mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Add support ( sans sniffer ) for FASTQ data type.
This commit is contained in:
@@ -44,6 +44,7 @@ class Registry( object ):
|
||||
'binseq.zip' : images.Binseq(),
|
||||
'customtrack' : interval.CustomTrack(),
|
||||
'fasta' : sequence.Fasta(),
|
||||
'fastq' : sequence.Fastq(),
|
||||
'gff' : interval.Gff(),
|
||||
'gff3' : interval.Gff3(),
|
||||
'interval' : interval.Interval(),
|
||||
@@ -65,6 +66,7 @@ class Registry( object ):
|
||||
'binseq.zip' : 'application/zip',
|
||||
'customtrack' : 'text/plain',
|
||||
'fasta' : 'text/plain',
|
||||
'fastq' : 'text/plain',
|
||||
'gff' : 'text/plain',
|
||||
'gff3' : 'text/plain',
|
||||
'interval' : 'text/plain',
|
||||
|
||||
@@ -88,6 +88,26 @@ class Fasta( Sequence ):
|
||||
except:
|
||||
return False
|
||||
|
||||
class Fastq( Sequence ):
|
||||
"""Class representing a FASTQ sequence"""
|
||||
# FASTQ format stores sequences and Phred qualities in a single file. It is concise and compact.
|
||||
# FASTQ is first widely used in the Sanger Institute and therefore we usually take the Sanger
|
||||
# specification and the standard FASTQ format, or simply FASTQ format. Although Solexa/Illumina
|
||||
# read file looks pretty much like FASTQ, they are different in that the qualities are scaled
|
||||
# differently. In the quality string, if you can see a character with its ASCII code higher than
|
||||
# 90, probably your file is in the Solexa/Illumina format.
|
||||
#
|
||||
# For details, see http://maq.sourceforge.net/fastq.shtml
|
||||
file_ext = "fastq"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
Sequence.set_peek( self, dataset )
|
||||
sequences = 0
|
||||
for line in file( dataset.file_name ):
|
||||
if line and line.startswith( "@" ):
|
||||
sequences += 1
|
||||
dataset.blurb = '%d sequences' % sequences
|
||||
|
||||
try:
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
import bx.align.maf
|
||||
|
||||
@@ -20,7 +20,7 @@
|
||||
|
||||
**Auto-detect**
|
||||
|
||||
The system will attempt to detect AXT, FASTA, Gff, HTML, LAV, Maf, Tabular, Wiggle, BED and Interval (BED with headers) formats. If your file is not detected properly as one of the known formats, it most likely means that it has some format problems (e.g., different number of columns on different rows). You can still coerce the system to set your data to the format you think it should be (please send us a note if you see a case when a valid format is not detected). You can also upload valid files that are compressed (gzipped), which will automatically be decompressed upon upload.
|
||||
The system will attempt to detect AXT, BED, FASTA, Gff, Gff3, Interval (BED with headers), LAV, Maf, Tabular and Wiggle formats. If your file is not detected properly as one of the known formats, it most likely means that it has some format problems (e.g., different number of columns on different rows). You can still coerce the system to set your data to the format you think it should be (please send us a note if you see a case when a valid format is not detected). You can also upload valid files that are compressed (gzipped), which will automatically be decompressed upon upload.
|
||||
|
||||
-----
|
||||
|
||||
@@ -36,12 +36,6 @@ blastz pairwise alignment format. Each alignment block in an axt file contains
|
||||
|
||||
-----
|
||||
|
||||
**Binseq.zip**
|
||||
|
||||
A zipped archive consisting of binary sequence files in either 'ab1' or 'scf' format. All files in this archive must have the same file extension which is one of '.ab1' or '.scf'. You must manually select this 'File Format' when uploading the file.
|
||||
|
||||
-----
|
||||
|
||||
**BED**
|
||||
|
||||
* Tab delimited format (tabular)
|
||||
@@ -71,6 +65,12 @@ A zipped archive consisting of binary sequence files in either 'ab1' or 'scf' fo
|
||||
|
||||
-----
|
||||
|
||||
**Binseq.zip**
|
||||
|
||||
A zipped archive consisting of binary sequence files in either 'ab1' or 'scf' format. All files in this archive must have the same file extension which is one of '.ab1' or '.scf'. You must manually select this 'File Format' when uploading the file.
|
||||
|
||||
-----
|
||||
|
||||
**FASTA**
|
||||
|
||||
A sequence in FASTA format consists of a single-line description, followed by lines of sequence data. The first character of the description line is a greater-than (">") symbol in the first column. All lines should be shorter than 80 charcters::
|
||||
@@ -84,6 +84,25 @@ A sequence in FASTA format consists of a single-line description, followed by li
|
||||
|
||||
-----
|
||||
|
||||
**FASTQ**
|
||||
|
||||
FASTQ format stores sequences and Phred qualities in a single file. FASTQ is first widely used in the Sanger Institute and therefore we usually take the Sanger specification and the standard FASTQ format, or simply FASTQ format. You must manually select this 'File Format' when uploading the file::
|
||||
|
||||
@EAS54_6_R1_2_1_413_324
|
||||
CCCTTCTTGTCTTCAGCGTTTCTCC
|
||||
+
|
||||
;;3;;;;;;;;;;;;7;;;;;;;88
|
||||
@EAS54_6_R1_2_1_540_792
|
||||
TTGGCAGGCCAAGGCCGATGGATCA
|
||||
+
|
||||
;;;;;;;;;;;7;;;;;-;;;3;83
|
||||
@EAS54_6_R1_2_1_443_348
|
||||
GTTGCTTCTGGCGTGGGTGGGGGGG
|
||||
+EAS54_6_R1_2_1_443_348
|
||||
;;;;;;;;;;;9;7;;.7;393333
|
||||
|
||||
-----
|
||||
|
||||
**Gff**
|
||||
|
||||
GFF lines have nine required fields that must be tab-separated.
|
||||
@@ -128,6 +147,20 @@ TBA and multiz multiple alignment format. The first line of a .maf file begins
|
||||
|
||||
-----
|
||||
|
||||
**Qual**
|
||||
|
||||
The qual sequence format is a FASTA-like format which stores numerical quality values for each nucleotide or amino acid. It is used by CAP3 and Phrap. You must manually select this 'File Format' when uploading the file::
|
||||
|
||||
>HSMETOO 134bp
|
||||
10 20 30 40 50 50 50 50 50 20 25 25 30 30 20 15 20 35 50 50 50 50 50 50
|
||||
50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50
|
||||
50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50
|
||||
50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50
|
||||
50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50 50
|
||||
50 50 50 20 30 20 10 10
|
||||
|
||||
-----
|
||||
|
||||
**Scf**
|
||||
|
||||
A binary sequence file in 'scf' format with a '.scf' file extension. You must manually select this 'File Format' when uploading the file.
|
||||
@@ -140,6 +173,18 @@ Any data in tab delimited format (tabular)
|
||||
|
||||
-----
|
||||
|
||||
**Taxonomy**
|
||||
|
||||
Tabular data containing at least 24 columns. You must manually select this 'File Format' when uploading the file.
|
||||
|
||||
-----
|
||||
|
||||
**Txt**
|
||||
|
||||
Any text file.
|
||||
|
||||
-----
|
||||
|
||||
**Txtseq.zip**
|
||||
|
||||
A zipped archive consisting of flat text sequence files. All files in this archive must have the same file extension of '.txt'. You must manually select this 'File Format' when uploading the file.
|
||||
@@ -150,11 +195,6 @@ A zipped archive consisting of flat text sequence files. All files in this arch
|
||||
|
||||
The wiggle format is line-oriented. Wiggle data is preceeded by a track definition line, which adds a number of options for controlling the default display of this track.
|
||||
|
||||
-----
|
||||
|
||||
**Other text type**
|
||||
|
||||
Any text file
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
@@ -174,7 +174,7 @@ binseq.zip = galaxy.datatypes.images:Binseq,application/zip
|
||||
customtrack = galaxy.datatypes.interval:CustomTrack
|
||||
data = galaxy.datatypes.data:Data,application/octet-stream
|
||||
fasta = galaxy.datatypes.sequence:Fasta
|
||||
gbrowsetrack = galaxy.datatypes.interval:GBrowseTrack
|
||||
fastq = galaxy.datatypes.sequence:Fastq
|
||||
gff = galaxy.datatypes.interval:Gff
|
||||
gff3 = galaxy.datatypes.interval:Gff3
|
||||
gif = galaxy.datatypes.images:Image,image/gif
|
||||
|
||||
Reference in New Issue
Block a user