Merge pull request #5793 from jmchilton/bounded_memory_datatypes

Sniffing framework with constrained memory and I/O.
This commit is contained in:
Martin Cech
2018-04-13 15:43:37 -04:00
committed by GitHub
34 changed files with 1625 additions and 865 deletions
+4 -4
View File
@@ -3,11 +3,13 @@ import tarfile
from galaxy.datatypes.binary import CompressedArchive
from galaxy.datatypes.data import get_file_peek, Text
from galaxy.datatypes.sniff import build_sniff_from_prefix
from galaxy.util import nice_size
log = logging.getLogger(__name__)
@build_sniff_from_prefix
class SnapHmm(Text):
file_ext = "snaphmm"
edam_data = "data_1364"
@@ -26,13 +28,11 @@ class SnapHmm(Text):
except Exception:
return "SNAP HMM model (%s)" % (nice_size(dataset.get_size()))
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
SNAP model files start with zoeHMM
"""
with open(filename, 'r') as handle:
return handle.read(6) == 'zoeHMM'
return False
return file_prefix.startswith('zoeHMM')
class Augustus(CompressedArchive):
+48 -49
View File
@@ -13,20 +13,20 @@ import sys
from galaxy.datatypes import data
from galaxy.datatypes import sequence
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import build_sniff_from_prefix
from galaxy.datatypes.text import Html
log = logging.getLogger(__name__)
@build_sniff_from_prefix
class Amos(data.Text):
"""Class describing the AMOS assembly file """
edam_data = "data_0925"
edam_format = "format_3582"
file_ext = 'afg'
def sniff(self, filename):
# FIXME: this method will read the entire file.
# It should call get_headers() like other sniff methods.
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is an amos assembly file format
Example::
@@ -50,25 +50,24 @@ class Amos(data.Text):
}
}
"""
with open(filename) as fh:
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if line.startswith('{'):
if re.match(r'{(RED|CTG|TLE)$', line):
return True
for line in file_prefix.line_iterator():
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if line.startswith('{'):
if re.match(r'{(RED|CTG|TLE)$', line):
return True
return False
@build_sniff_from_prefix
class Sequences(sequence.Fasta):
"""Class describing the Sequences file generated by velveth """
edam_data = "data_0925"
file_ext = 'sequences'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a velveth produced fasta format
The id line has 3 fields separated by tabs: sequence_name sequence_index category::
@@ -78,33 +77,33 @@ class Sequences(sequence.Fasta):
>SEQUENCE_1_length_35 2 1
CGACGAATGACAGGTCACGAATTTGGCGGGGATTA
"""
with open(filename) as fh:
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if line.startswith('>'):
if not re.match(r'>[^\t]+\t\d+\t\d+$', line):
break
# The next line.strip() must not be '', nor startwith '>'
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
return True
else:
break # we found a non-empty line, but it's not a fasta header
fh = file_prefix.string_io()
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if line.startswith('>'):
if not re.match(r'>[^\t]+\t\d+\t\d+$', line):
break
# The next line.strip() must not be '', nor startwith '>'
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
return True
else:
break # we found a non-empty line, but it's not a fasta header
return False
@build_sniff_from_prefix
class Roadmaps(data.Text):
"""Class describing the Sequences file generated by velveth """
edam_format = "format_2561"
file_ext = 'roadmaps'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a velveth produced RoadMap::
142858 21 1
@@ -113,22 +112,22 @@ class Roadmaps(data.Text):
...
"""
with open(filename) as fh:
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if not re.match(r'\d+\t\d+\t\d+$', line):
break
# The next line.strip() should be 'ROADMAP 1'
line = fh.readline().strip()
if not re.match(r'ROADMAP \d+$', line):
break
return True
else:
break # we found a non-empty line, but it's not a fasta header
fh = file_prefix.string_io()
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if not re.match(r'\d+\t\d+\t\d+$', line):
break
# The next line.strip() should be 'ROADMAP 1'
line = fh.readline().strip()
if not re.match(r'ROADMAP \d+$', line):
break
return True
else:
break # we found a non-empty line, but it's not a fasta header
return False
+14 -12
View File
@@ -39,11 +39,13 @@ from .data import (
get_file_peek,
Text
)
from .sniff import build_sniff_from_prefix
from .xml import GenericXml
log = logging.getLogger(__name__)
@build_sniff_from_prefix
class BlastXml(GenericXml):
"""NCBI Blast XML Output data"""
file_ext = "blastxml"
@@ -59,7 +61,7 @@ class BlastXml(GenericXml):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""Determines whether the file is blastxml
>>> from galaxy.datatypes.sniff import get_test_fname
@@ -73,17 +75,17 @@ class BlastXml(GenericXml):
>>> BlastXml().sniff(fname)
False
"""
with open(filename) as handle:
line = handle.readline()
if line.strip() != '<?xml version="1.0"?>':
return False
line = handle.readline()
if line.strip() not in ['<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "http://www.ncbi.nlm.nih.gov/dtd/NCBI_BlastOutput.dtd">',
'<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "NCBI_BlastOutput.dtd">']:
return False
line = handle.readline()
if line.strip() != '<BlastOutput>':
return False
handle = file_prefix.string_io()
line = handle.readline()
if line.strip() != '<?xml version="1.0"?>':
return False
line = handle.readline()
if line.strip() not in ['<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "http://www.ncbi.nlm.nih.gov/dtd/NCBI_BlastOutput.dtd">',
'<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "NCBI_BlastOutput.dtd">']:
return False
line = handle.readline()
if line.strip() != '<BlastOutput>':
return False
return True
def merge(split_files, output_file):
@@ -9,12 +9,14 @@ from galaxy.datatypes.binary import Binary
from galaxy.datatypes.data import get_file_peek
from galaxy.datatypes.data import nice_size
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import build_sniff_from_prefix
MAX_HEADER_LINES = 500
MAX_LINE_LEN = 2000
COLOR_OPTS = ['COLOR_SCALARS', 'red', 'green', 'blue']
@build_sniff_from_prefix
class Ply(object):
"""
The PLY format describes an object as a collection of vertices,
@@ -37,16 +39,14 @@ class Ply(object):
def __init__(self, **kwd):
raise NotImplementedError
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
The structure of a typical PLY file:
Header, Vertex List, Face List, (lists of other elements)
"""
with open(filename, "r") as fh:
if not self._is_ply_header(fh, self.subtype):
return False
return True
return False
if not self._is_ply_header(file_prefix.string_io(), self.subtype):
return False
return True
def _is_ply_header(self, fh, subtype):
"""
@@ -131,6 +131,7 @@ class PlyBinary(Ply, Binary):
Binary.__init__(self, **kwd)
@build_sniff_from_prefix
class Vtk(object):
r"""
The Visualization Toolkit provides a number of source and writer objects to
@@ -200,16 +201,14 @@ class Vtk(object):
def __init__(self, **kwd):
raise NotImplementedError
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
VTK files can be either ASCII or binary, with two different
styles of file formats: legacy or XML. We'll assume if the
file contains a valid VTK header, then it is a valid VTK file.
"""
with open(filename, "r") as fh:
if self._is_vtk_header(fh, self.subtype):
return True
return False
if self._is_vtk_header(file_prefix.string_io(), self.subtype):
return True
return False
def _is_vtk_header(self, fh, subtype):
+4 -8
View File
@@ -16,6 +16,7 @@ import six
from galaxy import util
from galaxy.datatypes.metadata import MetadataElement # import directly to maintain ease of use in Datatype class definitions
from galaxy.datatypes.sniff import build_sniff_from_prefix
from galaxy.util import (
compression_utils,
FILENAME_VALID_CHARS,
@@ -961,6 +962,7 @@ class Newick(Text):
return ['phyloviz']
@build_sniff_from_prefix
class Nexus(Text):
"""Nexus format as used By Paup, Mr Bayes, etc"""
edam_data = "data_0872"
@@ -974,15 +976,9 @@ class Nexus(Text):
def init_meta(self, dataset, copy_from=None):
Text.init_meta(self, dataset, copy_from=copy_from)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""All Nexus Files Simply puts a '#NEXUS' in its first line"""
with open(filename, "r") as f:
firstline = f.readline().upper()
if "#NEXUS" in firstline:
return True
else:
return False
return file_prefix.string_io().read(6).upper() == "#NEXUS"
def get_visualizations(self, dataset):
"""
+97 -106
View File
@@ -22,6 +22,7 @@ from six.moves.urllib.parse import quote_plus
from galaxy.datatypes import metadata
from galaxy.datatypes.data import Text
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import build_sniff_from_prefix
from galaxy.datatypes.tabular import Tabular
from galaxy.datatypes.text import Html
from galaxy.util import nice_size
@@ -35,6 +36,7 @@ VALID_GENOME_GRAPH_MARKERS = re.compile('^(chr.*|RH.*|rs.*|SNP_.*|CN.*|A_.*)')
VALID_GENOTYPES_LINE = re.compile('^([a-zA-Z0-9]+)(\\s([0-9]{2}|[A-Z]{2}|NC|\?\?))+\\s*$')
@build_sniff_from_prefix
class GenomeGraphs(Tabular):
"""
Tab delimited data containing a marker id and any number of numeric values
@@ -166,7 +168,7 @@ class GenomeGraphs(Tabular):
errors.append('row %d, %s' % (' '.join(badvals)))
return errors
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in gg format
@@ -178,9 +180,7 @@ class GenomeGraphs(Tabular):
>>> GenomeGraphs().sniff( fname )
True
"""
with open(filename, 'r') as f:
buf = f.read(1024)
buf = file_prefix.contents_header
rows = [l.split() for l in buf.splitlines()[1:4]] # break on lines and drop header, small sample
if len(rows) < 1:
@@ -247,14 +247,6 @@ class rgSampleList(rgTabList):
self.column_names[1] = 'IID'
# this is what Plink wants as at 2009
def sniff(self, filename):
with open(filename, "r") as infile:
header = next(infile) # header
if header[0] == 'FID' and header[1] == 'IID':
return True
else:
return False
class rgFeatureList(rgTabList):
"""
@@ -909,6 +901,7 @@ class LinkageStudies(Text):
self.max_lines = 10
@build_sniff_from_prefix
class GenotypeMatrix(LinkageStudies):
"""
Sample matrix of genotypes
@@ -918,7 +911,6 @@ class GenotypeMatrix(LinkageStudies):
def __init__(self, **kwd):
super(GenotypeMatrix, self).__init__(**kwd)
self.num_cols = -1
def header_check(self, fio):
header_elems = fio.readline().split('\t')
@@ -933,7 +925,7 @@ class GenotypeMatrix(LinkageStudies):
return True
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
>>> classname = GenotypeMatrix
>>> from galaxy.datatypes.sniff import get_test_fname
@@ -941,7 +933,6 @@ class GenotypeMatrix(LinkageStudies):
>>> file_true = get_test_fname("linkstudies." + extn_true)
>>> classname().sniff(file_true)
True
>>> false_files = list(LinkageStudies.test_files)
>>> false_files.remove("linkstudies." + extn_true)
>>> result_true = []
@@ -954,27 +945,29 @@ class GenotypeMatrix(LinkageStudies):
>>> result_true
[]
"""
with open(filename, "r") as fio:
fio = file_prefix.string_io()
num_cols = -1
if not self.header_check(fio):
if not self.header_check(fio):
return False
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return True
tokens = line.split('\t')
if num_cols == -1:
num_cols = len(tokens)
elif num_cols != len(tokens):
return False
if not VALID_GENOTYPES_LINE.match(line):
return False
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return True
tokens = line.split('\t')
if self.num_cols == -1:
self.num_cols = len(tokens)
elif self.num_cols != len(tokens):
return False
if not VALID_GENOTYPES_LINE.match(line):
return False
return True
return True
@build_sniff_from_prefix
class MarkerMap(LinkageStudies):
"""
Map of genetic markers including physical and genetic distance
@@ -992,7 +985,7 @@ class MarkerMap(LinkageStudies):
return False
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
>>> classname = MarkerMap
>>> from galaxy.datatypes.sniff import get_test_fname
@@ -1000,7 +993,6 @@ class MarkerMap(LinkageStudies):
>>> file_true = get_test_fname("linkstudies." + extn_true)
>>> classname().sniff(file_true)
True
>>> false_files = list(LinkageStudies.test_files)
>>> false_files.remove("linkstudies." + extn_true)
>>> result_true = []
@@ -1013,32 +1005,32 @@ class MarkerMap(LinkageStudies):
>>> result_true
[]
"""
with open(filename, "r") as fio:
fio = file_prefix.string_io()
if not self.header_check(fio):
return False
if not self.header_check(fio):
return False
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return True
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return True
try:
chrm, gpos, nam, bpos, row = line.split()
float(gpos)
int(bpos)
try:
chrm, gpos, nam, bpos, row = line.split()
float(gpos)
int(bpos)
try:
int(chrm)
except ValueError:
if not chrm.lower()[0] in ('x', 'y', 'm'):
return False
int(chrm)
except ValueError:
return False
if not chrm.lower()[0] in ('x', 'y', 'm'):
return False
return True
except ValueError:
return False
return True
@build_sniff_from_prefix
class DataIn(LinkageStudies):
"""
Common linkage input file for intermarker distances
@@ -1048,13 +1040,8 @@ class DataIn(LinkageStudies):
def __init__(self, **kwd):
super(DataIn, self).__init__(**kwd)
self.num_markers = None
self.intermarkers = 0
def eof_function(self):
return self.intermarkers > 0
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
>>> classname = DataIn
>>> from galaxy.datatypes.sniff import get_test_fname
@@ -1062,7 +1049,6 @@ class DataIn(LinkageStudies):
>>> file_true = get_test_fname("linkstudies." + extn_true)
>>> classname().sniff(file_true)
True
>>> false_files = list(LinkageStudies.test_files)
>>> false_files.remove("linkstudies." + extn_true)
>>> result_true = []
@@ -1075,41 +1061,47 @@ class DataIn(LinkageStudies):
>>> result_true
[]
"""
with open(filename, "r") as fio:
intermarkers = 0
num_markers = None
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return self.eof_function()
def eof_function():
return intermarkers > 0
tokens = line.split()
try:
if lcount == 0:
self.num_markers = int(tokens[0])
map(int, tokens[1:])
elif lcount == 1:
map(float, tokens)
fio = file_prefix.string_io()
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return eof_function()
if len(tokens) != 4:
return False
elif lcount == 2:
map(int, tokens)
last_token = int(tokens[-1])
tokens = line.split()
try:
if lcount == 0:
num_markers = int(tokens[0])
map(int, tokens[1:])
elif lcount == 1:
map(float, tokens)
if self.num_markers is None:
return False
if len(tokens) != last_token:
return False
if self.num_markers != last_token:
return False
elif tokens[0] == "3" and tokens[1] == "2":
self.intermarkers += 1
if len(tokens) != 4:
return False
elif lcount == 2:
map(int, tokens)
last_token = int(tokens[-1])
except (ValueError, IndexError):
return False
if num_markers is None:
return False
if len(tokens) != last_token:
return False
if num_markers != last_token:
return False
elif tokens[0] == "3" and tokens[1] == "2":
intermarkers += 1
return self.eof_function()
except (ValueError, IndexError):
return False
return eof_function()
@build_sniff_from_prefix
class AllegroLOD(LinkageStudies):
"""
Allegro output format for LOD scores
@@ -1125,7 +1117,7 @@ class AllegroLOD(LinkageStudies):
return False
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
>>> classname = AllegroLOD
>>> from galaxy.datatypes.sniff import get_test_fname
@@ -1133,7 +1125,6 @@ class AllegroLOD(LinkageStudies):
>>> file_true = get_test_fname("linkstudies." + extn_true)
>>> classname().sniff(file_true)
True
>>> false_files = list(LinkageStudies.test_files)
>>> false_files.remove("linkstudies." + extn_true)
>>> result_true = []
@@ -1146,28 +1137,28 @@ class AllegroLOD(LinkageStudies):
>>> result_true
[]
"""
with open(filename, "r") as fio:
fio = file_prefix.string_io()
if not self.header_check(fio):
if not self.header_check(fio):
return False
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return True
tokens = line.split()
try:
int(tokens[0])
float(tokens[1])
if tokens[2] != "-inf":
float(tokens[2])
except (ValueError, IndexError):
return False
for lcount, line in enumerate(fio):
if lcount > self.max_lines:
return True
tokens = line.split()
try:
int(tokens[0])
float(tokens[1])
if tokens[2] != "-inf":
float(tokens[2])
except (ValueError, IndexError):
return False
return True
return True
if __name__ == '__main__':
+80 -75
View File
@@ -13,6 +13,7 @@ from galaxy import util
from galaxy.datatypes import metadata
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import (
build_sniff_from_prefix,
get_headers,
iter_headers
)
@@ -50,6 +51,7 @@ VIEWPORT_MAX_READS_PER_LINE = 10
@dataproviders.decorators.has_dataproviders
@build_sniff_from_prefix
class Interval(Tabular):
"""Tab delimited data containing interval information"""
edam_data = "data_3002"
@@ -297,7 +299,7 @@ class Interval(Tabular):
"""Return options for removing errors along with a description"""
return [("lines", "Remove erroneous lines")]
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Checks for 'intervalness'
@@ -312,26 +314,23 @@ class Interval(Tabular):
>>> Interval().sniff( fname )
True
"""
found_valid_lines = False
try:
"""
If we got here, we already know the file is_column_based and is not bed,
so we'll just look for some valid data.
"""
headers = iter_headers(filename, '\t', comment_designator='#')
headers = iter_headers(file_prefix, '\t', comment_designator='#')
# If we got here, we already know the file is_column_based and is not bed,
# so we'll just look for some valid data.
for hdr in headers:
if hdr:
if len(hdr) < 3:
return False
try:
# Assume chrom start and end are in column positions 1 and 2
# respectively ( for 0 based columns )
int(hdr[1])
int(hdr[2])
except Exception:
return False
return True
# Assume chrom start and end are in column positions 1 and 2
# respectively ( for 0 based columns )
int(hdr[1])
int(hdr[2])
found_valid_lines = True
except Exception:
return False
return found_valid_lines
def get_track_resolution(self, dataset, start, end):
return None
@@ -464,7 +463,7 @@ class Bed(Interval):
except Exception:
return "This item contains no content"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Checks for 'bedness'
@@ -488,10 +487,10 @@ class Bed(Interval):
>>> Bed().sniff( fname )
True
"""
if not get_headers(filename, '\t', comment_designator='#', count=1):
if not get_headers(file_prefix, '\t', comment_designator='#', count=1):
return False
try:
headers = iter_headers(filename, '\t', comment_designator='#')
headers = iter_headers(file_prefix, '\t', comment_designator='#')
for hdr in headers:
if hdr[0] == '':
continue
@@ -635,6 +634,7 @@ class _RemoteCallMixin(object):
@dataproviders.decorators.has_dataproviders
@build_sniff_from_prefix
class Gff(Tabular, _RemoteCallMixin):
"""Tab delimited data in Gff format"""
edam_data = "data_1255"
@@ -822,7 +822,7 @@ class Gff(Tabular, _RemoteCallMixin):
ret_val.append((site_name, link))
return ret_val
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in gff format
@@ -831,17 +831,17 @@ class Gff(Tabular, _RemoteCallMixin):
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3
>>> from galaxy.datatypes.sniff import get_test_fname
>>> fname = get_test_fname( 'gff_version_3.gff' )
>>> fname = get_test_fname('gff_version_3.gff')
>>> Gff().sniff( fname )
False
>>> fname = get_test_fname( 'test.gff' )
>>> fname = get_test_fname('test.gff')
>>> Gff().sniff( fname )
True
"""
if len(get_headers(filename, '\t', count=2)) < 2:
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(filename, '\t')
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
return False
@@ -937,7 +937,7 @@ class Gff3(Gff):
break
Tabular.set_meta(self, dataset, overwrite=overwrite, skip=i)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in GFF version 3 format
@@ -970,10 +970,10 @@ class Gff3(Gff):
>>> Gff3().sniff( fname )
True
"""
if len(get_headers(filename, '\t', count=2)) < 2:
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(filename, '\t')
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
return True
@@ -1020,7 +1020,7 @@ class Gtf(Gff):
MetadataElement(name="column_types", default=['str', 'str', 'str', 'int', 'int', 'float', 'str', 'int', 'list'],
param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in gtf format
@@ -1045,10 +1045,10 @@ class Gtf(Gff):
>>> Gtf().sniff( fname )
True
"""
if len(get_headers(filename, '\t', count=2)) < 2:
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(filename, '\t')
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
return False
@@ -1085,6 +1085,7 @@ class Gtf(Gff):
@dataproviders.decorators.has_dataproviders
@build_sniff_from_prefix
class Wiggle(Tabular, _RemoteCallMixin):
"""Tab delimited data in wiggle format"""
edam_format = "format_3005"
@@ -1218,7 +1219,7 @@ class Wiggle(Tabular, _RemoteCallMixin):
max_data_lines = 100
Tabular.set_meta(self, dataset, overwrite=overwrite, skip=i, max_data_lines=max_data_lines)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines wether the file is in wiggle format
@@ -1242,7 +1243,7 @@ class Wiggle(Tabular, _RemoteCallMixin):
True
"""
try:
headers = iter_headers(filename, None)
headers = iter_headers(file_prefix, None)
for hdr in headers:
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
return True
@@ -1272,6 +1273,7 @@ class Wiggle(Tabular, _RemoteCallMixin):
return dataproviders.dataset.WiggleDataProvider(dataset_source, **settings)
@build_sniff_from_prefix
class CustomTrack(Tabular):
"""UCSC CustomTrack"""
edam_format = "format_3588"
@@ -1360,7 +1362,7 @@ class CustomTrack(Tabular):
ret_val.append((site_name, link))
return ret_val
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in customtrack format.
@@ -1377,7 +1379,8 @@ class CustomTrack(Tabular):
>>> CustomTrack().sniff( fname )
True
"""
headers = iter_headers(filename, None)
headers = iter_headers(file_prefix, None)
found_at_least_one_track = False
first_line = True
for hdr in headers:
if first_line:
@@ -1409,9 +1412,10 @@ class CustomTrack(Tabular):
int(hdr[2])
except Exception:
return False
found_at_least_one_track = True
except Exception:
return False
return True
return found_at_least_one_track
class ENCODEPeak(Interval):
@@ -1467,6 +1471,7 @@ class ChromatinInteractions(Interval):
return False
@build_sniff_from_prefix
class ScIdx(Tabular):
"""
ScIdx files are 1-based and consist of strand-specific coordinate counts.
@@ -1492,55 +1497,55 @@ class ScIdx(Tabular):
# line of the dataset displays them.
self.column_names = ['chrom', 'index', 'forward', 'reverse', 'value']
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Checks for 'scidx-ness.'
"""
count = 0
with open(filename, "r") as fh:
while True:
line = fh.readline()
if not line:
# EOF
if count > 1:
# The second line is always the labels:
# chrom index forward reverse value
# We need at least the column labels and a data line.
return True
return False
line = line.strip()
# The first line is always a comment like this:
# 2015-11-23 20:18:56.51;input.bam;READ1
if count == 0:
if line.startswith('#'):
count += 1
continue
else:
return False
# Skip first line.
fh = file_prefix.string_io()
while True:
line = fh.readline()
if not line:
# EOF
if count > 1:
items = line.split('\t')
if len(items) != 5:
return False
index = items[1]
if not index.isdigit():
return False
forward = items[2]
if not forward.isdigit():
return False
reverse = items[3]
if not reverse.isdigit():
return False
value = items[4]
if not value.isdigit():
return False
if int(forward) + int(reverse) != int(value):
return False
if count == 100:
# The second line is always the labels:
# chrom index forward reverse value
# We need at least the column labels and a data line.
return True
count += 1
if count < 100 and count > 0:
return False
line = line.strip()
# The first line is always a comment like this:
# 2015-11-23 20:18:56.51;input.bam;READ1
if count == 0:
if line.startswith('#'):
count += 1
continue
else:
return False
# Skip first line.
if count > 1:
items = line.split('\t')
if len(items) != 5:
return False
index = items[1]
if not index.isdigit():
return False
forward = items[2]
if not forward.isdigit():
return False
reverse = items[3]
if not reverse.isdigit():
return False
value = items[4]
if not value.isdigit():
return False
if int(forward) + int(reverse) != int(value):
return False
if count == 100:
return True
count += 1
if count < 100 and count > 0:
return True
return False
+48 -51
View File
@@ -11,6 +11,7 @@ from galaxy.datatypes.binary import Binary
from galaxy.datatypes.data import get_file_peek
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import (
build_sniff_from_prefix,
get_headers,
iter_headers
)
@@ -84,10 +85,11 @@ class MOL(GenericMolFile):
dataset.metadata.number_of_molecules = 1
@build_sniff_from_prefix
class SDF(GenericMolFile):
file_ext = "sdf"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess if the file is a SDF2 file.
@@ -102,11 +104,9 @@ class SDF(GenericMolFile):
>>> fname = get_test_fname('drugbank_drugs.sdf')
>>> SDF().sniff(fname)
True
>>> fname = get_test_fname('github88.v3k.sdf')
>>> SDF().sniff(fname)
True
>>> fname = get_test_fname('chebi_57262.v3k.mol')
>>> SDF().sniff(fname)
False
@@ -114,23 +114,22 @@ class SDF(GenericMolFile):
m_end_found = False
limit = 10000
idx = 0
with open(filename) as in_file:
for line in in_file:
idx += 1
line = line.rstrip('\n\r')
if idx < 4:
continue
elif idx == 4:
if len(line) != 39 or not(line.endswith(' V2000') or
line.endswith(' V3000')):
return False
elif not m_end_found:
if line == 'M END':
m_end_found = True
elif line == '$$$$':
return True
if idx == limit:
break
for line in file_prefix.line_iterator():
idx += 1
line = line.rstrip('\n\r')
if idx < 4:
continue
elif idx == 4:
if len(line) != 39 or not(line.endswith(' V2000') or
line.endswith(' V3000')):
return False
elif not m_end_found:
if line == 'M END':
m_end_found = True
elif line == '$$$$':
return True
if idx == limit:
break
return False
def set_meta(self, dataset, **kwd):
@@ -189,10 +188,11 @@ class SDF(GenericMolFile):
split = classmethod(split)
@build_sniff_from_prefix
class MOL2(GenericMolFile):
file_ext = "mol2"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess if the file is a MOL2 file.
@@ -200,21 +200,19 @@ class MOL2(GenericMolFile):
>>> fname = get_test_fname('drugbank_drugs.mol2')
>>> MOL2().sniff(fname)
True
>>> fname = get_test_fname('drugbank_drugs.cml')
>>> MOL2().sniff(fname)
False
"""
limit = 60
idx = 0
with open(filename) as in_file:
for line in in_file:
line = line.rstrip('\n\r')
if line == '@<TRIPOS>MOLECULE':
return True
idx += 1
if idx == limit:
break
for line in file_prefix.line_iterator():
line = line.rstrip('\n\r')
if line == '@<TRIPOS>MOLECULE':
return True
idx += 1
if idx == limit:
break
return False
def set_meta(self, dataset, **kwd):
@@ -277,13 +275,14 @@ class MOL2(GenericMolFile):
split = classmethod(split)
@build_sniff_from_prefix
class FPS(GenericMolFile):
"""
chemfp fingerprint file: http://code.google.com/p/chem-fingerprints/wiki/FPS
"""
file_ext = "fps"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess if the file is a FPS file.
@@ -291,12 +290,11 @@ class FPS(GenericMolFile):
>>> fname = get_test_fname('q.fps')
>>> FPS().sniff(fname)
True
>>> fname = get_test_fname('drugbank_drugs.cml')
>>> FPS().sniff(fname)
False
"""
header = get_headers(filename, sep='\t', count=1)
header = get_headers(file_prefix, sep='\t', count=1)
if header[0][0].strip() == '#FPS1':
return True
else:
@@ -473,6 +471,7 @@ class PHAR(GenericMolFile):
dataset.blurb = 'file purged from disk'
@build_sniff_from_prefix
class PDB(GenericMolFile):
"""
Protein Databank format.
@@ -480,7 +479,7 @@ class PDB(GenericMolFile):
"""
file_ext = "pdb"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess if the file is a PDB file.
@@ -488,12 +487,11 @@ class PDB(GenericMolFile):
>>> fname = get_test_fname('5e5z.pdb')
>>> PDB().sniff(fname)
True
>>> fname = get_test_fname('drugbank_drugs.cml')
>>> PDB().sniff(fname)
False
"""
headers = iter_headers(filename, sep=' ', count=300)
headers = iter_headers(file_prefix, sep=' ', count=300)
h = t = c = s = k = e = False
for line in headers:
section_name = line[0].strip()
@@ -526,6 +524,7 @@ class PDB(GenericMolFile):
dataset.blurb = 'file purged from disk'
@build_sniff_from_prefix
class PDBQT(GenericMolFile):
"""
PDBQT Autodock and Autodock Vina format
@@ -533,7 +532,7 @@ class PDBQT(GenericMolFile):
"""
file_ext = "pdbqt"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess if the file is a PDBQT file.
@@ -541,12 +540,11 @@ class PDBQT(GenericMolFile):
>>> fname = get_test_fname('NuBBE_1_obabel_3D.pdbqt')
>>> PDBQT().sniff(fname)
True
>>> fname = get_test_fname('drugbank_drugs.cml')
>>> PDBQT().sniff(fname)
False
"""
headers = iter_headers(filename, sep=' ', count=300)
headers = iter_headers(file_prefix, sep=' ', count=300)
h = t = c = s = k = False
for line in headers:
section_name = line[0].strip()
@@ -601,6 +599,7 @@ class grdtgz(Binary):
dataset.blurb = 'file purged from disk'
@build_sniff_from_prefix
class InChI(Tabular):
file_ext = "inchi"
column_names = ['InChI']
@@ -625,7 +624,7 @@ class InChI(Tabular):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess if the file is a InChI file.
@@ -633,16 +632,17 @@ class InChI(Tabular):
>>> fname = get_test_fname('drugbank_drugs.inchi')
>>> InChI().sniff(fname)
True
>>> fname = get_test_fname('drugbank_drugs.cml')
>>> InChI().sniff(fname)
False
"""
inchi_lines = iter_headers(filename, sep=' ', count=10)
inchi_lines = iter_headers(file_prefix, sep=' ', count=10)
found_lines = False
for inchi in inchi_lines:
if not inchi[0].startswith('InChI='):
return False
return True
found_lines = True
return found_lines
class SMILES(Tabular):
@@ -704,6 +704,7 @@ class SMILES(Tabular):
'''
@build_sniff_from_prefix
class CML(GenericXml):
"""
Chemical Markup Language
@@ -729,7 +730,7 @@ class CML(GenericXml):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess if the file is a CML file.
@@ -737,18 +738,14 @@ class CML(GenericXml):
>>> fname = get_test_fname('interval.interval')
>>> CML().sniff(fname)
False
>>> fname = get_test_fname('drugbank_drugs.cml')
>>> CML().sniff(fname)
True
"""
with open(filename) as handle:
line = handle.readline()
if line.strip() != '<?xml version="1.0"?>':
return False
line = handle.readline()
if line.strip().find('http://www.xml-cml.org/schema') == -1:
for expected_string in ['<?xml version="1.0"?>', 'http://www.xml-cml.org/schema']:
if expected_string not in file_prefix.contents_header:
return False
return True
def split(cls, input_datasets, subdir_generator_function, split_params):
+39 -28
View File
@@ -8,6 +8,7 @@ import sys
from galaxy.datatypes.data import Text
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import (
build_sniff_from_prefix,
get_headers,
iter_headers
)
@@ -16,6 +17,7 @@ from galaxy.datatypes.tabular import Tabular
log = logging.getLogger(__name__)
@build_sniff_from_prefix
class Otu(Text):
file_ext = 'mothur.otu'
MetadataElement(name="columns", default=0, desc="Number of columns", readonly=True, visible=True, no_value=0)
@@ -76,7 +78,7 @@ class Otu(Text):
dataset.metadata.otulabels = list(otulabel_names)
dataset.metadata.otulabels.sort()
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is otu (operational taxonomic unit) format
@@ -88,7 +90,7 @@ class Otu(Text):
>>> Otu().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -120,7 +122,7 @@ class Sabund(Otu):
def init_meta(self, dataset, copy_from=None):
super(Sabund, self).init_meta(dataset, copy_from=copy_from)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is otu (operational taxonomic unit) format
label<TAB>count[<TAB>value(1..n)]
@@ -133,7 +135,7 @@ class Sabund(Otu):
>>> Sabund().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -196,7 +198,7 @@ class GroupAbund(Otu):
dataset.metadata.groups.sort()
dataset.metadata.skip = skip
def sniff(self, filename, vals_are_int=False):
def sniff_prefix(self, file_prefix, vals_are_int=False):
"""
Determines whether the file is a otu (operational taxonomic unit)
Shared format
@@ -211,7 +213,7 @@ class GroupAbund(Otu):
>>> GroupAbund().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -235,6 +237,7 @@ class GroupAbund(Otu):
return False
@build_sniff_from_prefix
class SecondaryStructureMap(Tabular):
file_ext = 'mothur.map'
@@ -243,7 +246,7 @@ class SecondaryStructureMap(Tabular):
super(SecondaryStructureMap, self).__init__(**kwd)
self.column_names = ['Map']
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a secondary structure map format
A single column with an integer value which indicates the row that this
@@ -258,7 +261,7 @@ class SecondaryStructureMap(Tabular):
>>> SecondaryStructureMap().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
line_num = 0
rowidxmap = {}
for line in headers:
@@ -337,6 +340,7 @@ class DistanceMatrix(Text):
log.warning("DistanceMatrix set_meta %s" % e)
@build_sniff_from_prefix
class LowerTriangleDistanceMatrix(DistanceMatrix):
file_ext = 'mothur.lower.dist'
@@ -347,7 +351,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
def init_meta(self, dataset, copy_from=None):
super(LowerTriangleDistanceMatrix, self).init_meta(dataset, copy_from=copy_from)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a lower-triangle distance matrix (phylip) format
The first line has the number of sequences in the matrix.
@@ -368,7 +372,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
False
"""
numlines = 300
headers = iter_headers(filename, sep='\t', count=numlines)
headers = iter_headers(file_prefix, sep='\t', count=numlines)
line_num = 0
for line in headers:
if not line[0].startswith('@'):
@@ -400,6 +404,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
return False
@build_sniff_from_prefix
class SquareDistanceMatrix(DistanceMatrix):
file_ext = 'mothur.square.dist'
@@ -409,7 +414,7 @@ class SquareDistanceMatrix(DistanceMatrix):
def init_meta(self, dataset, copy_from=None):
super(SquareDistanceMatrix, self).init_meta(dataset, copy_from=copy_from)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a square distance matrix (Column-formatted distance matrix) format
The first line has the number of sequences in the matrix.
@@ -429,7 +434,7 @@ class SquareDistanceMatrix(DistanceMatrix):
False
"""
numlines = 300
headers = iter_headers(filename, sep='\t', count=numlines)
headers = iter_headers(file_prefix, sep='\t', count=numlines)
line_num = 0
for line in headers:
if not line[0].startswith('@'):
@@ -460,6 +465,7 @@ class SquareDistanceMatrix(DistanceMatrix):
return False
@build_sniff_from_prefix
class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
file_ext = 'mothur.pair.dist'
@@ -472,7 +478,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
def set_meta(self, dataset, overwrite=True, skip=None, **kwd):
super(PairwiseDistanceMatrix, self).set_meta(dataset, overwrite=overwrite, skip=skip, **kwd)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a pairwise distance matrix (Column-formatted distance matrix) format
The first and second columns have the sequence names and the third column is the distance between those sequences.
@@ -485,7 +491,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
>>> PairwiseDistanceMatrix().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -566,10 +572,11 @@ class AccNos(Tabular):
self.columns = 1
@build_sniff_from_prefix
class Oligos(Text):
file_ext = 'mothur.oligos'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
http://www.mothur.org/wiki/Oligos_File
Determines whether the file is a otu (operational taxonomic unit) format
@@ -582,7 +589,7 @@ class Oligos(Text):
>>> Oligos().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@') and not line[0].startswith('#'):
@@ -600,6 +607,7 @@ class Oligos(Text):
return False
@build_sniff_from_prefix
class Frequency(Tabular):
file_ext = 'mothur.freq'
@@ -609,7 +617,7 @@ class Frequency(Tabular):
self.column_names = ['position', 'frequency']
self.column_types = ['int', 'float']
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a frequency tabular format for chimera analysis
#1.14.0
@@ -625,13 +633,12 @@ class Frequency(Tabular):
>>> fname = get_test_fname( 'mothur_datatypetest_false.mothur.freq' )
>>> Frequency().sniff( fname )
False
# Expression count matrix (EdgeR wrapper)
>>> # Expression count matrix (EdgeR wrapper)
>>> fname = get_test_fname( 'mothur_datatypetest_false_2.mothur.freq' )
>>> Frequency().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -660,6 +667,7 @@ class Frequency(Tabular):
return False
@build_sniff_from_prefix
class Quantile(Tabular):
file_ext = 'mothur.quan'
MetadataElement(name="filtered", default=False, no_value=False, optional=True, desc="Quantiles calculated using a mask", readonly=True)
@@ -671,7 +679,7 @@ class Quantile(Tabular):
self.column_names = ['num', 'ten', 'twentyfive', 'fifty', 'seventyfive', 'ninetyfive', 'ninetynine']
self.column_types = ['int', 'float', 'float', 'float', 'float', 'float', 'float']
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a quantiles tabular format for chimera analysis
1 0 0 0 0 0 0
@@ -687,7 +695,7 @@ class Quantile(Tabular):
>>> Quantile().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@') and not line[0].startswith('#'):
@@ -710,10 +718,11 @@ class Quantile(Tabular):
return False
@build_sniff_from_prefix
class LaneMask(Text):
file_ext = 'mothur.filter'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a lane mask filter: 1 line consisting of zeros and ones.
@@ -725,7 +734,7 @@ class LaneMask(Text):
>>> LaneMask().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t', count=2)
headers = get_headers(file_prefix, sep='\t', count=2)
if len(headers) != 1 or len(headers[0]) != 1:
return False
@@ -775,6 +784,7 @@ class CountTable(Tabular):
dataset.metadata.data_lines -= 1
@build_sniff_from_prefix
class RefTaxonomy(Tabular):
file_ext = 'mothur.ref.taxonomy'
@@ -782,7 +792,7 @@ class RefTaxonomy(Tabular):
super(RefTaxonomy, self).__init__(**kwd)
self.column_names = ['name', 'taxonomy']
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is a Reference Taxonomy
@@ -808,7 +818,7 @@ class RefTaxonomy(Tabular):
>>> RefTaxonomy().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t', count=300)
headers = iter_headers(file_prefix, sep='\t', count=300)
count = 0
pat_prog = re.compile('^([^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?(;[^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?)*(;)?)$')
found_semicolons = False
@@ -852,6 +862,7 @@ class TaxonomySummary(Tabular):
self.column_names = ['taxlevel', 'rankID', 'taxon', 'daughterlevels', 'total']
@build_sniff_from_prefix
class Axes(Tabular):
file_ext = 'mothur.axes'
@@ -859,7 +870,7 @@ class Axes(Tabular):
"""Initialize axes datatype"""
super(Axes, self).__init__(**kwd)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is an axes format
The first line may have column headings.
@@ -883,7 +894,7 @@ class Axes(Tabular):
>>> Axes().sniff( fname )
False
"""
headers = iter_headers(filename, sep='\t')
headers = iter_headers(file_prefix, sep='\t')
count = 0
col_cnt = None
all_integers = True
+20 -24
View File
@@ -1,16 +1,21 @@
import abc
import logging
import os
import re
from galaxy.datatypes.binary import Binary
from galaxy.datatypes.data import get_file_peek, Text
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import build_sniff_from_prefix
from galaxy.datatypes.util import generic_util
from galaxy.util import nice_size
log = logging.getLogger(__name__)
STOCKHOLM_SEARCH_PATTERN = re.compile(r'#\s+STOCKHOLM\s+1\.0')
@build_sniff_from_prefix
class InfernalCM(Text):
file_ext = "cm"
@@ -32,20 +37,17 @@ class InfernalCM(Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
>>> from galaxy.datatypes.sniff import get_test_fname
>>> fname = get_test_fname( 'infernal_model.cm' )
>>> InfernalCM().sniff( fname )
True
>>> fname = get_test_fname( 'test.mz5' )
>>> fname = get_test_fname( '2.txt' )
>>> InfernalCM().sniff( fname )
False
"""
with open(filename, 'r') as f:
first_line = f.readline()
return first_line.startswith("INFERNAL")
return file_prefix.startswith("INFERNAL")
def set_meta(self, dataset, **kwd):
"""
@@ -58,6 +60,7 @@ class InfernalCM(Text):
dataset.metadata.cm_version = (first_line.split()[0]).replace('INFERNAL', '')
@build_sniff_from_prefix
class Hmmer(Text):
edam_data = "data_1364"
edam_format = "format_1370"
@@ -77,7 +80,7 @@ class Hmmer(Text):
return "HMMER database (%s)" % (nice_size(dataset.get_size()))
@abc.abstractmethod
def sniff(self, filename):
def sniff_prefix(self, filename):
raise NotImplementedError
@@ -85,24 +88,20 @@ class Hmmer2(Hmmer):
edam_format = "format_3328"
file_ext = "hmm2"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""HMMER2 files start with HMMER2.0
"""
with open(filename, 'r') as handle:
return handle.read(8) == 'HMMER2.0'
return False
return file_prefix.startswith('HMMER2.0')
class Hmmer3(Hmmer):
edam_format = "format_3329"
file_ext = "hmm3"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""HMMER3 files start with HMMER3/f
"""
with open(filename, 'r') as handle:
return handle.read(8) == 'HMMER3/f'
return False
return file_prefix.startswith('HMMER3/f')
class HmmerPress(Binary):
@@ -139,6 +138,7 @@ class HmmerPress(Binary):
self.add_composite_file('model.hmm.h3p', is_binary=True)
@build_sniff_from_prefix
class Stockholm_1_0(Text):
edam_data = "data_0863"
edam_format = "format_1961"
@@ -157,11 +157,8 @@ class Stockholm_1_0(Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
if generic_util.count_special_lines('^#[[:space:]+]STOCKHOLM[[:space:]+]1.0', filename) > 0:
return True
else:
return False
def sniff_prefix(self, file_prefix):
return file_prefix.search(STOCKHOLM_SEARCH_PATTERN)
def set_meta(self, dataset, **kwd):
"""
@@ -222,6 +219,7 @@ class Stockholm_1_0(Text):
split = classmethod(split)
@build_sniff_from_prefix
class MauveXmfa(Text):
file_ext = "xmfa"
@@ -238,10 +236,8 @@ class MauveXmfa(Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
with open(filename, 'r') as handle:
return handle.read(21) == '#FormatVersion Mauve1'
return False
def sniff_prefix(self, file_prefix):
return file_prefix.startswith('#FormatVersion Mauve1')
def set_meta(self, dataset, **kwd):
dataset.metadata.number_of_models = generic_util.count_special_lines('^#Sequence([[:digit:]]+)Entry', dataset.file_name)
+14 -12
View File
@@ -9,10 +9,12 @@ Phylip datatype sniffer
"""
from galaxy import util
from galaxy.datatypes.data import get_file_peek, Text
from galaxy.datatypes.sniff import build_sniff_from_prefix
from galaxy.util import nice_size
from .metadata import MetadataElement
@build_sniff_from_prefix
class Phylip(Text):
"""Phylip format stores a multiple sequence alignment"""
edam_data = "data_0863"
@@ -44,7 +46,7 @@ class Phylip(Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
All Phylip files starts with the number of sequences so we can use this
to count the following number of sequences in the first 'stack'
@@ -54,15 +56,15 @@ class Phylip(Text):
>>> Phylip().sniff(fname)
True
"""
with open(filename, "r") as f:
# Get number of sequence from first line
nb_seq = int(f.readline().split()[0])
# counts number of sequence from first stack
count = 0
for line in f:
if not line.split():
break
count += 1
if count > nb_seq:
return False
f = file_prefix.string_io()
# Get number of sequence from first line
nb_seq = int(f.readline().split()[0])
# counts number of sequence from first stack
count = 0
for line in f:
if not line.split():
break
count += 1
if count > nb_seq:
return False
return count == nb_seq
+22 -20
View File
@@ -3,13 +3,14 @@ import re
from galaxy.datatypes.data import get_file_peek, Text
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import get_headers
from galaxy.datatypes.sniff import build_sniff_from_prefix, get_headers
from galaxy.datatypes.tabular import Tabular
from galaxy.util import nice_size
log = logging.getLogger(__name__)
@build_sniff_from_prefix
class Smat(Text):
file_ext = "smat"
@@ -27,7 +28,7 @@ class Smat(Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
The use of ESTScan implies the creation of scores matrices which
reflect the codons preferences in the studied organisms. The
@@ -51,26 +52,27 @@ class Smat(Text):
True
"""
line_no = 0
with open(filename, "r") as fh:
for line in fh:
line_no += 1
if line_no > 10000:
return True
if line_no == 1 and not line.startswith('FORMAT'):
# The first line is always the start of a format section.
fh = file_prefix.string_io()
for line in fh:
line_no += 1
if line_no > 10000:
return True
if line_no == 1 and not line.startswith('FORMAT'):
# The first line is always the start of a format section.
return False
if not line.startswith('FORMAT'):
if line.find('\t') >= 0:
# Smat files are not tabular.
return False
if not line.startswith('FORMAT'):
if line.find('\t') >= 0:
# Smat files are not tabular.
items = line.split()
if len(items) != 4:
return False
for item in items:
# Make sure each item is an integer.
if re.match(r"[-+]?\d+$", item) is None:
return False
items = line.split()
if len(items) != 4:
return False
for item in items:
# Make sure each item is an integer.
if re.match(r"[-+]?\d+$", item) is None:
return False
return True
# Ensure at least a few matching lines are found.
return line_no > 2
# These commented classes are required by versions 1.0.0, 1.0.1 and 1.0.2 of the
+35 -33
View File
@@ -7,6 +7,7 @@ import re
from galaxy.datatypes import data
from galaxy.datatypes.binary import Binary
from galaxy.datatypes.data import Text
from galaxy.datatypes.sniff import build_sniff_from_prefix
from galaxy.datatypes.tabular import Tabular
from galaxy.datatypes.xml import GenericXml
from galaxy.util import nice_size
@@ -97,16 +98,16 @@ class ProteomicsXml(GenericXml):
edam_data = "data_2536"
edam_format = "format_2032"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
""" Determines whether the file is the correct XML type. """
with open(filename, 'r') as contents:
while True:
line = contents.readline()
if line is None or not line.startswith('<?'):
break
# pattern match <root or <ns:root for any ns string
pattern = '^<(\w*:)?%s' % self.root
return line is not None and re.match(pattern, line) is not None
contents = file_prefix.string_io()
while True:
line = contents.readline()
if line is None or not line.startswith('<?'):
break
# pattern match <root or <ns:root for any ns string
pattern = '^<(\w*:)?%s' % self.root
return line is not None and re.match(pattern, line) is not None
def set_peek(self, dataset, is_multi_byte=False):
"""Set the peek and blurb text"""
@@ -300,6 +301,7 @@ class ThermoRAW(Binary):
return "Thermo Finnigan RAW file (%s)" % (nice_size(dataset.get_size()))
@build_sniff_from_prefix
class Msp(Text):
""" Output of NIST MS Search Program chemdata.nist.gov/mass-spc/ftp/mass-spc/PepLib.pdf """
file_ext = "msp"
@@ -309,16 +311,15 @@ class Msp(Text):
next_line = contents.readline()
return next_line is not None and next_line.startswith(prefix)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
""" Determines whether the file is a NIST MSP output file."""
with open(filename, 'r') as f:
begin_contents = f.read(1024)
if "\n" not in begin_contents:
return False
lines = begin_contents.splitlines()
if len(lines) < 2:
return False
return lines[0].startswith("Name:") and lines[1].startswith("MW:")
begin_contents = file_prefix.contents_header
if "\n" not in begin_contents:
return False
lines = begin_contents.splitlines()
if len(lines) < 2:
return False
return lines[0].startswith("Name:") and lines[1].startswith("MW:")
class SPLibNoIndex(Text):
@@ -335,6 +336,7 @@ class SPLibNoIndex(Text):
dataset.blurb = 'file purged from disk'
@build_sniff_from_prefix
class SPLib(Msp):
"""SpectraST Spectral Library. Closely related to msp format"""
file_ext = "splib"
@@ -374,29 +376,29 @@ class SPLib(Msp):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
""" Determines whether the file is a SpectraST generated file.
"""
with open(filename, 'r') as contents:
return Msp.next_line_starts_with(contents, "Name:") and Msp.next_line_starts_with(contents, "LibID:")
contents = file_prefix.string_io()
return Msp.next_line_starts_with(contents, "Name:") and Msp.next_line_starts_with(contents, "LibID:")
@build_sniff_from_prefix
class Ms2(Text):
file_ext = "ms2"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
""" Determines whether the file is a valid ms2 file."""
with open(filename, 'r') as contents:
header_lines = []
while True:
line = contents.readline()
if line is None or len(line) == 0:
pass
elif line.startswith('H\t'):
header_lines.append(line)
else:
break
contents = file_prefix.string_io()
header_lines = []
while True:
line = contents.readline()
if line is None or len(line) == 0:
pass
elif line.startswith('H\t'):
header_lines.append(line)
else:
break
for header_field in ['CreationDate', 'Extractor', 'ExtractorVersion', 'ExtractorOptions']:
found_header = False
+53 -48
View File
@@ -3,7 +3,10 @@ Qualityscore class
"""
import logging
from . import data
from . import (
data,
sniff
)
log = logging.getLogger(__name__)
@@ -17,6 +20,7 @@ class QualityScore(data.Text):
file_ext = "qual"
@sniff.build_sniff_from_prefix
class QualityScoreSOLiD(QualityScore):
"""
until we know more about quality score formats
@@ -24,7 +28,7 @@ class QualityScoreSOLiD(QualityScore):
edam_format = "format_3610"
file_ext = "qualsolid"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
>>> from galaxy.datatypes.sniff import get_test_fname
>>> fname = get_test_fname( 'sequence.fasta' )
@@ -34,34 +38,34 @@ class QualityScoreSOLiD(QualityScore):
>>> QualityScoreSOLiD().sniff( fname )
True
"""
with open(filename) as fh:
readlen = None
goodblock = 0
while True:
line = fh.readline()
if not line:
if goodblock > 0:
fh = file_prefix.string_io()
readlen = None
goodblock = 0
while True:
line = fh.readline()
if not line:
if goodblock > 0:
return True
else:
break # EOF
line = line.strip()
if line and not line.startswith('#'): # first non-empty non-comment line
if line.startswith('>'):
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
try:
[int(x) for x in line.split()]
if not(readlen):
readlen = len(line.split())
assert len(line.split()) == readlen # SOLiD reads should be of the same length
except Exception:
break
goodblock += 1
if goodblock > 10:
return True
else:
break # EOF
line = line.strip()
if line and not line.startswith('#'): # first non-empty non-comment line
if line.startswith('>'):
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
try:
[int(x) for x in line.split()]
if not(readlen):
readlen = len(line.split())
assert len(line.split()) == readlen # SOLiD reads should be of the same length
except Exception:
break
goodblock += 1
if goodblock > 10:
return True
else:
break # we found a non-empty line, but it's not a header
else:
break # we found a non-empty line, but it's not a header
return False
def set_meta(self, dataset, **kwd):
@@ -71,6 +75,7 @@ class QualityScoreSOLiD(QualityScore):
return QualityScore.set_meta(self, dataset, **kwd)
@sniff.build_sniff_from_prefix
class QualityScore454(QualityScore):
"""
until we know more about quality score formats
@@ -78,7 +83,7 @@ class QualityScore454(QualityScore):
edam_format = "format_3611"
file_ext = "qual454"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
>>> from galaxy.datatypes.sniff import get_test_fname
>>> fname = get_test_fname( 'sequence.fasta' )
@@ -88,24 +93,24 @@ class QualityScore454(QualityScore):
>>> QualityScore454().sniff( fname )
True
"""
with open(filename) as fh:
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line and not line.startswith('#'): # first non-empty non-comment line
if line.startswith('>'):
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
try:
[int(x) for x in line.split()]
except Exception:
break
return True
else:
break # we found a non-empty line, but it's not a header
fh = file_prefix.string_io()
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line and not line.startswith('#'): # first non-empty non-comment line
if line.startswith('>'):
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
try:
[int(x) for x in line.split()]
except Exception:
break
return True
else:
break # we found a non-empty line, but it's not a header
return False
+120 -116
View File
@@ -22,8 +22,9 @@ from galaxy.datatypes.binary import (
)
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import (
build_sniff_from_prefix,
get_headers,
iter_headers
iter_headers,
)
from galaxy.util import (
compression_utils,
@@ -46,6 +47,7 @@ SNIFF_COMPRESSED_FASTAS = os.environ.get("GALAXY_ENABLE_BETA_COMPRESSED_FASTA_SN
SNIFF_COMPRESSED_GENBANKS = os.environ.get("GALAXY_ENABLE_BETA_COMPRESSED_GENBANK_SNIFFING", "0") == "1"
@build_sniff_from_prefix
class SequenceSplitLocations(data.Text):
"""
Class storing information about a sequence file composed of multiple gzip files concatenated as
@@ -75,10 +77,10 @@ class SequenceSplitLocations(data.Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
if os.path.getsize(filename) < 50000:
def sniff_prefix(self, file_prefix):
if file_prefix.file_size < 50000 and not file_prefix.truncated:
try:
data = json.load(open(filename))
data = json.loads(file_prefix.contents_header)
sections = data['sections']
for section in sections:
if 'start' not in section or 'end' not in section or 'sequences' not in section:
@@ -330,12 +332,13 @@ class FastaGz(Sequence, CompressedArchive):
return Sequence.sniff(self, filename)
@build_sniff_from_prefix
class Fasta(Sequence):
"""Class representing a FASTA sequence"""
edam_format = "format_1929"
file_ext = "fasta"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in fasta format
@@ -369,26 +372,26 @@ class Fasta(Sequence):
>>> Fasta().sniff( fname )
True
"""
with open(filename) as fh:
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if line.startswith('>'):
# The next line.strip() must not be '', nor startwith '>'
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
fh = file_prefix.string_io()
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line: # first non-empty line
if line.startswith('>'):
# The next line.strip() must not be '', nor startwith '>'
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
# If there is a third line, and it isn't a header line, it may not contain chars like '()[].' otherwise it's most likely a DotBracket file
line = fh.readline()
if not line.startswith('>') and re.search("[\(\)\[\]\.]", line):
break
return True
else:
break # we found a non-empty line, but it's not a fasta header
# If there is a third line, and it isn't a header line, it may not contain chars like '()[].' otherwise it's most likely a DotBracket file
line = fh.readline()
if not line.startswith('>') and re.search("[\(\)\[\]\.]", line):
break
return True
else:
break # we found a non-empty line, but it's not a fasta header
return False
def split(cls, input_datasets, subdir_generator_function, split_params):
@@ -514,12 +517,13 @@ class Fasta(Sequence):
_count_split = classmethod(_count_split)
@build_sniff_from_prefix
class csFasta(Sequence):
""" Class representing the SOLID Color-Space sequence ( csfasta ) """
edam_format = "format_3589"
file_ext = "csfasta"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Color-space sequence:
>2_15_85_F3
@@ -533,24 +537,24 @@ class csFasta(Sequence):
>>> csFasta().sniff( fname )
True
"""
with open(filename) as fh:
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line and not line.startswith('#'): # first non-empty non-comment line
if line.startswith('>'):
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
elif line[0] not in string.ascii_uppercase:
return False
elif len(line) > 1 and not re.search('^[\d.]+$', line[1:]):
return False
return True
else:
break # we found a non-empty line, but it's not a header
fh = file_prefix.string_io()
while True:
line = fh.readline()
if not line:
break # EOF
line = line.strip()
if line and not line.startswith('#'): # first non-empty non-comment line
if line.startswith('>'):
line = fh.readline().strip()
if line == '' or line.startswith('>'):
break
elif line[0] not in string.ascii_uppercase:
return False
elif len(line) > 1 and not re.search('^[\d.]+$', line[1:]):
return False
return True
else:
break # we found a non-empty line, but it's not a header
return False
def set_meta(self, dataset, **kwd):
@@ -561,6 +565,7 @@ class csFasta(Sequence):
return Sequence.set_meta(self, dataset, **kwd)
@build_sniff_from_prefix
class BaseFastq(Sequence):
"""Base class for FastQ sequences"""
edam_format = "format_1930"
@@ -599,7 +604,7 @@ class BaseFastq(Sequence):
dataset.metadata.data_lines = data_lines
dataset.metadata.sequences = sequences
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in generic fastq format
For details, see http://maq.sourceforge.net/fastq.shtml
@@ -620,11 +625,10 @@ class BaseFastq(Sequence):
>>> FastqSanger().sniff( fname )
False
"""
compressed = is_gzip(filename) or is_bz2(filename)
compressed = file_prefix.compressed_format is not None
if compressed and not isinstance(self, Binary):
return False
headers = iter_headers(filename, None, count=1000)
headers = iter_headers(file_prefix, None, count=1000)
# If this is a FastqSanger-derived class, then check to see if the base qualities match
if isinstance(self, FastqSanger) or isinstance(self, FastqSangerGz) or isinstance(self, FastqSangerBz2):
if not self.sangerQualities(headers):
@@ -633,7 +637,7 @@ class BaseFastq(Sequence):
bases_regexp = re.compile("^[NGTAC]*")
# check that first block looks like a fastq block
try:
headers = get_headers(filename, None, count=4)
headers = get_headers(file_prefix, None, count=4)
if len(headers) == 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]:
# Check the sequence line, make sure it contains only G/C/A/T/N
if not bases_regexp.match(headers[1][0]):
@@ -818,6 +822,7 @@ class FastqCSSangerBz2(FastqBz2):
file_ext = "fastqcssanger.bz2"
@build_sniff_from_prefix
class Maf(Alignment):
"""Class describing a Maf alignment"""
edam_format = "format_3008"
@@ -900,7 +905,7 @@ class Maf(Alignment):
out = "Can't create peek %s" % exc
return out
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines wether the file is in maf format
@@ -923,7 +928,7 @@ class Maf(Alignment):
>>> Maf().sniff( fname )
False
"""
headers = get_headers(filename, None)
headers = get_headers(file_prefix, None)
try:
if len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf":
return True
@@ -971,6 +976,7 @@ class MafCustomTrack(data.Text):
pass
@build_sniff_from_prefix
class Axt(data.Text):
"""Class describing an axt alignment"""
# gvk- 11/19/09 - This is really an alignment, but we no longer have tools that use this data type, and it is
@@ -981,7 +987,7 @@ class Axt(data.Text):
edam_format = "format_3013"
file_ext = "axt"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in axt format
@@ -1007,7 +1013,7 @@ class Axt(data.Text):
>>> Axt().sniff( fname )
False
"""
headers = get_headers(filename, None)
headers = get_headers(file_prefix, None)
if len(headers) < 4:
return False
for hdr in headers:
@@ -1026,6 +1032,7 @@ class Axt(data.Text):
return True
@build_sniff_from_prefix
class Lav(data.Text):
"""Class describing a LAV alignment"""
# gvk- 11/19/09 - This is really an alignment, but we no longer have tools that use this data type, and it is
@@ -1036,7 +1043,7 @@ class Lav(data.Text):
edam_format = "format_3014"
file_ext = "lav"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in lav format
@@ -1053,7 +1060,7 @@ class Lav(data.Text):
>>> Lav().sniff( fname )
False
"""
headers = get_headers(filename, None)
headers = get_headers(file_prefix, None)
try:
if len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav'):
return True
@@ -1096,6 +1103,7 @@ class RNADotPlotMatrix(data.Data):
return False
@build_sniff_from_prefix
class DotBracket(Sequence):
edam_data = "data_0880"
edam_format = "format_1457"
@@ -1128,7 +1136,7 @@ class DotBracket(Sequence):
dataset.metadata.data_lines = data_lines
dataset.metadata.sequences = sequences
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Galaxy Dbn (Dot-Bracket notation) rules:
@@ -1160,47 +1168,47 @@ class DotBracket(Sequence):
state = 0
with open(filename, "r") as handle:
for line in handle:
line = line.strip()
for line in file_prefix.line_iterator():
line = line.strip()
if line:
# header line
if state == 0:
if(line[0] != '>'):
return False
else:
state = 1
if line:
# header line
if state == 0:
if(line[0] != '>'):
return False
else:
state = 1
# sequence line
elif state == 1:
if not self.sequence_regexp.match(line):
return False
else:
sequence_size = len(line)
state = 2
# sequence line
elif state == 1:
if not self.sequence_regexp.match(line):
return False
else:
sequence_size = len(line)
state = 2
# dot-bracket structure line
elif state == 2:
if sequence_size != len(line) or not self.structure_regexp.match(line) or \
line.count('(') != line.count(')') or \
line.count('[') != line.count(']') or \
line.count('{') != line.count('}'):
return False
else:
return True
# dot-bracket structure line
elif state == 2:
if sequence_size != len(line) or not self.structure_regexp.match(line) or \
line.count('(') != line.count(')') or \
line.count('[') != line.count(']') or \
line.count('{') != line.count('}'):
return False
else:
return True
# Number of lines is less than 3
return False
@build_sniff_from_prefix
class Genbank(data.Text):
"""Class representing a Genbank sequence"""
edam_format = "format_1936"
edam_data = "data_0849"
file_ext = "genbank"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determine whether the file is in genbank format.
Works for compressed files.
@@ -1210,15 +1218,10 @@ class Genbank(data.Text):
>>> Genbank().sniff( fname )
True
"""
compressed = is_gzip(filename)
compressed = file_prefix.compressed_format
if compressed and not isinstance(self, Binary):
return False
try:
with compression_utils.get_fileobj(filename) as file:
return 'LOCUS ' == file.read(6)
except Exception:
pass
return False
return 'LOCUS ' == file_prefix.contents_header[0:6]
class GenbankGz(Genbank, CompressedArchive):
@@ -1244,11 +1247,12 @@ class GenbankGz(Genbank, CompressedArchive):
return Genbank.sniff(self, filename)
@build_sniff_from_prefix
class MemePsp(Sequence):
"""Class representing MEME Position Specific Priors"""
file_ext = "memepsp"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
The format of an entry in a PSP file is:
@@ -1274,34 +1278,34 @@ class MemePsp(Sequence):
return True
try:
num_lines = 0
with open(filename) as fh:
line = fh.readline()
if not line:
# EOF.
return False
num_lines += 1
if num_lines > 100:
return True
line = line.strip()
if line:
if line.startswith('>'):
# The line must not be blank, nor start with '>'
line = fh.readline().strip()
if line == '' or line.startswith('>'):
return False
# All items within the line must be floats.
fh = file_prefix.string_io()
line = fh.readline()
if not line:
# EOF.
return False
num_lines += 1
if num_lines > 100:
return True
line = line.strip()
if line:
if line.startswith('>'):
# The line must not be blank, nor start with '>'
line = fh.readline().strip()
if line == '' or line.startswith('>'):
return False
# All items within the line must be floats.
if not floats_verified(line):
return False
# If there is a second line within the ID section,
# all items within the line must be floats.
line = fh.readline().strip()
if line:
if not floats_verified(line):
return False
# If there is a second line within the ID section,
# all items within the line must be floats.
line = fh.readline().strip()
if line:
if not floats_verified(line):
return False
else:
# We found a non-empty line,
# but it's not a psp id width.
return False
else:
# We found a non-empty line,
# but it's not a psp id width.
return False
except Exception:
return False
# We've reached EOF in less than 100 lines.
+235 -41
View File
@@ -13,7 +13,7 @@ import sys
import tempfile
import zipfile
from six import text_type
from six import StringIO, text_type
from six.moves import filter
from six.moves.urllib.request import urlopen
@@ -35,6 +35,8 @@ else:
log = logging.getLogger(__name__)
SNIFF_PREFIX_BYTES = int(os.environ.get("GALAXY_SNIFF_PREFIX_BYTES", None) or 2 ** 20)
def get_test_fname(fname):
"""Returns test data filename"""
@@ -189,10 +191,10 @@ def convert_newlines_sep2tabs(fname, in_place=True, patt="\\s+", tmp_dir=None, t
return (i + 1, temp_name)
def iter_headers(fname, sep, count=60, comment_designator=None):
with compression_utils.get_fileobj(fname) as in_file:
def iter_headers(fname_or_file_prefix, sep, count=60, comment_designator=None):
if isinstance(fname_or_file_prefix, FilePrefix):
idx = 0
for line in in_file:
for line in fname_or_file_prefix.line_iterator():
line = line.rstrip('\n\r')
if comment_designator is not None and comment_designator != '' and line.startswith(comment_designator):
continue
@@ -200,9 +202,20 @@ def iter_headers(fname, sep, count=60, comment_designator=None):
idx += 1
if idx == count:
break
else:
with compression_utils.get_fileobj(fname_or_file_prefix) as in_file:
idx = 0
for line in in_file:
line = line.rstrip('\n\r')
if comment_designator is not None and comment_designator != '' and line.startswith(comment_designator):
continue
yield line.split(sep)
idx += 1
if idx == count:
break
def get_headers(fname, sep, count=60, comment_designator=None):
def get_headers(fname_or_file_prefix, sep, count=60, comment_designator=None):
"""
Returns a list with the first 'count' lines split by 'sep', ignoring lines
starting with 'comment_designator'
@@ -214,10 +227,10 @@ def get_headers(fname, sep, count=60, comment_designator=None):
>>> get_headers(fname, '\\t', count=5, comment_designator='#') == [[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']]
True
"""
return list(iter_headers(fname=fname, sep=sep, count=count, comment_designator=comment_designator))
return list(iter_headers(fname_or_file_prefix=fname_or_file_prefix, sep=sep, count=count, comment_designator=comment_designator))
def is_column_based(fname, sep='\t', skip=0):
def is_column_based(fname_or_file_prefix, sep='\t', skip=0):
"""
Checks whether the file is column based with respect to a separator
(defaults to tab separator).
@@ -245,8 +258,11 @@ def is_column_based(fname, sep='\t', skip=0):
>>> is_column_based(fname)
True
"""
if getattr(fname_or_file_prefix, "binary", None) is True:
return False
try:
headers = get_headers(fname, sep)
headers = get_headers(fname_or_file_prefix, sep)
except UnicodeDecodeError:
return False
count = 0
@@ -306,17 +322,14 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> fname = get_test_fname('gff_version_3.gff')
>>> guess_ext(fname, sniff_order)
'gff3'
>>> fname = get_test_fname('temp.txt')
>>> open(fname, 'wt').write("a\\t2")
>>> guess_ext(fname, sniff_order)
>>> fname = get_test_fname('2.txt')
>>> guess_ext(fname, sniff_order) # 2.txt
'txt'
>>> fname = get_test_fname('temp.txt')
>>> open(fname, 'wt').write("a\\t2\\nc\\t1\\nd\\t0")
>>> fname = get_test_fname('2.tabular')
>>> guess_ext(fname, sniff_order)
'tabular'
>>> fname = get_test_fname('temp.txt')
>>> open(fname, 'wt').write("a 1 2 x\\nb 3 4 y\\nc 5 6 z")
>>> guess_ext(fname, sniff_order)
>>> fname = get_test_fname('3.txt')
>>> guess_ext(fname, sniff_order) # 3.txt
'txt'
>>> fname = get_test_fname('test_tab1.tabular')
>>> guess_ext(fname, sniff_order)
@@ -363,6 +376,29 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.otu')
>>> guess_ext(fname, sniff_order)
'mothur.otu'
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.lower.dist')
>>> guess_ext(fname, sniff_order)
'mothur.lower.dist'
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.square.dist')
>>> guess_ext(fname, sniff_order)
'mothur.square.dist'
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.pair.dist')
>>> guess_ext(fname, sniff_order)
'mothur.pair.dist'
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.freq')
>>> guess_ext(fname, sniff_order)
'mothur.freq'
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.quan')
>>> guess_ext(fname, sniff_order)
'mothur.quan'
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.ref.taxonomy')
>>> guess_ext(fname, sniff_order)
'mothur.ref.taxonomy'
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.axes')
>>> guess_ext(fname, sniff_order)
'mothur.axes'
>>> guess_ext(get_test_fname('infernal_model.cm'), sniff_order)
'cm'
>>> fname = get_test_fname('1.gg')
>>> guess_ext(fname, sniff_order)
'gg'
@@ -378,7 +414,83 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> fname = get_test_fname('454Score.pdf')
>>> guess_ext(fname, sniff_order)
'pdf'
>>> fname = get_test_fname('1.obo')
>>> guess_ext(fname, sniff_order)
'obo'
>>> fname = get_test_fname('1.arff')
>>> guess_ext(fname, sniff_order)
'arff'
>>> fname = get_test_fname('1.afg')
>>> guess_ext(fname, sniff_order)
'afg'
>>> fname = get_test_fname('1.owl')
>>> guess_ext(fname, sniff_order)
'owl'
>>> fname = get_test_fname('Acanium.hmm')
>>> guess_ext(fname, sniff_order)
'snaphmm'
>>> fname = get_test_fname('wiggle.wig')
>>> guess_ext(fname, sniff_order)
'wig'
>>> fname = get_test_fname('example.iqtree')
>>> guess_ext(fname, sniff_order)
'iqtree'
>>> fname = get_test_fname('1.stockholm')
>>> guess_ext(fname, sniff_order)
'stockholm'
>>> fname = get_test_fname('1.xmfa')
>>> guess_ext(fname, sniff_order)
'xmfa'
>>> fname = get_test_fname('test.phylip')
>>> guess_ext(fname, sniff_order)
'phylip'
>>> fname = get_test_fname('1.smat')
>>> guess_ext(fname, sniff_order)
'smat'
>>> fname = get_test_fname('1.ttl')
>>> guess_ext(fname, sniff_order)
'ttl'
>>> fname = get_test_fname('1.hdt')
>>> guess_ext(fname, sniff_order)
'hdt'
>>> fname = get_test_fname('1.phyloxml')
>>> guess_ext(fname, sniff_order)
'phyloxml'
"""
file_prefix = FilePrefix(fname)
file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary)
# Ugly hack for tsv vs tabular sniffing, we want to prefer tabular
# to tsv but it doesn't have a sniffer - is TSV was sniffed just check
# if it is an okay tabular and use that instead.
if file_ext == 'tsv':
if is_column_based(file_prefix, '\t', 1):
file_ext = 'tabular'
if file_ext is not None:
return file_ext
# skip header check if data is already known to be binary
if is_binary:
return file_ext or 'binary'
try:
get_headers(file_prefix, None)
except UnicodeDecodeError:
return 'data' # default data type file extension
if is_column_based(file_prefix, '\t', 1):
return 'tabular' # default tabular data type file extension
return 'txt' # default text data type file extension
def run_sniffers_raw(filename_or_file_prefix, sniff_order, is_binary=False):
"""Run through sniffers specified by sniff_order, return None of None match.
"""
if isinstance(filename_or_file_prefix, FilePrefix):
fname = filename_or_file_prefix.filename
file_prefix = filename_or_file_prefix
else:
fname = filename_or_file_prefix
file_prefix = FilePrefix(filename_or_file_prefix)
file_ext = None
for datatype in sniff_order:
"""
@@ -390,31 +502,29 @@ def guess_ext(fname, sniff_order, is_binary=False):
successfully discovered.
"""
try:
if ((is_binary and datatype.is_binary) or
(not is_binary)) and datatype.sniff(fname):
if hasattr(datatype, "sniff_prefix"):
datatype_compressed = getattr(datatype, "compressed", False)
if datatype_compressed and not file_prefix.compressed_format:
continue
if not datatype_compressed and file_prefix.compressed_format:
continue
if file_prefix.compressed_format and getattr(datatype, "compressed_format"):
# In this case go a step further and compare the compressed format detected
# to the expected.
if file_prefix.compressed_format != datatype.compressed_format:
continue
if datatype.sniff_prefix(file_prefix):
file_ext = datatype.file_ext
break
elif is_binary and not datatype.is_binary:
continue
elif datatype.sniff(fname):
file_ext = datatype.file_ext
break
except Exception:
pass
# Ugly hack for tsv vs tabular sniffing, we want to prefer tabular
# to tsv but it doesn't have a sniffer - is TSV was sniffed just check
# if it is an okay tabular and use that instead.
if file_ext == 'tsv':
if is_column_based(fname, '\t', 1):
file_ext = 'tabular'
if file_ext is not None:
return file_ext
# skip header check if data is already known to be binary
if is_binary:
return file_ext or 'binary'
try:
get_headers(fname, None)
except UnicodeDecodeError:
return 'data' # default data type file extension
if is_column_based(fname, '\t', 1):
return 'tabular' # default tabular data type file extension
return 'txt' # default text data type file extension
return file_ext
def zip_single_fileobj(path):
@@ -424,6 +534,91 @@ def zip_single_fileobj(path):
return z.open(name)
class FilePrefix(object):
def __init__(self, filename):
binary = False
compressed_format = None
contents_header = None # First MAX_BYTES of the file.
truncated = False
# A future direction to optimize sniffing even more for sniffers at the top of the list
# is to lazy load contents_header based on what interface is requested. For instance instead
# of returning a StringIO directly in string_io() return an object that reads the contents and
# populates contents_header while providing a StringIO-like interface until the file is read
# but then would fallback to native string_io()
try:
compressed_format, f = compression_utils.get_fileobj_raw(filename)
try:
contents_header = f.read(SNIFF_PREFIX_BYTES)
truncated = len(contents_header) == SNIFF_PREFIX_BYTES
finally:
f.close()
except UnicodeDecodeError:
binary = True
self.truncated = truncated
self.filename = filename
self.binary = binary
self.compressed_format = compressed_format
self.contents_header = contents_header
self._file_size = None
@property
def file_size(self):
if self._file_size is None:
self._file_size = os.path.getsize(self.filename)
return self._file_size
def string_io(self):
if self.binary:
raise Exception("Attempting to create a StringIO object for binary data.")
rval = StringIO(self.contents_header)
return rval
def startswith(self, prefix):
return self.string_io().read(len(prefix)) == prefix
def line_iterator(self):
s = self.string_io()
for line in s:
if line.endswith("\n") or line.endswith("\r"):
yield line
elif s.pos == s.len and not self.truncated:
# At the end, return the last line if it wasn't truncated when reading it in.
yield line
# Convenience wrappers around contents_header, shielding contents_header means we can
# potentially do a better job lazy loading this data later on.
def search(self, pattern):
return pattern.search(self.contents_header)
def search_str(self, query_str):
return query_str in self.contents_header
def build_sniff_from_prefix(klass):
def auto_sniff(self, filename):
file_prefix = FilePrefix(filename)
datatype_compressed = getattr(self, "compressed", False)
if file_prefix.compressed_format and not datatype_compressed:
return False
if datatype_compressed and not file_prefix.compressed_format:
return False
if hasattr(self, "compressed_format"):
if self.compressed_format != file_prefix.compressed_format:
return False
return self.sniff_prefix(file_prefix)
klass.sniff = auto_sniff
return klass
def disable_parent_class_sniffing(klass):
klass.sniff = lambda self, filename: False
klass.sniff_prefix = lambda self, file_prefix: False
return klass
def handle_compressed_file(
filename,
datatypes_registry,
@@ -464,11 +659,10 @@ def handle_compressed_file(
if ext in AUTO_DETECT_EXTENSIONS:
# attempt to sniff for a keep-compressed datatype (observing the sniff order)
sniff_datatypes = filter(lambda d: getattr(d, 'compressed', False), datatypes_registry.sniff_order)
for datatype in sniff_datatypes:
if datatype.sniff(filename):
ext = datatype.file_ext
keep_compressed = True
break
sniffed_ext = run_sniffers_raw(filename, sniff_datatypes)
if sniffed_ext:
ext = sniffed_ext
keep_compressed = True
else:
datatype = datatypes_registry.get_datatype_by_extension(ext)
keep_compressed = getattr(datatype, 'compressed', False)
+106 -106
View File
@@ -21,11 +21,11 @@ from galaxy import util
from galaxy.datatypes import binary, data, metadata
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import (
build_sniff_from_prefix,
get_headers,
iter_headers
)
from galaxy.util import compression_utils
from galaxy.util.checkers import is_gzip
from . import dataproviders
if sys.version_info > (3,):
@@ -416,6 +416,7 @@ class Taxonomy(Tabular):
@dataproviders.decorators.has_dataproviders
@build_sniff_from_prefix
class Sam(Tabular):
edam_format = "format_2573"
edam_data = "data_0863"
@@ -434,7 +435,7 @@ class Sam(Tabular):
"""Returns formated html of peek"""
return self.make_html_table(dataset, column_names=self.column_names)
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in SAM format
@@ -463,31 +464,31 @@ class Sam(Tabular):
>>> Sam().sniff( fname )
True
"""
with open(filename) as fh:
count = 0
while True:
line = fh.readline()
line = line.strip()
if not line:
break # EOF
if line:
if line[0] != '@':
line_pieces = line.split('\t')
if len(line_pieces) < 11:
return False
try:
int(line_pieces[1])
int(line_pieces[3])
int(line_pieces[4])
int(line_pieces[7])
int(line_pieces[8])
except ValueError:
return False
count += 1
if count == 5:
return True
if count < 5 and count > 0:
return True
fh = file_prefix.string_io()
count = 0
while True:
line = fh.readline()
line = line.strip()
if not line:
break # EOF
if line:
if line[0] != '@':
line_pieces = line.split('\t')
if len(line_pieces) < 11:
return False
try:
int(line_pieces[1])
int(line_pieces[3])
int(line_pieces[4])
int(line_pieces[7])
int(line_pieces[8])
except ValueError:
return False
count += 1
if count == 5:
return True
if count < 5 and count > 0:
return True
return False
def set_meta(self, dataset, overwrite=True, skip=None, max_data_lines=5, **kwd):
@@ -592,6 +593,7 @@ class Sam(Tabular):
@dataproviders.decorators.has_dataproviders
@build_sniff_from_prefix
class Pileup(Tabular):
"""Tab delimited data in pileup (6- or 10-column) format"""
edam_format = "format_3015"
@@ -616,7 +618,7 @@ class Pileup(Tabular):
"""Return options for removing errors along with a description"""
return [("lines", "Remove erroneous lines")]
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Checks for 'pileup-ness'
@@ -634,24 +636,32 @@ class Pileup(Tabular):
>>> fname = get_test_fname( '10col.pileup' )
>>> Pileup().sniff( fname )
True
>>> fname = get_test_fname( '1.xls' )
>>> Pileup().sniff( fname )
False
>>> fname = get_test_fname( '2.txt' )
>>> Pileup().sniff( fname ) # 2.txt
False
>>> fname = get_test_fname( '2.tabular' )
>>> Pileup().sniff( fname )
False
"""
headers = iter_headers(filename, '\t')
found_non_comment_lines = False
try:
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and not hdr[0].startswith('#'):
if len(hdr) < 5:
return False
try:
# chrom start in column 1 (with 0-based columns)
# and reference base is in column 2
chrom = int(hdr[1])
assert chrom >= 0
assert hdr[2] in ['A', 'C', 'G', 'T', 'N', 'a', 'c', 'g', 't', 'n']
except Exception:
return False
return True
# chrom start in column 1 (with 0-based columns)
# and reference base is in column 2
chrom = int(hdr[1])
assert chrom >= 0
assert hdr[2] in ['A', 'C', 'G', 'T', 'N', 'a', 'c', 'g', 't', 'n']
found_non_comment_lines = True
except Exception:
return False
return found_non_comment_lines
# Dataproviders
@dataproviders.decorators.dataprovider_factory('genomic-region',
@@ -667,6 +677,7 @@ class Pileup(Tabular):
@dataproviders.decorators.has_dataproviders
@build_sniff_from_prefix
class BaseVcf(Tabular):
""" Variant Call Format for describing SNPs and other simple genome variations. """
edam_format = "format_3016"
@@ -680,15 +691,12 @@ class BaseVcf(Tabular):
MetadataElement(name="viz_filter_cols", desc="Score column for visualization", default=[5], param=metadata.ColumnParameter, optional=True, multiple=True, visible=False)
MetadataElement(name="sample_names", default=[], desc="Sample names", readonly=True, visible=False, optional=True, no_value=[])
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
# Because this sniffer is run on compressed files that might be BGZF (due to the VcfGz subclass), we should
# handle unicode decode errors. This should ultimately be done in get_headers(), but guess_ext() currently
# relies on get_headers() raising this exception.
try:
headers = get_headers(filename, '\n', count=1)
return headers[0][0].startswith("##fileformat=VCF")
except UnicodeDecodeError:
return False
headers = get_headers(file_prefix, '\n', count=1)
return headers[0][0].startswith("##fileformat=VCF")
def display_peek(self, dataset):
"""Returns formated html of peek"""
@@ -737,23 +745,14 @@ class BaseVcf(Tabular):
class Vcf(BaseVcf):
file_ext = 'vcf'
def sniff(self, filename):
if is_gzip(filename):
return False
return super(Vcf, self).sniff(filename)
class VcfGz(BaseVcf, binary.Binary):
file_ext = 'vcf_bgzip'
compressed = True
compressed_format = "gzip"
MetadataElement(name="tabix_index", desc="Vcf Index File", param=metadata.FileParameter, file_ext="tbi", readonly=True, no_value=None, visible=False, optional=True)
def sniff(self, filename):
if not is_gzip(filename):
return False
return super(VcfGz, self).sniff(filename)
def set_meta(self, dataset, **kwd):
super(BaseVcf, self).set_meta(dataset, **kwd)
""" Creates the index for the VCF file. """
@@ -769,8 +768,11 @@ class VcfGz(BaseVcf, binary.Binary):
dataset.metadata.tabix_index = index_file
@build_sniff_from_prefix
class Eland(Tabular):
"""Support for the export.txt.gz file used by Illumina's ELANDv2e aligner"""
compressed = True
compressed_format = "gzip"
file_ext = '_export.txt.gz'
MetadataElement(name="columns", default=0, desc="Number of columns", readonly=True, visible=False)
MetadataElement(name="column_types", default=[], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False, no_value=[])
@@ -811,7 +813,7 @@ class Eland(Tabular):
out = "Can't create peek %s" % exc
return out
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in ELAND export format
@@ -824,32 +826,31 @@ class Eland(Tabular):
- LANE, TILEm X, Y, INDEX, READ_NO, SEQ, QUAL, POSITION, *STRAND, FILT must be correct
- We will only check that up to the first 5 alignments are correctly formatted.
"""
with compression_utils.get_fileobj(filename, compressed_formats=['gzip']) as fh:
count = 0
while True:
line = fh.readline()
line = line.strip()
if not line:
break # EOF
if line:
line_pieces = line.split('\t')
if len(line_pieces) != 22:
return False
if long(line_pieces[1]) < 0:
raise Exception('Out of range')
if long(line_pieces[2]) < 0:
raise Exception('Out of range')
if long(line_pieces[3]) < 0:
raise Exception('Out of range')
int(line_pieces[4])
int(line_pieces[5])
# can get a lot more specific
count += 1
if count == 5:
break
if count > 0:
return True
return False
fh = file_prefix.string_io()
count = 0
while True:
line = fh.readline()
line = line.strip()
if not line:
break # EOF
if line:
line_pieces = line.split('\t')
if len(line_pieces) != 22:
return False
if long(line_pieces[1]) < 0:
raise Exception('Out of range')
if long(line_pieces[2]) < 0:
raise Exception('Out of range')
if long(line_pieces[3]) < 0:
raise Exception('Out of range')
int(line_pieces[4])
int(line_pieces[5])
# can get a lot more specific
count += 1
if count == 5:
break
if count > 0:
return True
def set_meta(self, dataset, overwrite=True, skip=None, max_data_lines=5, **kwd):
if dataset.has_data():
@@ -884,10 +885,11 @@ class Eland(Tabular):
dataset.metadata.reads = list(reads.keys())
@build_sniff_from_prefix
class ElandMulti(Tabular):
file_ext = 'elandmulti'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
return False
@@ -1056,6 +1058,7 @@ class TSV(BaseCSV):
strict_width = True # Leave files with different width to tabular
@build_sniff_from_prefix
class ConnectivityTable(Tabular):
edam_format = "format_3309"
file_ext = "ct"
@@ -1077,7 +1080,7 @@ class ConnectivityTable(Tabular):
dataset.metadata.data_lines = data_lines
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
The ConnectivityTable (CT) is a file format used for describing
RNA 2D structures by tools including MFOLD, UNAFOLD and
@@ -1112,31 +1115,28 @@ class ConnectivityTable(Tabular):
i = 0
j = 1
try:
with open(filename) as handle:
for line in handle:
line = line.strip()
handle = file_prefix.string_io()
for line in handle:
line = line.strip()
if len(line) > 0:
if i == 0:
if not self.header_regexp.match(line):
return False
else:
length = int(re.split('\W+', line, 1)[0])
if len(line) > 0:
if i == 0:
if not self.header_regexp.match(line):
return False
else:
length = int(re.split('\W+', line, 1)[0])
else:
if not self.structure_regexp.match(line.upper()):
return False
else:
if j != int(re.split('\W+', line, 1)[0]):
return False
elif j == length: # Last line of first sequence has been recheached
return True
else:
if not self.structure_regexp.match(line.upper()):
return False
else:
if j != int(re.split('\W+', line, 1)[0]):
return False
elif j == length: # Last line of first sequence has been recheached
return True
else:
j += 1
i += 1
return False
except Exception:
return False
j += 1
i += 1
return False
def get_chunk(self, trans, dataset, chunk):
ck_index = int(chunk)
+67
View File
@@ -0,0 +1,67 @@
{UNV
iid:1
com:
Generated by dsommer with tarchive2amos on Wed Aug 30 13:10:59 2006
.
}
{RED
iid:1
eid:zfishG-a2661d04.q1c
seq:
TAAAATAAATGTTATGTTATCATGTTGACAGATCAATGATAAAATAAAGCCTGGTGATTA
AAAACCTGCAATACCTTGACAAGAACTTTCATGTAAACTAAAGTACTAACTAAAAAAGTG
TCCTGAGAAATCTCGACAGTTTTTTGAGTTTGATAGCCCTGGGCTCAATCAGAAAACTAG
CCAGTCAGAAACACTCTTCATTTCACTCGTTCGGGTCTGCTGACACTGACTTTGCTGACA
AGTCTTTGGAGGTTGAGTTTTGGAGAGAGATGGCGTTAGCAAAAATGGCTGAAGTTAGCA
AAATGGCTGCAGTCGCCTACAGCATTCGAATTCATACCTTGTTTCTGAGACCATGTGTCA
CTCACCTGGGCGCTGACTTTGCTCTTGCTTGCCACGGCTTGTTTGAGAGCCTGTCTGATA
ATGACATGCAGTCGGAGCACCACAGCCTCGATGTCCTGTTCCCCCTTCATGGCAGCGGCG
AGTGACCCCAGTTCATCCACATATGCTCTCCAATGAGGCTGTACCTCCGTGGTTCCCCCC
AAACATGCTCCCATGGTGCTGACACACAGTGGACGGCACGGCCGGGCCTTCAGCAGGCTT
TGACAGTGCGGGCAGTACCACAGTTTCAGAAGTGAGCGCCCACAATCTCTACCTGCCCGC
AAATGCTGAGT
.
qlt:
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXX
.
clr:0,671
}
{RED
iid:2
eid:zfishI-a72c06.p1c
seq:
CACCCAAAAGCAATGGCAAAAGATCTGGCGGACGCATTGCGGGCTGGGCGGGTGCTCAGC
TTGGCACTTGCCGAAGGGTCAGAGGTGATGAACATTACGGAAACAGCAGGGTTATCAAAA
GAGTGCACACGGGCTCTGGTACGCATGCATTACTGCTCTCACTGCCGTGGACTCACCCTG
ATCCATGCGTGCAGCAACTACTGTCTTAATGTCATGCGCGGGTGCCTGGCGAGCTACTCC
GAGCTCCACCAGCCCTGGAGACAGTATGTCACCATATTGCAGGACCTCACGCAAATGGTT
GCCGGAGCTCACAATTTAGAGCTGGCCTTACTGGGGATCAGAGGTCAGGTCGAGGAGGCC
ATACTCTACGCTCAGCTTCACGGGCCCAGGCTAACTGCCACAGTGAGTACTAGCATTTTT
ACGCTTTTACAGCTAGCATTAGCTTGTATTGTAGCATGAAAAAAGGTCTACAGAGTTATG
AAGCACAAAGACCTCTTCTGCTATAAGCGGGTTTCTGAA
.
qlt:
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
.
clr:0,519
}
5B
+85
View File
@@ -0,0 +1,85 @@
% 1. Title: Database for fitting contact lenses
%
% 2. Sources:
% (a) Cendrowska, J. "PRISM: An algorithm for inducing modular rules",
% International Journal of Man-Machine Studies, 1987, 27, 349-370
% (b) Donor: Benoit Julien (Julien@ce.cmu.edu)
% (c) Date: 1 August 1990
%
% 3. Past Usage:
% 1. See above.
% 2. Witten, I. H. & MacDonald, B. A. (1988). Using concept
% learning for knowledge acquisition. International Journal of
% Man-Machine Studies, 27, (pp. 349-370).
%
% Notes: This database is complete (all possible combinations of
% attribute-value pairs are represented).
%
% Each instance is complete and correct.
%
% 9 rules cover the training set.
%
% 4. Relevant Information Paragraph:
% The examples are complete and noise free.
% The examples highly simplified the problem. The attributes do not
% fully describe all the factors affecting the decision as to which type,
% if any, to fit.
%
% 5. Number of Instances: 24
%
% 6. Number of Attributes: 4 (all nominal)
%
% 7. Attribute Information:
% -- 3 Classes
% 1 : the patient should be fitted with hard contact lenses,
% 2 : the patient should be fitted with soft contact lenses,
% 1 : the patient should not be fitted with contact lenses.
%
% 1. age of the patient: (1) young, (2) pre-presbyopic, (3) presbyopic
% 2. spectacle prescription: (1) myope, (2) hypermetrope
% 3. astigmatic: (1) no, (2) yes
% 4. tear production rate: (1) reduced, (2) normal
%
% 8. Number of Missing Attribute Values: 0
%
% 9. Class Distribution:
% 1. hard contact lenses: 4
% 2. soft contact lenses: 5
% 3. no contact lenses: 15
@relation contact-lenses
@attribute age {young, pre-presbyopic, presbyopic}
@attribute spectacle-prescrip {myope, hypermetrope}
@attribute astigmatism {no, yes}
@attribute tear-prod-rate {reduced, normal}
@attribute contact-lenses {soft, hard, none}
@data
%
% 24 instances
%
young,myope,no,reduced,none
young,myope,no,normal,soft
young,myope,yes,reduced,none
young,myope,yes,normal,hard
young,hypermetrope,no,reduced,none
young,hypermetrope,no,normal,soft
young,hypermetrope,yes,reduced,none
young,hypermetrope,yes,normal,hard
pre-presbyopic,myope,no,reduced,none
pre-presbyopic,myope,no,normal,soft
pre-presbyopic,myope,yes,reduced,none
pre-presbyopic,myope,yes,normal,hard
pre-presbyopic,hypermetrope,no,reduced,none
pre-presbyopic,hypermetrope,no,normal,soft
pre-presbyopic,hypermetrope,yes,reduced,none
pre-presbyopic,hypermetrope,yes,normal,none
presbyopic,myope,no,reduced,none
presbyopic,myope,no,normal,none
presbyopic,myope,yes,reduced,none
presbyopic,myope,yes,normal,hard
presbyopic,hypermetrope,no,reduced,none
presbyopic,hypermetrope,no,normal,soft
presbyopic,hypermetrope,yes,reduced,none
presbyopic,hypermetrope,yes,normal,none
Binary file not shown.
+31
View File
@@ -0,0 +1,31 @@
format-version: GO_1.0
!any comment here
typeref: relationship.types
subsetdef: goslim "Generic GO Slim"
version: $Revision: 1.18 $
date: April 18th, 2003
saved-by: jrichter
remark: Example file
[Term]
id: GO:0003674
name: molecular_function
def: "The action characteristic of a gene product." [GO:curators]
subset: goslim
[Term]
id: GO:0016209
name: antioxidant activity
is_a: GO:0003674
def: "Inhibition of the reactions brought about by dioxygen or peroxides. Usually the antioxidant is effective because it can itself be more easily oxidized than the substance protected. The term is often applied to components that can trap free radicals, thereby breaking the chain reaction that normally leads to extensive biological damage." [ISBN:0198506732]
[Term]
id: GO:0045174
name: glutathione dehydrogenase (ascorbate) activity
xref_analog: EC:1.8.5.1 ""
def: "Catalysis of the reaction: 2 glutathione + dehydroascorbate = glutathione disulfide + ascorbate." [EC:1.8.5.1]
synonym: dehydroascorbate reductase []
is_a: GO:0009055
is_a: GO:0015038
is_a: GO:0016672
+221
View File
@@ -0,0 +1,221 @@
<?xml version="1.0"?>
<rdf:RDF
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
xmlns:xsd="http://www.w3.org/2001/XMLSchema#"
xmlns:rdfs="http://www.w3.org/2000/01/rdf-schema#"
xmlns:mfg="http://www.workingontologist.org/Examples/Chapter3/Product.owl#"
xmlns:owl="http://www.w3.org/2002/07/owl#"
xmlns:daml="http://www.daml.org/2001/03/daml+oil#"
xml:base="http://www.workingontologist.org/Examples/Chapter3/Product.owl">
<owl:Ontology rdf:about="">
<owl:versionInfo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Created with TopBraid Spreadsheet converter</owl:versionInfo>
</owl:Ontology>
<owl:Class rdf:ID="Product"/>
<owl:DatatypeProperty rdf:ID="Product_SKU">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>SKU</rdfs:label>
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
<rdfs:domain rdf:resource="#Product"/>
</owl:DatatypeProperty>
<owl:DatatypeProperty rdf:ID="Product_Product_Line">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product Line</rdfs:label>
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
<rdfs:domain rdf:resource="#Product"/>
</owl:DatatypeProperty>
<owl:DatatypeProperty rdf:ID="Product_Manufacture_Location">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Manufacture Location</rdfs:label>
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
<rdfs:domain rdf:resource="#Product"/>
</owl:DatatypeProperty>
<owl:DatatypeProperty rdf:ID="Product_Available">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Available</rdfs:label>
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
<rdfs:domain rdf:resource="#Product"/>
</owl:DatatypeProperty>
<owl:DatatypeProperty rdf:ID="Product_Division">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Division</rdfs:label>
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
<rdfs:domain rdf:resource="#Product"/>
</owl:DatatypeProperty>
<owl:DatatypeProperty rdf:ID="Product_ModelNo">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>ModelNo</rdfs:label>
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
<rdfs:domain rdf:resource="#Product"/>
</owl:DatatypeProperty>
<owl:DatatypeProperty rdf:ID="Product_ID">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>ID</rdfs:label>
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
<rdfs:domain rdf:resource="#Product"/>
</owl:DatatypeProperty>
<mfg:Product rdf:ID="Product2">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 2</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>2</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>ZX-3P</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Manufacturing support</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Paper machine</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Sacramento</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>KD5243</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>4</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product6">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 6</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>6</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>B-1431</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Control Engineering</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Active sensor</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Seoul</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>KK3945</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>0</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product1">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 1</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>1</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>ZX-3</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Manufacturing support</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Papermachine</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Sacramento</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>FB3524</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>23</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product7">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 7</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>7</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>DBB-12</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Accessories</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Monitor</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Hong Kong</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>ND5520</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>100</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product9">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 9</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>9</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>SPX-1234</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Safety</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Safety</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>valve</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Cleveland</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>OP5333</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product4">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 4</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>4</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>B-1430</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Control Engineering</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Feedback line</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Elizabeth</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>KS4520</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>23</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product8">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 8</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>8</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>SP-1234</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Safety</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Safety</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>valve</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Cleveland</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>HI4554</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product5">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 5</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>5</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>B-1430X</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Control Engineering</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Feedback line</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Elizabeth</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>CL5934</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>14</mfg:Product_Available>
</mfg:Product>
<mfg:Product rdf:ID="Product3">
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Product 3</rdfs:label>
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>3</mfg:Product_ID>
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>ZX-3S</mfg:Product_ModelNo>
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Manufacturing support</mfg:Product_Division>
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Paper machine</mfg:Product_Product_Line>
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>Sacramento</mfg:Product_Manufacture_Location>
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>IL4028</mfg:Product_SKU>
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
>34</mfg:Product_Available>
</mfg:Product>
</rdf:RDF>
<!-- Created with TopBraid Composer -->
+48
View File
@@ -0,0 +1,48 @@
<phyloxml xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns="http://www.phyloxml.org" xsi:schemaLocation="http://www.phyloxml.org http://www.phyloxml.org/1.10/phyloxml.xsd">
<phylogeny rooted="true" rerootable="false">
<name>Alcohol dehydrogenases</name>
<description>contains examples of commonly used elements</description>
<clade>
<events>
<speciations>1</speciations>
</events>
<clade>
<taxonomy>
<id provider="ncbi">6645</id>
<scientific_name>Octopus vulgaris</scientific_name>
</taxonomy>
<sequence>
<accession source="UniProtKB">P81431</accession>
<name>Alcohol dehydrogenase class-3</name>
</sequence>
</clade>
<clade>
<confidence type="bootstrap">100</confidence>
<events>
<speciations>1</speciations>
</events>
<clade>
<taxonomy>
<id provider="ncbi">1423</id>
<scientific_name>Bacillus subtilis</scientific_name>
</taxonomy>
<sequence>
<accession source="UniProtKB">P71017</accession>
<name>Alcohol dehydrogenase</name>
</sequence>
</clade>
<clade>
<taxonomy>
<id provider="ncbi">562</id>
<scientific_name>Escherichia coli</scientific_name>
</taxonomy>
<sequence>
<accession source="UniProtKB">Q46856</accession>
<name>Alcohol dehydrogenase</name>
</sequence>
</clade>
</clade>
</clade>
</phylogeny>
</phyloxml>
+18
View File
@@ -0,0 +1,18 @@
# STOCKHOLM 1.0
#=GF ID UPSK
#=GF SE Predicted; Infernal
#=GF SS Published; PMID 9223489
#=GF RN [1]
#=GF RM 9223489
#=GF RT The role of the pseudoknot at the 3' end of turnip yellow mosaic
#=GF RT virus RNA in minus-strand synthesis by the viral RNA-dependent RNA
#=GF RT polymerase.
#=GF RA Deiman BA, Kortlever RM, Pleij CW;
#=GF RL J Virol 1997;71:5990-5996.
AF035635.1/619-641 UGAGUUCUCGAUCUCUAAAAUCG
M24804.1/82-104 UGAGUUCUCUAUCUCUAAAAUCG
J04373.1/6212-6234 UAAGUUCUCGAUCUUUAAAAUCG
M24803.1/1-23 UAAGUUCUCGAUCUCUAAAAUCG
#=GC SS_cons .AAA....<<<<aaa....>>>>
//
+10
View File
@@ -0,0 +1,10 @@
@prefix rdf: <http://www.w3.org/1999/02/22-rdf-syntax-ns#> .
@prefix dc: <http://purl.org/dc/elements/1.1/> .
@prefix ex: <http://example.org/stuff/1.0/> .
<http://www.w3.org/TR/rdf-syntax-grammar>
dc:title "RDF/XML Syntax Specification (Revised)" ;
ex:editor [
ex:fullname "Dave Beckett";
ex:homePage <http://purl.org/net/dajobe/>
] .
+30
View File
@@ -0,0 +1,30 @@
#FormatVersion Mauve1
#Sequence1File a.fa
#Sequence1Entry 1
#Sequence1Format FastA
#Sequence2File b.fa
#Sequence2Entry 2
#Sequence2Format FastA
#Sequence3File c.fa
#Sequence3Entry 3
#Sequence3Format FastA
#BackboneFile three.xmfa.bbcols
> 1:0-0 + a.fa
--------------------------------------------------------------------------------
--------------------------------------------------------------------------------
--------------------------------------------------------------------------------
> 2:5417-5968 + b.fa
TTTAAACATCCCTCGGCCCGTCGCCCTTTTATAATAGCAGTACGTGAGAGGAGCGCCCTAAGCTTTGGGAAATTCAAGC-
--------------------------------------------------------------------------------
CTGGAACGTACTTGCTGGTTTCGCTACTATTTCAAACAAGTTAGAGGCCGTTACCTCGGGCGAACGTATAAACCATTCTG
> 3:9476-10076 - c.fa
TTTAAACACCTTTTTGGATG--GCCCAGTTCGTTCAGTTGTG-GGGAGGAGATCGCCCCAAACGTATGGTGAGTCGGGCG
TTTCCTATAGCTATAGGACCAATCCACTTACCATACGCCCGGCGTCGCCCAGTCCGGTTCGGTACCCTCCATGACCCACG
---------------------------------------------------------AAATGAGGGCCCAGGGTATGCTT
=
> 2:5969-6015 + b.fa
-----------------------
GGGCGAACGTATAAACCATTCTG
> 3:9429-9476 - c.fa
TTCGGTACCCTCCATGACCCACG
AAATGAGGGCCCAGGGTATGCTT
+3
View File
@@ -0,0 +1,3 @@
a 2
c 1
d 0
+1
View File
@@ -0,0 +1 @@
a 2
+3
View File
@@ -0,0 +1,3 @@
a 1 2 x
b 3 4 y
c 5 6 z
+21
View File
@@ -0,0 +1,21 @@
zoeHMM Acanium.hmm 6 8 6 7
<STATES>
Einit 0 0 3 -1 explicit
Esngl 0 0 150 -1 explicit
Eterm 0 0 3 -1 explicit
Exon 0 0 6 -1 explicit
Inter 0.9 0.9 0 0 geometric
Intron 0.1 0.1 0 0 geometric
<STATE_TRANSITIONS>
Einit Intron 1
Esngl Inter 1
Eterm Inter 1
Exon Intron 1
Inter Einit 0.852754
Inter Esngl 0.147246
Intron Eterm 0.129065
Intron Exon 0.870935
+61 -59
View File
@@ -14,12 +14,13 @@ from six.moves import shlex_quote
from galaxy.datatypes.data import get_file_peek, Text
from galaxy.datatypes.metadata import MetadataElement, MetadataParameter
from galaxy.datatypes.sniff import iter_headers
from galaxy.datatypes.sniff import build_sniff_from_prefix, iter_headers
from galaxy.util import nice_size, string_as_bool
log = logging.getLogger(__name__)
@build_sniff_from_prefix
class Html(Text):
"""Class describing an html file"""
edam_format = "format_2331"
@@ -37,7 +38,7 @@ class Html(Text):
"""Returns the mime type of the datatype"""
return 'text/html'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is in html format
@@ -49,13 +50,14 @@ class Html(Text):
>>> Html().sniff( fname )
True
"""
headers = iter_headers(filename, None)
headers = iter_headers(file_prefix, None)
for i, hdr in enumerate(headers):
if hdr and hdr[0].lower().find('<html>') >= 0:
return True
return False
@build_sniff_from_prefix
class Json(Text):
edam_format = "format_3464"
file_ext = "json"
@@ -72,30 +74,27 @@ class Json(Text):
"""Returns the mime type of the datatype"""
return 'application/json'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to load the string with the json module. If successful it's a json file.
"""
return self._looks_like_json(filename)
return self._looks_like_json(file_prefix)
def _looks_like_json(self, filename):
def _looks_like_json(self, file_prefix):
# Pattern used by SequenceSplitLocations
if os.path.getsize(filename) < 50000:
if file_prefix.file_size < 50000 and not file_prefix.truncated:
# If the file is small enough - don't guess just check.
try:
json.load(open(filename, "r"))
json.loads(file_prefix.contents_header)
return True
except Exception:
return False
else:
with open(filename, "r") as fh:
while True:
# Grab first chunk of file and see if it looks like json.
start = fh.read(100).strip()
if start:
# simple types are valid JSON as well - but would such a file
# be interesting as JSON in Galaxy?
return start.startswith("[") or start.startswith("{")
start = file_prefix.string_io().read(100).strip()
if start:
# simple types are valid JSON as well - but would such a file
# be interesting as JSON in Galaxy?
return start.startswith("[") or start.startswith("{")
return False
def display_peek(self, dataset):
@@ -105,6 +104,7 @@ class Json(Text):
return "JSON file (%s)" % (nice_size(dataset.get_size()))
@build_sniff_from_prefix
class Ipynb(Json):
file_ext = "ipynb"
@@ -116,13 +116,14 @@ class Ipynb(Json):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to load the string with the json module. If successful it's a json file.
"""
if self._looks_like_json(filename):
if self._looks_like_json(file_prefix):
try:
ipynb = json.load(open(filename))
with open(file_prefix.filename) as f:
ipynb = json.load(f)
if ipynb.get('nbformat', False) is not False and ipynb.get('metadata', False):
return True
else:
@@ -161,6 +162,7 @@ class Ipynb(Json):
pass
@build_sniff_from_prefix
class Biom1(Json):
"""
BIOM version 1.0 file format description
@@ -186,13 +188,13 @@ class Biom1(Json):
if not dataset.dataset.purged:
dataset.blurb = "Biological Observation Matrix v1"
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
is_biom = False
if self._looks_like_json(filename):
is_biom = self._looks_like_biom(filename)
if self._looks_like_json(file_prefix):
is_biom = self._looks_like_biom(file_prefix)
return is_biom
def _looks_like_biom(self, filepath, load_size=50000):
def _looks_like_biom(self, file_prefix, load_size=50000):
"""
@param filepath: [str] The path to the evaluated file.
@param load_size: [int] The size of the file block load in RAM (in
@@ -201,7 +203,7 @@ class Biom1(Json):
is_biom = False
segment_size = int(load_size / 2)
try:
with open(filepath, "r") as fh:
with open(file_prefix.filename, "r") as fh:
prev_str = ""
segment_str = fh.read(segment_size)
if segment_str.strip().startswith('{'):
@@ -255,6 +257,7 @@ class Biom1(Json):
pass
@build_sniff_from_prefix
class Obo(Text):
"""
OBO file format description
@@ -272,25 +275,26 @@ class Obo(Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess the Obo filetype.
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
"""
stanza = re.compile(r'^\[.*\]$')
with open(filename) as handle:
first_line = handle.readline()
if not first_line.startswith('format-version:'):
return False
handle = file_prefix.string_io()
first_line = handle.readline()
if not first_line.startswith('format-version:'):
return False
for line in handle:
if stanza.match(line.strip()):
# a stanza needs to begin with an ID tag
if handle.next().startswith('id:'):
return True
for line in handle:
if stanza.match(line.strip()):
# a stanza needs to begin with an ID tag
if handle.next().startswith('id:'):
return True
return False
@build_sniff_from_prefix
class Arff(Text):
"""
An ARFF (Attribute-Relation File Format) file is an ASCII text file that describes a list of instances sharing a set of attributes.
@@ -312,31 +316,31 @@ class Arff(Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Try to guess the Arff filetype.
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
"""
with open(filename) as handle:
relation_found = False
attribute_found = False
for line_count, line in enumerate(handle):
if line_count > 1000:
# only investigate the first 1000 lines
return False
line = line.strip()
if not line:
continue
handle = file_prefix.string_io()
relation_found = False
attribute_found = False
for line_count, line in enumerate(handle):
if line_count > 1000:
# only investigate the first 1000 lines
return False
line = line.strip()
if not line:
continue
start_string = line[:20].upper()
if start_string.startswith("@RELATION"):
relation_found = True
elif start_string.startswith("@ATTRIBUTE"):
attribute_found = True
elif start_string.startswith("@DATA"):
# @DATA should be the last data block
if relation_found and attribute_found:
return True
start_string = line[:20].upper()
if start_string.startswith("@RELATION"):
relation_found = True
elif start_string.startswith("@ATTRIBUTE"):
attribute_found = True
elif start_string.startswith("@DATA"):
# @DATA should be the last data block
if relation_found and attribute_found:
return True
return False
def set_meta(self, dataset, **kwd):
@@ -543,11 +547,12 @@ class SnpSiftDbNSFP(Text):
dataset.blurb = 'file purged from disc'
@build_sniff_from_prefix
class IQTree(Text):
"""IQ-TREE format"""
file_ext = 'iqtree'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Detect the IQTree file
@@ -567,7 +572,4 @@ class IQTree(Text):
>>> IQTree().sniff(fname)
False
"""
with open(filename, 'r') as fio:
return fio.read(7) == "IQ-TREE"
return False
return file_prefix.startswith("IQ-TREE")
+30 -27
View File
@@ -4,6 +4,9 @@ Triple format classes
import logging
import re
from galaxy.datatypes.sniff import (
build_sniff_from_prefix,
)
from . import (
binary,
data,
@@ -13,6 +16,9 @@ from . import (
log = logging.getLogger(__name__)
TURTLE_PREFIX_PATTERN = re.compile(r'@prefix\s+[^:]*:\s+<[^>]*>\s\.')
TURTLE_BASE_PATTERN = re.compile(r'@base\s+<[^>]*>\s\.')
class Triples(data.Data):
"""
@@ -38,6 +44,7 @@ class Triples(data.Data):
dataset.blurb = 'file purged from disk'
@build_sniff_from_prefix
class NTriples(data.Text, Triples):
"""
The N-Triples triple data format
@@ -45,11 +52,10 @@ class NTriples(data.Text, Triples):
edam_format = "format_3256"
file_ext = "nt"
def sniff(self, filename):
with open(filename, "r") as f:
# <http://example.org/dir/relfile> <http://www.w3.org/1999/02/22-rdf-syntax-ns#type> <http://example.org/type> .
if re.compile(r'<[^>]*>\s<[^>]*>\s<[^>]*>\s\.').search(f.readline(1024)):
return True
def sniff_prefix(self, file_prefix):
# <http://example.org/dir/relfile> <http://www.w3.org/1999/02/22-rdf-syntax-ns#type> <http://example.org/type> .
if re.compile(r'<[^>]*>\s<[^>]*>\s<[^>]*>\s\.').search(file_prefix.contents_header):
return True
return False
def set_peek(self, dataset, is_multi_byte=False):
@@ -85,6 +91,7 @@ class N3(data.Text, Triples):
dataset.blurb = 'file purged from disk'
@build_sniff_from_prefix
class Turtle(data.Text, Triples):
"""
The Turtle triple data format
@@ -92,14 +99,13 @@ class Turtle(data.Text, Triples):
edam_format = "format_3255"
file_ext = "ttl"
def sniff(self, filename):
with open(filename, "r") as f:
# @prefix rdf: <http://www.w3.org/1999/02/22-rdf-syntax-ns#> .
line = f.readline(1024)
if re.compile(r'@prefix\s+[^:]*:\s+<[^>]*>\s\.').search(line):
return True
if re.compile(r'@base\s+<[^>]*>\s\.').search(line):
return True
def sniff_prefix(self, file_prefix):
# @prefix rdf: <http://www.w3.org/1999/02/22-rdf-syntax-ns#> .
if file_prefix.search(TURTLE_PREFIX_PATTERN):
return True
if file_prefix.search(TURTLE_BASE_PATTERN):
return True
return False
def set_peek(self, dataset, is_multi_byte=False):
@@ -113,6 +119,7 @@ class Turtle(data.Text, Triples):
# TODO: we might want to look at rdflib or a similar, larger lib/egg
@build_sniff_from_prefix
class Rdf(xml.GenericXml, Triples):
"""
Resource Description Framework format (http://www.w3.org/RDF/).
@@ -120,13 +127,11 @@ class Rdf(xml.GenericXml, Triples):
edam_format = "format_3261"
file_ext = "rdf"
def sniff(self, filename):
with open(filename, "r") as f:
firstlines = "".join(f.readlines(5000))
# <rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" ...
match = re.compile(r'xmlns:([^=]*)="http://www.w3.org/1999/02/22-rdf-syntax-ns#"').search(firstlines)
if not match and (match.group(1) + ":RDF") in firstlines:
return True
def sniff_prefix(self, file_prefix):
# <rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" ...
match = re.compile(r'xmlns:([^=]*)="http://www.w3.org/1999/02/22-rdf-syntax-ns#"').search(file_prefix.contents_header)
if not match and (match.group(1) + ":RDF") in file_prefix.contents_header:
return True
return False
def set_peek(self, dataset, is_multi_byte=False):
@@ -139,6 +144,7 @@ class Rdf(xml.GenericXml, Triples):
dataset.blurb = 'file purged from disk'
@build_sniff_from_prefix
class Jsonld(text.Json, Triples):
"""
The JSON-LD data format
@@ -147,12 +153,10 @@ class Jsonld(text.Json, Triples):
edam_format = "format_3464"
file_ext = "jsonld"
def sniff(self, filename):
if self._looks_like_json(filename):
with open(filename, "r") as f:
firstlines = "".join(f.readlines(5000))
if "\"@id\"" in firstlines or "\"@context\"" in firstlines:
return True
def sniff_prefix(self, file_prefix):
if self._looks_like_json(file_prefix):
if "\"@id\"" in file_prefix.contents_header or "\"@context\"" in file_prefix.contents_header:
return True
return False
def set_peek(self, dataset, is_multi_byte=False):
@@ -176,7 +180,6 @@ class HDT(binary.Binary, Triples):
with open(filename, "rb") as f:
if f.read(4) == "$HDT":
return True
return False
def set_peek(self, dataset, is_multi_byte=False):
"""Set the peek and blurb text"""
+35 -31
View File
@@ -6,13 +6,17 @@ import re
from . import (
data,
dataproviders
dataproviders,
sniff
)
log = logging.getLogger(__name__)
OWL_MARKER = re.compile(r'\<owl:')
@dataproviders.decorators.has_dataproviders
@sniff.build_sniff_from_prefix
class GenericXml(data.Text):
"""Base format class for any XML file."""
edam_format = "format_2332"
@@ -27,7 +31,17 @@ class GenericXml(data.Text):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
def _has_root_element_in_prefix(self, file_prefix, root):
contents = file_prefix.string_io()
while True:
line = contents.readline()
if line is None or not line.startswith('<?'):
break
# pattern match <root or <ns:root for any ns string
pattern = '^<(\w*:)?%s' % root
return line is not None and re.match(pattern, line) is not None
def sniff_prefix(self, file_prefix):
"""
Determines whether the file is XML or not
@@ -39,12 +53,7 @@ class GenericXml(data.Text):
>>> GenericXml().sniff( fname )
False
"""
# TODO - Use a context manager on Python 2.5+ to close handle
with open(filename) as handle:
line = handle.readline()
# TODO - Is there a more robust way to do this?
return line.startswith('<?xml ')
return file_prefix.startswith('<?xml ')
def merge(split_files, output_file):
"""Merging multiple XML files is non-trivial and must be done in subclasses."""
@@ -60,6 +69,7 @@ class GenericXml(data.Text):
return dataproviders.hierarchy.XMLDataProvider(dataset_source, **settings)
@sniff.disable_parent_class_sniffing
class MEMEXml(GenericXml):
"""MEME XML Output data"""
file_ext = "memexml"
@@ -73,10 +83,8 @@ class MEMEXml(GenericXml):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
return False
@sniff.disable_parent_class_sniffing
class CisML(GenericXml):
"""CisML XML data""" # see: http://www.ncbi.nlm.nih.gov/pubmed/15001475
file_ext = "cisml"
@@ -90,9 +98,6 @@ class CisML(GenericXml):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
return False
class Phyloxml(GenericXml):
"""Format for defining phyloxml data http://www.phyloxml.org/"""
@@ -109,15 +114,21 @@ class Phyloxml(GenericXml):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
def sniff(self, filename):
""""Checking for keyword - 'phyloxml' always in lowercase in the first few lines"""
def sniff_prefix(self, file_prefix):
""""Checking for keyword - 'phyloxml' always in lowercase in the first few lines.
with open(filename, "r") as f:
firstlines = "".join(f.readlines(5))
if "phyloxml" in firstlines:
return True
return False
>>> from galaxy.datatypes.sniff import get_test_fname
>>> fname = get_test_fname( '1.phyloxml' )
>>> Phyloxml().sniff( fname )
True
>>> fname = get_test_fname( 'interval.interval' )
>>> Phyloxml().sniff( fname )
False
>>> fname = get_test_fname( 'megablast_xml_parser_test1.blastxml' )
>>> Phyloxml().sniff( fname )
False
"""
return self._has_root_element_in_prefix(file_prefix, "phyloxml")
def get_visualizations(self, dataset):
"""
@@ -143,15 +154,8 @@ class Owl(GenericXml):
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disc'
def sniff(self, filename):
def sniff_prefix(self, file_prefix):
"""
Checking for keyword - '<owl' in the first 200 lines.
"""
owl_marker = re.compile(r'\<owl:')
with open(filename) as handle:
# Check first 200 lines for the string "<owl:"
first_lines = handle.readlines(200)
for line in first_lines:
if owl_marker.search(line):
return True
return False
return file_prefix.search(OWL_MARKER)
+12 -4
View File
@@ -26,6 +26,10 @@ def get_fileobj(filename, mode="r", compressed_formats=None):
:param compressed_formats: list of allowed compressed file formats among
'bz2', 'gzip' and 'zip'. If left to None, all 3 formats are allowed
"""
return get_fileobj_raw(filename, mode, compressed_formats)[1]
def get_fileobj_raw(filename, mode="r", compressed_formats=None):
if compressed_formats is None:
compressed_formats = ['bz2', 'gzip', 'zip']
# Remove 't' from mode, which may cause an error for compressed files
@@ -36,22 +40,26 @@ def get_fileobj(filename, mode="r", compressed_formats=None):
cmode = 'r'
else:
cmode = mode
compressed_format = None
if 'gzip' in compressed_formats and is_gzip(filename):
fh = gzip.GzipFile(filename, cmode)
compressed_format = 'gzip'
elif 'bz2' in compressed_formats and is_bz2(filename):
fh = bz2.BZ2File(filename, cmode)
compressed_format = 'bz2'
elif 'zip' in compressed_formats and zipfile.is_zipfile(filename):
# Return fileobj for the first file in a zip file.
with zipfile.ZipFile(filename, cmode) as zh:
fh = zh.open(zh.namelist()[0], cmode)
compressed_format = 'zip'
elif 'b' in mode:
return open(filename, mode)
return compressed_format, open(filename, mode)
else:
return io.open(filename, mode, encoding='utf-8')
return compressed_format, io.open(filename, mode, encoding='utf-8')
if 'b' not in mode:
return io.TextIOWrapper(fh, encoding='utf-8')
return compressed_format, io.TextIOWrapper(fh, encoding='utf-8')
else:
return fh
return compressed_format, fh
class CompressedFile(object):