diff --git a/lib/galaxy/datatypes/annotation.py b/lib/galaxy/datatypes/annotation.py index 32de5d10e1d..64ed7220c91 100644 --- a/lib/galaxy/datatypes/annotation.py +++ b/lib/galaxy/datatypes/annotation.py @@ -3,11 +3,13 @@ import tarfile from galaxy.datatypes.binary import CompressedArchive from galaxy.datatypes.data import get_file_peek, Text +from galaxy.datatypes.sniff import build_sniff_from_prefix from galaxy.util import nice_size log = logging.getLogger(__name__) +@build_sniff_from_prefix class SnapHmm(Text): file_ext = "snaphmm" edam_data = "data_1364" @@ -26,13 +28,11 @@ class SnapHmm(Text): except Exception: return "SNAP HMM model (%s)" % (nice_size(dataset.get_size())) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ SNAP model files start with zoeHMM """ - with open(filename, 'r') as handle: - return handle.read(6) == 'zoeHMM' - return False + return file_prefix.startswith('zoeHMM') class Augustus(CompressedArchive): diff --git a/lib/galaxy/datatypes/assembly.py b/lib/galaxy/datatypes/assembly.py index 7b60e606087..64b5662c092 100644 --- a/lib/galaxy/datatypes/assembly.py +++ b/lib/galaxy/datatypes/assembly.py @@ -13,20 +13,20 @@ import sys from galaxy.datatypes import data from galaxy.datatypes import sequence from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes.sniff import build_sniff_from_prefix from galaxy.datatypes.text import Html log = logging.getLogger(__name__) +@build_sniff_from_prefix class Amos(data.Text): """Class describing the AMOS assembly file """ edam_data = "data_0925" edam_format = "format_3582" file_ext = 'afg' - def sniff(self, filename): - # FIXME: this method will read the entire file. - # It should call get_headers() like other sniff methods. + def sniff_prefix(self, file_prefix): """ Determines whether the file is an amos assembly file format Example:: @@ -50,25 +50,24 @@ class Amos(data.Text): } } """ - with open(filename) as fh: - while True: - line = fh.readline() - if not line: - break # EOF - line = line.strip() - if line: # first non-empty line - if line.startswith('{'): - if re.match(r'{(RED|CTG|TLE)$', line): - return True + for line in file_prefix.line_iterator(): + if not line: + break # EOF + line = line.strip() + if line: # first non-empty line + if line.startswith('{'): + if re.match(r'{(RED|CTG|TLE)$', line): + return True return False +@build_sniff_from_prefix class Sequences(sequence.Fasta): """Class describing the Sequences file generated by velveth """ edam_data = "data_0925" file_ext = 'sequences' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a velveth produced fasta format The id line has 3 fields separated by tabs: sequence_name sequence_index category:: @@ -78,33 +77,33 @@ class Sequences(sequence.Fasta): >SEQUENCE_1_length_35 2 1 CGACGAATGACAGGTCACGAATTTGGCGGGGATTA """ - - with open(filename) as fh: - while True: - line = fh.readline() - if not line: - break # EOF - line = line.strip() - if line: # first non-empty line - if line.startswith('>'): - if not re.match(r'>[^\t]+\t\d+\t\d+$', line): - break - # The next line.strip() must not be '', nor startwith '>' - line = fh.readline().strip() - if line == '' or line.startswith('>'): - break - return True - else: - break # we found a non-empty line, but it's not a fasta header + fh = file_prefix.string_io() + while True: + line = fh.readline() + if not line: + break # EOF + line = line.strip() + if line: # first non-empty line + if line.startswith('>'): + if not re.match(r'>[^\t]+\t\d+\t\d+$', line): + break + # The next line.strip() must not be '', nor startwith '>' + line = fh.readline().strip() + if line == '' or line.startswith('>'): + break + return True + else: + break # we found a non-empty line, but it's not a fasta header return False +@build_sniff_from_prefix class Roadmaps(data.Text): """Class describing the Sequences file generated by velveth """ edam_format = "format_2561" file_ext = 'roadmaps' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a velveth produced RoadMap:: 142858 21 1 @@ -113,22 +112,22 @@ class Roadmaps(data.Text): ... """ - with open(filename) as fh: - while True: - line = fh.readline() - if not line: - break # EOF - line = line.strip() - if line: # first non-empty line - if not re.match(r'\d+\t\d+\t\d+$', line): - break - # The next line.strip() should be 'ROADMAP 1' - line = fh.readline().strip() - if not re.match(r'ROADMAP \d+$', line): - break - return True - else: - break # we found a non-empty line, but it's not a fasta header + fh = file_prefix.string_io() + while True: + line = fh.readline() + if not line: + break # EOF + line = line.strip() + if line: # first non-empty line + if not re.match(r'\d+\t\d+\t\d+$', line): + break + # The next line.strip() should be 'ROADMAP 1' + line = fh.readline().strip() + if not re.match(r'ROADMAP \d+$', line): + break + return True + else: + break # we found a non-empty line, but it's not a fasta header return False diff --git a/lib/galaxy/datatypes/blast.py b/lib/galaxy/datatypes/blast.py index 62709b2361d..f1924449d34 100644 --- a/lib/galaxy/datatypes/blast.py +++ b/lib/galaxy/datatypes/blast.py @@ -39,11 +39,13 @@ from .data import ( get_file_peek, Text ) +from .sniff import build_sniff_from_prefix from .xml import GenericXml log = logging.getLogger(__name__) +@build_sniff_from_prefix class BlastXml(GenericXml): """NCBI Blast XML Output data""" file_ext = "blastxml" @@ -59,7 +61,7 @@ class BlastXml(GenericXml): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disk' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """Determines whether the file is blastxml >>> from galaxy.datatypes.sniff import get_test_fname @@ -73,17 +75,17 @@ class BlastXml(GenericXml): >>> BlastXml().sniff(fname) False """ - with open(filename) as handle: - line = handle.readline() - if line.strip() != '': - return False - line = handle.readline() - if line.strip() not in ['', - '']: - return False - line = handle.readline() - if line.strip() != '': - return False + handle = file_prefix.string_io() + line = handle.readline() + if line.strip() != '': + return False + line = handle.readline() + if line.strip() not in ['', + '']: + return False + line = handle.readline() + if line.strip() != '': + return False return True def merge(split_files, output_file): diff --git a/lib/galaxy/datatypes/constructive_solid_geometry.py b/lib/galaxy/datatypes/constructive_solid_geometry.py index 1c59f64ccf6..eaccbf137a9 100644 --- a/lib/galaxy/datatypes/constructive_solid_geometry.py +++ b/lib/galaxy/datatypes/constructive_solid_geometry.py @@ -9,12 +9,14 @@ from galaxy.datatypes.binary import Binary from galaxy.datatypes.data import get_file_peek from galaxy.datatypes.data import nice_size from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes.sniff import build_sniff_from_prefix MAX_HEADER_LINES = 500 MAX_LINE_LEN = 2000 COLOR_OPTS = ['COLOR_SCALARS', 'red', 'green', 'blue'] +@build_sniff_from_prefix class Ply(object): """ The PLY format describes an object as a collection of vertices, @@ -37,16 +39,14 @@ class Ply(object): def __init__(self, **kwd): raise NotImplementedError - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ The structure of a typical PLY file: Header, Vertex List, Face List, (lists of other elements) """ - with open(filename, "r") as fh: - if not self._is_ply_header(fh, self.subtype): - return False - return True - return False + if not self._is_ply_header(file_prefix.string_io(), self.subtype): + return False + return True def _is_ply_header(self, fh, subtype): """ @@ -131,6 +131,7 @@ class PlyBinary(Ply, Binary): Binary.__init__(self, **kwd) +@build_sniff_from_prefix class Vtk(object): r""" The Visualization Toolkit provides a number of source and writer objects to @@ -200,16 +201,14 @@ class Vtk(object): def __init__(self, **kwd): raise NotImplementedError - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ VTK files can be either ASCII or binary, with two different styles of file formats: legacy or XML. We'll assume if the file contains a valid VTK header, then it is a valid VTK file. """ - with open(filename, "r") as fh: - if self._is_vtk_header(fh, self.subtype): - return True - return False + if self._is_vtk_header(file_prefix.string_io(), self.subtype): + return True return False def _is_vtk_header(self, fh, subtype): diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 66842ac9ef7..2e237070690 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -16,6 +16,7 @@ import six from galaxy import util from galaxy.datatypes.metadata import MetadataElement # import directly to maintain ease of use in Datatype class definitions +from galaxy.datatypes.sniff import build_sniff_from_prefix from galaxy.util import ( compression_utils, FILENAME_VALID_CHARS, @@ -961,6 +962,7 @@ class Newick(Text): return ['phyloviz'] +@build_sniff_from_prefix class Nexus(Text): """Nexus format as used By Paup, Mr Bayes, etc""" edam_data = "data_0872" @@ -974,15 +976,9 @@ class Nexus(Text): def init_meta(self, dataset, copy_from=None): Text.init_meta(self, dataset, copy_from=copy_from) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """All Nexus Files Simply puts a '#NEXUS' in its first line""" - with open(filename, "r") as f: - firstline = f.readline().upper() - - if "#NEXUS" in firstline: - return True - else: - return False + return file_prefix.string_io().read(6).upper() == "#NEXUS" def get_visualizations(self, dataset): """ diff --git a/lib/galaxy/datatypes/genetics.py b/lib/galaxy/datatypes/genetics.py index 3c8b5ede0ca..5563fdf9980 100644 --- a/lib/galaxy/datatypes/genetics.py +++ b/lib/galaxy/datatypes/genetics.py @@ -22,6 +22,7 @@ from six.moves.urllib.parse import quote_plus from galaxy.datatypes import metadata from galaxy.datatypes.data import Text from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes.sniff import build_sniff_from_prefix from galaxy.datatypes.tabular import Tabular from galaxy.datatypes.text import Html from galaxy.util import nice_size @@ -35,6 +36,7 @@ VALID_GENOME_GRAPH_MARKERS = re.compile('^(chr.*|RH.*|rs.*|SNP_.*|CN.*|A_.*)') VALID_GENOTYPES_LINE = re.compile('^([a-zA-Z0-9]+)(\\s([0-9]{2}|[A-Z]{2}|NC|\?\?))+\\s*$') +@build_sniff_from_prefix class GenomeGraphs(Tabular): """ Tab delimited data containing a marker id and any number of numeric values @@ -166,7 +168,7 @@ class GenomeGraphs(Tabular): errors.append('row %d, %s' % (' '.join(badvals))) return errors - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in gg format @@ -178,9 +180,7 @@ class GenomeGraphs(Tabular): >>> GenomeGraphs().sniff( fname ) True """ - with open(filename, 'r') as f: - buf = f.read(1024) - + buf = file_prefix.contents_header rows = [l.split() for l in buf.splitlines()[1:4]] # break on lines and drop header, small sample if len(rows) < 1: @@ -247,14 +247,6 @@ class rgSampleList(rgTabList): self.column_names[1] = 'IID' # this is what Plink wants as at 2009 - def sniff(self, filename): - with open(filename, "r") as infile: - header = next(infile) # header - if header[0] == 'FID' and header[1] == 'IID': - return True - else: - return False - class rgFeatureList(rgTabList): """ @@ -909,6 +901,7 @@ class LinkageStudies(Text): self.max_lines = 10 +@build_sniff_from_prefix class GenotypeMatrix(LinkageStudies): """ Sample matrix of genotypes @@ -918,7 +911,6 @@ class GenotypeMatrix(LinkageStudies): def __init__(self, **kwd): super(GenotypeMatrix, self).__init__(**kwd) - self.num_cols = -1 def header_check(self, fio): header_elems = fio.readline().split('\t') @@ -933,7 +925,7 @@ class GenotypeMatrix(LinkageStudies): return True - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ >>> classname = GenotypeMatrix >>> from galaxy.datatypes.sniff import get_test_fname @@ -941,7 +933,6 @@ class GenotypeMatrix(LinkageStudies): >>> file_true = get_test_fname("linkstudies." + extn_true) >>> classname().sniff(file_true) True - >>> false_files = list(LinkageStudies.test_files) >>> false_files.remove("linkstudies." + extn_true) >>> result_true = [] @@ -954,27 +945,29 @@ class GenotypeMatrix(LinkageStudies): >>> result_true [] """ - with open(filename, "r") as fio: + fio = file_prefix.string_io() + num_cols = -1 - if not self.header_check(fio): + if not self.header_check(fio): + return False + + for lcount, line in enumerate(fio): + if lcount > self.max_lines: + return True + + tokens = line.split('\t') + + if num_cols == -1: + num_cols = len(tokens) + elif num_cols != len(tokens): + return False + if not VALID_GENOTYPES_LINE.match(line): return False - for lcount, line in enumerate(fio): - if lcount > self.max_lines: - return True - - tokens = line.split('\t') - - if self.num_cols == -1: - self.num_cols = len(tokens) - elif self.num_cols != len(tokens): - return False - if not VALID_GENOTYPES_LINE.match(line): - return False - - return True + return True +@build_sniff_from_prefix class MarkerMap(LinkageStudies): """ Map of genetic markers including physical and genetic distance @@ -992,7 +985,7 @@ class MarkerMap(LinkageStudies): return False - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ >>> classname = MarkerMap >>> from galaxy.datatypes.sniff import get_test_fname @@ -1000,7 +993,6 @@ class MarkerMap(LinkageStudies): >>> file_true = get_test_fname("linkstudies." + extn_true) >>> classname().sniff(file_true) True - >>> false_files = list(LinkageStudies.test_files) >>> false_files.remove("linkstudies." + extn_true) >>> result_true = [] @@ -1013,32 +1005,32 @@ class MarkerMap(LinkageStudies): >>> result_true [] """ - with open(filename, "r") as fio: + fio = file_prefix.string_io() + if not self.header_check(fio): + return False - if not self.header_check(fio): - return False + for lcount, line in enumerate(fio): + if lcount > self.max_lines: + return True - for lcount, line in enumerate(fio): - if lcount > self.max_lines: - return True + try: + chrm, gpos, nam, bpos, row = line.split() + float(gpos) + int(bpos) try: - chrm, gpos, nam, bpos, row = line.split() - float(gpos) - int(bpos) - - try: - int(chrm) - except ValueError: - if not chrm.lower()[0] in ('x', 'y', 'm'): - return False - + int(chrm) except ValueError: - return False + if not chrm.lower()[0] in ('x', 'y', 'm'): + return False - return True + except ValueError: + return False + + return True +@build_sniff_from_prefix class DataIn(LinkageStudies): """ Common linkage input file for intermarker distances @@ -1048,13 +1040,8 @@ class DataIn(LinkageStudies): def __init__(self, **kwd): super(DataIn, self).__init__(**kwd) - self.num_markers = None - self.intermarkers = 0 - def eof_function(self): - return self.intermarkers > 0 - - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ >>> classname = DataIn >>> from galaxy.datatypes.sniff import get_test_fname @@ -1062,7 +1049,6 @@ class DataIn(LinkageStudies): >>> file_true = get_test_fname("linkstudies." + extn_true) >>> classname().sniff(file_true) True - >>> false_files = list(LinkageStudies.test_files) >>> false_files.remove("linkstudies." + extn_true) >>> result_true = [] @@ -1075,41 +1061,47 @@ class DataIn(LinkageStudies): >>> result_true [] """ - with open(filename, "r") as fio: + intermarkers = 0 + num_markers = None - for lcount, line in enumerate(fio): - if lcount > self.max_lines: - return self.eof_function() + def eof_function(): + return intermarkers > 0 - tokens = line.split() - try: - if lcount == 0: - self.num_markers = int(tokens[0]) - map(int, tokens[1:]) - elif lcount == 1: - map(float, tokens) + fio = file_prefix.string_io() + for lcount, line in enumerate(fio): + if lcount > self.max_lines: + return eof_function() - if len(tokens) != 4: - return False - elif lcount == 2: - map(int, tokens) - last_token = int(tokens[-1]) + tokens = line.split() + try: + if lcount == 0: + num_markers = int(tokens[0]) + map(int, tokens[1:]) + elif lcount == 1: + map(float, tokens) - if self.num_markers is None: - return False - if len(tokens) != last_token: - return False - if self.num_markers != last_token: - return False - elif tokens[0] == "3" and tokens[1] == "2": - self.intermarkers += 1 + if len(tokens) != 4: + return False + elif lcount == 2: + map(int, tokens) + last_token = int(tokens[-1]) - except (ValueError, IndexError): - return False + if num_markers is None: + return False + if len(tokens) != last_token: + return False + if num_markers != last_token: + return False + elif tokens[0] == "3" and tokens[1] == "2": + intermarkers += 1 - return self.eof_function() + except (ValueError, IndexError): + return False + + return eof_function() +@build_sniff_from_prefix class AllegroLOD(LinkageStudies): """ Allegro output format for LOD scores @@ -1125,7 +1117,7 @@ class AllegroLOD(LinkageStudies): return False - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ >>> classname = AllegroLOD >>> from galaxy.datatypes.sniff import get_test_fname @@ -1133,7 +1125,6 @@ class AllegroLOD(LinkageStudies): >>> file_true = get_test_fname("linkstudies." + extn_true) >>> classname().sniff(file_true) True - >>> false_files = list(LinkageStudies.test_files) >>> false_files.remove("linkstudies." + extn_true) >>> result_true = [] @@ -1146,28 +1137,28 @@ class AllegroLOD(LinkageStudies): >>> result_true [] """ - with open(filename, "r") as fio: + fio = file_prefix.string_io() - if not self.header_check(fio): + if not self.header_check(fio): + return False + + for lcount, line in enumerate(fio): + if lcount > self.max_lines: + return True + + tokens = line.split() + + try: + int(tokens[0]) + float(tokens[1]) + + if tokens[2] != "-inf": + float(tokens[2]) + + except (ValueError, IndexError): return False - for lcount, line in enumerate(fio): - if lcount > self.max_lines: - return True - - tokens = line.split() - - try: - int(tokens[0]) - float(tokens[1]) - - if tokens[2] != "-inf": - float(tokens[2]) - - except (ValueError, IndexError): - return False - - return True + return True if __name__ == '__main__': diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 8531a15f3c5..5d90b9b1940 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -13,6 +13,7 @@ from galaxy import util from galaxy.datatypes import metadata from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes.sniff import ( + build_sniff_from_prefix, get_headers, iter_headers ) @@ -50,6 +51,7 @@ VIEWPORT_MAX_READS_PER_LINE = 10 @dataproviders.decorators.has_dataproviders +@build_sniff_from_prefix class Interval(Tabular): """Tab delimited data containing interval information""" edam_data = "data_3002" @@ -297,7 +299,7 @@ class Interval(Tabular): """Return options for removing errors along with a description""" return [("lines", "Remove erroneous lines")] - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Checks for 'intervalness' @@ -312,26 +314,23 @@ class Interval(Tabular): >>> Interval().sniff( fname ) True """ + found_valid_lines = False try: - """ - If we got here, we already know the file is_column_based and is not bed, - so we'll just look for some valid data. - """ - headers = iter_headers(filename, '\t', comment_designator='#') + headers = iter_headers(file_prefix, '\t', comment_designator='#') + # If we got here, we already know the file is_column_based and is not bed, + # so we'll just look for some valid data. for hdr in headers: if hdr: if len(hdr) < 3: return False - try: - # Assume chrom start and end are in column positions 1 and 2 - # respectively ( for 0 based columns ) - int(hdr[1]) - int(hdr[2]) - except Exception: - return False - return True + # Assume chrom start and end are in column positions 1 and 2 + # respectively ( for 0 based columns ) + int(hdr[1]) + int(hdr[2]) + found_valid_lines = True except Exception: return False + return found_valid_lines def get_track_resolution(self, dataset, start, end): return None @@ -464,7 +463,7 @@ class Bed(Interval): except Exception: return "This item contains no content" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Checks for 'bedness' @@ -488,10 +487,10 @@ class Bed(Interval): >>> Bed().sniff( fname ) True """ - if not get_headers(filename, '\t', comment_designator='#', count=1): + if not get_headers(file_prefix, '\t', comment_designator='#', count=1): return False try: - headers = iter_headers(filename, '\t', comment_designator='#') + headers = iter_headers(file_prefix, '\t', comment_designator='#') for hdr in headers: if hdr[0] == '': continue @@ -635,6 +634,7 @@ class _RemoteCallMixin(object): @dataproviders.decorators.has_dataproviders +@build_sniff_from_prefix class Gff(Tabular, _RemoteCallMixin): """Tab delimited data in Gff format""" edam_data = "data_1255" @@ -822,7 +822,7 @@ class Gff(Tabular, _RemoteCallMixin): ret_val.append((site_name, link)) return ret_val - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in gff format @@ -831,17 +831,17 @@ class Gff(Tabular, _RemoteCallMixin): For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3 >>> from galaxy.datatypes.sniff import get_test_fname - >>> fname = get_test_fname( 'gff_version_3.gff' ) + >>> fname = get_test_fname('gff_version_3.gff') >>> Gff().sniff( fname ) False - >>> fname = get_test_fname( 'test.gff' ) + >>> fname = get_test_fname('test.gff') >>> Gff().sniff( fname ) True """ - if len(get_headers(filename, '\t', count=2)) < 2: + if len(get_headers(file_prefix, '\t', count=2)) < 2: return False try: - headers = iter_headers(filename, '\t') + headers = iter_headers(file_prefix, '\t') for hdr in headers: if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0: return False @@ -937,7 +937,7 @@ class Gff3(Gff): break Tabular.set_meta(self, dataset, overwrite=overwrite, skip=i) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in GFF version 3 format @@ -970,10 +970,10 @@ class Gff3(Gff): >>> Gff3().sniff( fname ) True """ - if len(get_headers(filename, '\t', count=2)) < 2: + if len(get_headers(file_prefix, '\t', count=2)) < 2: return False try: - headers = iter_headers(filename, '\t') + headers = iter_headers(file_prefix, '\t') for hdr in headers: if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0: return True @@ -1020,7 +1020,7 @@ class Gtf(Gff): MetadataElement(name="column_types", default=['str', 'str', 'str', 'int', 'int', 'float', 'str', 'int', 'list'], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in gtf format @@ -1045,10 +1045,10 @@ class Gtf(Gff): >>> Gtf().sniff( fname ) True """ - if len(get_headers(filename, '\t', count=2)) < 2: + if len(get_headers(file_prefix, '\t', count=2)) < 2: return False try: - headers = iter_headers(filename, '\t') + headers = iter_headers(file_prefix, '\t') for hdr in headers: if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0: return False @@ -1085,6 +1085,7 @@ class Gtf(Gff): @dataproviders.decorators.has_dataproviders +@build_sniff_from_prefix class Wiggle(Tabular, _RemoteCallMixin): """Tab delimited data in wiggle format""" edam_format = "format_3005" @@ -1218,7 +1219,7 @@ class Wiggle(Tabular, _RemoteCallMixin): max_data_lines = 100 Tabular.set_meta(self, dataset, overwrite=overwrite, skip=i, max_data_lines=max_data_lines) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines wether the file is in wiggle format @@ -1242,7 +1243,7 @@ class Wiggle(Tabular, _RemoteCallMixin): True """ try: - headers = iter_headers(filename, None) + headers = iter_headers(file_prefix, None) for hdr in headers: if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'): return True @@ -1272,6 +1273,7 @@ class Wiggle(Tabular, _RemoteCallMixin): return dataproviders.dataset.WiggleDataProvider(dataset_source, **settings) +@build_sniff_from_prefix class CustomTrack(Tabular): """UCSC CustomTrack""" edam_format = "format_3588" @@ -1360,7 +1362,7 @@ class CustomTrack(Tabular): ret_val.append((site_name, link)) return ret_val - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in customtrack format. @@ -1377,7 +1379,8 @@ class CustomTrack(Tabular): >>> CustomTrack().sniff( fname ) True """ - headers = iter_headers(filename, None) + headers = iter_headers(file_prefix, None) + found_at_least_one_track = False first_line = True for hdr in headers: if first_line: @@ -1409,9 +1412,10 @@ class CustomTrack(Tabular): int(hdr[2]) except Exception: return False + found_at_least_one_track = True except Exception: return False - return True + return found_at_least_one_track class ENCODEPeak(Interval): @@ -1467,6 +1471,7 @@ class ChromatinInteractions(Interval): return False +@build_sniff_from_prefix class ScIdx(Tabular): """ ScIdx files are 1-based and consist of strand-specific coordinate counts. @@ -1492,55 +1497,55 @@ class ScIdx(Tabular): # line of the dataset displays them. self.column_names = ['chrom', 'index', 'forward', 'reverse', 'value'] - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Checks for 'scidx-ness.' """ count = 0 - with open(filename, "r") as fh: - while True: - line = fh.readline() - if not line: - # EOF - if count > 1: - # The second line is always the labels: - # chrom index forward reverse value - # We need at least the column labels and a data line. - return True - return False - line = line.strip() - # The first line is always a comment like this: - # 2015-11-23 20:18:56.51;input.bam;READ1 - if count == 0: - if line.startswith('#'): - count += 1 - continue - else: - return False - # Skip first line. + fh = file_prefix.string_io() + while True: + line = fh.readline() + if not line: + # EOF if count > 1: - items = line.split('\t') - if len(items) != 5: - return False - index = items[1] - if not index.isdigit(): - return False - forward = items[2] - if not forward.isdigit(): - return False - reverse = items[3] - if not reverse.isdigit(): - return False - value = items[4] - if not value.isdigit(): - return False - if int(forward) + int(reverse) != int(value): - return False - if count == 100: + # The second line is always the labels: + # chrom index forward reverse value + # We need at least the column labels and a data line. return True - count += 1 - if count < 100 and count > 0: + return False + line = line.strip() + # The first line is always a comment like this: + # 2015-11-23 20:18:56.51;input.bam;READ1 + if count == 0: + if line.startswith('#'): + count += 1 + continue + else: + return False + # Skip first line. + if count > 1: + items = line.split('\t') + if len(items) != 5: + return False + index = items[1] + if not index.isdigit(): + return False + forward = items[2] + if not forward.isdigit(): + return False + reverse = items[3] + if not reverse.isdigit(): + return False + value = items[4] + if not value.isdigit(): + return False + if int(forward) + int(reverse) != int(value): + return False + if count == 100: return True + count += 1 + if count < 100 and count > 0: + return True return False diff --git a/lib/galaxy/datatypes/molecules.py b/lib/galaxy/datatypes/molecules.py index ce3b3e0d755..9276c1be76d 100644 --- a/lib/galaxy/datatypes/molecules.py +++ b/lib/galaxy/datatypes/molecules.py @@ -11,6 +11,7 @@ from galaxy.datatypes.binary import Binary from galaxy.datatypes.data import get_file_peek from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes.sniff import ( + build_sniff_from_prefix, get_headers, iter_headers ) @@ -84,10 +85,11 @@ class MOL(GenericMolFile): dataset.metadata.number_of_molecules = 1 +@build_sniff_from_prefix class SDF(GenericMolFile): file_ext = "sdf" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess if the file is a SDF2 file. @@ -102,11 +104,9 @@ class SDF(GenericMolFile): >>> fname = get_test_fname('drugbank_drugs.sdf') >>> SDF().sniff(fname) True - >>> fname = get_test_fname('github88.v3k.sdf') >>> SDF().sniff(fname) True - >>> fname = get_test_fname('chebi_57262.v3k.mol') >>> SDF().sniff(fname) False @@ -114,23 +114,22 @@ class SDF(GenericMolFile): m_end_found = False limit = 10000 idx = 0 - with open(filename) as in_file: - for line in in_file: - idx += 1 - line = line.rstrip('\n\r') - if idx < 4: - continue - elif idx == 4: - if len(line) != 39 or not(line.endswith(' V2000') or - line.endswith(' V3000')): - return False - elif not m_end_found: - if line == 'M END': - m_end_found = True - elif line == '$$$$': - return True - if idx == limit: - break + for line in file_prefix.line_iterator(): + idx += 1 + line = line.rstrip('\n\r') + if idx < 4: + continue + elif idx == 4: + if len(line) != 39 or not(line.endswith(' V2000') or + line.endswith(' V3000')): + return False + elif not m_end_found: + if line == 'M END': + m_end_found = True + elif line == '$$$$': + return True + if idx == limit: + break return False def set_meta(self, dataset, **kwd): @@ -189,10 +188,11 @@ class SDF(GenericMolFile): split = classmethod(split) +@build_sniff_from_prefix class MOL2(GenericMolFile): file_ext = "mol2" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess if the file is a MOL2 file. @@ -200,21 +200,19 @@ class MOL2(GenericMolFile): >>> fname = get_test_fname('drugbank_drugs.mol2') >>> MOL2().sniff(fname) True - >>> fname = get_test_fname('drugbank_drugs.cml') >>> MOL2().sniff(fname) False """ limit = 60 idx = 0 - with open(filename) as in_file: - for line in in_file: - line = line.rstrip('\n\r') - if line == '@MOLECULE': - return True - idx += 1 - if idx == limit: - break + for line in file_prefix.line_iterator(): + line = line.rstrip('\n\r') + if line == '@MOLECULE': + return True + idx += 1 + if idx == limit: + break return False def set_meta(self, dataset, **kwd): @@ -277,13 +275,14 @@ class MOL2(GenericMolFile): split = classmethod(split) +@build_sniff_from_prefix class FPS(GenericMolFile): """ chemfp fingerprint file: http://code.google.com/p/chem-fingerprints/wiki/FPS """ file_ext = "fps" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess if the file is a FPS file. @@ -291,12 +290,11 @@ class FPS(GenericMolFile): >>> fname = get_test_fname('q.fps') >>> FPS().sniff(fname) True - >>> fname = get_test_fname('drugbank_drugs.cml') >>> FPS().sniff(fname) False """ - header = get_headers(filename, sep='\t', count=1) + header = get_headers(file_prefix, sep='\t', count=1) if header[0][0].strip() == '#FPS1': return True else: @@ -473,6 +471,7 @@ class PHAR(GenericMolFile): dataset.blurb = 'file purged from disk' +@build_sniff_from_prefix class PDB(GenericMolFile): """ Protein Databank format. @@ -480,7 +479,7 @@ class PDB(GenericMolFile): """ file_ext = "pdb" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess if the file is a PDB file. @@ -488,12 +487,11 @@ class PDB(GenericMolFile): >>> fname = get_test_fname('5e5z.pdb') >>> PDB().sniff(fname) True - >>> fname = get_test_fname('drugbank_drugs.cml') >>> PDB().sniff(fname) False """ - headers = iter_headers(filename, sep=' ', count=300) + headers = iter_headers(file_prefix, sep=' ', count=300) h = t = c = s = k = e = False for line in headers: section_name = line[0].strip() @@ -526,6 +524,7 @@ class PDB(GenericMolFile): dataset.blurb = 'file purged from disk' +@build_sniff_from_prefix class PDBQT(GenericMolFile): """ PDBQT Autodock and Autodock Vina format @@ -533,7 +532,7 @@ class PDBQT(GenericMolFile): """ file_ext = "pdbqt" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess if the file is a PDBQT file. @@ -541,12 +540,11 @@ class PDBQT(GenericMolFile): >>> fname = get_test_fname('NuBBE_1_obabel_3D.pdbqt') >>> PDBQT().sniff(fname) True - >>> fname = get_test_fname('drugbank_drugs.cml') >>> PDBQT().sniff(fname) False """ - headers = iter_headers(filename, sep=' ', count=300) + headers = iter_headers(file_prefix, sep=' ', count=300) h = t = c = s = k = False for line in headers: section_name = line[0].strip() @@ -601,6 +599,7 @@ class grdtgz(Binary): dataset.blurb = 'file purged from disk' +@build_sniff_from_prefix class InChI(Tabular): file_ext = "inchi" column_names = ['InChI'] @@ -625,7 +624,7 @@ class InChI(Tabular): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disk' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess if the file is a InChI file. @@ -633,16 +632,17 @@ class InChI(Tabular): >>> fname = get_test_fname('drugbank_drugs.inchi') >>> InChI().sniff(fname) True - >>> fname = get_test_fname('drugbank_drugs.cml') >>> InChI().sniff(fname) False """ - inchi_lines = iter_headers(filename, sep=' ', count=10) + inchi_lines = iter_headers(file_prefix, sep=' ', count=10) + found_lines = False for inchi in inchi_lines: if not inchi[0].startswith('InChI='): return False - return True + found_lines = True + return found_lines class SMILES(Tabular): @@ -704,6 +704,7 @@ class SMILES(Tabular): ''' +@build_sniff_from_prefix class CML(GenericXml): """ Chemical Markup Language @@ -729,7 +730,7 @@ class CML(GenericXml): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disk' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess if the file is a CML file. @@ -737,18 +738,14 @@ class CML(GenericXml): >>> fname = get_test_fname('interval.interval') >>> CML().sniff(fname) False - >>> fname = get_test_fname('drugbank_drugs.cml') >>> CML().sniff(fname) True """ - with open(filename) as handle: - line = handle.readline() - if line.strip() != '': - return False - line = handle.readline() - if line.strip().find('http://www.xml-cml.org/schema') == -1: + for expected_string in ['', 'http://www.xml-cml.org/schema']: + if expected_string not in file_prefix.contents_header: return False + return True def split(cls, input_datasets, subdir_generator_function, split_params): diff --git a/lib/galaxy/datatypes/mothur.py b/lib/galaxy/datatypes/mothur.py index 89fa0d0154a..cec2b21dd06 100644 --- a/lib/galaxy/datatypes/mothur.py +++ b/lib/galaxy/datatypes/mothur.py @@ -8,6 +8,7 @@ import sys from galaxy.datatypes.data import Text from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes.sniff import ( + build_sniff_from_prefix, get_headers, iter_headers ) @@ -16,6 +17,7 @@ from galaxy.datatypes.tabular import Tabular log = logging.getLogger(__name__) +@build_sniff_from_prefix class Otu(Text): file_ext = 'mothur.otu' MetadataElement(name="columns", default=0, desc="Number of columns", readonly=True, visible=True, no_value=0) @@ -76,7 +78,7 @@ class Otu(Text): dataset.metadata.otulabels = list(otulabel_names) dataset.metadata.otulabels.sort() - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is otu (operational taxonomic unit) format @@ -88,7 +90,7 @@ class Otu(Text): >>> Otu().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -120,7 +122,7 @@ class Sabund(Otu): def init_meta(self, dataset, copy_from=None): super(Sabund, self).init_meta(dataset, copy_from=copy_from) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is otu (operational taxonomic unit) format labelcount[value(1..n)] @@ -133,7 +135,7 @@ class Sabund(Otu): >>> Sabund().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -196,7 +198,7 @@ class GroupAbund(Otu): dataset.metadata.groups.sort() dataset.metadata.skip = skip - def sniff(self, filename, vals_are_int=False): + def sniff_prefix(self, file_prefix, vals_are_int=False): """ Determines whether the file is a otu (operational taxonomic unit) Shared format @@ -211,7 +213,7 @@ class GroupAbund(Otu): >>> GroupAbund().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -235,6 +237,7 @@ class GroupAbund(Otu): return False +@build_sniff_from_prefix class SecondaryStructureMap(Tabular): file_ext = 'mothur.map' @@ -243,7 +246,7 @@ class SecondaryStructureMap(Tabular): super(SecondaryStructureMap, self).__init__(**kwd) self.column_names = ['Map'] - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a secondary structure map format A single column with an integer value which indicates the row that this @@ -258,7 +261,7 @@ class SecondaryStructureMap(Tabular): >>> SecondaryStructureMap().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') line_num = 0 rowidxmap = {} for line in headers: @@ -337,6 +340,7 @@ class DistanceMatrix(Text): log.warning("DistanceMatrix set_meta %s" % e) +@build_sniff_from_prefix class LowerTriangleDistanceMatrix(DistanceMatrix): file_ext = 'mothur.lower.dist' @@ -347,7 +351,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix): def init_meta(self, dataset, copy_from=None): super(LowerTriangleDistanceMatrix, self).init_meta(dataset, copy_from=copy_from) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a lower-triangle distance matrix (phylip) format The first line has the number of sequences in the matrix. @@ -368,7 +372,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix): False """ numlines = 300 - headers = iter_headers(filename, sep='\t', count=numlines) + headers = iter_headers(file_prefix, sep='\t', count=numlines) line_num = 0 for line in headers: if not line[0].startswith('@'): @@ -400,6 +404,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix): return False +@build_sniff_from_prefix class SquareDistanceMatrix(DistanceMatrix): file_ext = 'mothur.square.dist' @@ -409,7 +414,7 @@ class SquareDistanceMatrix(DistanceMatrix): def init_meta(self, dataset, copy_from=None): super(SquareDistanceMatrix, self).init_meta(dataset, copy_from=copy_from) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a square distance matrix (Column-formatted distance matrix) format The first line has the number of sequences in the matrix. @@ -429,7 +434,7 @@ class SquareDistanceMatrix(DistanceMatrix): False """ numlines = 300 - headers = iter_headers(filename, sep='\t', count=numlines) + headers = iter_headers(file_prefix, sep='\t', count=numlines) line_num = 0 for line in headers: if not line[0].startswith('@'): @@ -460,6 +465,7 @@ class SquareDistanceMatrix(DistanceMatrix): return False +@build_sniff_from_prefix class PairwiseDistanceMatrix(DistanceMatrix, Tabular): file_ext = 'mothur.pair.dist' @@ -472,7 +478,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular): def set_meta(self, dataset, overwrite=True, skip=None, **kwd): super(PairwiseDistanceMatrix, self).set_meta(dataset, overwrite=overwrite, skip=skip, **kwd) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a pairwise distance matrix (Column-formatted distance matrix) format The first and second columns have the sequence names and the third column is the distance between those sequences. @@ -485,7 +491,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular): >>> PairwiseDistanceMatrix().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -566,10 +572,11 @@ class AccNos(Tabular): self.columns = 1 +@build_sniff_from_prefix class Oligos(Text): file_ext = 'mothur.oligos' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ http://www.mothur.org/wiki/Oligos_File Determines whether the file is a otu (operational taxonomic unit) format @@ -582,7 +589,7 @@ class Oligos(Text): >>> Oligos().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 for line in headers: if not line[0].startswith('@') and not line[0].startswith('#'): @@ -600,6 +607,7 @@ class Oligos(Text): return False +@build_sniff_from_prefix class Frequency(Tabular): file_ext = 'mothur.freq' @@ -609,7 +617,7 @@ class Frequency(Tabular): self.column_names = ['position', 'frequency'] self.column_types = ['int', 'float'] - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a frequency tabular format for chimera analysis #1.14.0 @@ -625,13 +633,12 @@ class Frequency(Tabular): >>> fname = get_test_fname( 'mothur_datatypetest_false.mothur.freq' ) >>> Frequency().sniff( fname ) False - - # Expression count matrix (EdgeR wrapper) + >>> # Expression count matrix (EdgeR wrapper) >>> fname = get_test_fname( 'mothur_datatypetest_false_2.mothur.freq' ) >>> Frequency().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -660,6 +667,7 @@ class Frequency(Tabular): return False +@build_sniff_from_prefix class Quantile(Tabular): file_ext = 'mothur.quan' MetadataElement(name="filtered", default=False, no_value=False, optional=True, desc="Quantiles calculated using a mask", readonly=True) @@ -671,7 +679,7 @@ class Quantile(Tabular): self.column_names = ['num', 'ten', 'twentyfive', 'fifty', 'seventyfive', 'ninetyfive', 'ninetynine'] self.column_types = ['int', 'float', 'float', 'float', 'float', 'float', 'float'] - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a quantiles tabular format for chimera analysis 1 0 0 0 0 0 0 @@ -687,7 +695,7 @@ class Quantile(Tabular): >>> Quantile().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 for line in headers: if not line[0].startswith('@') and not line[0].startswith('#'): @@ -710,10 +718,11 @@ class Quantile(Tabular): return False +@build_sniff_from_prefix class LaneMask(Text): file_ext = 'mothur.filter' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a lane mask filter: 1 line consisting of zeros and ones. @@ -725,7 +734,7 @@ class LaneMask(Text): >>> LaneMask().sniff( fname ) False """ - headers = get_headers(filename, sep='\t', count=2) + headers = get_headers(file_prefix, sep='\t', count=2) if len(headers) != 1 or len(headers[0]) != 1: return False @@ -775,6 +784,7 @@ class CountTable(Tabular): dataset.metadata.data_lines -= 1 +@build_sniff_from_prefix class RefTaxonomy(Tabular): file_ext = 'mothur.ref.taxonomy' @@ -782,7 +792,7 @@ class RefTaxonomy(Tabular): super(RefTaxonomy, self).__init__(**kwd) self.column_names = ['name', 'taxonomy'] - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is a Reference Taxonomy @@ -808,7 +818,7 @@ class RefTaxonomy(Tabular): >>> RefTaxonomy().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t', count=300) + headers = iter_headers(file_prefix, sep='\t', count=300) count = 0 pat_prog = re.compile('^([^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?(;[^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?)*(;)?)$') found_semicolons = False @@ -852,6 +862,7 @@ class TaxonomySummary(Tabular): self.column_names = ['taxlevel', 'rankID', 'taxon', 'daughterlevels', 'total'] +@build_sniff_from_prefix class Axes(Tabular): file_ext = 'mothur.axes' @@ -859,7 +870,7 @@ class Axes(Tabular): """Initialize axes datatype""" super(Axes, self).__init__(**kwd) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is an axes format The first line may have column headings. @@ -883,7 +894,7 @@ class Axes(Tabular): >>> Axes().sniff( fname ) False """ - headers = iter_headers(filename, sep='\t') + headers = iter_headers(file_prefix, sep='\t') count = 0 col_cnt = None all_integers = True diff --git a/lib/galaxy/datatypes/msa.py b/lib/galaxy/datatypes/msa.py index da8f0a7ae63..770f797e444 100644 --- a/lib/galaxy/datatypes/msa.py +++ b/lib/galaxy/datatypes/msa.py @@ -1,16 +1,21 @@ import abc import logging import os +import re from galaxy.datatypes.binary import Binary from galaxy.datatypes.data import get_file_peek, Text from galaxy.datatypes.metadata import MetadataElement +from galaxy.datatypes.sniff import build_sniff_from_prefix from galaxy.datatypes.util import generic_util from galaxy.util import nice_size log = logging.getLogger(__name__) +STOCKHOLM_SEARCH_PATTERN = re.compile(r'#\s+STOCKHOLM\s+1\.0') + +@build_sniff_from_prefix class InfernalCM(Text): file_ext = "cm" @@ -32,20 +37,17 @@ class InfernalCM(Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ >>> from galaxy.datatypes.sniff import get_test_fname >>> fname = get_test_fname( 'infernal_model.cm' ) >>> InfernalCM().sniff( fname ) True - >>> fname = get_test_fname( 'test.mz5' ) + >>> fname = get_test_fname( '2.txt' ) >>> InfernalCM().sniff( fname ) False """ - with open(filename, 'r') as f: - first_line = f.readline() - - return first_line.startswith("INFERNAL") + return file_prefix.startswith("INFERNAL") def set_meta(self, dataset, **kwd): """ @@ -58,6 +60,7 @@ class InfernalCM(Text): dataset.metadata.cm_version = (first_line.split()[0]).replace('INFERNAL', '') +@build_sniff_from_prefix class Hmmer(Text): edam_data = "data_1364" edam_format = "format_1370" @@ -77,7 +80,7 @@ class Hmmer(Text): return "HMMER database (%s)" % (nice_size(dataset.get_size())) @abc.abstractmethod - def sniff(self, filename): + def sniff_prefix(self, filename): raise NotImplementedError @@ -85,24 +88,20 @@ class Hmmer2(Hmmer): edam_format = "format_3328" file_ext = "hmm2" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """HMMER2 files start with HMMER2.0 """ - with open(filename, 'r') as handle: - return handle.read(8) == 'HMMER2.0' - return False + return file_prefix.startswith('HMMER2.0') class Hmmer3(Hmmer): edam_format = "format_3329" file_ext = "hmm3" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """HMMER3 files start with HMMER3/f """ - with open(filename, 'r') as handle: - return handle.read(8) == 'HMMER3/f' - return False + return file_prefix.startswith('HMMER3/f') class HmmerPress(Binary): @@ -139,6 +138,7 @@ class HmmerPress(Binary): self.add_composite_file('model.hmm.h3p', is_binary=True) +@build_sniff_from_prefix class Stockholm_1_0(Text): edam_data = "data_0863" edam_format = "format_1961" @@ -157,11 +157,8 @@ class Stockholm_1_0(Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): - if generic_util.count_special_lines('^#[[:space:]+]STOCKHOLM[[:space:]+]1.0', filename) > 0: - return True - else: - return False + def sniff_prefix(self, file_prefix): + return file_prefix.search(STOCKHOLM_SEARCH_PATTERN) def set_meta(self, dataset, **kwd): """ @@ -222,6 +219,7 @@ class Stockholm_1_0(Text): split = classmethod(split) +@build_sniff_from_prefix class MauveXmfa(Text): file_ext = "xmfa" @@ -238,10 +236,8 @@ class MauveXmfa(Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): - with open(filename, 'r') as handle: - return handle.read(21) == '#FormatVersion Mauve1' - return False + def sniff_prefix(self, file_prefix): + return file_prefix.startswith('#FormatVersion Mauve1') def set_meta(self, dataset, **kwd): dataset.metadata.number_of_models = generic_util.count_special_lines('^#Sequence([[:digit:]]+)Entry', dataset.file_name) diff --git a/lib/galaxy/datatypes/phylip.py b/lib/galaxy/datatypes/phylip.py index 6c62945e463..7de64e296b0 100644 --- a/lib/galaxy/datatypes/phylip.py +++ b/lib/galaxy/datatypes/phylip.py @@ -9,10 +9,12 @@ Phylip datatype sniffer """ from galaxy import util from galaxy.datatypes.data import get_file_peek, Text +from galaxy.datatypes.sniff import build_sniff_from_prefix from galaxy.util import nice_size from .metadata import MetadataElement +@build_sniff_from_prefix class Phylip(Text): """Phylip format stores a multiple sequence alignment""" edam_data = "data_0863" @@ -44,7 +46,7 @@ class Phylip(Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disk' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ All Phylip files starts with the number of sequences so we can use this to count the following number of sequences in the first 'stack' @@ -54,15 +56,15 @@ class Phylip(Text): >>> Phylip().sniff(fname) True """ - with open(filename, "r") as f: - # Get number of sequence from first line - nb_seq = int(f.readline().split()[0]) - # counts number of sequence from first stack - count = 0 - for line in f: - if not line.split(): - break - count += 1 - if count > nb_seq: - return False + f = file_prefix.string_io() + # Get number of sequence from first line + nb_seq = int(f.readline().split()[0]) + # counts number of sequence from first stack + count = 0 + for line in f: + if not line.split(): + break + count += 1 + if count > nb_seq: + return False return count == nb_seq diff --git a/lib/galaxy/datatypes/plant_tribes.py b/lib/galaxy/datatypes/plant_tribes.py index 1b804160970..017c6473b8c 100644 --- a/lib/galaxy/datatypes/plant_tribes.py +++ b/lib/galaxy/datatypes/plant_tribes.py @@ -3,13 +3,14 @@ import re from galaxy.datatypes.data import get_file_peek, Text from galaxy.datatypes.metadata import MetadataElement -from galaxy.datatypes.sniff import get_headers +from galaxy.datatypes.sniff import build_sniff_from_prefix, get_headers from galaxy.datatypes.tabular import Tabular from galaxy.util import nice_size log = logging.getLogger(__name__) +@build_sniff_from_prefix class Smat(Text): file_ext = "smat" @@ -27,7 +28,7 @@ class Smat(Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ The use of ESTScan implies the creation of scores matrices which reflect the codons preferences in the studied organisms. The @@ -51,26 +52,27 @@ class Smat(Text): True """ line_no = 0 - with open(filename, "r") as fh: - for line in fh: - line_no += 1 - if line_no > 10000: - return True - if line_no == 1 and not line.startswith('FORMAT'): - # The first line is always the start of a format section. + fh = file_prefix.string_io() + for line in fh: + line_no += 1 + if line_no > 10000: + return True + if line_no == 1 and not line.startswith('FORMAT'): + # The first line is always the start of a format section. + return False + if not line.startswith('FORMAT'): + if line.find('\t') >= 0: + # Smat files are not tabular. return False - if not line.startswith('FORMAT'): - if line.find('\t') >= 0: - # Smat files are not tabular. + items = line.split() + if len(items) != 4: + return False + for item in items: + # Make sure each item is an integer. + if re.match(r"[-+]?\d+$", item) is None: return False - items = line.split() - if len(items) != 4: - return False - for item in items: - # Make sure each item is an integer. - if re.match(r"[-+]?\d+$", item) is None: - return False - return True + # Ensure at least a few matching lines are found. + return line_no > 2 # These commented classes are required by versions 1.0.0, 1.0.1 and 1.0.2 of the diff --git a/lib/galaxy/datatypes/proteomics.py b/lib/galaxy/datatypes/proteomics.py index 2ca7ff891f3..b3de1f5a97a 100644 --- a/lib/galaxy/datatypes/proteomics.py +++ b/lib/galaxy/datatypes/proteomics.py @@ -7,6 +7,7 @@ import re from galaxy.datatypes import data from galaxy.datatypes.binary import Binary from galaxy.datatypes.data import Text +from galaxy.datatypes.sniff import build_sniff_from_prefix from galaxy.datatypes.tabular import Tabular from galaxy.datatypes.xml import GenericXml from galaxy.util import nice_size @@ -97,16 +98,16 @@ class ProteomicsXml(GenericXml): edam_data = "data_2536" edam_format = "format_2032" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is the correct XML type. """ - with open(filename, 'r') as contents: - while True: - line = contents.readline() - if line is None or not line.startswith('>> from galaxy.datatypes.sniff import get_test_fname >>> fname = get_test_fname( 'sequence.fasta' ) @@ -34,34 +38,34 @@ class QualityScoreSOLiD(QualityScore): >>> QualityScoreSOLiD().sniff( fname ) True """ - with open(filename) as fh: - readlen = None - goodblock = 0 - while True: - line = fh.readline() - if not line: - if goodblock > 0: + fh = file_prefix.string_io() + readlen = None + goodblock = 0 + while True: + line = fh.readline() + if not line: + if goodblock > 0: + return True + else: + break # EOF + line = line.strip() + if line and not line.startswith('#'): # first non-empty non-comment line + if line.startswith('>'): + line = fh.readline().strip() + if line == '' or line.startswith('>'): + break + try: + [int(x) for x in line.split()] + if not(readlen): + readlen = len(line.split()) + assert len(line.split()) == readlen # SOLiD reads should be of the same length + except Exception: + break + goodblock += 1 + if goodblock > 10: return True - else: - break # EOF - line = line.strip() - if line and not line.startswith('#'): # first non-empty non-comment line - if line.startswith('>'): - line = fh.readline().strip() - if line == '' or line.startswith('>'): - break - try: - [int(x) for x in line.split()] - if not(readlen): - readlen = len(line.split()) - assert len(line.split()) == readlen # SOLiD reads should be of the same length - except Exception: - break - goodblock += 1 - if goodblock > 10: - return True - else: - break # we found a non-empty line, but it's not a header + else: + break # we found a non-empty line, but it's not a header return False def set_meta(self, dataset, **kwd): @@ -71,6 +75,7 @@ class QualityScoreSOLiD(QualityScore): return QualityScore.set_meta(self, dataset, **kwd) +@sniff.build_sniff_from_prefix class QualityScore454(QualityScore): """ until we know more about quality score formats @@ -78,7 +83,7 @@ class QualityScore454(QualityScore): edam_format = "format_3611" file_ext = "qual454" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ >>> from galaxy.datatypes.sniff import get_test_fname >>> fname = get_test_fname( 'sequence.fasta' ) @@ -88,24 +93,24 @@ class QualityScore454(QualityScore): >>> QualityScore454().sniff( fname ) True """ - with open(filename) as fh: - while True: - line = fh.readline() - if not line: - break # EOF - line = line.strip() - if line and not line.startswith('#'): # first non-empty non-comment line - if line.startswith('>'): - line = fh.readline().strip() - if line == '' or line.startswith('>'): - break - try: - [int(x) for x in line.split()] - except Exception: - break - return True - else: - break # we found a non-empty line, but it's not a header + fh = file_prefix.string_io() + while True: + line = fh.readline() + if not line: + break # EOF + line = line.strip() + if line and not line.startswith('#'): # first non-empty non-comment line + if line.startswith('>'): + line = fh.readline().strip() + if line == '' or line.startswith('>'): + break + try: + [int(x) for x in line.split()] + except Exception: + break + return True + else: + break # we found a non-empty line, but it's not a header return False diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index d55a0442a7b..c2f51095afb 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -22,8 +22,9 @@ from galaxy.datatypes.binary import ( ) from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes.sniff import ( + build_sniff_from_prefix, get_headers, - iter_headers + iter_headers, ) from galaxy.util import ( compression_utils, @@ -46,6 +47,7 @@ SNIFF_COMPRESSED_FASTAS = os.environ.get("GALAXY_ENABLE_BETA_COMPRESSED_FASTA_SN SNIFF_COMPRESSED_GENBANKS = os.environ.get("GALAXY_ENABLE_BETA_COMPRESSED_GENBANK_SNIFFING", "0") == "1" +@build_sniff_from_prefix class SequenceSplitLocations(data.Text): """ Class storing information about a sequence file composed of multiple gzip files concatenated as @@ -75,10 +77,10 @@ class SequenceSplitLocations(data.Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disk' - def sniff(self, filename): - if os.path.getsize(filename) < 50000: + def sniff_prefix(self, file_prefix): + if file_prefix.file_size < 50000 and not file_prefix.truncated: try: - data = json.load(open(filename)) + data = json.loads(file_prefix.contents_header) sections = data['sections'] for section in sections: if 'start' not in section or 'end' not in section or 'sequences' not in section: @@ -330,12 +332,13 @@ class FastaGz(Sequence, CompressedArchive): return Sequence.sniff(self, filename) +@build_sniff_from_prefix class Fasta(Sequence): """Class representing a FASTA sequence""" edam_format = "format_1929" file_ext = "fasta" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in fasta format @@ -369,26 +372,26 @@ class Fasta(Sequence): >>> Fasta().sniff( fname ) True """ - with open(filename) as fh: - while True: - line = fh.readline() - if not line: - break # EOF - line = line.strip() - if line: # first non-empty line - if line.startswith('>'): - # The next line.strip() must not be '', nor startwith '>' - line = fh.readline().strip() - if line == '' or line.startswith('>'): - break + fh = file_prefix.string_io() + while True: + line = fh.readline() + if not line: + break # EOF + line = line.strip() + if line: # first non-empty line + if line.startswith('>'): + # The next line.strip() must not be '', nor startwith '>' + line = fh.readline().strip() + if line == '' or line.startswith('>'): + break - # If there is a third line, and it isn't a header line, it may not contain chars like '()[].' otherwise it's most likely a DotBracket file - line = fh.readline() - if not line.startswith('>') and re.search("[\(\)\[\]\.]", line): - break - return True - else: - break # we found a non-empty line, but it's not a fasta header + # If there is a third line, and it isn't a header line, it may not contain chars like '()[].' otherwise it's most likely a DotBracket file + line = fh.readline() + if not line.startswith('>') and re.search("[\(\)\[\]\.]", line): + break + return True + else: + break # we found a non-empty line, but it's not a fasta header return False def split(cls, input_datasets, subdir_generator_function, split_params): @@ -514,12 +517,13 @@ class Fasta(Sequence): _count_split = classmethod(_count_split) +@build_sniff_from_prefix class csFasta(Sequence): """ Class representing the SOLID Color-Space sequence ( csfasta ) """ edam_format = "format_3589" file_ext = "csfasta" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Color-space sequence: >2_15_85_F3 @@ -533,24 +537,24 @@ class csFasta(Sequence): >>> csFasta().sniff( fname ) True """ - with open(filename) as fh: - while True: - line = fh.readline() - if not line: - break # EOF - line = line.strip() - if line and not line.startswith('#'): # first non-empty non-comment line - if line.startswith('>'): - line = fh.readline().strip() - if line == '' or line.startswith('>'): - break - elif line[0] not in string.ascii_uppercase: - return False - elif len(line) > 1 and not re.search('^[\d.]+$', line[1:]): - return False - return True - else: - break # we found a non-empty line, but it's not a header + fh = file_prefix.string_io() + while True: + line = fh.readline() + if not line: + break # EOF + line = line.strip() + if line and not line.startswith('#'): # first non-empty non-comment line + if line.startswith('>'): + line = fh.readline().strip() + if line == '' or line.startswith('>'): + break + elif line[0] not in string.ascii_uppercase: + return False + elif len(line) > 1 and not re.search('^[\d.]+$', line[1:]): + return False + return True + else: + break # we found a non-empty line, but it's not a header return False def set_meta(self, dataset, **kwd): @@ -561,6 +565,7 @@ class csFasta(Sequence): return Sequence.set_meta(self, dataset, **kwd) +@build_sniff_from_prefix class BaseFastq(Sequence): """Base class for FastQ sequences""" edam_format = "format_1930" @@ -599,7 +604,7 @@ class BaseFastq(Sequence): dataset.metadata.data_lines = data_lines dataset.metadata.sequences = sequences - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in generic fastq format For details, see http://maq.sourceforge.net/fastq.shtml @@ -620,11 +625,10 @@ class BaseFastq(Sequence): >>> FastqSanger().sniff( fname ) False """ - compressed = is_gzip(filename) or is_bz2(filename) + compressed = file_prefix.compressed_format is not None if compressed and not isinstance(self, Binary): return False - headers = iter_headers(filename, None, count=1000) - + headers = iter_headers(file_prefix, None, count=1000) # If this is a FastqSanger-derived class, then check to see if the base qualities match if isinstance(self, FastqSanger) or isinstance(self, FastqSangerGz) or isinstance(self, FastqSangerBz2): if not self.sangerQualities(headers): @@ -633,7 +637,7 @@ class BaseFastq(Sequence): bases_regexp = re.compile("^[NGTAC]*") # check that first block looks like a fastq block try: - headers = get_headers(filename, None, count=4) + headers = get_headers(file_prefix, None, count=4) if len(headers) == 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]: # Check the sequence line, make sure it contains only G/C/A/T/N if not bases_regexp.match(headers[1][0]): @@ -818,6 +822,7 @@ class FastqCSSangerBz2(FastqBz2): file_ext = "fastqcssanger.bz2" +@build_sniff_from_prefix class Maf(Alignment): """Class describing a Maf alignment""" edam_format = "format_3008" @@ -900,7 +905,7 @@ class Maf(Alignment): out = "Can't create peek %s" % exc return out - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines wether the file is in maf format @@ -923,7 +928,7 @@ class Maf(Alignment): >>> Maf().sniff( fname ) False """ - headers = get_headers(filename, None) + headers = get_headers(file_prefix, None) try: if len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf": return True @@ -971,6 +976,7 @@ class MafCustomTrack(data.Text): pass +@build_sniff_from_prefix class Axt(data.Text): """Class describing an axt alignment""" # gvk- 11/19/09 - This is really an alignment, but we no longer have tools that use this data type, and it is @@ -981,7 +987,7 @@ class Axt(data.Text): edam_format = "format_3013" file_ext = "axt" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in axt format @@ -1007,7 +1013,7 @@ class Axt(data.Text): >>> Axt().sniff( fname ) False """ - headers = get_headers(filename, None) + headers = get_headers(file_prefix, None) if len(headers) < 4: return False for hdr in headers: @@ -1026,6 +1032,7 @@ class Axt(data.Text): return True +@build_sniff_from_prefix class Lav(data.Text): """Class describing a LAV alignment""" # gvk- 11/19/09 - This is really an alignment, but we no longer have tools that use this data type, and it is @@ -1036,7 +1043,7 @@ class Lav(data.Text): edam_format = "format_3014" file_ext = "lav" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in lav format @@ -1053,7 +1060,7 @@ class Lav(data.Text): >>> Lav().sniff( fname ) False """ - headers = get_headers(filename, None) + headers = get_headers(file_prefix, None) try: if len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav'): return True @@ -1096,6 +1103,7 @@ class RNADotPlotMatrix(data.Data): return False +@build_sniff_from_prefix class DotBracket(Sequence): edam_data = "data_0880" edam_format = "format_1457" @@ -1128,7 +1136,7 @@ class DotBracket(Sequence): dataset.metadata.data_lines = data_lines dataset.metadata.sequences = sequences - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Galaxy Dbn (Dot-Bracket notation) rules: @@ -1160,47 +1168,47 @@ class DotBracket(Sequence): state = 0 - with open(filename, "r") as handle: - for line in handle: - line = line.strip() + for line in file_prefix.line_iterator(): + line = line.strip() - if line: - # header line - if state == 0: - if(line[0] != '>'): - return False - else: - state = 1 + if line: + # header line + if state == 0: + if(line[0] != '>'): + return False + else: + state = 1 - # sequence line - elif state == 1: - if not self.sequence_regexp.match(line): - return False - else: - sequence_size = len(line) - state = 2 + # sequence line + elif state == 1: + if not self.sequence_regexp.match(line): + return False + else: + sequence_size = len(line) + state = 2 - # dot-bracket structure line - elif state == 2: - if sequence_size != len(line) or not self.structure_regexp.match(line) or \ - line.count('(') != line.count(')') or \ - line.count('[') != line.count(']') or \ - line.count('{') != line.count('}'): - return False - else: - return True + # dot-bracket structure line + elif state == 2: + if sequence_size != len(line) or not self.structure_regexp.match(line) or \ + line.count('(') != line.count(')') or \ + line.count('[') != line.count(']') or \ + line.count('{') != line.count('}'): + return False + else: + return True # Number of lines is less than 3 return False +@build_sniff_from_prefix class Genbank(data.Text): """Class representing a Genbank sequence""" edam_format = "format_1936" edam_data = "data_0849" file_ext = "genbank" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determine whether the file is in genbank format. Works for compressed files. @@ -1210,15 +1218,10 @@ class Genbank(data.Text): >>> Genbank().sniff( fname ) True """ - compressed = is_gzip(filename) + compressed = file_prefix.compressed_format if compressed and not isinstance(self, Binary): return False - try: - with compression_utils.get_fileobj(filename) as file: - return 'LOCUS ' == file.read(6) - except Exception: - pass - return False + return 'LOCUS ' == file_prefix.contents_header[0:6] class GenbankGz(Genbank, CompressedArchive): @@ -1244,11 +1247,12 @@ class GenbankGz(Genbank, CompressedArchive): return Genbank.sniff(self, filename) +@build_sniff_from_prefix class MemePsp(Sequence): """Class representing MEME Position Specific Priors""" file_ext = "memepsp" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ The format of an entry in a PSP file is: @@ -1274,34 +1278,34 @@ class MemePsp(Sequence): return True try: num_lines = 0 - with open(filename) as fh: - line = fh.readline() - if not line: - # EOF. - return False - num_lines += 1 - if num_lines > 100: - return True - line = line.strip() - if line: - if line.startswith('>'): - # The line must not be blank, nor start with '>' - line = fh.readline().strip() - if line == '' or line.startswith('>'): - return False - # All items within the line must be floats. + fh = file_prefix.string_io() + line = fh.readline() + if not line: + # EOF. + return False + num_lines += 1 + if num_lines > 100: + return True + line = line.strip() + if line: + if line.startswith('>'): + # The line must not be blank, nor start with '>' + line = fh.readline().strip() + if line == '' or line.startswith('>'): + return False + # All items within the line must be floats. + if not floats_verified(line): + return False + # If there is a second line within the ID section, + # all items within the line must be floats. + line = fh.readline().strip() + if line: if not floats_verified(line): return False - # If there is a second line within the ID section, - # all items within the line must be floats. - line = fh.readline().strip() - if line: - if not floats_verified(line): - return False - else: - # We found a non-empty line, - # but it's not a psp id width. - return False + else: + # We found a non-empty line, + # but it's not a psp id width. + return False except Exception: return False # We've reached EOF in less than 100 lines. diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 709d5b02eb7..ddf686384b8 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -13,7 +13,7 @@ import sys import tempfile import zipfile -from six import text_type +from six import StringIO, text_type from six.moves import filter from six.moves.urllib.request import urlopen @@ -35,6 +35,8 @@ else: log = logging.getLogger(__name__) +SNIFF_PREFIX_BYTES = int(os.environ.get("GALAXY_SNIFF_PREFIX_BYTES", None) or 2 ** 20) + def get_test_fname(fname): """Returns test data filename""" @@ -189,10 +191,10 @@ def convert_newlines_sep2tabs(fname, in_place=True, patt="\\s+", tmp_dir=None, t return (i + 1, temp_name) -def iter_headers(fname, sep, count=60, comment_designator=None): - with compression_utils.get_fileobj(fname) as in_file: +def iter_headers(fname_or_file_prefix, sep, count=60, comment_designator=None): + if isinstance(fname_or_file_prefix, FilePrefix): idx = 0 - for line in in_file: + for line in fname_or_file_prefix.line_iterator(): line = line.rstrip('\n\r') if comment_designator is not None and comment_designator != '' and line.startswith(comment_designator): continue @@ -200,9 +202,20 @@ def iter_headers(fname, sep, count=60, comment_designator=None): idx += 1 if idx == count: break + else: + with compression_utils.get_fileobj(fname_or_file_prefix) as in_file: + idx = 0 + for line in in_file: + line = line.rstrip('\n\r') + if comment_designator is not None and comment_designator != '' and line.startswith(comment_designator): + continue + yield line.split(sep) + idx += 1 + if idx == count: + break -def get_headers(fname, sep, count=60, comment_designator=None): +def get_headers(fname_or_file_prefix, sep, count=60, comment_designator=None): """ Returns a list with the first 'count' lines split by 'sep', ignoring lines starting with 'comment_designator' @@ -214,10 +227,10 @@ def get_headers(fname, sep, count=60, comment_designator=None): >>> get_headers(fname, '\\t', count=5, comment_designator='#') == [[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']] True """ - return list(iter_headers(fname=fname, sep=sep, count=count, comment_designator=comment_designator)) + return list(iter_headers(fname_or_file_prefix=fname_or_file_prefix, sep=sep, count=count, comment_designator=comment_designator)) -def is_column_based(fname, sep='\t', skip=0): +def is_column_based(fname_or_file_prefix, sep='\t', skip=0): """ Checks whether the file is column based with respect to a separator (defaults to tab separator). @@ -245,8 +258,11 @@ def is_column_based(fname, sep='\t', skip=0): >>> is_column_based(fname) True """ + if getattr(fname_or_file_prefix, "binary", None) is True: + return False + try: - headers = get_headers(fname, sep) + headers = get_headers(fname_or_file_prefix, sep) except UnicodeDecodeError: return False count = 0 @@ -306,17 +322,14 @@ def guess_ext(fname, sniff_order, is_binary=False): >>> fname = get_test_fname('gff_version_3.gff') >>> guess_ext(fname, sniff_order) 'gff3' - >>> fname = get_test_fname('temp.txt') - >>> open(fname, 'wt').write("a\\t2") - >>> guess_ext(fname, sniff_order) + >>> fname = get_test_fname('2.txt') + >>> guess_ext(fname, sniff_order) # 2.txt 'txt' - >>> fname = get_test_fname('temp.txt') - >>> open(fname, 'wt').write("a\\t2\\nc\\t1\\nd\\t0") + >>> fname = get_test_fname('2.tabular') >>> guess_ext(fname, sniff_order) 'tabular' - >>> fname = get_test_fname('temp.txt') - >>> open(fname, 'wt').write("a 1 2 x\\nb 3 4 y\\nc 5 6 z") - >>> guess_ext(fname, sniff_order) + >>> fname = get_test_fname('3.txt') + >>> guess_ext(fname, sniff_order) # 3.txt 'txt' >>> fname = get_test_fname('test_tab1.tabular') >>> guess_ext(fname, sniff_order) @@ -363,6 +376,29 @@ def guess_ext(fname, sniff_order, is_binary=False): >>> fname = get_test_fname('mothur_datatypetest_true.mothur.otu') >>> guess_ext(fname, sniff_order) 'mothur.otu' + >>> fname = get_test_fname('mothur_datatypetest_true.mothur.lower.dist') + >>> guess_ext(fname, sniff_order) + 'mothur.lower.dist' + >>> fname = get_test_fname('mothur_datatypetest_true.mothur.square.dist') + >>> guess_ext(fname, sniff_order) + 'mothur.square.dist' + >>> fname = get_test_fname('mothur_datatypetest_true.mothur.pair.dist') + >>> guess_ext(fname, sniff_order) + 'mothur.pair.dist' + >>> fname = get_test_fname('mothur_datatypetest_true.mothur.freq') + >>> guess_ext(fname, sniff_order) + 'mothur.freq' + >>> fname = get_test_fname('mothur_datatypetest_true.mothur.quan') + >>> guess_ext(fname, sniff_order) + 'mothur.quan' + >>> fname = get_test_fname('mothur_datatypetest_true.mothur.ref.taxonomy') + >>> guess_ext(fname, sniff_order) + 'mothur.ref.taxonomy' + >>> fname = get_test_fname('mothur_datatypetest_true.mothur.axes') + >>> guess_ext(fname, sniff_order) + 'mothur.axes' + >>> guess_ext(get_test_fname('infernal_model.cm'), sniff_order) + 'cm' >>> fname = get_test_fname('1.gg') >>> guess_ext(fname, sniff_order) 'gg' @@ -378,7 +414,83 @@ def guess_ext(fname, sniff_order, is_binary=False): >>> fname = get_test_fname('454Score.pdf') >>> guess_ext(fname, sniff_order) 'pdf' + >>> fname = get_test_fname('1.obo') + >>> guess_ext(fname, sniff_order) + 'obo' + >>> fname = get_test_fname('1.arff') + >>> guess_ext(fname, sniff_order) + 'arff' + >>> fname = get_test_fname('1.afg') + >>> guess_ext(fname, sniff_order) + 'afg' + >>> fname = get_test_fname('1.owl') + >>> guess_ext(fname, sniff_order) + 'owl' + >>> fname = get_test_fname('Acanium.hmm') + >>> guess_ext(fname, sniff_order) + 'snaphmm' + >>> fname = get_test_fname('wiggle.wig') + >>> guess_ext(fname, sniff_order) + 'wig' + >>> fname = get_test_fname('example.iqtree') + >>> guess_ext(fname, sniff_order) + 'iqtree' + >>> fname = get_test_fname('1.stockholm') + >>> guess_ext(fname, sniff_order) + 'stockholm' + >>> fname = get_test_fname('1.xmfa') + >>> guess_ext(fname, sniff_order) + 'xmfa' + >>> fname = get_test_fname('test.phylip') + >>> guess_ext(fname, sniff_order) + 'phylip' + >>> fname = get_test_fname('1.smat') + >>> guess_ext(fname, sniff_order) + 'smat' + >>> fname = get_test_fname('1.ttl') + >>> guess_ext(fname, sniff_order) + 'ttl' + >>> fname = get_test_fname('1.hdt') + >>> guess_ext(fname, sniff_order) + 'hdt' + >>> fname = get_test_fname('1.phyloxml') + >>> guess_ext(fname, sniff_order) + 'phyloxml' """ + file_prefix = FilePrefix(fname) + file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary) + + # Ugly hack for tsv vs tabular sniffing, we want to prefer tabular + # to tsv but it doesn't have a sniffer - is TSV was sniffed just check + # if it is an okay tabular and use that instead. + if file_ext == 'tsv': + if is_column_based(file_prefix, '\t', 1): + file_ext = 'tabular' + if file_ext is not None: + return file_ext + + # skip header check if data is already known to be binary + if is_binary: + return file_ext or 'binary' + try: + get_headers(file_prefix, None) + except UnicodeDecodeError: + return 'data' # default data type file extension + if is_column_based(file_prefix, '\t', 1): + return 'tabular' # default tabular data type file extension + return 'txt' # default text data type file extension + + +def run_sniffers_raw(filename_or_file_prefix, sniff_order, is_binary=False): + """Run through sniffers specified by sniff_order, return None of None match. + """ + if isinstance(filename_or_file_prefix, FilePrefix): + fname = filename_or_file_prefix.filename + file_prefix = filename_or_file_prefix + else: + fname = filename_or_file_prefix + file_prefix = FilePrefix(filename_or_file_prefix) + file_ext = None for datatype in sniff_order: """ @@ -390,31 +502,29 @@ def guess_ext(fname, sniff_order, is_binary=False): successfully discovered. """ try: - if ((is_binary and datatype.is_binary) or - (not is_binary)) and datatype.sniff(fname): + if hasattr(datatype, "sniff_prefix"): + datatype_compressed = getattr(datatype, "compressed", False) + if datatype_compressed and not file_prefix.compressed_format: + continue + if not datatype_compressed and file_prefix.compressed_format: + continue + if file_prefix.compressed_format and getattr(datatype, "compressed_format"): + # In this case go a step further and compare the compressed format detected + # to the expected. + if file_prefix.compressed_format != datatype.compressed_format: + continue + if datatype.sniff_prefix(file_prefix): + file_ext = datatype.file_ext + break + elif is_binary and not datatype.is_binary: + continue + elif datatype.sniff(fname): file_ext = datatype.file_ext break except Exception: pass - # Ugly hack for tsv vs tabular sniffing, we want to prefer tabular - # to tsv but it doesn't have a sniffer - is TSV was sniffed just check - # if it is an okay tabular and use that instead. - if file_ext == 'tsv': - if is_column_based(fname, '\t', 1): - file_ext = 'tabular' - if file_ext is not None: - return file_ext - # skip header check if data is already known to be binary - if is_binary: - return file_ext or 'binary' - try: - get_headers(fname, None) - except UnicodeDecodeError: - return 'data' # default data type file extension - if is_column_based(fname, '\t', 1): - return 'tabular' # default tabular data type file extension - return 'txt' # default text data type file extension + return file_ext def zip_single_fileobj(path): @@ -424,6 +534,91 @@ def zip_single_fileobj(path): return z.open(name) +class FilePrefix(object): + + def __init__(self, filename): + binary = False + compressed_format = None + contents_header = None # First MAX_BYTES of the file. + truncated = False + # A future direction to optimize sniffing even more for sniffers at the top of the list + # is to lazy load contents_header based on what interface is requested. For instance instead + # of returning a StringIO directly in string_io() return an object that reads the contents and + # populates contents_header while providing a StringIO-like interface until the file is read + # but then would fallback to native string_io() + try: + compressed_format, f = compression_utils.get_fileobj_raw(filename) + try: + contents_header = f.read(SNIFF_PREFIX_BYTES) + truncated = len(contents_header) == SNIFF_PREFIX_BYTES + finally: + f.close() + except UnicodeDecodeError: + binary = True + + self.truncated = truncated + self.filename = filename + self.binary = binary + self.compressed_format = compressed_format + self.contents_header = contents_header + self._file_size = None + + @property + def file_size(self): + if self._file_size is None: + self._file_size = os.path.getsize(self.filename) + return self._file_size + + def string_io(self): + if self.binary: + raise Exception("Attempting to create a StringIO object for binary data.") + rval = StringIO(self.contents_header) + return rval + + def startswith(self, prefix): + return self.string_io().read(len(prefix)) == prefix + + def line_iterator(self): + s = self.string_io() + for line in s: + if line.endswith("\n") or line.endswith("\r"): + yield line + elif s.pos == s.len and not self.truncated: + # At the end, return the last line if it wasn't truncated when reading it in. + yield line + + # Convenience wrappers around contents_header, shielding contents_header means we can + # potentially do a better job lazy loading this data later on. + def search(self, pattern): + return pattern.search(self.contents_header) + + def search_str(self, query_str): + return query_str in self.contents_header + + +def build_sniff_from_prefix(klass): + def auto_sniff(self, filename): + file_prefix = FilePrefix(filename) + datatype_compressed = getattr(self, "compressed", False) + if file_prefix.compressed_format and not datatype_compressed: + return False + if datatype_compressed and not file_prefix.compressed_format: + return False + if hasattr(self, "compressed_format"): + if self.compressed_format != file_prefix.compressed_format: + return False + return self.sniff_prefix(file_prefix) + + klass.sniff = auto_sniff + return klass + + +def disable_parent_class_sniffing(klass): + klass.sniff = lambda self, filename: False + klass.sniff_prefix = lambda self, file_prefix: False + return klass + + def handle_compressed_file( filename, datatypes_registry, @@ -464,11 +659,10 @@ def handle_compressed_file( if ext in AUTO_DETECT_EXTENSIONS: # attempt to sniff for a keep-compressed datatype (observing the sniff order) sniff_datatypes = filter(lambda d: getattr(d, 'compressed', False), datatypes_registry.sniff_order) - for datatype in sniff_datatypes: - if datatype.sniff(filename): - ext = datatype.file_ext - keep_compressed = True - break + sniffed_ext = run_sniffers_raw(filename, sniff_datatypes) + if sniffed_ext: + ext = sniffed_ext + keep_compressed = True else: datatype = datatypes_registry.get_datatype_by_extension(ext) keep_compressed = getattr(datatype, 'compressed', False) diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index 364bf048431..62f847e3efa 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -21,11 +21,11 @@ from galaxy import util from galaxy.datatypes import binary, data, metadata from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes.sniff import ( + build_sniff_from_prefix, get_headers, iter_headers ) from galaxy.util import compression_utils -from galaxy.util.checkers import is_gzip from . import dataproviders if sys.version_info > (3,): @@ -416,6 +416,7 @@ class Taxonomy(Tabular): @dataproviders.decorators.has_dataproviders +@build_sniff_from_prefix class Sam(Tabular): edam_format = "format_2573" edam_data = "data_0863" @@ -434,7 +435,7 @@ class Sam(Tabular): """Returns formated html of peek""" return self.make_html_table(dataset, column_names=self.column_names) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in SAM format @@ -463,31 +464,31 @@ class Sam(Tabular): >>> Sam().sniff( fname ) True """ - with open(filename) as fh: - count = 0 - while True: - line = fh.readline() - line = line.strip() - if not line: - break # EOF - if line: - if line[0] != '@': - line_pieces = line.split('\t') - if len(line_pieces) < 11: - return False - try: - int(line_pieces[1]) - int(line_pieces[3]) - int(line_pieces[4]) - int(line_pieces[7]) - int(line_pieces[8]) - except ValueError: - return False - count += 1 - if count == 5: - return True - if count < 5 and count > 0: - return True + fh = file_prefix.string_io() + count = 0 + while True: + line = fh.readline() + line = line.strip() + if not line: + break # EOF + if line: + if line[0] != '@': + line_pieces = line.split('\t') + if len(line_pieces) < 11: + return False + try: + int(line_pieces[1]) + int(line_pieces[3]) + int(line_pieces[4]) + int(line_pieces[7]) + int(line_pieces[8]) + except ValueError: + return False + count += 1 + if count == 5: + return True + if count < 5 and count > 0: + return True return False def set_meta(self, dataset, overwrite=True, skip=None, max_data_lines=5, **kwd): @@ -592,6 +593,7 @@ class Sam(Tabular): @dataproviders.decorators.has_dataproviders +@build_sniff_from_prefix class Pileup(Tabular): """Tab delimited data in pileup (6- or 10-column) format""" edam_format = "format_3015" @@ -616,7 +618,7 @@ class Pileup(Tabular): """Return options for removing errors along with a description""" return [("lines", "Remove erroneous lines")] - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Checks for 'pileup-ness' @@ -634,24 +636,32 @@ class Pileup(Tabular): >>> fname = get_test_fname( '10col.pileup' ) >>> Pileup().sniff( fname ) True + >>> fname = get_test_fname( '1.xls' ) + >>> Pileup().sniff( fname ) + False + >>> fname = get_test_fname( '2.txt' ) + >>> Pileup().sniff( fname ) # 2.txt + False + >>> fname = get_test_fname( '2.tabular' ) + >>> Pileup().sniff( fname ) + False """ - headers = iter_headers(filename, '\t') + found_non_comment_lines = False try: + headers = iter_headers(file_prefix, '\t') for hdr in headers: if hdr and not hdr[0].startswith('#'): if len(hdr) < 5: return False - try: - # chrom start in column 1 (with 0-based columns) - # and reference base is in column 2 - chrom = int(hdr[1]) - assert chrom >= 0 - assert hdr[2] in ['A', 'C', 'G', 'T', 'N', 'a', 'c', 'g', 't', 'n'] - except Exception: - return False - return True + # chrom start in column 1 (with 0-based columns) + # and reference base is in column 2 + chrom = int(hdr[1]) + assert chrom >= 0 + assert hdr[2] in ['A', 'C', 'G', 'T', 'N', 'a', 'c', 'g', 't', 'n'] + found_non_comment_lines = True except Exception: return False + return found_non_comment_lines # Dataproviders @dataproviders.decorators.dataprovider_factory('genomic-region', @@ -667,6 +677,7 @@ class Pileup(Tabular): @dataproviders.decorators.has_dataproviders +@build_sniff_from_prefix class BaseVcf(Tabular): """ Variant Call Format for describing SNPs and other simple genome variations. """ edam_format = "format_3016" @@ -680,15 +691,12 @@ class BaseVcf(Tabular): MetadataElement(name="viz_filter_cols", desc="Score column for visualization", default=[5], param=metadata.ColumnParameter, optional=True, multiple=True, visible=False) MetadataElement(name="sample_names", default=[], desc="Sample names", readonly=True, visible=False, optional=True, no_value=[]) - def sniff(self, filename): + def sniff_prefix(self, file_prefix): # Because this sniffer is run on compressed files that might be BGZF (due to the VcfGz subclass), we should # handle unicode decode errors. This should ultimately be done in get_headers(), but guess_ext() currently # relies on get_headers() raising this exception. - try: - headers = get_headers(filename, '\n', count=1) - return headers[0][0].startswith("##fileformat=VCF") - except UnicodeDecodeError: - return False + headers = get_headers(file_prefix, '\n', count=1) + return headers[0][0].startswith("##fileformat=VCF") def display_peek(self, dataset): """Returns formated html of peek""" @@ -737,23 +745,14 @@ class BaseVcf(Tabular): class Vcf(BaseVcf): file_ext = 'vcf' - def sniff(self, filename): - if is_gzip(filename): - return False - return super(Vcf, self).sniff(filename) - class VcfGz(BaseVcf, binary.Binary): file_ext = 'vcf_bgzip' compressed = True + compressed_format = "gzip" MetadataElement(name="tabix_index", desc="Vcf Index File", param=metadata.FileParameter, file_ext="tbi", readonly=True, no_value=None, visible=False, optional=True) - def sniff(self, filename): - if not is_gzip(filename): - return False - return super(VcfGz, self).sniff(filename) - def set_meta(self, dataset, **kwd): super(BaseVcf, self).set_meta(dataset, **kwd) """ Creates the index for the VCF file. """ @@ -769,8 +768,11 @@ class VcfGz(BaseVcf, binary.Binary): dataset.metadata.tabix_index = index_file +@build_sniff_from_prefix class Eland(Tabular): """Support for the export.txt.gz file used by Illumina's ELANDv2e aligner""" + compressed = True + compressed_format = "gzip" file_ext = '_export.txt.gz' MetadataElement(name="columns", default=0, desc="Number of columns", readonly=True, visible=False) MetadataElement(name="column_types", default=[], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False, no_value=[]) @@ -811,7 +813,7 @@ class Eland(Tabular): out = "Can't create peek %s" % exc return out - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in ELAND export format @@ -824,32 +826,31 @@ class Eland(Tabular): - LANE, TILEm X, Y, INDEX, READ_NO, SEQ, QUAL, POSITION, *STRAND, FILT must be correct - We will only check that up to the first 5 alignments are correctly formatted. """ - with compression_utils.get_fileobj(filename, compressed_formats=['gzip']) as fh: - count = 0 - while True: - line = fh.readline() - line = line.strip() - if not line: - break # EOF - if line: - line_pieces = line.split('\t') - if len(line_pieces) != 22: - return False - if long(line_pieces[1]) < 0: - raise Exception('Out of range') - if long(line_pieces[2]) < 0: - raise Exception('Out of range') - if long(line_pieces[3]) < 0: - raise Exception('Out of range') - int(line_pieces[4]) - int(line_pieces[5]) - # can get a lot more specific - count += 1 - if count == 5: - break - if count > 0: - return True - return False + fh = file_prefix.string_io() + count = 0 + while True: + line = fh.readline() + line = line.strip() + if not line: + break # EOF + if line: + line_pieces = line.split('\t') + if len(line_pieces) != 22: + return False + if long(line_pieces[1]) < 0: + raise Exception('Out of range') + if long(line_pieces[2]) < 0: + raise Exception('Out of range') + if long(line_pieces[3]) < 0: + raise Exception('Out of range') + int(line_pieces[4]) + int(line_pieces[5]) + # can get a lot more specific + count += 1 + if count == 5: + break + if count > 0: + return True def set_meta(self, dataset, overwrite=True, skip=None, max_data_lines=5, **kwd): if dataset.has_data(): @@ -884,10 +885,11 @@ class Eland(Tabular): dataset.metadata.reads = list(reads.keys()) +@build_sniff_from_prefix class ElandMulti(Tabular): file_ext = 'elandmulti' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): return False @@ -1056,6 +1058,7 @@ class TSV(BaseCSV): strict_width = True # Leave files with different width to tabular +@build_sniff_from_prefix class ConnectivityTable(Tabular): edam_format = "format_3309" file_ext = "ct" @@ -1077,7 +1080,7 @@ class ConnectivityTable(Tabular): dataset.metadata.data_lines = data_lines - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ The ConnectivityTable (CT) is a file format used for describing RNA 2D structures by tools including MFOLD, UNAFOLD and @@ -1112,31 +1115,28 @@ class ConnectivityTable(Tabular): i = 0 j = 1 - try: - with open(filename) as handle: - for line in handle: - line = line.strip() + handle = file_prefix.string_io() + for line in handle: + line = line.strip() - if len(line) > 0: - if i == 0: - if not self.header_regexp.match(line): - return False - else: - length = int(re.split('\W+', line, 1)[0]) + if len(line) > 0: + if i == 0: + if not self.header_regexp.match(line): + return False + else: + length = int(re.split('\W+', line, 1)[0]) + else: + if not self.structure_regexp.match(line.upper()): + return False + else: + if j != int(re.split('\W+', line, 1)[0]): + return False + elif j == length: # Last line of first sequence has been recheached + return True else: - if not self.structure_regexp.match(line.upper()): - return False - else: - if j != int(re.split('\W+', line, 1)[0]): - return False - elif j == length: # Last line of first sequence has been recheached - return True - else: - j += 1 - i += 1 - return False - except Exception: - return False + j += 1 + i += 1 + return False def get_chunk(self, trans, dataset, chunk): ck_index = int(chunk) diff --git a/lib/galaxy/datatypes/test/1.afg b/lib/galaxy/datatypes/test/1.afg new file mode 100644 index 00000000000..bb8059d9fca --- /dev/null +++ b/lib/galaxy/datatypes/test/1.afg @@ -0,0 +1,67 @@ +{UNV +iid:1 +com: +Generated by dsommer with tarchive2amos on Wed Aug 30 13:10:59 2006 +. +} +{RED +iid:1 +eid:zfishG-a2661d04.q1c +seq: +TAAAATAAATGTTATGTTATCATGTTGACAGATCAATGATAAAATAAAGCCTGGTGATTA +AAAACCTGCAATACCTTGACAAGAACTTTCATGTAAACTAAAGTACTAACTAAAAAAGTG +TCCTGAGAAATCTCGACAGTTTTTTGAGTTTGATAGCCCTGGGCTCAATCAGAAAACTAG +CCAGTCAGAAACACTCTTCATTTCACTCGTTCGGGTCTGCTGACACTGACTTTGCTGACA +AGTCTTTGGAGGTTGAGTTTTGGAGAGAGATGGCGTTAGCAAAAATGGCTGAAGTTAGCA +AAATGGCTGCAGTCGCCTACAGCATTCGAATTCATACCTTGTTTCTGAGACCATGTGTCA +CTCACCTGGGCGCTGACTTTGCTCTTGCTTGCCACGGCTTGTTTGAGAGCCTGTCTGATA +ATGACATGCAGTCGGAGCACCACAGCCTCGATGTCCTGTTCCCCCTTCATGGCAGCGGCG +AGTGACCCCAGTTCATCCACATATGCTCTCCAATGAGGCTGTACCTCCGTGGTTCCCCCC +AAACATGCTCCCATGGTGCTGACACACAGTGGACGGCACGGCCGGGCCTTCAGCAGGCTT +TGACAGTGCGGGCAGTACCACAGTTTCAGAAGTGAGCGCCCACAATCTCTACCTGCCCGC +AAATGCTGAGT +. +qlt: +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXX +. +clr:0,671 +} +{RED +iid:2 +eid:zfishI-a72c06.p1c +seq: +CACCCAAAAGCAATGGCAAAAGATCTGGCGGACGCATTGCGGGCTGGGCGGGTGCTCAGC +TTGGCACTTGCCGAAGGGTCAGAGGTGATGAACATTACGGAAACAGCAGGGTTATCAAAA +GAGTGCACACGGGCTCTGGTACGCATGCATTACTGCTCTCACTGCCGTGGACTCACCCTG +ATCCATGCGTGCAGCAACTACTGTCTTAATGTCATGCGCGGGTGCCTGGCGAGCTACTCC +GAGCTCCACCAGCCCTGGAGACAGTATGTCACCATATTGCAGGACCTCACGCAAATGGTT +GCCGGAGCTCACAATTTAGAGCTGGCCTTACTGGGGATCAGAGGTCAGGTCGAGGAGGCC +ATACTCTACGCTCAGCTTCACGGGCCCAGGCTAACTGCCACAGTGAGTACTAGCATTTTT +ACGCTTTTACAGCTAGCATTAGCTTGTATTGTAGCATGAAAAAAGGTCTACAGAGTTATG +AAGCACAAAGACCTCTTCTGCTATAAGCGGGTTTCTGAA +. +qlt: +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX +. +clr:0,519 +} +5B \ No newline at end of file diff --git a/lib/galaxy/datatypes/test/1.arff b/lib/galaxy/datatypes/test/1.arff new file mode 100644 index 00000000000..34fe9b00d2b --- /dev/null +++ b/lib/galaxy/datatypes/test/1.arff @@ -0,0 +1,85 @@ +% 1. Title: Database for fitting contact lenses +% +% 2. Sources: +% (a) Cendrowska, J. "PRISM: An algorithm for inducing modular rules", +% International Journal of Man-Machine Studies, 1987, 27, 349-370 +% (b) Donor: Benoit Julien (Julien@ce.cmu.edu) +% (c) Date: 1 August 1990 +% +% 3. Past Usage: +% 1. See above. +% 2. Witten, I. H. & MacDonald, B. A. (1988). Using concept +% learning for knowledge acquisition. International Journal of +% Man-Machine Studies, 27, (pp. 349-370). +% +% Notes: This database is complete (all possible combinations of +% attribute-value pairs are represented). +% +% Each instance is complete and correct. +% +% 9 rules cover the training set. +% +% 4. Relevant Information Paragraph: +% The examples are complete and noise free. +% The examples highly simplified the problem. The attributes do not +% fully describe all the factors affecting the decision as to which type, +% if any, to fit. +% +% 5. Number of Instances: 24 +% +% 6. Number of Attributes: 4 (all nominal) +% +% 7. Attribute Information: +% -- 3 Classes +% 1 : the patient should be fitted with hard contact lenses, +% 2 : the patient should be fitted with soft contact lenses, +% 1 : the patient should not be fitted with contact lenses. +% +% 1. age of the patient: (1) young, (2) pre-presbyopic, (3) presbyopic +% 2. spectacle prescription: (1) myope, (2) hypermetrope +% 3. astigmatic: (1) no, (2) yes +% 4. tear production rate: (1) reduced, (2) normal +% +% 8. Number of Missing Attribute Values: 0 +% +% 9. Class Distribution: +% 1. hard contact lenses: 4 +% 2. soft contact lenses: 5 +% 3. no contact lenses: 15 + +@relation contact-lenses + +@attribute age {young, pre-presbyopic, presbyopic} +@attribute spectacle-prescrip {myope, hypermetrope} +@attribute astigmatism {no, yes} +@attribute tear-prod-rate {reduced, normal} +@attribute contact-lenses {soft, hard, none} + +@data +% +% 24 instances +% +young,myope,no,reduced,none +young,myope,no,normal,soft +young,myope,yes,reduced,none +young,myope,yes,normal,hard +young,hypermetrope,no,reduced,none +young,hypermetrope,no,normal,soft +young,hypermetrope,yes,reduced,none +young,hypermetrope,yes,normal,hard +pre-presbyopic,myope,no,reduced,none +pre-presbyopic,myope,no,normal,soft +pre-presbyopic,myope,yes,reduced,none +pre-presbyopic,myope,yes,normal,hard +pre-presbyopic,hypermetrope,no,reduced,none +pre-presbyopic,hypermetrope,no,normal,soft +pre-presbyopic,hypermetrope,yes,reduced,none +pre-presbyopic,hypermetrope,yes,normal,none +presbyopic,myope,no,reduced,none +presbyopic,myope,no,normal,none +presbyopic,myope,yes,reduced,none +presbyopic,myope,yes,normal,hard +presbyopic,hypermetrope,no,reduced,none +presbyopic,hypermetrope,no,normal,soft +presbyopic,hypermetrope,yes,reduced,none +presbyopic,hypermetrope,yes,normal,none diff --git a/lib/galaxy/datatypes/test/1.hdt b/lib/galaxy/datatypes/test/1.hdt new file mode 100644 index 00000000000..b67bf29ee48 Binary files /dev/null and b/lib/galaxy/datatypes/test/1.hdt differ diff --git a/lib/galaxy/datatypes/test/1.obo b/lib/galaxy/datatypes/test/1.obo new file mode 100644 index 00000000000..57725e58d02 --- /dev/null +++ b/lib/galaxy/datatypes/test/1.obo @@ -0,0 +1,31 @@ +format-version: GO_1.0 +!any comment here +typeref: relationship.types +subsetdef: goslim "Generic GO Slim" +version: $Revision: 1.18 $ +date: April 18th, 2003 +saved-by: jrichter +remark: Example file + +[Term] +id: GO:0003674 +name: molecular_function +def: "The action characteristic of a gene product." [GO:curators] +subset: goslim + +[Term] +id: GO:0016209 +name: antioxidant activity +is_a: GO:0003674 +def: "Inhibition of the reactions brought about by dioxygen or peroxides. Usually the antioxidant is effective because it can itself be more easily oxidized than the substance protected. The term is often applied to components that can trap free radicals, thereby breaking the chain reaction that normally leads to extensive biological damage." [ISBN:0198506732] + +[Term] +id: GO:0045174 +name: glutathione dehydrogenase (ascorbate) activity +xref_analog: EC:1.8.5.1 "" +def: "Catalysis of the reaction: 2 glutathione + dehydroascorbate = glutathione disulfide + ascorbate." [EC:1.8.5.1] +synonym: dehydroascorbate reductase [] +is_a: GO:0009055 +is_a: GO:0015038 +is_a: GO:0016672 + diff --git a/lib/galaxy/datatypes/test/1.owl b/lib/galaxy/datatypes/test/1.owl new file mode 100644 index 00000000000..19d9ce88b6e --- /dev/null +++ b/lib/galaxy/datatypes/test/1.owl @@ -0,0 +1,221 @@ + + + + Created with TopBraid Spreadsheet converter + + + + SKU + + + + + Product Line + + + + + Manufacture Location + + + + + Available + + + + + Division + + + + + ModelNo + + + + + ID + + + + + Product 2 + 2 + ZX-3P + Manufacturing support + Paper machine + Sacramento + KD5243 + 4 + + + Product 6 + 6 + B-1431 + Control Engineering + Active sensor + Seoul + KK3945 + 0 + + + Product 1 + 1 + ZX-3 + Manufacturing support + Papermachine + Sacramento + FB3524 + 23 + + + Product 7 + 7 + DBB-12 + Accessories + Monitor + Hong Kong + ND5520 + 100 + + + Product 9 + 9 + SPX-1234 + Safety + Safety + valve + Cleveland + OP5333 + + + Product 4 + 4 + B-1430 + Control Engineering + Feedback line + Elizabeth + KS4520 + 23 + + + Product 8 + 8 + SP-1234 + Safety + Safety + valve + Cleveland + HI4554 + + + Product 5 + 5 + B-1430X + Control Engineering + Feedback line + Elizabeth + CL5934 + 14 + + + Product 3 + 3 + ZX-3S + Manufacturing support + Paper machine + Sacramento + IL4028 + 34 + + + + diff --git a/lib/galaxy/datatypes/test/1.phyloxml b/lib/galaxy/datatypes/test/1.phyloxml new file mode 100644 index 00000000000..b448ab53c27 --- /dev/null +++ b/lib/galaxy/datatypes/test/1.phyloxml @@ -0,0 +1,48 @@ + + + Alcohol dehydrogenases + contains examples of commonly used elements + + + 1 + + + + 6645 + Octopus vulgaris + + + P81431 + Alcohol dehydrogenase class-3 + + + + 100 + + 1 + + + + 1423 + Bacillus subtilis + + + P71017 + Alcohol dehydrogenase + + + + + 562 + Escherichia coli + + + Q46856 + Alcohol dehydrogenase + + + + + + + diff --git a/lib/galaxy/datatypes/test/1.stockholm b/lib/galaxy/datatypes/test/1.stockholm new file mode 100644 index 00000000000..4e768964d51 --- /dev/null +++ b/lib/galaxy/datatypes/test/1.stockholm @@ -0,0 +1,18 @@ +# STOCKHOLM 1.0 +#=GF ID UPSK +#=GF SE Predicted; Infernal +#=GF SS Published; PMID 9223489 +#=GF RN [1] +#=GF RM 9223489 +#=GF RT The role of the pseudoknot at the 3' end of turnip yellow mosaic +#=GF RT virus RNA in minus-strand synthesis by the viral RNA-dependent RNA +#=GF RT polymerase. +#=GF RA Deiman BA, Kortlever RM, Pleij CW; +#=GF RL J Virol 1997;71:5990-5996. + +AF035635.1/619-641 UGAGUUCUCGAUCUCUAAAAUCG +M24804.1/82-104 UGAGUUCUCUAUCUCUAAAAUCG +J04373.1/6212-6234 UAAGUUCUCGAUCUUUAAAAUCG +M24803.1/1-23 UAAGUUCUCGAUCUCUAAAAUCG +#=GC SS_cons .AAA....<<<>>> +// \ No newline at end of file diff --git a/lib/galaxy/datatypes/test/1.ttl b/lib/galaxy/datatypes/test/1.ttl new file mode 100644 index 00000000000..c2a48846140 --- /dev/null +++ b/lib/galaxy/datatypes/test/1.ttl @@ -0,0 +1,10 @@ +@prefix rdf: . +@prefix dc: . +@prefix ex: . + + + dc:title "RDF/XML Syntax Specification (Revised)" ; + ex:editor [ + ex:fullname "Dave Beckett"; + ex:homePage + ] . diff --git a/lib/galaxy/datatypes/test/1.xmfa b/lib/galaxy/datatypes/test/1.xmfa new file mode 100644 index 00000000000..14e7c3a79fe --- /dev/null +++ b/lib/galaxy/datatypes/test/1.xmfa @@ -0,0 +1,30 @@ +#FormatVersion Mauve1 +#Sequence1File a.fa +#Sequence1Entry 1 +#Sequence1Format FastA +#Sequence2File b.fa +#Sequence2Entry 2 +#Sequence2Format FastA +#Sequence3File c.fa +#Sequence3Entry 3 +#Sequence3Format FastA +#BackboneFile three.xmfa.bbcols +> 1:0-0 + a.fa +-------------------------------------------------------------------------------- +-------------------------------------------------------------------------------- +-------------------------------------------------------------------------------- +> 2:5417-5968 + b.fa +TTTAAACATCCCTCGGCCCGTCGCCCTTTTATAATAGCAGTACGTGAGAGGAGCGCCCTAAGCTTTGGGAAATTCAAGC- +-------------------------------------------------------------------------------- +CTGGAACGTACTTGCTGGTTTCGCTACTATTTCAAACAAGTTAGAGGCCGTTACCTCGGGCGAACGTATAAACCATTCTG +> 3:9476-10076 - c.fa +TTTAAACACCTTTTTGGATG--GCCCAGTTCGTTCAGTTGTG-GGGAGGAGATCGCCCCAAACGTATGGTGAGTCGGGCG +TTTCCTATAGCTATAGGACCAATCCACTTACCATACGCCCGGCGTCGCCCAGTCCGGTTCGGTACCCTCCATGACCCACG +---------------------------------------------------------AAATGAGGGCCCAGGGTATGCTT += +> 2:5969-6015 + b.fa +----------------------- +GGGCGAACGTATAAACCATTCTG +> 3:9429-9476 - c.fa +TTCGGTACCCTCCATGACCCACG +AAATGAGGGCCCAGGGTATGCTT diff --git a/lib/galaxy/datatypes/test/2.tabular b/lib/galaxy/datatypes/test/2.tabular new file mode 100644 index 00000000000..d56c0bee10d --- /dev/null +++ b/lib/galaxy/datatypes/test/2.tabular @@ -0,0 +1,3 @@ +a 2 +c 1 +d 0 diff --git a/lib/galaxy/datatypes/test/2.txt b/lib/galaxy/datatypes/test/2.txt new file mode 100644 index 00000000000..c9e51fd9bac --- /dev/null +++ b/lib/galaxy/datatypes/test/2.txt @@ -0,0 +1 @@ +a 2 \ No newline at end of file diff --git a/lib/galaxy/datatypes/test/3.txt b/lib/galaxy/datatypes/test/3.txt new file mode 100644 index 00000000000..afb3a194860 --- /dev/null +++ b/lib/galaxy/datatypes/test/3.txt @@ -0,0 +1,3 @@ +a 1 2 x +b 3 4 y +c 5 6 z \ No newline at end of file diff --git a/lib/galaxy/datatypes/test/Acanium.hmm b/lib/galaxy/datatypes/test/Acanium.hmm new file mode 100644 index 00000000000..102dbd7ecae --- /dev/null +++ b/lib/galaxy/datatypes/test/Acanium.hmm @@ -0,0 +1,21 @@ +zoeHMM Acanium.hmm 6 8 6 7 + + + +Einit 0 0 3 -1 explicit +Esngl 0 0 150 -1 explicit +Eterm 0 0 3 -1 explicit +Exon 0 0 6 -1 explicit +Inter 0.9 0.9 0 0 geometric +Intron 0.1 0.1 0 0 geometric + + + +Einit Intron 1 +Esngl Inter 1 +Eterm Inter 1 +Exon Intron 1 +Inter Einit 0.852754 +Inter Esngl 0.147246 +Intron Eterm 0.129065 +Intron Exon 0.870935 diff --git a/lib/galaxy/datatypes/text.py b/lib/galaxy/datatypes/text.py index 89289ced29c..dd5abf6e88f 100644 --- a/lib/galaxy/datatypes/text.py +++ b/lib/galaxy/datatypes/text.py @@ -14,12 +14,13 @@ from six.moves import shlex_quote from galaxy.datatypes.data import get_file_peek, Text from galaxy.datatypes.metadata import MetadataElement, MetadataParameter -from galaxy.datatypes.sniff import iter_headers +from galaxy.datatypes.sniff import build_sniff_from_prefix, iter_headers from galaxy.util import nice_size, string_as_bool log = logging.getLogger(__name__) +@build_sniff_from_prefix class Html(Text): """Class describing an html file""" edam_format = "format_2331" @@ -37,7 +38,7 @@ class Html(Text): """Returns the mime type of the datatype""" return 'text/html' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Determines whether the file is in html format @@ -49,13 +50,14 @@ class Html(Text): >>> Html().sniff( fname ) True """ - headers = iter_headers(filename, None) + headers = iter_headers(file_prefix, None) for i, hdr in enumerate(headers): if hdr and hdr[0].lower().find('') >= 0: return True return False +@build_sniff_from_prefix class Json(Text): edam_format = "format_3464" file_ext = "json" @@ -72,30 +74,27 @@ class Json(Text): """Returns the mime type of the datatype""" return 'application/json' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to load the string with the json module. If successful it's a json file. """ - return self._looks_like_json(filename) + return self._looks_like_json(file_prefix) - def _looks_like_json(self, filename): + def _looks_like_json(self, file_prefix): # Pattern used by SequenceSplitLocations - if os.path.getsize(filename) < 50000: + if file_prefix.file_size < 50000 and not file_prefix.truncated: # If the file is small enough - don't guess just check. try: - json.load(open(filename, "r")) + json.loads(file_prefix.contents_header) return True except Exception: return False else: - with open(filename, "r") as fh: - while True: - # Grab first chunk of file and see if it looks like json. - start = fh.read(100).strip() - if start: - # simple types are valid JSON as well - but would such a file - # be interesting as JSON in Galaxy? - return start.startswith("[") or start.startswith("{") + start = file_prefix.string_io().read(100).strip() + if start: + # simple types are valid JSON as well - but would such a file + # be interesting as JSON in Galaxy? + return start.startswith("[") or start.startswith("{") return False def display_peek(self, dataset): @@ -105,6 +104,7 @@ class Json(Text): return "JSON file (%s)" % (nice_size(dataset.get_size())) +@build_sniff_from_prefix class Ipynb(Json): file_ext = "ipynb" @@ -116,13 +116,14 @@ class Ipynb(Json): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to load the string with the json module. If successful it's a json file. """ - if self._looks_like_json(filename): + if self._looks_like_json(file_prefix): try: - ipynb = json.load(open(filename)) + with open(file_prefix.filename) as f: + ipynb = json.load(f) if ipynb.get('nbformat', False) is not False and ipynb.get('metadata', False): return True else: @@ -161,6 +162,7 @@ class Ipynb(Json): pass +@build_sniff_from_prefix class Biom1(Json): """ BIOM version 1.0 file format description @@ -186,13 +188,13 @@ class Biom1(Json): if not dataset.dataset.purged: dataset.blurb = "Biological Observation Matrix v1" - def sniff(self, filename): + def sniff_prefix(self, file_prefix): is_biom = False - if self._looks_like_json(filename): - is_biom = self._looks_like_biom(filename) + if self._looks_like_json(file_prefix): + is_biom = self._looks_like_biom(file_prefix) return is_biom - def _looks_like_biom(self, filepath, load_size=50000): + def _looks_like_biom(self, file_prefix, load_size=50000): """ @param filepath: [str] The path to the evaluated file. @param load_size: [int] The size of the file block load in RAM (in @@ -201,7 +203,7 @@ class Biom1(Json): is_biom = False segment_size = int(load_size / 2) try: - with open(filepath, "r") as fh: + with open(file_prefix.filename, "r") as fh: prev_str = "" segment_str = fh.read(segment_size) if segment_str.strip().startswith('{'): @@ -255,6 +257,7 @@ class Biom1(Json): pass +@build_sniff_from_prefix class Obo(Text): """ OBO file format description @@ -272,25 +275,26 @@ class Obo(Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess the Obo filetype. It usually starts with a "format-version:" string and has several stanzas which starts with "id:". """ stanza = re.compile(r'^\[.*\]$') - with open(filename) as handle: - first_line = handle.readline() - if not first_line.startswith('format-version:'): - return False + handle = file_prefix.string_io() + first_line = handle.readline() + if not first_line.startswith('format-version:'): + return False - for line in handle: - if stanza.match(line.strip()): - # a stanza needs to begin with an ID tag - if handle.next().startswith('id:'): - return True + for line in handle: + if stanza.match(line.strip()): + # a stanza needs to begin with an ID tag + if handle.next().startswith('id:'): + return True return False +@build_sniff_from_prefix class Arff(Text): """ An ARFF (Attribute-Relation File Format) file is an ASCII text file that describes a list of instances sharing a set of attributes. @@ -312,31 +316,31 @@ class Arff(Text): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Try to guess the Arff filetype. It usually starts with a "format-version:" string and has several stanzas which starts with "id:". """ - with open(filename) as handle: - relation_found = False - attribute_found = False - for line_count, line in enumerate(handle): - if line_count > 1000: - # only investigate the first 1000 lines - return False - line = line.strip() - if not line: - continue + handle = file_prefix.string_io() + relation_found = False + attribute_found = False + for line_count, line in enumerate(handle): + if line_count > 1000: + # only investigate the first 1000 lines + return False + line = line.strip() + if not line: + continue - start_string = line[:20].upper() - if start_string.startswith("@RELATION"): - relation_found = True - elif start_string.startswith("@ATTRIBUTE"): - attribute_found = True - elif start_string.startswith("@DATA"): - # @DATA should be the last data block - if relation_found and attribute_found: - return True + start_string = line[:20].upper() + if start_string.startswith("@RELATION"): + relation_found = True + elif start_string.startswith("@ATTRIBUTE"): + attribute_found = True + elif start_string.startswith("@DATA"): + # @DATA should be the last data block + if relation_found and attribute_found: + return True return False def set_meta(self, dataset, **kwd): @@ -543,11 +547,12 @@ class SnpSiftDbNSFP(Text): dataset.blurb = 'file purged from disc' +@build_sniff_from_prefix class IQTree(Text): """IQ-TREE format""" file_ext = 'iqtree' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Detect the IQTree file @@ -567,7 +572,4 @@ class IQTree(Text): >>> IQTree().sniff(fname) False """ - with open(filename, 'r') as fio: - return fio.read(7) == "IQ-TREE" - - return False + return file_prefix.startswith("IQ-TREE") diff --git a/lib/galaxy/datatypes/triples.py b/lib/galaxy/datatypes/triples.py index 28a1b7c810f..f5c724a3712 100644 --- a/lib/galaxy/datatypes/triples.py +++ b/lib/galaxy/datatypes/triples.py @@ -4,6 +4,9 @@ Triple format classes import logging import re +from galaxy.datatypes.sniff import ( + build_sniff_from_prefix, +) from . import ( binary, data, @@ -13,6 +16,9 @@ from . import ( log = logging.getLogger(__name__) +TURTLE_PREFIX_PATTERN = re.compile(r'@prefix\s+[^:]*:\s+<[^>]*>\s\.') +TURTLE_BASE_PATTERN = re.compile(r'@base\s+<[^>]*>\s\.') + class Triples(data.Data): """ @@ -38,6 +44,7 @@ class Triples(data.Data): dataset.blurb = 'file purged from disk' +@build_sniff_from_prefix class NTriples(data.Text, Triples): """ The N-Triples triple data format @@ -45,11 +52,10 @@ class NTriples(data.Text, Triples): edam_format = "format_3256" file_ext = "nt" - def sniff(self, filename): - with open(filename, "r") as f: - # . - if re.compile(r'<[^>]*>\s<[^>]*>\s<[^>]*>\s\.').search(f.readline(1024)): - return True + def sniff_prefix(self, file_prefix): + # . + if re.compile(r'<[^>]*>\s<[^>]*>\s<[^>]*>\s\.').search(file_prefix.contents_header): + return True return False def set_peek(self, dataset, is_multi_byte=False): @@ -85,6 +91,7 @@ class N3(data.Text, Triples): dataset.blurb = 'file purged from disk' +@build_sniff_from_prefix class Turtle(data.Text, Triples): """ The Turtle triple data format @@ -92,14 +99,13 @@ class Turtle(data.Text, Triples): edam_format = "format_3255" file_ext = "ttl" - def sniff(self, filename): - with open(filename, "r") as f: - # @prefix rdf: . - line = f.readline(1024) - if re.compile(r'@prefix\s+[^:]*:\s+<[^>]*>\s\.').search(line): - return True - if re.compile(r'@base\s+<[^>]*>\s\.').search(line): - return True + def sniff_prefix(self, file_prefix): + # @prefix rdf: . + if file_prefix.search(TURTLE_PREFIX_PATTERN): + return True + + if file_prefix.search(TURTLE_BASE_PATTERN): + return True return False def set_peek(self, dataset, is_multi_byte=False): @@ -113,6 +119,7 @@ class Turtle(data.Text, Triples): # TODO: we might want to look at rdflib or a similar, larger lib/egg +@build_sniff_from_prefix class Rdf(xml.GenericXml, Triples): """ Resource Description Framework format (http://www.w3.org/RDF/). @@ -120,13 +127,11 @@ class Rdf(xml.GenericXml, Triples): edam_format = "format_3261" file_ext = "rdf" - def sniff(self, filename): - with open(filename, "r") as f: - firstlines = "".join(f.readlines(5000)) - # >> GenericXml().sniff( fname ) False """ - # TODO - Use a context manager on Python 2.5+ to close handle - with open(filename) as handle: - line = handle.readline() - - # TODO - Is there a more robust way to do this? - return line.startswith('>> from galaxy.datatypes.sniff import get_test_fname + >>> fname = get_test_fname( '1.phyloxml' ) + >>> Phyloxml().sniff( fname ) + True + >>> fname = get_test_fname( 'interval.interval' ) + >>> Phyloxml().sniff( fname ) + False + >>> fname = get_test_fname( 'megablast_xml_parser_test1.blastxml' ) + >>> Phyloxml().sniff( fname ) + False + """ + return self._has_root_element_in_prefix(file_prefix, "phyloxml") def get_visualizations(self, dataset): """ @@ -143,15 +154,8 @@ class Owl(GenericXml): dataset.peek = 'file does not exist' dataset.blurb = 'file purged from disc' - def sniff(self, filename): + def sniff_prefix(self, file_prefix): """ Checking for keyword - '