From fe2d607675acf5e18994957fb8ff4422079987ea Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Tue, 8 Aug 2017 14:25:32 +0200 Subject: [PATCH 1/3] Replace list with generator when iterating headers This should be more memory-friendly. @blankenberg suggested doing this in #4319. --- lib/galaxy/datatypes/interval.py | 35 +++++++++++++----------- lib/galaxy/datatypes/molecules.py | 11 +++++--- lib/galaxy/datatypes/mothur.py | 44 +++++++++++++++++-------------- lib/galaxy/datatypes/sequence.py | 13 ++++++--- lib/galaxy/datatypes/sniff.py | 32 +++++++++++----------- lib/galaxy/datatypes/tabular.py | 7 +++-- lib/galaxy/datatypes/text.py | 4 +-- 7 files changed, 83 insertions(+), 63 deletions(-) diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 658d767af3a..b3eaa50665a 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -13,7 +13,10 @@ from six.moves.urllib.parse import quote_plus from galaxy import util from galaxy.datatypes import metadata from galaxy.datatypes.metadata import MetadataElement -from galaxy.datatypes.sniff import get_headers +from galaxy.datatypes.sniff import ( + get_headers, + iter_headers +) from galaxy.datatypes.tabular import Tabular from galaxy.datatypes.util.gff_util import parse_gff3_attributes, parse_gff_attributes from galaxy.web import url_for @@ -313,7 +316,7 @@ class Interval( Tabular ): >>> Interval().sniff( fname ) True """ - headers = get_headers( filename, '\t', comment_designator='#' ) + headers = iter_headers( filename, '\t', comment_designator='#' ) try: """ If we got here, we already know the file is_column_based and is not bed, @@ -489,10 +492,10 @@ class Bed( Interval ): >>> Bed().sniff( fname ) True """ - headers = get_headers( filename, '\t', comment_designator='#' ) + if not get_headers( filename, '\t', comment_designator='#', count=1 ): + return False + headers = iter_headers( filename, '\t', comment_designator='#' ) try: - if not headers: - return False for hdr in headers: if hdr[0] == '': continue @@ -832,10 +835,10 @@ class Gff( Tabular, _RemoteCallMixin ): >>> Gff().sniff( fname ) True """ - headers = get_headers( filename, '\t' ) + if len(get_headers( filename, '\t', count=2 )) < 2: + return False + headers = iter_headers( filename, '\t' ) try: - if len(headers) < 2: - return False for hdr in headers: if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0: return False @@ -964,10 +967,10 @@ class Gff3( Gff ): >>> Gff3().sniff( fname ) True """ - headers = get_headers( filename, '\t' ) + if len(get_headers( filename, '\t', count=2 )) < 2: + return False + headers = iter_headers( filename, '\t' ) try: - if len(headers) < 2: - return False for hdr in headers: if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '3' ) >= 0: return True @@ -1039,10 +1042,10 @@ class Gtf( Gff ): >>> Gtf().sniff( fname ) True """ - headers = get_headers( filename, '\t' ) + if len(get_headers( filename, '\t', count=2 )) < 2: + return False + headers = iter_headers( filename, '\t' ) try: - if len(headers) < 2: - return False for hdr in headers: if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0: return False @@ -1235,7 +1238,7 @@ class Wiggle( Tabular, _RemoteCallMixin ): >>> Wiggle().sniff( fname ) True """ - headers = get_headers( filename, None ) + headers = iter_headers( filename, None ) try: for hdr in headers: if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'): @@ -1371,7 +1374,7 @@ class CustomTrack ( Tabular ): >>> CustomTrack().sniff( fname ) True """ - headers = get_headers( filename, None ) + headers = iter_headers( filename, None ) first_line = True for hdr in headers: if first_line: diff --git a/lib/galaxy/datatypes/molecules.py b/lib/galaxy/datatypes/molecules.py index 703e143e6c8..bfacad3c169 100644 --- a/lib/galaxy/datatypes/molecules.py +++ b/lib/galaxy/datatypes/molecules.py @@ -10,7 +10,10 @@ from galaxy.datatypes import ( from galaxy.datatypes.binary import Binary from galaxy.datatypes.data import get_file_peek from galaxy.datatypes.metadata import MetadataElement -from galaxy.datatypes.sniff import get_headers +from galaxy.datatypes.sniff import ( + get_headers, + iter_headers +) from galaxy.datatypes.tabular import Tabular from galaxy.datatypes.xml import GenericXml @@ -461,7 +464,7 @@ class PDB(GenericMolFile): >>> PDB().sniff(fname) False """ - headers = get_headers(filename, sep=' ', count=300) + headers = iter_headers(filename, sep=' ', count=300) h = t = c = s = k = e = False for line in headers: section_name = line[0].strip() @@ -514,7 +517,7 @@ class PDBQT(GenericMolFile): >>> PDBQT().sniff(fname) False """ - headers = get_headers(filename, sep=' ', count=300) + headers = iter_headers(filename, sep=' ', count=300) h = t = c = s = k = False for line in headers: section_name = line[0].strip() @@ -607,7 +610,7 @@ class InChI(Tabular): >>> InChI().sniff(fname) False """ - inchi_lines = get_headers(filename, sep=' ', count=10) + inchi_lines = iter_headers(filename, sep=' ', count=10) for inchi in inchi_lines: if not inchi[0].startswith('InChI='): return False diff --git a/lib/galaxy/datatypes/mothur.py b/lib/galaxy/datatypes/mothur.py index c6a6ce44cb1..3c0df2cdb3e 100644 --- a/lib/galaxy/datatypes/mothur.py +++ b/lib/galaxy/datatypes/mothur.py @@ -7,7 +7,10 @@ import sys from galaxy.datatypes.data import Text from galaxy.datatypes.metadata import MetadataElement -from galaxy.datatypes.sniff import get_headers +from galaxy.datatypes.sniff import ( + get_headers, + iter_headers +) from galaxy.datatypes.tabular import Tabular log = logging.getLogger(__name__) @@ -32,10 +35,11 @@ class Otu(Text): data_lines = 0 comment_lines = 0 - headers = get_headers(dataset.file_name, sep='\t', count=-1) + headers = iter_headers(dataset.file_name, sep='\t', count=-1) + first_line = next(headers) # set otulabels - if len(headers[0]) > 2: - otulabel_names = headers[0][2:] + if len(first_line) > 2: + otulabel_names = first_line[2:] # set label names and number of lines for line in headers: if len(line) >= 2 and not line[0].startswith('@'): @@ -64,7 +68,7 @@ class Otu(Text): >>> Otu().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -109,7 +113,7 @@ class Sabund(Otu): >>> Sabund().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -151,7 +155,7 @@ class GroupAbund(Otu): comment_lines = 0 ncols = 0 - headers = get_headers(dataset.file_name, sep='\t', count=-1) + headers = iter_headers(dataset.file_name, sep='\t', count=-1) for line in headers: if line[0] == 'label' and line[1] == 'Group': skip = 1 @@ -187,7 +191,7 @@ class GroupAbund(Otu): >>> GroupAbund().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -234,7 +238,7 @@ class SecondaryStructureMap(Tabular): >>> SecondaryStructureMap().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') line_num = 0 rowidxmap = {} for line in headers: @@ -302,7 +306,7 @@ class DistanceMatrix(Text): def set_meta(self, dataset, overwrite=True, skip=0, **kwd): super(DistanceMatrix, self).set_meta(dataset, overwrite=overwrite, skip=skip, **kwd) - headers = get_headers(dataset.file_name, sep='\t') + headers = iter_headers(dataset.file_name, sep='\t') for line in headers: if not line[0].startswith('@'): try: @@ -344,7 +348,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix): False """ numlines = 300 - headers = get_headers(filename, sep='\t', count=numlines) + headers = iter_headers(filename, sep='\t', count=numlines) line_num = 0 for line in headers: if not line[0].startswith('@'): @@ -405,7 +409,7 @@ class SquareDistanceMatrix(DistanceMatrix): False """ numlines = 300 - headers = get_headers(filename, sep='\t', count=numlines) + headers = iter_headers(filename, sep='\t', count=numlines) line_num = 0 for line in headers: if not line[0].startswith('@'): @@ -461,7 +465,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular): >>> PairwiseDistanceMatrix().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -525,7 +529,7 @@ class Group(Tabular): super(Group, self).set_meta(dataset, overwrite, skip, max_data_lines) group_names = set() - headers = get_headers(dataset.file_name, sep='\t', count=-1) + headers = iter_headers(dataset.file_name, sep='\t', count=-1) for line in headers: if len(line) > 1: group_names.add(line[1]) @@ -558,7 +562,7 @@ class Oligos(Text): >>> Oligos().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 for line in headers: if not line[0].startswith('@') and not line[0].startswith('#'): @@ -602,7 +606,7 @@ class Frequency(Tabular): >>> Frequency().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 for line in headers: if not line[0].startswith('@'): @@ -653,7 +657,7 @@ class Quantile(Tabular): >>> Quantile().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 for line in headers: if not line[0].startswith('@') and not line[0].startswith('#'): @@ -691,7 +695,7 @@ class LaneMask(Text): >>> LaneMask().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = get_headers(filename, sep='\t', count=2) if len(headers) != 1 or len(headers[0]) != 1: return False @@ -774,7 +778,7 @@ class RefTaxonomy(Tabular): >>> RefTaxonomy().sniff( fname ) False """ - headers = get_headers(filename, sep='\t', count=300) + headers = iter_headers(filename, sep='\t', count=300) count = 0 pat_prog = re.compile('^([^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?(;[^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?)*(;)?)$') found_semicolons = False @@ -849,7 +853,7 @@ class Axes(Tabular): >>> Axes().sniff( fname ) False """ - headers = get_headers(filename, sep='\t') + headers = iter_headers(filename, sep='\t') count = 0 col_cnt = None all_integers = True diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index b3c4f65f2ba..ee0a646cc73 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -9,6 +9,7 @@ import re import string import sys from cgi import escape +from itertools import islice import bx.align.maf @@ -16,7 +17,10 @@ from galaxy import util from galaxy.datatypes import metadata from galaxy.datatypes.binary import Binary from galaxy.datatypes.metadata import MetadataElement -from galaxy.datatypes.sniff import get_headers +from galaxy.datatypes.sniff import ( + get_headers, + iter_headers +) from galaxy.util import ( compression_utils, nice_size @@ -611,7 +615,7 @@ class BaseFastq ( Sequence ): compressed = is_gzip(filename) or is_bz2(filename) if compressed and not isinstance(self, Binary): return False - headers = get_headers( filename, None, count=1000 ) + headers = iter_headers( filename, None, count=1000 ) # If this is a FastqSanger-derived class, then check to see if the base qualities match if isinstance(self, FastqSanger) or isinstance(self, FastqSangerGz) or isinstance(self, FastqSangerBz2): @@ -621,7 +625,8 @@ class BaseFastq ( Sequence ): bases_regexp = re.compile( "^[NGTAC]*" ) # check that first block looks like a fastq block try: - if len( headers ) >= 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]: + headers = get_headers( filename, None, count=4 ) + if len( headers ) == 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]: # Check the sequence line, make sure it contains only G/C/A/T/N if not bases_regexp.match( headers[1][0] ): return False @@ -695,7 +700,7 @@ class BaseFastq ( Sequence ): @staticmethod def sangerQualities( lines ): """Presuming lines are lines from a fastq file, return True if the qualities are compatible with sanger encoding""" - for line in lines[3::4]: + for line in islice(lines, 3, None, 4): if not all(_ >= '!' and _ <= 'M' for _ in line[0]): return False return True diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 4e1aa31b223..f2e0ea6e07a 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -200,19 +200,7 @@ def convert_newlines_sep2tabs( fname, in_place=True, patt="\\s+", tmp_dir=None, return ( i + 1, temp_name ) -def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ): - """ - Returns a list with the first 'count' lines split by 'sep', ignoring lines - starting with 'comment_designator' - - >>> fname = get_test_fname('complete.bed') - >>> get_headers(fname,'\\t') - [['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']] - >>> fname = get_test_fname('test.gff') - >>> get_headers(fname, '\\t', count=5, comment_designator='#') - [[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']] - """ - headers = [] +def iter_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ): with compression_utils.get_fileobj(fname) as in_file: idx = 0 for line in in_file: @@ -225,11 +213,25 @@ def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=N comment_designator = comment_designator.encode( 'utf-8' ) if comment_designator is not None and comment_designator != '' and line.startswith( comment_designator ): continue - headers.append( line.split(sep) ) + yield line.split(sep) idx += 1 if idx == count: break - return headers + + +def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ): + """ + Returns a list with the first 'count' lines split by 'sep', ignoring lines + starting with 'comment_designator' + + >>> fname = get_test_fname('complete.bed') + >>> get_headers(fname,'\\t') + [['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']] + >>> fname = get_test_fname('test.gff') + >>> get_headers(fname, '\\t', count=5, comment_designator='#') + [[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']] + """ + return list(iter_headers(fname=fname, sep=sep, count=count, is_multi_byte=is_multi_byte, comment_designator=comment_designator)) def is_column_based( fname, sep='\t', skip=0, is_multi_byte=False ): diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index fbcba2b43a9..5525033a224 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -17,7 +17,10 @@ from json import dumps from galaxy import util from galaxy.datatypes import data, metadata from galaxy.datatypes.metadata import MetadataElement -from galaxy.datatypes.sniff import get_headers +from galaxy.datatypes.sniff import ( + get_headers, + iter_headers +) from galaxy.util import compression_utils from . import dataproviders @@ -638,7 +641,7 @@ class Pileup( Tabular ): >>> Pileup().sniff( fname ) True """ - headers = get_headers( filename, '\t' ) + headers = iter_headers( filename, '\t' ) try: for hdr in headers: if hdr and not hdr[0].startswith( '#' ): diff --git a/lib/galaxy/datatypes/text.py b/lib/galaxy/datatypes/text.py index bffb3ad24bc..22b8ca8db11 100644 --- a/lib/galaxy/datatypes/text.py +++ b/lib/galaxy/datatypes/text.py @@ -12,7 +12,7 @@ import tempfile from galaxy.datatypes.data import get_file_peek, Text from galaxy.datatypes.metadata import MetadataElement, MetadataParameter -from galaxy.datatypes.sniff import get_headers +from galaxy.datatypes.sniff import iter_headers from galaxy.util import nice_size, string_as_bool log = logging.getLogger(__name__) @@ -47,7 +47,7 @@ class Html( Text ): >>> Html().sniff( fname ) True """ - headers = get_headers( filename, None ) + headers = iter_headers( filename, None ) try: for i, hdr in enumerate(headers): if hdr and hdr[0].lower().find( '' ) >= 0: From e6c238e234ae4d16d1fd1118c1b7e585ba541645 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Tue, 8 Aug 2017 17:35:04 +0200 Subject: [PATCH 2/3] Move iter_headers into try clause (thx @nsoranzo) --- lib/galaxy/datatypes/interval.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index b3eaa50665a..5b67e9ea0fe 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -316,12 +316,12 @@ class Interval( Tabular ): >>> Interval().sniff( fname ) True """ - headers = iter_headers( filename, '\t', comment_designator='#' ) try: """ If we got here, we already know the file is_column_based and is not bed, so we'll just look for some valid data. """ + headers = iter_headers( filename, '\t', comment_designator='#' ) for hdr in headers: if hdr: if len(hdr) < 3: @@ -494,8 +494,8 @@ class Bed( Interval ): """ if not get_headers( filename, '\t', comment_designator='#', count=1 ): return False - headers = iter_headers( filename, '\t', comment_designator='#' ) try: + headers = iter_headers( filename, '\t', comment_designator='#' ) for hdr in headers: if hdr[0] == '': continue @@ -837,8 +837,8 @@ class Gff( Tabular, _RemoteCallMixin ): """ if len(get_headers( filename, '\t', count=2 )) < 2: return False - headers = iter_headers( filename, '\t' ) try: + headers = iter_headers( filename, '\t' ) for hdr in headers: if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0: return False @@ -969,8 +969,8 @@ class Gff3( Gff ): """ if len(get_headers( filename, '\t', count=2 )) < 2: return False - headers = iter_headers( filename, '\t' ) try: + headers = iter_headers( filename, '\t' ) for hdr in headers: if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '3' ) >= 0: return True @@ -1044,8 +1044,8 @@ class Gtf( Gff ): """ if len(get_headers( filename, '\t', count=2 )) < 2: return False - headers = iter_headers( filename, '\t' ) try: + headers = iter_headers( filename, '\t' ) for hdr in headers: if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0: return False @@ -1238,8 +1238,8 @@ class Wiggle( Tabular, _RemoteCallMixin ): >>> Wiggle().sniff( fname ) True """ - headers = iter_headers( filename, None ) try: + headers = iter_headers( filename, None ) for hdr in headers: if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'): return True From 43702ccacdbfe925d921dac413c02bb7f65b9c4f Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Tue, 8 Aug 2017 17:53:24 +0200 Subject: [PATCH 3/3] Process the first line when setting metdata for mother.otu files --- lib/galaxy/datatypes/mothur.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/mothur.py b/lib/galaxy/datatypes/mothur.py index 3c0df2cdb3e..75bb29be8c2 100644 --- a/lib/galaxy/datatypes/mothur.py +++ b/lib/galaxy/datatypes/mothur.py @@ -36,7 +36,7 @@ class Otu(Text): comment_lines = 0 headers = iter_headers(dataset.file_name, sep='\t', count=-1) - first_line = next(headers) + first_line = get_headers(dataset.file_name, sep='\t', count=1) # set otulabels if len(first_line) > 2: otulabel_names = first_line[2:]