Merge pull request #4388 from mvdbeek/iter_headers

Replace list with generator when iterating headers
This commit is contained in:
John Chilton
2017-08-08 13:55:02 -04:00
committed by GitHub
7 changed files with 83 additions and 63 deletions
+19 -16
View File
@@ -13,7 +13,10 @@ from six.moves.urllib.parse import quote_plus
from galaxy import util
from galaxy.datatypes import metadata
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import get_headers
from galaxy.datatypes.sniff import (
get_headers,
iter_headers
)
from galaxy.datatypes.tabular import Tabular
from galaxy.datatypes.util.gff_util import parse_gff3_attributes, parse_gff_attributes
from galaxy.web import url_for
@@ -313,12 +316,12 @@ class Interval( Tabular ):
>>> Interval().sniff( fname )
True
"""
headers = get_headers( filename, '\t', comment_designator='#' )
try:
"""
If we got here, we already know the file is_column_based and is not bed,
so we'll just look for some valid data.
"""
headers = iter_headers( filename, '\t', comment_designator='#' )
for hdr in headers:
if hdr:
if len(hdr) < 3:
@@ -489,10 +492,10 @@ class Bed( Interval ):
>>> Bed().sniff( fname )
True
"""
headers = get_headers( filename, '\t', comment_designator='#' )
if not get_headers( filename, '\t', comment_designator='#', count=1 ):
return False
try:
if not headers:
return False
headers = iter_headers( filename, '\t', comment_designator='#' )
for hdr in headers:
if hdr[0] == '':
continue
@@ -832,10 +835,10 @@ class Gff( Tabular, _RemoteCallMixin ):
>>> Gff().sniff( fname )
True
"""
headers = get_headers( filename, '\t' )
if len(get_headers( filename, '\t', count=2 )) < 2:
return False
try:
if len(headers) < 2:
return False
headers = iter_headers( filename, '\t' )
for hdr in headers:
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0:
return False
@@ -964,10 +967,10 @@ class Gff3( Gff ):
>>> Gff3().sniff( fname )
True
"""
headers = get_headers( filename, '\t' )
if len(get_headers( filename, '\t', count=2 )) < 2:
return False
try:
if len(headers) < 2:
return False
headers = iter_headers( filename, '\t' )
for hdr in headers:
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '3' ) >= 0:
return True
@@ -1039,10 +1042,10 @@ class Gtf( Gff ):
>>> Gtf().sniff( fname )
True
"""
headers = get_headers( filename, '\t' )
if len(get_headers( filename, '\t', count=2 )) < 2:
return False
try:
if len(headers) < 2:
return False
headers = iter_headers( filename, '\t' )
for hdr in headers:
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0:
return False
@@ -1235,8 +1238,8 @@ class Wiggle( Tabular, _RemoteCallMixin ):
>>> Wiggle().sniff( fname )
True
"""
headers = get_headers( filename, None )
try:
headers = iter_headers( filename, None )
for hdr in headers:
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
return True
@@ -1371,7 +1374,7 @@ class CustomTrack ( Tabular ):
>>> CustomTrack().sniff( fname )
True
"""
headers = get_headers( filename, None )
headers = iter_headers( filename, None )
first_line = True
for hdr in headers:
if first_line:
+7 -4
View File
@@ -10,7 +10,10 @@ from galaxy.datatypes import (
from galaxy.datatypes.binary import Binary
from galaxy.datatypes.data import get_file_peek
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import get_headers
from galaxy.datatypes.sniff import (
get_headers,
iter_headers
)
from galaxy.datatypes.tabular import Tabular
from galaxy.datatypes.xml import GenericXml
@@ -461,7 +464,7 @@ class PDB(GenericMolFile):
>>> PDB().sniff(fname)
False
"""
headers = get_headers(filename, sep=' ', count=300)
headers = iter_headers(filename, sep=' ', count=300)
h = t = c = s = k = e = False
for line in headers:
section_name = line[0].strip()
@@ -514,7 +517,7 @@ class PDBQT(GenericMolFile):
>>> PDBQT().sniff(fname)
False
"""
headers = get_headers(filename, sep=' ', count=300)
headers = iter_headers(filename, sep=' ', count=300)
h = t = c = s = k = False
for line in headers:
section_name = line[0].strip()
@@ -607,7 +610,7 @@ class InChI(Tabular):
>>> InChI().sniff(fname)
False
"""
inchi_lines = get_headers(filename, sep=' ', count=10)
inchi_lines = iter_headers(filename, sep=' ', count=10)
for inchi in inchi_lines:
if not inchi[0].startswith('InChI='):
return False
+24 -20
View File
@@ -7,7 +7,10 @@ import sys
from galaxy.datatypes.data import Text
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import get_headers
from galaxy.datatypes.sniff import (
get_headers,
iter_headers
)
from galaxy.datatypes.tabular import Tabular
log = logging.getLogger(__name__)
@@ -32,10 +35,11 @@ class Otu(Text):
data_lines = 0
comment_lines = 0
headers = get_headers(dataset.file_name, sep='\t', count=-1)
headers = iter_headers(dataset.file_name, sep='\t', count=-1)
first_line = get_headers(dataset.file_name, sep='\t', count=1)
# set otulabels
if len(headers[0]) > 2:
otulabel_names = headers[0][2:]
if len(first_line) > 2:
otulabel_names = first_line[2:]
# set label names and number of lines
for line in headers:
if len(line) >= 2 and not line[0].startswith('@'):
@@ -64,7 +68,7 @@ class Otu(Text):
>>> Otu().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -109,7 +113,7 @@ class Sabund(Otu):
>>> Sabund().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -151,7 +155,7 @@ class GroupAbund(Otu):
comment_lines = 0
ncols = 0
headers = get_headers(dataset.file_name, sep='\t', count=-1)
headers = iter_headers(dataset.file_name, sep='\t', count=-1)
for line in headers:
if line[0] == 'label' and line[1] == 'Group':
skip = 1
@@ -187,7 +191,7 @@ class GroupAbund(Otu):
>>> GroupAbund().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -234,7 +238,7 @@ class SecondaryStructureMap(Tabular):
>>> SecondaryStructureMap().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
line_num = 0
rowidxmap = {}
for line in headers:
@@ -302,7 +306,7 @@ class DistanceMatrix(Text):
def set_meta(self, dataset, overwrite=True, skip=0, **kwd):
super(DistanceMatrix, self).set_meta(dataset, overwrite=overwrite, skip=skip, **kwd)
headers = get_headers(dataset.file_name, sep='\t')
headers = iter_headers(dataset.file_name, sep='\t')
for line in headers:
if not line[0].startswith('@'):
try:
@@ -344,7 +348,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
False
"""
numlines = 300
headers = get_headers(filename, sep='\t', count=numlines)
headers = iter_headers(filename, sep='\t', count=numlines)
line_num = 0
for line in headers:
if not line[0].startswith('@'):
@@ -405,7 +409,7 @@ class SquareDistanceMatrix(DistanceMatrix):
False
"""
numlines = 300
headers = get_headers(filename, sep='\t', count=numlines)
headers = iter_headers(filename, sep='\t', count=numlines)
line_num = 0
for line in headers:
if not line[0].startswith('@'):
@@ -461,7 +465,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
>>> PairwiseDistanceMatrix().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -525,7 +529,7 @@ class Group(Tabular):
super(Group, self).set_meta(dataset, overwrite, skip, max_data_lines)
group_names = set()
headers = get_headers(dataset.file_name, sep='\t', count=-1)
headers = iter_headers(dataset.file_name, sep='\t', count=-1)
for line in headers:
if len(line) > 1:
group_names.add(line[1])
@@ -558,7 +562,7 @@ class Oligos(Text):
>>> Oligos().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@') and not line[0].startswith('#'):
@@ -602,7 +606,7 @@ class Frequency(Tabular):
>>> Frequency().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@'):
@@ -653,7 +657,7 @@ class Quantile(Tabular):
>>> Quantile().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
for line in headers:
if not line[0].startswith('@') and not line[0].startswith('#'):
@@ -691,7 +695,7 @@ class LaneMask(Text):
>>> LaneMask().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = get_headers(filename, sep='\t', count=2)
if len(headers) != 1 or len(headers[0]) != 1:
return False
@@ -774,7 +778,7 @@ class RefTaxonomy(Tabular):
>>> RefTaxonomy().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t', count=300)
headers = iter_headers(filename, sep='\t', count=300)
count = 0
pat_prog = re.compile('^([^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?(;[^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?)*(;)?)$')
found_semicolons = False
@@ -849,7 +853,7 @@ class Axes(Tabular):
>>> Axes().sniff( fname )
False
"""
headers = get_headers(filename, sep='\t')
headers = iter_headers(filename, sep='\t')
count = 0
col_cnt = None
all_integers = True
+9 -4
View File
@@ -9,6 +9,7 @@ import re
import string
import sys
from cgi import escape
from itertools import islice
import bx.align.maf
@@ -16,7 +17,10 @@ from galaxy import util
from galaxy.datatypes import metadata
from galaxy.datatypes.binary import Binary
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import get_headers
from galaxy.datatypes.sniff import (
get_headers,
iter_headers
)
from galaxy.util import (
compression_utils,
nice_size
@@ -611,7 +615,7 @@ class BaseFastq ( Sequence ):
compressed = is_gzip(filename) or is_bz2(filename)
if compressed and not isinstance(self, Binary):
return False
headers = get_headers( filename, None, count=1000 )
headers = iter_headers( filename, None, count=1000 )
# If this is a FastqSanger-derived class, then check to see if the base qualities match
if isinstance(self, FastqSanger) or isinstance(self, FastqSangerGz) or isinstance(self, FastqSangerBz2):
@@ -621,7 +625,8 @@ class BaseFastq ( Sequence ):
bases_regexp = re.compile( "^[NGTAC]*" )
# check that first block looks like a fastq block
try:
if len( headers ) >= 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]:
headers = get_headers( filename, None, count=4 )
if len( headers ) == 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]:
# Check the sequence line, make sure it contains only G/C/A/T/N
if not bases_regexp.match( headers[1][0] ):
return False
@@ -695,7 +700,7 @@ class BaseFastq ( Sequence ):
@staticmethod
def sangerQualities( lines ):
"""Presuming lines are lines from a fastq file, return True if the qualities are compatible with sanger encoding"""
for line in lines[3::4]:
for line in islice(lines, 3, None, 4):
if not all(_ >= '!' and _ <= 'M' for _ in line[0]):
return False
return True
+17 -15
View File
@@ -200,19 +200,7 @@ def convert_newlines_sep2tabs( fname, in_place=True, patt="\\s+", tmp_dir=None,
return ( i + 1, temp_name )
def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ):
"""
Returns a list with the first 'count' lines split by 'sep', ignoring lines
starting with 'comment_designator'
>>> fname = get_test_fname('complete.bed')
>>> get_headers(fname,'\\t')
[['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']]
>>> fname = get_test_fname('test.gff')
>>> get_headers(fname, '\\t', count=5, comment_designator='#')
[[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']]
"""
headers = []
def iter_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ):
with compression_utils.get_fileobj(fname) as in_file:
idx = 0
for line in in_file:
@@ -225,11 +213,25 @@ def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=N
comment_designator = comment_designator.encode( 'utf-8' )
if comment_designator is not None and comment_designator != '' and line.startswith( comment_designator ):
continue
headers.append( line.split(sep) )
yield line.split(sep)
idx += 1
if idx == count:
break
return headers
def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ):
"""
Returns a list with the first 'count' lines split by 'sep', ignoring lines
starting with 'comment_designator'
>>> fname = get_test_fname('complete.bed')
>>> get_headers(fname,'\\t')
[['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']]
>>> fname = get_test_fname('test.gff')
>>> get_headers(fname, '\\t', count=5, comment_designator='#')
[[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']]
"""
return list(iter_headers(fname=fname, sep=sep, count=count, is_multi_byte=is_multi_byte, comment_designator=comment_designator))
def is_column_based( fname, sep='\t', skip=0, is_multi_byte=False ):
+5 -2
View File
@@ -17,7 +17,10 @@ from json import dumps
from galaxy import util
from galaxy.datatypes import data, metadata
from galaxy.datatypes.metadata import MetadataElement
from galaxy.datatypes.sniff import get_headers
from galaxy.datatypes.sniff import (
get_headers,
iter_headers
)
from galaxy.util import compression_utils
from . import dataproviders
@@ -638,7 +641,7 @@ class Pileup( Tabular ):
>>> Pileup().sniff( fname )
True
"""
headers = get_headers( filename, '\t' )
headers = iter_headers( filename, '\t' )
try:
for hdr in headers:
if hdr and not hdr[0].startswith( '#' ):
+2 -2
View File
@@ -12,7 +12,7 @@ import tempfile
from galaxy.datatypes.data import get_file_peek, Text
from galaxy.datatypes.metadata import MetadataElement, MetadataParameter
from galaxy.datatypes.sniff import get_headers
from galaxy.datatypes.sniff import iter_headers
from galaxy.util import nice_size, string_as_bool
log = logging.getLogger(__name__)
@@ -47,7 +47,7 @@ class Html( Text ):
>>> Html().sniff( fname )
True
"""
headers = get_headers( filename, None )
headers = iter_headers( filename, None )
try:
for i, hdr in enumerate(headers):
if hdr and hdr[0].lower().find( '<html>' ) >= 0: