mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merge pull request #4388 from mvdbeek/iter_headers
Replace list with generator when iterating headers
This commit is contained in:
@@ -13,7 +13,10 @@ from six.moves.urllib.parse import quote_plus
|
||||
from galaxy import util
|
||||
from galaxy.datatypes import metadata
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import get_headers
|
||||
from galaxy.datatypes.sniff import (
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
from galaxy.datatypes.tabular import Tabular
|
||||
from galaxy.datatypes.util.gff_util import parse_gff3_attributes, parse_gff_attributes
|
||||
from galaxy.web import url_for
|
||||
@@ -313,12 +316,12 @@ class Interval( Tabular ):
|
||||
>>> Interval().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, '\t', comment_designator='#' )
|
||||
try:
|
||||
"""
|
||||
If we got here, we already know the file is_column_based and is not bed,
|
||||
so we'll just look for some valid data.
|
||||
"""
|
||||
headers = iter_headers( filename, '\t', comment_designator='#' )
|
||||
for hdr in headers:
|
||||
if hdr:
|
||||
if len(hdr) < 3:
|
||||
@@ -489,10 +492,10 @@ class Bed( Interval ):
|
||||
>>> Bed().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, '\t', comment_designator='#' )
|
||||
if not get_headers( filename, '\t', comment_designator='#', count=1 ):
|
||||
return False
|
||||
try:
|
||||
if not headers:
|
||||
return False
|
||||
headers = iter_headers( filename, '\t', comment_designator='#' )
|
||||
for hdr in headers:
|
||||
if hdr[0] == '':
|
||||
continue
|
||||
@@ -832,10 +835,10 @@ class Gff( Tabular, _RemoteCallMixin ):
|
||||
>>> Gff().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, '\t' )
|
||||
if len(get_headers( filename, '\t', count=2 )) < 2:
|
||||
return False
|
||||
try:
|
||||
if len(headers) < 2:
|
||||
return False
|
||||
headers = iter_headers( filename, '\t' )
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0:
|
||||
return False
|
||||
@@ -964,10 +967,10 @@ class Gff3( Gff ):
|
||||
>>> Gff3().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, '\t' )
|
||||
if len(get_headers( filename, '\t', count=2 )) < 2:
|
||||
return False
|
||||
try:
|
||||
if len(headers) < 2:
|
||||
return False
|
||||
headers = iter_headers( filename, '\t' )
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '3' ) >= 0:
|
||||
return True
|
||||
@@ -1039,10 +1042,10 @@ class Gtf( Gff ):
|
||||
>>> Gtf().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, '\t' )
|
||||
if len(get_headers( filename, '\t', count=2 )) < 2:
|
||||
return False
|
||||
try:
|
||||
if len(headers) < 2:
|
||||
return False
|
||||
headers = iter_headers( filename, '\t' )
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith( '##gff-version' ) and hdr[0].find( '2' ) < 0:
|
||||
return False
|
||||
@@ -1235,8 +1238,8 @@ class Wiggle( Tabular, _RemoteCallMixin ):
|
||||
>>> Wiggle().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
try:
|
||||
headers = iter_headers( filename, None )
|
||||
for hdr in headers:
|
||||
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
|
||||
return True
|
||||
@@ -1371,7 +1374,7 @@ class CustomTrack ( Tabular ):
|
||||
>>> CustomTrack().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
headers = iter_headers( filename, None )
|
||||
first_line = True
|
||||
for hdr in headers:
|
||||
if first_line:
|
||||
|
||||
@@ -10,7 +10,10 @@ from galaxy.datatypes import (
|
||||
from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.data import get_file_peek
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import get_headers
|
||||
from galaxy.datatypes.sniff import (
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
from galaxy.datatypes.tabular import Tabular
|
||||
from galaxy.datatypes.xml import GenericXml
|
||||
|
||||
@@ -461,7 +464,7 @@ class PDB(GenericMolFile):
|
||||
>>> PDB().sniff(fname)
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep=' ', count=300)
|
||||
headers = iter_headers(filename, sep=' ', count=300)
|
||||
h = t = c = s = k = e = False
|
||||
for line in headers:
|
||||
section_name = line[0].strip()
|
||||
@@ -514,7 +517,7 @@ class PDBQT(GenericMolFile):
|
||||
>>> PDBQT().sniff(fname)
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep=' ', count=300)
|
||||
headers = iter_headers(filename, sep=' ', count=300)
|
||||
h = t = c = s = k = False
|
||||
for line in headers:
|
||||
section_name = line[0].strip()
|
||||
@@ -607,7 +610,7 @@ class InChI(Tabular):
|
||||
>>> InChI().sniff(fname)
|
||||
False
|
||||
"""
|
||||
inchi_lines = get_headers(filename, sep=' ', count=10)
|
||||
inchi_lines = iter_headers(filename, sep=' ', count=10)
|
||||
for inchi in inchi_lines:
|
||||
if not inchi[0].startswith('InChI='):
|
||||
return False
|
||||
|
||||
@@ -7,7 +7,10 @@ import sys
|
||||
|
||||
from galaxy.datatypes.data import Text
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import get_headers
|
||||
from galaxy.datatypes.sniff import (
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
from galaxy.datatypes.tabular import Tabular
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -32,10 +35,11 @@ class Otu(Text):
|
||||
data_lines = 0
|
||||
comment_lines = 0
|
||||
|
||||
headers = get_headers(dataset.file_name, sep='\t', count=-1)
|
||||
headers = iter_headers(dataset.file_name, sep='\t', count=-1)
|
||||
first_line = get_headers(dataset.file_name, sep='\t', count=1)
|
||||
# set otulabels
|
||||
if len(headers[0]) > 2:
|
||||
otulabel_names = headers[0][2:]
|
||||
if len(first_line) > 2:
|
||||
otulabel_names = first_line[2:]
|
||||
# set label names and number of lines
|
||||
for line in headers:
|
||||
if len(line) >= 2 and not line[0].startswith('@'):
|
||||
@@ -64,7 +68,7 @@ class Otu(Text):
|
||||
>>> Otu().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -109,7 +113,7 @@ class Sabund(Otu):
|
||||
>>> Sabund().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -151,7 +155,7 @@ class GroupAbund(Otu):
|
||||
comment_lines = 0
|
||||
ncols = 0
|
||||
|
||||
headers = get_headers(dataset.file_name, sep='\t', count=-1)
|
||||
headers = iter_headers(dataset.file_name, sep='\t', count=-1)
|
||||
for line in headers:
|
||||
if line[0] == 'label' and line[1] == 'Group':
|
||||
skip = 1
|
||||
@@ -187,7 +191,7 @@ class GroupAbund(Otu):
|
||||
>>> GroupAbund().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -234,7 +238,7 @@ class SecondaryStructureMap(Tabular):
|
||||
>>> SecondaryStructureMap().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
line_num = 0
|
||||
rowidxmap = {}
|
||||
for line in headers:
|
||||
@@ -302,7 +306,7 @@ class DistanceMatrix(Text):
|
||||
def set_meta(self, dataset, overwrite=True, skip=0, **kwd):
|
||||
super(DistanceMatrix, self).set_meta(dataset, overwrite=overwrite, skip=skip, **kwd)
|
||||
|
||||
headers = get_headers(dataset.file_name, sep='\t')
|
||||
headers = iter_headers(dataset.file_name, sep='\t')
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
try:
|
||||
@@ -344,7 +348,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
|
||||
False
|
||||
"""
|
||||
numlines = 300
|
||||
headers = get_headers(filename, sep='\t', count=numlines)
|
||||
headers = iter_headers(filename, sep='\t', count=numlines)
|
||||
line_num = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -405,7 +409,7 @@ class SquareDistanceMatrix(DistanceMatrix):
|
||||
False
|
||||
"""
|
||||
numlines = 300
|
||||
headers = get_headers(filename, sep='\t', count=numlines)
|
||||
headers = iter_headers(filename, sep='\t', count=numlines)
|
||||
line_num = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -461,7 +465,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
|
||||
>>> PairwiseDistanceMatrix().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -525,7 +529,7 @@ class Group(Tabular):
|
||||
super(Group, self).set_meta(dataset, overwrite, skip, max_data_lines)
|
||||
|
||||
group_names = set()
|
||||
headers = get_headers(dataset.file_name, sep='\t', count=-1)
|
||||
headers = iter_headers(dataset.file_name, sep='\t', count=-1)
|
||||
for line in headers:
|
||||
if len(line) > 1:
|
||||
group_names.add(line[1])
|
||||
@@ -558,7 +562,7 @@ class Oligos(Text):
|
||||
>>> Oligos().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@') and not line[0].startswith('#'):
|
||||
@@ -602,7 +606,7 @@ class Frequency(Tabular):
|
||||
>>> Frequency().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -653,7 +657,7 @@ class Quantile(Tabular):
|
||||
>>> Quantile().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@') and not line[0].startswith('#'):
|
||||
@@ -691,7 +695,7 @@ class LaneMask(Text):
|
||||
>>> LaneMask().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = get_headers(filename, sep='\t', count=2)
|
||||
if len(headers) != 1 or len(headers[0]) != 1:
|
||||
return False
|
||||
|
||||
@@ -774,7 +778,7 @@ class RefTaxonomy(Tabular):
|
||||
>>> RefTaxonomy().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t', count=300)
|
||||
headers = iter_headers(filename, sep='\t', count=300)
|
||||
count = 0
|
||||
pat_prog = re.compile('^([^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?(;[^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?)*(;)?)$')
|
||||
found_semicolons = False
|
||||
@@ -849,7 +853,7 @@ class Axes(Tabular):
|
||||
>>> Axes().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t')
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
count = 0
|
||||
col_cnt = None
|
||||
all_integers = True
|
||||
|
||||
@@ -9,6 +9,7 @@ import re
|
||||
import string
|
||||
import sys
|
||||
from cgi import escape
|
||||
from itertools import islice
|
||||
|
||||
import bx.align.maf
|
||||
|
||||
@@ -16,7 +17,10 @@ from galaxy import util
|
||||
from galaxy.datatypes import metadata
|
||||
from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import get_headers
|
||||
from galaxy.datatypes.sniff import (
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
from galaxy.util import (
|
||||
compression_utils,
|
||||
nice_size
|
||||
@@ -611,7 +615,7 @@ class BaseFastq ( Sequence ):
|
||||
compressed = is_gzip(filename) or is_bz2(filename)
|
||||
if compressed and not isinstance(self, Binary):
|
||||
return False
|
||||
headers = get_headers( filename, None, count=1000 )
|
||||
headers = iter_headers( filename, None, count=1000 )
|
||||
|
||||
# If this is a FastqSanger-derived class, then check to see if the base qualities match
|
||||
if isinstance(self, FastqSanger) or isinstance(self, FastqSangerGz) or isinstance(self, FastqSangerBz2):
|
||||
@@ -621,7 +625,8 @@ class BaseFastq ( Sequence ):
|
||||
bases_regexp = re.compile( "^[NGTAC]*" )
|
||||
# check that first block looks like a fastq block
|
||||
try:
|
||||
if len( headers ) >= 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]:
|
||||
headers = get_headers( filename, None, count=4 )
|
||||
if len( headers ) == 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]:
|
||||
# Check the sequence line, make sure it contains only G/C/A/T/N
|
||||
if not bases_regexp.match( headers[1][0] ):
|
||||
return False
|
||||
@@ -695,7 +700,7 @@ class BaseFastq ( Sequence ):
|
||||
@staticmethod
|
||||
def sangerQualities( lines ):
|
||||
"""Presuming lines are lines from a fastq file, return True if the qualities are compatible with sanger encoding"""
|
||||
for line in lines[3::4]:
|
||||
for line in islice(lines, 3, None, 4):
|
||||
if not all(_ >= '!' and _ <= 'M' for _ in line[0]):
|
||||
return False
|
||||
return True
|
||||
|
||||
@@ -200,19 +200,7 @@ def convert_newlines_sep2tabs( fname, in_place=True, patt="\\s+", tmp_dir=None,
|
||||
return ( i + 1, temp_name )
|
||||
|
||||
|
||||
def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ):
|
||||
"""
|
||||
Returns a list with the first 'count' lines split by 'sep', ignoring lines
|
||||
starting with 'comment_designator'
|
||||
|
||||
>>> fname = get_test_fname('complete.bed')
|
||||
>>> get_headers(fname,'\\t')
|
||||
[['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']]
|
||||
>>> fname = get_test_fname('test.gff')
|
||||
>>> get_headers(fname, '\\t', count=5, comment_designator='#')
|
||||
[[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']]
|
||||
"""
|
||||
headers = []
|
||||
def iter_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ):
|
||||
with compression_utils.get_fileobj(fname) as in_file:
|
||||
idx = 0
|
||||
for line in in_file:
|
||||
@@ -225,11 +213,25 @@ def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=N
|
||||
comment_designator = comment_designator.encode( 'utf-8' )
|
||||
if comment_designator is not None and comment_designator != '' and line.startswith( comment_designator ):
|
||||
continue
|
||||
headers.append( line.split(sep) )
|
||||
yield line.split(sep)
|
||||
idx += 1
|
||||
if idx == count:
|
||||
break
|
||||
return headers
|
||||
|
||||
|
||||
def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ):
|
||||
"""
|
||||
Returns a list with the first 'count' lines split by 'sep', ignoring lines
|
||||
starting with 'comment_designator'
|
||||
|
||||
>>> fname = get_test_fname('complete.bed')
|
||||
>>> get_headers(fname,'\\t')
|
||||
[['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']]
|
||||
>>> fname = get_test_fname('test.gff')
|
||||
>>> get_headers(fname, '\\t', count=5, comment_designator='#')
|
||||
[[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']]
|
||||
"""
|
||||
return list(iter_headers(fname=fname, sep=sep, count=count, is_multi_byte=is_multi_byte, comment_designator=comment_designator))
|
||||
|
||||
|
||||
def is_column_based( fname, sep='\t', skip=0, is_multi_byte=False ):
|
||||
|
||||
@@ -17,7 +17,10 @@ from json import dumps
|
||||
from galaxy import util
|
||||
from galaxy.datatypes import data, metadata
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import get_headers
|
||||
from galaxy.datatypes.sniff import (
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
from galaxy.util import compression_utils
|
||||
|
||||
from . import dataproviders
|
||||
@@ -638,7 +641,7 @@ class Pileup( Tabular ):
|
||||
>>> Pileup().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, '\t' )
|
||||
headers = iter_headers( filename, '\t' )
|
||||
try:
|
||||
for hdr in headers:
|
||||
if hdr and not hdr[0].startswith( '#' ):
|
||||
|
||||
@@ -12,7 +12,7 @@ import tempfile
|
||||
|
||||
from galaxy.datatypes.data import get_file_peek, Text
|
||||
from galaxy.datatypes.metadata import MetadataElement, MetadataParameter
|
||||
from galaxy.datatypes.sniff import get_headers
|
||||
from galaxy.datatypes.sniff import iter_headers
|
||||
from galaxy.util import nice_size, string_as_bool
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -47,7 +47,7 @@ class Html( Text ):
|
||||
>>> Html().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = get_headers( filename, None )
|
||||
headers = iter_headers( filename, None )
|
||||
try:
|
||||
for i, hdr in enumerate(headers):
|
||||
if hdr and hdr[0].lower().find( '<html>' ) >= 0:
|
||||
|
||||
Reference in New Issue
Block a user