mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Fix sniffers of interval datatypes for files starting with many comments
For files starting with >60 comment lines, the `sniff_prefix()` method of Bed, Gff, Gff3 and Gtf classes was returning `True` when actually all headers lines were skipped. Also: - Don't skip lines having more than one column just because the first column is empty. - Properly skip comment lines in `is_column_based()` - Add test file which would have been incorrectly sniffed as `gtf` before. It still isn't sniffed as `gff3`, but that's fine because it has too many comment lines for the default header prefix.
This commit is contained in:
@@ -477,9 +477,9 @@ class Bed(Interval):
|
||||
if not get_headers(file_prefix, '\t', comment_designator='#', count=1):
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(file_prefix, '\t', comment_designator='#')
|
||||
for hdr in headers:
|
||||
if hdr[0] == '':
|
||||
found_valid_lines = False
|
||||
for hdr in iter_headers(file_prefix, '\t', comment_designator='#'):
|
||||
if not hdr or hdr == ['']:
|
||||
continue
|
||||
if len(hdr) < 3 or len(hdr) > 12:
|
||||
return False
|
||||
@@ -542,7 +542,8 @@ class Bed(Interval):
|
||||
return False
|
||||
if len(block_sizes) != block_count or len(block_starts) != block_count:
|
||||
return False
|
||||
return True
|
||||
found_valid_lines = True
|
||||
return found_valid_lines
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
@@ -818,28 +819,33 @@ class Gff(Tabular, _RemoteCallMixin):
|
||||
if len(get_headers(file_prefix, '\t', count=2)) < 2:
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(file_prefix, '\t')
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
|
||||
found_valid_lines = False
|
||||
for hdr in iter_headers(file_prefix, '\t'):
|
||||
if not hdr or hdr == ['']:
|
||||
continue
|
||||
if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
|
||||
return False
|
||||
if hdr and hdr[0] and not hdr[0].startswith('#'):
|
||||
if len(hdr) != 9:
|
||||
return False
|
||||
# The gff-version header comment may have been stripped, so inspect the data
|
||||
if hdr[0].startswith('#'):
|
||||
continue
|
||||
if len(hdr) != 9:
|
||||
return False
|
||||
try:
|
||||
int(hdr[3])
|
||||
int(hdr[4])
|
||||
except Exception:
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
int(hdr[3])
|
||||
int(hdr[4])
|
||||
float(hdr[5])
|
||||
except Exception:
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
float(hdr[5])
|
||||
except Exception:
|
||||
return False
|
||||
if hdr[6] not in data.valid_strand:
|
||||
return False
|
||||
if hdr[7] not in self.valid_gff_frame:
|
||||
return False
|
||||
return True
|
||||
if hdr[6] not in data.valid_strand:
|
||||
return False
|
||||
if hdr[7] not in self.valid_gff_frame:
|
||||
return False
|
||||
found_valid_lines = True
|
||||
return found_valid_lines
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
@@ -953,37 +959,41 @@ class Gff3(Gff):
|
||||
if len(get_headers(file_prefix, '\t', count=2)) < 2:
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(file_prefix, '\t')
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
|
||||
found_valid_lines = False
|
||||
for hdr in iter_headers(file_prefix, '\t'):
|
||||
if not hdr or hdr == ['']:
|
||||
continue
|
||||
if hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
|
||||
return True
|
||||
elif hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0:
|
||||
elif hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0:
|
||||
return False
|
||||
# Header comments may have been stripped, so inspect the data
|
||||
if hdr and hdr[0] and not hdr[0].startswith('#'):
|
||||
if len(hdr) != 9:
|
||||
# The gff-version header comment may have been stripped, so inspect the data
|
||||
if hdr[0].startswith('#'):
|
||||
continue
|
||||
if len(hdr) != 9:
|
||||
return False
|
||||
try:
|
||||
int(hdr[3])
|
||||
except Exception:
|
||||
if hdr[3] != '.':
|
||||
return False
|
||||
try:
|
||||
int(hdr[4])
|
||||
except Exception:
|
||||
if hdr[4] != '.':
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
int(hdr[3])
|
||||
float(hdr[5])
|
||||
except Exception:
|
||||
if hdr[3] != '.':
|
||||
return False
|
||||
try:
|
||||
int(hdr[4])
|
||||
except Exception:
|
||||
if hdr[4] != '.':
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
float(hdr[5])
|
||||
except Exception:
|
||||
return False
|
||||
if hdr[6] not in self.valid_gff3_strand:
|
||||
return False
|
||||
if hdr[7] not in self.valid_gff3_phase:
|
||||
return False
|
||||
parse_gff3_attributes(hdr[8])
|
||||
return True
|
||||
if hdr[6] not in self.valid_gff3_strand:
|
||||
return False
|
||||
if hdr[7] not in self.valid_gff3_phase:
|
||||
return False
|
||||
parse_gff3_attributes(hdr[8])
|
||||
found_valid_lines = True
|
||||
return found_valid_lines
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
@@ -1031,34 +1041,38 @@ class Gtf(Gff):
|
||||
if len(get_headers(file_prefix, '\t', count=2)) < 2:
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(file_prefix, '\t')
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
|
||||
found_valid_lines = False
|
||||
for hdr in iter_headers(file_prefix, '\t'):
|
||||
if not hdr or hdr == ['']:
|
||||
continue
|
||||
if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
|
||||
return False
|
||||
if hdr and hdr[0] and not hdr[0].startswith('#'):
|
||||
if len(hdr) != 9:
|
||||
return False
|
||||
# The gff-version header comment may have been stripped, so inspect the data
|
||||
if hdr[0].startswith('#'):
|
||||
continue
|
||||
if len(hdr) != 9:
|
||||
return False
|
||||
try:
|
||||
int(hdr[3])
|
||||
int(hdr[4])
|
||||
except Exception:
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
int(hdr[3])
|
||||
int(hdr[4])
|
||||
float(hdr[5])
|
||||
except Exception:
|
||||
return False
|
||||
if hdr[5] != '.':
|
||||
try:
|
||||
float(hdr[5])
|
||||
except Exception:
|
||||
return False
|
||||
if hdr[6] not in data.valid_strand:
|
||||
return False
|
||||
if hdr[7] not in self.valid_gff_frame:
|
||||
return False
|
||||
|
||||
# Check attributes for gene_id (transcript_id is also mandatory
|
||||
# but not for genes)
|
||||
attributes = parse_gff_attributes(hdr[8])
|
||||
if 'gene_id' not in attributes:
|
||||
return False
|
||||
return True
|
||||
if hdr[6] not in data.valid_strand:
|
||||
return False
|
||||
if hdr[7] not in self.valid_gff_frame:
|
||||
return False
|
||||
# Check attributes for gene_id (transcript_id is also mandatory
|
||||
# but not for genes)
|
||||
attributes = parse_gff_attributes(hdr[8])
|
||||
if 'gene_id' not in attributes:
|
||||
return False
|
||||
found_valid_lines = True
|
||||
return found_valid_lines
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
@@ -235,24 +235,22 @@ def is_column_based(fname_or_file_prefix, sep='\t', skip=0):
|
||||
return False
|
||||
|
||||
try:
|
||||
headers = get_headers(fname_or_file_prefix, sep)
|
||||
headers = get_headers(fname_or_file_prefix, sep, comment_designator='#')[skip:]
|
||||
except UnicodeDecodeError:
|
||||
return False
|
||||
count = 0
|
||||
if not headers:
|
||||
return False
|
||||
for hdr in headers[skip:]:
|
||||
if hdr and hdr[0] and not hdr[0].startswith('#'):
|
||||
if len(hdr) > 1:
|
||||
for hdr in headers:
|
||||
if hdr and hdr != ['']:
|
||||
if count:
|
||||
if len(hdr) != count:
|
||||
return False
|
||||
else:
|
||||
count = len(hdr)
|
||||
break
|
||||
if count < 2:
|
||||
return False
|
||||
for hdr in headers[skip:]:
|
||||
if hdr and hdr[0] and not hdr[0].startswith('#'):
|
||||
if len(hdr) != count:
|
||||
return False
|
||||
return True
|
||||
if count < 2:
|
||||
return False
|
||||
return count >= 2
|
||||
|
||||
|
||||
def guess_ext(fname, sniff_order, is_binary=False):
|
||||
@@ -303,13 +301,13 @@ def guess_ext(fname, sniff_order, is_binary=False):
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'gff3'
|
||||
>>> fname = get_test_fname('2.txt')
|
||||
>>> guess_ext(fname, sniff_order) # 2.txt
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'txt'
|
||||
>>> fname = get_test_fname('2.tabular')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'tabular'
|
||||
>>> fname = get_test_fname('3.txt')
|
||||
>>> guess_ext(fname, sniff_order) # 3.txt
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'txt'
|
||||
>>> fname = get_test_fname('test_tab1.tabular')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
@@ -451,6 +449,9 @@ def guess_ext(fname, sniff_order, is_binary=False):
|
||||
>>> fname = get_test_fname('1imzml')
|
||||
>>> guess_ext(fname, sniff_order) # This test case is ensuring doesn't throw exception, actual value could change if non-utf encoding handling improves.
|
||||
'data'
|
||||
>>> fname = get_test_fname('too_many_comments_gff3.tabular')
|
||||
>>> guess_ext(fname, sniff_order) # It's a VCF but is sniffed as tabular because of the limit on the number of header lines we read
|
||||
'tabular'
|
||||
"""
|
||||
file_prefix = FilePrefix(fname)
|
||||
file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary)
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
#
|
||||
ctgA est match 5410 7503 . - . ID=EST:agt830.3;Target=agt830.3+1+595
|
||||
ctgA est HSP 7000 7503 . - . Parent=EST:agt830.3;Target=agt830.3+1+504
|
||||
ctgA est HSP 5410 5500 . ? . Parent=EST:agt830.3;Target=agt830.3+505+595
|
||||
Reference in New Issue
Block a user