Fix sniffers of interval datatypes for files starting with many comments

For files starting with >60 comment lines, the `sniff_prefix()` method of
Bed, Gff, Gff3 and Gtf classes was returning `True` when actually all
headers lines were skipped.

Also:
- Don't skip lines having more than one column just because the first
column is empty.
- Properly skip comment lines in `is_column_based()`
- Add test file which would have been incorrectly sniffed as `gtf` before.
  It still isn't sniffed as `gff3`, but that's fine because it has too many
  comment lines for the default header prefix.
This commit is contained in:
Nicola Soranzo
2021-02-19 11:50:52 +00:00
parent ae6b37eec6
commit 26f731f060
3 changed files with 165 additions and 85 deletions
+85 -71
View File
@@ -477,9 +477,9 @@ class Bed(Interval):
if not get_headers(file_prefix, '\t', comment_designator='#', count=1):
return False
try:
headers = iter_headers(file_prefix, '\t', comment_designator='#')
for hdr in headers:
if hdr[0] == '':
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t', comment_designator='#'):
if not hdr or hdr == ['']:
continue
if len(hdr) < 3 or len(hdr) > 12:
return False
@@ -542,7 +542,8 @@ class Bed(Interval):
return False
if len(block_sizes) != block_count or len(block_starts) != block_count:
return False
return True
found_valid_lines = True
return found_valid_lines
except Exception:
return False
@@ -818,28 +819,33 @@ class Gff(Tabular, _RemoteCallMixin):
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t'):
if not hdr or hdr == ['']:
continue
if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
return False
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != 9:
return False
# The gff-version header comment may have been stripped, so inspect the data
if hdr[0].startswith('#'):
continue
if len(hdr) != 9:
return False
try:
int(hdr[3])
int(hdr[4])
except Exception:
return False
if hdr[5] != '.':
try:
int(hdr[3])
int(hdr[4])
float(hdr[5])
except Exception:
return False
if hdr[5] != '.':
try:
float(hdr[5])
except Exception:
return False
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
return True
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
found_valid_lines = True
return found_valid_lines
except Exception:
return False
@@ -953,37 +959,41 @@ class Gff3(Gff):
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t'):
if not hdr or hdr == ['']:
continue
if hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
return True
elif hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0:
elif hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0:
return False
# Header comments may have been stripped, so inspect the data
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != 9:
# The gff-version header comment may have been stripped, so inspect the data
if hdr[0].startswith('#'):
continue
if len(hdr) != 9:
return False
try:
int(hdr[3])
except Exception:
if hdr[3] != '.':
return False
try:
int(hdr[4])
except Exception:
if hdr[4] != '.':
return False
if hdr[5] != '.':
try:
int(hdr[3])
float(hdr[5])
except Exception:
if hdr[3] != '.':
return False
try:
int(hdr[4])
except Exception:
if hdr[4] != '.':
return False
if hdr[5] != '.':
try:
float(hdr[5])
except Exception:
return False
if hdr[6] not in self.valid_gff3_strand:
return False
if hdr[7] not in self.valid_gff3_phase:
return False
parse_gff3_attributes(hdr[8])
return True
if hdr[6] not in self.valid_gff3_strand:
return False
if hdr[7] not in self.valid_gff3_phase:
return False
parse_gff3_attributes(hdr[8])
found_valid_lines = True
return found_valid_lines
except Exception:
return False
@@ -1031,34 +1041,38 @@ class Gtf(Gff):
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t'):
if not hdr or hdr == ['']:
continue
if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
return False
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != 9:
return False
# The gff-version header comment may have been stripped, so inspect the data
if hdr[0].startswith('#'):
continue
if len(hdr) != 9:
return False
try:
int(hdr[3])
int(hdr[4])
except Exception:
return False
if hdr[5] != '.':
try:
int(hdr[3])
int(hdr[4])
float(hdr[5])
except Exception:
return False
if hdr[5] != '.':
try:
float(hdr[5])
except Exception:
return False
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
# Check attributes for gene_id (transcript_id is also mandatory
# but not for genes)
attributes = parse_gff_attributes(hdr[8])
if 'gene_id' not in attributes:
return False
return True
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
# Check attributes for gene_id (transcript_id is also mandatory
# but not for genes)
attributes = parse_gff_attributes(hdr[8])
if 'gene_id' not in attributes:
return False
found_valid_lines = True
return found_valid_lines
except Exception:
return False
+15 -14
View File
@@ -235,24 +235,22 @@ def is_column_based(fname_or_file_prefix, sep='\t', skip=0):
return False
try:
headers = get_headers(fname_or_file_prefix, sep)
headers = get_headers(fname_or_file_prefix, sep, comment_designator='#')[skip:]
except UnicodeDecodeError:
return False
count = 0
if not headers:
return False
for hdr in headers[skip:]:
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) > 1:
for hdr in headers:
if hdr and hdr != ['']:
if count:
if len(hdr) != count:
return False
else:
count = len(hdr)
break
if count < 2:
return False
for hdr in headers[skip:]:
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != count:
return False
return True
if count < 2:
return False
return count >= 2
def guess_ext(fname, sniff_order, is_binary=False):
@@ -303,13 +301,13 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> guess_ext(fname, sniff_order)
'gff3'
>>> fname = get_test_fname('2.txt')
>>> guess_ext(fname, sniff_order) # 2.txt
>>> guess_ext(fname, sniff_order)
'txt'
>>> fname = get_test_fname('2.tabular')
>>> guess_ext(fname, sniff_order)
'tabular'
>>> fname = get_test_fname('3.txt')
>>> guess_ext(fname, sniff_order) # 3.txt
>>> guess_ext(fname, sniff_order)
'txt'
>>> fname = get_test_fname('test_tab1.tabular')
>>> guess_ext(fname, sniff_order)
@@ -451,6 +449,9 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> fname = get_test_fname('1imzml')
>>> guess_ext(fname, sniff_order) # This test case is ensuring doesn't throw exception, actual value could change if non-utf encoding handling improves.
'data'
>>> fname = get_test_fname('too_many_comments_gff3.tabular')
>>> guess_ext(fname, sniff_order) # It's a VCF but is sniffed as tabular because of the limit on the number of header lines we read
'tabular'
"""
file_prefix = FilePrefix(fname)
file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary)
@@ -0,0 +1,65 @@
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
ctgA est match 5410 7503 . - . ID=EST:agt830.3;Target=agt830.3+1+595
ctgA est HSP 7000 7503 . - . Parent=EST:agt830.3;Target=agt830.3+1+504
ctgA est HSP 5410 5500 . ? . Parent=EST:agt830.3;Target=agt830.3+505+595