Merge pull request #11416 from nsoranzo/release_20.09_fix_sniffers_for_too_many_comments

[20.09] Fix sniffers of interval datatypes for files starting with many comments
This commit is contained in:
Marius van den Beek
2021-02-19 15:25:06 +01:00
committed by GitHub
4 changed files with 171 additions and 92 deletions
+85 -71
View File
@@ -477,9 +477,9 @@ class Bed(Interval):
if not get_headers(file_prefix, '\t', comment_designator='#', count=1):
return False
try:
headers = iter_headers(file_prefix, '\t', comment_designator='#')
for hdr in headers:
if hdr[0] == '':
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t', comment_designator='#'):
if not hdr or hdr == ['']:
continue
if len(hdr) < 3 or len(hdr) > 12:
return False
@@ -542,7 +542,8 @@ class Bed(Interval):
return False
if len(block_sizes) != block_count or len(block_starts) != block_count:
return False
return True
found_valid_lines = True
return found_valid_lines
except Exception:
return False
@@ -818,28 +819,33 @@ class Gff(Tabular, _RemoteCallMixin):
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t'):
if not hdr or hdr == ['']:
continue
if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
return False
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != 9:
return False
# The gff-version header comment may have been stripped, so inspect the data
if hdr[0].startswith('#'):
continue
if len(hdr) != 9:
return False
try:
int(hdr[3])
int(hdr[4])
except Exception:
return False
if hdr[5] != '.':
try:
int(hdr[3])
int(hdr[4])
float(hdr[5])
except Exception:
return False
if hdr[5] != '.':
try:
float(hdr[5])
except Exception:
return False
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
return True
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
found_valid_lines = True
return found_valid_lines
except Exception:
return False
@@ -953,37 +959,41 @@ class Gff3(Gff):
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t'):
if not hdr or hdr == ['']:
continue
if hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
return True
elif hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0:
elif hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0:
return False
# Header comments may have been stripped, so inspect the data
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != 9:
# The gff-version header comment may have been stripped, so inspect the data
if hdr[0].startswith('#'):
continue
if len(hdr) != 9:
return False
try:
int(hdr[3])
except Exception:
if hdr[3] != '.':
return False
try:
int(hdr[4])
except Exception:
if hdr[4] != '.':
return False
if hdr[5] != '.':
try:
int(hdr[3])
float(hdr[5])
except Exception:
if hdr[3] != '.':
return False
try:
int(hdr[4])
except Exception:
if hdr[4] != '.':
return False
if hdr[5] != '.':
try:
float(hdr[5])
except Exception:
return False
if hdr[6] not in self.valid_gff3_strand:
return False
if hdr[7] not in self.valid_gff3_phase:
return False
parse_gff3_attributes(hdr[8])
return True
if hdr[6] not in self.valid_gff3_strand:
return False
if hdr[7] not in self.valid_gff3_phase:
return False
parse_gff3_attributes(hdr[8])
found_valid_lines = True
return found_valid_lines
except Exception:
return False
@@ -1031,34 +1041,38 @@ class Gtf(Gff):
if len(get_headers(file_prefix, '\t', count=2)) < 2:
return False
try:
headers = iter_headers(file_prefix, '\t')
for hdr in headers:
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
found_valid_lines = False
for hdr in iter_headers(file_prefix, '\t'):
if not hdr or hdr == ['']:
continue
if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
return False
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != 9:
return False
# The gff-version header comment may have been stripped, so inspect the data
if hdr[0].startswith('#'):
continue
if len(hdr) != 9:
return False
try:
int(hdr[3])
int(hdr[4])
except Exception:
return False
if hdr[5] != '.':
try:
int(hdr[3])
int(hdr[4])
float(hdr[5])
except Exception:
return False
if hdr[5] != '.':
try:
float(hdr[5])
except Exception:
return False
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
# Check attributes for gene_id (transcript_id is also mandatory
# but not for genes)
attributes = parse_gff_attributes(hdr[8])
if 'gene_id' not in attributes:
return False
return True
if hdr[6] not in data.valid_strand:
return False
if hdr[7] not in self.valid_gff_frame:
return False
# Check attributes for gene_id (transcript_id is also mandatory
# but not for genes)
attributes = parse_gff_attributes(hdr[8])
if 'gene_id' not in attributes:
return False
found_valid_lines = True
return found_valid_lines
except Exception:
return False
+15 -14
View File
@@ -235,24 +235,22 @@ def is_column_based(fname_or_file_prefix, sep='\t', skip=0):
return False
try:
headers = get_headers(fname_or_file_prefix, sep)
headers = get_headers(fname_or_file_prefix, sep, comment_designator='#')[skip:]
except UnicodeDecodeError:
return False
count = 0
if not headers:
return False
for hdr in headers[skip:]:
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) > 1:
for hdr in headers:
if hdr and hdr != ['']:
if count:
if len(hdr) != count:
return False
else:
count = len(hdr)
break
if count < 2:
return False
for hdr in headers[skip:]:
if hdr and hdr[0] and not hdr[0].startswith('#'):
if len(hdr) != count:
return False
return True
if count < 2:
return False
return count >= 2
def guess_ext(fname, sniff_order, is_binary=False):
@@ -303,13 +301,13 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> guess_ext(fname, sniff_order)
'gff3'
>>> fname = get_test_fname('2.txt')
>>> guess_ext(fname, sniff_order) # 2.txt
>>> guess_ext(fname, sniff_order)
'txt'
>>> fname = get_test_fname('2.tabular')
>>> guess_ext(fname, sniff_order)
'tabular'
>>> fname = get_test_fname('3.txt')
>>> guess_ext(fname, sniff_order) # 3.txt
>>> guess_ext(fname, sniff_order)
'txt'
>>> fname = get_test_fname('test_tab1.tabular')
>>> guess_ext(fname, sniff_order)
@@ -451,6 +449,9 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> fname = get_test_fname('1imzml')
>>> guess_ext(fname, sniff_order) # This test case is ensuring doesn't throw exception, actual value could change if non-utf encoding handling improves.
'data'
>>> fname = get_test_fname('too_many_comments_gff3.tabular')
>>> guess_ext(fname, sniff_order) # It's a VCF but is sniffed as tabular because of the limit on the number of header lines we read
'tabular'
"""
file_prefix = FilePrefix(fname)
file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary)
+6 -7
View File
@@ -743,13 +743,12 @@ class BaseVcf(Tabular):
def set_meta(self, dataset, **kwd):
super().set_meta(dataset, **kwd)
source = open(dataset.file_name)
# Skip comments.
line = None
for line in source:
if not line.startswith('##'):
break
with compression_utils.get_fileobj(dataset.file_name) as fh:
# Skip comments.
for line in fh:
if not line.startswith('##'):
break
if line and line.startswith('#'):
# Found header line, get sample names.
@@ -815,7 +814,7 @@ class VcfGz(BaseVcf, binary.Binary):
return binascii.hexlify(last28) == b'1f8b08040000000000ff0600424302001b0003000000000000000000'
def set_meta(self, dataset, **kwd):
super(BaseVcf, self).set_meta(dataset, **kwd)
super().set_meta(dataset, **kwd)
""" Creates the index for the VCF file. """
# These metadata values are not accessible by users, always overwrite
index_file = dataset.metadata.tabix_index
@@ -0,0 +1,65 @@
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
#
ctgA est match 5410 7503 . - . ID=EST:agt830.3;Target=agt830.3+1+595
ctgA est HSP 7000 7503 . - . Parent=EST:agt830.3;Target=agt830.3+1+504
ctgA est HSP 5410 5500 . ? . Parent=EST:agt830.3;Target=agt830.3+505+595