diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 89d771b42e6..20728b5d3d8 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -477,9 +477,9 @@ class Bed(Interval): if not get_headers(file_prefix, '\t', comment_designator='#', count=1): return False try: - headers = iter_headers(file_prefix, '\t', comment_designator='#') - for hdr in headers: - if hdr[0] == '': + found_valid_lines = False + for hdr in iter_headers(file_prefix, '\t', comment_designator='#'): + if not hdr or hdr == ['']: continue if len(hdr) < 3 or len(hdr) > 12: return False @@ -542,7 +542,8 @@ class Bed(Interval): return False if len(block_sizes) != block_count or len(block_starts) != block_count: return False - return True + found_valid_lines = True + return found_valid_lines except Exception: return False @@ -818,28 +819,33 @@ class Gff(Tabular, _RemoteCallMixin): if len(get_headers(file_prefix, '\t', count=2)) < 2: return False try: - headers = iter_headers(file_prefix, '\t') - for hdr in headers: - if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0: + found_valid_lines = False + for hdr in iter_headers(file_prefix, '\t'): + if not hdr or hdr == ['']: + continue + if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0: return False - if hdr and hdr[0] and not hdr[0].startswith('#'): - if len(hdr) != 9: - return False + # The gff-version header comment may have been stripped, so inspect the data + if hdr[0].startswith('#'): + continue + if len(hdr) != 9: + return False + try: + int(hdr[3]) + int(hdr[4]) + except Exception: + return False + if hdr[5] != '.': try: - int(hdr[3]) - int(hdr[4]) + float(hdr[5]) except Exception: return False - if hdr[5] != '.': - try: - float(hdr[5]) - except Exception: - return False - if hdr[6] not in data.valid_strand: - return False - if hdr[7] not in self.valid_gff_frame: - return False - return True + if hdr[6] not in data.valid_strand: + return False + if hdr[7] not in self.valid_gff_frame: + return False + found_valid_lines = True + return found_valid_lines except Exception: return False @@ -953,37 +959,41 @@ class Gff3(Gff): if len(get_headers(file_prefix, '\t', count=2)) < 2: return False try: - headers = iter_headers(file_prefix, '\t') - for hdr in headers: - if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0: + found_valid_lines = False + for hdr in iter_headers(file_prefix, '\t'): + if not hdr or hdr == ['']: + continue + if hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0: return True - elif hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0: + elif hdr[0].startswith('##gff-version') and hdr[0].find('3') < 0: return False - # Header comments may have been stripped, so inspect the data - if hdr and hdr[0] and not hdr[0].startswith('#'): - if len(hdr) != 9: + # The gff-version header comment may have been stripped, so inspect the data + if hdr[0].startswith('#'): + continue + if len(hdr) != 9: + return False + try: + int(hdr[3]) + except Exception: + if hdr[3] != '.': return False + try: + int(hdr[4]) + except Exception: + if hdr[4] != '.': + return False + if hdr[5] != '.': try: - int(hdr[3]) + float(hdr[5]) except Exception: - if hdr[3] != '.': - return False - try: - int(hdr[4]) - except Exception: - if hdr[4] != '.': - return False - if hdr[5] != '.': - try: - float(hdr[5]) - except Exception: - return False - if hdr[6] not in self.valid_gff3_strand: return False - if hdr[7] not in self.valid_gff3_phase: - return False - parse_gff3_attributes(hdr[8]) - return True + if hdr[6] not in self.valid_gff3_strand: + return False + if hdr[7] not in self.valid_gff3_phase: + return False + parse_gff3_attributes(hdr[8]) + found_valid_lines = True + return found_valid_lines except Exception: return False @@ -1031,34 +1041,38 @@ class Gtf(Gff): if len(get_headers(file_prefix, '\t', count=2)) < 2: return False try: - headers = iter_headers(file_prefix, '\t') - for hdr in headers: - if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0: + found_valid_lines = False + for hdr in iter_headers(file_prefix, '\t'): + if not hdr or hdr == ['']: + continue + if hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0: return False - if hdr and hdr[0] and not hdr[0].startswith('#'): - if len(hdr) != 9: - return False + # The gff-version header comment may have been stripped, so inspect the data + if hdr[0].startswith('#'): + continue + if len(hdr) != 9: + return False + try: + int(hdr[3]) + int(hdr[4]) + except Exception: + return False + if hdr[5] != '.': try: - int(hdr[3]) - int(hdr[4]) + float(hdr[5]) except Exception: return False - if hdr[5] != '.': - try: - float(hdr[5]) - except Exception: - return False - if hdr[6] not in data.valid_strand: - return False - if hdr[7] not in self.valid_gff_frame: - return False - - # Check attributes for gene_id (transcript_id is also mandatory - # but not for genes) - attributes = parse_gff_attributes(hdr[8]) - if 'gene_id' not in attributes: - return False - return True + if hdr[6] not in data.valid_strand: + return False + if hdr[7] not in self.valid_gff_frame: + return False + # Check attributes for gene_id (transcript_id is also mandatory + # but not for genes) + attributes = parse_gff_attributes(hdr[8]) + if 'gene_id' not in attributes: + return False + found_valid_lines = True + return found_valid_lines except Exception: return False diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 272042ece26..a63d5340bb9 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -235,24 +235,22 @@ def is_column_based(fname_or_file_prefix, sep='\t', skip=0): return False try: - headers = get_headers(fname_or_file_prefix, sep) + headers = get_headers(fname_or_file_prefix, sep, comment_designator='#')[skip:] except UnicodeDecodeError: return False count = 0 if not headers: return False - for hdr in headers[skip:]: - if hdr and hdr[0] and not hdr[0].startswith('#'): - if len(hdr) > 1: + for hdr in headers: + if hdr and hdr != ['']: + if count: + if len(hdr) != count: + return False + else: count = len(hdr) - break - if count < 2: - return False - for hdr in headers[skip:]: - if hdr and hdr[0] and not hdr[0].startswith('#'): - if len(hdr) != count: - return False - return True + if count < 2: + return False + return count >= 2 def guess_ext(fname, sniff_order, is_binary=False): @@ -303,13 +301,13 @@ def guess_ext(fname, sniff_order, is_binary=False): >>> guess_ext(fname, sniff_order) 'gff3' >>> fname = get_test_fname('2.txt') - >>> guess_ext(fname, sniff_order) # 2.txt + >>> guess_ext(fname, sniff_order) 'txt' >>> fname = get_test_fname('2.tabular') >>> guess_ext(fname, sniff_order) 'tabular' >>> fname = get_test_fname('3.txt') - >>> guess_ext(fname, sniff_order) # 3.txt + >>> guess_ext(fname, sniff_order) 'txt' >>> fname = get_test_fname('test_tab1.tabular') >>> guess_ext(fname, sniff_order) @@ -451,6 +449,9 @@ def guess_ext(fname, sniff_order, is_binary=False): >>> fname = get_test_fname('1imzml') >>> guess_ext(fname, sniff_order) # This test case is ensuring doesn't throw exception, actual value could change if non-utf encoding handling improves. 'data' + >>> fname = get_test_fname('too_many_comments_gff3.tabular') + >>> guess_ext(fname, sniff_order) # It's a VCF but is sniffed as tabular because of the limit on the number of header lines we read + 'tabular' """ file_prefix = FilePrefix(fname) file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary) diff --git a/lib/galaxy/datatypes/test/too_many_comments_gff3.tabular b/lib/galaxy/datatypes/test/too_many_comments_gff3.tabular new file mode 100644 index 00000000000..e3fb5113dc6 --- /dev/null +++ b/lib/galaxy/datatypes/test/too_many_comments_gff3.tabular @@ -0,0 +1,65 @@ +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +# +ctgA est match 5410 7503 . - . ID=EST:agt830.3;Target=agt830.3+1+595 +ctgA est HSP 7000 7503 . - . Parent=EST:agt830.3;Target=agt830.3+1+504 +ctgA est HSP 5410 5500 . ? . Parent=EST:agt830.3;Target=agt830.3+505+595