From 9c10b6ce5de83494eb4ac6ad66d85b780cb5f4b5 Mon Sep 17 00:00:00 2001 From: Devon Ryan Date: Tue, 27 Jun 2017 10:53:31 +0200 Subject: [PATCH 1/4] For file sniffing, allow get_headers() to skip comment lines, which should prevent at least some instances of #3148. --- lib/galaxy/datatypes/interval.py | 8 ++++---- lib/galaxy/datatypes/sniff.py | 16 +++++++++++++--- 2 files changed, 17 insertions(+), 7 deletions(-) diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 1992479e8bb..424faa8efe0 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -314,14 +314,14 @@ class Interval( Tabular ): >>> Interval().sniff( fname ) True """ - headers = get_headers( filename, '\t' ) + headers = get_headers( filename, '\t', commentDesignator='#' ) try: """ If we got here, we already know the file is_column_based and is not bed, so we'll just look for some valid data. """ for hdr in headers: - if hdr and not hdr[0].startswith( '#' ): + if hdr: if len(hdr) < 3: return False try: @@ -504,12 +504,12 @@ class Bed( Interval ): >>> Bed().sniff( fname ) True """ - headers = get_headers( filename, '\t' ) + headers = get_headers( filename, '\t', commentDesignator='#' ) try: if not headers: return False for hdr in headers: - if (hdr[0] == '' or hdr[0].startswith( '#' )): + if (hdr[0] == ''): continue valid_col1 = False if len(hdr) < 3 or len(hdr) > 12: diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index b2a82e9f8bf..0692fa3bef9 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -198,24 +198,34 @@ def convert_newlines_sep2tabs( fname, in_place=True, patt="\\s+", tmp_dir=None, return ( i + 1, temp_name ) -def get_headers( fname, sep, count=60, is_multi_byte=False ): +def get_headers( fname, sep, count=60, is_multi_byte=False, commentDesignator=None ): """ - Returns a list with the first 'count' lines split by 'sep' + Returns a list with the first 'count' lines split by 'sep', ignoring lines + starting with 'commentDesignator' >>> fname = get_test_fname('complete.bed') >>> get_headers(fname,'\\t') [['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']] + >>> fname = get_test_fname('test.gff') + >>> get_headers(fname, '\\t', count=5, commentDesignator='#') + [[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']] """ headers = [] in_file = compression_utils.get_fileobj(fname) try: - for idx, line in enumerate(in_file): + idx = 0 + for line in in_file: line = line.rstrip('\n\r') if is_multi_byte: # TODO: fix this - sep is never found in line line = unicodify( line, 'utf-8' ) sep = sep.encode( 'utf-8' ) + if commentDesignator is not None: + commentDesignator = commentDesignator.encode( 'utf-8' ) + if commentLine is not None and line.startswith( commentDesignator ): + continue headers.append( line.split(sep) ) + idx += 1 if idx == count: break finally: From 3bb3ecb09bb0ac77668d08cb1d5ecc0b7d61fa29 Mon Sep 17 00:00:00 2001 From: Devon Ryan Date: Tue, 27 Jun 2017 11:31:35 +0200 Subject: [PATCH 2/4] Typo --- lib/galaxy/datatypes/sniff.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 0692fa3bef9..d7a0af6be25 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -222,7 +222,7 @@ def get_headers( fname, sep, count=60, is_multi_byte=False, commentDesignator=No sep = sep.encode( 'utf-8' ) if commentDesignator is not None: commentDesignator = commentDesignator.encode( 'utf-8' ) - if commentLine is not None and line.startswith( commentDesignator ): + if commentDesignator is not None and line.startswith( commentDesignator ): continue headers.append( line.split(sep) ) idx += 1 From e6278a778d3e2b1e02c6cba49516a9f0ebdf5e5b Mon Sep 17 00:00:00 2001 From: Devon Ryan Date: Wed, 28 Jun 2017 00:27:44 +0200 Subject: [PATCH 3/4] Remove camel case --- lib/galaxy/datatypes/interval.py | 4 ++-- lib/galaxy/datatypes/sniff.py | 12 ++++++------ 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 424faa8efe0..4ffb9cd8829 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -314,7 +314,7 @@ class Interval( Tabular ): >>> Interval().sniff( fname ) True """ - headers = get_headers( filename, '\t', commentDesignator='#' ) + headers = get_headers( filename, '\t', comment_designator='#' ) try: """ If we got here, we already know the file is_column_based and is not bed, @@ -504,7 +504,7 @@ class Bed( Interval ): >>> Bed().sniff( fname ) True """ - headers = get_headers( filename, '\t', commentDesignator='#' ) + headers = get_headers( filename, '\t', comment_designator='#' ) try: if not headers: return False diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index d7a0af6be25..b1cb96e565e 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -198,16 +198,16 @@ def convert_newlines_sep2tabs( fname, in_place=True, patt="\\s+", tmp_dir=None, return ( i + 1, temp_name ) -def get_headers( fname, sep, count=60, is_multi_byte=False, commentDesignator=None ): +def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=None ): """ Returns a list with the first 'count' lines split by 'sep', ignoring lines - starting with 'commentDesignator' + starting with 'comment_designator' >>> fname = get_test_fname('complete.bed') >>> get_headers(fname,'\\t') [['chr7', '127475281', '127491632', 'NM_000230', '0', '+', '127486022', '127488767', '0', '3', '29,172,3225,', '0,10713,13126,'], ['chr7', '127486011', '127488900', 'D49487', '0', '+', '127486022', '127488767', '0', '2', '155,490,', '0,2399']] >>> fname = get_test_fname('test.gff') - >>> get_headers(fname, '\\t', count=5, commentDesignator='#') + >>> get_headers(fname, '\\t', count=5, comment_designator='#') [[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']] """ headers = [] @@ -220,9 +220,9 @@ def get_headers( fname, sep, count=60, is_multi_byte=False, commentDesignator=No # TODO: fix this - sep is never found in line line = unicodify( line, 'utf-8' ) sep = sep.encode( 'utf-8' ) - if commentDesignator is not None: - commentDesignator = commentDesignator.encode( 'utf-8' ) - if commentDesignator is not None and line.startswith( commentDesignator ): + if comment_designator is not None: + comment_designator = comment_designator.encode( 'utf-8' ) + if comment_designator is not None and line.startswith( comment_designator ): continue headers.append( line.split(sep) ) idx += 1 From e5e83833d13d54b695952d73279b011c3b996af8 Mon Sep 17 00:00:00 2001 From: Devon Ryan Date: Wed, 28 Jun 2017 07:19:16 +0200 Subject: [PATCH 4/4] Style and ignore '' comment_designators --- lib/galaxy/datatypes/interval.py | 2 +- lib/galaxy/datatypes/sniff.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 4ffb9cd8829..578c02e748a 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -509,7 +509,7 @@ class Bed( Interval ): if not headers: return False for hdr in headers: - if (hdr[0] == ''): + if hdr[0] == '': continue valid_col1 = False if len(hdr) < 3 or len(hdr) > 12: diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index b1cb96e565e..2fa4f42c7c8 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -220,9 +220,9 @@ def get_headers( fname, sep, count=60, is_multi_byte=False, comment_designator=N # TODO: fix this - sep is never found in line line = unicodify( line, 'utf-8' ) sep = sep.encode( 'utf-8' ) - if comment_designator is not None: + if comment_designator is not None and comment_designator != '': comment_designator = comment_designator.encode( 'utf-8' ) - if comment_designator is not None and line.startswith( comment_designator ): + if comment_designator is not None and comment_designator != '' and line.startswith( comment_designator ): continue headers.append( line.split(sep) ) idx += 1