From 79f2bb59911a9381d8a2f01e3c366f619c693160 Mon Sep 17 00:00:00 2001 From: Wolfgang Maier Date: Wed, 10 Jul 2019 15:46:00 +0200 Subject: [PATCH 1/6] Sniff fastqsanger with telltale low quality scores When a fastq file contains quality scores that are too low to be compatible with any +64 encoding scheme, the assumption should be that the file is Phred+33 encoded even if some scores are above the current upper threshold of 44. --- lib/galaxy/datatypes/sequence.py | 17 ++++++++++++++++- lib/galaxy/datatypes/test/2.fastqsanger | 8 ++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) create mode 100644 lib/galaxy/datatypes/test/2.fastqsanger diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index ffa11e25a2f..03a8e8fb0fa 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -818,9 +818,24 @@ class FastqSanger(Fastq): @staticmethod def quality_check(lines): """Presuming lines are lines from a fastq file, return True if the qualities are compatible with sanger encoding""" + is_ambiguous = True + within_upper_bound = True for line in islice(lines, 3, None, 4): - if not all(_ >= '!' and _ <= 'M' for _ in line[0]) or ' ' in line: + if ' ' in line: return False + if is_ambiguous: + for q in line[0]: + if q < '!' or q > '~': + return False + if q < ';': + is_ambiguous = False + elif q > 'M': + within_upper_bound = False + else: + if any(_ < '!' or _ > '~' for _ in line[0]): + return False + if is_ambiguous and not within_upper_bound: + return False return True diff --git a/lib/galaxy/datatypes/test/2.fastqsanger b/lib/galaxy/datatypes/test/2.fastqsanger new file mode 100644 index 00000000000..ab753439029 --- /dev/null +++ b/lib/galaxy/datatypes/test/2.fastqsanger @@ -0,0 +1,8 @@ +@DCW97JN1:309:C0C42ACXX:4:2206:12976:57510/1 +AAATGGGCATAATATAGATGTAGAGATGTGTTGAATTTATGACTCATTT ++ +ADECJJJJDCCECCCCMCCNCCMCNBCLBJBCLADCDC=CNDJILEDDD +@DCW97JN1:309:C0C42ACXX:4:2211:6915:3569/1 +AAAGAAATTAAATGGGCATAATATAGATGTAGAGATGTGTTGAATTTAT ++ +@DEJAAA?B@ECCEFFGDBBFC8ABGCBK=BI=66@H?E@@JACCCDCCICHIGECCCCLDCHDEI Date: Wed, 10 Jul 2019 16:04:05 +0200 Subject: [PATCH 2/6] Add unit test and fix test file --- lib/galaxy/datatypes/sequence.py | 5 ++++- lib/galaxy/datatypes/test/2.fastqsanger | 2 +- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 03a8e8fb0fa..89249bf5432 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -700,7 +700,7 @@ class BaseFastq(Sequence): >>> Fastq().sniff(fname) True >>> fname = get_test_fname('2.fastq') - >>> Fastq().sniff( fname ) + >>> Fastq().sniff(fname) True >>> FastqSanger().sniff(fname) False @@ -714,6 +714,9 @@ class BaseFastq(Sequence): True >>> FastqCSSanger().sniff(fname) True + >>> fname = get_test_fname('2.fastqsanger') + >>> FastqSanger().sniff(fname) + True """ compressed = file_prefix.compressed_format is not None if compressed and not isinstance(self, Binary): diff --git a/lib/galaxy/datatypes/test/2.fastqsanger b/lib/galaxy/datatypes/test/2.fastqsanger index ab753439029..ff60d92d068 100644 --- a/lib/galaxy/datatypes/test/2.fastqsanger +++ b/lib/galaxy/datatypes/test/2.fastqsanger @@ -5,4 +5,4 @@ ADECJJJJDCCECCCCMCCNCCMCNBCLBJBCLADCDC=CNDJILEDDD @DCW97JN1:309:C0C42ACXX:4:2211:6915:3569/1 AAAGAAATTAAATGGGCATAATATAGATGTAGAGATGTGTTGAATTTAT + -@DEJAAA?B@ECCEFFGDBBFC8ABGCBK=BI=66@H?E@@JACCCDCCICHIGECCCCLDCHDEI Date: Wed, 10 Jul 2019 17:22:14 +0200 Subject: [PATCH 3/6] Avoid previously used test file names --- lib/galaxy/datatypes/sequence.py | 2 +- lib/galaxy/datatypes/test/{2.fastqsanger => 4.fastqsanger} | 0 2 files changed, 1 insertion(+), 1 deletion(-) rename lib/galaxy/datatypes/test/{2.fastqsanger => 4.fastqsanger} (100%) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 89249bf5432..97196fac3da 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -714,7 +714,7 @@ class BaseFastq(Sequence): True >>> FastqCSSanger().sniff(fname) True - >>> fname = get_test_fname('2.fastqsanger') + >>> fname = get_test_fname('4.fastqsanger') >>> FastqSanger().sniff(fname) True """ diff --git a/lib/galaxy/datatypes/test/2.fastqsanger b/lib/galaxy/datatypes/test/4.fastqsanger similarity index 100% rename from lib/galaxy/datatypes/test/2.fastqsanger rename to lib/galaxy/datatypes/test/4.fastqsanger From 37b3a0fe13c9630d26e5ef4b15ffac19e6b2931e Mon Sep 17 00:00:00 2001 From: Wolfgang Maier Date: Wed, 10 Jul 2019 17:37:56 +0200 Subject: [PATCH 4/6] Improve readability --- lib/galaxy/datatypes/sequence.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 97196fac3da..595e365f5e9 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -822,7 +822,7 @@ class FastqSanger(Fastq): def quality_check(lines): """Presuming lines are lines from a fastq file, return True if the qualities are compatible with sanger encoding""" is_ambiguous = True - within_upper_bound = True + within_typical_upper_bound = True for line in islice(lines, 3, None, 4): if ' ' in line: return False @@ -831,13 +831,14 @@ class FastqSanger(Fastq): if q < '!' or q > '~': return False if q < ';': + # encoding is Phred+33 is_ambiguous = False elif q > 'M': - within_upper_bound = False + within_typical_upper_bound = False else: if any(_ < '!' or _ > '~' for _ in line[0]): return False - if is_ambiguous and not within_upper_bound: + if is_ambiguous and not within_typical_upper_bound: return False return True From ff6f1084a7cc514e8a2da280636646c9998eeeca Mon Sep 17 00:00:00 2001 From: Wolfgang Maier Date: Thu, 11 Jul 2019 16:47:25 +0200 Subject: [PATCH 5/6] Reduce to minimal required change --- lib/galaxy/datatypes/sequence.py | 24 ++++-------------------- 1 file changed, 4 insertions(+), 20 deletions(-) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 595e365f5e9..ddc63176491 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -694,6 +694,9 @@ class BaseFastq(Sequence): >>> fname = get_test_fname('1.fastqsanger') >>> FastqSanger().sniff(fname) True + >>> fname = get_test_fname('4.fastqsanger') + >>> FastqSanger().sniff(fname) + True >>> fname = get_test_fname('3.fastq') >>> FastqSanger().sniff(fname) False @@ -714,9 +717,6 @@ class BaseFastq(Sequence): True >>> FastqCSSanger().sniff(fname) True - >>> fname = get_test_fname('4.fastqsanger') - >>> FastqSanger().sniff(fname) - True """ compressed = file_prefix.compressed_format is not None if compressed and not isinstance(self, Binary): @@ -821,25 +821,9 @@ class FastqSanger(Fastq): @staticmethod def quality_check(lines): """Presuming lines are lines from a fastq file, return True if the qualities are compatible with sanger encoding""" - is_ambiguous = True - within_typical_upper_bound = True for line in islice(lines, 3, None, 4): - if ' ' in line: + if not all(_ >= '!' and _ <= 'S' for _ in line[0]) or ' ' in line: return False - if is_ambiguous: - for q in line[0]: - if q < '!' or q > '~': - return False - if q < ';': - # encoding is Phred+33 - is_ambiguous = False - elif q > 'M': - within_typical_upper_bound = False - else: - if any(_ < '!' or _ > '~' for _ in line[0]): - return False - if is_ambiguous and not within_typical_upper_bound: - return False return True From 80b9be3cf51d132f1156a80083251187fd641c46 Mon Sep 17 00:00:00 2001 From: Wolfgang Maier Date: Thu, 11 Jul 2019 21:14:05 +0200 Subject: [PATCH 6/6] Clean up code --- lib/galaxy/datatypes/sequence.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index ddc63176491..28fbe4754be 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -822,7 +822,7 @@ class FastqSanger(Fastq): def quality_check(lines): """Presuming lines are lines from a fastq file, return True if the qualities are compatible with sanger encoding""" for line in islice(lines, 3, None, 4): - if not all(_ >= '!' and _ <= 'S' for _ in line[0]) or ' ' in line: + if not all(_ >= '!' and _ <= 'S' for _ in line[0]): return False return True