From f5af9539b9d50e0c0e99ade18eea8af6f7beb0e4 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 10 Dec 2017 15:02:31 +0100 Subject: [PATCH 01/24] add BamNative datatype --- lib/galaxy/datatypes/binary.py | 190 +++++++++++++++++++-------------- 1 file changed, 108 insertions(+), 82 deletions(-) diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index daefe7a7e97..4ed2a390e09 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -187,14 +187,11 @@ class GenericAsn1Binary(Binary): edam_data = "data_0849" -@dataproviders.decorators.has_dataproviders -class Bam(Binary): - """Class describing a BAM binary file""" +class BamNative(Binary): + """Class describing a BAM binary file that is not necessarily sorted""" edam_format = "format_2572" edam_data = "data_0863" - file_ext = "bam" - track_type = "ReadTrack" - data_sources = {"data": "bai", "index": "bigwig"} + file_ext = "bam_native" MetadataElement(name="bam_index", desc="BAM Index File", param=metadata.FileParameter, file_ext="bai", readonly=True, no_value=None, visible=False, optional=True) MetadataElement(name="bam_version", default=None, desc="BAM Version", param=MetadataParameter, readonly=True, visible=False, optional=True, no_value=None) @@ -210,13 +207,117 @@ class Bam(Binary): @staticmethod def merge(split_files, output_file): """ - Merges BAM files + Merges Bam files :param split_files: List of bam file paths to merge :param output_file: Write merged bam file to this location """ pysam.merge('-O', 'BAM', output_file, *split_files) + def init_meta(self, dataset, copy_from=None): + Binary.init_meta(self, dataset, copy_from=copy_from) + + def set_meta(self, dataset, overwrite=True, **kwd): + try: + bam_file = pysam.AlignmentFile(dataset.file_name, mode='rb') + # TODO: Reference names, lengths, read_groups and headers can become very large, truncate when necessary + dataset.metadata.reference_names = list(bam_file.references) + dataset.metadata.reference_lengths = list(bam_file.lengths) + dataset.metadata.bam_header = bam_file.header + dataset.metadata.read_groups = [read_group['ID'] for read_group in dataset.metadata.bam_header.get('RG', []) if 'ID' in read_group] + dataset.metadata.sort_order = bam_file.header.get('HD', {}).get('SO', None) + dataset.metadata.bam_version = bam_file.header.get('HD', {}).get('VN', None) + except Exception: + # Per Dan, don't log here because doing so will cause datasets that + # fail metadata to end in the error state + pass + + def set_peek(self, dataset, is_multi_byte=False): + if not dataset.dataset.purged: + dataset.peek = "Binary bam alignments file" + dataset.blurb = nice_size(dataset.get_size()) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' + + def display_peek(self, dataset): + try: + return dataset.peek + except Exception: + return "Binary bam alignments file (%s)" % (nice_size(dataset.get_size())) + + def to_archive(self, trans, dataset, name=""): + rel_paths = [] + file_paths = [] + rel_paths.append("%s.%s" % (name or dataset.file_name, dataset.extension)) + file_paths.append(dataset.file_name) + rel_paths.append("%s.%s.bai" % (name or dataset.file_name, dataset.extension)) + file_paths.append(dataset.metadata.bam_index.file_name) + return zip(file_paths, rel_paths) + + def get_chunk(self, trans, dataset, offset=0, ck_size=None): + index_file = dataset.metadata.bam_index + if index_file: + index_filename = index_file.file_name + else: + index_filename = None + with pysam.AlignmentFile(dataset.file_name, "rb", index_filename=index_filename) as bamfile: + ck_size = 300 # 300 lines + ck_data = "" + header_line_count = 0 + if offset == 0: + ck_data = bamfile.text.replace('\t', ' ') + header_line_count = bamfile.text.count('\n') + else: + bamfile.seek(offset) + for line_number, alignment in enumerate(bamfile) : + # return only Header lines if 'header_line_count' exceeds 'ck_size' + # FIXME: Can be problematic if bam has million lines of header + offset = bamfile.tell() + if (line_number + header_line_count) > ck_size: + break + else: + bamline = alignment.tostring(bamfile) + # Galaxy display each tag as separate column because 'tostring()' funcition put tabs in between each tag of tags column. + # Below code will remove spaces between each tag. + bamline_modified = ('\t').join(bamline.split()[:11] + [(' ').join(bamline.split()[11:])]) + ck_data = "%s\n%s" % (ck_data, bamline_modified) + return dumps({'ck_data': util.unicodify(ck_data), + 'offset': offset}) + + def display_data(self, trans, dataset, preview=False, filename=None, to_ext=None, offset=None, ck_size=None, **kwd): + preview = util.string_as_bool(preview) + if offset is not None: + return self.get_chunk(trans, dataset, offset, ck_size) + elif to_ext or not preview: + return super(Bam, self).display_data(trans, dataset, preview, filename, to_ext, **kwd) + else: + column_names = dataset.metadata.column_names + if not column_names: + column_names = [] + column_types = dataset.metadata.column_types + if not column_types: + column_types = [] + column_number = dataset.metadata.columns + if column_number is None: + column_number = 1 + return trans.fill_template("/dataset/tabular_chunked.mako", + dataset=dataset, + chunk=self.get_chunk(trans, dataset, 0), + column_number=column_number, + column_names=column_names, + column_types=column_types) + + +@dataproviders.decorators.has_dataproviders +class Bam(BamNative): + """Class describing a BAM binary file""" + edam_format = "format_2572" + edam_data = "data_0863" + file_ext = "bam" + track_type = "ReadTrack" + data_sources = {"data": "bai", "index": "bigwig"} + def dataset_content_needs_grooming(self, file_name): """ Check if file_name is a coordinate-sorted BAM file @@ -266,9 +367,6 @@ class Bam(Binary): # Remove temp file and empty temporary directory os.rmdir(tmp_dir) - def init_meta(self, dataset, copy_from=None): - Binary.init_meta(self, dataset, copy_from=copy_from) - def set_meta(self, dataset, overwrite=True, **kwd): # These metadata values are not accessible by users, always overwrite index_file = dataset.metadata.bam_index @@ -302,78 +400,6 @@ class Bam(Binary): except Exception: return False - def set_peek(self, dataset, is_multi_byte=False): - if not dataset.dataset.purged: - dataset.peek = "Binary bam alignments file" - dataset.blurb = nice_size(dataset.get_size()) - else: - dataset.peek = 'file does not exist' - dataset.blurb = 'file purged from disk' - - def display_peek(self, dataset): - try: - return dataset.peek - except Exception: - return "Binary bam alignments file (%s)" % (nice_size(dataset.get_size())) - - def to_archive(self, trans, dataset, name=""): - rel_paths = [] - file_paths = [] - rel_paths.append("%s.%s" % (name or dataset.file_name, dataset.extension)) - file_paths.append(dataset.file_name) - rel_paths.append("%s.%s.bai" % (name or dataset.file_name, dataset.extension)) - file_paths.append(dataset.metadata.bam_index.file_name) - return zip(file_paths, rel_paths) - - def get_chunk(self, trans, dataset, offset=0, ck_size=None): - index_file = dataset.metadata.bam_index - with pysam.AlignmentFile(dataset.file_name, "rb", index_filename=index_file.file_name) as bamfile: - ck_size = 300 # 300 lines - ck_data = "" - header_line_count = 0 - if offset == 0: - ck_data = bamfile.text.replace('\t', ' ') - header_line_count = bamfile.text.count('\n') - else: - bamfile.seek(offset) - for line_number, alignment in enumerate(bamfile) : - # return only Header lines if 'header_line_count' exceeds 'ck_size' - # FIXME: Can be problematic if bam has million lines of header - offset = bamfile.tell() - if (line_number + header_line_count) > ck_size: - break - else: - bamline = alignment.tostring(bamfile) - # Galaxy display each tag as separate column because 'tostring()' funcition put tabs in between each tag of tags column. - # Below code will remove spaces between each tag. - bamline_modified = ('\t').join(bamline.split()[:11] + [(' ').join(bamline.split()[11:])]) - ck_data = "%s\n%s" % (ck_data, bamline_modified) - return dumps({'ck_data': util.unicodify(ck_data), - 'offset': offset}) - - def display_data(self, trans, dataset, preview=False, filename=None, to_ext=None, offset=None, ck_size=None, **kwd): - preview = util.string_as_bool(preview) - if offset is not None: - return self.get_chunk(trans, dataset, offset, ck_size) - elif to_ext or not preview: - return super(Bam, self).display_data(trans, dataset, preview, filename, to_ext, **kwd) - else: - column_names = dataset.metadata.column_names - if not column_names: - column_names = [] - column_types = dataset.metadata.column_types - if not column_types: - column_types = [] - column_number = dataset.metadata.columns - if column_number is None: - column_number = 1 - return trans.fill_template("/dataset/tabular_chunked.mako", - dataset=dataset, - chunk=self.get_chunk(trans, dataset, 0), - column_number=column_number, - column_names=column_names, - column_types=column_types) - # ------------- Dataproviders # pipe through samtools view # ALSO: (as Sam) From 017dad025db9a06c9c6d2430203f52fdec959f68 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 10 Dec 2017 15:03:05 +0100 Subject: [PATCH 02/24] add new test file --- test-data/bam_native_from_sam.bam | Bin 0 -> 484 bytes 1 file changed, 0 insertions(+), 0 deletions(-) create mode 100644 test-data/bam_native_from_sam.bam diff --git a/test-data/bam_native_from_sam.bam b/test-data/bam_native_from_sam.bam new file mode 100644 index 0000000000000000000000000000000000000000..0052a80bd03036c571776cb753de60e344fce87e GIT binary patch literal 484 zcmb2|=3rp}f&Xj_PR>jWz6`}hUs6NT5)ukH_@3~5+pOi~`}oVp%|?b#oj4>KU3qrQ zXgC$oAfa(&0r!$f#T;ceUIk``y2tl2Jb;GCqnTF8SUS=yR+w!%7-Cc8cywVDi z5;$Y{a{6@@mp@J#B1_#Jzr@&7o~b&u@&7^lCs#MytDIXF9BcOSfVbIZ>+@&cSvRH5 znKb8J@9h>oi!FKXG2lua!la8DY zJMpN?h&ODt4y(n6s(s%Roqn2}S6$TZm3DVmg{jpmO_?7pwaJ@4KmDPYvZcdQ@T4m1 zk{AyWrLek?kYqFOWzr=aht)V{95xQ#zGLz}?J)UO{_QXKEm6OuC?8d;e9yUOs)uTs zR_^hz*=q6Y9vx@iE@1a>|EfOQbyp6*57BN~*e&$hVLQ()QwMjm44&ds(#fUQdcJZioQ^!Uj2ScJr5S*N GPyhf^LCc%~ literal 0 HcmV?d00001 From eea55a9258342e8de32162c9e62563de0f6438b8 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 10 Dec 2017 15:03:47 +0100 Subject: [PATCH 03/24] add new test tool --- test/functional/tools/sam_to_bam_native.xml | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 test/functional/tools/sam_to_bam_native.xml diff --git a/test/functional/tools/sam_to_bam_native.xml b/test/functional/tools/sam_to_bam_native.xml new file mode 100644 index 00000000000..578b40f3549 --- /dev/null +++ b/test/functional/tools/sam_to_bam_native.xml @@ -0,0 +1,20 @@ + + '$outfile1' + ]]> + + + + + + + + + + + + + + + + From 2a093f25f6fdb6ebf4a8bba84050130c4929e6c3 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 10 Dec 2017 15:05:28 +0100 Subject: [PATCH 04/24] add new bam_native datatype to the registry --- config/datatypes_conf.xml.sample | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/config/datatypes_conf.xml.sample b/config/datatypes_conf.xml.sample index 6782d490ae0..4045ac713fc 100644 --- a/config/datatypes_conf.xml.sample +++ b/config/datatypes_conf.xml.sample @@ -16,6 +16,10 @@ + + + + @@ -285,6 +289,7 @@ + From 4d7bc9a778afa171bd7498004733943543af5730 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 10 Dec 2017 15:06:28 +0100 Subject: [PATCH 05/24] bam to bigwig converter, converts also bam_native --- lib/galaxy/datatypes/converters/bam_to_bigwig_converter.xml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/converters/bam_to_bigwig_converter.xml b/lib/galaxy/datatypes/converters/bam_to_bigwig_converter.xml index 6307bcf0d64..50b8c1ca14c 100644 --- a/lib/galaxy/datatypes/converters/bam_to_bigwig_converter.xml +++ b/lib/galaxy/datatypes/converters/bam_to_bigwig_converter.xml @@ -17,7 +17,7 @@ > temp.bg && bedGraphToBigWig temp.bg '$chromInfo' '$output']]> - + From 2569851498f76d2fc416bd96ab4007c579bb0ffd Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 10 Dec 2017 15:06:53 +0100 Subject: [PATCH 06/24] use unsorted sam file for tests --- test-data/sam_with_header.sam | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/test-data/sam_with_header.sam b/test-data/sam_with_header.sam index 33449b176bc..f2428278d55 100644 --- a/test-data/sam_with_header.sam +++ b/test-data/sam_with_header.sam @@ -1,14 +1,14 @@ @SQ SN:ref LN:45 @SQ SN:ref2 LN:40 +r003 16 ref 29 30 6H5M * 0 0 TAGGC * +r001 83 ref 37 30 9M = 7 -39 CAGCGCCAT * +x2 0 ref2 2 30 21M * 0 0 ggttttataaaacaaataatt ????????????????????? r001 163 ref 7 30 8M4I4M1D3M = 37 39 TTAGATAAAGAGGATACTG * XX:B:S,12561,2,20,112 r002 0 ref 9 30 1S2I6M1P1I1P1I4M2I * 0 0 AAAAGATAAGGGATAAA * r003 0 ref 9 30 5H6M * 0 0 AGCTAA * r004 0 ref 16 30 6M14N1I5M * 0 0 ATAGCTCTCAGC * -r003 16 ref 29 30 6H5M * 0 0 TAGGC * -r001 83 ref 37 30 9M = 7 -39 CAGCGCCAT * -x1 0 ref2 1 30 20M * 0 0 aggttttataaaacaaataa ???????????????????? -x2 0 ref2 2 30 21M * 0 0 ggttttataaaacaaataatt ????????????????????? x3 0 ref2 6 30 9M4I13M * 0 0 ttataaaacAAATaattaagtctaca ?????????????????????????? x4 0 ref2 10 30 25M * 0 0 CaaaTaattaagtctacagagcaac ????????????????????????? x5 0 ref2 12 30 24M * 0 0 aaTaattaagtctacagagcaact ???????????????????????? +x1 0 ref2 1 30 20M * 0 0 aggttttataaaacaaataa ???????????????????? x6 0 ref2 14 30 23M * 0 0 Taattaagtctacagagcaacta ??????????????????????? From 8b55e0a33a52fe45fc5fdaa171607c7f253021c9 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 10 Dec 2017 15:28:39 +0100 Subject: [PATCH 07/24] add two converters --- .../bam_native_to_bam_converter.xml | 23 +++++++++++++++++++ .../converters/sam_to_bam_native.xml | 22 ++++++++++++++++++ 2 files changed, 45 insertions(+) create mode 100644 lib/galaxy/datatypes/converters/bam_native_to_bam_converter.xml create mode 100644 lib/galaxy/datatypes/converters/sam_to_bam_native.xml diff --git a/lib/galaxy/datatypes/converters/bam_native_to_bam_converter.xml b/lib/galaxy/datatypes/converters/bam_native_to_bam_converter.xml new file mode 100644 index 00000000000..8ac1dd7b633 --- /dev/null +++ b/lib/galaxy/datatypes/converters/bam_native_to_bam_converter.xml @@ -0,0 +1,23 @@ + diff --git a/lib/galaxy/datatypes/converters/sam_to_bam_native.xml b/lib/galaxy/datatypes/converters/sam_to_bam_native.xml new file mode 100644 index 00000000000..363a3b57b68 --- /dev/null +++ b/lib/galaxy/datatypes/converters/sam_to_bam_native.xml @@ -0,0 +1,22 @@ + + + + samtools + + + + + + + + + + + + From 640cfb1e7d8be8ceaba96920cfb1be72c48f4974 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Thu, 4 Jan 2018 00:12:13 +0100 Subject: [PATCH 08/24] Bam to BAM --- lib/galaxy/datatypes/binary.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index 4ed2a390e09..d73b3c9ad63 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -207,7 +207,7 @@ class BamNative(Binary): @staticmethod def merge(split_files, output_file): """ - Merges Bam files + Merges BAM files :param split_files: List of bam file paths to merge :param output_file: Write merged bam file to this location From e519f4e7efcbb055a66a861588ceabd09e0b9b58 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Thu, 4 Jan 2018 00:27:22 +0100 Subject: [PATCH 09/24] rearrange sniffer --- lib/galaxy/datatypes/binary.py | 30 ++++++++++++++---------------- 1 file changed, 14 insertions(+), 16 deletions(-) diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index d73b3c9ad63..06f0a4f76e5 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -217,6 +217,17 @@ class BamNative(Binary): def init_meta(self, dataset, copy_from=None): Binary.init_meta(self, dataset, copy_from=copy_from) + def sniff(self, filename): + # BAM is compressed in the BGZF format, and must not be uncompressed in Galaxy. + # The first 4 bytes of any bam file is 'BAM\1', and the file is binary. + try: + header = gzip.open(filename).read(4) + if header == b'BAM\1': + return True + return False + except Exception: + return False + def set_meta(self, dataset, overwrite=True, **kwd): try: bam_file = pysam.AlignmentFile(dataset.file_name, mode='rb') @@ -256,12 +267,7 @@ class BamNative(Binary): return zip(file_paths, rel_paths) def get_chunk(self, trans, dataset, offset=0, ck_size=None): - index_file = dataset.metadata.bam_index - if index_file: - index_filename = index_file.file_name - else: - index_filename = None - with pysam.AlignmentFile(dataset.file_name, "rb", index_filename=index_filename) as bamfile: + with pysam.AlignmentFile(dataset.file_name, "rb") as bamfile: ck_size = 300 # 300 lines ck_data = "" header_line_count = 0 @@ -389,16 +395,8 @@ class Bam(BamNative): # fail metadata to end in the error state pass - def sniff(self, filename): - # BAM is compressed in the BGZF format, and must not be uncompressed in Galaxy. - # The first 4 bytes of any bam file is 'BAM\1', and the file is binary. - try: - header = gzip.open(filename).read(4) - if header == b'BAM\1': - return True - return False - except Exception: - return False + def sniff(self, file_name): + return super(Bam, self).sniff(file_name) and not self.dataset_content_needs_grooming(file_name) # ------------- Dataproviders # pipe through samtools view From 295e5712ee2c020292754ddcb75d6d5744c4ed10 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Thu, 4 Jan 2018 00:45:02 +0100 Subject: [PATCH 10/24] adjust tests --- test/functional/tools/sam_to_bam_native.xml | 15 +++++++++++---- test/functional/tools/samples_tool_conf.xml | 1 + 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/test/functional/tools/sam_to_bam_native.xml b/test/functional/tools/sam_to_bam_native.xml index 578b40f3549..81222e2f309 100644 --- a/test/functional/tools/sam_to_bam_native.xml +++ b/test/functional/tools/sam_to_bam_native.xml @@ -1,18 +1,25 @@ + + samtools + '$outfile1' + samtools view + -b + -@ \${GALAXY_SLOTS:-2} + -o '${output}' + '$input' ]]> - + - + - + diff --git a/test/functional/tools/samples_tool_conf.xml b/test/functional/tools/samples_tool_conf.xml index 5bd62a7cbb8..c467eb82ee1 100644 --- a/test/functional/tools/samples_tool_conf.xml +++ b/test/functional/tools/samples_tool_conf.xml @@ -81,6 +81,7 @@ + From 2b5b6cf74fb29eb4608318477c062982d8ad955c Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Thu, 4 Jan 2018 13:35:56 +0200 Subject: [PATCH 11/24] Fix test and add datatypes to test/functional/tools/sample_datatypes_conf.xml This test is still failing because `CONVERTER_sam_to_bam` will be invoked instead of CONVERTER_sam_to_bam_native. Possibly becasue Bam inherits from BamNative, so this isn't strictly speaking wrong, though not what one would expect. --- test/functional/tools/sam_to_bam_native.xml | 17 ++++++++++++++--- test/functional/tools/sample_datatypes_conf.xml | 10 +++++++++- 2 files changed, 23 insertions(+), 4 deletions(-) diff --git a/test/functional/tools/sam_to_bam_native.xml b/test/functional/tools/sam_to_bam_native.xml index 81222e2f309..df3a728d4a0 100644 --- a/test/functional/tools/sam_to_bam_native.xml +++ b/test/functional/tools/sam_to_bam_native.xml @@ -3,23 +3,34 @@ samtools - + + + - + + + + + + diff --git a/test/functional/tools/sample_datatypes_conf.xml b/test/functional/tools/sample_datatypes_conf.xml index 8b366098630..98991439526 100644 --- a/test/functional/tools/sample_datatypes_conf.xml +++ b/test/functional/tools/sample_datatypes_conf.xml @@ -14,8 +14,16 @@ - + + + + + + + + + From a8b3d23f1438f31436b86d24b48868fd3e133dad Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Thu, 4 Jan 2018 14:11:11 +0200 Subject: [PATCH 12/24] Adjust sniffer test for unsorted bam --- lib/galaxy/datatypes/sniff.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 48ac2d46e6e..1d3681f97c4 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -337,7 +337,7 @@ def guess_ext(fname, sniff_order): 'bam' >>> fname = get_test_fname('3unsorted.bam') >>> guess_ext(fname, sniff_order) - 'bam' + 'bam_native' >>> fname = get_test_fname('test.idpDB') >>> guess_ext(fname, sniff_order) 'idpdb' From 446643159540f8db67dfd0752560b387b63fc4b8 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Thu, 4 Jan 2018 23:10:38 +0100 Subject: [PATCH 13/24] resort the converters so that the accepted_formats are infront --- lib/galaxy/datatypes/registry.py | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index b893f869b73..72f818e55e3 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -804,6 +804,7 @@ class Registry(object): converters = odict() source_datatype = type(self.get_datatype_by_extension(ext)) for ext2, converters_dict in self.datatype_converters.items(): + converter_datatype = type(self.get_datatype_by_extension(ext2)) if issubclass(source_datatype, converter_datatype): converters.update(converters_dict) @@ -822,7 +823,21 @@ class Registry(object): def find_conversion_destination_for_dataset_by_extensions(self, dataset, accepted_formats, converter_safe=True): """Returns ( target_ext, existing converted dataset )""" - for convert_ext in self.get_converters_by_datatype(dataset.ext): + + converters = self.get_converters_by_datatype(dataset.ext) + new_order = odict() + + accepted_format_keys = list() + for k in accepted_formats: + accepted_format_keys.append(k.file_ext) + if k.file_ext in converters: + new_order[k.file_ext] = converters[k.file_ext] + + for k,v in converters.items(): + if not k in accepted_format_keys: + new_order[k] = v + + for convert_ext in new_order: convert_ext_datatype = self.get_datatype_by_extension(convert_ext) if convert_ext_datatype is None: self.log.warning("Datatype class not found for extension '%s', which is used as target for conversion from datatype '%s'" % (convert_ext, dataset.ext)) From 8928b34506c1ba1f4856fad82dc42aef0f7c68f6 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 7 Jan 2018 21:51:36 +0100 Subject: [PATCH 14/24] implement priority converters --- lib/galaxy/datatypes/registry.py | 41 +++++++++++++++++--------------- 1 file changed, 22 insertions(+), 19 deletions(-) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 72f818e55e3..0d2b14f3808 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -798,20 +798,34 @@ class Registry(object): tabular.CSV() ] - def get_converters_by_datatype(self, ext): - """Returns available converters by source type""" + def get_converters_by_datatype(self, ext, priority_formats=None): + """ + Returns available converters by source type. + `priority_formats` will contain a lis of format extensions that should + be handled with priority. This means they will end up infront of the + returned ordered dictionary. + """ if ext not in self._converters_by_datatype: converters = odict() + prio_converters = odict() source_datatype = type(self.get_datatype_by_extension(ext)) for ext2, converters_dict in self.datatype_converters.items(): - converter_datatype = type(self.get_datatype_by_extension(ext2)) if issubclass(source_datatype, converter_datatype): - converters.update(converters_dict) + for k, v in converters_dict.items(): + if k in priority_formats: + prio_converters[k] = v + else: + converters[k] = v # Ensure ext-level converters are present if ext in self.datatype_converters.keys(): - converters.update(self.datatype_converters[ext]) - self._converters_by_datatype[ext] = converters + for k, v in self.datatype_converters[ext].items(): + if k in priority_formats: + prio_converters[k] = v + else: + converters[k] = v + prio_converters.update(converters) + self._converters_by_datatype[ext] = prio_converters return self._converters_by_datatype[ext] def get_converter_by_target_type(self, source_ext, target_ext): @@ -824,20 +838,9 @@ class Registry(object): def find_conversion_destination_for_dataset_by_extensions(self, dataset, accepted_formats, converter_safe=True): """Returns ( target_ext, existing converted dataset )""" - converters = self.get_converters_by_datatype(dataset.ext) - new_order = odict() + converters = self.get_converters_by_datatype(dataset.ext, [k.file_ext for k in accepted_formats]) - accepted_format_keys = list() - for k in accepted_formats: - accepted_format_keys.append(k.file_ext) - if k.file_ext in converters: - new_order[k.file_ext] = converters[k.file_ext] - - for k,v in converters.items(): - if not k in accepted_format_keys: - new_order[k] = v - - for convert_ext in new_order: + for convert_ext in converters: convert_ext_datatype = self.get_datatype_by_extension(convert_ext) if convert_ext_datatype is None: self.log.warning("Datatype class not found for extension '%s', which is used as target for conversion from datatype '%s'" % (convert_ext, dataset.ext)) From b139726ed380192ecf03f10e91220a0488b573f3 Mon Sep 17 00:00:00 2001 From: Bjoern Gruening Date: Sun, 7 Jan 2018 21:54:44 +0100 Subject: [PATCH 15/24] fix spelling --- lib/galaxy/datatypes/registry.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 0d2b14f3808..426dcb492c1 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -801,7 +801,7 @@ class Registry(object): def get_converters_by_datatype(self, ext, priority_formats=None): """ Returns available converters by source type. - `priority_formats` will contain a lis of format extensions that should + `priority_formats` will contain a list of format extensions that should be handled with priority. This means they will end up infront of the returned ordered dictionary. """ From 49dd02f4090b525beba735fc9ff16e32388b3fcd Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Thu, 11 Jan 2018 14:10:49 +0100 Subject: [PATCH 16/24] Keep header when converting sam to bam Otherwise the converted datasets are not valid BAM files. --- lib/galaxy/datatypes/converters/sam_to_bam_native.xml | 1 + test/functional/tools/sam_to_bam_native.xml | 1 + 2 files changed, 2 insertions(+) diff --git a/lib/galaxy/datatypes/converters/sam_to_bam_native.xml b/lib/galaxy/datatypes/converters/sam_to_bam_native.xml index 363a3b57b68..5e8475ea52a 100644 --- a/lib/galaxy/datatypes/converters/sam_to_bam_native.xml +++ b/lib/galaxy/datatypes/converters/sam_to_bam_native.xml @@ -6,6 +6,7 @@ Date: Thu, 11 Jan 2018 14:14:15 +0100 Subject: [PATCH 17/24] Revert "fix spelling" This reverts commit b139726ed380192ecf03f10e91220a0488b573f3. --- lib/galaxy/datatypes/registry.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 426dcb492c1..0d2b14f3808 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -801,7 +801,7 @@ class Registry(object): def get_converters_by_datatype(self, ext, priority_formats=None): """ Returns available converters by source type. - `priority_formats` will contain a list of format extensions that should + `priority_formats` will contain a lis of format extensions that should be handled with priority. This means they will end up infront of the returned ordered dictionary. """ From 38c5f711983033bb01072a769e6a618528724955 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Thu, 11 Jan 2018 14:14:33 +0100 Subject: [PATCH 18/24] Revert "implement priority converters" This reverts commit 8928b34506c1ba1f4856fad82dc42aef0f7c68f6. --- lib/galaxy/datatypes/registry.py | 41 +++++++++++++++----------------- 1 file changed, 19 insertions(+), 22 deletions(-) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 0d2b14f3808..72f818e55e3 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -798,34 +798,20 @@ class Registry(object): tabular.CSV() ] - def get_converters_by_datatype(self, ext, priority_formats=None): - """ - Returns available converters by source type. - `priority_formats` will contain a lis of format extensions that should - be handled with priority. This means they will end up infront of the - returned ordered dictionary. - """ + def get_converters_by_datatype(self, ext): + """Returns available converters by source type""" if ext not in self._converters_by_datatype: converters = odict() - prio_converters = odict() source_datatype = type(self.get_datatype_by_extension(ext)) for ext2, converters_dict in self.datatype_converters.items(): + converter_datatype = type(self.get_datatype_by_extension(ext2)) if issubclass(source_datatype, converter_datatype): - for k, v in converters_dict.items(): - if k in priority_formats: - prio_converters[k] = v - else: - converters[k] = v + converters.update(converters_dict) # Ensure ext-level converters are present if ext in self.datatype_converters.keys(): - for k, v in self.datatype_converters[ext].items(): - if k in priority_formats: - prio_converters[k] = v - else: - converters[k] = v - prio_converters.update(converters) - self._converters_by_datatype[ext] = prio_converters + converters.update(self.datatype_converters[ext]) + self._converters_by_datatype[ext] = converters return self._converters_by_datatype[ext] def get_converter_by_target_type(self, source_ext, target_ext): @@ -838,9 +824,20 @@ class Registry(object): def find_conversion_destination_for_dataset_by_extensions(self, dataset, accepted_formats, converter_safe=True): """Returns ( target_ext, existing converted dataset )""" - converters = self.get_converters_by_datatype(dataset.ext, [k.file_ext for k in accepted_formats]) + converters = self.get_converters_by_datatype(dataset.ext) + new_order = odict() - for convert_ext in converters: + accepted_format_keys = list() + for k in accepted_formats: + accepted_format_keys.append(k.file_ext) + if k.file_ext in converters: + new_order[k.file_ext] = converters[k.file_ext] + + for k,v in converters.items(): + if not k in accepted_format_keys: + new_order[k] = v + + for convert_ext in new_order: convert_ext_datatype = self.get_datatype_by_extension(convert_ext) if convert_ext_datatype is None: self.log.warning("Datatype class not found for extension '%s', which is used as target for conversion from datatype '%s'" % (convert_ext, dataset.ext)) From 6b444f609d3a48d17a783e581dc077b0731673ca Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Sat, 13 Jan 2018 13:10:46 +0100 Subject: [PATCH 19/24] Revert "resort the converters so that the accepted_formats are infront" This reverts commit 446643159540f8db67dfd0752560b387b63fc4b8. --- lib/galaxy/datatypes/registry.py | 17 +---------------- 1 file changed, 1 insertion(+), 16 deletions(-) diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 72f818e55e3..b893f869b73 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -804,7 +804,6 @@ class Registry(object): converters = odict() source_datatype = type(self.get_datatype_by_extension(ext)) for ext2, converters_dict in self.datatype_converters.items(): - converter_datatype = type(self.get_datatype_by_extension(ext2)) if issubclass(source_datatype, converter_datatype): converters.update(converters_dict) @@ -823,21 +822,7 @@ class Registry(object): def find_conversion_destination_for_dataset_by_extensions(self, dataset, accepted_formats, converter_safe=True): """Returns ( target_ext, existing converted dataset )""" - - converters = self.get_converters_by_datatype(dataset.ext) - new_order = odict() - - accepted_format_keys = list() - for k in accepted_formats: - accepted_format_keys.append(k.file_ext) - if k.file_ext in converters: - new_order[k.file_ext] = converters[k.file_ext] - - for k,v in converters.items(): - if not k in accepted_format_keys: - new_order[k] = v - - for convert_ext in new_order: + for convert_ext in self.get_converters_by_datatype(dataset.ext): convert_ext_datatype = self.get_datatype_by_extension(convert_ext) if convert_ext_datatype is None: self.log.warning("Datatype class not found for extension '%s', which is used as target for conversion from datatype '%s'" % (convert_ext, dataset.ext)) From c2328e32eb18bd7a4a5c791306203b933fc73603 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Thu, 11 Jan 2018 15:04:33 +0100 Subject: [PATCH 20/24] Move up sam_to_bam_native converter That should allow implicit conversion from sam to bam_native for tools that support bam_native. --- config/datatypes_conf.xml.sample | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/config/datatypes_conf.xml.sample b/config/datatypes_conf.xml.sample index 4045ac713fc..7c590f5e7df 100644 --- a/config/datatypes_conf.xml.sample +++ b/config/datatypes_conf.xml.sample @@ -288,8 +288,8 @@ - + From a36ebe35521e3e45c5c3301eef9535ebdf630f9d Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Sat, 13 Jan 2018 14:17:56 +0100 Subject: [PATCH 21/24] Enhance tests for sam_to_bam conversion This demonstrates that the `sam_to_bam` converter will be used for an input with `format="bam"`, and not `sam_to_bam_native`. --- test/functional/tools/sam_to_bam_native.xml | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/test/functional/tools/sam_to_bam_native.xml b/test/functional/tools/sam_to_bam_native.xml index b960bb4c87a..aed8f24892c 100644 --- a/test/functional/tools/sam_to_bam_native.xml +++ b/test/functional/tools/sam_to_bam_native.xml @@ -8,30 +8,39 @@ -b -h -@ \${GALAXY_SLOTS:-2} - -o '${output}' + -o '$bam_native_output' '$input1' - #else - cp '$input2' '$output' + #elif $input2: + cp '$input2' '$bam_native_output' + #elif $input3: + cp '$input3' '$bam_output' #end if ]]> + - + + - + - + + + + + + From aaaf7b50a7a227e5344712f6441fce337a20d1ba Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Wed, 17 Jan 2018 17:10:17 +0100 Subject: [PATCH 22/24] Enable conda_auto_init and conda_auto_install for framework tests --- scripts/functional_tests.py | 2 ++ test/base/driver_util.py | 9 ++++++++- 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/scripts/functional_tests.py b/scripts/functional_tests.py index 51228a384bb..0493a49bd3b 100644 --- a/scripts/functional_tests.py +++ b/scripts/functional_tests.py @@ -71,6 +71,8 @@ class FrameworkToolsGalaxyTestDriver(DefaultGalaxyTestDriver): """Galaxy-style nose TestDriver for testing framework Galaxy tools.""" framework_tool_and_types = True + conda_auto_init = True + conda_auto_install = True class DataManagersGalaxyTestDriver(driver_util.GalaxyTestDriver): diff --git a/test/base/driver_util.py b/test/base/driver_util.py index 29e99a8aef7..131a72b173a 100644 --- a/test/base/driver_util.py +++ b/test/base/driver_util.py @@ -129,6 +129,8 @@ def setup_galaxy_config( update_integrated_tool_panel=False, prefer_template_database=False, log_format=None, + conda_auto_init=False, + conda_auto_install=False ): """Setup environment and build config for test Galaxy instance.""" if not os.path.exists(tmpdir): @@ -188,7 +190,8 @@ def setup_galaxy_config( api_allow_run_as='test@bx.psu.edu', auto_configure_logging=logging_config_file is None, check_migrate_tools=False, - conda_auto_init=False, + conda_auto_init=conda_auto_init, + conda_auto_install=conda_auto_install, cleanup_job='onsuccess', data_manager_config_file=data_manager_config_file, enable_beta_tool_formats=True, @@ -793,6 +796,8 @@ class GalaxyTestDriver(TestDriver): """Instantial a Galaxy-style nose TestDriver for testing Galaxy.""" testing_shed_tools = False + conda_auto_init = False + conda_auto_install = False def setup(self, config_object=None): """Setup a Galaxy server for functional test (if needed). @@ -851,6 +856,8 @@ class GalaxyTestDriver(TestDriver): datatypes_conf=datatypes_conf_override, prefer_template_database=getattr(config_object, "prefer_template_database", False), log_format=log_format, + conda_auto_init=self.conda_auto_init, + conda_auto_install=self.conda_auto_install, ) galaxy_config = setup_galaxy_config( galaxy_db_path, From 680bf685c2dfdc874277321e2c7d0673516bf62a Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Wed, 17 Jan 2018 18:11:31 +0100 Subject: [PATCH 23/24] Fix datatype order --- test/functional/tools/sample_datatypes_conf.xml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/functional/tools/sample_datatypes_conf.xml b/test/functional/tools/sample_datatypes_conf.xml index 98991439526..c2f8edbe9c6 100644 --- a/test/functional/tools/sample_datatypes_conf.xml +++ b/test/functional/tools/sample_datatypes_conf.xml @@ -15,8 +15,8 @@ - + From 8ecc86cb677bc3971baa75bf0e8c1481d8e3aef4 Mon Sep 17 00:00:00 2001 From: mvdbeek Date: Wed, 17 Jan 2018 19:32:15 +0100 Subject: [PATCH 24/24] Simplify conda framework setup Thanks @nsoranzo! --- test/base/driver_util.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/test/base/driver_util.py b/test/base/driver_util.py index 131a72b173a..7f9ff9ede21 100644 --- a/test/base/driver_util.py +++ b/test/base/driver_util.py @@ -796,8 +796,6 @@ class GalaxyTestDriver(TestDriver): """Instantial a Galaxy-style nose TestDriver for testing Galaxy.""" testing_shed_tools = False - conda_auto_init = False - conda_auto_install = False def setup(self, config_object=None): """Setup a Galaxy server for functional test (if needed). @@ -856,8 +854,8 @@ class GalaxyTestDriver(TestDriver): datatypes_conf=datatypes_conf_override, prefer_template_database=getattr(config_object, "prefer_template_database", False), log_format=log_format, - conda_auto_init=self.conda_auto_init, - conda_auto_install=self.conda_auto_install, + conda_auto_init=getattr(config_object, "conda_auto_init", False), + conda_auto_install=getattr(config_object, "conda_auto_install", False), ) galaxy_config = setup_galaxy_config( galaxy_db_path,