From 6c08780fee122d7f82e3675b63f15ed4bf1dd952 Mon Sep 17 00:00:00 2001 From: Wolfgang Maier Date: Thu, 20 Apr 2023 23:41:59 +0200 Subject: [PATCH 1/3] Improve display chunk generation for BAMs The new version is faster even for lines containing multiple tabs and optimizes the no or one tag cases. It also fixes a subtle bug that would drop a line at chunk boundaries. --- lib/galaxy/datatypes/binary.py | 30 +++++++++++++++++------------- 1 file changed, 17 insertions(+), 13 deletions(-) diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index 1908dc124dd..f95bbdce563 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -609,30 +609,34 @@ class BamNative(CompressedArchive, _BamOrSam): try: with pysam.AlignmentFile(dataset.file_name, "rb", check_sq=False) as bamfile: ck_size = 300 # 300 lines - ck_data = "" - header_line_count = 0 if offset == 0: - ck_data = bamfile.text.replace("\t", " ") # type: ignore[attr-defined] - header_line_count = bamfile.text.count("\n") # type: ignore[attr-defined] + offset = bamfile.tell() + ck_lines = bamfile.text.strip().replace("\t", " ").splitlines() # type: ignore[attr-defined] else: bamfile.seek(offset) - for line_number, alignment in enumerate(bamfile): + ck_lines = [] + for line_number, alignment in enumerate(bamfile, len(ck_lines)): # return only Header lines if 'header_line_count' exceeds 'ck_size' # FIXME: Can be problematic if bam has million lines of header - offset = bamfile.tell() - if (line_number + header_line_count) > ck_size: + if line_number > ck_size: break - else: - bamline = alignment.tostring(bamfile) - # Galaxy display each tag as separate column because 'tostring()' funcition put tabs in between each tag of tags column. - # Below code will remove spaces between each tag. - bamline_modified = ("\t").join(bamline.split()[:11] + [(" ").join(bamline.split()[11:])]) - ck_data = f"{ck_data}\n{bamline_modified}" + + offset = bamfile.tell() + bamline = alignment.tostring(bamfile) + # With multiple tags, Galaxy would display each as a separate column + # because the 'tostring()' function uses tabs also between tags. + # Below code will turn these extra tabs into spaces. + n_tabs = bamline.count('\t') + if n_tabs > 11: + bamline, *extra_tags = bamline.rsplit('\t', maxsplit=n_tabs - 11) + bamline = f"{bamline} {' '.join(extra_tags) + ck_lines.append(f"{bamline} {' '.join(extra_tags)}") else: # Nothing to enumerate; we've either offset to the end # of the bamfile, or there is no data. (possible with # header-only bams) offset = -1 + ck_data = "\n".join(ck_lines) except Exception as e: offset = -1 ck_data = f"Could not display BAM file, error was:\n{e}" From 8d4f434402acc3cd8000be8d6c6c00d4d85155e9 Mon Sep 17 00:00:00 2001 From: Wolfgang Maier Date: Thu, 20 Apr 2023 23:57:57 +0200 Subject: [PATCH 2/3] Fix wrongly saved file --- lib/galaxy/datatypes/binary.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index f95bbdce563..a7023538f6e 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -629,8 +629,8 @@ class BamNative(CompressedArchive, _BamOrSam): n_tabs = bamline.count('\t') if n_tabs > 11: bamline, *extra_tags = bamline.rsplit('\t', maxsplit=n_tabs - 11) - bamline = f"{bamline} {' '.join(extra_tags) - ck_lines.append(f"{bamline} {' '.join(extra_tags)}") + bamline = f"{bamline} {' '.join(extra_tags)}" + ck_lines.append(bamline) else: # Nothing to enumerate; we've either offset to the end # of the bamfile, or there is no data. (possible with From 999162c8843f39e6f608428dc38e4b8efb14b729 Mon Sep 17 00:00:00 2001 From: Wolfgang Maier Date: Fri, 21 Apr 2023 09:12:23 +0200 Subject: [PATCH 3/3] Fix linting --- lib/galaxy/datatypes/binary.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index a7023538f6e..7a30e9f6f55 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -626,9 +626,9 @@ class BamNative(CompressedArchive, _BamOrSam): # With multiple tags, Galaxy would display each as a separate column # because the 'tostring()' function uses tabs also between tags. # Below code will turn these extra tabs into spaces. - n_tabs = bamline.count('\t') + n_tabs = bamline.count("\t") if n_tabs > 11: - bamline, *extra_tags = bamline.rsplit('\t', maxsplit=n_tabs - 11) + bamline, *extra_tags = bamline.rsplit("\t", maxsplit=n_tabs - 11) bamline = f"{bamline} {' '.join(extra_tags)}" ck_lines.append(bamline) else: