From 56b5002aababf39a11d8e6fd794a19382f5f767f Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 22 Jun 2016 10:29:57 -0400 Subject: [PATCH 1/2] Tabular dataset chunking fix which should prevent duplicated lines. Skipping to the newline previously shifted the chunk past where it should read. --- lib/galaxy/datatypes/tabular.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index e2f021fa273..cac3e7b47de 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -56,13 +56,16 @@ class TabularData( data.Text ): def get_chunk(self, trans, dataset, chunk): ck_index = int(chunk) f = open(dataset.file_name) + t_start = (ck_index * trans.app.config.display_chunk_size) f.seek(ck_index * trans.app.config.display_chunk_size) # If we aren't at the start of the file, seek to next newline. Do this better eventually. if f.tell() != 0: cursor = f.read(1) while cursor and cursor != '\n': cursor = f.read(1) - ck_data = f.read(trans.app.config.display_chunk_size) + t_firstnewline = f.tell() + t_skipped = t_firstnewline - t_start + ck_data = f.read(trans.app.config.display_chunk_size - (t_skipped)) cursor = f.read(1) while cursor and ck_data[-1] != '\n': ck_data += cursor From 0ab502baa1ae0e16c9345c3ac816a1dde5cb54d9 Mon Sep 17 00:00:00 2001 From: Dannon Baker Date: Wed, 22 Jun 2016 10:37:42 -0400 Subject: [PATCH 2/2] Rename vars to be sensible, add comment to clarify what this is doing. --- lib/galaxy/datatypes/tabular.py | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index cac3e7b47de..0a2c03334e7 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -56,16 +56,18 @@ class TabularData( data.Text ): def get_chunk(self, trans, dataset, chunk): ck_index = int(chunk) f = open(dataset.file_name) - t_start = (ck_index * trans.app.config.display_chunk_size) - f.seek(ck_index * trans.app.config.display_chunk_size) + initial_offset = (ck_index * trans.app.config.display_chunk_size) + f.seek(initial_offset) # If we aren't at the start of the file, seek to next newline. Do this better eventually. if f.tell() != 0: cursor = f.read(1) while cursor and cursor != '\n': cursor = f.read(1) - t_firstnewline = f.tell() - t_skipped = t_firstnewline - t_start - ck_data = f.read(trans.app.config.display_chunk_size - (t_skipped)) + read_start_offset = f.tell() + prechunk_skip = read_start_offset - initial_offset + # We subtract the prechunk skip out of the primary chunk read to avoid + # shifting the chunk tail onto a line it shouldn't have. + ck_data = f.read(trans.app.config.display_chunk_size - prechunk_skip) cursor = f.read(1) while cursor and ck_data[-1] != '\n': ck_data += cursor