From 6000f79b6d7ebf467f3a6b398f5dfe67e0c5b870 Mon Sep 17 00:00:00 2001 From: John Chilton Date: Sun, 12 May 2019 21:49:14 -0400 Subject: [PATCH] Fix extra newlines in peek code. Behavior changes: - Now includes a trailing new line if the input contained a trailing newline. - Now returns only as many lines as are present if the file contains fewer lines than LINE_COUNT. --- lib/galaxy/datatypes/data.py | 35 ++++++++++++++++-------- lib/galaxy/datatypes/test/0_nonewline | 1 + test/integration/test_datatype_upload.py | 2 +- test/unit/datatypes/test_data.py | 2 +- 4 files changed, 27 insertions(+), 13 deletions(-) create mode 100644 lib/galaxy/datatypes/test/0_nonewline diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index c0feebd7d52..c0ba39e006b 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -1030,8 +1030,14 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc :param is_multi_byte: deprecated :type is_multi_byte: bool - >>> fname = get_test_fname('4.bed') - >>> assert get_file_peek(fname, LINE_COUNT=1) == u'chr22\\t30128507\\t31828507\\tuc003bnx.1_cds_2_0_chr22_29227_f\\t0\\t+\\n' + >>> def assert_peek_is(file_name, expected, *args, **kwd): + ... path = get_test_fname(file_name) + ... peek = get_file_peek(path, *args, **kwd) + ... assert peek == expected, "%s != %s" % (peek, expected) + >>> assert_peek_is('0_nonewline', u'0') + >>> assert_peek_is('0.txt', u'0\\n') + >>> assert_peek_is('4.bed', u'chr22\\t30128507\\t31828507\\tuc003bnx.1_cds_2_0_chr22_29227_f\\t0\\t+\\n', LINE_COUNT=1) + >>> assert_peek_is('1.bed', u'chr1\\t147962192\\t147962580\\tCCDS989.1_cds_0_0_chr1_147962193_r\\t0\\t-\\nchr1\\t147984545\\t147984630\\tCCDS990.1_cds_0_0_chr1_147984546_f\\t0\\t+\\n', LINE_COUNT=2) """ # Set size for file.readline() to a negative number to force it to # read until either a newline or EOF. Needed for datasets with very @@ -1042,20 +1048,27 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc skipchars = [] lines = [] count = 0 + + last_line_break = False with compression_utils.get_fileobj(file_name, "U") as temp: while count < LINE_COUNT: try: line = temp.readline(WIDTH) except UnicodeDecodeError: return "binary file" - if not line_wrap: - if line.endswith('\n'): - line = line[:-1] - else: - while True: - i = temp.read(1) - if not i or i == '\n': - break + if line == "": + break + last_line_break = False + if line.endswith('\n'): + line = line[:-1] + last_line_break = True + elif not line_wrap: + while True: + i = temp.read(1) + if i == '\n': + last_line_break = True + if not i or i == '\n': + break skip_line = False for skipchar in skipchars: if line.startswith(skipchar): @@ -1064,4 +1077,4 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc if not skip_line: lines.append(line) count += 1 - return '\n'.join(lines) + return '\n'.join(lines) + ('\n' if last_line_break else '') diff --git a/lib/galaxy/datatypes/test/0_nonewline b/lib/galaxy/datatypes/test/0_nonewline new file mode 100644 index 00000000000..c227083464f --- /dev/null +++ b/lib/galaxy/datatypes/test/0_nonewline @@ -0,0 +1 @@ +0 \ No newline at end of file diff --git a/test/integration/test_datatype_upload.py b/test/integration/test_datatype_upload.py index 4c7c742c534..58e45b0bb55 100644 --- a/test/integration/test_datatype_upload.py +++ b/test/integration/test_datatype_upload.py @@ -32,7 +32,7 @@ def find_datatype(registry, filename): def collect_test_data(registry): - test_files = os.listdir(TEST_FILE_DIR) + test_files = [f for f in os.listdir(TEST_FILE_DIR) if "." in f] files = [os.path.join(TEST_FILE_DIR, f) for f in test_files] datatypes = [find_datatype(registry, f) for f in test_files] uploadable = [datatype.file_ext in registry.upload_file_formats for datatype in datatypes] diff --git a/test/unit/datatypes/test_data.py b/test/unit/datatypes/test_data.py index 969916d57b9..d1f85571714 100644 --- a/test/unit/datatypes/test_data.py +++ b/test/unit/datatypes/test_data.py @@ -10,4 +10,4 @@ from galaxy.util import galaxy_directory def test_get_file_peek(): # should get the first 5 lines of the file without a trailing newline character - assert get_file_peek(os.path.join(galaxy_directory(), 'test-data/1.tabular'), line_wrap=False) == 'chr22\t1000\tNM_17\nchr22\t2000\tNM_18\nchr10\t2200\tNM_10\nchr10\thap\ttest\nchr10\t1200\tNM_11' + assert get_file_peek(os.path.join(galaxy_directory(), 'test-data/1.tabular'), line_wrap=False) == 'chr22\t1000\tNM_17\nchr22\t2000\tNM_18\nchr10\t2200\tNM_10\nchr10\thap\ttest\nchr10\t1200\tNM_11\n'