diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index c0feebd7d52..c0ba39e006b 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -1030,8 +1030,14 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc :param is_multi_byte: deprecated :type is_multi_byte: bool - >>> fname = get_test_fname('4.bed') - >>> assert get_file_peek(fname, LINE_COUNT=1) == u'chr22\\t30128507\\t31828507\\tuc003bnx.1_cds_2_0_chr22_29227_f\\t0\\t+\\n' + >>> def assert_peek_is(file_name, expected, *args, **kwd): + ... path = get_test_fname(file_name) + ... peek = get_file_peek(path, *args, **kwd) + ... assert peek == expected, "%s != %s" % (peek, expected) + >>> assert_peek_is('0_nonewline', u'0') + >>> assert_peek_is('0.txt', u'0\\n') + >>> assert_peek_is('4.bed', u'chr22\\t30128507\\t31828507\\tuc003bnx.1_cds_2_0_chr22_29227_f\\t0\\t+\\n', LINE_COUNT=1) + >>> assert_peek_is('1.bed', u'chr1\\t147962192\\t147962580\\tCCDS989.1_cds_0_0_chr1_147962193_r\\t0\\t-\\nchr1\\t147984545\\t147984630\\tCCDS990.1_cds_0_0_chr1_147984546_f\\t0\\t+\\n', LINE_COUNT=2) """ # Set size for file.readline() to a negative number to force it to # read until either a newline or EOF. Needed for datasets with very @@ -1042,20 +1048,27 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc skipchars = [] lines = [] count = 0 + + last_line_break = False with compression_utils.get_fileobj(file_name, "U") as temp: while count < LINE_COUNT: try: line = temp.readline(WIDTH) except UnicodeDecodeError: return "binary file" - if not line_wrap: - if line.endswith('\n'): - line = line[:-1] - else: - while True: - i = temp.read(1) - if not i or i == '\n': - break + if line == "": + break + last_line_break = False + if line.endswith('\n'): + line = line[:-1] + last_line_break = True + elif not line_wrap: + while True: + i = temp.read(1) + if i == '\n': + last_line_break = True + if not i or i == '\n': + break skip_line = False for skipchar in skipchars: if line.startswith(skipchar): @@ -1064,4 +1077,4 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc if not skip_line: lines.append(line) count += 1 - return '\n'.join(lines) + return '\n'.join(lines) + ('\n' if last_line_break else '') diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 404f9c53a11..d35c83ffd4c 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -14,7 +14,11 @@ import sys import tempfile import zipfile -from six import StringIO, text_type +from six import ( + PY3, + StringIO, + text_type, +) from six.moves import filter from six.moves.urllib.request import urlopen @@ -103,28 +107,40 @@ def stream_to_file(stream, suffix='', prefix='', dir=None, text=False, **kwd): return stream_to_open_named_file(stream, fd, temp_name, **kwd) -def convert_newlines(fname, in_place=True, tmp_dir=None, tmp_prefix="gxupload"): +def convert_newlines(fname, in_place=True, tmp_dir=None, tmp_prefix="gxupload", block_size=128 * 1024, regexp=None): """ Converts in place a file from universal line endings to Posix line endings. - - >>> fname = get_test_fname('temp.txt') - >>> with open(fname, 'wt') as fh: - ... _ = fh.write("1 2\\r3 4") - >>> convert_newlines(fname, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir()) - (2, None) - >>> open(fname).read() - '1 2\\n3 4\\n' """ fd, temp_name = tempfile.mkstemp(prefix=tmp_prefix, dir=tmp_dir) - with io.open(fd, mode="wt", encoding='utf-8') as fp: - i = None - for i, line in enumerate(io.open(fname, encoding='utf-8')): - fp.write("%s\n" % line.rstrip("\r\n")) - if i is None: - i = 0 + i = 0 + if PY3: + NEWLINE_BYTE = 10 + CR_BYTE = 13 else: - i += 1 + NEWLINE_BYTE = "\n" + CR_BYTE = "\r" + with io.open(fd, mode="wb") as fp, io.open(fname, mode="rb") as fi: + last_char = None + block = fi.read(block_size) + last_block = b"" + while block: + if last_char == CR_BYTE and block.startswith(b"\n"): + # Last block ended with CR, new block startswith newline. + # Since we replace CR with newline in the previous iteration we skip the first byte + block = block[1:] + if block: + last_char = block[-1] + block = block.replace(b"\r\n", b"\n").replace(b"\r", b"\n") + if regexp: + block = b"\t".join(regexp.split(block)) + fp.write(block) + i += block.count(b"\n") + last_block = block + block = fi.read(block_size) + if last_block and last_block[-1] != NEWLINE_BYTE: + i += 1 + fp.write(b"\n") if in_place: shutil.move(temp_name, fname) # Return number of lines in file. @@ -133,47 +149,9 @@ def convert_newlines(fname, in_place=True, tmp_dir=None, tmp_prefix="gxupload"): return (i, temp_name) -def sep2tabs(fname, in_place=True, patt=r"\s+", tmp_dir=None, tmp_prefix="gxupload"): +def convert_newlines_sep2tabs(fname, in_place=True, patt=br"[^\S\n]+", tmp_dir=None, tmp_prefix="gxupload"): """ - Transforms in place a 'sep' separated file to a tab separated one - - >>> fname = get_test_fname('temp.txt') - >>> with open(fname, 'wt') as fh: - ... _ = fh.write(u"1 2\\n3 4\\n") - >>> sep2tabs(fname) - (2, None) - >>> open(fname).read() - '1\\t2\\n3\\t4\\n' - """ - regexp = re.compile(patt) - fd, temp_name = tempfile.mkstemp(prefix=tmp_prefix, dir=tmp_dir) - with io.open(fd, mode="wt", encoding='utf-8') as fp: - i = None - for i, line in enumerate(io.open(fname, encoding='utf-8')): - if line.endswith("\r"): - line = line.rstrip('\r') - elems = regexp.split(line) - fp.write(u"%s\r" % '\t'.join(elems)) - else: - line = line.rstrip('\n') - elems = regexp.split(line) - fp.write(u"%s\n" % '\t'.join(elems)) - if i is None: - i = 0 - else: - i += 1 - if in_place: - shutil.move(temp_name, fname) - # Return number of lines in file. - return (i, None) - else: - return (i, temp_name) - - -def convert_newlines_sep2tabs(fname, in_place=True, patt=r"\s+", tmp_dir=None, tmp_prefix="gxupload"): - """ - Combines above methods: convert_newlines() and sep2tabs() - so that files do not need to be read twice + Converts newlines in a file to posix newlines and replaces spaces with tabs. >>> fname = get_test_fname('temp.txt') >>> with open(fname, 'wt') as fh: @@ -184,18 +162,7 @@ def convert_newlines_sep2tabs(fname, in_place=True, patt=r"\s+", tmp_dir=None, t '1\\t2\\n3\\t4\\n' """ regexp = re.compile(patt) - fd, temp_name = tempfile.mkstemp(prefix=tmp_prefix, dir=tmp_dir) - with io.open(fd, mode="wt", encoding='utf-8') as fp: - for i, line in enumerate(io.open(fname, encoding='utf-8')): - line = line.rstrip('\r\n') - elems = regexp.split(line) - fp.write(u"%s\n" % '\t'.join(elems)) - if in_place: - shutil.move(temp_name, fname) - # Return number of lines in file. - return (i + 1, None) - else: - return (i + 1, temp_name) + return convert_newlines(fname, in_place, tmp_dir, tmp_prefix, regexp=regexp) def iter_headers(fname_or_file_prefix, sep, count=60, comment_designator=None): @@ -474,6 +441,9 @@ def guess_ext(fname, sniff_order, is_binary=False): >>> fname = get_test_fname('1.mtx') >>> guess_ext(fname, sniff_order) 'mtx' + >>> fname = get_test_fname('1imzml') + >>> guess_ext(fname, sniff_order) # This test case is ensuring doesn't throw exception, actual value could change if non-utf encoding handling improves. + 'data' """ file_prefix = FilePrefix(fname) file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary) @@ -555,8 +525,9 @@ def zip_single_fileobj(path): class FilePrefix(object): def __init__(self, filename): - binary = False + non_utf8_error = None compressed_format = None + contents_header_bytes = None contents_header = None # First MAX_BYTES of the file. truncated = False # A future direction to optimize sniffing even more for sniffers at the top of the list @@ -565,20 +536,23 @@ class FilePrefix(object): # populates contents_header while providing a StringIO-like interface until the file is read # but then would fallback to native string_io() try: - compressed_format, f = compression_utils.get_fileobj_raw(filename) + compressed_format, f = compression_utils.get_fileobj_raw(filename, "rb") try: - contents_header = f.read(SNIFF_PREFIX_BYTES) - truncated = len(contents_header) == SNIFF_PREFIX_BYTES + contents_header_bytes = f.read(SNIFF_PREFIX_BYTES) + truncated = len(contents_header_bytes) == SNIFF_PREFIX_BYTES + contents_header = contents_header_bytes.decode("utf-8") finally: f.close() - except UnicodeDecodeError: - binary = True + except UnicodeDecodeError as e: + non_utf8_error = e self.truncated = truncated self.filename = filename - self.binary = binary + self.non_utf8_error = non_utf8_error + self.binary = non_utf8_error is not None # obviously wrong self.compressed_format = compressed_format self.contents_header = contents_header + self.contents_header_bytes = contents_header_bytes self._file_size = None @property @@ -588,8 +562,8 @@ class FilePrefix(object): return self._file_size def string_io(self): - if self.binary: - raise Exception("Attempting to create a StringIO object for binary data.") + if self.non_utf8_error is not None: + raise self.non_utf8_error rval = StringIO(self.contents_header) return rval diff --git a/lib/galaxy/datatypes/test/0_nonewline b/lib/galaxy/datatypes/test/0_nonewline new file mode 100644 index 00000000000..c227083464f --- /dev/null +++ b/lib/galaxy/datatypes/test/0_nonewline @@ -0,0 +1 @@ +0 \ No newline at end of file diff --git a/lib/galaxy/datatypes/test/1imzml b/lib/galaxy/datatypes/test/1imzml new file mode 100644 index 00000000000..6b45d2a3667 --- /dev/null +++ b/lib/galaxy/datatypes/test/1imzml @@ -0,0 +1,380 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/lib/galaxy/datatypes/test/dosimzml b/lib/galaxy/datatypes/test/dosimzml new file mode 100644 index 00000000000..6ff4f5c078d --- /dev/null +++ b/lib/galaxy/datatypes/test/dosimzml @@ -0,0 +1,380 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/test/integration/test_datatype_upload.py b/test/integration/test_datatype_upload.py index 4c7c742c534..58e45b0bb55 100644 --- a/test/integration/test_datatype_upload.py +++ b/test/integration/test_datatype_upload.py @@ -32,7 +32,7 @@ def find_datatype(registry, filename): def collect_test_data(registry): - test_files = os.listdir(TEST_FILE_DIR) + test_files = [f for f in os.listdir(TEST_FILE_DIR) if "." in f] files = [os.path.join(TEST_FILE_DIR, f) for f in test_files] datatypes = [find_datatype(registry, f) for f in test_files] uploadable = [datatype.file_ext in registry.upload_file_formats for datatype in datatypes] diff --git a/test/unit/datatypes/test_data.py b/test/unit/datatypes/test_data.py index 969916d57b9..d1f85571714 100644 --- a/test/unit/datatypes/test_data.py +++ b/test/unit/datatypes/test_data.py @@ -10,4 +10,4 @@ from galaxy.util import galaxy_directory def test_get_file_peek(): # should get the first 5 lines of the file without a trailing newline character - assert get_file_peek(os.path.join(galaxy_directory(), 'test-data/1.tabular'), line_wrap=False) == 'chr22\t1000\tNM_17\nchr22\t2000\tNM_18\nchr10\t2200\tNM_10\nchr10\thap\ttest\nchr10\t1200\tNM_11' + assert get_file_peek(os.path.join(galaxy_directory(), 'test-data/1.tabular'), line_wrap=False) == 'chr22\t1000\tNM_17\nchr22\t2000\tNM_18\nchr10\t2200\tNM_10\nchr10\thap\ttest\nchr10\t1200\tNM_11\n' diff --git a/test/unit/datatypes/test_sniff.py b/test/unit/datatypes/test_sniff.py new file mode 100644 index 00000000000..faf75443281 --- /dev/null +++ b/test/unit/datatypes/test_sniff.py @@ -0,0 +1,79 @@ +import tempfile + +import pytest + +from galaxy.datatypes.sniff import ( + convert_newlines, + convert_newlines_sep2tabs, + get_test_fname, +) + + +def assert_converts_to_1234_convert_sep2tabs(content, expected='1\t2\n3\t4\n'): + with tempfile.NamedTemporaryFile(delete=False, mode='w') as tf: + tf.write(content) + rval = convert_newlines_sep2tabs(tf.name, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir()) + assert expected == open(tf.name).read() + assert rval == (2, None), rval + + +def assert_converts_to_1234_convert(content, block_size=1024): + fname = get_test_fname('temp2.txt') + with open(fname, 'w') as fh: + fh.write(content) + rval = convert_newlines(fname, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir(), block_size=block_size) + actual_contents = open(fname).read() + assert '1 2\n3 4\n' == actual_contents, actual_contents + assert rval == (2, None), "rval != %s for %s" % (rval, content) + + +@pytest.mark.parametrize('source,block_size', [ + ("1 2\r3 4", None), + ("1 2\n3 4", None), + ("1 2\r\n3 4", None), + ("1 2\r3 4\r", None), + ("1 2\n3 4\n", None), + ("1 2\r\n3 4\r\n", None), + ("1 2\r3 4", 2), + ("1 2\n3 4", 2), + ("1 2\r\n3 4", 2), + ("1 2\r3 4\r", 2), + ("1 2\n3 4\n", 2), + ("1 2\r\n3 4\r\n", 2), + ("1 2\r3 4", 3), + ("1 2\n3 4", 3), + ("1 2\r\n3 4", 3), + ("1 2\r3 4\r", 3), + ("1 2\n3 4\n", 3), + ("1 2\r\n3 4\r\n", 3), +]) +def test_convert_newlines(source, block_size): + # Verify ends with newline - with or without that on inputs - for any of + # \r \\n or \\r\\n newlines. + if block_size: + assert_converts_to_1234_convert(source, block_size) + else: + assert_converts_to_1234_convert(source) + + +def test_convert_newlines_non_utf(): + fname = get_test_fname("dosimzml") + rval = convert_newlines(fname, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir(), in_place=False) + new_file = rval[1] + assert open(new_file, "rb").read() == open(get_test_fname("1imzml"), "rb").read() + + +@pytest.mark.parametrize('source,expected', [ + ("1 2\n3 4\n", None), + ("1 2\n3 4\n", None), + ("1\t2\n3\t4\n", None), + ("1\t2\r3\t4\r", None), + ("1\t2\r\n3\t4\r\n", None), + ("1 2\r\n3 4\r\n", None), + ("1 2 \n3 4 \n", '1\t2\t\n3\t4\t\n'), +]) +def test_convert_sep2tabs(source, expected): + if expected: + assert_converts_to_1234_convert_sep2tabs(source, expected=expected) + else: + assert_converts_to_1234_convert_sep2tabs(source)