diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index 6d352abb08d..35f7dba4a5f 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -14,7 +14,11 @@ import sys import tempfile import zipfile -from six import StringIO, text_type +from six import ( + PY3, + StringIO, + text_type, +) from six.moves import filter from six.moves.urllib.request import urlopen @@ -110,27 +114,30 @@ def convert_newlines(fname, in_place=True, tmp_dir=None, tmp_prefix="gxupload", """ fd, temp_name = tempfile.mkstemp(prefix=tmp_prefix, dir=tmp_dir) i = 0 + if PY3: + NEWLINE_BYTE = 10 + CR_BYTE = 13 + else: + NEWLINE_BYTE = "\n" + CR_BYTE = "\r" with io.open(fd, mode="wb") as fp: with io.open(fname, mode="rb") as fi: - line = b'' - converted_line = b"" last_char = None block = fi.read(block_size) + last_block = b"" while block: - if last_char == "\r" and block.startswith(b"\n"): - # last block ended with "\r", new block startswith "\n" - # since we replace "\r" with "\n" in the previous iteration we skip the first byte + if last_char == CR_BYTE and block.startswith(b"\n"): + # Last block ended with CR, new block startswith newline. + # Since we replace CR with newline in the previous iteration we skip the first byte block = block[1:] - # splitlines(True) splits at line terminators but keeps them so we can replace them - lines = block.splitlines(True) - for line in lines: - converted_line = line.replace(b"\r\n", b"\n").replace(b"\r", b"\n") - if b"\n" in converted_line: - i += 1 - fp.write(converted_line) - last_char = util.unicodify(line, error='replace')[-1] - block = fi.read(block_size) - if not converted_line.endswith(b"\n"): + if block: + last_char = block[-1] + block = block.replace(b"\r\n", b"\n").replace(b"\r", b"\n") + fp.write(block) + i += block.count(b"\n") + last_block = block + block = fi.read(block_size) + if last_block and last_block[-1] != NEWLINE_BYTE: i += 1 fp.write(b"\n") if in_place: