diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py
index c0feebd7d52..c0ba39e006b 100644
--- a/lib/galaxy/datatypes/data.py
+++ b/lib/galaxy/datatypes/data.py
@@ -1030,8 +1030,14 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc
:param is_multi_byte: deprecated
:type is_multi_byte: bool
- >>> fname = get_test_fname('4.bed')
- >>> assert get_file_peek(fname, LINE_COUNT=1) == u'chr22\\t30128507\\t31828507\\tuc003bnx.1_cds_2_0_chr22_29227_f\\t0\\t+\\n'
+ >>> def assert_peek_is(file_name, expected, *args, **kwd):
+ ... path = get_test_fname(file_name)
+ ... peek = get_file_peek(path, *args, **kwd)
+ ... assert peek == expected, "%s != %s" % (peek, expected)
+ >>> assert_peek_is('0_nonewline', u'0')
+ >>> assert_peek_is('0.txt', u'0\\n')
+ >>> assert_peek_is('4.bed', u'chr22\\t30128507\\t31828507\\tuc003bnx.1_cds_2_0_chr22_29227_f\\t0\\t+\\n', LINE_COUNT=1)
+ >>> assert_peek_is('1.bed', u'chr1\\t147962192\\t147962580\\tCCDS989.1_cds_0_0_chr1_147962193_r\\t0\\t-\\nchr1\\t147984545\\t147984630\\tCCDS990.1_cds_0_0_chr1_147984546_f\\t0\\t+\\n', LINE_COUNT=2)
"""
# Set size for file.readline() to a negative number to force it to
# read until either a newline or EOF. Needed for datasets with very
@@ -1042,20 +1048,27 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc
skipchars = []
lines = []
count = 0
+
+ last_line_break = False
with compression_utils.get_fileobj(file_name, "U") as temp:
while count < LINE_COUNT:
try:
line = temp.readline(WIDTH)
except UnicodeDecodeError:
return "binary file"
- if not line_wrap:
- if line.endswith('\n'):
- line = line[:-1]
- else:
- while True:
- i = temp.read(1)
- if not i or i == '\n':
- break
+ if line == "":
+ break
+ last_line_break = False
+ if line.endswith('\n'):
+ line = line[:-1]
+ last_line_break = True
+ elif not line_wrap:
+ while True:
+ i = temp.read(1)
+ if i == '\n':
+ last_line_break = True
+ if not i or i == '\n':
+ break
skip_line = False
for skipchar in skipchars:
if line.startswith(skipchar):
@@ -1064,4 +1077,4 @@ def get_file_peek(file_name, is_multi_byte=False, WIDTH=256, LINE_COUNT=5, skipc
if not skip_line:
lines.append(line)
count += 1
- return '\n'.join(lines)
+ return '\n'.join(lines) + ('\n' if last_line_break else '')
diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py
index 404f9c53a11..d35c83ffd4c 100644
--- a/lib/galaxy/datatypes/sniff.py
+++ b/lib/galaxy/datatypes/sniff.py
@@ -14,7 +14,11 @@ import sys
import tempfile
import zipfile
-from six import StringIO, text_type
+from six import (
+ PY3,
+ StringIO,
+ text_type,
+)
from six.moves import filter
from six.moves.urllib.request import urlopen
@@ -103,28 +107,40 @@ def stream_to_file(stream, suffix='', prefix='', dir=None, text=False, **kwd):
return stream_to_open_named_file(stream, fd, temp_name, **kwd)
-def convert_newlines(fname, in_place=True, tmp_dir=None, tmp_prefix="gxupload"):
+def convert_newlines(fname, in_place=True, tmp_dir=None, tmp_prefix="gxupload", block_size=128 * 1024, regexp=None):
"""
Converts in place a file from universal line endings
to Posix line endings.
-
- >>> fname = get_test_fname('temp.txt')
- >>> with open(fname, 'wt') as fh:
- ... _ = fh.write("1 2\\r3 4")
- >>> convert_newlines(fname, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir())
- (2, None)
- >>> open(fname).read()
- '1 2\\n3 4\\n'
"""
fd, temp_name = tempfile.mkstemp(prefix=tmp_prefix, dir=tmp_dir)
- with io.open(fd, mode="wt", encoding='utf-8') as fp:
- i = None
- for i, line in enumerate(io.open(fname, encoding='utf-8')):
- fp.write("%s\n" % line.rstrip("\r\n"))
- if i is None:
- i = 0
+ i = 0
+ if PY3:
+ NEWLINE_BYTE = 10
+ CR_BYTE = 13
else:
- i += 1
+ NEWLINE_BYTE = "\n"
+ CR_BYTE = "\r"
+ with io.open(fd, mode="wb") as fp, io.open(fname, mode="rb") as fi:
+ last_char = None
+ block = fi.read(block_size)
+ last_block = b""
+ while block:
+ if last_char == CR_BYTE and block.startswith(b"\n"):
+ # Last block ended with CR, new block startswith newline.
+ # Since we replace CR with newline in the previous iteration we skip the first byte
+ block = block[1:]
+ if block:
+ last_char = block[-1]
+ block = block.replace(b"\r\n", b"\n").replace(b"\r", b"\n")
+ if regexp:
+ block = b"\t".join(regexp.split(block))
+ fp.write(block)
+ i += block.count(b"\n")
+ last_block = block
+ block = fi.read(block_size)
+ if last_block and last_block[-1] != NEWLINE_BYTE:
+ i += 1
+ fp.write(b"\n")
if in_place:
shutil.move(temp_name, fname)
# Return number of lines in file.
@@ -133,47 +149,9 @@ def convert_newlines(fname, in_place=True, tmp_dir=None, tmp_prefix="gxupload"):
return (i, temp_name)
-def sep2tabs(fname, in_place=True, patt=r"\s+", tmp_dir=None, tmp_prefix="gxupload"):
+def convert_newlines_sep2tabs(fname, in_place=True, patt=br"[^\S\n]+", tmp_dir=None, tmp_prefix="gxupload"):
"""
- Transforms in place a 'sep' separated file to a tab separated one
-
- >>> fname = get_test_fname('temp.txt')
- >>> with open(fname, 'wt') as fh:
- ... _ = fh.write(u"1 2\\n3 4\\n")
- >>> sep2tabs(fname)
- (2, None)
- >>> open(fname).read()
- '1\\t2\\n3\\t4\\n'
- """
- regexp = re.compile(patt)
- fd, temp_name = tempfile.mkstemp(prefix=tmp_prefix, dir=tmp_dir)
- with io.open(fd, mode="wt", encoding='utf-8') as fp:
- i = None
- for i, line in enumerate(io.open(fname, encoding='utf-8')):
- if line.endswith("\r"):
- line = line.rstrip('\r')
- elems = regexp.split(line)
- fp.write(u"%s\r" % '\t'.join(elems))
- else:
- line = line.rstrip('\n')
- elems = regexp.split(line)
- fp.write(u"%s\n" % '\t'.join(elems))
- if i is None:
- i = 0
- else:
- i += 1
- if in_place:
- shutil.move(temp_name, fname)
- # Return number of lines in file.
- return (i, None)
- else:
- return (i, temp_name)
-
-
-def convert_newlines_sep2tabs(fname, in_place=True, patt=r"\s+", tmp_dir=None, tmp_prefix="gxupload"):
- """
- Combines above methods: convert_newlines() and sep2tabs()
- so that files do not need to be read twice
+ Converts newlines in a file to posix newlines and replaces spaces with tabs.
>>> fname = get_test_fname('temp.txt')
>>> with open(fname, 'wt') as fh:
@@ -184,18 +162,7 @@ def convert_newlines_sep2tabs(fname, in_place=True, patt=r"\s+", tmp_dir=None, t
'1\\t2\\n3\\t4\\n'
"""
regexp = re.compile(patt)
- fd, temp_name = tempfile.mkstemp(prefix=tmp_prefix, dir=tmp_dir)
- with io.open(fd, mode="wt", encoding='utf-8') as fp:
- for i, line in enumerate(io.open(fname, encoding='utf-8')):
- line = line.rstrip('\r\n')
- elems = regexp.split(line)
- fp.write(u"%s\n" % '\t'.join(elems))
- if in_place:
- shutil.move(temp_name, fname)
- # Return number of lines in file.
- return (i + 1, None)
- else:
- return (i + 1, temp_name)
+ return convert_newlines(fname, in_place, tmp_dir, tmp_prefix, regexp=regexp)
def iter_headers(fname_or_file_prefix, sep, count=60, comment_designator=None):
@@ -474,6 +441,9 @@ def guess_ext(fname, sniff_order, is_binary=False):
>>> fname = get_test_fname('1.mtx')
>>> guess_ext(fname, sniff_order)
'mtx'
+ >>> fname = get_test_fname('1imzml')
+ >>> guess_ext(fname, sniff_order) # This test case is ensuring doesn't throw exception, actual value could change if non-utf encoding handling improves.
+ 'data'
"""
file_prefix = FilePrefix(fname)
file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary)
@@ -555,8 +525,9 @@ def zip_single_fileobj(path):
class FilePrefix(object):
def __init__(self, filename):
- binary = False
+ non_utf8_error = None
compressed_format = None
+ contents_header_bytes = None
contents_header = None # First MAX_BYTES of the file.
truncated = False
# A future direction to optimize sniffing even more for sniffers at the top of the list
@@ -565,20 +536,23 @@ class FilePrefix(object):
# populates contents_header while providing a StringIO-like interface until the file is read
# but then would fallback to native string_io()
try:
- compressed_format, f = compression_utils.get_fileobj_raw(filename)
+ compressed_format, f = compression_utils.get_fileobj_raw(filename, "rb")
try:
- contents_header = f.read(SNIFF_PREFIX_BYTES)
- truncated = len(contents_header) == SNIFF_PREFIX_BYTES
+ contents_header_bytes = f.read(SNIFF_PREFIX_BYTES)
+ truncated = len(contents_header_bytes) == SNIFF_PREFIX_BYTES
+ contents_header = contents_header_bytes.decode("utf-8")
finally:
f.close()
- except UnicodeDecodeError:
- binary = True
+ except UnicodeDecodeError as e:
+ non_utf8_error = e
self.truncated = truncated
self.filename = filename
- self.binary = binary
+ self.non_utf8_error = non_utf8_error
+ self.binary = non_utf8_error is not None # obviously wrong
self.compressed_format = compressed_format
self.contents_header = contents_header
+ self.contents_header_bytes = contents_header_bytes
self._file_size = None
@property
@@ -588,8 +562,8 @@ class FilePrefix(object):
return self._file_size
def string_io(self):
- if self.binary:
- raise Exception("Attempting to create a StringIO object for binary data.")
+ if self.non_utf8_error is not None:
+ raise self.non_utf8_error
rval = StringIO(self.contents_header)
return rval
diff --git a/lib/galaxy/datatypes/test/0_nonewline b/lib/galaxy/datatypes/test/0_nonewline
new file mode 100644
index 00000000000..c227083464f
--- /dev/null
+++ b/lib/galaxy/datatypes/test/0_nonewline
@@ -0,0 +1 @@
+0
\ No newline at end of file
diff --git a/lib/galaxy/datatypes/test/1imzml b/lib/galaxy/datatypes/test/1imzml
new file mode 100644
index 00000000000..6b45d2a3667
--- /dev/null
+++ b/lib/galaxy/datatypes/test/1imzml
@@ -0,0 +1,380 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
diff --git a/lib/galaxy/datatypes/test/dosimzml b/lib/galaxy/datatypes/test/dosimzml
new file mode 100644
index 00000000000..6ff4f5c078d
--- /dev/null
+++ b/lib/galaxy/datatypes/test/dosimzml
@@ -0,0 +1,380 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
diff --git a/test/integration/test_datatype_upload.py b/test/integration/test_datatype_upload.py
index 4c7c742c534..58e45b0bb55 100644
--- a/test/integration/test_datatype_upload.py
+++ b/test/integration/test_datatype_upload.py
@@ -32,7 +32,7 @@ def find_datatype(registry, filename):
def collect_test_data(registry):
- test_files = os.listdir(TEST_FILE_DIR)
+ test_files = [f for f in os.listdir(TEST_FILE_DIR) if "." in f]
files = [os.path.join(TEST_FILE_DIR, f) for f in test_files]
datatypes = [find_datatype(registry, f) for f in test_files]
uploadable = [datatype.file_ext in registry.upload_file_formats for datatype in datatypes]
diff --git a/test/unit/datatypes/test_data.py b/test/unit/datatypes/test_data.py
index 969916d57b9..d1f85571714 100644
--- a/test/unit/datatypes/test_data.py
+++ b/test/unit/datatypes/test_data.py
@@ -10,4 +10,4 @@ from galaxy.util import galaxy_directory
def test_get_file_peek():
# should get the first 5 lines of the file without a trailing newline character
- assert get_file_peek(os.path.join(galaxy_directory(), 'test-data/1.tabular'), line_wrap=False) == 'chr22\t1000\tNM_17\nchr22\t2000\tNM_18\nchr10\t2200\tNM_10\nchr10\thap\ttest\nchr10\t1200\tNM_11'
+ assert get_file_peek(os.path.join(galaxy_directory(), 'test-data/1.tabular'), line_wrap=False) == 'chr22\t1000\tNM_17\nchr22\t2000\tNM_18\nchr10\t2200\tNM_10\nchr10\thap\ttest\nchr10\t1200\tNM_11\n'
diff --git a/test/unit/datatypes/test_sniff.py b/test/unit/datatypes/test_sniff.py
new file mode 100644
index 00000000000..faf75443281
--- /dev/null
+++ b/test/unit/datatypes/test_sniff.py
@@ -0,0 +1,79 @@
+import tempfile
+
+import pytest
+
+from galaxy.datatypes.sniff import (
+ convert_newlines,
+ convert_newlines_sep2tabs,
+ get_test_fname,
+)
+
+
+def assert_converts_to_1234_convert_sep2tabs(content, expected='1\t2\n3\t4\n'):
+ with tempfile.NamedTemporaryFile(delete=False, mode='w') as tf:
+ tf.write(content)
+ rval = convert_newlines_sep2tabs(tf.name, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir())
+ assert expected == open(tf.name).read()
+ assert rval == (2, None), rval
+
+
+def assert_converts_to_1234_convert(content, block_size=1024):
+ fname = get_test_fname('temp2.txt')
+ with open(fname, 'w') as fh:
+ fh.write(content)
+ rval = convert_newlines(fname, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir(), block_size=block_size)
+ actual_contents = open(fname).read()
+ assert '1 2\n3 4\n' == actual_contents, actual_contents
+ assert rval == (2, None), "rval != %s for %s" % (rval, content)
+
+
+@pytest.mark.parametrize('source,block_size', [
+ ("1 2\r3 4", None),
+ ("1 2\n3 4", None),
+ ("1 2\r\n3 4", None),
+ ("1 2\r3 4\r", None),
+ ("1 2\n3 4\n", None),
+ ("1 2\r\n3 4\r\n", None),
+ ("1 2\r3 4", 2),
+ ("1 2\n3 4", 2),
+ ("1 2\r\n3 4", 2),
+ ("1 2\r3 4\r", 2),
+ ("1 2\n3 4\n", 2),
+ ("1 2\r\n3 4\r\n", 2),
+ ("1 2\r3 4", 3),
+ ("1 2\n3 4", 3),
+ ("1 2\r\n3 4", 3),
+ ("1 2\r3 4\r", 3),
+ ("1 2\n3 4\n", 3),
+ ("1 2\r\n3 4\r\n", 3),
+])
+def test_convert_newlines(source, block_size):
+ # Verify ends with newline - with or without that on inputs - for any of
+ # \r \\n or \\r\\n newlines.
+ if block_size:
+ assert_converts_to_1234_convert(source, block_size)
+ else:
+ assert_converts_to_1234_convert(source)
+
+
+def test_convert_newlines_non_utf():
+ fname = get_test_fname("dosimzml")
+ rval = convert_newlines(fname, tmp_prefix="gxtest", tmp_dir=tempfile.gettempdir(), in_place=False)
+ new_file = rval[1]
+ assert open(new_file, "rb").read() == open(get_test_fname("1imzml"), "rb").read()
+
+
+@pytest.mark.parametrize('source,expected', [
+ ("1 2\n3 4\n", None),
+ ("1 2\n3 4\n", None),
+ ("1\t2\n3\t4\n", None),
+ ("1\t2\r3\t4\r", None),
+ ("1\t2\r\n3\t4\r\n", None),
+ ("1 2\r\n3 4\r\n", None),
+ ("1 2 \n3 4 \n", '1\t2\t\n3\t4\t\n'),
+])
+def test_convert_sep2tabs(source, expected):
+ if expected:
+ assert_converts_to_1234_convert_sep2tabs(source, expected=expected)
+ else:
+ assert_converts_to_1234_convert_sep2tabs(source)