diff --git a/config/datatypes_conf.xml.sample b/config/datatypes_conf.xml.sample
index 31bc824eaa6..a9d714cdf53 100644
--- a/config/datatypes_conf.xml.sample
+++ b/config/datatypes_conf.xml.sample
@@ -52,9 +52,8 @@
-
-
+
@@ -574,6 +573,7 @@
+
diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py
index ff9584d8b47..ebbbec01ea7 100644
--- a/lib/galaxy/datatypes/sniff.py
+++ b/lib/galaxy/datatypes/sniff.py
@@ -325,6 +325,9 @@ def guess_ext( fname, sniff_order, is_multi_byte=False ):
>>> fname = get_test_fname('test.mz5')
>>> guess_ext(fname, sniff_order)
'h5'
+ >>> fname = get_test_fname('issue1818.tabular')
+ >>> guess_ext(fname, sniff_order)
+ 'tabular'
>>> fname = get_test_fname('drugbank_drugs.cml')
>>> guess_ext(fname, sniff_order)
'cml'
@@ -344,6 +347,7 @@ def guess_ext( fname, sniff_order, is_multi_byte=False ):
>>> guess_ext(fname, sniff_order)
'pdb'
"""
+ file_ext = None
for datatype in sniff_order:
"""
Some classes may not have a sniff function, which is ok. In fact, the
@@ -355,9 +359,19 @@ def guess_ext( fname, sniff_order, is_multi_byte=False ):
"""
try:
if datatype.sniff( fname ):
- return datatype.file_ext
+ file_ext = datatype.file_ext
+ break
except:
pass
+ # Ugly hack for tsv vs tabular sniffing, we want to prefer tabular
+ # to tsv but it doesn't have a sniffer - is TSV was sniffed just check
+ # if it is an okay tabular and use that instead.
+ if file_ext == 'tsv':
+ if is_column_based( fname, '\t', 1, is_multi_byte=is_multi_byte ):
+ file_ext = 'tabular'
+ if file_ext is not None:
+ return file_ext
+
headers = get_headers( fname, None )
is_binary = False
if is_multi_byte:
diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py
index 817d58750c6..4c5038c17ed 100644
--- a/lib/galaxy/datatypes/tabular.py
+++ b/lib/galaxy/datatypes/tabular.py
@@ -875,15 +875,18 @@ class FeatureLocationIndex( Tabular ):
@dataproviders.decorators.has_dataproviders
-class CSV( TabularData ):
+class BaseCSV( TabularData ):
"""
Delimiter-separated table data.
This includes CSV, TSV and other dialects understood by the
Python 'csv' module https://docs.python.org/2/library/csv.html
+ Must be extended to define the dialect to use, strict_width: and file_ext.
+ See Python module csv for documentation of dialect settings
"""
delimiter = ','
file_ext = 'csv' # File extension
peek_size = 1024 # File chunk used for sniffing CSV dialect
+ big_peek_size = 10240 # Large File chunk used for sniffing CSV dialect
def is_int( self, column_text ):
try:
@@ -911,18 +914,60 @@ class CSV( TabularData ):
def sniff( self, filename ):
""" Return True if if recognizes dialect and header. """
- if not csv.Sniffer().has_header(open(filename, 'r').read(self.peek_size)):
+ try:
+ # check the dialect works
+ reader = csv.reader(open(filename, 'r'), self.dialect)
+ # Check we can read header and get columns
+ header_row = reader.next()
+ if len(header_row) < 2:
+ # No columns so not separated by this dialect.
+ return False
+
+ # check all rows can be read as otherwise set_meta throws an exception
+ if self.strict_width:
+ num_columns = len(header_row)
+ for data_row in reader:
+ # All columns must be the same length
+ if num_columns != len(data_row):
+ return False
+ else:
+ # Check the next row as it is used by set_meta
+ data_row = reader.next()
+ if len(data_row) < 2:
+ # No columns so not separated by this dialect.
+ return False
+ # ignore the length in the rest
+ for data_row in reader:
+ pass
+
+ # Optional: Check Python's csv comes up with a similar dialect
+ auto_dialect = csv.Sniffer().sniff(open(filename, 'r').read(self.big_peek_size))
+ if (auto_dialect.delimiter != self.dialect.delimiter):
+ return False
+ if (auto_dialect.quotechar != self.dialect.quotechar):
+ return False
+ """
+ Not checking for other dialect options
+ They may be mis detected from just the sample.
+ Or not effect the read such as doublequote
+
+ Optional: Check for headers as in the past.
+ Note No way around Python's csv calling Sniffer.sniff again.
+ Note Without checking the dialect returned by sniff
+ this test may be checking the wrong dialect.
+ """
+ if not csv.Sniffer().has_header(open(filename, 'r').read(self.big_peek_size)):
+ return False
+
+ return True
+ except:
+ # Not readable by Python's csv using this dialect
return False
- # Fetch at least three consecutive lines to be reasonably sure
- reader = csv.reader(open(filename, 'r'))
- for i in range(0, 3):
- reader.next()
- return True
def set_meta( self, dataset, **kwd ):
with open(dataset.file_name, 'r') as csvfile:
- # Parse file
- reader = csv.reader(csvfile)
+ # Parse file with the correct dialect
+ reader = csv.reader(csvfile, self.dialect)
data_row = None
header_row = None
try:
@@ -947,6 +992,40 @@ class CSV( TabularData ):
dataset.metadata.delimiter = reader.dialect.delimiter
+@dataproviders.decorators.has_dataproviders
+class CSV( BaseCSV ):
+ """
+ Comma separated table data.
+ Only sniffs comma separated files with at least 2 columns
+ """
+
+ def __init__(self, **kwd):
+ BaseCSV.__init__( self, **kwd )
+ self.dialect = csv.excel # This is the default
+ self.file_ext = 'csv' # File extension
+ self.strict_width = False # Previous csv type did not check column width
+
+
+@dataproviders.decorators.has_dataproviders
+class TSV( BaseCSV ):
+ """
+ Comma separated table data.
+ Only sniff tab separated files with at least two columns
+
+ Note: Use of this datatype is optional as the general tabular format will handle most tab separated files.
+ This datatype would only be required for dataset with tabs INSIDE double quotes.
+
+ This datatype currently does not support tsv files where the header has one column less to indicate first column is row names
+ This kind of file is handled fine by tabular.
+ """
+
+ def __init__(self, **kwd):
+ BaseCSV.__init__( self, **kwd )
+ self.dialect = csv.excel_tab
+ self.file_ext = 'tsv' # File extension
+ self.strict_width = True # Leave files with different width to tabular
+
+
class ConnectivityTable( Tabular ):
edam_format = "format_3309"
file_ext = "ct"
diff --git a/lib/galaxy/datatypes/test/issue1818.tabular b/lib/galaxy/datatypes/test/issue1818.tabular
new file mode 100644
index 00000000000..cd575ebcab0
--- /dev/null
+++ b/lib/galaxy/datatypes/test/issue1818.tabular
@@ -0,0 +1,84 @@
+Name Major Score
+Ignatius Engineering 83
+Austin Life-Sciences 91
+Zackery Engineering 54
+Marques Arts 58
+Darren Business 94
+Darius Social-Sciences 51
+Thanh Engineering 53
+Joe'Quann Engineering 75
+Bryan Arts 68
+Devin Engineering 92
+Joseph Social-Sciences 61
+Joshua Life-Sciences 14
+Ja'Won Social-Sciences 37
+Tyreque Arts 74
+Sage Arts 55
+Antonio Engineering 88
+Michael Engineering 39
+Randy Social-Sciences 68
+Dilan Health-Medicine 84
+Omar Engineering 99
+Zachary Arts 80
+Faison Engineering 47
+Angel Health-Medicine 100
+Gabriel Health-Medicine 100
+John Life-Sciences 70
+Leonard Business 87
+Juan Business 79
+Jonathan Health-Medicine 100
+Christopher Life-Sciences 59
+Brandon Life-Sciences 72
+D'Angelo Health-Medicine 90
+Justin Social-Sciences 90
+Israel Health-Medicine 81
+William Arts 46
+David Social-Sciences 69
+Drake Social-Sciences 59
+Drake Social-Sciences 76
+Nathan Arts 71
+Trevon Arts 74
+Aaron Business 83
+Daniel Health-Medicine 91
+Kevin Health-Medicine 100
+Antonio Engineering 56
+Donovan Arts 75
+Kerris Business 82
+Andre Health-Medicine 72
+Dakota Business 83
+Aaron Life-Sciences 58
+Walter Arts 75
+Isaiah Arts 80
+Christian Life-Sciences 67
+Dalton Health-Medicine 100
+Jesse Social-Sciences 32
+Diego Health-Medicine 82
+Nathen Life-Sciences 46
+Anthony Life-Sciences 32
+Christian Business 88
+David Business 92
+Avery Engineering 51
+Paul Arts 63
+Derek Arts 60
+Levi Arts 76
+Lance Social-Sciences 65
+Sonny Engineering 50
+Shawn Arts 65
+Leonardo Engineering 78
+Yeng Life-Sciences 39
+Leroy Social-Sciences 74
+Gurnam Life-Sciences 66
+Fernando Arts 78
+Williams Social-Sciences 62
+Roberto Arts 65
+Teriuse Business 94
+Nathaniel Arts 88
+Chase Social-Sciences 27
+Caleb Business 87
+Tysza Business 92
+Nico Arts 59
+Manuel Social-Sciences 61
+Patrick Health-Medicine 92
+Peter Health-Medicine 86
+Allen Life-Sciences 50
+Joel Social-Sciences 72