diff --git a/lib/galaxy/datatypes/tabular.py b/lib/galaxy/datatypes/tabular.py index 2c4bf3be072..fbb66d6da9c 100644 --- a/lib/galaxy/datatypes/tabular.py +++ b/lib/galaxy/datatypes/tabular.py @@ -875,6 +875,9 @@ class CSV( TabularData ): Delimiter-separated table data. This includes CSV, TSV and other dialects understood by the Python 'csv' module https://docs.python.org/2/library/csv.html + + WARNING: This type is BUGGY it is kept purely for backward compatability + It will incorrectly sniff tab seperated files for which the get_meta method fails! """ delimiter = ',' file_ext = 'csv' # File extension @@ -905,6 +908,7 @@ class CSV( TabularData ): return 'str' def sniff( self, filename ): + log.info ("all csv sniff called") """ Return True if if recognizes dialect and header. """ if not csv.Sniffer().has_header(open(filename, 'r').read(self.peek_size)): return False @@ -942,6 +946,144 @@ class CSV( TabularData ): dataset.metadata.delimiter = reader.dialect.delimiter +@dataproviders.decorators.has_dataproviders +class Base_CSV( CSV ): + """ + Delimiter-separated table data. + This includes CSV, TSV and other dialects understood by the + Python 'csv' module https://docs.python.org/2/library/csv.html + Must be extended to define the dialect to use, strict_width: and file_ext. + See Python module csv for documentation of dialect settings + """ + #dialect Set by subclass + #file_ext Set by subclass + #strict_width Set by subclass + #If set sniff fails is a single row is incorrect. + #Python's csv is more tollerant + big_peek_size = 10240 # Large File chunk used for sniffing CSV dialect + + def sniff( self, filename ): + """ Return True if if recognizes dialect and header. """ + try: + #check the dialect works + reader = csv.reader(open(filename, 'r'), self.dialect) + #Check we can read header and get columns + header_row = reader.next() + if len(header_row) < 2: + #No columns so not seperated by this dialect. + return False + + #check all rows can be read as otherwise set_meta throws an exception + if self.strict_width: + num_columns = len(header_row) + for data_row in reader: + #All columns must be the same length + if num_columns != len(data_row): + return False + else: + #Check the next row as it is used by set_meta + data_row = reader.next() + if len(data_row) < 2: + #No columns so not seperated by this dialect. + return False + #ignore the length in the rest + for data_row in reader: + pass + + #Optional: Check Python's csv comes up with a similar dialect + auto_dialect = csv.Sniffer().sniff(open(filename, 'r').read(self.big_peek_size)) + if (auto_dialect.delimiter != self.dialect.delimiter): + return False + if (auto_dialect.quotechar != self.dialect.quotechar): + return False + """ + Not checking for other dialect options + They may be mis detected from just the sample. + Or not effect the read such as doublequote + + Optional: Check for headers as in the past. + Note No way around Python's csv calling Sniffer.sniff again. + Note Without checking the dialect returned by sniff + this test may be checking the wrong dialect. + """ + if not csv.Sniffer().has_header(open(filename, 'r').read(self.big_peek_size)): + return False + + return True + except: + #Not readable by Python's csv using this dialect + return False + + def set_meta( self, dataset, **kwd ): + with open(dataset.file_name, 'r') as csvfile: + # Parse file with the correct dialect + reader = csv.reader(csvfile, self.dialect) + data_row = None + header_row = None + try: + header_row = reader.next() + data_row = reader.next() + for row in reader: + pass + except csv.Error as e: + raise Exception('CSV reader error - line %d: %s' % (reader.line_num, e)) + + # Guess column types + column_types = [] + for cell in data_row: + column_types.append(self.guess_type(cell)) + + # Set metadata + dataset.metadata.data_lines = reader.line_num - 1 + dataset.metadata.comment_lines = 1 + dataset.metadata.column_types = column_types + dataset.metadata.columns = max( len( header_row ), len( data_row ) ) + dataset.metadata.column_names = header_row + dataset.metadata.delimiter = reader.dialect.delimiter + + +@dataproviders.decorators.has_dataproviders +class Excell_CSV( Base_CSV ): + """ + Comma separated table data. + Only sniffs comma seperated files with at least 2 columns + """ + + def __init__(self, **kwd): + Base_CSV.__init__( self, **kwd ) + self.dialect = csv.excel # This is the default + #delimiter = ',' + #quotechar = '"' + #doublequote = True + #skipinitialspace = False + self.file_ext = 'csv' # File extension + self.strict_width = False # Previous csv type did not check column width + + +@dataproviders.decorators.has_dataproviders +class Excell_TSV( Base_CSV ): + """ + Comma separated table data. + Only sniff tab seperated files with at least two columns + + Note: Use of this datatype is optional as the general tabular format will handle most tab seperated files. + This datatye would only be required for dataset with tabs INSIDE double quotes. + + This datatype currently does not support tsv files where the header has one column less to indicate first column is row names + This kind of file is handled fine by tabular. + """ + + def __init__(self, **kwd): + Base_CSV.__init__( self, **kwd ) + self.dialect = csv.excel_tab + #delimiter = '\t' + #quotechar = '"' + #doublequote = True + #skipinitialspace = False + self.file_ext = 'tsv' # File extension + self.strict_width = True # Leave files with different width to tabular + + class ConnectivityTable( Tabular ): edam_format = "format_3309" file_ext = "ct"