mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Alternative stricter csv formats
This commit is contained in:
@@ -875,6 +875,9 @@ class CSV( TabularData ):
|
||||
Delimiter-separated table data.
|
||||
This includes CSV, TSV and other dialects understood by the
|
||||
Python 'csv' module https://docs.python.org/2/library/csv.html
|
||||
|
||||
WARNING: This type is BUGGY it is kept purely for backward compatability
|
||||
It will incorrectly sniff tab seperated files for which the get_meta method fails!
|
||||
"""
|
||||
delimiter = ','
|
||||
file_ext = 'csv' # File extension
|
||||
@@ -905,6 +908,7 @@ class CSV( TabularData ):
|
||||
return 'str'
|
||||
|
||||
def sniff( self, filename ):
|
||||
log.info ("all csv sniff called")
|
||||
""" Return True if if recognizes dialect and header. """
|
||||
if not csv.Sniffer().has_header(open(filename, 'r').read(self.peek_size)):
|
||||
return False
|
||||
@@ -942,6 +946,144 @@ class CSV( TabularData ):
|
||||
dataset.metadata.delimiter = reader.dialect.delimiter
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
class Base_CSV( CSV ):
|
||||
"""
|
||||
Delimiter-separated table data.
|
||||
This includes CSV, TSV and other dialects understood by the
|
||||
Python 'csv' module https://docs.python.org/2/library/csv.html
|
||||
Must be extended to define the dialect to use, strict_width: and file_ext.
|
||||
See Python module csv for documentation of dialect settings
|
||||
"""
|
||||
#dialect Set by subclass
|
||||
#file_ext Set by subclass
|
||||
#strict_width Set by subclass
|
||||
#If set sniff fails is a single row is incorrect.
|
||||
#Python's csv is more tollerant
|
||||
big_peek_size = 10240 # Large File chunk used for sniffing CSV dialect
|
||||
|
||||
def sniff( self, filename ):
|
||||
""" Return True if if recognizes dialect and header. """
|
||||
try:
|
||||
#check the dialect works
|
||||
reader = csv.reader(open(filename, 'r'), self.dialect)
|
||||
#Check we can read header and get columns
|
||||
header_row = reader.next()
|
||||
if len(header_row) < 2:
|
||||
#No columns so not seperated by this dialect.
|
||||
return False
|
||||
|
||||
#check all rows can be read as otherwise set_meta throws an exception
|
||||
if self.strict_width:
|
||||
num_columns = len(header_row)
|
||||
for data_row in reader:
|
||||
#All columns must be the same length
|
||||
if num_columns != len(data_row):
|
||||
return False
|
||||
else:
|
||||
#Check the next row as it is used by set_meta
|
||||
data_row = reader.next()
|
||||
if len(data_row) < 2:
|
||||
#No columns so not seperated by this dialect.
|
||||
return False
|
||||
#ignore the length in the rest
|
||||
for data_row in reader:
|
||||
pass
|
||||
|
||||
#Optional: Check Python's csv comes up with a similar dialect
|
||||
auto_dialect = csv.Sniffer().sniff(open(filename, 'r').read(self.big_peek_size))
|
||||
if (auto_dialect.delimiter != self.dialect.delimiter):
|
||||
return False
|
||||
if (auto_dialect.quotechar != self.dialect.quotechar):
|
||||
return False
|
||||
"""
|
||||
Not checking for other dialect options
|
||||
They may be mis detected from just the sample.
|
||||
Or not effect the read such as doublequote
|
||||
|
||||
Optional: Check for headers as in the past.
|
||||
Note No way around Python's csv calling Sniffer.sniff again.
|
||||
Note Without checking the dialect returned by sniff
|
||||
this test may be checking the wrong dialect.
|
||||
"""
|
||||
if not csv.Sniffer().has_header(open(filename, 'r').read(self.big_peek_size)):
|
||||
return False
|
||||
|
||||
return True
|
||||
except:
|
||||
#Not readable by Python's csv using this dialect
|
||||
return False
|
||||
|
||||
def set_meta( self, dataset, **kwd ):
|
||||
with open(dataset.file_name, 'r') as csvfile:
|
||||
# Parse file with the correct dialect
|
||||
reader = csv.reader(csvfile, self.dialect)
|
||||
data_row = None
|
||||
header_row = None
|
||||
try:
|
||||
header_row = reader.next()
|
||||
data_row = reader.next()
|
||||
for row in reader:
|
||||
pass
|
||||
except csv.Error as e:
|
||||
raise Exception('CSV reader error - line %d: %s' % (reader.line_num, e))
|
||||
|
||||
# Guess column types
|
||||
column_types = []
|
||||
for cell in data_row:
|
||||
column_types.append(self.guess_type(cell))
|
||||
|
||||
# Set metadata
|
||||
dataset.metadata.data_lines = reader.line_num - 1
|
||||
dataset.metadata.comment_lines = 1
|
||||
dataset.metadata.column_types = column_types
|
||||
dataset.metadata.columns = max( len( header_row ), len( data_row ) )
|
||||
dataset.metadata.column_names = header_row
|
||||
dataset.metadata.delimiter = reader.dialect.delimiter
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
class Excell_CSV( Base_CSV ):
|
||||
"""
|
||||
Comma separated table data.
|
||||
Only sniffs comma seperated files with at least 2 columns
|
||||
"""
|
||||
|
||||
def __init__(self, **kwd):
|
||||
Base_CSV.__init__( self, **kwd )
|
||||
self.dialect = csv.excel # This is the default
|
||||
#delimiter = ','
|
||||
#quotechar = '"'
|
||||
#doublequote = True
|
||||
#skipinitialspace = False
|
||||
self.file_ext = 'csv' # File extension
|
||||
self.strict_width = False # Previous csv type did not check column width
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
class Excell_TSV( Base_CSV ):
|
||||
"""
|
||||
Comma separated table data.
|
||||
Only sniff tab seperated files with at least two columns
|
||||
|
||||
Note: Use of this datatype is optional as the general tabular format will handle most tab seperated files.
|
||||
This datatye would only be required for dataset with tabs INSIDE double quotes.
|
||||
|
||||
This datatype currently does not support tsv files where the header has one column less to indicate first column is row names
|
||||
This kind of file is handled fine by tabular.
|
||||
"""
|
||||
|
||||
def __init__(self, **kwd):
|
||||
Base_CSV.__init__( self, **kwd )
|
||||
self.dialect = csv.excel_tab
|
||||
#delimiter = '\t'
|
||||
#quotechar = '"'
|
||||
#doublequote = True
|
||||
#skipinitialspace = False
|
||||
self.file_ext = 'tsv' # File extension
|
||||
self.strict_width = True # Leave files with different width to tabular
|
||||
|
||||
|
||||
class ConnectivityTable( Tabular ):
|
||||
edam_format = "format_3309"
|
||||
file_ext = "ct"
|
||||
|
||||
Reference in New Issue
Block a user