From bb769367ed1cbc66bcb7fa21ea17fdec0789e6c5 Mon Sep 17 00:00:00 2001 From: James Taylor Date: Thu, 8 Jul 2010 14:36:59 -0400 Subject: [PATCH 1/3] Make "loc files" more flexible by adding "tool data tables". These are configured at the application level. Specific tabular data files are specified in a application config file and bound to names, the tools then refer to these names. Thus users can configure where location files are located without modifying tool configs. Also: - Simpler column name configuration - Columns can be referred to by name in addition to index in all dynamic option filters - A data table can merge multiple files - Design can support other types of data files --- lib/galaxy/app.py | 3 + lib/galaxy/config.py | 1 + .../tools/parameters/dynamic_options.py | 97 +++++++++++++------ lib/galaxy/util/__init__.py | 16 +-- setup.sh | 1 + tools/maf/interval2maf.xml | 12 ++- tools/sr_mapping/bowtie_wrapper.xml | 3 + tools/sr_mapping/bwa_wrapper.xml | 3 + 8 files changed, 93 insertions(+), 43 deletions(-) diff --git a/lib/galaxy/app.py b/lib/galaxy/app.py index 6314ca7754a..0730ef6bc4b 100644 --- a/lib/galaxy/app.py +++ b/lib/galaxy/app.py @@ -2,6 +2,7 @@ import sys, os, atexit from galaxy import config, jobs, util, tools, web import galaxy.tools.search +import galaxy.tools.data from galaxy.web import security import galaxy.model import galaxy.datatypes.registry @@ -36,6 +37,8 @@ class UniverseApplication( object ): self.security = security.SecurityHelper( id_secret=self.config.id_secret ) # Tag handler self.tag_handler = GalaxyTagHandler() + # Tool data tables + self.tool_data_tables = galaxy.tools.data.ToolDataTableManager( self.config.tool_data_table_config_path ) # Initialize the tools self.toolbox = tools.ToolBox( self.config.tool_config, self.config.tool_path, self ) # Search support for tools diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index a32895ec51d..02188d05126 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -48,6 +48,7 @@ class Configuration( object ): self.tool_data_path = resolve_path( kwargs.get( "tool_data_path", "tool-data" ), os.getcwd() ) self.test_conf = resolve_path( kwargs.get( "test_conf", "" ), self.root ) self.tool_config = resolve_path( kwargs.get( 'tool_config_file', 'tool_conf.xml' ), self.root ) + self.tool_data_table_config_path = resolve_path( kwargs.get( 'tool_data_table_config_path', 'tool_data_table_conf.xml' ), self.root ) self.tool_secret = kwargs.get( "tool_secret", "" ) self.id_secret = kwargs.get( "id_secret", "USING THE DEFAULT IS NOT SECURE!" ) self.set_metadata_externally = string_as_bool( kwargs.get( "set_metadata_externally", "False" ) ) diff --git a/lib/galaxy/tools/parameters/dynamic_options.py b/lib/galaxy/tools/parameters/dynamic_options.py index 50837cfd08f..493ba53851e 100644 --- a/lib/galaxy/tools/parameters/dynamic_options.py +++ b/lib/galaxy/tools/parameters/dynamic_options.py @@ -46,9 +46,9 @@ class StaticValueFilter( Filter ): Filter.__init__( self, d_option, elem ) self.value = elem.get( "value", None ) assert self.value is not None, "Required 'value' attribute missing from filter" - self.column = elem.get( "column", None ) - assert self.column is not None, "Required 'column' attribute missing from filter, when loading from file" - self.column = int ( self.column ) + column = elem.get( "column", None ) + assert column is not None, "Required 'column' attribute missing from filter, when loading from file" + self.column = d_option.column_spec_to_index( column ) self.keep = string_as_bool( elem.get( "keep", 'True' ) ) def filter_options( self, options, trans, other_values ): rval = [] @@ -81,11 +81,11 @@ class DataMetaFilter( Filter ): d_option.has_dataset_dependencies = True self.key = elem.get( "key", None ) assert self.key is not None, "Required 'key' attribute missing from filter" - self.column = elem.get( "column", None ) - if self.column is None: + column = elem.get( "column", None ) + if column is None: assert self.dynamic_option.file_fields is None and self.dynamic_option.dataset_ref_name is None, "Required 'column' attribute missing from filter, when loading from file" else: - self.column = int ( self.column ) + self.column = d_option.column_spec_to_index( column ) self.multiple = string_as_bool( elem.get( "multiple", "False" ) ) self.separator = elem.get( "separator", "," ) def get_dependency_name( self ): @@ -141,9 +141,9 @@ class ParamValueFilter( Filter ): Filter.__init__( self, d_option, elem ) self.ref_name = elem.get( "ref", None ) assert self.ref_name is not None, "Required 'ref' attribute missing from filter" - self.column = elem.get( "column", None ) - assert self.column is not None, "Required 'column' attribute missing from filter" - self.column = int ( self.column ) + column = elem.get( "column", None ) + assert column is not None, "Required 'column' attribute missing from filter" + self.column = d_option.column_spec_to_index( column ) self.keep = string_as_bool( elem.get( "keep", 'True' ) ) def get_dependency_name( self ): return self.ref_name @@ -168,9 +168,9 @@ class UniqueValueFilter( Filter ): """ def __init__( self, d_option, elem ): Filter.__init__( self, d_option, elem ) - self.column = elem.get( "column", None ) - assert self.column is not None, "Required 'column' attribute missing from filter" - self.column = int ( self.column ) + column = elem.get( "column", None ) + assert column is not None, "Required 'column' attribute missing from filter" + self.column = d_option.column_spec_to_index( column ) def get_dependency_name( self ): return self.dynamic_option.dataset_ref_name def filter_options( self, options, trans, other_values ): @@ -196,9 +196,9 @@ class MultipleSplitterFilter( Filter ): def __init__( self, d_option, elem ): Filter.__init__( self, d_option, elem ) self.separator = elem.get( "separator", "," ) - self.columns = elem.get( "column", None ) - assert self.columns is not None, "Required 'columns' attribute missing from filter" - self.columns = [ int ( column ) for column in self.columns.split( "," ) ] + columns = elem.get( "column", None ) + assert columns is not None, "Required 'columns' attribute missing from filter" + self.columns = [ d_option.column_spec_to_index( column ) for column in columns.split( "," ) ] def filter_options( self, options, trans, other_values ): rval = [] for fields in options: @@ -302,9 +302,9 @@ class SortByColumnFilter( Filter ): """ def __init__( self, d_option, elem ): Filter.__init__( self, d_option, elem ) - self.column = elem.get( "column", None ) - assert self.column is not None, "Required 'column' attribute missing from filter" - self.column = int( self.column ) + column = elem.get( "column", None ) + assert column is not None, "Required 'column' attribute missing from filter" + self.column = d_option.column_spec_to_index( column ) def filter_options( self, options, trans, other_values ): rval = [] for i, fields in enumerate( options ): @@ -354,20 +354,25 @@ class DynamicOptions( object ): data_file = elem.get( 'from_file', None ) dataset_file = elem.get( 'from_dataset', None ) from_parameter = elem.get( 'from_parameter', None ) - if data_file is not None or dataset_file is not None or from_parameter is not None: - for column_elem in elem.findall( 'column' ): - name = column_elem.get( 'name', None ) - assert name is not None, "Required 'name' attribute missing from column def" - index = column_elem.get( 'index', None ) - assert index is not None, "Required 'index' attribute missing from column def" - index = int( index ) - self.columns[name] = index - if index > self.largest_index: - self.largest_index = index - assert 'value' in self.columns, "Required 'value' column missing from column def" - if 'name' not in self.columns: - self.columns['name'] = self.columns['value'] + tool_data_table_name = elem.get( 'from_data_table', None ) + + # Options are defined from a data table loaded by the app + self.tool_data_table = None + if tool_data_table_name: + app = tool_param.tool.app + assert tool_data_table_name in app.tool_data_tables, \ + "Data table named '%s' is required by tool but not configured" % tool_data_table_name + self.tool_data_table = app.tool_data_tables[ tool_data_table_name ] + # Column definitions are optional, but if provided override those from the table + if elem.find( "column" ) is not None: + self.parse_column_definitions( elem ) + else: + self.columns = self.tool_data_table.columns + # Options are defined by parsing tabular text data from an data file + # on disk, a dataset, or the value of another parameter + elif data_file is not None or dataset_file is not None or from_parameter is not None: + self.parse_column_definitions( elem ) if data_file is not None: data_file = data_file.strip() if not os.path.isabs( data_file ): @@ -388,6 +393,20 @@ class DynamicOptions( object ): # Load Validators for validator in elem.findall( 'validator' ): self.validators.append( validation.Validator.from_element( self.tool_param, validator ) ) + + def parse_column_definitions( self, elem ): + for column_elem in elem.findall( 'column' ): + name = column_elem.get( 'name', None ) + assert name is not None, "Required 'name' attribute missing from column def" + index = column_elem.get( 'index', None ) + assert index is not None, "Required 'index' attribute missing from column def" + index = int( index ) + self.columns[name] = index + if index > self.largest_index: + self.largest_index = index + assert 'value' in self.columns, "Required 'value' column missing from column def" + if 'name' not in self.columns: + self.columns['name'] = self.columns['value'] def parse_file_fields( self, reader ): rval = [] @@ -421,6 +440,8 @@ class DynamicOptions( object ): assert dataset is not None, "Required dataset '%s' missing from input" % self.dataset_ref_name if not dataset: return [] #no valid dataset in history options = self.parse_file_fields( open( dataset.file_name ) ) + elif self.tool_data_table: + options = self.tool_data_table.get_fields() else: options = list( self.file_fields ) for filter in self.filters: @@ -429,7 +450,7 @@ class DynamicOptions( object ): def get_options( self, trans, other_values ): rval = [] - if self.file_fields is not None or self.dataset_ref_name is not None: + if self.file_fields is not None or self.tool_data_table is not None or self.dataset_ref_name is not None: options = self.get_fields( trans, other_values ) for fields in options: rval.append( ( fields[self.columns['name']], fields[self.columns['value']], False ) ) @@ -437,3 +458,15 @@ class DynamicOptions( object ): for filter in self.filters: rval = filter.filter_options( rval, trans, other_values ) return rval + + def column_spec_to_index( self, column_spec ): + """ + Convert a column specification (as read from the config file), to an + index. A column specification can just be a number, a column name, or + a column alias. + """ + # Name? + if column_spec in self.columns: + return self.columns[column_spec] + # Int? + return int( column_spec ) diff --git a/lib/galaxy/util/__init__.py b/lib/galaxy/util/__init__.py index 2b5671802f1..8d0a4767535 100644 --- a/lib/galaxy/util/__init__.py +++ b/lib/galaxy/util/__init__.py @@ -231,13 +231,17 @@ def rst_to_html( s ): log.warn( str ) return docutils.core.publish_string( s, writer=HTMLFragWriter(), settings_overrides=dict( warning_stream=FakeStream() ) ) -def xml_text(root, name): +def xml_text(root, name=None): """Returns the text inside an element""" - # Try attribute first - val = root.get(name) - if val: return val - # Then try as element - elem = root.find(name) + if name is not None: + # Try attribute first + val = root.get(name) + if val: + return val + # Then try as element + elem = root.find(name) + else: + elem = root if elem is not None and elem.text: text = ''.join(elem.text.splitlines()) return text.strip() diff --git a/setup.sh b/setup.sh index f51e968df68..36e8d995da7 100644 --- a/setup.sh +++ b/setup.sh @@ -7,6 +7,7 @@ SAMPLES=" datatypes_conf.xml.sample reports_wsgi.ini.sample tool_conf.xml.sample +tool_data_table_conf.xml.sample universe_wsgi.ini.sample tool-data/alignseq.loc.sample tool-data/annotation_profiler_options.xml.sample diff --git a/tools/maf/interval2maf.xml b/tools/maf/interval2maf.xml index f3e0f69dd48..eaf6f9c8689 100644 --- a/tools/maf/interval2maf.xml +++ b/tools/maf/interval2maf.xml @@ -32,22 +32,24 @@ - + + + - + - - + + diff --git a/tools/sr_mapping/bowtie_wrapper.xml b/tools/sr_mapping/bowtie_wrapper.xml index e2db54c5a93..98031b7e383 100644 --- a/tools/sr_mapping/bowtie_wrapper.xml +++ b/tools/sr_mapping/bowtie_wrapper.xml @@ -192,10 +192,13 @@ + + diff --git a/tools/sr_mapping/bwa_wrapper.xml b/tools/sr_mapping/bwa_wrapper.xml index 1474f4bb75a..f33a461cce5 100644 --- a/tools/sr_mapping/bwa_wrapper.xml +++ b/tools/sr_mapping/bwa_wrapper.xml @@ -34,10 +34,13 @@ + + From c664344f11329407d8c10cbdac1c38e968d778bd Mon Sep 17 00:00:00 2001 From: James Taylor Date: Thu, 8 Jul 2010 15:01:19 -0400 Subject: [PATCH 2/3] Two missing files from previous commit (tool data tables) --- lib/galaxy/tools/data/__init__.py | 131 ++++++++++++++++++++++++++++++ tool_data_table_conf.xml.sample | 6 ++ 2 files changed, 137 insertions(+) create mode 100644 lib/galaxy/tools/data/__init__.py create mode 100644 tool_data_table_conf.xml.sample diff --git a/lib/galaxy/tools/data/__init__.py b/lib/galaxy/tools/data/__init__.py new file mode 100644 index 00000000000..07a7fa37c11 --- /dev/null +++ b/lib/galaxy/tools/data/__init__.py @@ -0,0 +1,131 @@ +""" +Manage tool data tables, which store (at the application level) data that is +used by tools, for example in the generation of dynamic options. Tables are +loaded and stored by names which tools use to refer to them. This allows +users to configure data tables for a local Galaxy instance without needing +to modify the tool configurations. +""" + +import logging, sys, os.path +from galaxy import util + +log = logging.getLogger( __name__ ) + +class ToolDataTableManager( object ): + """ + Manages a collection of tool data tables + """ + + def __init__( self, config_filename=None ): + self.data_tables = {} + if config_filename: + self.add_from_config_file( config_filename ) + + def __getitem__( self, key ): + return self.data_tables.__getitem__( key ) + + def __contains__( self, key ): + return self.data_tables.__contains__( key ) + + def add_from_config_file( self, config_filename ): + tree = util.parse_xml( config_filename ) + root = tree.getroot() + for table_elem in root.findall( 'table' ): + type = table_elem.get( 'type', 'tabular' ) + assert type in tool_data_table_types, "Unknown data table type '%s'" % type + table = tool_data_table_types[ type ]( table_elem ) + self.data_tables[ table.name ] = table + log.debug( "Loaded tool data table '%s", table.name ) + print >> sys.stderr, repr( self.data_tables ) + +class ToolDataTable( object ): + def __init__( self, config_element ): + self.name = config_element.get( 'name' ) + +class TabularToolDataTable( ToolDataTable ): + """ + Data stored in a tabular / separated value format on disk, allows multiple + files to be merged but all must have the same column definitions. + + + + + +
+ """ + + type_key = 'tabular' + + def __init__( self, config_element ): + super( TabularToolDataTable, self ).__init__( config_element ) + self.configure_and_load( config_element ) + + def configure_and_load( self, config_element ): + """ + Configure and load table from an XML element. + """ + self.separator = config_element.get( 'separator', '\t' ) + self.comment_char = config_element.get( 'comment_char', '#' ) + # Configure columns + self.parse_column_spec( config_element ) + # Read every file + all_rows = [] + for file_element in config_element.findall( 'file' ): + filename = file_element.get( 'path' ) + assert os.path.exists( filename ), \ + "Cannot find index file '%s' for tool data table '%s'" % ( filename, self.name ) + all_rows.extend( self.parse_file_fields( open( filename ) ) ) + self.data = all_rows + + def get_fields( self ): + return self.data + + def parse_column_spec( self, config_element ): + """ + Parse column definitions, which can either be a set of 'column' elements + with a name and index (as in dynamic options config), or a shorthand + comma separated list of names in order as the text of a 'column_names' + element. + + A column named 'value' is required. + """ + self.columns = {} + if config_element.find( 'columns' ) is not None: + column_names = util.xml_text( config_element.find( 'columns' ) ) + column_names = [ n.strip() for n in column_names.split( ',' ) ] + for index, name in enumerate( column_names ): + self.columns[ name ] = index + self.largest_index = index + else: + for column_elem in config_element.findall( 'column' ): + name = column_elem.get( 'name', None ) + assert name is not None, "Required 'name' attribute missing from column def" + index = column_elem.get( 'index', None ) + assert index is not None, "Required 'index' attribute missing from column def" + index = int( index ) + self.columns[name] = index + if index > self.largest_index: + self.largest_index = index + assert 'value' in self.columns, "Required 'value' column missing from column def" + if 'name' not in self.columns: + self.columns['name'] = self.columns['value'] + + def parse_file_fields( self, reader ): + """ + Parse separated lines from file and return a list of tuples. + + TODO: Allow named access to fields using the column names. + """ + rval = [] + for line in reader: + if line.lstrip().startswith( self.comment_char ): + continue + line = line.rstrip( "\n\r" ) + if line: + fields = line.split( self.separator ) + if self.largest_index < len( fields ): + rval.append( fields ) + return rval + +# Registry of tool data types by type_key +tool_data_table_types = dict( [ ( cls.type_key, cls ) for cls in [ TabularToolDataTable ] ] ) diff --git a/tool_data_table_conf.xml.sample b/tool_data_table_conf.xml.sample new file mode 100644 index 00000000000..21709578c3d --- /dev/null +++ b/tool_data_table_conf.xml.sample @@ -0,0 +1,6 @@ + + + name, value, dbkey, species + +
+
From 4fa4d17c888ddbfffd6549bcf8025fdb3348c225 Mon Sep 17 00:00:00 2001 From: James Taylor Date: Thu, 8 Jul 2010 15:21:38 -0400 Subject: [PATCH 3/3] Update tool data table config file --- tool_data_table_conf.xml.sample | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/tool_data_table_conf.xml.sample b/tool_data_table_conf.xml.sample index 21709578c3d..f7fb377f1af 100644 --- a/tool_data_table_conf.xml.sample +++ b/tool_data_table_conf.xml.sample @@ -1,6 +1,17 @@ + - name, value, dbkey, species - + name, value, dbkey, species + +
+ + + name, value + +
+ + + name, value +