From e25946e9b97127f29709407d13622af44bbc516f Mon Sep 17 00:00:00 2001 From: Guruprasad Anada Date: Tue, 13 Jan 2009 12:18:08 -0500 Subject: [PATCH 01/18] Tool to find the least common ancestor added under Taxonomy section. --- tools/taxonomy/lca.py | 145 +++++++++++++++++++++++++++++++++++++++++ tools/taxonomy/lca.xml | 12 ++++ 2 files changed, 157 insertions(+) create mode 100644 tools/taxonomy/lca.py create mode 100644 tools/taxonomy/lca.xml diff --git a/tools/taxonomy/lca.py b/tools/taxonomy/lca.py new file mode 100644 index 00000000000..beb356f160b --- /dev/null +++ b/tools/taxonomy/lca.py @@ -0,0 +1,145 @@ +#!/usr/bin/env python +#Guruprasad Ananda +""" +This tool provides the SQL "group by" functionality. +""" +import sys, string, re, commands, tempfile, random +#from rpy import * + +def stop_err(msg): + sys.stderr.write(msg) + sys.exit() + +def main(): + try: + inputfile = sys.argv[1] + outfile = sys.argv[2] + except: + stop_err("Syntax error: Use correct syntax: program infile outfile") + group_col = 0 + + tmpfile = tempfile.NamedTemporaryFile() + + try: + """ + The -k option for the Posix sort command is as follows: + -k, --key=POS1[,POS2] + start a key at POS1, end it at POS2 (origin 1) + In other words, column positions start at 1 rather than 0, so + we need to add 1 to group_col. + if POS2 is not specified, the newer versions of sort will consider the entire line for sorting. To prevent this, we set POS2=POS1. + """ + command_line = "sort -f -k " + str(group_col+1) +"," + str(group_col+1) + " -o " + tmpfile.name + " " + inputfile + except Exception, exc: + stop_err( 'Initialization error -> %s' %str(exc) ) + + error_code, stdout = commands.getstatusoutput(command_line) + + if error_code != 0: + stop_err( "Sorting input dataset resulted in error: %s: %s" %( error_code, stdout )) + + prev_item = "" + prev_vals = [] + remaining_vals = [] + skipped_lines = 0 + first_invalid_line = 0 + invalid_line = '' + invalid_value = '' + invalid_column = 0 + fout = open(outfile, "w") + cols = range(1,25) + block_valid = False + + for ii, line in enumerate( file( tmpfile.name )): + if line and not line.startswith( '#' ): + line = line.rstrip( '\r\n' ) + try: + fields = line.split("\t") + item = fields[group_col] + if prev_item != "": + # At this level, we're grouping on values (item and prev_item) in group_col + if item == prev_item: + # Keep iterating and storing values until a new value is encountered. + if block_valid: + for i, col in enumerate(cols): + if col >= 3: + prev_vals[i].append(fields[col].strip()) + if len(set(prev_vals[i])) > 1: + block_valid = False + break + + else: + """ + When a new value is encountered, write the previous value and the + corresponding aggregate values into the output file. This works + due to the sort on group_col we've applied to the data above. + """ + out_list = ['']*25 + out_list[0] = str(prev_item) + out_list[1] = str(prev_vals[0][0]) + out_list[2] = str(prev_vals[1][0]) + out_list[24] = str(prev_vals[23][0]) + #print >> fout, prev_vals + #sys.exit() + for k, col in enumerate(cols): + if col >= 3 and col < 24: + if len(set(prev_vals[k])) == 1: + out_list[col] = prev_vals[k][0] + else: + break + while k < 23: + out_list[k+1] = 'n' + k += 1 + + print >>fout, '\t'.join(out_list) + + block_valid = True + prev_item = item + prev_vals = [] + for col in cols: + val_list = [] + val_list.append(fields[col].strip()) + prev_vals.append(val_list) + + else: + # This only occurs once, right at the start of the iteration. + block_valid = True + prev_item = item #groupby item + for col in cols: #everyting else + val_list = [] + val_list.append(fields[col].strip()) + prev_vals.append(val_list) + + except Exception, exc: + skipped_lines += 1 + if not first_invalid_line: + first_invalid_line = ii+1 + else: + skipped_lines += 1 + if not first_invalid_line: + first_invalid_line = ii+1 + + # Handle the last grouped value + out_list = ['']*25 + out_list[0] = str(prev_item) + out_list[1] = str(prev_vals[0][0]) + out_list[2] = str(prev_vals[1][0]) + out_list[24] = str(prev_vals[23][0]) + for k, col in enumerate(cols): + if col >= 3 and col < 24: + if len(set(prev_vals[k])) == 1: + out_list[col] = prev_vals[k][0] + else: + break + while k < 23: + out_list[k+1] = 'n' + k += 1 + + print >>fout, '\t'.join(out_list) + + if skipped_lines > 0: + msg= "Skipped %d invalid lines starting with line %d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column ) + print msg + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/tools/taxonomy/lca.xml b/tools/taxonomy/lca.xml new file mode 100644 index 00000000000..8ba556c50dd --- /dev/null +++ b/tools/taxonomy/lca.xml @@ -0,0 +1,12 @@ + + + + lca.py $input1 $out_file1 + + + + + + + + From 1b3978609f8a0f29138d913a691e3af45eb9218c Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Tue, 13 Jan 2009 16:36:46 -0500 Subject: [PATCH 02/18] Add SOLiD Quality datatype and add a sniffer for Color Space FASTA. --- lib/galaxy/datatypes/qualityscore.py | 40 +++++++++++++++++++++++++++- lib/galaxy/datatypes/sequence.py | 39 +++++++++++++++++++++------ 2 files changed, 70 insertions(+), 9 deletions(-) diff --git a/lib/galaxy/datatypes/qualityscore.py b/lib/galaxy/datatypes/qualityscore.py index cfd5b280adb..766761e82dc 100644 --- a/lib/galaxy/datatypes/qualityscore.py +++ b/lib/galaxy/datatypes/qualityscore.py @@ -28,5 +28,43 @@ class QualityScore ( data.Text ): except: return "Quality score file (%s)" % ( data.nice_size( dataset.get_size() ) ) +class SolidQualityScore( data.Text ): + """ + Quality scores generated by ABI SOLiD + """ + file_ext = "solidqual" - \ No newline at end of file + def set_peek( self, dataset ): + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + def sniff( self, filename ): + """ + >>> fname = get_test_fname( 'sequence.fasta' ) + >>> SolidQualityScore().sniff( fname ) + False + >>> fname = get_test_fname( 'sequence.solidqual' ) + >>> SolidQualityScore().sniff( fname ) + True + """ + try: + fh = open( filename ) + while True: + line = fh.readline() + if not line: + break #EOF + line = line.strip() + if line and not line.startswith( '#' ): #first non-empty non-comment line + if line.startswith( '>' ): + line = fh.readline().strip() + if line == '' or line.startswith( '>' ): + break + try: + [ int( x ) for x in line.split() ] + except: + break + return True + else: + break #we found a non-empty line, but it's not a header + except: + pass + return False diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 2ec55345cc3..0033ff79a1f 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -5,6 +5,7 @@ Image classes import data import logging import re +import string from cgi import escape from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes import metadata @@ -94,15 +95,37 @@ class csFasta( Sequence ): Color-space sequence: >2_15_85_F3 T213021013012303002332212012112221222112212222 - - TODO: - add sniff function - """ - - return False - - + >>> fname = get_test_fname( 'sequence.fasta' ) + >>> csFasta().sniff( fname ) + False + >>> fname = get_test_fname( 'sequence.csfasta' ) + >>> csFasta().sniff( fname ) + True + """ + try: + fh = open( filename ) + while True: + line = fh.readline() + if not line: + break #EOF + line = line.strip() + if line and not line.startswith( '#' ): #first non-empty non-comment line + if line.startswith( '>' ): + line = fh.readline().strip() + if line == '' or line.startswith( '>' ): + break + elif line[0] not in string.ascii_uppercase: + return False + elif len( line ) > 1 and not re.search( '^\d+$', line[1:] ): + return False + return True + else: + break #we found a non-empty line, but it's not a header + except: + pass + return False + class FastqSolexa( Sequence ): """Class representing a FASTQ sequence ( the Solexa variant )""" file_ext = "fastqsolexa" From e9d2b617be50a496037fbc935959a1edbb4b114d Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Tue, 13 Jan 2009 16:40:23 -0500 Subject: [PATCH 03/18] Datatypes config for solidqual, as well as solidqual and csfasta sniffers --- datatypes_conf.xml.sample | 23 +++++++++++++---------- 1 file changed, 13 insertions(+), 10 deletions(-) diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index 431d9da7c95..3d1f94b94d5 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -38,6 +38,7 @@ + @@ -175,15 +176,17 @@ - - - - - - - - - - + + + + + + + + + + + + From f37e747d49711a879b45a7c61bc9239e60b417c0 Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Wed, 14 Jan 2009 08:43:00 -0500 Subject: [PATCH 04/18] Set the working directory in the wrapper when recovering jobs at startup. --- lib/galaxy/jobs/__init__.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 058bc31e1d0..9c67615ee04 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -125,6 +125,8 @@ class JobQueue( object ): if job.job_runner_name is not None: # why are we passing the queue to the wrapper? job_wrapper = JobWrapper( job, self.app.toolbox.tools_by_id[ job.tool_id ], self ) + job_wrapper.working_directory = \ + os.path.join( self.app.config.job_working_directory, str( job.id ) ) self.dispatcher.recover( job, job_wrapper ) def __monitor( self ): From 19e13df63aa67d6b3f358579443decf4f8abd66a Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Wed, 14 Jan 2009 08:50:31 -0500 Subject: [PATCH 05/18] Move setting of the working directory to the wrapper's init method instead. --- lib/galaxy/jobs/__init__.py | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 9c67615ee04..67b4c37a25d 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -125,8 +125,6 @@ class JobQueue( object ): if job.job_runner_name is not None: # why are we passing the queue to the wrapper? job_wrapper = JobWrapper( job, self.app.toolbox.tools_by_id[ job.tool_id ], self ) - job_wrapper.working_directory = \ - os.path.join( self.app.config.job_working_directory, str( job.id ) ) self.dispatcher.recover( job, job_wrapper ) def __monitor( self ): @@ -300,9 +298,13 @@ class JobWrapper( object ): self.queue = queue self.app = queue.app self.extra_filenames = [] - self.working_directory = None self.command_line = None self.galaxy_lib_dir = None + # With job outputs in the working directory, we need the working + # directory to be set before prepare is run, or else premature deletion + # and job recovery fail. + self.working_directory = \ + os.path.join( self.app.config.job_working_directory, str( self.job_id ) ) def get_param_dict( self ): """ @@ -319,9 +321,6 @@ class JobWrapper( object ): config files. """ mapping.context.current.clear() #this prevents the metadata reverting that has been seen in conjunction with the PBS job runner - # Create the working directory - self.working_directory = \ - os.path.join( self.app.config.job_working_directory, str( self.job_id ) ) if not os.path.exists( self.working_directory ): os.mkdir( self.working_directory ) # Restore parameters from the database From 8e376dea50c88faa1cf88779459c6a9860f15a98 Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Wed, 14 Jan 2009 10:49:14 -0500 Subject: [PATCH 06/18] Improvements to job recovery and clarification of the false path's job id --- lib/galaxy/jobs/__init__.py | 35 ++++++++++++++++++----------------- lib/galaxy/tools/__init__.py | 2 +- 2 files changed, 19 insertions(+), 18 deletions(-) diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 67b4c37a25d..427c0fe7424 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -111,19 +111,19 @@ class JobQueue( object ): def __check_jobs_at_startup( self ): """ - Checks all jobs that are in the 'running' or 'queued' state in the - database and requeues or cleans up as necessary. Only run as the + Checks all jobs that are in the 'new', 'queued' or 'running' state in + the database and requeues or cleans up as necessary. Only run as the job manager starts. """ model = self.app.model - # Jobs in the NEW state won't be requeued unless we're tracking in the database - if not self.track_jobs_in_database: - for job in model.Job.filter( model.Job.c.state==model.Job.states.NEW ).all(): - log.debug( "no runner: %s is still in new state, adding to the jobs queue" %job.id ) - self.queue.put( ( job.id, job.tool_id ) ) + for job in model.Job.filter( model.Job.c.state==model.Job.states.NEW ).all(): + log.debug( "no runner: %s is still in new state, adding to the jobs queue" %job.id ) + self.queue.put( ( job.id, job.tool_id ) ) for job in model.Job.filter( (model.Job.c.state == model.Job.states.RUNNING) | (model.Job.c.state == model.Job.states.QUEUED) ).all(): - if job.job_runner_name is not None: - # why are we passing the queue to the wrapper? + if job.job_runner_name is None: + log.debug( "no runner: %s is still in queued state, adding to the jobs queue" %job.id ) + self.queue.put( ( job.id, job.tool_id ) ) + else: job_wrapper = JobWrapper( job, self.app.toolbox.tools_by_id[ job.tool_id ], self ) self.dispatcher.recover( job, job_wrapper ) @@ -383,12 +383,12 @@ class JobWrapper( object ): if not job.state == model.Job.states.DELETED: for dataset_assoc in job.output_datasets: if self.app.config.outputs_to_working_directory: - false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.id ) ) - if os.path.exists( false_path ): + false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.dataset.id ) ) + try: shutil.move( false_path, dataset_assoc.dataset.file_name ) log.debug( "fail(): Moved %s to %s" % ( false_path, dataset_assoc.dataset.file_name ) ) - else: - log.warning( "fail(): Missing output file in working directory: %s" % false_path ) + except ( IOError, OSError ), e: + log.error( "fail(): Missing output file in working directory: %s" % e ) dataset = dataset_assoc.dataset dataset.refresh() dataset.state = dataset.states.ERROR @@ -453,12 +453,13 @@ class JobWrapper( object ): job.state = 'ok' for dataset_assoc in job.output_datasets: if self.app.config.outputs_to_working_directory: - false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.id ) ) - if os.path.exists( false_path ): + false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.dataset.id ) ) + try: shutil.move( false_path, dataset_assoc.dataset.file_name ) log.debug( "finish(): Moved %s to %s" % ( false_path, dataset_assoc.dataset.file_name ) ) - else: - log.warning( "finish(): Missing output file in working directory: %s" % false_path ) + except ( IOError, OSError ): + self.fail( "The job's output dataset(s) could not be read" ) + return for dataset in dataset_assoc.dataset.dataset.history_associations: #need to update all associated output hdas, i.e. history was shared with job running dataset.blurb = 'done' dataset.peek = 'no peek' diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index 796cfc1332f..f2837e77878 100644 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -1098,7 +1098,7 @@ class Tool: for name, data in output_datasets.items(): # Write outputs to the working directory (for security purposes) if desired. if self.app.config.outputs_to_working_directory and working_directory is not None: - false_path = os.path.abspath( os.path.join( working_directory, "galaxy_dataset_%d.dat" % data.id ) ) + false_path = os.path.abspath( os.path.join( working_directory, "galaxy_dataset_%d.dat" % data.dataset.id ) ) param_dict[name] = DatasetFilenameWrapper( data, false_path = false_path ) open( false_path, 'w' ).close() else: From 6976c383ad86520162d0865b15d15c4c7e5a719f Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Wed, 14 Jan 2009 16:15:51 -0500 Subject: [PATCH 07/18] When copying a history, only copy active_datasets - fixes sharing a history containing purged datasets. --- lib/galaxy/model/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 90aea556ea8..e3fe89703fa 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -376,7 +376,7 @@ class History( object ): des.flush() des.name = self.name des.user_id = self.user_id - for data in self.datasets: + for data in self.active_datasets: new_data = data.copy( copy_children = True ) des.add_dataset( new_data ) new_data.flush() From 87a77c62f2e132f02dab3b87570331f5d8832c2a Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Wed, 14 Jan 2009 17:20:22 -0500 Subject: [PATCH 08/18] Add a SelectToolParameterWrapper object and a wrap method to Tool.execute() that allows for things like the following in tool configs: Also replace EOL chars with
in a tool's output data.info so that they are displayed in HTML. --- lib/galaxy/datatypes/data.py | 3 ++- lib/galaxy/tools/__init__.py | 23 +++++++++++++++++++---- lib/galaxy/tools/actions/__init__.py | 27 +++++++++++++++++++++++++++ 3 files changed, 48 insertions(+), 5 deletions(-) diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 93a8e8adfcb..6be6f808f51 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -132,7 +132,8 @@ class Data( object ): def display_info(self, dataset): """Returns formated html of dataset info""" try: - return escape(dataset.info) + # Change new line chars to html + return escape( dataset.info ).replace( "\r", "\n" ).replace( "\n", "
" ) except: return "info unavailable" def validate(self, dataset): diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index f2837e77878..02fae427386 100644 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -1072,10 +1072,11 @@ class Tool: datatypes_registry = self.app.datatypes_registry, tool = self, name = input.name ) + elif isinstance( input, SelectToolParameter ): + input_values[ input.name ] = SelectToolParameterWrapper( input, input_values[ input.name ], self.app ) else: - input_values[ input.name ] = \ - InputValueWrapper( input, input_values[ input.name ], param_dict ) - # HACK: only wrap if check_values is false, this deals with external + input_values[ input.name ] = InputValueWrapper( input, input_values[ input.name ], param_dict ) + # HACK: only wrap if check_values is not false, this deals with external # tools where the inputs don't even get passed through. These # tools (e.g. UCSC) should really be handled in a special way. if self.check_values: @@ -1416,7 +1417,21 @@ class InputValueWrapper( object ): return self.input.to_param_dict_string( self.value, self._other_values ) def __getattr__( self, key ): return getattr( self.value, key ) - + +class SelectToolParameterWrapper( object ): + """ + Wraps a SelectTooParameter so that __str__ returns the selected value, but all other + attributes are accessible. + """ + def __init__( self, input, value, app ): + self.input = input + self.value = value + self.input.value_label = input.value_to_display_text( value, app ) + def __str__( self ): + return self.input.to_param_dict_string( self.value ) + def __getattr__( self, key ): + return getattr( self.input, key ) + class DatasetFilenameWrapper( object ): """ Wraps a dataset so that __str__ returns the filename, but all other diff --git a/lib/galaxy/tools/actions/__init__.py b/lib/galaxy/tools/actions/__init__.py index f4f153a1321..3e0093d1b60 100644 --- a/lib/galaxy/tools/actions/__init__.py +++ b/lib/galaxy/tools/actions/__init__.py @@ -1,9 +1,11 @@ from galaxy.util.bunch import Bunch from galaxy.tools.parameters import * +from galaxy.tools.parameters.grouping import * from galaxy.util.template import fill_template from galaxy.util.none_like import NoneDataset from galaxy.web import url_for from galaxy.jobs import JOB_OK +import galaxy.tools import logging log = logging.getLogger( __name__ ) @@ -65,6 +67,26 @@ class DefaultToolAction( object ): return input_datasets def execute(self, tool, trans, incoming={}, set_output_hid=True ): + def wrap_values( inputs, input_values ): + # Wrap tool inputs as necessary + for input in inputs.itervalues(): + if isinstance( input, Repeat ): + for d in input_values[ input.name ]: + wrap_values( input.inputs, d ) + elif isinstance( input, Conditional ): + values = input_values[ input.name ] + current = values["__current_case__"] + wrap_values( input.cases[current].inputs, values ) + elif isinstance( input, DataToolParameter ): + input_values[ input.name ] = \ + galaxy.tools.DatasetFilenameWrapper( input_values[ input.name ], + datatypes_registry = trans.app.datatypes_registry, + tool = tool, + name = input.name ) + elif isinstance( input, SelectToolParameter ): + input_values[ input.name ] = galaxy.tools.SelectToolParameterWrapper( input, input_values[ input.name ], tool.app ) + else: + input_values[ input.name ] = galaxy.tools.InputValueWrapper( input, input_values[ input.name ], incoming ) out_data = {} # Collect any input datasets from the incoming parameters inp_data = self.collect_input_datasets( tool, incoming, trans ) @@ -151,6 +173,11 @@ class DefaultToolAction( object ): # Set output label if output.label: params = dict( incoming ) + # wrapping the params allows the tool config to contain things like + # + # + # + wrap_values( tool.inputs, params ) params['tool'] = tool params['on_string'] = on_text data.name = fill_template( output.label, context=params ) From 43b925d59bac06c3e7bf033ea401f9de96751a30 Mon Sep 17 00:00:00 2001 From: James Taylor Date: Wed, 14 Jan 2009 17:56:30 -0500 Subject: [PATCH 09/18] New tool menu styles for labels and top-level tools. --- static/june_2007_style/blue/tool_menu.css | 15 +++++++++++--- static/june_2007_style/tool_menu.css.tmpl | 15 +++++++++++--- templates/root/tool_menu.mako | 25 +++++++++-------------- 3 files changed, 34 insertions(+), 21 deletions(-) diff --git a/static/june_2007_style/blue/tool_menu.css b/static/june_2007_style/blue/tool_menu.css index b0da725086e..b18e01624e5 100644 --- a/static/june_2007_style/blue/tool_menu.css +++ b/static/june_2007_style/blue/tool_menu.css @@ -31,15 +31,16 @@ div.toolSectionDetailsInner div.toolSectionTitle { - padding-bottom: 0px; font-weight: bold; } div.toolPanelLabel { - padding-top: 5px; + padding-top: 10px; padding-bottom: 5px; font-weight: bold; + color: gray; + text-transform: uppercase; } div.toolTitle @@ -52,9 +53,17 @@ div.toolTitle list-style: square outside; } +div.toolSectionBody div.toolPanelLabel +{ + padding-top: 5px; + padding-bottom: 5px; + margin-left: 16px; + margin-right: 10px; + display: list-item; + list-style: none outside; +} div.toolTitleNoSection { padding-bottom: 0px; - font-weight: bold; } diff --git a/static/june_2007_style/tool_menu.css.tmpl b/static/june_2007_style/tool_menu.css.tmpl index aa1178971a2..ef4b4ca7b28 100644 --- a/static/june_2007_style/tool_menu.css.tmpl +++ b/static/june_2007_style/tool_menu.css.tmpl @@ -31,15 +31,16 @@ div.toolSectionDetailsInner div.toolSectionTitle { - padding-bottom: 0px; font-weight: bold; } div.toolPanelLabel { - padding-top: 5px; + padding-top: 10px; padding-bottom: 5px; font-weight: bold; + color: gray; + text-transform: uppercase; } div.toolTitle @@ -52,9 +53,17 @@ div.toolTitle list-style: square outside; } +div.toolSectionBody div.toolPanelLabel +{ + padding-top: 5px; + padding-bottom: 5px; + margin-left: 16px; + margin-right: 10px; + display: list-item; + list-style: none outside; +} div.toolTitleNoSection { padding-bottom: 0px; - font-weight: bold; } diff --git a/templates/root/tool_menu.mako b/templates/root/tool_menu.mako index c35d9868519..7fb4017a999 100644 --- a/templates/root/tool_menu.mako +++ b/templates/root/tool_menu.mako @@ -38,11 +38,9 @@ ## Render a label <%def name="render_label( label )"> -
${label.text}
-
@@ -91,7 +89,6 @@ %for key, val in toolbox.tool_panel.items(): %if key.startswith( 'tool' ): ${render_tool( val, False )} -
%elif key.startswith( 'workflow' ): ${render_workflow( key, val, False )} %elif key.startswith( 'section' ): @@ -112,10 +109,10 @@ %endfor -
%elif key.startswith( 'label' ): ${render_label( val )} %endif +
%endfor ## Link to workflow management. The location of this may change, but eventually @@ -128,18 +125,16 @@ Workflow (beta)
-
-
- Manage workflows -
- %if t.user: - %for m in t.user.stored_workflow_menu_entries: - - %endfor - %endif +
+ Manage workflows
+ %if t.user: + %for m in t.user.stored_workflow_menu_entries: + + %endfor + %endif
From d46171ea21ca3874897c9f25d041163b252ff899 Mon Sep 17 00:00:00 2001 From: Guruprasad Anada Date: Thu, 15 Jan 2009 10:42:22 -0500 Subject: [PATCH 10/18] Fixed a bug in microsatellite mutability tool. --- tools/regVariation/microsats_mutability.py | 10 ---------- 1 file changed, 10 deletions(-) diff --git a/tools/regVariation/microsats_mutability.py b/tools/regVariation/microsats_mutability.py index 1261dbbba73..c4f66368b79 100644 --- a/tools/regVariation/microsats_mutability.py +++ b/tools/regVariation/microsats_mutability.py @@ -409,19 +409,9 @@ def main(): blk=0 win=0 linestr="" - ff=open("junkmabc","w") if region == 'win': - """ - sorted_infile = tempfile.NamedTemporaryFile() - if winspecies == 1: - cmdline = "sort -n -k 3"+" -o "+sorted_infile.name+" "+infile - elif winspecies == 2: - cmdline = "sort -n -k 10"+" -o "+sorted_infile.name+" "+infile - os.system(cmdline) - print >>ff, "Finished sorting" - """ msats = NiceReaderWrapper( fileinput.FileInput( infile ), chrom_col = speciesind, start_col = speciesind+1, From b7aaa17bc35a5f206b068995658c1345ae74986f Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Thu, 15 Jan 2009 10:49:01 -0500 Subject: [PATCH 11/18] Change History.copy() back to use datasets rather than active_datasets. Add checks in set_peek() to properly set peek on purged datasets. Fixes sharing a history that contains purged datasets. --- lib/galaxy/datatypes/data.py | 30 ++++--- lib/galaxy/datatypes/genetics.py | 21 +++-- lib/galaxy/datatypes/images.py | 114 +++++++++++++++++---------- lib/galaxy/datatypes/interval.py | 12 ++- lib/galaxy/datatypes/qualityscore.py | 12 ++- lib/galaxy/datatypes/sequence.py | 24 ++++-- lib/galaxy/datatypes/xml.py | 8 +- lib/galaxy/model/__init__.py | 18 ++++- tools/data_source/echo.xml | 8 +- 9 files changed, 169 insertions(+), 78 deletions(-) diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 6be6f808f51..db8188a38fa 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -103,8 +103,12 @@ class Data( object ): return False def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = '' - dataset.blurb = 'data' + if not dataset.dataset.purged: + dataset.peek = '' + dataset.blurb = 'data' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): """Create HTML table, used for displaying peek""" out = [''] @@ -275,19 +279,27 @@ class Text( Data ): return 'text/plain' def set_peek( self, dataset, line_count=None ): - dataset.peek = get_file_peek( dataset.file_name ) - if line_count is None: - dataset.blurb = "%s lines" % util.commaify( str( get_line_count( dataset.file_name ) ) ) + if not dataset.dataset.purged: + # The file must exist on disk for the get_file_peek() method + dataset.peek = get_file_peek( dataset.file_name ) + if line_count is None: + dataset.blurb = "%s lines" % util.commaify( str( get_line_count( dataset.file_name ) ) ) + else: + dataset.blurb = "%s lines" % util.commaify( str( line_count ) ) else: - dataset.blurb = "%s lines" % util.commaify( str( line_count ) ) + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' class Binary( Data ): """Binary data""" - def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = 'binary data' - dataset.blurb = 'data' + if not dataset.dataset.purged: + dataset.peek = 'binary data' + dataset.blurb = 'data' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def get_test_fname( fname ): """Returns test data filename""" diff --git a/lib/galaxy/datatypes/genetics.py b/lib/galaxy/datatypes/genetics.py index 9857eb7d36d..10d4ad9dc28 100644 --- a/lib/galaxy/datatypes/genetics.py +++ b/lib/galaxy/datatypes/genetics.py @@ -43,11 +43,14 @@ class GenomeGraphs( Tabular ): def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = data.get_file_peek( dataset.file_name ) - ## dataset.peek = self.make_html_table( dataset.peek ) - dataset.blurb = util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + " rows" - #i don't think set_meta should not be called here, it should be called separately - self.set_meta( dataset ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + " rows" + #i don't think set_meta should not be called here, it should be called separately + self.set_meta( dataset ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def get_estimated_display_viewport( self, dataset ): """Return a chrom, start, stop tuple for viewing a file.""" @@ -130,8 +133,12 @@ class SNPMatrix(Rgenetics): file_ext="snpmatrix" def set_peek( self, dataset ): - dataset.peek = "Binary RGenetics file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = "Binary RGenetics file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ """ diff --git a/lib/galaxy/datatypes/images.py b/lib/galaxy/datatypes/images.py index 14d9d6cd756..dab8e4a636e 100644 --- a/lib/galaxy/datatypes/images.py +++ b/lib/galaxy/datatypes/images.py @@ -14,9 +14,13 @@ class Ab1( data.Data ): """Class describing an ab1 binary sequence file""" file_ext = "ab1" def set_peek( self, dataset ): - export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'ab1','name':'ab1 sequence','info':'Sequence file','dbkey':dataset.dbkey}) - dataset.peek = "Binary ab1 sequence file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + export_url = "/history_add_to?" + urlencode({'history_id':dataset.history_id,'ext':'ab1','name':'ab1 sequence','info':'Sequence file','dbkey':dataset.dbkey}) + dataset.peek = "Binary ab1 sequence file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -27,9 +31,13 @@ class Scf( data.Data ): """Class describing an scf binary sequence file""" file_ext = "scf" def set_peek( self, dataset ): - export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'scf','name':'scf sequence','info':'Sequence file','dbkey':dataset.dbkey}) - dataset.peek = "Binary scf sequence file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + export_url = "/history_add_to?" + urlencode({'history_id':dataset.history_id,'ext':'scf','name':'scf sequence','info':'Sequence file','dbkey':dataset.dbkey}) + dataset.peek = "Binary scf sequence file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -40,10 +48,14 @@ class Binseq( data.Data ): """Class describing a zip archive of binary sequence files""" file_ext = "binseq.zip" def set_peek( self, dataset ): - zip_file = zipfile.ZipFile( dataset.file_name, "r" ) - num_files = len( zip_file.namelist() ) - dataset.peek = "Archive of %s binary sequence files" % ( str( num_files ) ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + zip_file = zipfile.ZipFile( dataset.file_name, "r" ) + num_files = len( zip_file.namelist() ) + dataset.peek = "Archive of %s binary sequence files" % ( str( num_files ) ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -57,10 +69,14 @@ class Txtseq( data.Data ): """Class describing a zip archive of text sequence files""" file_ext = "txtseq.zip" def set_peek( self, dataset ): - zip_file = zipfile.ZipFile( dataset.file_name, "r" ) - num_files = len( zip_file.namelist() ) - dataset.peek = "Archive of %s text sequence files" % ( str( num_files ) ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + zip_file = zipfile.ZipFile( dataset.file_name, "r" ) + num_files = len( zip_file.namelist() ) + dataset.peek = "Archive of %s text sequence files" % ( str( num_files ) ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -73,8 +89,12 @@ class Txtseq( data.Data ): class Image( data.Data ): """Class describing an image""" def set_peek( self, dataset ): - dataset.peek = 'Image in %s format' % dataset.extension - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = 'Image in %s format' % dataset.extension + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def create_applet_tag_peek( class_name, archive, params ): text = """ @@ -104,18 +124,22 @@ class Gmaj( data.Data ): """Class describing a GMAJ Applet""" file_ext = "gmaj.zip" def set_peek( self, dataset ): - params = { - "bundle":"display?id=%s&tofile=yes&toext=.zip" % dataset.id, - "buttonlabel": "Launch GMAJ", - "nobutton": "false", - "urlpause" :"100", - "debug": "false", - "posturl": "history_add_to?%s" % urlencode( { 'history_id': dataset.history_id, 'ext': 'maf', 'name': 'GMAJ Output on data %s' % dataset.hid, 'info': 'Added by GMAJ', 'dbkey': dataset.dbkey } ) - } - class_name = "edu.psu.bx.gmaj.MajApplet.class" - archive = "/static/gmaj/gmaj.jar" - dataset.peek = create_applet_tag_peek( class_name, archive, params ) - dataset.blurb = 'GMAJ Multiple Alignment Viewer' + if not dataset.dataset.purged: + params = { + "bundle":"display?id=%s&tofile=yes&toext=.zip" % dataset.id, + "buttonlabel": "Launch GMAJ", + "nobutton": "false", + "urlpause" :"100", + "debug": "false", + "posturl": "history_add_to?%s" % urlencode( { 'history_id': dataset.history_id, 'ext': 'maf', 'name': 'GMAJ Output on data %s' % dataset.hid, 'info': 'Added by GMAJ', 'dbkey': dataset.dbkey } ) + } + class_name = "edu.psu.bx.gmaj.MajApplet.class" + archive = "/static/gmaj/gmaj.jar" + dataset.peek = create_applet_tag_peek( class_name, archive, params ) + dataset.blurb = 'GMAJ Multiple Alignment Viewer' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -147,8 +171,12 @@ class Html( data.Text ): """Class describing an html file""" file_ext = "html" def set_peek( self, dataset ): - dataset.peek = "HTML file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = "HTML file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def get_mime(self): """Returns the mime type of the datatype""" return 'text/html' @@ -176,17 +204,21 @@ class Laj( data.Text ): """Class describing a LAJ Applet""" file_ext = "laj" def set_peek( self, dataset ): - params = { - "alignfile1": "display?id=%s" % dataset.id, - "buttonlabel": "Launch LAJ", - "title": "LAJ in Galaxy", - "posturl": "history_add_to?%s" % urlencode( { 'history_id': dataset.history_id, 'ext': 'lav', 'name': 'LAJ Output', 'info': 'Added by LAJ', 'dbkey': dataset.dbkey } ), - "noseq": "true" - } - class_name = "edu.psu.cse.bio.laj.LajApplet.class" - archive = "/static/laj/laj.jar" - dataset.peek = create_applet_tag_peek( class_name, archive, params ) - dataset.blurb = 'LAJ Multiple Alignment Viewer' + if not dataset.dataset.purged: + params = { + "alignfile1": "display?id=%s" % dataset.id, + "buttonlabel": "Launch LAJ", + "title": "LAJ in Galaxy", + "posturl": "history_add_to?%s" % urlencode( { 'history_id': dataset.history_id, 'ext': 'lav', 'name': 'LAJ Output', 'info': 'Added by LAJ', 'dbkey': dataset.dbkey } ), + "noseq": "true" + } + class_name = "edu.psu.cse.bio.laj.LajApplet.class" + archive = "/static/laj/laj.jar" + dataset.peek = create_applet_tag_peek( class_name, archive, params ) + dataset.blurb = 'LAJ Multiple Alignment Viewer' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 940af654523..3ee543ca454 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -59,11 +59,15 @@ class Interval( Tabular ): def set_peek( self, dataset, line_count=None ): """Set the peek and blurb text""" - dataset.peek = data.get_file_peek( dataset.file_name ) - if line_count is None: - dataset.blurb = "%s regions" % util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + if line_count is None: + dataset.blurb = "%s regions" % util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + else: + dataset.blurb = "%s regions" % util.commaify( str( line_count ) ) else: - dataset.blurb = "%s regions" % util.commaify( str( line_count ) ) + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def set_meta( self, dataset, overwrite = True, first_line_is_header = False, **kwd ): Tabular.set_meta( self, dataset, overwrite = overwrite, skip = 0 ) diff --git a/lib/galaxy/datatypes/qualityscore.py b/lib/galaxy/datatypes/qualityscore.py index cfd5b280adb..188cd413b0d 100644 --- a/lib/galaxy/datatypes/qualityscore.py +++ b/lib/galaxy/datatypes/qualityscore.py @@ -16,11 +16,15 @@ class QualityScore ( data.Text ): file_ext = "qual" def set_peek( self, dataset, line_count=None ): - dataset.peek = data.get_file_peek( dataset.file_name ) - if line_count is None: - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + if line_count is None: + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.blurb = "%s lines, Quality score file" % util.commaify( str( line_count ) ) else: - dataset.blurb = "%s lines, Quality score file" % util.commaify( str( line_count ) ) + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 2ec55345cc3..13144ef652c 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -31,8 +31,12 @@ class Fasta( Sequence ): file_ext = "fasta" def set_peek( self, dataset ): - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ @@ -86,8 +90,12 @@ class csFasta( Sequence ): file_ext = "csfasta" def set_peek( self, dataset ): - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ @@ -108,8 +116,12 @@ class FastqSolexa( Sequence ): file_ext = "fastqsolexa" def set_peek( self, dataset ): - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ diff --git a/lib/galaxy/datatypes/xml.py b/lib/galaxy/datatypes/xml.py index d658e33308f..e496aad00ab 100644 --- a/lib/galaxy/datatypes/xml.py +++ b/lib/galaxy/datatypes/xml.py @@ -12,8 +12,12 @@ class BlastXml( data.Text ): file_ext = "blastxml" def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = 'NCBI Blast XML data' + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = 'NCBI Blast XML data' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ Determines whether the file is blastxml diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index e3fe89703fa..1586d90c067 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -287,14 +287,26 @@ class HistoryDatasetAssociation( object ): return self.datatype.find_conversion_destination( self, accepted_formats, datatypes_registry, **kwd ) def copy( self, copy_children = False, parent_id = None ): - des = HistoryDatasetAssociation( hid=self.hid, name=self.name, info=self.info, blurb=self.blurb, peek=self.peek, extension=self.extension, dbkey=self.dbkey, dataset = self.dataset, visible=self.visible, deleted=self.deleted, parent_id=parent_id, copied_from_history_dataset_association = self ) + des = HistoryDatasetAssociation( hid=self.hid, + name=self.name, + info=self.info, + blurb=self.blurb, + peek=self.peek, + extension=self.extension, + dbkey=self.dbkey, + dataset=self.dataset, + visible=self.visible, + deleted=self.deleted, + parent_id=parent_id, + copied_from_history_dataset_association=self ) des.flush() des.set_size() des.metadata = self.metadata #need to set after flushed, as MetadataFiles require dataset.id if copy_children: for child in self.children: child_copy = child.copy( copy_children = copy_children, parent_id = des.id ) - des.set_peek() #in some instances peek relies on dataset_id, i.e. gmaj.zip for viewing MAFs + # In some instances peek relies on dataset_id ( e.g., gmaj.zip for viewing MAFs ) + des.set_peek() des.flush() return des @@ -376,7 +388,7 @@ class History( object ): des.flush() des.name = self.name des.user_id = self.user_id - for data in self.active_datasets: + for data in self.datasets: new_data = data.copy( copy_children = True ) des.add_dataset( new_data ) new_data.flush() diff --git a/tools/data_source/echo.xml b/tools/data_source/echo.xml index 860e6a12656..e3c3a2a84a2 100644 --- a/tools/data_source/echo.xml +++ b/tools/data_source/echo.xml @@ -6,14 +6,18 @@ echoes parameters - echo.py $input $output + echo.py $input $database $output + + + + - + From ee8779987e5ac7da4847321a425ff149ee8b7dec Mon Sep 17 00:00:00 2001 From: Guruprasad Anada Date: Thu, 15 Jan 2009 13:10:04 -0500 Subject: [PATCH 12/18] More changes to mutability tool. --- tools/regVariation/microsats_mutability.py | 29 +++++++++------------ tools/regVariation/microsats_mutability.xml | 4 +++ 2 files changed, 16 insertions(+), 17 deletions(-) diff --git a/tools/regVariation/microsats_mutability.py b/tools/regVariation/microsats_mutability.py index c4f66368b79..eab8c3c0829 100644 --- a/tools/regVariation/microsats_mutability.py +++ b/tools/regVariation/microsats_mutability.py @@ -128,11 +128,13 @@ def output_writer(blk, blk_lines): uniq_s_elems_2 = get_binned_lists(uniq_s_elems_2,s_bin_size) for pitem1 in uniq_elems_1: - repeats1 = [] - repeats2 = [] + #repeats1 = [] + #repeats2 = [] thresholds = [] if s_group_cols[0] != -1: #Sub-group by feature is not None for sitem1 in uniq_s_elems_1: + repeats1 = [] + repeats2 = [] if type(sitem1) == type(''): sitem1 = sitem1.strip() for bline in blk_lines: @@ -223,11 +225,13 @@ def output_writer(blk, blk_lines): count1[str(pitem1)]=sum(repeats2) for pitem2 in uniq_elems_2: - repeats1 = [] - repeats2 = [] + #repeats1 = [] + #repeats2 = [] thresholds = [] if s_group_cols[0] != -1: #Sub-group by feature is not None for sitem2 in uniq_s_elems_2: + repeats1 = [] + repeats2 = [] if type(sitem2)==type(''): sitem2 = sitem2.strip() for bline in blk_lines: @@ -349,11 +353,10 @@ def output_writer(blk, blk_lines): count = count1[key] mut = "%.2e" %(mut/num_generations) if region == 'align': - print >>fout, str(blk) + '\t'+seq1 + '\t' + start1+ '\t'+end1+ '\t'+seq2 + '\t'+start2+ '\t'+end2+ '\t'+key.strip()+ '\t'+str(mut) + '\t'+ str(count) + print >>fout, str(blk) + '\t'+seq1 + '\t' + seq2 + '\t' +key.strip()+ '\t'+str(mut) + '\t'+ str(count) elif region == 'win': fout.write("%s\t%s\t%s\t%s\n" %(blk,key.strip(),mut,count)) fout.flush() - #print >>fout, blk + '\t'+key.strip()+ '\t'+str(mut)+ '\t'+ str(count) #catch any remaining repeats, for instance if the orthologous position contained different repeat units for remaining_key in mut2.keys(): @@ -361,7 +364,7 @@ def output_writer(blk, blk_lines): mut = "%.2e" %(mut/num_generations) count = count2[remaining_key] if region == 'align': - print >>fout, str(blk) + '\t'+seq1 + '\t' + start1+ '\t'+end1+ '\t'+seq2 + '\t'+start2+ '\t'+end2+ '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count) + print >>fout, str(blk) + '\t'+seq1 + '\t'+seq2 + '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count) elif region == 'win': fout.write("%s\t%s\t%s\t%s\n" %(blk,remaining_key.strip(),mut,count)) fout.flush() @@ -420,17 +423,9 @@ def main(): fix_strand = True) msatTree = quicksect.IntervalTree() for item in msats: - #print >>sys.stderr, item if type( item ) is GenomicInterval: msatTree.insert( item, msats.linenum, item.fields ) - """ - result = [] - msatTree.traverse(lambda node: result.append( node )) - for n in result: - print >>sys.stderr,n.other - print >>sys.stderr,msatTree.chroms - #sys.exit() - """ + for iline in fint: try: iline = iline.rstrip('\r\n') @@ -460,7 +455,7 @@ def main(): print "Skipped %d intervals as invalid." %(skipped) elif region == 'align': if s_group_cols[0] != -1: - print >>fout, "#Window\tSpecies_1\tWindow_Start\tWindow_End\tSpecies_2\tGroupby_Feature\tSubGroupby_Feature\tMutability\tCount" + print >>fout, "#Window\tSpecies_1\tSpecies_2\tGroupby_Feature\tSubGroupby_Feature\tMutability\tCount" else: print >>fout, "#Window\tSpecies_1\tWindow_Start\tWindow_End\tSpecies_2\tGroupby_Feature\tMutability\tCount" prev_bnum = -1 diff --git a/tools/regVariation/microsats_mutability.xml b/tools/regVariation/microsats_mutability.xml index 141cadd31b0..e5ac10925b2 100644 --- a/tools/regVariation/microsats_mutability.xml +++ b/tools/regVariation/microsats_mutability.xml @@ -106,6 +106,10 @@ This tool computes microsatellite mutability for the orthologous microsatellites fetched from 'Extract Orthologous Microsatellites from pair-wise alignments' tool. +Mutability is computed according to the method described in the following paper: + +*Webster et al., Microsatellite evolution inferred from human-chimpanzee genomic sequence alignments, Proc Natl Acad Sci 2002 June 25; 99(13): 8748-8753* + ----- .. class:: warningmark From fa6155cc3f36ef419dc440b5b3391621b10d733f Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Thu, 15 Jan 2009 14:23:53 -0500 Subject: [PATCH 13/18] Eliminate the use of the "order=" for the sniffers in datatypes.conf - since we've moved from using an ini file to an xml file, ordering is implicit. --- datatypes_conf.xml.sample | 30 ++++++++++++++++-------------- lib/galaxy/datatypes/registry.py | 12 +++++------- 2 files changed, 21 insertions(+), 21 deletions(-) diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index 3bb5997181e..a3b173adfac 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -146,20 +146,22 @@ - - - - - - - - - - - - - + + + + + + + + + + + + + diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index c27cce91188..b7cd9b58b7c 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -59,18 +59,15 @@ class Registry( object ): self.converters.append( ( converter_config, extension, target_datatype ) ) except Exception, e: self.log.warning( 'Error loading datatype "%s", problem: %s' % ( extension, str( e ) ) ) - # Load datatype sniffers from config + # Load datatype sniffers from the config sniff_order = [] sniffers = root.find( 'sniffers' ) for elem in sniffers.findall( 'sniffer' ): - order = elem.get( 'order', None ) type = elem.get( 'type', None ) - if order and type: - sniff_order.append( ( order, type ) ) - sniff_order.sort() - for ele in sniff_order: + if type: + sniff_order.append( type ) + for type in sniff_order: try: - type = ele[1] fields = type.split( ":" ) datatype_module = fields[0] datatype_class = fields[1] @@ -147,6 +144,7 @@ class Registry( object ): sequence.Maf(), sequence.Lav(), sequence.Fasta(), + sequence.FastqSolexa(), interval.Wiggle(), images.Html(), sequence.Axt(), From 50b46e60e37e92233d902566f7d94b96ffb8a5a8 Mon Sep 17 00:00:00 2001 From: Guruprasad Anada Date: Fri, 16 Jan 2009 13:30:01 -0500 Subject: [PATCH 14/18] Removed an option from grouping tool, which is now part of the LCA tool under Taxonomy. --- tools/stats/grouping.py | 85 ++++------------------ tools/stats/grouping.xml | 149 +++++++++++++++++++-------------------- 2 files changed, 84 insertions(+), 150 deletions(-) diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py index eea1a135ed0..bcd3714b9b0 100644 --- a/tools/stats/grouping.py +++ b/tools/stats/grouping.py @@ -12,15 +12,13 @@ def stop_err(msg): def main(): inputfile = sys.argv[2] - in_columns = int( sys.argv[5] ) - show_remaining_cols = sys.argv[4] ops = [] cols = [] rounds = [] elems = [] - for var in sys.argv[6:]: + for var in sys.argv[4:]: ops.append(var.split()[0]) cols.append(var.split()[1]) rounds.append(var.split()[2]) @@ -81,19 +79,9 @@ def main(): if error_code != 0: stop_err( "Sorting input dataset resulted in error: %s: %s" %( error_code, stdout )) - - if show_remaining_cols == 'yes': - show_cols_list = [1]*in_columns - show_cols_list[group_col] = 0 - for c in cols: - c = int(c)-1 - show_cols_list[c] = 0 - #at the end of this, only the indices of the remaining columns will be set to 1 - remaining_cols = [j for j,k in enumerate(show_cols_list) if k==1] #this is the list of remaining column indices - + prev_item = "" prev_vals = [] - remaining_vals = [] skipped_lines = 0 first_invalid_line = 0 invalid_line = '' @@ -128,30 +116,22 @@ def main(): invalid_column = col+1 if valid: prev_vals[i].append(fields[col].strip()) - #Store values from all the remaning columns - if show_remaining_cols == 'yes': - for j, index in enumerate(remaining_cols): - remaining_vals[j].append(fields[index].strip()) else: """ When a new value is encountered, write the previous value and the corresponding aggregate values into the output file. This works due to the sort on group_col we've applied to the data above. """ - out_list = ['']*in_columns - out_list[group_col] = str(prev_item) - + out_str = prev_item + for i, op in enumerate( ops ): rfunc = "r." + op if op not in ['c','length','unique','random']: for j, elem in enumerate( prev_vals[i] ): prev_vals[i][j] = float( elem ) + rout = "%g" %( eval( rfunc )( prev_vals[i] )) if rounds[i] == 'yes': - rout = "%f" %( eval( rfunc )( prev_vals[i] )) rout = int(round(float(rout))) - else: - rout = "%g" %( eval( rfunc )( prev_vals[i] )) - else: if op != 'random': rout = eval( rfunc )( prev_vals[i] ) @@ -162,21 +142,9 @@ def main(): if op == 'unique': rfunc = "r.length" rout = eval( rfunc )( rout ) - - out_list[int(cols[i])-1] = str(rout) - - if show_remaining_cols == 'yes': - for index,el in enumerate(remaining_cols): - if index == 0: - try: - random_index = random.randint(0,len(remaining_vals[index])-1) - except: - random_index = 0 - #pick a random value from each of the remaning columns - rand_out = remaining_vals[index][random_index] - out_list[el] = str(rand_out) - - print >>fout, '\t'.join([elem for elem in out_list if elem != '']) + out_str += "\t" + str(rout) + + print >>fout, out_str prev_item = item prev_vals = [] @@ -185,14 +153,6 @@ def main(): val_list = [] val_list.append(fields[col].strip()) prev_vals.append(val_list) - - if show_remaining_cols == 'yes': - remaining_vals = [] - for index in remaining_cols: - remaining_val_list = [] - remaining_val_list.append(fields[index].strip()) - remaining_vals.append(remaining_val_list) - else: # This only occurs once, right at the start of the iteration. prev_item = item @@ -201,15 +161,8 @@ def main(): val_list = [] val_list.append(fields[col].strip()) prev_vals.append(val_list) - - if show_remaining_cols == 'yes': - remaining_vals = [] - for index in remaining_cols: - remaining_val_list = [] - remaining_val_list.append(fields[index].strip()) - remaining_vals.append(remaining_val_list) - except Exception: + except Exception, exc: skipped_lines += 1 if not first_invalid_line: first_invalid_line = ii+1 @@ -219,8 +172,7 @@ def main(): first_invalid_line = ii+1 # Handle the last grouped value - out_list = ['']*in_columns - out_list[group_col] = str(prev_item) + out_str = prev_item for i, op in enumerate(ops): rfunc = "r." + op @@ -228,11 +180,9 @@ def main(): if op not in ['c','length','unique','random']: for j, elem in enumerate( prev_vals[i] ): prev_vals[i][j] = float( elem ) + rout = '%g' %( eval( rfunc )( prev_vals[i] )) if rounds[i] == 'yes': - rout = '%f' %( eval( rfunc )( prev_vals[i] )) rout = int(round(float(rout))) - else: - rout = '%g' %( eval( rfunc )( prev_vals[i] )) else: if op != 'random': rout = eval( rfunc )( prev_vals[i] ) @@ -243,22 +193,13 @@ def main(): if op == 'unique': rfunc = "r.length" rout = eval( rfunc )( rout ) - out_list[int(cols[i])-1] = str(rout) + out_str += "\t" + str( rout ) except: skipped_lines += 1 if not first_invalid_line: first_invalid_line = ii+1 - if show_remaining_cols == 'yes': - for index,el in enumerate(remaining_cols): - if index == 0: - try: - random_index = random.randint(0,len(remaining_vals[index])-1) - except: - random_index = 0 - rand_out = remaining_vals[index][random_index] - out_list[el] = str(rand_out) - print >>fout, '\t'.join([elem for elem in out_list if elem != '']) + print >>fout, out_str # Generate a useful info message. msg = "--Group by c%d: " %(group_col+1) diff --git a/tools/stats/grouping.xml b/tools/stats/grouping.xml index fa00abd1dd3..8749b5dc922 100644 --- a/tools/stats/grouping.xml +++ b/tools/stats/grouping.xml @@ -1,49 +1,43 @@ - - data by a column and perform aggregate operation on other columns. - - grouping.py - $out_file1 - $input1 + + data by a column and perform aggregate operation on other columns. + + grouping.py + $out_file1 + $input1 $groupcol - $othercols - ${input1.metadata.columns} - #for $op in $operations - '${op.optype} - ${op.opcol} - ${op.opround}' - #end for - - - - - - - - - - - - + #for $op in $operations + '${op.optype} + ${op.opcol} + ${op.opround}' + #end for + + + + + + + + + + + + - - - - - - - - - + + + + - - - - - - - rpy - + + + + + + + + rpy + @@ -52,9 +46,9 @@ - + @@ -62,39 +56,38 @@ - - - - - -.. class:: infomark - -**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert* - ------ - -**Syntax** - -This tool allows you to group the input dataset by a particular column and perform aggregate functions like Mean, Sum, Max, Min and Concatenate on other columns. - -- All invalid, blank and comment lines are skipped when performing the aggregate functions. The number of skipped lines is displayed in the resulting history item. - ------ - -**Example** - -- For the following input:: - - chr22 1000 NM_17 - chr22 2000 NM_18 - chr10 2200 NM_10 - chr10 1200 NM_11 - chr22 1600 NM_19 - -- running this tool with **Group by column 1**, Operations **Mean on column 2** and **Concatenate on column 3** will return:: - - chr10 1700.00 ['NM_11', 'NM_10'] - chr22 1533.33 ['NM_17', 'NM_19', 'NM_18'] - - + + + + +.. class:: infomark + +**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert* + +----- + +**Syntax** + +This tool allows you to group the input dataset by a particular column and perform aggregate functions like Mean, Sum, Max, Min and Concatenate on other columns. + +- All invalid, blank and comment lines are skipped when performing the aggregate functions. The number of skipped lines is displayed in the resulting history item. + +----- + +**Example** + +- For the following input:: + + chr22 1000 NM_17 + chr22 2000 NM_18 + chr10 2200 NM_10 + chr10 1200 NM_11 + chr22 1600 NM_19 + +- running this tool with **Group by column 1**, Operations **Mean on column 2** and **Concatenate on column 3** will return:: + + chr10 1700.00 ['NM_11', 'NM_10'] + chr22 1533.33 ['NM_17', 'NM_19', 'NM_18'] + + From 85632afd4f4723a1f072a4ef963fe6a8e95155f8 Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Fri, 16 Jan 2009 15:02:44 -0500 Subject: [PATCH 15/18] Add the ability to keep specified permissions boxes from being displayed on certain forms. --- templates/dataset/security_common.mako | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/templates/dataset/security_common.mako b/templates/dataset/security_common.mako index 039a9a7ba6d..0643e22f1f3 100644 --- a/templates/dataset/security_common.mako +++ b/templates/dataset/security_common.mako @@ -29,7 +29,8 @@ -<%def name="render_permission_form( obj, obj_name, form_url, id_name, id, all_roles )"> +## Any permission ( e.g., 'DATASET_ACCESS' ) included in the do_not_render param will not be rendered on the page. +<%def name="render_permission_form( obj, obj_name, form_url, id_name, id, all_roles, do_not_render=[] )"> <% if isinstance( obj, trans.app.model.User ): current_actions = obj.default_permissions @@ -73,9 +74,11 @@
%for k, v in trans.app.model.Dataset.permitted_actions.items(): -
- ${render_select( current_actions, k, v, all_roles )} -
+ %if k not in do_not_render: +
+ ${render_select( current_actions, k, v, all_roles )} +
+ %endif %endfor
From 4bb40fba5485d1f78beee26e2c12e43ea3c21786 Mon Sep 17 00:00:00 2001 From: Nate Coraor Date: Mon, 19 Jan 2009 11:47:29 -0500 Subject: [PATCH 16/18] Flush newly created users directly in __get_or_create_remote_user, solves the new user creation when using remote_user bug found by Ross Lazarus. Also, external users get a random password now (makes the secret key safer if a bad guy gets the database). --- lib/galaxy/web/framework/__init__.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/lib/galaxy/web/framework/__init__.py b/lib/galaxy/web/framework/__init__.py index 4bceb09b491..7abd179d478 100644 --- a/lib/galaxy/web/framework/__init__.py +++ b/lib/galaxy/web/framework/__init__.py @@ -4,7 +4,7 @@ Galaxy web application framework import pkg_resources -import os, sys, time +import os, sys, time, random, string pkg_resources.require( "Cheetah" ) from Cheetah.Template import Template import base @@ -229,6 +229,9 @@ class UniverseWebTransaction( base.DefaultWebTransaction ): user_for_new_session = self.__get_or_create_remote_user( remote_user_email ) log.warning( "User logged in as '%s' externally, but has a cookie as '%s' invalidating session", remote_user_email, prev_galaxy_session.user.email ) + else: + # No session exists, get/create user for new session + user_for_new_session = self.__get_or_create_remote_user( remote_user_email ) else: if galaxy_session is not None and galaxy_session.user and galaxy_session.user.external: # Remote user support is not enabled, but there is an existing @@ -282,15 +285,15 @@ class UniverseWebTransaction( base.DefaultWebTransaction ): def __get_or_create_remote_user( self, remote_user_email ): """ Return the user in $HTTP_REMOTE_USER and create if necessary - - Caller is responsible for flushing the returned user. """ # remote_user middleware ensures HTTP_REMOTE_USER exists user = self.app.model.User.filter_by( email=remote_user_email ).first() if user is None: + random.seed() user = self.app.model.User( email=remote_user_email ) - user.set_password_cleartext( 'external' ) + user.set_password_cleartext( ''.join( random.sample( string.letters + string.digits, 12 ) ) ) user.external = True + user.flush() #self.log_event( "Automatically created account '%s'", user.email ) return user def __update_session_cookie( self, name='galaxysession' ): From 2c85c6fee962f972722afc60781ee2090f3c48d1 Mon Sep 17 00:00:00 2001 From: Anton Nekrutenko Date: Mon, 19 Jan 2009 12:12:36 -0500 Subject: [PATCH 17/18] Preliminary commity for lca. Still in progress. --- static/welcome.html | 107 +++++++++++++++++++++---------- tools/taxonomy/find_diag_hits.py | 7 +- tools/taxonomy/lca.py | 47 +++++++++++++- tools/taxonomy/lca.xml | 38 ++++++++++- 4 files changed, 159 insertions(+), 40 deletions(-) diff --git a/static/welcome.html b/static/welcome.html index 5bd2ab245cd..08021b9fe94 100644 --- a/static/welcome.html +++ b/static/welcome.html @@ -6,47 +6,88 @@ + + + + + + + + + + +
-
- Workflows are finally here! -
- Watch how you can (Click link to play)... - -
- For more screencasts click here. -
-
+
+ The Galaxy Main Server was recently upgraded. -
- Unsequenced Genomes of the World | October 2008 -
-
- -
- Costa's hummingbird (Calypte costae) | Kings Canyon NP, California -
-
+
+ There was some instability as Galaxy was restarted numerous times between 8:30 and 10:00 AM EST (UTC -0500) this morning (Wednesday, January 14). Any jobs from this time that ended in error can be tried again. If you continue to have any problems, please contact the Galaxy Team by using the 'report this error' link within a red history item, or emailing galaxy-bugs@bx.psu.edu. +
+ +

+

+ We are hiring! +
+ Thanks to your support and an unprecedented level of usage, we are looking for an experienced software developer to join our team. For more information about this position, please, click here.

-
- Galaxy is for Biologists -
- Use this site to access popular sources of data like the UCSC Table Browser. Run analyses right on the spot using a variety of integrated tools. Your results are always available and can be easily shared with others. Just watch how. -
-
-
- Galaxy is for Developers -
- Galaxy is an easy-to-use, open-source, scalable framework for tool and data integration. Stop wasting time writing interfaces and get your tools used by biologists! Galaxy includes everything you need to get started, so download and start integrating! -
+ +
+Introducing Galactic Quickies +
+ Galactic quickies are super-short screencasts that are always under 5 minutes. We thought it may be a good way to spread the word about Galaxy's functionality while keeping the "annoyance factor" to the minimum. The quickies will be updated weekly. +
+
+ + + +
+ +
+ + + +
+
+ For a high resolution version click here. +
+
+ + +
+

Galaxy team is a part of BX at Penn State.


This project is supported in part by NSF and the Huck Institutes of the Life Sciences.

Galaxy build: $Rev$

diff --git a/tools/taxonomy/find_diag_hits.py b/tools/taxonomy/find_diag_hits.py index a481cab65dd..993c066bb2b 100644 --- a/tools/taxonomy/find_diag_hits.py +++ b/tools/taxonomy/find_diag_hits.py @@ -71,7 +71,8 @@ taxRank = { 'genus' :20, 'subgenus' :21, 'species' :22, - 'subspecies' :23 + 'subspecies' :23, + 'order' :13 } @@ -157,12 +158,16 @@ try: for item in cur.fetchall(): out_string = '%s\t%s\t%d\t' % ( item[0], item[1], item[2] ) out_string += rankName + out_string += '\t' + out_string += str(taxRank[rankName]) print >>out_file, out_string else: cur.execute('select rank, count(*) from %s_count where N = 1 and length(rank)>1 group by rank' % rank) for item in cur.fetchall(): out_string = '%s\t%s\t' % ( item[0], item[1] ) out_string += rankName + out_string += '\t' + out_string += str(taxRank[rankName]) print >>out_file, out_string except Exception, e: stop_err("%s\n" % e) diff --git a/tools/taxonomy/lca.py b/tools/taxonomy/lca.py index beb356f160b..0b9785d9dd5 100644 --- a/tools/taxonomy/lca.py +++ b/tools/taxonomy/lca.py @@ -14,6 +14,32 @@ def main(): try: inputfile = sys.argv[1] outfile = sys.argv[2] + rank_bound = int( sys.argv[3] ) + """ + Mapping of ranks: + root :2, + superkingdom:3, + kingdom :4, + subkingdom :5, + superphylum :6, + phylum :7, + subphylum :8, + superclass :9, + class :10, + subclass :11, + superorder :12, + order :13, + suborder :14, + superfamily :15, + family :16, + subfamily :17, + tribe :18, + subtribe :19, + genus :20, + subgenus :21, + species :22, + subspecies :23, + """ except: stop_err("Syntax error: Use correct syntax: program infile outfile") group_col = 0 @@ -91,7 +117,16 @@ def main(): out_list[k+1] = 'n' k += 1 - print >>fout, '\t'.join(out_list) + # print >>fout, '\t'.join(out_list) + + if rank_bound == 0: + print >>fout, ''.join(out_list) + print 'n'*( 24 - rank_bound ) + else: + print '\t'.join(out_list[rank_bound:24]) + if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ): + print >>fout, '\t'.join(out_list) + block_valid = True prev_item = item @@ -134,9 +169,15 @@ def main(): while k < 23: out_list[k+1] = 'n' k += 1 - - print >>fout, '\t'.join(out_list) + if rank_bound == 0: + print >>fout, '\t'.join(out_list) + else: + print ''.join(out_list[rank_bound:24]) + print 'n'*( 24 - rank_bound ) + if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ): + print >>fout, '\t'.join(out_list) + if skipped_lines > 0: msg= "Skipped %d invalid lines starting with line %d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column ) print msg diff --git a/tools/taxonomy/lca.xml b/tools/taxonomy/lca.xml index 8ba556c50dd..f608a16cfc3 100644 --- a/tools/taxonomy/lca.xml +++ b/tools/taxonomy/lca.xml @@ -1,12 +1,44 @@ - lca.py $input1 $out_file1 + lca.py $input1 $out_file1 $rank_bound - + + + + + + + + + + + + + + + + + + + + + + + + + - + + + +**What it does** + +When performing metagenomic analyses it is often necessary to identify sequence reads corresponding to a particular taxonomic group, or, in other words, diagnostic of a particular taxonomic rank. This utility performs this analysis. It takes data generated by *Taxonomy manipulation->Fetch Taxonomic Ranks* as input and outputs either a list of sequence reads unique to a particular taxonomic rank, or a list of taxonomic ranks and the count of unique reads corresponding to each rank. + + + \ No newline at end of file From 498fbac868b1e20178fb89d9141edb9b59d48bf8 Mon Sep 17 00:00:00 2001 From: Wen-Yu Chung Date: Mon, 19 Jan 2009 15:47:02 -0500 Subject: [PATCH 18/18] remove the usage of hast tables in the fasta-tools --- tools/fasta_tools/fasta_compute_length.py | 50 +++++++-------- tools/fasta_tools/fasta_filter_by_length.py | 69 ++++++++++++--------- tools/fasta_tools/fasta_to_tabular.py | 39 +++++------- 3 files changed, 80 insertions(+), 78 deletions(-) diff --git a/tools/fasta_tools/fasta_compute_length.py b/tools/fasta_tools/fasta_compute_length.py index 3893072844b..6086ac2716a 100644 --- a/tools/fasta_tools/fasta_compute_length.py +++ b/tools/fasta_tools/fasta_compute_length.py @@ -1,8 +1,8 @@ #! /usr/bin/python """ -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. +Input: fasta, int +Output: tabular +Return titles with lengths of corresponding seq """ import sys, os @@ -10,41 +10,37 @@ import sys, os assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): - input_filename = sys.argv[1] - output_filename = sys.argv[2] + + infile = sys.argv[1] + outfile = sys.argv[2] keep_first = int( sys.argv[3] ) - tmp_title = tmp_seq = '' - tmp_seq_count = 0 - seq_hash = {} + + fasta_title = fasta_seq = '' + # number of char to keep in the title if keep_first == 0: keep_first = None else: keep_first += 1 - for i, line in enumerate( file( input_filename ) ): + out = open(outfile, 'w') + + for i, line in enumerate( file( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue if line[0] == '>': - if len( tmp_seq ) > 0: - tmp_seq_count += 1 - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - tmp_title = line - tmp_seq = '' + if len( fasta_seq ) > 0 : + out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) ) + fasta_title = line + fasta_seq = '' else: - tmp_seq = "%s%s" % ( tmp_seq, line ) - if line.split() and line.split()[0].isdigit(): - tmp_seq = "%s " % tmp_seq - if len( tmp_seq ) > 0: - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - - title_keys = seq_hash.keys() - title_keys.sort() - output_handle = open( output_filename, 'w' ) - for i, fasta_title in title_keys: - tmp_seq = seq_hash[ ( i, fasta_title ) ] - output_handle.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( tmp_seq ) ) ) - output_handle.close() + fasta_seq = "%s%s" % ( fasta_seq, line ) + + # check the last sequence + if len( fasta_seq ) > 0: + out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) ) + + out.close() if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_filter_by_length.py b/tools/fasta_tools/fasta_filter_by_length.py index 4947e7b333f..9906300e7ef 100644 --- a/tools/fasta_tools/fasta_filter_by_length.py +++ b/tools/fasta_tools/fasta_filter_by_length.py @@ -15,7 +15,7 @@ def stop_err( msg ): def __main__(): - input_filename = sys.argv[1] + infile = sys.argv[1] try: min_length = int( sys.argv[2] ) except: @@ -24,49 +24,62 @@ def __main__(): max_length = int( sys.argv[3] ) except: stop_err( "Maximum length of the return sequence requires a numerical value." ) - output_filename = sys.argv[4] - tmp_title = tmp_seq = '' - tmp_seq_count = 0 - seq_hash = {} + outfile = sys.argv[4] + fasta_title = fasta_seq = '' + at_least_one = 0 + + out = open( outfile, 'w' ) - for i, line in enumerate( file( input_filename ) ): + for i, line in enumerate( file( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue + if line[0] == '>': - if len( tmp_seq ) > 0: - tmp_seq_count += 1 - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - tmp_title = line - tmp_seq = '' + if len( fasta_seq ) > 0: + + if max_length <= 0: + compare_max_length = len( fasta_seq ) + 1 + else: + compare_max_length = max_length + + l = len( fasta_seq ) + + if l >= min_length and l <= compare_max_length: + at_least_one += 1 + out.write( "%s\n" % fasta_title ) + c = 0 + s = fasta_seq + while c < l: + b = min( c + 50, l ) + out.write( "%s\n" % s[ c:b ] ) + c = b + + fasta_title = line + fasta_seq = '' else: - tmp_seq = "%s%s" % ( tmp_seq, line ) - if line.split()[0].isdigit(): - tmp_seq = "%s " % tmp_seq - if len( tmp_seq ) > 0: - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq + fasta_seq = "%s%s" % ( fasta_seq, line ) - title_keys = seq_hash.keys() - title_keys.sort() - output_handle = open( output_filename, 'w' ) - at_least_one = 0 - for i, fasta_title in title_keys: - tmp_seq = seq_hash[ ( i, fasta_title ) ] + if len( fasta_seq ) > 0: + if max_length <= 0: - compare_max_length = len( tmp_seq ) + 1 + compare_max_length = len( fasta_seq ) + 1 else: compare_max_length = max_length - l = len( tmp_seq ) + + l = len( fasta_seq ) + if l >= min_length and l <= compare_max_length: at_least_one += 1 - output_handle.write( "%s\n" % fasta_title ) + out.write( "%s\n" % fasta_title ) c = 0 - s = tmp_seq + s = fasta_seq while c < l: b = min( c + 50, l ) - output_handle.write( "%s\n" % s[ c:b ] ) + out.write( "%s\n" % s[ c:b ] ) c = b - output_handle.close() + + out.close() if at_least_one == 0: print "There is no sequence that falls within your range." diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py index 42e79e63e2a..8a18d0c69c4 100644 --- a/tools/fasta_tools/fasta_to_tabular.py +++ b/tools/fasta_tools/fasta_to_tabular.py @@ -1,9 +1,9 @@ #! /usr/bin/python # This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools """ -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. +Input: fasta, int +Output: tabular +format convert: fasta to tabular """ import sys, os @@ -14,38 +14,31 @@ def __main__(): infile = sys.argv[1] outfile = sys.argv[2] keep_first = int( sys.argv[3] ) - title = '' - sequence = '' - sequence_count = 0 + fasta_title = fasta_seq = '' if keep_first == 0: keep_first = None else: keep_first += 1 + out = open( outfile, 'w' ) + for i, line in enumerate( open( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue if line.startswith( '>' ): - if sequence: - sequence_count += 1 - seq_hash[( sequence_count, title )] = sequence - title = line - sequence = '' + if fasta_seq: + out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) ) + fasta_title = line + fasta_seq = '' else: - sequence = "%s%s" % ( sequence, line ) - if line.split() and line.split()[0].isdigit(): - sequence += ' ' - if sequence: - seq_hash[( sequence_count, title )] = sequence - # return only those lengths are in the range - title_keys = seq_hash.keys() - title_keys.sort() - out = open( outfile, 'w' ) - for i, fasta_title in title_keys: - sequence = seq_hash[( i, fasta_title )] - out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], sequence ) ) + if line: + fasta_seq = "%s%s" % ( fasta_seq, line ) + + if fasta_seq: + out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) ) + out.close() if __name__ == "__main__" : __main__() \ No newline at end of file