diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index 391f7937663..76bba775482 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -38,6 +38,7 @@ + @@ -172,20 +173,24 @@ - - - - - - - - - - - - - + + + + + + + + + + + + + + + diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index cf5f430c7a7..863fd01222d 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -106,8 +106,12 @@ class Data( object ): return False def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = '' - dataset.blurb = 'data' + if not dataset.dataset.purged: + dataset.peek = '' + dataset.blurb = 'data' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): """Create HTML table, used for displaying peek""" out = [''] @@ -135,7 +139,8 @@ class Data( object ): def display_info(self, dataset): """Returns formated html of dataset info""" try: - return escape(dataset.info) + # Change new line chars to html + return escape( dataset.info ).replace( "\r", "\n" ).replace( "\n", "
" ) except: return "info unavailable" def validate(self, dataset): @@ -282,19 +287,27 @@ class Text( Data ): return 'text/plain' def set_peek( self, dataset, line_count=None ): - dataset.peek = get_file_peek( dataset.file_name ) - if line_count is None: - dataset.blurb = "%s lines" % util.commaify( str( get_line_count( dataset.file_name ) ) ) + if not dataset.dataset.purged: + # The file must exist on disk for the get_file_peek() method + dataset.peek = get_file_peek( dataset.file_name ) + if line_count is None: + dataset.blurb = "%s lines" % util.commaify( str( get_line_count( dataset.file_name ) ) ) + else: + dataset.blurb = "%s lines" % util.commaify( str( line_count ) ) else: - dataset.blurb = "%s lines" % util.commaify( str( line_count ) ) + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' class Binary( Data ): """Binary data""" - def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = 'binary data' - dataset.blurb = 'data' + if not dataset.dataset.purged: + dataset.peek = 'binary data' + dataset.blurb = 'data' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def get_test_fname( fname ): """Returns test data filename""" diff --git a/lib/galaxy/datatypes/genetics.py b/lib/galaxy/datatypes/genetics.py index 27032ecc2b2..6edfbc64264 100644 --- a/lib/galaxy/datatypes/genetics.py +++ b/lib/galaxy/datatypes/genetics.py @@ -43,11 +43,14 @@ class GenomeGraphs( Tabular ): def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = data.get_file_peek( dataset.file_name ) - ## dataset.peek = self.make_html_table( dataset.peek ) - dataset.blurb = util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + " rows" - #i don't think set_meta should not be called here, it should be called separately - self.set_meta( dataset ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + " rows" + #i don't think set_meta should not be called here, it should be called separately + self.set_meta( dataset ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def get_estimated_display_viewport( self, dataset ): """Return a chrom, start, stop tuple for viewing a file.""" @@ -130,8 +133,12 @@ class SNPMatrix(Rgenetics): file_ext="snpmatrix" def set_peek( self, dataset ): - dataset.peek = "Binary RGenetics file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = "Binary RGenetics file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' #def sniff( self, filename ): # """ # """ diff --git a/lib/galaxy/datatypes/images.py b/lib/galaxy/datatypes/images.py index afb17fb4e21..3d498578933 100644 --- a/lib/galaxy/datatypes/images.py +++ b/lib/galaxy/datatypes/images.py @@ -14,9 +14,13 @@ class Ab1( data.Data ): """Class describing an ab1 binary sequence file""" file_ext = "ab1" def set_peek( self, dataset ): - export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'ab1','name':'ab1 sequence','info':'Sequence file','dbkey':dataset.dbkey}) - dataset.peek = "Binary ab1 sequence file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + export_url = "/history_add_to?" + urlencode({'history_id':dataset.history_id,'ext':'ab1','name':'ab1 sequence','info':'Sequence file','dbkey':dataset.dbkey}) + dataset.peek = "Binary ab1 sequence file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -27,9 +31,13 @@ class Scf( data.Data ): """Class describing an scf binary sequence file""" file_ext = "scf" def set_peek( self, dataset ): - export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'scf','name':'scf sequence','info':'Sequence file','dbkey':dataset.dbkey}) - dataset.peek = "Binary scf sequence file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + export_url = "/history_add_to?" + urlencode({'history_id':dataset.history_id,'ext':'scf','name':'scf sequence','info':'Sequence file','dbkey':dataset.dbkey}) + dataset.peek = "Binary scf sequence file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -40,10 +48,14 @@ class Binseq( data.Data ): """Class describing a zip archive of binary sequence files""" file_ext = "binseq.zip" def set_peek( self, dataset ): - zip_file = zipfile.ZipFile( dataset.file_name, "r" ) - num_files = len( zip_file.namelist() ) - dataset.peek = "Archive of %s binary sequence files" % ( str( num_files ) ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + zip_file = zipfile.ZipFile( dataset.file_name, "r" ) + num_files = len( zip_file.namelist() ) + dataset.peek = "Archive of %s binary sequence files" % ( str( num_files ) ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -57,10 +69,14 @@ class Txtseq( data.Data ): """Class describing a zip archive of text sequence files""" file_ext = "txtseq.zip" def set_peek( self, dataset ): - zip_file = zipfile.ZipFile( dataset.file_name, "r" ) - num_files = len( zip_file.namelist() ) - dataset.peek = "Archive of %s text sequence files" % ( str( num_files ) ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + zip_file = zipfile.ZipFile( dataset.file_name, "r" ) + num_files = len( zip_file.namelist() ) + dataset.peek = "Archive of %s text sequence files" % ( str( num_files ) ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -73,8 +89,12 @@ class Txtseq( data.Data ): class Image( data.Data ): """Class describing an image""" def set_peek( self, dataset ): - dataset.peek = 'Image in %s format' % dataset.extension - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = 'Image in %s format' % dataset.extension + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def create_applet_tag_peek( class_name, archive, params ): text = """ @@ -105,21 +125,26 @@ class Gmaj( data.Data ): file_ext = "gmaj.zip" copy_safe_peek = False def set_peek( self, dataset ): - if hasattr( dataset, 'history_id' ): - params = { - "bundle":"display?id=%s&tofile=yes&toext=.zip" % dataset.id, - "buttonlabel": "Launch GMAJ", - "nobutton": "false", - "urlpause" :"100", - "debug": "false", - "posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'maf', 'name': 'GMAJ Output on data %s' % dataset.hid, 'info': 'Added by GMAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) ) - } - class_name = "edu.psu.bx.gmaj.MajApplet.class" - archive = "/static/gmaj/gmaj.jar" - dataset.peek = create_applet_tag_peek( class_name, archive, params ) + if not dataset.dataset.purged: + if hasattr( dataset, 'history_id' ): + params = { + "bundle":"display?id=%s&tofile=yes&toext=.zip" % dataset.id, + "buttonlabel": "Launch GMAJ", + "nobutton": "false", + "urlpause" :"100", + "debug": "false", + "posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'maf', 'name': 'GMAJ Output on data %s' % dataset.hid, 'info': 'Added by GMAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) ) + } + class_name = "edu.psu.bx.gmaj.MajApplet.class" + archive = "/static/gmaj/gmaj.jar" + dataset.peek = create_applet_tag_peek( class_name, archive, params ) + dataset.blurb = 'GMAJ Multiple Alignment Viewer' + else: + dataset.peek = "After you add this item to your history, you will be able to launch the GMAJ applet." + dataset.blurb = 'GMAJ Multiple Alignment Viewer' else: - dataset.peek = "After you add this item to your history, you will be able to launch the GMAJ applet." - dataset.blurb = 'GMAJ Multiple Alignment Viewer' + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek @@ -151,8 +176,12 @@ class Html( data.Text ): """Class describing an html file""" file_ext = "html" def set_peek( self, dataset ): - dataset.peek = "HTML file" - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = "HTML file" + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def get_mime(self): """Returns the mime type of the datatype""" return 'text/html' @@ -181,20 +210,24 @@ class Laj( data.Text ): file_ext = "laj" copy_safe_peek = False def set_peek( self, dataset ): - if hasattr( dataset, 'history_id' ): - params = { - "alignfile1": "display?id=%s" % dataset.id, - "buttonlabel": "Launch LAJ", - "title": "LAJ in Galaxy", - "posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'lav', 'name': 'LAJ Output', 'info': 'Added by LAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) ), - "noseq": "true" - } - class_name = "edu.psu.cse.bio.laj.LajApplet.class" - archive = "/static/laj/laj.jar" - dataset.peek = create_applet_tag_peek( class_name, archive, params ) + if not dataset.dataset.purged: + if hasattr( dataset, 'history_id' ): + params = { + "alignfile1": "display?id=%s" % dataset.id, + "buttonlabel": "Launch LAJ", + "title": "LAJ in Galaxy", + "posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'lav', 'name': 'LAJ Output', 'info': 'Added by LAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) ), + "noseq": "true" + } + class_name = "edu.psu.cse.bio.laj.LajApplet.class" + archive = "/static/laj/laj.jar" + dataset.peek = create_applet_tag_peek( class_name, archive, params ) + else: + dataset.peek = "After you add this item to your history, you will be able to launch the LAJ applet." + dataset.blurb = 'LAJ Multiple Alignment Viewer' else: - dataset.peek = "After you add this item to your history, you will be able to launch the LAJ applet." - dataset.blurb = 'LAJ Multiple Alignment Viewer' + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: return dataset.peek diff --git a/lib/galaxy/datatypes/interval.py b/lib/galaxy/datatypes/interval.py index 940af654523..3ee543ca454 100644 --- a/lib/galaxy/datatypes/interval.py +++ b/lib/galaxy/datatypes/interval.py @@ -59,11 +59,15 @@ class Interval( Tabular ): def set_peek( self, dataset, line_count=None ): """Set the peek and blurb text""" - dataset.peek = data.get_file_peek( dataset.file_name ) - if line_count is None: - dataset.blurb = "%s regions" % util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + if line_count is None: + dataset.blurb = "%s regions" % util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + else: + dataset.blurb = "%s regions" % util.commaify( str( line_count ) ) else: - dataset.blurb = "%s regions" % util.commaify( str( line_count ) ) + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def set_meta( self, dataset, overwrite = True, first_line_is_header = False, **kwd ): Tabular.set_meta( self, dataset, overwrite = overwrite, skip = 0 ) diff --git a/lib/galaxy/datatypes/qualityscore.py b/lib/galaxy/datatypes/qualityscore.py index cfd5b280adb..cdd66b287f4 100644 --- a/lib/galaxy/datatypes/qualityscore.py +++ b/lib/galaxy/datatypes/qualityscore.py @@ -16,11 +16,15 @@ class QualityScore ( data.Text ): file_ext = "qual" def set_peek( self, dataset, line_count=None ): - dataset.peek = data.get_file_peek( dataset.file_name ) - if line_count is None: - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + if line_count is None: + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.blurb = "%s lines, Quality score file" % util.commaify( str( line_count ) ) else: - dataset.blurb = "%s lines, Quality score file" % util.commaify( str( line_count ) ) + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def display_peek(self, dataset): try: @@ -28,5 +32,43 @@ class QualityScore ( data.Text ): except: return "Quality score file (%s)" % ( data.nice_size( dataset.get_size() ) ) +class SolidQualityScore( data.Text ): + """ + Quality scores generated by ABI SOLiD + """ + file_ext = "solidqual" - \ No newline at end of file + def set_peek( self, dataset ): + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + def sniff( self, filename ): + """ + >>> fname = get_test_fname( 'sequence.fasta' ) + >>> SolidQualityScore().sniff( fname ) + False + >>> fname = get_test_fname( 'sequence.solidqual' ) + >>> SolidQualityScore().sniff( fname ) + True + """ + try: + fh = open( filename ) + while True: + line = fh.readline() + if not line: + break #EOF + line = line.strip() + if line and not line.startswith( '#' ): #first non-empty non-comment line + if line.startswith( '>' ): + line = fh.readline().strip() + if line == '' or line.startswith( '>' ): + break + try: + [ int( x ) for x in line.split() ] + except: + break + return True + else: + break #we found a non-empty line, but it's not a header + except: + pass + return False diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 1639e904964..816851d7437 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -59,18 +59,15 @@ class Registry( object ): self.converters.append( ( converter_config, extension, target_datatype ) ) except Exception, e: self.log.warning( 'Error loading datatype "%s", problem: %s' % ( extension, str( e ) ) ) - # Load datatype sniffers from config + # Load datatype sniffers from the config sniff_order = [] sniffers = root.find( 'sniffers' ) for elem in sniffers.findall( 'sniffer' ): - order = elem.get( 'order', None ) type = elem.get( 'type', None ) - if order and type: - sniff_order.append( ( order, type ) ) - sniff_order.sort() - for ele in sniff_order: + if type: + sniff_order.append( type ) + for type in sniff_order: try: - type = ele[1] fields = type.split( ":" ) datatype_module = fields[0] datatype_class = fields[1] @@ -149,6 +146,7 @@ class Registry( object ): sequence.Maf(), sequence.Lav(), sequence.Fasta(), + sequence.FastqSolexa(), interval.Wiggle(), images.Html(), sequence.Axt(), diff --git a/lib/galaxy/datatypes/sequence.py b/lib/galaxy/datatypes/sequence.py index 2ec55345cc3..10d54cffa03 100644 --- a/lib/galaxy/datatypes/sequence.py +++ b/lib/galaxy/datatypes/sequence.py @@ -5,6 +5,7 @@ Image classes import data import logging import re +import string from cgi import escape from galaxy.datatypes.metadata import MetadataElement from galaxy.datatypes import metadata @@ -31,8 +32,12 @@ class Fasta( Sequence ): file_ext = "fasta" def set_peek( self, dataset ): - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ @@ -86,30 +91,60 @@ class csFasta( Sequence ): file_ext = "csfasta" def set_peek( self, dataset ): - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ Color-space sequence: >2_15_85_F3 T213021013012303002332212012112221222112212222 - - TODO: - add sniff function - """ - - return False - - + >>> fname = get_test_fname( 'sequence.fasta' ) + >>> csFasta().sniff( fname ) + False + >>> fname = get_test_fname( 'sequence.csfasta' ) + >>> csFasta().sniff( fname ) + True + """ + try: + fh = open( filename ) + while True: + line = fh.readline() + if not line: + break #EOF + line = line.strip() + if line and not line.startswith( '#' ): #first non-empty non-comment line + if line.startswith( '>' ): + line = fh.readline().strip() + if line == '' or line.startswith( '>' ): + break + elif line[0] not in string.ascii_uppercase: + return False + elif len( line ) > 1 and not re.search( '^\d+$', line[1:] ): + return False + return True + else: + break #we found a non-empty line, but it's not a header + except: + pass + return False + class FastqSolexa( Sequence ): """Class representing a FASTQ sequence ( the Solexa variant )""" file_ext = "fastqsolexa" def set_peek( self, dataset ): - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = data.nice_size( dataset.get_size() ) + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = data.nice_size( dataset.get_size() ) + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ diff --git a/lib/galaxy/datatypes/xml.py b/lib/galaxy/datatypes/xml.py index d658e33308f..e496aad00ab 100644 --- a/lib/galaxy/datatypes/xml.py +++ b/lib/galaxy/datatypes/xml.py @@ -12,8 +12,12 @@ class BlastXml( data.Text ): file_ext = "blastxml" def set_peek( self, dataset ): """Set the peek and blurb text""" - dataset.peek = data.get_file_peek( dataset.file_name ) - dataset.blurb = 'NCBI Blast XML data' + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name ) + dataset.blurb = 'NCBI Blast XML data' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' def sniff( self, filename ): """ Determines whether the file is blastxml diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 058bc31e1d0..427c0fe7424 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -111,19 +111,19 @@ class JobQueue( object ): def __check_jobs_at_startup( self ): """ - Checks all jobs that are in the 'running' or 'queued' state in the - database and requeues or cleans up as necessary. Only run as the + Checks all jobs that are in the 'new', 'queued' or 'running' state in + the database and requeues or cleans up as necessary. Only run as the job manager starts. """ model = self.app.model - # Jobs in the NEW state won't be requeued unless we're tracking in the database - if not self.track_jobs_in_database: - for job in model.Job.filter( model.Job.c.state==model.Job.states.NEW ).all(): - log.debug( "no runner: %s is still in new state, adding to the jobs queue" %job.id ) - self.queue.put( ( job.id, job.tool_id ) ) + for job in model.Job.filter( model.Job.c.state==model.Job.states.NEW ).all(): + log.debug( "no runner: %s is still in new state, adding to the jobs queue" %job.id ) + self.queue.put( ( job.id, job.tool_id ) ) for job in model.Job.filter( (model.Job.c.state == model.Job.states.RUNNING) | (model.Job.c.state == model.Job.states.QUEUED) ).all(): - if job.job_runner_name is not None: - # why are we passing the queue to the wrapper? + if job.job_runner_name is None: + log.debug( "no runner: %s is still in queued state, adding to the jobs queue" %job.id ) + self.queue.put( ( job.id, job.tool_id ) ) + else: job_wrapper = JobWrapper( job, self.app.toolbox.tools_by_id[ job.tool_id ], self ) self.dispatcher.recover( job, job_wrapper ) @@ -298,9 +298,13 @@ class JobWrapper( object ): self.queue = queue self.app = queue.app self.extra_filenames = [] - self.working_directory = None self.command_line = None self.galaxy_lib_dir = None + # With job outputs in the working directory, we need the working + # directory to be set before prepare is run, or else premature deletion + # and job recovery fail. + self.working_directory = \ + os.path.join( self.app.config.job_working_directory, str( self.job_id ) ) def get_param_dict( self ): """ @@ -317,9 +321,6 @@ class JobWrapper( object ): config files. """ mapping.context.current.clear() #this prevents the metadata reverting that has been seen in conjunction with the PBS job runner - # Create the working directory - self.working_directory = \ - os.path.join( self.app.config.job_working_directory, str( self.job_id ) ) if not os.path.exists( self.working_directory ): os.mkdir( self.working_directory ) # Restore parameters from the database @@ -382,12 +383,12 @@ class JobWrapper( object ): if not job.state == model.Job.states.DELETED: for dataset_assoc in job.output_datasets: if self.app.config.outputs_to_working_directory: - false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.id ) ) - if os.path.exists( false_path ): + false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.dataset.id ) ) + try: shutil.move( false_path, dataset_assoc.dataset.file_name ) log.debug( "fail(): Moved %s to %s" % ( false_path, dataset_assoc.dataset.file_name ) ) - else: - log.warning( "fail(): Missing output file in working directory: %s" % false_path ) + except ( IOError, OSError ), e: + log.error( "fail(): Missing output file in working directory: %s" % e ) dataset = dataset_assoc.dataset dataset.refresh() dataset.state = dataset.states.ERROR @@ -452,12 +453,13 @@ class JobWrapper( object ): job.state = 'ok' for dataset_assoc in job.output_datasets: if self.app.config.outputs_to_working_directory: - false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.id ) ) - if os.path.exists( false_path ): + false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.dataset.id ) ) + try: shutil.move( false_path, dataset_assoc.dataset.file_name ) log.debug( "finish(): Moved %s to %s" % ( false_path, dataset_assoc.dataset.file_name ) ) - else: - log.warning( "finish(): Missing output file in working directory: %s" % false_path ) + except ( IOError, OSError ): + self.fail( "The job's output dataset(s) could not be read" ) + return for dataset in dataset_assoc.dataset.dataset.history_associations: #need to update all associated output hdas, i.e. history was shared with job running dataset.blurb = 'done' dataset.peek = 'no peek' diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index 4c64f850cc1..655fe6893da 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -912,3 +912,4 @@ def directory_hash_id( id ): # Break into chunks of three return [ padded[i*3:(i+1)*3] for i in range( len( padded ) // 3 ) ] + diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index f552367aabb..85d0b4bcc42 100644 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -1072,10 +1072,11 @@ class Tool: datatypes_registry = self.app.datatypes_registry, tool = self, name = input.name ) + elif isinstance( input, SelectToolParameter ): + input_values[ input.name ] = SelectToolParameterWrapper( input, input_values[ input.name ], self.app ) else: - input_values[ input.name ] = \ - InputValueWrapper( input, input_values[ input.name ], param_dict ) - # HACK: only wrap if check_values is false, this deals with external + input_values[ input.name ] = InputValueWrapper( input, input_values[ input.name ], param_dict ) + # HACK: only wrap if check_values is not false, this deals with external # tools where the inputs don't even get passed through. These # tools (e.g. UCSC) should really be handled in a special way. if self.check_values: @@ -1098,7 +1099,7 @@ class Tool: for name, data in output_datasets.items(): # Write outputs to the working directory (for security purposes) if desired. if self.app.config.outputs_to_working_directory and working_directory is not None: - false_path = os.path.abspath( os.path.join( working_directory, "galaxy_dataset_%d.dat" % data.id ) ) + false_path = os.path.abspath( os.path.join( working_directory, "galaxy_dataset_%d.dat" % data.dataset.id ) ) param_dict[name] = DatasetFilenameWrapper( data, false_path = false_path ) open( false_path, 'w' ).close() else: @@ -1415,7 +1416,21 @@ class InputValueWrapper( object ): return self.input.to_param_dict_string( self.value, self._other_values ) def __getattr__( self, key ): return getattr( self.value, key ) - + +class SelectToolParameterWrapper( object ): + """ + Wraps a SelectTooParameter so that __str__ returns the selected value, but all other + attributes are accessible. + """ + def __init__( self, input, value, app ): + self.input = input + self.value = value + self.input.value_label = input.value_to_display_text( value, app ) + def __str__( self ): + return self.input.to_param_dict_string( self.value ) + def __getattr__( self, key ): + return getattr( self.input, key ) + class DatasetFilenameWrapper( object ): """ Wraps a dataset so that __str__ returns the filename, but all other diff --git a/lib/galaxy/tools/actions/__init__.py b/lib/galaxy/tools/actions/__init__.py index 11e587916e2..98c9c1f31b4 100644 --- a/lib/galaxy/tools/actions/__init__.py +++ b/lib/galaxy/tools/actions/__init__.py @@ -1,9 +1,11 @@ from galaxy.util.bunch import Bunch from galaxy.tools.parameters import * +from galaxy.tools.parameters.grouping import * from galaxy.util.template import fill_template from galaxy.util.none_like import NoneDataset from galaxy.web import url_for from galaxy.jobs import JOB_OK +import galaxy.tools import logging log = logging.getLogger( __name__ ) @@ -68,6 +70,26 @@ class DefaultToolAction( object ): return input_datasets def execute(self, tool, trans, incoming={}, set_output_hid=True ): + def wrap_values( inputs, input_values ): + # Wrap tool inputs as necessary + for input in inputs.itervalues(): + if isinstance( input, Repeat ): + for d in input_values[ input.name ]: + wrap_values( input.inputs, d ) + elif isinstance( input, Conditional ): + values = input_values[ input.name ] + current = values["__current_case__"] + wrap_values( input.cases[current].inputs, values ) + elif isinstance( input, DataToolParameter ): + input_values[ input.name ] = \ + galaxy.tools.DatasetFilenameWrapper( input_values[ input.name ], + datatypes_registry = trans.app.datatypes_registry, + tool = tool, + name = input.name ) + elif isinstance( input, SelectToolParameter ): + input_values[ input.name ] = galaxy.tools.SelectToolParameterWrapper( input, input_values[ input.name ], tool.app ) + else: + input_values[ input.name ] = galaxy.tools.InputValueWrapper( input, input_values[ input.name ], incoming ) out_data = {} # Collect any input datasets from the incoming parameters inp_data = self.collect_input_datasets( tool, incoming, trans ) @@ -163,6 +185,11 @@ class DefaultToolAction( object ): # Set output label if output.label: params = dict( incoming ) + # wrapping the params allows the tool config to contain things like + # + # + # + wrap_values( tool.inputs, params ) params['tool'] = tool params['on_string'] = on_text data.name = fill_template( output.label, context=params ) diff --git a/lib/galaxy/web/framework/__init__.py b/lib/galaxy/web/framework/__init__.py index 6605f3cd578..93226dcd98e 100644 --- a/lib/galaxy/web/framework/__init__.py +++ b/lib/galaxy/web/framework/__init__.py @@ -4,7 +4,7 @@ Galaxy web application framework import pkg_resources -import os, sys, time, socket +import os, sys, time, socket, random, string pkg_resources.require( "Cheetah" ) from Cheetah.Template import Template import base @@ -254,6 +254,9 @@ class UniverseWebTransaction( base.DefaultWebTransaction ): user_for_new_session = self.__get_or_create_remote_user( remote_user_email ) log.warning( "User logged in as '%s' externally, but has a cookie as '%s' invalidating session", remote_user_email, prev_galaxy_session.user.email ) + else: + # No session exists, get/create user for new session + user_for_new_session = self.__get_or_create_remote_user( remote_user_email ) else: if galaxy_session is not None and galaxy_session.user and galaxy_session.user.external: # Remote user support is not enabled, but there is an existing @@ -329,14 +332,16 @@ class UniverseWebTransaction( base.DefaultWebTransaction ): def __get_or_create_remote_user( self, remote_user_email ): """ Return the user in $HTTP_REMOTE_USER and create if necessary - Caller is responsible for flushing the returned user. """ # remote_user middleware ensures HTTP_REMOTE_USER exists user = self.app.model.User.filter( self.app.model.User.table.c.email==remote_user_email ).first() if user is None: + random.seed() user = self.app.model.User( email=remote_user_email ) - user.set_password_cleartext( 'external' ) + user.set_password_cleartext( ''.join( random.sample( string.letters + string.digits, 12 ) ) ) user.external = True + user.flush() + #self.log_event( "Automatically created account '%s'", user.email ) elif user.deleted: return self.show_error_message( "Your account is no longer valid, contact your Galaxy administrator to activate your account." ) return user diff --git a/static/june_2007_style/blue/tool_menu.css b/static/june_2007_style/blue/tool_menu.css index b0da725086e..b18e01624e5 100644 --- a/static/june_2007_style/blue/tool_menu.css +++ b/static/june_2007_style/blue/tool_menu.css @@ -31,15 +31,16 @@ div.toolSectionDetailsInner div.toolSectionTitle { - padding-bottom: 0px; font-weight: bold; } div.toolPanelLabel { - padding-top: 5px; + padding-top: 10px; padding-bottom: 5px; font-weight: bold; + color: gray; + text-transform: uppercase; } div.toolTitle @@ -52,9 +53,17 @@ div.toolTitle list-style: square outside; } +div.toolSectionBody div.toolPanelLabel +{ + padding-top: 5px; + padding-bottom: 5px; + margin-left: 16px; + margin-right: 10px; + display: list-item; + list-style: none outside; +} div.toolTitleNoSection { padding-bottom: 0px; - font-weight: bold; } diff --git a/static/june_2007_style/tool_menu.css.tmpl b/static/june_2007_style/tool_menu.css.tmpl index aa1178971a2..ef4b4ca7b28 100644 --- a/static/june_2007_style/tool_menu.css.tmpl +++ b/static/june_2007_style/tool_menu.css.tmpl @@ -31,15 +31,16 @@ div.toolSectionDetailsInner div.toolSectionTitle { - padding-bottom: 0px; font-weight: bold; } div.toolPanelLabel { - padding-top: 5px; + padding-top: 10px; padding-bottom: 5px; font-weight: bold; + color: gray; + text-transform: uppercase; } div.toolTitle @@ -52,9 +53,17 @@ div.toolTitle list-style: square outside; } +div.toolSectionBody div.toolPanelLabel +{ + padding-top: 5px; + padding-bottom: 5px; + margin-left: 16px; + margin-right: 10px; + display: list-item; + list-style: none outside; +} div.toolTitleNoSection { padding-bottom: 0px; - font-weight: bold; } diff --git a/static/welcome.html b/static/welcome.html index 5bd2ab245cd..08021b9fe94 100644 --- a/static/welcome.html +++ b/static/welcome.html @@ -6,47 +6,88 @@ + + + + + + + + + + +
-
- Workflows are finally here! -
- Watch how you can (Click link to play)... - -
- For more screencasts click here. -
-
+
+ The Galaxy Main Server was recently upgraded. -
- Unsequenced Genomes of the World | October 2008 -
-
- -
- Costa's hummingbird (Calypte costae) | Kings Canyon NP, California -
-
+
+ There was some instability as Galaxy was restarted numerous times between 8:30 and 10:00 AM EST (UTC -0500) this morning (Wednesday, January 14). Any jobs from this time that ended in error can be tried again. If you continue to have any problems, please contact the Galaxy Team by using the 'report this error' link within a red history item, or emailing galaxy-bugs@bx.psu.edu. +
+ +

+

+ We are hiring! +
+ Thanks to your support and an unprecedented level of usage, we are looking for an experienced software developer to join our team. For more information about this position, please, click here.

-
- Galaxy is for Biologists -
- Use this site to access popular sources of data like the UCSC Table Browser. Run analyses right on the spot using a variety of integrated tools. Your results are always available and can be easily shared with others. Just watch how. -
-
-
- Galaxy is for Developers -
- Galaxy is an easy-to-use, open-source, scalable framework for tool and data integration. Stop wasting time writing interfaces and get your tools used by biologists! Galaxy includes everything you need to get started, so download and start integrating! -
+ +
+Introducing Galactic Quickies +
+ Galactic quickies are super-short screencasts that are always under 5 minutes. We thought it may be a good way to spread the word about Galaxy's functionality while keeping the "annoyance factor" to the minimum. The quickies will be updated weekly. +
+
+ + + +
+ +
+ + + +
+
+ For a high resolution version click here. +
+
+ + +
+

Galaxy team is a part of BX at Penn State.


This project is supported in part by NSF and the Huck Institutes of the Life Sciences.

Galaxy build: $Rev$

diff --git a/templates/dataset/security_common.mako b/templates/dataset/security_common.mako index 039a9a7ba6d..0643e22f1f3 100644 --- a/templates/dataset/security_common.mako +++ b/templates/dataset/security_common.mako @@ -29,7 +29,8 @@ -<%def name="render_permission_form( obj, obj_name, form_url, id_name, id, all_roles )"> +## Any permission ( e.g., 'DATASET_ACCESS' ) included in the do_not_render param will not be rendered on the page. +<%def name="render_permission_form( obj, obj_name, form_url, id_name, id, all_roles, do_not_render=[] )"> <% if isinstance( obj, trans.app.model.User ): current_actions = obj.default_permissions @@ -73,9 +74,11 @@
%for k, v in trans.app.model.Dataset.permitted_actions.items(): -
- ${render_select( current_actions, k, v, all_roles )} -
+ %if k not in do_not_render: +
+ ${render_select( current_actions, k, v, all_roles )} +
+ %endif %endfor
diff --git a/templates/root/tool_menu.mako b/templates/root/tool_menu.mako index c35d9868519..7fb4017a999 100644 --- a/templates/root/tool_menu.mako +++ b/templates/root/tool_menu.mako @@ -38,11 +38,9 @@ ## Render a label <%def name="render_label( label )"> -
${label.text}
-
@@ -91,7 +89,6 @@ %for key, val in toolbox.tool_panel.items(): %if key.startswith( 'tool' ): ${render_tool( val, False )} -
%elif key.startswith( 'workflow' ): ${render_workflow( key, val, False )} %elif key.startswith( 'section' ): @@ -112,10 +109,10 @@ %endfor
-
%elif key.startswith( 'label' ): ${render_label( val )} %endif +
%endfor ## Link to workflow management. The location of this may change, but eventually @@ -128,18 +125,16 @@ Workflow (beta)
-
-
- Manage workflows -
- %if t.user: - %for m in t.user.stored_workflow_menu_entries: - - %endfor - %endif +
+ Manage workflows
+ %if t.user: + %for m in t.user.stored_workflow_menu_entries: + + %endfor + %endif
diff --git a/tools/data_source/echo.xml b/tools/data_source/echo.xml index 860e6a12656..e3c3a2a84a2 100644 --- a/tools/data_source/echo.xml +++ b/tools/data_source/echo.xml @@ -6,14 +6,18 @@ echoes parameters - echo.py $input $output + echo.py $input $database $output + + + + - + diff --git a/tools/fasta_tools/fasta_compute_length.py b/tools/fasta_tools/fasta_compute_length.py index 3893072844b..6086ac2716a 100644 --- a/tools/fasta_tools/fasta_compute_length.py +++ b/tools/fasta_tools/fasta_compute_length.py @@ -1,8 +1,8 @@ #! /usr/bin/python """ -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. +Input: fasta, int +Output: tabular +Return titles with lengths of corresponding seq """ import sys, os @@ -10,41 +10,37 @@ import sys, os assert sys.version_info[:2] >= ( 2, 4 ) def __main__(): - input_filename = sys.argv[1] - output_filename = sys.argv[2] + + infile = sys.argv[1] + outfile = sys.argv[2] keep_first = int( sys.argv[3] ) - tmp_title = tmp_seq = '' - tmp_seq_count = 0 - seq_hash = {} + + fasta_title = fasta_seq = '' + # number of char to keep in the title if keep_first == 0: keep_first = None else: keep_first += 1 - for i, line in enumerate( file( input_filename ) ): + out = open(outfile, 'w') + + for i, line in enumerate( file( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue if line[0] == '>': - if len( tmp_seq ) > 0: - tmp_seq_count += 1 - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - tmp_title = line - tmp_seq = '' + if len( fasta_seq ) > 0 : + out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) ) + fasta_title = line + fasta_seq = '' else: - tmp_seq = "%s%s" % ( tmp_seq, line ) - if line.split() and line.split()[0].isdigit(): - tmp_seq = "%s " % tmp_seq - if len( tmp_seq ) > 0: - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - - title_keys = seq_hash.keys() - title_keys.sort() - output_handle = open( output_filename, 'w' ) - for i, fasta_title in title_keys: - tmp_seq = seq_hash[ ( i, fasta_title ) ] - output_handle.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( tmp_seq ) ) ) - output_handle.close() + fasta_seq = "%s%s" % ( fasta_seq, line ) + + # check the last sequence + if len( fasta_seq ) > 0: + out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) ) + + out.close() if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_filter_by_length.py b/tools/fasta_tools/fasta_filter_by_length.py index 4947e7b333f..9906300e7ef 100644 --- a/tools/fasta_tools/fasta_filter_by_length.py +++ b/tools/fasta_tools/fasta_filter_by_length.py @@ -15,7 +15,7 @@ def stop_err( msg ): def __main__(): - input_filename = sys.argv[1] + infile = sys.argv[1] try: min_length = int( sys.argv[2] ) except: @@ -24,49 +24,62 @@ def __main__(): max_length = int( sys.argv[3] ) except: stop_err( "Maximum length of the return sequence requires a numerical value." ) - output_filename = sys.argv[4] - tmp_title = tmp_seq = '' - tmp_seq_count = 0 - seq_hash = {} + outfile = sys.argv[4] + fasta_title = fasta_seq = '' + at_least_one = 0 + + out = open( outfile, 'w' ) - for i, line in enumerate( file( input_filename ) ): + for i, line in enumerate( file( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue + if line[0] == '>': - if len( tmp_seq ) > 0: - tmp_seq_count += 1 - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq - tmp_title = line - tmp_seq = '' + if len( fasta_seq ) > 0: + + if max_length <= 0: + compare_max_length = len( fasta_seq ) + 1 + else: + compare_max_length = max_length + + l = len( fasta_seq ) + + if l >= min_length and l <= compare_max_length: + at_least_one += 1 + out.write( "%s\n" % fasta_title ) + c = 0 + s = fasta_seq + while c < l: + b = min( c + 50, l ) + out.write( "%s\n" % s[ c:b ] ) + c = b + + fasta_title = line + fasta_seq = '' else: - tmp_seq = "%s%s" % ( tmp_seq, line ) - if line.split()[0].isdigit(): - tmp_seq = "%s " % tmp_seq - if len( tmp_seq ) > 0: - seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq + fasta_seq = "%s%s" % ( fasta_seq, line ) - title_keys = seq_hash.keys() - title_keys.sort() - output_handle = open( output_filename, 'w' ) - at_least_one = 0 - for i, fasta_title in title_keys: - tmp_seq = seq_hash[ ( i, fasta_title ) ] + if len( fasta_seq ) > 0: + if max_length <= 0: - compare_max_length = len( tmp_seq ) + 1 + compare_max_length = len( fasta_seq ) + 1 else: compare_max_length = max_length - l = len( tmp_seq ) + + l = len( fasta_seq ) + if l >= min_length and l <= compare_max_length: at_least_one += 1 - output_handle.write( "%s\n" % fasta_title ) + out.write( "%s\n" % fasta_title ) c = 0 - s = tmp_seq + s = fasta_seq while c < l: b = min( c + 50, l ) - output_handle.write( "%s\n" % s[ c:b ] ) + out.write( "%s\n" % s[ c:b ] ) c = b - output_handle.close() + + out.close() if at_least_one == 0: print "There is no sequence that falls within your range." diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py index 42e79e63e2a..8a18d0c69c4 100644 --- a/tools/fasta_tools/fasta_to_tabular.py +++ b/tools/fasta_tools/fasta_to_tabular.py @@ -1,9 +1,9 @@ #! /usr/bin/python # This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools """ -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. +Input: fasta, int +Output: tabular +format convert: fasta to tabular """ import sys, os @@ -14,38 +14,31 @@ def __main__(): infile = sys.argv[1] outfile = sys.argv[2] keep_first = int( sys.argv[3] ) - title = '' - sequence = '' - sequence_count = 0 + fasta_title = fasta_seq = '' if keep_first == 0: keep_first = None else: keep_first += 1 + out = open( outfile, 'w' ) + for i, line in enumerate( open( infile ) ): line = line.rstrip( '\r\n' ) if not line or line.startswith( '#' ): continue if line.startswith( '>' ): - if sequence: - sequence_count += 1 - seq_hash[( sequence_count, title )] = sequence - title = line - sequence = '' + if fasta_seq: + out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) ) + fasta_title = line + fasta_seq = '' else: - sequence = "%s%s" % ( sequence, line ) - if line.split() and line.split()[0].isdigit(): - sequence += ' ' - if sequence: - seq_hash[( sequence_count, title )] = sequence - # return only those lengths are in the range - title_keys = seq_hash.keys() - title_keys.sort() - out = open( outfile, 'w' ) - for i, fasta_title in title_keys: - sequence = seq_hash[( i, fasta_title )] - out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], sequence ) ) + if line: + fasta_seq = "%s%s" % ( fasta_seq, line ) + + if fasta_seq: + out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) ) + out.close() if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/regVariation/microsats_mutability.py b/tools/regVariation/microsats_mutability.py index 1261dbbba73..eab8c3c0829 100644 --- a/tools/regVariation/microsats_mutability.py +++ b/tools/regVariation/microsats_mutability.py @@ -128,11 +128,13 @@ def output_writer(blk, blk_lines): uniq_s_elems_2 = get_binned_lists(uniq_s_elems_2,s_bin_size) for pitem1 in uniq_elems_1: - repeats1 = [] - repeats2 = [] + #repeats1 = [] + #repeats2 = [] thresholds = [] if s_group_cols[0] != -1: #Sub-group by feature is not None for sitem1 in uniq_s_elems_1: + repeats1 = [] + repeats2 = [] if type(sitem1) == type(''): sitem1 = sitem1.strip() for bline in blk_lines: @@ -223,11 +225,13 @@ def output_writer(blk, blk_lines): count1[str(pitem1)]=sum(repeats2) for pitem2 in uniq_elems_2: - repeats1 = [] - repeats2 = [] + #repeats1 = [] + #repeats2 = [] thresholds = [] if s_group_cols[0] != -1: #Sub-group by feature is not None for sitem2 in uniq_s_elems_2: + repeats1 = [] + repeats2 = [] if type(sitem2)==type(''): sitem2 = sitem2.strip() for bline in blk_lines: @@ -349,11 +353,10 @@ def output_writer(blk, blk_lines): count = count1[key] mut = "%.2e" %(mut/num_generations) if region == 'align': - print >>fout, str(blk) + '\t'+seq1 + '\t' + start1+ '\t'+end1+ '\t'+seq2 + '\t'+start2+ '\t'+end2+ '\t'+key.strip()+ '\t'+str(mut) + '\t'+ str(count) + print >>fout, str(blk) + '\t'+seq1 + '\t' + seq2 + '\t' +key.strip()+ '\t'+str(mut) + '\t'+ str(count) elif region == 'win': fout.write("%s\t%s\t%s\t%s\n" %(blk,key.strip(),mut,count)) fout.flush() - #print >>fout, blk + '\t'+key.strip()+ '\t'+str(mut)+ '\t'+ str(count) #catch any remaining repeats, for instance if the orthologous position contained different repeat units for remaining_key in mut2.keys(): @@ -361,7 +364,7 @@ def output_writer(blk, blk_lines): mut = "%.2e" %(mut/num_generations) count = count2[remaining_key] if region == 'align': - print >>fout, str(blk) + '\t'+seq1 + '\t' + start1+ '\t'+end1+ '\t'+seq2 + '\t'+start2+ '\t'+end2+ '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count) + print >>fout, str(blk) + '\t'+seq1 + '\t'+seq2 + '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count) elif region == 'win': fout.write("%s\t%s\t%s\t%s\n" %(blk,remaining_key.strip(),mut,count)) fout.flush() @@ -409,19 +412,9 @@ def main(): blk=0 win=0 linestr="" - ff=open("junkmabc","w") if region == 'win': - """ - sorted_infile = tempfile.NamedTemporaryFile() - if winspecies == 1: - cmdline = "sort -n -k 3"+" -o "+sorted_infile.name+" "+infile - elif winspecies == 2: - cmdline = "sort -n -k 10"+" -o "+sorted_infile.name+" "+infile - os.system(cmdline) - print >>ff, "Finished sorting" - """ msats = NiceReaderWrapper( fileinput.FileInput( infile ), chrom_col = speciesind, start_col = speciesind+1, @@ -430,17 +423,9 @@ def main(): fix_strand = True) msatTree = quicksect.IntervalTree() for item in msats: - #print >>sys.stderr, item if type( item ) is GenomicInterval: msatTree.insert( item, msats.linenum, item.fields ) - """ - result = [] - msatTree.traverse(lambda node: result.append( node )) - for n in result: - print >>sys.stderr,n.other - print >>sys.stderr,msatTree.chroms - #sys.exit() - """ + for iline in fint: try: iline = iline.rstrip('\r\n') @@ -470,7 +455,7 @@ def main(): print "Skipped %d intervals as invalid." %(skipped) elif region == 'align': if s_group_cols[0] != -1: - print >>fout, "#Window\tSpecies_1\tWindow_Start\tWindow_End\tSpecies_2\tGroupby_Feature\tSubGroupby_Feature\tMutability\tCount" + print >>fout, "#Window\tSpecies_1\tSpecies_2\tGroupby_Feature\tSubGroupby_Feature\tMutability\tCount" else: print >>fout, "#Window\tSpecies_1\tWindow_Start\tWindow_End\tSpecies_2\tGroupby_Feature\tMutability\tCount" prev_bnum = -1 diff --git a/tools/regVariation/microsats_mutability.xml b/tools/regVariation/microsats_mutability.xml index 141cadd31b0..e5ac10925b2 100644 --- a/tools/regVariation/microsats_mutability.xml +++ b/tools/regVariation/microsats_mutability.xml @@ -106,6 +106,10 @@ This tool computes microsatellite mutability for the orthologous microsatellites fetched from 'Extract Orthologous Microsatellites from pair-wise alignments' tool. +Mutability is computed according to the method described in the following paper: + +*Webster et al., Microsatellite evolution inferred from human-chimpanzee genomic sequence alignments, Proc Natl Acad Sci 2002 June 25; 99(13): 8748-8753* + ----- .. class:: warningmark diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py index eea1a135ed0..bcd3714b9b0 100644 --- a/tools/stats/grouping.py +++ b/tools/stats/grouping.py @@ -12,15 +12,13 @@ def stop_err(msg): def main(): inputfile = sys.argv[2] - in_columns = int( sys.argv[5] ) - show_remaining_cols = sys.argv[4] ops = [] cols = [] rounds = [] elems = [] - for var in sys.argv[6:]: + for var in sys.argv[4:]: ops.append(var.split()[0]) cols.append(var.split()[1]) rounds.append(var.split()[2]) @@ -81,19 +79,9 @@ def main(): if error_code != 0: stop_err( "Sorting input dataset resulted in error: %s: %s" %( error_code, stdout )) - - if show_remaining_cols == 'yes': - show_cols_list = [1]*in_columns - show_cols_list[group_col] = 0 - for c in cols: - c = int(c)-1 - show_cols_list[c] = 0 - #at the end of this, only the indices of the remaining columns will be set to 1 - remaining_cols = [j for j,k in enumerate(show_cols_list) if k==1] #this is the list of remaining column indices - + prev_item = "" prev_vals = [] - remaining_vals = [] skipped_lines = 0 first_invalid_line = 0 invalid_line = '' @@ -128,30 +116,22 @@ def main(): invalid_column = col+1 if valid: prev_vals[i].append(fields[col].strip()) - #Store values from all the remaning columns - if show_remaining_cols == 'yes': - for j, index in enumerate(remaining_cols): - remaining_vals[j].append(fields[index].strip()) else: """ When a new value is encountered, write the previous value and the corresponding aggregate values into the output file. This works due to the sort on group_col we've applied to the data above. """ - out_list = ['']*in_columns - out_list[group_col] = str(prev_item) - + out_str = prev_item + for i, op in enumerate( ops ): rfunc = "r." + op if op not in ['c','length','unique','random']: for j, elem in enumerate( prev_vals[i] ): prev_vals[i][j] = float( elem ) + rout = "%g" %( eval( rfunc )( prev_vals[i] )) if rounds[i] == 'yes': - rout = "%f" %( eval( rfunc )( prev_vals[i] )) rout = int(round(float(rout))) - else: - rout = "%g" %( eval( rfunc )( prev_vals[i] )) - else: if op != 'random': rout = eval( rfunc )( prev_vals[i] ) @@ -162,21 +142,9 @@ def main(): if op == 'unique': rfunc = "r.length" rout = eval( rfunc )( rout ) - - out_list[int(cols[i])-1] = str(rout) - - if show_remaining_cols == 'yes': - for index,el in enumerate(remaining_cols): - if index == 0: - try: - random_index = random.randint(0,len(remaining_vals[index])-1) - except: - random_index = 0 - #pick a random value from each of the remaning columns - rand_out = remaining_vals[index][random_index] - out_list[el] = str(rand_out) - - print >>fout, '\t'.join([elem for elem in out_list if elem != '']) + out_str += "\t" + str(rout) + + print >>fout, out_str prev_item = item prev_vals = [] @@ -185,14 +153,6 @@ def main(): val_list = [] val_list.append(fields[col].strip()) prev_vals.append(val_list) - - if show_remaining_cols == 'yes': - remaining_vals = [] - for index in remaining_cols: - remaining_val_list = [] - remaining_val_list.append(fields[index].strip()) - remaining_vals.append(remaining_val_list) - else: # This only occurs once, right at the start of the iteration. prev_item = item @@ -201,15 +161,8 @@ def main(): val_list = [] val_list.append(fields[col].strip()) prev_vals.append(val_list) - - if show_remaining_cols == 'yes': - remaining_vals = [] - for index in remaining_cols: - remaining_val_list = [] - remaining_val_list.append(fields[index].strip()) - remaining_vals.append(remaining_val_list) - except Exception: + except Exception, exc: skipped_lines += 1 if not first_invalid_line: first_invalid_line = ii+1 @@ -219,8 +172,7 @@ def main(): first_invalid_line = ii+1 # Handle the last grouped value - out_list = ['']*in_columns - out_list[group_col] = str(prev_item) + out_str = prev_item for i, op in enumerate(ops): rfunc = "r." + op @@ -228,11 +180,9 @@ def main(): if op not in ['c','length','unique','random']: for j, elem in enumerate( prev_vals[i] ): prev_vals[i][j] = float( elem ) + rout = '%g' %( eval( rfunc )( prev_vals[i] )) if rounds[i] == 'yes': - rout = '%f' %( eval( rfunc )( prev_vals[i] )) rout = int(round(float(rout))) - else: - rout = '%g' %( eval( rfunc )( prev_vals[i] )) else: if op != 'random': rout = eval( rfunc )( prev_vals[i] ) @@ -243,22 +193,13 @@ def main(): if op == 'unique': rfunc = "r.length" rout = eval( rfunc )( rout ) - out_list[int(cols[i])-1] = str(rout) + out_str += "\t" + str( rout ) except: skipped_lines += 1 if not first_invalid_line: first_invalid_line = ii+1 - if show_remaining_cols == 'yes': - for index,el in enumerate(remaining_cols): - if index == 0: - try: - random_index = random.randint(0,len(remaining_vals[index])-1) - except: - random_index = 0 - rand_out = remaining_vals[index][random_index] - out_list[el] = str(rand_out) - print >>fout, '\t'.join([elem for elem in out_list if elem != '']) + print >>fout, out_str # Generate a useful info message. msg = "--Group by c%d: " %(group_col+1) diff --git a/tools/stats/grouping.xml b/tools/stats/grouping.xml index fa00abd1dd3..8749b5dc922 100644 --- a/tools/stats/grouping.xml +++ b/tools/stats/grouping.xml @@ -1,49 +1,43 @@ - - data by a column and perform aggregate operation on other columns. - - grouping.py - $out_file1 - $input1 + + data by a column and perform aggregate operation on other columns. + + grouping.py + $out_file1 + $input1 $groupcol - $othercols - ${input1.metadata.columns} - #for $op in $operations - '${op.optype} - ${op.opcol} - ${op.opround}' - #end for - - - - - - - - - - - - + #for $op in $operations + '${op.optype} + ${op.opcol} + ${op.opround}' + #end for + + + + + + + + + + + + - - - - - - - - - + + + + - - - - - - - rpy - + + + + + + + + rpy + @@ -52,9 +46,9 @@ - + @@ -62,39 +56,38 @@ - - - - - -.. class:: infomark - -**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert* - ------ - -**Syntax** - -This tool allows you to group the input dataset by a particular column and perform aggregate functions like Mean, Sum, Max, Min and Concatenate on other columns. - -- All invalid, blank and comment lines are skipped when performing the aggregate functions. The number of skipped lines is displayed in the resulting history item. - ------ - -**Example** - -- For the following input:: - - chr22 1000 NM_17 - chr22 2000 NM_18 - chr10 2200 NM_10 - chr10 1200 NM_11 - chr22 1600 NM_19 - -- running this tool with **Group by column 1**, Operations **Mean on column 2** and **Concatenate on column 3** will return:: - - chr10 1700.00 ['NM_11', 'NM_10'] - chr22 1533.33 ['NM_17', 'NM_19', 'NM_18'] - - + + + + +.. class:: infomark + +**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert* + +----- + +**Syntax** + +This tool allows you to group the input dataset by a particular column and perform aggregate functions like Mean, Sum, Max, Min and Concatenate on other columns. + +- All invalid, blank and comment lines are skipped when performing the aggregate functions. The number of skipped lines is displayed in the resulting history item. + +----- + +**Example** + +- For the following input:: + + chr22 1000 NM_17 + chr22 2000 NM_18 + chr10 2200 NM_10 + chr10 1200 NM_11 + chr22 1600 NM_19 + +- running this tool with **Group by column 1**, Operations **Mean on column 2** and **Concatenate on column 3** will return:: + + chr10 1700.00 ['NM_11', 'NM_10'] + chr22 1533.33 ['NM_17', 'NM_19', 'NM_18'] + + diff --git a/tools/taxonomy/find_diag_hits.py b/tools/taxonomy/find_diag_hits.py index a481cab65dd..993c066bb2b 100644 --- a/tools/taxonomy/find_diag_hits.py +++ b/tools/taxonomy/find_diag_hits.py @@ -71,7 +71,8 @@ taxRank = { 'genus' :20, 'subgenus' :21, 'species' :22, - 'subspecies' :23 + 'subspecies' :23, + 'order' :13 } @@ -157,12 +158,16 @@ try: for item in cur.fetchall(): out_string = '%s\t%s\t%d\t' % ( item[0], item[1], item[2] ) out_string += rankName + out_string += '\t' + out_string += str(taxRank[rankName]) print >>out_file, out_string else: cur.execute('select rank, count(*) from %s_count where N = 1 and length(rank)>1 group by rank' % rank) for item in cur.fetchall(): out_string = '%s\t%s\t' % ( item[0], item[1] ) out_string += rankName + out_string += '\t' + out_string += str(taxRank[rankName]) print >>out_file, out_string except Exception, e: stop_err("%s\n" % e) diff --git a/tools/taxonomy/lca.py b/tools/taxonomy/lca.py new file mode 100644 index 00000000000..0b9785d9dd5 --- /dev/null +++ b/tools/taxonomy/lca.py @@ -0,0 +1,186 @@ +#!/usr/bin/env python +#Guruprasad Ananda +""" +This tool provides the SQL "group by" functionality. +""" +import sys, string, re, commands, tempfile, random +#from rpy import * + +def stop_err(msg): + sys.stderr.write(msg) + sys.exit() + +def main(): + try: + inputfile = sys.argv[1] + outfile = sys.argv[2] + rank_bound = int( sys.argv[3] ) + """ + Mapping of ranks: + root :2, + superkingdom:3, + kingdom :4, + subkingdom :5, + superphylum :6, + phylum :7, + subphylum :8, + superclass :9, + class :10, + subclass :11, + superorder :12, + order :13, + suborder :14, + superfamily :15, + family :16, + subfamily :17, + tribe :18, + subtribe :19, + genus :20, + subgenus :21, + species :22, + subspecies :23, + """ + except: + stop_err("Syntax error: Use correct syntax: program infile outfile") + group_col = 0 + + tmpfile = tempfile.NamedTemporaryFile() + + try: + """ + The -k option for the Posix sort command is as follows: + -k, --key=POS1[,POS2] + start a key at POS1, end it at POS2 (origin 1) + In other words, column positions start at 1 rather than 0, so + we need to add 1 to group_col. + if POS2 is not specified, the newer versions of sort will consider the entire line for sorting. To prevent this, we set POS2=POS1. + """ + command_line = "sort -f -k " + str(group_col+1) +"," + str(group_col+1) + " -o " + tmpfile.name + " " + inputfile + except Exception, exc: + stop_err( 'Initialization error -> %s' %str(exc) ) + + error_code, stdout = commands.getstatusoutput(command_line) + + if error_code != 0: + stop_err( "Sorting input dataset resulted in error: %s: %s" %( error_code, stdout )) + + prev_item = "" + prev_vals = [] + remaining_vals = [] + skipped_lines = 0 + first_invalid_line = 0 + invalid_line = '' + invalid_value = '' + invalid_column = 0 + fout = open(outfile, "w") + cols = range(1,25) + block_valid = False + + for ii, line in enumerate( file( tmpfile.name )): + if line and not line.startswith( '#' ): + line = line.rstrip( '\r\n' ) + try: + fields = line.split("\t") + item = fields[group_col] + if prev_item != "": + # At this level, we're grouping on values (item and prev_item) in group_col + if item == prev_item: + # Keep iterating and storing values until a new value is encountered. + if block_valid: + for i, col in enumerate(cols): + if col >= 3: + prev_vals[i].append(fields[col].strip()) + if len(set(prev_vals[i])) > 1: + block_valid = False + break + + else: + """ + When a new value is encountered, write the previous value and the + corresponding aggregate values into the output file. This works + due to the sort on group_col we've applied to the data above. + """ + out_list = ['']*25 + out_list[0] = str(prev_item) + out_list[1] = str(prev_vals[0][0]) + out_list[2] = str(prev_vals[1][0]) + out_list[24] = str(prev_vals[23][0]) + #print >> fout, prev_vals + #sys.exit() + for k, col in enumerate(cols): + if col >= 3 and col < 24: + if len(set(prev_vals[k])) == 1: + out_list[col] = prev_vals[k][0] + else: + break + while k < 23: + out_list[k+1] = 'n' + k += 1 + + # print >>fout, '\t'.join(out_list) + + if rank_bound == 0: + print >>fout, ''.join(out_list) + print 'n'*( 24 - rank_bound ) + else: + print '\t'.join(out_list[rank_bound:24]) + if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ): + print >>fout, '\t'.join(out_list) + + + block_valid = True + prev_item = item + prev_vals = [] + for col in cols: + val_list = [] + val_list.append(fields[col].strip()) + prev_vals.append(val_list) + + else: + # This only occurs once, right at the start of the iteration. + block_valid = True + prev_item = item #groupby item + for col in cols: #everyting else + val_list = [] + val_list.append(fields[col].strip()) + prev_vals.append(val_list) + + except Exception, exc: + skipped_lines += 1 + if not first_invalid_line: + first_invalid_line = ii+1 + else: + skipped_lines += 1 + if not first_invalid_line: + first_invalid_line = ii+1 + + # Handle the last grouped value + out_list = ['']*25 + out_list[0] = str(prev_item) + out_list[1] = str(prev_vals[0][0]) + out_list[2] = str(prev_vals[1][0]) + out_list[24] = str(prev_vals[23][0]) + for k, col in enumerate(cols): + if col >= 3 and col < 24: + if len(set(prev_vals[k])) == 1: + out_list[col] = prev_vals[k][0] + else: + break + while k < 23: + out_list[k+1] = 'n' + k += 1 + + if rank_bound == 0: + print >>fout, '\t'.join(out_list) + else: + print ''.join(out_list[rank_bound:24]) + print 'n'*( 24 - rank_bound ) + if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ): + print >>fout, '\t'.join(out_list) + + if skipped_lines > 0: + msg= "Skipped %d invalid lines starting with line %d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column ) + print msg + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/tools/taxonomy/lca.xml b/tools/taxonomy/lca.xml new file mode 100644 index 00000000000..f608a16cfc3 --- /dev/null +++ b/tools/taxonomy/lca.xml @@ -0,0 +1,44 @@ + + + + lca.py $input1 $out_file1 $rank_bound + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +**What it does** + +When performing metagenomic analyses it is often necessary to identify sequence reads corresponding to a particular taxonomic group, or, in other words, diagnostic of a particular taxonomic rank. This utility performs this analysis. It takes data generated by *Taxonomy manipulation->Fetch Taxonomic Ranks* as input and outputs either a list of sequence reads unique to a particular taxonomic rank, or a list of taxonomic ranks and the count of unique reads corresponding to each rank. + + + \ No newline at end of file