mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merged in changes from remote head
This commit is contained in:
+19
-14
@@ -38,6 +38,7 @@
|
||||
<datatype extension="png" type="galaxy.datatypes.images:Image" mimetype="image/png"/>
|
||||
<datatype extension="qual" type="galaxy.datatypes.qualityscore:QualityScore" display_in_upload="true"/>
|
||||
<datatype extension="scf" type="galaxy.datatypes.images:Scf" mimetype="application/octet-stream" display_in_upload="true"/>
|
||||
<datatype extension="solidqual" type="galaxy.datatypes.qualityscore:SolidQualityScore" display_in_upload="true"/>
|
||||
<datatype extension="taxonomy" type="galaxy.datatypes.tabular:Taxonomy" display_in_upload="true"/>
|
||||
<datatype extension="tabular" type="galaxy.datatypes.tabular:Tabular" display_in_upload="true"/>
|
||||
<datatype extension="txt" type="galaxy.datatypes.data:Text" display_in_upload="true"/>
|
||||
@@ -172,20 +173,24 @@
|
||||
<!--
|
||||
The order in which Galaxy attempts to determine data types is
|
||||
important because some formats are much more loosely defined
|
||||
than others.
|
||||
than others. The following list should be the most rigidly
|
||||
defined format first, followed by next-most rigidly defined,
|
||||
and so on.
|
||||
-->
|
||||
<sniffer order="005" type="galaxy.datatypes.xml:BlastXml"/>
|
||||
<sniffer order="010" type="galaxy.datatypes.sequence:Maf"/>
|
||||
<sniffer order="015" type="galaxy.datatypes.sequence:Lav"/>
|
||||
<sniffer order="020" type="galaxy.datatypes.sequence:Fasta"/>
|
||||
<sniffer order="025" type="galaxy.datatypes.sequence:FastqSolexa"/>
|
||||
<sniffer order="030" type="galaxy.datatypes.interval:Wiggle"/>
|
||||
<sniffer order="035" type="galaxy.datatypes.images:Html"/>
|
||||
<sniffer order="040" type="galaxy.datatypes.sequence:Axt"/>
|
||||
<sniffer order="045" type="galaxy.datatypes.interval:Bed"/>
|
||||
<sniffer order="050" type="galaxy.datatypes.interval:CustomTrack"/>
|
||||
<sniffer order="055" type="galaxy.datatypes.interval:Gff"/>
|
||||
<sniffer order="060" type="galaxy.datatypes.interval:Gff3"/>
|
||||
<sniffer order="065" type="galaxy.datatypes.interval:Interval"/>
|
||||
<sniffer type="galaxy.datatypes.xml:BlastXml"/>
|
||||
<sniffer type="galaxy.datatypes.sequence:Maf"/>
|
||||
<sniffer type="galaxy.datatypes.sequence:Lav"/>
|
||||
<sniffer type="galaxy.datatypes.sequence:csFasta"/>
|
||||
<sniffer type="galaxy.datatypes.qualityscore:SolidQualityScore"/>
|
||||
<sniffer type="galaxy.datatypes.sequence:Fasta"/>
|
||||
<sniffer type="galaxy.datatypes.sequence:FastqSolexa"/>
|
||||
<sniffer type="galaxy.datatypes.interval:Wiggle"/>
|
||||
<sniffer type="galaxy.datatypes.images:Html"/>
|
||||
<sniffer type="galaxy.datatypes.sequence:Axt"/>
|
||||
<sniffer type="galaxy.datatypes.interval:Bed"/>
|
||||
<sniffer type="galaxy.datatypes.interval:CustomTrack"/>
|
||||
<sniffer type="galaxy.datatypes.interval:Gff"/>
|
||||
<sniffer type="galaxy.datatypes.interval:Gff3"/>
|
||||
<sniffer type="galaxy.datatypes.interval:Interval"/>
|
||||
</sniffers>
|
||||
</datatypes>
|
||||
|
||||
@@ -106,8 +106,12 @@ class Data( object ):
|
||||
return False
|
||||
def set_peek( self, dataset ):
|
||||
"""Set the peek and blurb text"""
|
||||
dataset.peek = ''
|
||||
dataset.blurb = 'data'
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = ''
|
||||
dataset.blurb = 'data'
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def display_peek(self, dataset):
|
||||
"""Create HTML table, used for displaying peek"""
|
||||
out = ['<table cellspacing="0" cellpadding="3">']
|
||||
@@ -135,7 +139,8 @@ class Data( object ):
|
||||
def display_info(self, dataset):
|
||||
"""Returns formated html of dataset info"""
|
||||
try:
|
||||
return escape(dataset.info)
|
||||
# Change new line chars to html
|
||||
return escape( dataset.info ).replace( "\r", "\n" ).replace( "\n", "<br>" )
|
||||
except:
|
||||
return "info unavailable"
|
||||
def validate(self, dataset):
|
||||
@@ -282,19 +287,27 @@ class Text( Data ):
|
||||
return 'text/plain'
|
||||
|
||||
def set_peek( self, dataset, line_count=None ):
|
||||
dataset.peek = get_file_peek( dataset.file_name )
|
||||
if line_count is None:
|
||||
dataset.blurb = "%s lines" % util.commaify( str( get_line_count( dataset.file_name ) ) )
|
||||
if not dataset.dataset.purged:
|
||||
# The file must exist on disk for the get_file_peek() method
|
||||
dataset.peek = get_file_peek( dataset.file_name )
|
||||
if line_count is None:
|
||||
dataset.blurb = "%s lines" % util.commaify( str( get_line_count( dataset.file_name ) ) )
|
||||
else:
|
||||
dataset.blurb = "%s lines" % util.commaify( str( line_count ) )
|
||||
else:
|
||||
dataset.blurb = "%s lines" % util.commaify( str( line_count ) )
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
class Binary( Data ):
|
||||
"""Binary data"""
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
"""Set the peek and blurb text"""
|
||||
dataset.peek = 'binary data'
|
||||
dataset.blurb = 'data'
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = 'binary data'
|
||||
dataset.blurb = 'data'
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def get_test_fname( fname ):
|
||||
"""Returns test data filename"""
|
||||
|
||||
@@ -43,11 +43,14 @@ class GenomeGraphs( Tabular ):
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
"""Set the peek and blurb text"""
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
## dataset.peek = self.make_html_table( dataset.peek )
|
||||
dataset.blurb = util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + " rows"
|
||||
#i don't think set_meta should not be called here, it should be called separately
|
||||
self.set_meta( dataset )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = util.commaify( str( data.get_line_count( dataset.file_name ) ) ) + " rows"
|
||||
#i don't think set_meta should not be called here, it should be called separately
|
||||
self.set_meta( dataset )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def get_estimated_display_viewport( self, dataset ):
|
||||
"""Return a chrom, start, stop tuple for viewing a file."""
|
||||
@@ -130,8 +133,12 @@ class SNPMatrix(Rgenetics):
|
||||
file_ext="snpmatrix"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = "Binary RGenetics file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = "Binary RGenetics file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
#def sniff( self, filename ):
|
||||
# """
|
||||
# """
|
||||
|
||||
@@ -14,9 +14,13 @@ class Ab1( data.Data ):
|
||||
"""Class describing an ab1 binary sequence file"""
|
||||
file_ext = "ab1"
|
||||
def set_peek( self, dataset ):
|
||||
export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'ab1','name':'ab1 sequence','info':'Sequence file','dbkey':dataset.dbkey})
|
||||
dataset.peek = "Binary ab1 sequence file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
export_url = "/history_add_to?" + urlencode({'history_id':dataset.history_id,'ext':'ab1','name':'ab1 sequence','info':'Sequence file','dbkey':dataset.dbkey})
|
||||
dataset.peek = "Binary ab1 sequence file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
@@ -27,9 +31,13 @@ class Scf( data.Data ):
|
||||
"""Class describing an scf binary sequence file"""
|
||||
file_ext = "scf"
|
||||
def set_peek( self, dataset ):
|
||||
export_url = "/history_add_to?"+urlencode({'history_id':dataset.history_id,'ext':'scf','name':'scf sequence','info':'Sequence file','dbkey':dataset.dbkey})
|
||||
dataset.peek = "Binary scf sequence file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
export_url = "/history_add_to?" + urlencode({'history_id':dataset.history_id,'ext':'scf','name':'scf sequence','info':'Sequence file','dbkey':dataset.dbkey})
|
||||
dataset.peek = "Binary scf sequence file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
@@ -40,10 +48,14 @@ class Binseq( data.Data ):
|
||||
"""Class describing a zip archive of binary sequence files"""
|
||||
file_ext = "binseq.zip"
|
||||
def set_peek( self, dataset ):
|
||||
zip_file = zipfile.ZipFile( dataset.file_name, "r" )
|
||||
num_files = len( zip_file.namelist() )
|
||||
dataset.peek = "Archive of %s binary sequence files" % ( str( num_files ) )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
zip_file = zipfile.ZipFile( dataset.file_name, "r" )
|
||||
num_files = len( zip_file.namelist() )
|
||||
dataset.peek = "Archive of %s binary sequence files" % ( str( num_files ) )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
@@ -57,10 +69,14 @@ class Txtseq( data.Data ):
|
||||
"""Class describing a zip archive of text sequence files"""
|
||||
file_ext = "txtseq.zip"
|
||||
def set_peek( self, dataset ):
|
||||
zip_file = zipfile.ZipFile( dataset.file_name, "r" )
|
||||
num_files = len( zip_file.namelist() )
|
||||
dataset.peek = "Archive of %s text sequence files" % ( str( num_files ) )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
zip_file = zipfile.ZipFile( dataset.file_name, "r" )
|
||||
num_files = len( zip_file.namelist() )
|
||||
dataset.peek = "Archive of %s text sequence files" % ( str( num_files ) )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
@@ -73,8 +89,12 @@ class Txtseq( data.Data ):
|
||||
class Image( data.Data ):
|
||||
"""Class describing an image"""
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = 'Image in %s format' % dataset.extension
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = 'Image in %s format' % dataset.extension
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def create_applet_tag_peek( class_name, archive, params ):
|
||||
text = """
|
||||
@@ -105,21 +125,26 @@ class Gmaj( data.Data ):
|
||||
file_ext = "gmaj.zip"
|
||||
copy_safe_peek = False
|
||||
def set_peek( self, dataset ):
|
||||
if hasattr( dataset, 'history_id' ):
|
||||
params = {
|
||||
"bundle":"display?id=%s&tofile=yes&toext=.zip" % dataset.id,
|
||||
"buttonlabel": "Launch GMAJ",
|
||||
"nobutton": "false",
|
||||
"urlpause" :"100",
|
||||
"debug": "false",
|
||||
"posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'maf', 'name': 'GMAJ Output on data %s' % dataset.hid, 'info': 'Added by GMAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) )
|
||||
}
|
||||
class_name = "edu.psu.bx.gmaj.MajApplet.class"
|
||||
archive = "/static/gmaj/gmaj.jar"
|
||||
dataset.peek = create_applet_tag_peek( class_name, archive, params )
|
||||
if not dataset.dataset.purged:
|
||||
if hasattr( dataset, 'history_id' ):
|
||||
params = {
|
||||
"bundle":"display?id=%s&tofile=yes&toext=.zip" % dataset.id,
|
||||
"buttonlabel": "Launch GMAJ",
|
||||
"nobutton": "false",
|
||||
"urlpause" :"100",
|
||||
"debug": "false",
|
||||
"posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'maf', 'name': 'GMAJ Output on data %s' % dataset.hid, 'info': 'Added by GMAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) )
|
||||
}
|
||||
class_name = "edu.psu.bx.gmaj.MajApplet.class"
|
||||
archive = "/static/gmaj/gmaj.jar"
|
||||
dataset.peek = create_applet_tag_peek( class_name, archive, params )
|
||||
dataset.blurb = 'GMAJ Multiple Alignment Viewer'
|
||||
else:
|
||||
dataset.peek = "After you add this item to your history, you will be able to launch the GMAJ applet."
|
||||
dataset.blurb = 'GMAJ Multiple Alignment Viewer'
|
||||
else:
|
||||
dataset.peek = "After you add this item to your history, you will be able to launch the GMAJ applet."
|
||||
dataset.blurb = 'GMAJ Multiple Alignment Viewer'
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
@@ -151,8 +176,12 @@ class Html( data.Text ):
|
||||
"""Class describing an html file"""
|
||||
file_ext = "html"
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = "HTML file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = "HTML file"
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def get_mime(self):
|
||||
"""Returns the mime type of the datatype"""
|
||||
return 'text/html'
|
||||
@@ -181,20 +210,24 @@ class Laj( data.Text ):
|
||||
file_ext = "laj"
|
||||
copy_safe_peek = False
|
||||
def set_peek( self, dataset ):
|
||||
if hasattr( dataset, 'history_id' ):
|
||||
params = {
|
||||
"alignfile1": "display?id=%s" % dataset.id,
|
||||
"buttonlabel": "Launch LAJ",
|
||||
"title": "LAJ in Galaxy",
|
||||
"posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'lav', 'name': 'LAJ Output', 'info': 'Added by LAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) ),
|
||||
"noseq": "true"
|
||||
}
|
||||
class_name = "edu.psu.cse.bio.laj.LajApplet.class"
|
||||
archive = "/static/laj/laj.jar"
|
||||
dataset.peek = create_applet_tag_peek( class_name, archive, params )
|
||||
if not dataset.dataset.purged:
|
||||
if hasattr( dataset, 'history_id' ):
|
||||
params = {
|
||||
"alignfile1": "display?id=%s" % dataset.id,
|
||||
"buttonlabel": "Launch LAJ",
|
||||
"title": "LAJ in Galaxy",
|
||||
"posturl": quote_plus( "history_add_to?%s" % "&".join( [ "%s=%s" % ( key, value ) for key, value in { 'history_id': dataset.history_id, 'ext': 'lav', 'name': 'LAJ Output', 'info': 'Added by LAJ', 'dbkey': dataset.dbkey, 'copy_access_from': dataset.id }.items() ] ) ),
|
||||
"noseq": "true"
|
||||
}
|
||||
class_name = "edu.psu.cse.bio.laj.LajApplet.class"
|
||||
archive = "/static/laj/laj.jar"
|
||||
dataset.peek = create_applet_tag_peek( class_name, archive, params )
|
||||
else:
|
||||
dataset.peek = "After you add this item to your history, you will be able to launch the LAJ applet."
|
||||
dataset.blurb = 'LAJ Multiple Alignment Viewer'
|
||||
else:
|
||||
dataset.peek = "After you add this item to your history, you will be able to launch the LAJ applet."
|
||||
dataset.blurb = 'LAJ Multiple Alignment Viewer'
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
return dataset.peek
|
||||
|
||||
@@ -59,11 +59,15 @@ class Interval( Tabular ):
|
||||
|
||||
def set_peek( self, dataset, line_count=None ):
|
||||
"""Set the peek and blurb text"""
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
if line_count is None:
|
||||
dataset.blurb = "%s regions" % util.commaify( str( data.get_line_count( dataset.file_name ) ) )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
if line_count is None:
|
||||
dataset.blurb = "%s regions" % util.commaify( str( data.get_line_count( dataset.file_name ) ) )
|
||||
else:
|
||||
dataset.blurb = "%s regions" % util.commaify( str( line_count ) )
|
||||
else:
|
||||
dataset.blurb = "%s regions" % util.commaify( str( line_count ) )
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def set_meta( self, dataset, overwrite = True, first_line_is_header = False, **kwd ):
|
||||
Tabular.set_meta( self, dataset, overwrite = overwrite, skip = 0 )
|
||||
|
||||
@@ -16,11 +16,15 @@ class QualityScore ( data.Text ):
|
||||
file_ext = "qual"
|
||||
|
||||
def set_peek( self, dataset, line_count=None ):
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
if line_count is None:
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
if line_count is None:
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.blurb = "%s lines, Quality score file" % util.commaify( str( line_count ) )
|
||||
else:
|
||||
dataset.blurb = "%s lines, Quality score file" % util.commaify( str( line_count ) )
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def display_peek(self, dataset):
|
||||
try:
|
||||
@@ -28,5 +32,43 @@ class QualityScore ( data.Text ):
|
||||
except:
|
||||
return "Quality score file (%s)" % ( data.nice_size( dataset.get_size() ) )
|
||||
|
||||
class SolidQualityScore( data.Text ):
|
||||
"""
|
||||
Quality scores generated by ABI SOLiD
|
||||
"""
|
||||
file_ext = "solidqual"
|
||||
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
>>> fname = get_test_fname( 'sequence.fasta' )
|
||||
>>> SolidQualityScore().sniff( fname )
|
||||
False
|
||||
>>> fname = get_test_fname( 'sequence.solidqual' )
|
||||
>>> SolidQualityScore().sniff( fname )
|
||||
True
|
||||
"""
|
||||
try:
|
||||
fh = open( filename )
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break #EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith( '#' ): #first non-empty non-comment line
|
||||
if line.startswith( '>' ):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith( '>' ):
|
||||
break
|
||||
try:
|
||||
[ int( x ) for x in line.split() ]
|
||||
except:
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break #we found a non-empty line, but it's not a header
|
||||
except:
|
||||
pass
|
||||
return False
|
||||
|
||||
@@ -59,18 +59,15 @@ class Registry( object ):
|
||||
self.converters.append( ( converter_config, extension, target_datatype ) )
|
||||
except Exception, e:
|
||||
self.log.warning( 'Error loading datatype "%s", problem: %s' % ( extension, str( e ) ) )
|
||||
# Load datatype sniffers from config
|
||||
# Load datatype sniffers from the config
|
||||
sniff_order = []
|
||||
sniffers = root.find( 'sniffers' )
|
||||
for elem in sniffers.findall( 'sniffer' ):
|
||||
order = elem.get( 'order', None )
|
||||
type = elem.get( 'type', None )
|
||||
if order and type:
|
||||
sniff_order.append( ( order, type ) )
|
||||
sniff_order.sort()
|
||||
for ele in sniff_order:
|
||||
if type:
|
||||
sniff_order.append( type )
|
||||
for type in sniff_order:
|
||||
try:
|
||||
type = ele[1]
|
||||
fields = type.split( ":" )
|
||||
datatype_module = fields[0]
|
||||
datatype_class = fields[1]
|
||||
@@ -149,6 +146,7 @@ class Registry( object ):
|
||||
sequence.Maf(),
|
||||
sequence.Lav(),
|
||||
sequence.Fasta(),
|
||||
sequence.FastqSolexa(),
|
||||
interval.Wiggle(),
|
||||
images.Html(),
|
||||
sequence.Axt(),
|
||||
|
||||
@@ -5,6 +5,7 @@ Image classes
|
||||
import data
|
||||
import logging
|
||||
import re
|
||||
import string
|
||||
from cgi import escape
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes import metadata
|
||||
@@ -31,8 +32,12 @@ class Fasta( Sequence ):
|
||||
file_ext = "fasta"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
@@ -86,30 +91,60 @@ class csFasta( Sequence ):
|
||||
file_ext = "csfasta"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Color-space sequence:
|
||||
>2_15_85_F3
|
||||
T213021013012303002332212012112221222112212222
|
||||
|
||||
TODO:
|
||||
add sniff function
|
||||
"""
|
||||
|
||||
return False
|
||||
|
||||
|
||||
|
||||
>>> fname = get_test_fname( 'sequence.fasta' )
|
||||
>>> csFasta().sniff( fname )
|
||||
False
|
||||
>>> fname = get_test_fname( 'sequence.csfasta' )
|
||||
>>> csFasta().sniff( fname )
|
||||
True
|
||||
"""
|
||||
try:
|
||||
fh = open( filename )
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break #EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith( '#' ): #first non-empty non-comment line
|
||||
if line.startswith( '>' ):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith( '>' ):
|
||||
break
|
||||
elif line[0] not in string.ascii_uppercase:
|
||||
return False
|
||||
elif len( line ) > 1 and not re.search( '^\d+$', line[1:] ):
|
||||
return False
|
||||
return True
|
||||
else:
|
||||
break #we found a non-empty line, but it's not a header
|
||||
except:
|
||||
pass
|
||||
return False
|
||||
|
||||
class FastqSolexa( Sequence ):
|
||||
"""Class representing a FASTQ sequence ( the Solexa variant )"""
|
||||
file_ext = "fastqsolexa"
|
||||
|
||||
def set_peek( self, dataset ):
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = data.nice_size( dataset.get_size() )
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
|
||||
@@ -12,8 +12,12 @@ class BlastXml( data.Text ):
|
||||
file_ext = "blastxml"
|
||||
def set_peek( self, dataset ):
|
||||
"""Set the peek and blurb text"""
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = 'NCBI Blast XML data'
|
||||
if not dataset.dataset.purged:
|
||||
dataset.peek = data.get_file_peek( dataset.file_name )
|
||||
dataset.blurb = 'NCBI Blast XML data'
|
||||
else:
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
def sniff( self, filename ):
|
||||
"""
|
||||
Determines whether the file is blastxml
|
||||
|
||||
+23
-21
@@ -111,19 +111,19 @@ class JobQueue( object ):
|
||||
|
||||
def __check_jobs_at_startup( self ):
|
||||
"""
|
||||
Checks all jobs that are in the 'running' or 'queued' state in the
|
||||
database and requeues or cleans up as necessary. Only run as the
|
||||
Checks all jobs that are in the 'new', 'queued' or 'running' state in
|
||||
the database and requeues or cleans up as necessary. Only run as the
|
||||
job manager starts.
|
||||
"""
|
||||
model = self.app.model
|
||||
# Jobs in the NEW state won't be requeued unless we're tracking in the database
|
||||
if not self.track_jobs_in_database:
|
||||
for job in model.Job.filter( model.Job.c.state==model.Job.states.NEW ).all():
|
||||
log.debug( "no runner: %s is still in new state, adding to the jobs queue" %job.id )
|
||||
self.queue.put( ( job.id, job.tool_id ) )
|
||||
for job in model.Job.filter( model.Job.c.state==model.Job.states.NEW ).all():
|
||||
log.debug( "no runner: %s is still in new state, adding to the jobs queue" %job.id )
|
||||
self.queue.put( ( job.id, job.tool_id ) )
|
||||
for job in model.Job.filter( (model.Job.c.state == model.Job.states.RUNNING) | (model.Job.c.state == model.Job.states.QUEUED) ).all():
|
||||
if job.job_runner_name is not None:
|
||||
# why are we passing the queue to the wrapper?
|
||||
if job.job_runner_name is None:
|
||||
log.debug( "no runner: %s is still in queued state, adding to the jobs queue" %job.id )
|
||||
self.queue.put( ( job.id, job.tool_id ) )
|
||||
else:
|
||||
job_wrapper = JobWrapper( job, self.app.toolbox.tools_by_id[ job.tool_id ], self )
|
||||
self.dispatcher.recover( job, job_wrapper )
|
||||
|
||||
@@ -298,9 +298,13 @@ class JobWrapper( object ):
|
||||
self.queue = queue
|
||||
self.app = queue.app
|
||||
self.extra_filenames = []
|
||||
self.working_directory = None
|
||||
self.command_line = None
|
||||
self.galaxy_lib_dir = None
|
||||
# With job outputs in the working directory, we need the working
|
||||
# directory to be set before prepare is run, or else premature deletion
|
||||
# and job recovery fail.
|
||||
self.working_directory = \
|
||||
os.path.join( self.app.config.job_working_directory, str( self.job_id ) )
|
||||
|
||||
def get_param_dict( self ):
|
||||
"""
|
||||
@@ -317,9 +321,6 @@ class JobWrapper( object ):
|
||||
config files.
|
||||
"""
|
||||
mapping.context.current.clear() #this prevents the metadata reverting that has been seen in conjunction with the PBS job runner
|
||||
# Create the working directory
|
||||
self.working_directory = \
|
||||
os.path.join( self.app.config.job_working_directory, str( self.job_id ) )
|
||||
if not os.path.exists( self.working_directory ):
|
||||
os.mkdir( self.working_directory )
|
||||
# Restore parameters from the database
|
||||
@@ -382,12 +383,12 @@ class JobWrapper( object ):
|
||||
if not job.state == model.Job.states.DELETED:
|
||||
for dataset_assoc in job.output_datasets:
|
||||
if self.app.config.outputs_to_working_directory:
|
||||
false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.id ) )
|
||||
if os.path.exists( false_path ):
|
||||
false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.dataset.id ) )
|
||||
try:
|
||||
shutil.move( false_path, dataset_assoc.dataset.file_name )
|
||||
log.debug( "fail(): Moved %s to %s" % ( false_path, dataset_assoc.dataset.file_name ) )
|
||||
else:
|
||||
log.warning( "fail(): Missing output file in working directory: %s" % false_path )
|
||||
except ( IOError, OSError ), e:
|
||||
log.error( "fail(): Missing output file in working directory: %s" % e )
|
||||
dataset = dataset_assoc.dataset
|
||||
dataset.refresh()
|
||||
dataset.state = dataset.states.ERROR
|
||||
@@ -452,12 +453,13 @@ class JobWrapper( object ):
|
||||
job.state = 'ok'
|
||||
for dataset_assoc in job.output_datasets:
|
||||
if self.app.config.outputs_to_working_directory:
|
||||
false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.id ) )
|
||||
if os.path.exists( false_path ):
|
||||
false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % dataset_assoc.dataset.dataset.id ) )
|
||||
try:
|
||||
shutil.move( false_path, dataset_assoc.dataset.file_name )
|
||||
log.debug( "finish(): Moved %s to %s" % ( false_path, dataset_assoc.dataset.file_name ) )
|
||||
else:
|
||||
log.warning( "finish(): Missing output file in working directory: %s" % false_path )
|
||||
except ( IOError, OSError ):
|
||||
self.fail( "The job's output dataset(s) could not be read" )
|
||||
return
|
||||
for dataset in dataset_assoc.dataset.dataset.history_associations: #need to update all associated output hdas, i.e. history was shared with job running
|
||||
dataset.blurb = 'done'
|
||||
dataset.peek = 'no peek'
|
||||
|
||||
@@ -912,3 +912,4 @@ def directory_hash_id( id ):
|
||||
# Break into chunks of three
|
||||
return [ padded[i*3:(i+1)*3] for i in range( len( padded ) // 3 ) ]
|
||||
|
||||
|
||||
|
||||
@@ -1072,10 +1072,11 @@ class Tool:
|
||||
datatypes_registry = self.app.datatypes_registry,
|
||||
tool = self,
|
||||
name = input.name )
|
||||
elif isinstance( input, SelectToolParameter ):
|
||||
input_values[ input.name ] = SelectToolParameterWrapper( input, input_values[ input.name ], self.app )
|
||||
else:
|
||||
input_values[ input.name ] = \
|
||||
InputValueWrapper( input, input_values[ input.name ], param_dict )
|
||||
# HACK: only wrap if check_values is false, this deals with external
|
||||
input_values[ input.name ] = InputValueWrapper( input, input_values[ input.name ], param_dict )
|
||||
# HACK: only wrap if check_values is not false, this deals with external
|
||||
# tools where the inputs don't even get passed through. These
|
||||
# tools (e.g. UCSC) should really be handled in a special way.
|
||||
if self.check_values:
|
||||
@@ -1098,7 +1099,7 @@ class Tool:
|
||||
for name, data in output_datasets.items():
|
||||
# Write outputs to the working directory (for security purposes) if desired.
|
||||
if self.app.config.outputs_to_working_directory and working_directory is not None:
|
||||
false_path = os.path.abspath( os.path.join( working_directory, "galaxy_dataset_%d.dat" % data.id ) )
|
||||
false_path = os.path.abspath( os.path.join( working_directory, "galaxy_dataset_%d.dat" % data.dataset.id ) )
|
||||
param_dict[name] = DatasetFilenameWrapper( data, false_path = false_path )
|
||||
open( false_path, 'w' ).close()
|
||||
else:
|
||||
@@ -1415,7 +1416,21 @@ class InputValueWrapper( object ):
|
||||
return self.input.to_param_dict_string( self.value, self._other_values )
|
||||
def __getattr__( self, key ):
|
||||
return getattr( self.value, key )
|
||||
|
||||
|
||||
class SelectToolParameterWrapper( object ):
|
||||
"""
|
||||
Wraps a SelectTooParameter so that __str__ returns the selected value, but all other
|
||||
attributes are accessible.
|
||||
"""
|
||||
def __init__( self, input, value, app ):
|
||||
self.input = input
|
||||
self.value = value
|
||||
self.input.value_label = input.value_to_display_text( value, app )
|
||||
def __str__( self ):
|
||||
return self.input.to_param_dict_string( self.value )
|
||||
def __getattr__( self, key ):
|
||||
return getattr( self.input, key )
|
||||
|
||||
class DatasetFilenameWrapper( object ):
|
||||
"""
|
||||
Wraps a dataset so that __str__ returns the filename, but all other
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
from galaxy.util.bunch import Bunch
|
||||
from galaxy.tools.parameters import *
|
||||
from galaxy.tools.parameters.grouping import *
|
||||
from galaxy.util.template import fill_template
|
||||
from galaxy.util.none_like import NoneDataset
|
||||
from galaxy.web import url_for
|
||||
from galaxy.jobs import JOB_OK
|
||||
import galaxy.tools
|
||||
|
||||
import logging
|
||||
log = logging.getLogger( __name__ )
|
||||
@@ -68,6 +70,26 @@ class DefaultToolAction( object ):
|
||||
return input_datasets
|
||||
|
||||
def execute(self, tool, trans, incoming={}, set_output_hid=True ):
|
||||
def wrap_values( inputs, input_values ):
|
||||
# Wrap tool inputs as necessary
|
||||
for input in inputs.itervalues():
|
||||
if isinstance( input, Repeat ):
|
||||
for d in input_values[ input.name ]:
|
||||
wrap_values( input.inputs, d )
|
||||
elif isinstance( input, Conditional ):
|
||||
values = input_values[ input.name ]
|
||||
current = values["__current_case__"]
|
||||
wrap_values( input.cases[current].inputs, values )
|
||||
elif isinstance( input, DataToolParameter ):
|
||||
input_values[ input.name ] = \
|
||||
galaxy.tools.DatasetFilenameWrapper( input_values[ input.name ],
|
||||
datatypes_registry = trans.app.datatypes_registry,
|
||||
tool = tool,
|
||||
name = input.name )
|
||||
elif isinstance( input, SelectToolParameter ):
|
||||
input_values[ input.name ] = galaxy.tools.SelectToolParameterWrapper( input, input_values[ input.name ], tool.app )
|
||||
else:
|
||||
input_values[ input.name ] = galaxy.tools.InputValueWrapper( input, input_values[ input.name ], incoming )
|
||||
out_data = {}
|
||||
# Collect any input datasets from the incoming parameters
|
||||
inp_data = self.collect_input_datasets( tool, incoming, trans )
|
||||
@@ -163,6 +185,11 @@ class DefaultToolAction( object ):
|
||||
# Set output label
|
||||
if output.label:
|
||||
params = dict( incoming )
|
||||
# wrapping the params allows the tool config to contain things like
|
||||
# <outputs>
|
||||
# <data format="input" name="output" label="Blat on ${<input_param>.name}" />
|
||||
# </outputs>
|
||||
wrap_values( tool.inputs, params )
|
||||
params['tool'] = tool
|
||||
params['on_string'] = on_text
|
||||
data.name = fill_template( output.label, context=params )
|
||||
|
||||
@@ -4,7 +4,7 @@ Galaxy web application framework
|
||||
|
||||
import pkg_resources
|
||||
|
||||
import os, sys, time, socket
|
||||
import os, sys, time, socket, random, string
|
||||
pkg_resources.require( "Cheetah" )
|
||||
from Cheetah.Template import Template
|
||||
import base
|
||||
@@ -254,6 +254,9 @@ class UniverseWebTransaction( base.DefaultWebTransaction ):
|
||||
user_for_new_session = self.__get_or_create_remote_user( remote_user_email )
|
||||
log.warning( "User logged in as '%s' externally, but has a cookie as '%s' invalidating session",
|
||||
remote_user_email, prev_galaxy_session.user.email )
|
||||
else:
|
||||
# No session exists, get/create user for new session
|
||||
user_for_new_session = self.__get_or_create_remote_user( remote_user_email )
|
||||
else:
|
||||
if galaxy_session is not None and galaxy_session.user and galaxy_session.user.external:
|
||||
# Remote user support is not enabled, but there is an existing
|
||||
@@ -329,14 +332,16 @@ class UniverseWebTransaction( base.DefaultWebTransaction ):
|
||||
def __get_or_create_remote_user( self, remote_user_email ):
|
||||
"""
|
||||
Return the user in $HTTP_REMOTE_USER and create if necessary
|
||||
Caller is responsible for flushing the returned user.
|
||||
"""
|
||||
# remote_user middleware ensures HTTP_REMOTE_USER exists
|
||||
user = self.app.model.User.filter( self.app.model.User.table.c.email==remote_user_email ).first()
|
||||
if user is None:
|
||||
random.seed()
|
||||
user = self.app.model.User( email=remote_user_email )
|
||||
user.set_password_cleartext( 'external' )
|
||||
user.set_password_cleartext( ''.join( random.sample( string.letters + string.digits, 12 ) ) )
|
||||
user.external = True
|
||||
user.flush()
|
||||
#self.log_event( "Automatically created account '%s'", user.email )
|
||||
elif user.deleted:
|
||||
return self.show_error_message( "Your account is no longer valid, contact your Galaxy administrator to activate your account." )
|
||||
return user
|
||||
|
||||
@@ -31,15 +31,16 @@ div.toolSectionDetailsInner
|
||||
|
||||
div.toolSectionTitle
|
||||
{
|
||||
padding-bottom: 0px;
|
||||
font-weight: bold;
|
||||
}
|
||||
|
||||
div.toolPanelLabel
|
||||
{
|
||||
padding-top: 5px;
|
||||
padding-top: 10px;
|
||||
padding-bottom: 5px;
|
||||
font-weight: bold;
|
||||
color: gray;
|
||||
text-transform: uppercase;
|
||||
}
|
||||
|
||||
div.toolTitle
|
||||
@@ -52,9 +53,17 @@ div.toolTitle
|
||||
list-style: square outside;
|
||||
}
|
||||
|
||||
div.toolSectionBody div.toolPanelLabel
|
||||
{
|
||||
padding-top: 5px;
|
||||
padding-bottom: 5px;
|
||||
margin-left: 16px;
|
||||
margin-right: 10px;
|
||||
display: list-item;
|
||||
list-style: none outside;
|
||||
}
|
||||
|
||||
div.toolTitleNoSection
|
||||
{
|
||||
padding-bottom: 0px;
|
||||
font-weight: bold;
|
||||
}
|
||||
|
||||
@@ -31,15 +31,16 @@ div.toolSectionDetailsInner
|
||||
|
||||
div.toolSectionTitle
|
||||
{
|
||||
padding-bottom: 0px;
|
||||
font-weight: bold;
|
||||
}
|
||||
|
||||
div.toolPanelLabel
|
||||
{
|
||||
padding-top: 5px;
|
||||
padding-top: 10px;
|
||||
padding-bottom: 5px;
|
||||
font-weight: bold;
|
||||
color: gray;
|
||||
text-transform: uppercase;
|
||||
}
|
||||
|
||||
div.toolTitle
|
||||
@@ -52,9 +53,17 @@ div.toolTitle
|
||||
list-style: square outside;
|
||||
}
|
||||
|
||||
div.toolSectionBody div.toolPanelLabel
|
||||
{
|
||||
padding-top: 5px;
|
||||
padding-bottom: 5px;
|
||||
margin-left: 16px;
|
||||
margin-right: 10px;
|
||||
display: list-item;
|
||||
list-style: none outside;
|
||||
}
|
||||
|
||||
div.toolTitleNoSection
|
||||
{
|
||||
padding-bottom: 0px;
|
||||
font-weight: bold;
|
||||
}
|
||||
|
||||
+74
-33
@@ -6,47 +6,88 @@
|
||||
<meta name="generator" content="Docutils 0.3.9: http://docutils.sourceforge.net/" />
|
||||
<title></title>
|
||||
<link rel="stylesheet" href="style/base.css" type="text/css" />
|
||||
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
||||
<meta name="generator" content="Docutils 0.3.9: http://docutils.sourceforge.net/" />
|
||||
<title></title>
|
||||
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
||||
<meta name="generator" content="Docutils 0.6: http://docutils.sourceforge.net/"/>
|
||||
<script src="http://www.apple.com/library/quicktime/scripts/ac_quicktime.js" language="JavaScript" type="text/javascript"></script>
|
||||
<script src="http://www.apple.com/library/quicktime/scripts/qtp_library.js" language="JavaScript" type="text/javascript"></script>
|
||||
<link href="http://www.apple.com/library/quicktime/stylesheets/qtp_library.css" rel="StyleSheet" type="text/css" />
|
||||
<link rel="stylesheet" href="style/base.css" type="text/css" />
|
||||
</head>
|
||||
|
||||
</head>
|
||||
<body>
|
||||
|
||||
<div class="document">
|
||||
<div class="donemessage">
|
||||
<strong>Workflows are finally here!</strong>
|
||||
<hr>
|
||||
Watch how you can (<i>Click link to play</i>)...
|
||||
<ul>
|
||||
<li><a target="_blank" href="http://screencast.g2.bx.psu.edu/galaxy/WorkFlow_SC4/">Create Workflows by Example</a>: Convert your Galaxy History into a Workflow.</li>
|
||||
<li><a target="_blank" href="http://screencast.g2.bx.psu.edu/galaxy/WorkFlow_SC5/">Edit Workflows</a>: I want to repeat an analysis...but...with different parameters.</li>
|
||||
<li><a target="_blank" href="http://screencast.g2.bx.psu.edu/galaxy/WorkFlow_SC7/">Workflows from scratch</a>: Drag, drag, drag...</li>
|
||||
</ul>
|
||||
<hr>
|
||||
For more screencasts click <a target="_blank" href="http://galaxy.psu.edu/screencasts.html">here</a>.
|
||||
</div>
|
||||
<hr>
|
||||
<div class="warningmessage">
|
||||
<strong>The Galaxy Main Server was recently upgraded.</strong>
|
||||
|
||||
<div class="section" align="center">
|
||||
<strong>Unsequenced Genomes of the World</strong> | October 2008
|
||||
<br>
|
||||
<br>
|
||||
<img src="images/welcomePhoto.jpg" border="0">
|
||||
<br>
|
||||
Costa's hummingbird (<i>Calypte costae</i>) | Kings Canyon NP, California
|
||||
<br>
|
||||
<br>
|
||||
<hr>
|
||||
There was some instability as Galaxy was restarted numerous times between 8:30 and 10:00 AM EST (UTC -0500) this morning (Wednesday, January 14). Any jobs from this time that ended in error can be tried again. If you continue to have any problems, please contact the Galaxy Team by using the 'report this error' link within a red history item, or emailing <a href="mailto:galaxy-bugs@bx.psu.edu">galaxy-bugs@bx.psu.edu</a>.
|
||||
</div>
|
||||
|
||||
<p/>
|
||||
<div class="infomessage">
|
||||
<strong>We are hiring!</strong>
|
||||
<hr>
|
||||
Thanks to your support and an unprecedented level of usage, we are looking for an experienced software developer to join our team. For more information about this position, please, click <a target="_blank" href="https://www.bx.psu.edu/cgi-bin/trac.cgi/galaxy/wiki/PythonDeveloper">here</a>.
|
||||
</div>
|
||||
|
||||
<hr>
|
||||
<div class="welcomeBlue">
|
||||
<strong>Galaxy is for Biologists</strong>
|
||||
<br>
|
||||
Use this site to access popular sources of data like the UCSC Table Browser. Run analyses right on the spot using a variety of integrated tools. Your results are always available and can be easily shared with others. Just <a target="_blank" href="http://g2.trac.bx.psu.edu/wiki/ScreenCasts">watch</a> how.
|
||||
</div>
|
||||
<br>
|
||||
<div class="welcomeRed">
|
||||
<strong>Galaxy is for Developers</strong>
|
||||
<br>
|
||||
Galaxy is an easy-to-use, open-source, scalable framework for tool and data integration. Stop wasting time writing interfaces and get your tools used by biologists! Galaxy includes everything you need to get started, so <a target="_blank" href="http://g2.trac.bx.psu.edu/wiki/HowToInstall">download</a> and start <a target="_blank" href="http://g2.trac.bx.psu.edu/wiki/HowToInstall">integrating</a>!
|
||||
</div>
|
||||
|
||||
<div class="welcomeBlue" id="screencasts" align="center">
|
||||
<strong>Introducing Galactic Quickies</strong>
|
||||
<hr>
|
||||
Galactic quickies are <i>super-short</i> screencasts that are <i>always</i> under 5 minutes. We thought it may be a good way to spread the word about Galaxy's functionality while keeping the "annoyance factor" to the minimum. The quickies will be updated weekly.
|
||||
<hr>
|
||||
<table width="100%" border="0" height="100%">
|
||||
<tr>
|
||||
<td valign="middle" align="center">
|
||||
|
||||
<br>
|
||||
<script type="text/javascript"><!--
|
||||
QT_WritePoster_XHTML('click to play', 'http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq-poster.jpg',
|
||||
'http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.mov',
|
||||
'640', '480', '',
|
||||
'controller', 'true',
|
||||
'autoplay', 'true',
|
||||
'bgcolor', '#CCCCFF',
|
||||
'scale', 'aspect');
|
||||
//-->
|
||||
|
||||
</script>
|
||||
<noscript>
|
||||
<object width="640" height="480" classid="clsid:02BF25D5-8C17-4B23-BC80-D3488ABDDC6B" codebase="http://www.apple.com/qtactivex/qtplugin.cab">
|
||||
<param name="src" value="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.jpg" />
|
||||
<param name="href" value="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.mov" />
|
||||
<param name="target" value="myself" />
|
||||
<param name="controller" value="false" />
|
||||
<param name="autoplay" value="false" />
|
||||
<param name="scale" value="aspect" />
|
||||
<embed width="640" height="480" type="video/quicktime" pluginspage="http://www.apple.com/quicktime/download/"
|
||||
src="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.jpg"
|
||||
href="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.mov"
|
||||
target="myself"
|
||||
controller="false"
|
||||
autoplay="false"
|
||||
scale="aspect">
|
||||
</embed>
|
||||
</object>
|
||||
</noscript>
|
||||
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
<br>
|
||||
For a high resolution version <a target="_blank" class="reference" href="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/">click here</a>.
|
||||
<br>
|
||||
<hr>
|
||||
</div>
|
||||
|
||||
<hr>
|
||||
|
||||
<p><a target="_blank" class="reference" href="http://g2.trac.bx.psu.edu/wiki/GalaxyTeam">Galaxy team</a> is a part of <a target="_blank" class="reference" href="http://www.bx.psu.edu">BX</a> at <a target="_blank" class="reference" href="http://www.psu.edu">Penn State</a>.</p>
|
||||
<hr class="docutils" /> This project is supported in part by <a target="_blank" class="reference" href="http://www.nsf.gov">NSF</a> and <a target="_blank" class="reference" href="http://www.huck.psu.edu">the Huck Institutes of the Life Sciences</a>.
|
||||
<p><small>Galaxy build: <b>$Rev$</b></small></p>
|
||||
|
||||
@@ -29,7 +29,8 @@
|
||||
</div>
|
||||
</%def>
|
||||
|
||||
<%def name="render_permission_form( obj, obj_name, form_url, id_name, id, all_roles )">
|
||||
## Any permission ( e.g., 'DATASET_ACCESS' ) included in the do_not_render param will not be rendered on the page.
|
||||
<%def name="render_permission_form( obj, obj_name, form_url, id_name, id, all_roles, do_not_render=[] )">
|
||||
<%
|
||||
if isinstance( obj, trans.app.model.User ):
|
||||
current_actions = obj.default_permissions
|
||||
@@ -73,9 +74,11 @@
|
||||
<input type="hidden" name="${id_name}" value="${id}"/>
|
||||
<div class="form-row"></div>
|
||||
%for k, v in trans.app.model.Dataset.permitted_actions.items():
|
||||
<div class="form-row">
|
||||
${render_select( current_actions, k, v, all_roles )}
|
||||
</div>
|
||||
%if k not in do_not_render:
|
||||
<div class="form-row">
|
||||
${render_select( current_actions, k, v, all_roles )}
|
||||
</div>
|
||||
%endif
|
||||
%endfor
|
||||
<div class="form-row">
|
||||
<input type="submit" name="update_roles" value="Save"/>
|
||||
|
||||
@@ -38,11 +38,9 @@
|
||||
|
||||
## Render a label
|
||||
<%def name="render_label( label )">
|
||||
<div class="toolSectionPad"></div>
|
||||
<div class="toolPanelLabel" id="title_${label.id}">
|
||||
<span>${label.text}</span>
|
||||
</div>
|
||||
<div class="toolSectionPad"></div>
|
||||
</%def>
|
||||
|
||||
<!DOCTYPE HTML PUBLIC "-//W3C//DTD HTML 4.01 Transitional//EN" "http://www.w3.org/TR/html4/loose.dtd">
|
||||
@@ -91,7 +89,6 @@
|
||||
%for key, val in toolbox.tool_panel.items():
|
||||
%if key.startswith( 'tool' ):
|
||||
${render_tool( val, False )}
|
||||
<div class="toolSectionPad"></div>
|
||||
%elif key.startswith( 'workflow' ):
|
||||
${render_workflow( key, val, False )}
|
||||
%elif key.startswith( 'section' ):
|
||||
@@ -112,10 +109,10 @@
|
||||
%endfor
|
||||
</div>
|
||||
</div>
|
||||
<div class="toolSectionPad"></div>
|
||||
%elif key.startswith( 'label' ):
|
||||
${render_label( val )}
|
||||
%endif
|
||||
<div class="toolSectionPad"></div>
|
||||
%endfor
|
||||
|
||||
## Link to workflow management. The location of this may change, but eventually
|
||||
@@ -128,18 +125,16 @@
|
||||
<span>Workflow <i>(beta)</i></span>
|
||||
</div>
|
||||
<div id="XXinternalXXworkflow" class="toolSectionBody">
|
||||
<div class="toolSectionBg">
|
||||
<div class="toolTitle">
|
||||
<a href="${h.url_for( controller='workflow', action='index' )}" target="galaxy_main">Manage</a> workflows
|
||||
</div>
|
||||
%if t.user:
|
||||
%for m in t.user.stored_workflow_menu_entries:
|
||||
<div class="toolTitle">
|
||||
<a href="${h.url_for( controller='workflow', action='run', id=trans.security.encode_id(m.stored_workflow_id) )}" target="galaxy_main">${m.stored_workflow.name}</a>
|
||||
</div>
|
||||
%endfor
|
||||
%endif
|
||||
<div class="toolTitle">
|
||||
<a href="${h.url_for( controller='workflow', action='index' )}" target="galaxy_main">Manage</a> workflows
|
||||
</div>
|
||||
%if t.user:
|
||||
%for m in t.user.stored_workflow_menu_entries:
|
||||
<div class="toolTitle">
|
||||
<a href="${h.url_for( controller='workflow', action='run', id=trans.security.encode_id(m.stored_workflow_id) )}" target="galaxy_main">${m.stored_workflow.name}</a>
|
||||
</div>
|
||||
%endfor
|
||||
%endif
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -6,14 +6,18 @@
|
||||
echoes parameters
|
||||
</description>
|
||||
|
||||
<command interpreter="python">echo.py $input $output </command>
|
||||
<command interpreter="python">echo.py $input $database $output </command>
|
||||
|
||||
<inputs>
|
||||
<param format="tabular" name="input" type="data" label="Input stuff"/>
|
||||
<param type="select" name="database" label="Database">
|
||||
<option value="alignseq.loc">Human (hg18)</option>
|
||||
<option value="faseq.loc">Fly (dm3)</option>
|
||||
</param>
|
||||
</inputs>
|
||||
|
||||
<outputs>
|
||||
<data format="input" name="output" />
|
||||
<data format="input" name="output" label="Blat on ${database.value_label}" />
|
||||
</outputs>
|
||||
|
||||
</tool>
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
#! /usr/bin/python
|
||||
"""
|
||||
Input: fasta, minimal length, maximal length
|
||||
Output: fasta
|
||||
Return sequences whose lengths are within the range.
|
||||
Input: fasta, int
|
||||
Output: tabular
|
||||
Return titles with lengths of corresponding seq
|
||||
"""
|
||||
|
||||
import sys, os
|
||||
@@ -10,41 +10,37 @@ import sys, os
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
|
||||
def __main__():
|
||||
input_filename = sys.argv[1]
|
||||
output_filename = sys.argv[2]
|
||||
|
||||
infile = sys.argv[1]
|
||||
outfile = sys.argv[2]
|
||||
keep_first = int( sys.argv[3] )
|
||||
tmp_title = tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
seq_hash = {}
|
||||
|
||||
fasta_title = fasta_seq = ''
|
||||
|
||||
# number of char to keep in the title
|
||||
if keep_first == 0:
|
||||
keep_first = None
|
||||
else:
|
||||
keep_first += 1
|
||||
|
||||
for i, line in enumerate( file( input_filename ) ):
|
||||
out = open(outfile, 'w')
|
||||
|
||||
for i, line in enumerate( file( infile ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
continue
|
||||
if line[0] == '>':
|
||||
if len( tmp_seq ) > 0:
|
||||
tmp_seq_count += 1
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
tmp_title = line
|
||||
tmp_seq = ''
|
||||
if len( fasta_seq ) > 0 :
|
||||
out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) )
|
||||
fasta_title = line
|
||||
fasta_seq = ''
|
||||
else:
|
||||
tmp_seq = "%s%s" % ( tmp_seq, line )
|
||||
if line.split() and line.split()[0].isdigit():
|
||||
tmp_seq = "%s " % tmp_seq
|
||||
if len( tmp_seq ) > 0:
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
output_handle = open( output_filename, 'w' )
|
||||
for i, fasta_title in title_keys:
|
||||
tmp_seq = seq_hash[ ( i, fasta_title ) ]
|
||||
output_handle.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( tmp_seq ) ) )
|
||||
output_handle.close()
|
||||
fasta_seq = "%s%s" % ( fasta_seq, line )
|
||||
|
||||
# check the last sequence
|
||||
if len( fasta_seq ) > 0:
|
||||
out.write( "%s\t%d\n" % ( fasta_title[ 1:keep_first ], len( fasta_seq ) ) )
|
||||
|
||||
out.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -15,7 +15,7 @@ def stop_err( msg ):
|
||||
|
||||
def __main__():
|
||||
|
||||
input_filename = sys.argv[1]
|
||||
infile = sys.argv[1]
|
||||
try:
|
||||
min_length = int( sys.argv[2] )
|
||||
except:
|
||||
@@ -24,49 +24,62 @@ def __main__():
|
||||
max_length = int( sys.argv[3] )
|
||||
except:
|
||||
stop_err( "Maximum length of the return sequence requires a numerical value." )
|
||||
output_filename = sys.argv[4]
|
||||
tmp_title = tmp_seq = ''
|
||||
tmp_seq_count = 0
|
||||
seq_hash = {}
|
||||
outfile = sys.argv[4]
|
||||
fasta_title = fasta_seq = ''
|
||||
at_least_one = 0
|
||||
|
||||
out = open( outfile, 'w' )
|
||||
|
||||
for i, line in enumerate( file( input_filename ) ):
|
||||
for i, line in enumerate( file( infile ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
continue
|
||||
|
||||
if line[0] == '>':
|
||||
if len( tmp_seq ) > 0:
|
||||
tmp_seq_count += 1
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
tmp_title = line
|
||||
tmp_seq = ''
|
||||
if len( fasta_seq ) > 0:
|
||||
|
||||
if max_length <= 0:
|
||||
compare_max_length = len( fasta_seq ) + 1
|
||||
else:
|
||||
compare_max_length = max_length
|
||||
|
||||
l = len( fasta_seq )
|
||||
|
||||
if l >= min_length and l <= compare_max_length:
|
||||
at_least_one += 1
|
||||
out.write( "%s\n" % fasta_title )
|
||||
c = 0
|
||||
s = fasta_seq
|
||||
while c < l:
|
||||
b = min( c + 50, l )
|
||||
out.write( "%s\n" % s[ c:b ] )
|
||||
c = b
|
||||
|
||||
fasta_title = line
|
||||
fasta_seq = ''
|
||||
else:
|
||||
tmp_seq = "%s%s" % ( tmp_seq, line )
|
||||
if line.split()[0].isdigit():
|
||||
tmp_seq = "%s " % tmp_seq
|
||||
if len( tmp_seq ) > 0:
|
||||
seq_hash[ ( tmp_seq_count, tmp_title ) ] = tmp_seq
|
||||
fasta_seq = "%s%s" % ( fasta_seq, line )
|
||||
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
output_handle = open( output_filename, 'w' )
|
||||
at_least_one = 0
|
||||
for i, fasta_title in title_keys:
|
||||
tmp_seq = seq_hash[ ( i, fasta_title ) ]
|
||||
if len( fasta_seq ) > 0:
|
||||
|
||||
if max_length <= 0:
|
||||
compare_max_length = len( tmp_seq ) + 1
|
||||
compare_max_length = len( fasta_seq ) + 1
|
||||
else:
|
||||
compare_max_length = max_length
|
||||
l = len( tmp_seq )
|
||||
|
||||
l = len( fasta_seq )
|
||||
|
||||
if l >= min_length and l <= compare_max_length:
|
||||
at_least_one += 1
|
||||
output_handle.write( "%s\n" % fasta_title )
|
||||
out.write( "%s\n" % fasta_title )
|
||||
c = 0
|
||||
s = tmp_seq
|
||||
s = fasta_seq
|
||||
while c < l:
|
||||
b = min( c + 50, l )
|
||||
output_handle.write( "%s\n" % s[ c:b ] )
|
||||
out.write( "%s\n" % s[ c:b ] )
|
||||
c = b
|
||||
output_handle.close()
|
||||
|
||||
out.close()
|
||||
|
||||
if at_least_one == 0:
|
||||
print "There is no sequence that falls within your range."
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
#! /usr/bin/python
|
||||
# This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools
|
||||
"""
|
||||
Input: fasta, minimal length, maximal length
|
||||
Output: fasta
|
||||
Return sequences whose lengths are within the range.
|
||||
Input: fasta, int
|
||||
Output: tabular
|
||||
format convert: fasta to tabular
|
||||
"""
|
||||
|
||||
import sys, os
|
||||
@@ -14,38 +14,31 @@ def __main__():
|
||||
infile = sys.argv[1]
|
||||
outfile = sys.argv[2]
|
||||
keep_first = int( sys.argv[3] )
|
||||
title = ''
|
||||
sequence = ''
|
||||
sequence_count = 0
|
||||
fasta_title = fasta_seq = ''
|
||||
|
||||
if keep_first == 0:
|
||||
keep_first = None
|
||||
else:
|
||||
keep_first += 1
|
||||
|
||||
out = open( outfile, 'w' )
|
||||
|
||||
for i, line in enumerate( open( infile ) ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
if not line or line.startswith( '#' ):
|
||||
continue
|
||||
if line.startswith( '>' ):
|
||||
if sequence:
|
||||
sequence_count += 1
|
||||
seq_hash[( sequence_count, title )] = sequence
|
||||
title = line
|
||||
sequence = ''
|
||||
if fasta_seq:
|
||||
out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) )
|
||||
fasta_title = line
|
||||
fasta_seq = ''
|
||||
else:
|
||||
sequence = "%s%s" % ( sequence, line )
|
||||
if line.split() and line.split()[0].isdigit():
|
||||
sequence += ' '
|
||||
if sequence:
|
||||
seq_hash[( sequence_count, title )] = sequence
|
||||
# return only those lengths are in the range
|
||||
title_keys = seq_hash.keys()
|
||||
title_keys.sort()
|
||||
out = open( outfile, 'w' )
|
||||
for i, fasta_title in title_keys:
|
||||
sequence = seq_hash[( i, fasta_title )]
|
||||
out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], sequence ) )
|
||||
if line:
|
||||
fasta_seq = "%s%s" % ( fasta_seq, line )
|
||||
|
||||
if fasta_seq:
|
||||
out.write( "%s\t%s\n" %( fasta_title[ 1:keep_first ], fasta_seq ) )
|
||||
|
||||
out.close()
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -128,11 +128,13 @@ def output_writer(blk, blk_lines):
|
||||
uniq_s_elems_2 = get_binned_lists(uniq_s_elems_2,s_bin_size)
|
||||
|
||||
for pitem1 in uniq_elems_1:
|
||||
repeats1 = []
|
||||
repeats2 = []
|
||||
#repeats1 = []
|
||||
#repeats2 = []
|
||||
thresholds = []
|
||||
if s_group_cols[0] != -1: #Sub-group by feature is not None
|
||||
for sitem1 in uniq_s_elems_1:
|
||||
repeats1 = []
|
||||
repeats2 = []
|
||||
if type(sitem1) == type(''):
|
||||
sitem1 = sitem1.strip()
|
||||
for bline in blk_lines:
|
||||
@@ -223,11 +225,13 @@ def output_writer(blk, blk_lines):
|
||||
count1[str(pitem1)]=sum(repeats2)
|
||||
|
||||
for pitem2 in uniq_elems_2:
|
||||
repeats1 = []
|
||||
repeats2 = []
|
||||
#repeats1 = []
|
||||
#repeats2 = []
|
||||
thresholds = []
|
||||
if s_group_cols[0] != -1: #Sub-group by feature is not None
|
||||
for sitem2 in uniq_s_elems_2:
|
||||
repeats1 = []
|
||||
repeats2 = []
|
||||
if type(sitem2)==type(''):
|
||||
sitem2 = sitem2.strip()
|
||||
for bline in blk_lines:
|
||||
@@ -349,11 +353,10 @@ def output_writer(blk, blk_lines):
|
||||
count = count1[key]
|
||||
mut = "%.2e" %(mut/num_generations)
|
||||
if region == 'align':
|
||||
print >>fout, str(blk) + '\t'+seq1 + '\t' + start1+ '\t'+end1+ '\t'+seq2 + '\t'+start2+ '\t'+end2+ '\t'+key.strip()+ '\t'+str(mut) + '\t'+ str(count)
|
||||
print >>fout, str(blk) + '\t'+seq1 + '\t' + seq2 + '\t' +key.strip()+ '\t'+str(mut) + '\t'+ str(count)
|
||||
elif region == 'win':
|
||||
fout.write("%s\t%s\t%s\t%s\n" %(blk,key.strip(),mut,count))
|
||||
fout.flush()
|
||||
#print >>fout, blk + '\t'+key.strip()+ '\t'+str(mut)+ '\t'+ str(count)
|
||||
|
||||
#catch any remaining repeats, for instance if the orthologous position contained different repeat units
|
||||
for remaining_key in mut2.keys():
|
||||
@@ -361,7 +364,7 @@ def output_writer(blk, blk_lines):
|
||||
mut = "%.2e" %(mut/num_generations)
|
||||
count = count2[remaining_key]
|
||||
if region == 'align':
|
||||
print >>fout, str(blk) + '\t'+seq1 + '\t' + start1+ '\t'+end1+ '\t'+seq2 + '\t'+start2+ '\t'+end2+ '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count)
|
||||
print >>fout, str(blk) + '\t'+seq1 + '\t'+seq2 + '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count)
|
||||
elif region == 'win':
|
||||
fout.write("%s\t%s\t%s\t%s\n" %(blk,remaining_key.strip(),mut,count))
|
||||
fout.flush()
|
||||
@@ -409,19 +412,9 @@ def main():
|
||||
blk=0
|
||||
win=0
|
||||
linestr=""
|
||||
ff=open("junkmabc","w")
|
||||
|
||||
if region == 'win':
|
||||
|
||||
"""
|
||||
sorted_infile = tempfile.NamedTemporaryFile()
|
||||
if winspecies == 1:
|
||||
cmdline = "sort -n -k 3"+" -o "+sorted_infile.name+" "+infile
|
||||
elif winspecies == 2:
|
||||
cmdline = "sort -n -k 10"+" -o "+sorted_infile.name+" "+infile
|
||||
os.system(cmdline)
|
||||
print >>ff, "Finished sorting"
|
||||
"""
|
||||
msats = NiceReaderWrapper( fileinput.FileInput( infile ),
|
||||
chrom_col = speciesind,
|
||||
start_col = speciesind+1,
|
||||
@@ -430,17 +423,9 @@ def main():
|
||||
fix_strand = True)
|
||||
msatTree = quicksect.IntervalTree()
|
||||
for item in msats:
|
||||
#print >>sys.stderr, item
|
||||
if type( item ) is GenomicInterval:
|
||||
msatTree.insert( item, msats.linenum, item.fields )
|
||||
"""
|
||||
result = []
|
||||
msatTree.traverse(lambda node: result.append( node ))
|
||||
for n in result:
|
||||
print >>sys.stderr,n.other
|
||||
print >>sys.stderr,msatTree.chroms
|
||||
#sys.exit()
|
||||
"""
|
||||
|
||||
for iline in fint:
|
||||
try:
|
||||
iline = iline.rstrip('\r\n')
|
||||
@@ -470,7 +455,7 @@ def main():
|
||||
print "Skipped %d intervals as invalid." %(skipped)
|
||||
elif region == 'align':
|
||||
if s_group_cols[0] != -1:
|
||||
print >>fout, "#Window\tSpecies_1\tWindow_Start\tWindow_End\tSpecies_2\tGroupby_Feature\tSubGroupby_Feature\tMutability\tCount"
|
||||
print >>fout, "#Window\tSpecies_1\tSpecies_2\tGroupby_Feature\tSubGroupby_Feature\tMutability\tCount"
|
||||
else:
|
||||
print >>fout, "#Window\tSpecies_1\tWindow_Start\tWindow_End\tSpecies_2\tGroupby_Feature\tMutability\tCount"
|
||||
prev_bnum = -1
|
||||
|
||||
@@ -106,6 +106,10 @@
|
||||
|
||||
This tool computes microsatellite mutability for the orthologous microsatellites fetched from 'Extract Orthologous Microsatellites from pair-wise alignments' tool.
|
||||
|
||||
Mutability is computed according to the method described in the following paper:
|
||||
|
||||
*Webster et al., Microsatellite evolution inferred from human-chimpanzee genomic sequence alignments, Proc Natl Acad Sci 2002 June 25; 99(13): 8748-8753*
|
||||
|
||||
-----
|
||||
|
||||
.. class:: warningmark
|
||||
|
||||
+13
-72
@@ -12,15 +12,13 @@ def stop_err(msg):
|
||||
|
||||
def main():
|
||||
inputfile = sys.argv[2]
|
||||
in_columns = int( sys.argv[5] )
|
||||
show_remaining_cols = sys.argv[4]
|
||||
|
||||
ops = []
|
||||
cols = []
|
||||
rounds = []
|
||||
elems = []
|
||||
|
||||
for var in sys.argv[6:]:
|
||||
for var in sys.argv[4:]:
|
||||
ops.append(var.split()[0])
|
||||
cols.append(var.split()[1])
|
||||
rounds.append(var.split()[2])
|
||||
@@ -81,19 +79,9 @@ def main():
|
||||
|
||||
if error_code != 0:
|
||||
stop_err( "Sorting input dataset resulted in error: %s: %s" %( error_code, stdout ))
|
||||
|
||||
if show_remaining_cols == 'yes':
|
||||
show_cols_list = [1]*in_columns
|
||||
show_cols_list[group_col] = 0
|
||||
for c in cols:
|
||||
c = int(c)-1
|
||||
show_cols_list[c] = 0
|
||||
#at the end of this, only the indices of the remaining columns will be set to 1
|
||||
remaining_cols = [j for j,k in enumerate(show_cols_list) if k==1] #this is the list of remaining column indices
|
||||
|
||||
|
||||
prev_item = ""
|
||||
prev_vals = []
|
||||
remaining_vals = []
|
||||
skipped_lines = 0
|
||||
first_invalid_line = 0
|
||||
invalid_line = ''
|
||||
@@ -128,30 +116,22 @@ def main():
|
||||
invalid_column = col+1
|
||||
if valid:
|
||||
prev_vals[i].append(fields[col].strip())
|
||||
#Store values from all the remaning columns
|
||||
if show_remaining_cols == 'yes':
|
||||
for j, index in enumerate(remaining_cols):
|
||||
remaining_vals[j].append(fields[index].strip())
|
||||
else:
|
||||
"""
|
||||
When a new value is encountered, write the previous value and the
|
||||
corresponding aggregate values into the output file. This works
|
||||
due to the sort on group_col we've applied to the data above.
|
||||
"""
|
||||
out_list = ['']*in_columns
|
||||
out_list[group_col] = str(prev_item)
|
||||
|
||||
out_str = prev_item
|
||||
|
||||
for i, op in enumerate( ops ):
|
||||
rfunc = "r." + op
|
||||
if op not in ['c','length','unique','random']:
|
||||
for j, elem in enumerate( prev_vals[i] ):
|
||||
prev_vals[i][j] = float( elem )
|
||||
rout = "%g" %( eval( rfunc )( prev_vals[i] ))
|
||||
if rounds[i] == 'yes':
|
||||
rout = "%f" %( eval( rfunc )( prev_vals[i] ))
|
||||
rout = int(round(float(rout)))
|
||||
else:
|
||||
rout = "%g" %( eval( rfunc )( prev_vals[i] ))
|
||||
|
||||
else:
|
||||
if op != 'random':
|
||||
rout = eval( rfunc )( prev_vals[i] )
|
||||
@@ -162,21 +142,9 @@ def main():
|
||||
if op == 'unique':
|
||||
rfunc = "r.length"
|
||||
rout = eval( rfunc )( rout )
|
||||
|
||||
out_list[int(cols[i])-1] = str(rout)
|
||||
|
||||
if show_remaining_cols == 'yes':
|
||||
for index,el in enumerate(remaining_cols):
|
||||
if index == 0:
|
||||
try:
|
||||
random_index = random.randint(0,len(remaining_vals[index])-1)
|
||||
except:
|
||||
random_index = 0
|
||||
#pick a random value from each of the remaning columns
|
||||
rand_out = remaining_vals[index][random_index]
|
||||
out_list[el] = str(rand_out)
|
||||
|
||||
print >>fout, '\t'.join([elem for elem in out_list if elem != ''])
|
||||
out_str += "\t" + str(rout)
|
||||
|
||||
print >>fout, out_str
|
||||
|
||||
prev_item = item
|
||||
prev_vals = []
|
||||
@@ -185,14 +153,6 @@ def main():
|
||||
val_list = []
|
||||
val_list.append(fields[col].strip())
|
||||
prev_vals.append(val_list)
|
||||
|
||||
if show_remaining_cols == 'yes':
|
||||
remaining_vals = []
|
||||
for index in remaining_cols:
|
||||
remaining_val_list = []
|
||||
remaining_val_list.append(fields[index].strip())
|
||||
remaining_vals.append(remaining_val_list)
|
||||
|
||||
else:
|
||||
# This only occurs once, right at the start of the iteration.
|
||||
prev_item = item
|
||||
@@ -201,15 +161,8 @@ def main():
|
||||
val_list = []
|
||||
val_list.append(fields[col].strip())
|
||||
prev_vals.append(val_list)
|
||||
|
||||
if show_remaining_cols == 'yes':
|
||||
remaining_vals = []
|
||||
for index in remaining_cols:
|
||||
remaining_val_list = []
|
||||
remaining_val_list.append(fields[index].strip())
|
||||
remaining_vals.append(remaining_val_list)
|
||||
|
||||
except Exception:
|
||||
except Exception, exc:
|
||||
skipped_lines += 1
|
||||
if not first_invalid_line:
|
||||
first_invalid_line = ii+1
|
||||
@@ -219,8 +172,7 @@ def main():
|
||||
first_invalid_line = ii+1
|
||||
|
||||
# Handle the last grouped value
|
||||
out_list = ['']*in_columns
|
||||
out_list[group_col] = str(prev_item)
|
||||
out_str = prev_item
|
||||
|
||||
for i, op in enumerate(ops):
|
||||
rfunc = "r." + op
|
||||
@@ -228,11 +180,9 @@ def main():
|
||||
if op not in ['c','length','unique','random']:
|
||||
for j, elem in enumerate( prev_vals[i] ):
|
||||
prev_vals[i][j] = float( elem )
|
||||
rout = '%g' %( eval( rfunc )( prev_vals[i] ))
|
||||
if rounds[i] == 'yes':
|
||||
rout = '%f' %( eval( rfunc )( prev_vals[i] ))
|
||||
rout = int(round(float(rout)))
|
||||
else:
|
||||
rout = '%g' %( eval( rfunc )( prev_vals[i] ))
|
||||
else:
|
||||
if op != 'random':
|
||||
rout = eval( rfunc )( prev_vals[i] )
|
||||
@@ -243,22 +193,13 @@ def main():
|
||||
if op == 'unique':
|
||||
rfunc = "r.length"
|
||||
rout = eval( rfunc )( rout )
|
||||
out_list[int(cols[i])-1] = str(rout)
|
||||
out_str += "\t" + str( rout )
|
||||
except:
|
||||
skipped_lines += 1
|
||||
if not first_invalid_line:
|
||||
first_invalid_line = ii+1
|
||||
if show_remaining_cols == 'yes':
|
||||
for index,el in enumerate(remaining_cols):
|
||||
if index == 0:
|
||||
try:
|
||||
random_index = random.randint(0,len(remaining_vals[index])-1)
|
||||
except:
|
||||
random_index = 0
|
||||
rand_out = remaining_vals[index][random_index]
|
||||
out_list[el] = str(rand_out)
|
||||
|
||||
print >>fout, '\t'.join([elem for elem in out_list if elem != ''])
|
||||
print >>fout, out_str
|
||||
|
||||
# Generate a useful info message.
|
||||
msg = "--Group by c%d: " %(group_col+1)
|
||||
|
||||
+71
-78
@@ -1,49 +1,43 @@
|
||||
<tool id="Grouping1" name="Group" version="1.5.0">
|
||||
<description>data by a column and perform aggregate operation on other columns.</description>
|
||||
<command interpreter="python">
|
||||
grouping.py
|
||||
$out_file1
|
||||
$input1
|
||||
<tool id="Grouping1" name="Group" version="1.4.0">
|
||||
<description>data by a column and perform aggregate operation on other columns.</description>
|
||||
<command interpreter="python">
|
||||
grouping.py
|
||||
$out_file1
|
||||
$input1
|
||||
$groupcol
|
||||
$othercols
|
||||
${input1.metadata.columns}
|
||||
#for $op in $operations
|
||||
'${op.optype}
|
||||
${op.opcol}
|
||||
${op.opround}'
|
||||
#end for
|
||||
</command>
|
||||
<inputs>
|
||||
<param format="tabular" name="input1" type="data" label="Select data" help="Query missing? See TIP below."/>
|
||||
<param name="groupcol" label="Group by column" type="data_column" data_ref="input1" />
|
||||
<repeat name="operations" title="Operation">
|
||||
<param name="optype" type="select" label="Type">
|
||||
<option value="mean">Mean</option>
|
||||
<option value="max">Maximum</option>
|
||||
<option value="min">Minimum</option>
|
||||
<option value="sum">Sum</option>
|
||||
<option value="length">Count</option>
|
||||
<option value="unique">Count Distinct</option>
|
||||
#for $op in $operations
|
||||
'${op.optype}
|
||||
${op.opcol}
|
||||
${op.opround}'
|
||||
#end for
|
||||
</command>
|
||||
<inputs>
|
||||
<param format="tabular" name="input1" type="data" label="Select data" help="Query missing? See TIP below."/>
|
||||
<param name="groupcol" label="Group by column" type="data_column" data_ref="input1" />
|
||||
<repeat name="operations" title="Operation">
|
||||
<param name="optype" type="select" label="Type">
|
||||
<option value="mean">Mean</option>
|
||||
<option value="max">Maximum</option>
|
||||
<option value="min">Minimum</option>
|
||||
<option value="sum">Sum</option>
|
||||
<option value="length">Count</option>
|
||||
<option value="unique">Count Distinct</option>
|
||||
<option value="c">Concatenate</option>
|
||||
<option value="random">Randomly pick</option>
|
||||
</param>
|
||||
<param name="opcol" label="On column" type="data_column" data_ref="input1" />
|
||||
<param name="opround" type="select" label="Round result to nearest integer?">
|
||||
<option value="no">NO</option>
|
||||
<option value="yes">YES</option>
|
||||
</param>
|
||||
</repeat>
|
||||
<param name="othercols" type="select" label="Randomly pick an entry from each of the remaining columns (besides the columns chosen for group and aggregate operations above) ?">
|
||||
<option value="random">Randomly pick</option>
|
||||
</param>
|
||||
<param name="opcol" label="On column" type="data_column" data_ref="input1" />
|
||||
<param name="opround" type="select" label="Round result to nearest integer?">
|
||||
<option value="no">NO</option>
|
||||
<option value="yes">YES</option>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="input" name="out_file1" metadata_source="input1" />
|
||||
</outputs>
|
||||
<requirements>
|
||||
<requirement type="python-module">rpy</requirement>
|
||||
</requirements>
|
||||
</param>
|
||||
</repeat>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="input" name="out_file1" metadata_source="input1" />
|
||||
</outputs>
|
||||
<requirements>
|
||||
<requirement type="python-module">rpy</requirement>
|
||||
</requirements>
|
||||
<tests>
|
||||
<!-- Test valid data -->
|
||||
<test>
|
||||
@@ -52,9 +46,9 @@
|
||||
<param name="optype" value="mean"/>
|
||||
<param name="opcol" value="2"/>
|
||||
<param name="opround" value="no"/>
|
||||
<param name="othercols" value="no"/>
|
||||
<output name="out_file1" file="groupby_out1.dat"/>
|
||||
</test>
|
||||
|
||||
<!-- Test data with an invalid value in a column -->
|
||||
<test>
|
||||
<param name="input1" value="1.tabular"/>
|
||||
@@ -62,39 +56,38 @@
|
||||
<param name="optype" value="mean"/>
|
||||
<param name="opcol" value="2"/>
|
||||
<param name="opround" value="no"/>
|
||||
<param name="othercols" value="no"/>
|
||||
<output name="out_file1" file="groupby_out2.dat"/>
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert*
|
||||
|
||||
-----
|
||||
|
||||
**Syntax**
|
||||
|
||||
This tool allows you to group the input dataset by a particular column and perform aggregate functions like Mean, Sum, Max, Min and Concatenate on other columns.
|
||||
|
||||
- All invalid, blank and comment lines are skipped when performing the aggregate functions. The number of skipped lines is displayed in the resulting history item.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
- For the following input::
|
||||
|
||||
chr22 1000 NM_17
|
||||
chr22 2000 NM_18
|
||||
chr10 2200 NM_10
|
||||
chr10 1200 NM_11
|
||||
chr22 1600 NM_19
|
||||
|
||||
- running this tool with **Group by column 1**, Operations **Mean on column 2** and **Concatenate on column 3** will return::
|
||||
|
||||
chr10 1700.00 ['NM_11', 'NM_10']
|
||||
chr22 1533.33 ['NM_17', 'NM_19', 'NM_18']
|
||||
</help>
|
||||
</tool>
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert*
|
||||
|
||||
-----
|
||||
|
||||
**Syntax**
|
||||
|
||||
This tool allows you to group the input dataset by a particular column and perform aggregate functions like Mean, Sum, Max, Min and Concatenate on other columns.
|
||||
|
||||
- All invalid, blank and comment lines are skipped when performing the aggregate functions. The number of skipped lines is displayed in the resulting history item.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
- For the following input::
|
||||
|
||||
chr22 1000 NM_17
|
||||
chr22 2000 NM_18
|
||||
chr10 2200 NM_10
|
||||
chr10 1200 NM_11
|
||||
chr22 1600 NM_19
|
||||
|
||||
- running this tool with **Group by column 1**, Operations **Mean on column 2** and **Concatenate on column 3** will return::
|
||||
|
||||
chr10 1700.00 ['NM_11', 'NM_10']
|
||||
chr22 1533.33 ['NM_17', 'NM_19', 'NM_18']
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
@@ -71,7 +71,8 @@ taxRank = {
|
||||
'genus' :20,
|
||||
'subgenus' :21,
|
||||
'species' :22,
|
||||
'subspecies' :23
|
||||
'subspecies' :23,
|
||||
'order' :13
|
||||
}
|
||||
|
||||
|
||||
@@ -157,12 +158,16 @@ try:
|
||||
for item in cur.fetchall():
|
||||
out_string = '%s\t%s\t%d\t' % ( item[0], item[1], item[2] )
|
||||
out_string += rankName
|
||||
out_string += '\t'
|
||||
out_string += str(taxRank[rankName])
|
||||
print >>out_file, out_string
|
||||
else:
|
||||
cur.execute('select rank, count(*) from %s_count where N = 1 and length(rank)>1 group by rank' % rank)
|
||||
for item in cur.fetchall():
|
||||
out_string = '%s\t%s\t' % ( item[0], item[1] )
|
||||
out_string += rankName
|
||||
out_string += '\t'
|
||||
out_string += str(taxRank[rankName])
|
||||
print >>out_file, out_string
|
||||
except Exception, e:
|
||||
stop_err("%s\n" % e)
|
||||
|
||||
@@ -0,0 +1,186 @@
|
||||
#!/usr/bin/env python
|
||||
#Guruprasad Ananda
|
||||
"""
|
||||
This tool provides the SQL "group by" functionality.
|
||||
"""
|
||||
import sys, string, re, commands, tempfile, random
|
||||
#from rpy import *
|
||||
|
||||
def stop_err(msg):
|
||||
sys.stderr.write(msg)
|
||||
sys.exit()
|
||||
|
||||
def main():
|
||||
try:
|
||||
inputfile = sys.argv[1]
|
||||
outfile = sys.argv[2]
|
||||
rank_bound = int( sys.argv[3] )
|
||||
"""
|
||||
Mapping of ranks:
|
||||
root :2,
|
||||
superkingdom:3,
|
||||
kingdom :4,
|
||||
subkingdom :5,
|
||||
superphylum :6,
|
||||
phylum :7,
|
||||
subphylum :8,
|
||||
superclass :9,
|
||||
class :10,
|
||||
subclass :11,
|
||||
superorder :12,
|
||||
order :13,
|
||||
suborder :14,
|
||||
superfamily :15,
|
||||
family :16,
|
||||
subfamily :17,
|
||||
tribe :18,
|
||||
subtribe :19,
|
||||
genus :20,
|
||||
subgenus :21,
|
||||
species :22,
|
||||
subspecies :23,
|
||||
"""
|
||||
except:
|
||||
stop_err("Syntax error: Use correct syntax: program infile outfile")
|
||||
group_col = 0
|
||||
|
||||
tmpfile = tempfile.NamedTemporaryFile()
|
||||
|
||||
try:
|
||||
"""
|
||||
The -k option for the Posix sort command is as follows:
|
||||
-k, --key=POS1[,POS2]
|
||||
start a key at POS1, end it at POS2 (origin 1)
|
||||
In other words, column positions start at 1 rather than 0, so
|
||||
we need to add 1 to group_col.
|
||||
if POS2 is not specified, the newer versions of sort will consider the entire line for sorting. To prevent this, we set POS2=POS1.
|
||||
"""
|
||||
command_line = "sort -f -k " + str(group_col+1) +"," + str(group_col+1) + " -o " + tmpfile.name + " " + inputfile
|
||||
except Exception, exc:
|
||||
stop_err( 'Initialization error -> %s' %str(exc) )
|
||||
|
||||
error_code, stdout = commands.getstatusoutput(command_line)
|
||||
|
||||
if error_code != 0:
|
||||
stop_err( "Sorting input dataset resulted in error: %s: %s" %( error_code, stdout ))
|
||||
|
||||
prev_item = ""
|
||||
prev_vals = []
|
||||
remaining_vals = []
|
||||
skipped_lines = 0
|
||||
first_invalid_line = 0
|
||||
invalid_line = ''
|
||||
invalid_value = ''
|
||||
invalid_column = 0
|
||||
fout = open(outfile, "w")
|
||||
cols = range(1,25)
|
||||
block_valid = False
|
||||
|
||||
for ii, line in enumerate( file( tmpfile.name )):
|
||||
if line and not line.startswith( '#' ):
|
||||
line = line.rstrip( '\r\n' )
|
||||
try:
|
||||
fields = line.split("\t")
|
||||
item = fields[group_col]
|
||||
if prev_item != "":
|
||||
# At this level, we're grouping on values (item and prev_item) in group_col
|
||||
if item == prev_item:
|
||||
# Keep iterating and storing values until a new value is encountered.
|
||||
if block_valid:
|
||||
for i, col in enumerate(cols):
|
||||
if col >= 3:
|
||||
prev_vals[i].append(fields[col].strip())
|
||||
if len(set(prev_vals[i])) > 1:
|
||||
block_valid = False
|
||||
break
|
||||
|
||||
else:
|
||||
"""
|
||||
When a new value is encountered, write the previous value and the
|
||||
corresponding aggregate values into the output file. This works
|
||||
due to the sort on group_col we've applied to the data above.
|
||||
"""
|
||||
out_list = ['']*25
|
||||
out_list[0] = str(prev_item)
|
||||
out_list[1] = str(prev_vals[0][0])
|
||||
out_list[2] = str(prev_vals[1][0])
|
||||
out_list[24] = str(prev_vals[23][0])
|
||||
#print >> fout, prev_vals
|
||||
#sys.exit()
|
||||
for k, col in enumerate(cols):
|
||||
if col >= 3 and col < 24:
|
||||
if len(set(prev_vals[k])) == 1:
|
||||
out_list[col] = prev_vals[k][0]
|
||||
else:
|
||||
break
|
||||
while k < 23:
|
||||
out_list[k+1] = 'n'
|
||||
k += 1
|
||||
|
||||
# print >>fout, '\t'.join(out_list)
|
||||
|
||||
if rank_bound == 0:
|
||||
print >>fout, ''.join(out_list)
|
||||
print 'n'*( 24 - rank_bound )
|
||||
else:
|
||||
print '\t'.join(out_list[rank_bound:24])
|
||||
if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ):
|
||||
print >>fout, '\t'.join(out_list)
|
||||
|
||||
|
||||
block_valid = True
|
||||
prev_item = item
|
||||
prev_vals = []
|
||||
for col in cols:
|
||||
val_list = []
|
||||
val_list.append(fields[col].strip())
|
||||
prev_vals.append(val_list)
|
||||
|
||||
else:
|
||||
# This only occurs once, right at the start of the iteration.
|
||||
block_valid = True
|
||||
prev_item = item #groupby item
|
||||
for col in cols: #everyting else
|
||||
val_list = []
|
||||
val_list.append(fields[col].strip())
|
||||
prev_vals.append(val_list)
|
||||
|
||||
except Exception, exc:
|
||||
skipped_lines += 1
|
||||
if not first_invalid_line:
|
||||
first_invalid_line = ii+1
|
||||
else:
|
||||
skipped_lines += 1
|
||||
if not first_invalid_line:
|
||||
first_invalid_line = ii+1
|
||||
|
||||
# Handle the last grouped value
|
||||
out_list = ['']*25
|
||||
out_list[0] = str(prev_item)
|
||||
out_list[1] = str(prev_vals[0][0])
|
||||
out_list[2] = str(prev_vals[1][0])
|
||||
out_list[24] = str(prev_vals[23][0])
|
||||
for k, col in enumerate(cols):
|
||||
if col >= 3 and col < 24:
|
||||
if len(set(prev_vals[k])) == 1:
|
||||
out_list[col] = prev_vals[k][0]
|
||||
else:
|
||||
break
|
||||
while k < 23:
|
||||
out_list[k+1] = 'n'
|
||||
k += 1
|
||||
|
||||
if rank_bound == 0:
|
||||
print >>fout, '\t'.join(out_list)
|
||||
else:
|
||||
print ''.join(out_list[rank_bound:24])
|
||||
print 'n'*( 24 - rank_bound )
|
||||
if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ):
|
||||
print >>fout, '\t'.join(out_list)
|
||||
|
||||
if skipped_lines > 0:
|
||||
msg= "Skipped %d invalid lines starting with line %d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column )
|
||||
print msg
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,44 @@
|
||||
<tool id="lca1" name="Least Common Ancestor" version="1.0.0">
|
||||
<description></description>
|
||||
<command interpreter="python">
|
||||
lca.py $input1 $out_file1 $rank_bound
|
||||
</command>
|
||||
<inputs>
|
||||
<param format="taxonomy" name="input1" type="data" label="Select taxonomy dataset"/>
|
||||
<param name="rank_bound" label="rank bound" type="select" help="Choose smallest diagnostic level">
|
||||
<option value="0">Everything</option>
|
||||
<option value="3">Superkingdom</option>
|
||||
<option value="4">Kingdom</option>
|
||||
<option value="5">Subkingdom</option>
|
||||
<option value="6">Superphylum</option>
|
||||
<option value="7">Phylum</option>
|
||||
<option value="8">Subphylum</option>
|
||||
<option value="9">Superclass</option>
|
||||
<option value="10">Class</option>
|
||||
<option value="11">Subclass</option>
|
||||
<option value="12">Superorder</option>
|
||||
<option value="13">Order</option>
|
||||
<option value="14">Suborder</option>
|
||||
<option value="15">Superfamily</option>
|
||||
<option value="16">Family</option>
|
||||
<option value="17">Subfamily</option>
|
||||
<option value="18">Tribe</option>
|
||||
<option value="19">Subtribe</option>
|
||||
<option value="20">Genus</option>
|
||||
<option value="21">Subgenus</option>
|
||||
<option value="22">Species</option>
|
||||
<option value="23">Subspecies</option>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="taxonomy" name="out_file1" metadata_source="input1" />
|
||||
</outputs>
|
||||
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
When performing metagenomic analyses it is often necessary to identify sequence reads corresponding to a particular taxonomic group, or, in other words, diagnostic of a particular taxonomic rank. This utility performs this analysis. It takes data generated by *Taxonomy manipulation->Fetch Taxonomic Ranks* as input and outputs either a list of sequence reads unique to a particular taxonomic rank, or a list of taxonomic ranks and the count of unique reads corresponding to each rank.
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
Reference in New Issue
Block a user