diff --git a/cron/get_ncbi.py b/cron/get_ncbi.py new file mode 100644 index 00000000000..74fed7ae67e --- /dev/null +++ b/cron/get_ncbi.py @@ -0,0 +1,93 @@ +import urllib, pkg_resources, os +pkg_resources.require( 'elementtree' ) +from elementtree import ElementTree, ElementInclude +from xml.parsers.expat import ExpatError as XMLParseErrorThing +import sys + +import pkg_resources + +class GetListing: + def __init__( self, data ): + self.tree = ElementTree.parse( data ) + self.root = self.tree.getroot() + ElementInclude.include(self.root) + + def xml_text(self, name=None): + """Returns the text inside an element""" + root = self.root + if name is not None: + # Try attribute first + val = root.get(name) + if val: + return val + # Then try as element + elem = root.find(name) + else: + elem = root + if elem is not None and elem.text: + text = ''.join(elem.text.splitlines()) + return text.strip() + # No luck, return empty string + return '' + +def dlcachefile( webenv, querykey, i, results ): + url = 'http://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=nuccore&usehistory=y&term=nuccore_assembly[filter]%20AND%20refseq[filter]' + fp = urllib.urlopen( url ) + search = GetListing( fp ) + fp.close() + webenv = search.xml_text( 'WebEnv' ) + querykey = search.xml_text( 'QueryKey' ) + url = 'http://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi?db=nuccore&WebEnv=%s&query_key=%s&retstart=%d&retmax=%d' % ( webenv, querykey, i, results ) + fp = urllib.urlopen( url ) + cachefile = os.tmpfile() + for line in fp: + cachefile.write( line ) + fp.close() + cachefile.flush() + cachefile.seek(0) + return cachefile + + +url = 'http://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi?db=nuccore&usehistory=y&term=nuccore_assembly[filter]%20AND%20refseq[filter]' +fp = urllib.urlopen( url ) +results = GetListing( fp ) +fp.close() + +webenv = results.xml_text( 'WebEnv' ) +querykey = results.xml_text( 'QueryKey' ) +counts = int( results.xml_text( 'Count' ) ) +results = 10000 +found = 0 + +for i in range(0, counts + results, results): + rets = dict() + cache = dlcachefile( webenv, querykey, i, results ) + try: + xmldoc = GetListing( cache ) + except (IOError, XMLParseErrorThing): + cache = dlcachefile( webenv, querykey, i, results ) + try: + xmldoc = GetListing( cache ) + except (IOError, XMLParseErrorThing): + cache.close() + exit() + pass + finally: + cache.close() + entries = xmldoc.root.findall( 'DocSum' ) + for entry in entries: + dbkey = None + children = entry.findall('Item') + for item in children: + rets[ item.get('Name') ] = item.text + if not rets['Caption'].startswith('NC_'): + continue + + for ret in rets['Extra'].split('|'): + if not ret.startswith('NC_'): + continue + else: + dbkey = ret + break + if dbkey is not None: + print '\t'.join( [ dbkey, rets['Title'] ] ) diff --git a/cron/updatencbi.sh.sample b/cron/updatencbi.sh.sample new file mode 100644 index 00000000000..b0b84b0afc5 --- /dev/null +++ b/cron/updatencbi.sh.sample @@ -0,0 +1,42 @@ +#!/bin/sh +# +# Script to update NCBI shared data tables. The idea is to update, but if +# the update fails, not replace current data/tables with error +# messages. + +# Edit this line to refer to galaxy's path: +GALAXY=/path/to/galaxy +PYTHONPATH=${GALAXY}/lib +export PYTHONPATH + +# setup directories +echo "Creating required directories." +DIRS=" +${GALAXY}/tool-data/shared/ncbi +${GALAXY}/tool-data/shared/ncbi/new +" +for dir in $DIRS; do + if [ ! -d $dir ]; then + echo "Creating $dir" + mkdir $dir + else + echo "$dir already exists, continuing." + fi +done + +date +echo "Updating NCBI shared data tables." + +# Try to build "builds.txt" +echo "Updating builds.txt" +python ${GALAXY}/cron/get_ncbi.py > ${GALAXY}/tool-data/shared/ncbi/new/builds.txt +if [ $? -eq 0 ] +then + diff ${GALAXY}/tool-data/shared/ncbi/new/builds.txt ${GALAXY}/tool-data/shared/ncbi/builds.txt > /dev/null 2>&1 + if [ $? -ne 0 ] + then + cp -f ${GALAXY}/tool-data/shared/ncbi/new/builds.txt ${GALAXY}/tool-data/shared/ncbi/builds.txt + fi +else + echo "Failed to update builds.txt" >&2 +fi diff --git a/datatypes_conf.xml.sample b/datatypes_conf.xml.sample index cb2f1b3af88..6b0fdce731a 100644 --- a/datatypes_conf.xml.sample +++ b/datatypes_conf.xml.sample @@ -4,6 +4,7 @@ + @@ -22,7 +23,9 @@ - + + + @@ -79,6 +82,7 @@ + @@ -167,7 +171,6 @@ - @@ -244,7 +247,6 @@ - diff --git a/display_applications/gbrowse/gbrowse_gff.xml b/display_applications/gbrowse/gbrowse_gff.xml index 6e2bceedfd6..ee1438b6174 100644 --- a/display_applications/gbrowse/gbrowse_gff.xml +++ b/display_applications/gbrowse/gbrowse_gff.xml @@ -1,16 +1,24 @@ - + + - - + + + + + ${site_id in $APP.config.gbrowse_display_sites} - ${dataset.dbkey in $builds} + ${dataset.dbkey in $site_dbkeys} + - ${gbrowse_link}/?${position}eurl=${gff_file.qp} + ${site_link}${site_organism}/?${position}eurl=${gff_file.qp} + + $site_organisms[ $site_dbkeys.index( $gff_file.dbkey ) ] + #set chrom, start, end = $gff_file.datatype.get_estimated_display_viewport( $gff_file ) #if $chrom is not None: diff --git a/display_applications/gbrowse/gbrowse_interval_as_bed.xml b/display_applications/gbrowse/gbrowse_interval_as_bed.xml index a81153dcfa3..cc0bb359eae 100644 --- a/display_applications/gbrowse/gbrowse_interval_as_bed.xml +++ b/display_applications/gbrowse/gbrowse_interval_as_bed.xml @@ -1,16 +1,24 @@ - + + - - + + + + + ${site_id in $APP.config.gbrowse_display_sites} - ${dataset.dbkey in $builds} + ${dataset.dbkey in $site_dbkeys} + - ${gbrowse_link}/?${position}eurl=${bed_file.qp} + ${site_link}${site_organism}/?${position}eurl=${bed_file.qp} + + $site_organisms[ $site_dbkeys.index( $bed_file.dbkey ) ] + #set chrom, start, end = $bed_file.datatype.get_estimated_display_viewport( $bed_file ) #if $chrom is not None: diff --git a/display_applications/gbrowse/gbrowse_wig.xml b/display_applications/gbrowse/gbrowse_wig.xml index c5d1597e1b7..56873d0cc84 100644 --- a/display_applications/gbrowse/gbrowse_wig.xml +++ b/display_applications/gbrowse/gbrowse_wig.xml @@ -1,16 +1,24 @@ - + + - - + + + + + ${site_id in $APP.config.gbrowse_display_sites} - ${dataset.dbkey in $builds} + ${dataset.dbkey in $site_dbkeys} + - ${gbrowse_link}/?${position}eurl=${wig_file.qp} + ${site_link}${site_organism}/?${position}eurl=${wig_file.qp} + + $site_organisms[ $site_dbkeys.index( $wig_file.dbkey ) ] + #set chrom, start, end = $wig_file.datatype.get_estimated_display_viewport( $wig_file ) #if $chrom is not None: diff --git a/eggs.ini b/eggs.ini index 93f8800e3a4..bd64da38894 100644 --- a/eggs.ini +++ b/eggs.ini @@ -17,7 +17,7 @@ Cheetah = 2.2.2 ctypes = 1.0.2 DRMAA_python = 0.2 MarkupSafe = 0.12 -mercurial = 2.1.2 +mercurial = 2.2.3 MySQL_python = 1.2.3c1 numpy = 1.6.0 pbs_python = 4.1.0 diff --git a/lib/galaxy/config.py b/lib/galaxy/config.py index e0893cb2c43..b5df8416b07 100644 --- a/lib/galaxy/config.py +++ b/lib/galaxy/config.py @@ -132,7 +132,7 @@ class Configuration( object ): self.log_events = string_as_bool( kwargs.get( 'log_events', 'False' ) ) self.sanitize_all_html = string_as_bool( kwargs.get( 'sanitize_all_html', True ) ) self.ucsc_display_sites = kwargs.get( 'ucsc_display_sites', "main,test,archaea,ucla" ).lower().split(",") - self.gbrowse_display_sites = kwargs.get( 'gbrowse_display_sites', "wormbase,tair,modencode_worm,modencode_fly,sgd_yeast" ).lower().split(",") + self.gbrowse_display_sites = kwargs.get( 'gbrowse_display_sites', "modencode,sgd_yeast,tair,wormbase,wormbase_ws120,wormbase_ws140,wormbase_ws170,wormbase_ws180,wormbase_ws190,wormbase_ws200,wormbase_ws204,wormbase_ws210,wormbase_ws220,wormbase_ws225" ).lower().split(",") self.genetrack_display_sites = kwargs.get( 'genetrack_display_sites', "main,test" ).lower().split(",") self.brand = kwargs.get( 'brand', None ) self.support_url = kwargs.get( 'support_url', 'http://wiki.g2.bx.psu.edu/Support' ) @@ -169,10 +169,21 @@ class Configuration( object ): if self.nginx_upload_store: self.nginx_upload_store = os.path.abspath( self.nginx_upload_store ) self.object_store = kwargs.get( 'object_store', 'disk' ) - self.aws_access_key = kwargs.get( 'aws_access_key', None ) - self.aws_secret_key = kwargs.get( 'aws_secret_key', None ) - self.s3_bucket = kwargs.get( 's3_bucket', None) - self.use_reduced_redundancy = kwargs.get( 'use_reduced_redundancy', False ) + # Handle AWS-specific config options for backward compatibility + if kwargs.get( 'aws_access_key', None) is not None: + self.os_access_key= kwargs.get( 'aws_access_key', None ) + self.os_secret_key= kwargs.get( 'aws_secret_key', None ) + self.os_bucket_name= kwargs.get( 's3_bucket', None ) + self.os_use_reduced_redundancy = kwargs.get( 'use_reduced_redundancy', False ) + else: + self.os_access_key = kwargs.get( 'os_access_key', None ) + self.os_secret_key = kwargs.get( 'os_secret_key', None ) + self.os_bucket_name = kwargs.get( 'os_bucket_name', None ) + self.os_use_reduced_redundancy = kwargs.get( 'os_use_reduced_redundancy', False ) + self.os_host = kwargs.get( 'os_host', None ) + self.os_port = kwargs.get( 'os_port', None ) + self.os_is_secure = string_as_bool( kwargs.get( 'os_is_secure', True ) ) + self.os_conn_path = kwargs.get( 'os_conn_path', '/' ) self.object_store_cache_size = float(kwargs.get( 'object_store_cache_size', -1 )) self.distributed_object_store_config_file = kwargs.get( 'distributed_object_store_config_file', None ) # Parse global_conf and save the parser diff --git a/lib/galaxy/datatypes/converters/bedgraph_to_bigwig_converter.xml b/lib/galaxy/datatypes/converters/bedgraph_to_bigwig_converter.xml new file mode 100644 index 00000000000..77d2c2a42e6 --- /dev/null +++ b/lib/galaxy/datatypes/converters/bedgraph_to_bigwig_converter.xml @@ -0,0 +1,14 @@ + \ No newline at end of file diff --git a/lib/galaxy/datatypes/converters/gff_to_fli.py b/lib/galaxy/datatypes/converters/gff_to_fli.py new file mode 100644 index 00000000000..ef38576a5a3 --- /dev/null +++ b/lib/galaxy/datatypes/converters/gff_to_fli.py @@ -0,0 +1,57 @@ +''' +Creates a feature location index for a given GFF file. +''' + +import sys +from galaxy import eggs +from galaxy.datatypes.util.gff_util import read_unordered_gtf, convert_gff_coords_to_bed + +def main(): + # Process arguments. + in_fname = sys.argv[1] + out_fname = sys.argv[2] + + # Create dict of name-location pairings. + name_loc_dict = {} + for feature in read_unordered_gtf( open( in_fname, 'r' ) ): + for name in feature.attributes: + val = feature.attributes[ name ] + try: + float( val ) + continue + except: + convert_gff_coords_to_bed( feature ) + # Value is not a number, so it can be indexed. + if val not in name_loc_dict: + # Value is not in dictionary. + name_loc_dict[ val ] = { + 'contig': feature.chrom, + 'start': feature.start, + 'end': feature.end + } + else: + # Value already in dictionary, so update dictionary. + loc = name_loc_dict[ val ] + if feature.start < loc[ 'start' ]: + loc[ 'start' ] = feature.start + if feature.end > loc[ 'end' ]: + loc[ 'end' ] = feature.end + + # Print name, loc in sorted order. + out = open( out_fname, 'w' ) + max_len = 0 + entries = [] + for name in sorted( name_loc_dict.iterkeys() ): + loc = name_loc_dict[ name ] + entry = '%s\t%s' % ( name, '%s:%i-%i' % ( loc[ 'contig' ], loc[ 'start' ], loc[ 'end' ] ) ) + if len( entry ) > max_len: + max_len = len( entry ) + entries.append( entry ) + + out.write( str( max_len + 1 ).ljust( max_len ) + '\n' ) + for entry in entries: + out.write( entry.ljust( max_len ) + '\n' ) + out.close() + +if __name__ == '__main__': + main() \ No newline at end of file diff --git a/lib/galaxy/datatypes/converters/gff_to_fli_converter.xml b/lib/galaxy/datatypes/converters/gff_to_fli_converter.xml new file mode 100644 index 00000000000..22f739438a0 --- /dev/null +++ b/lib/galaxy/datatypes/converters/gff_to_fli_converter.xml @@ -0,0 +1,13 @@ + + + + gff_to_fli.py $input1 $output1 + + + + + + + + + diff --git a/lib/galaxy/datatypes/converters/interval_to_interval_index_converter.py.orig b/lib/galaxy/datatypes/converters/interval_to_interval_index_converter.py.orig deleted file mode 100644 index 15d84bd4141..00000000000 --- a/lib/galaxy/datatypes/converters/interval_to_interval_index_converter.py.orig +++ /dev/null @@ -1,79 +0,0 @@ -#!/usr/bin/env python - -""" -Convert from interval file to interval index file. Default input file format is BED (0-based, half-open intervals). - -usage: %prog in_file out_file - -G, --gff: input is GFF format, meaning start and end coordinates are 1-based, closed interval -""" - -from __future__ import division - -import sys, fileinput -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) -from galaxy.visualization.tracks.summary import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.gff_util import convert_gff_coords_to_bed -from bx.interval_index_file import Indexes -from galaxy.tools.util.gff_util import parse_gff_attributes - -def main(): - - # Read options, args. - options, args = doc_optparse.parse( __doc__ ) - try: - gff_format = bool( options.gff ) - input_fname, out_fname = args - except: - doc_optparse.exception() - - # Do conversion. - # TODO: take column numbers from command line. - if gff_format: - chr_col, start_col, end_col = ( 0, 3, 4 ) - else: - chr_col, start_col, end_col = ( 0, 1, 2 ) - index = Indexes() - offset = 0 - # Need to keep track of last gene, transcript id for indexing GTF files. - last_gene_id = None - last_transcript_id = None - for line in open(input_fname, "r"): - feature = line.strip().split('\t') - if not feature or feature[0].startswith("track") or feature[0].startswith("#"): - offset += len(line) - continue - chrom = feature[ chr_col ] - chrom_start = int( feature[ start_col ] ) - chrom_end = int( feature[ end_col ] ) - if gff_format: - chrom_start, chrom_end = convert_gff_coords_to_bed( [chrom_start, chrom_end ] ) - - # Only add feature if gene_id, transcript_id are different from last - # values. - if len( feature ) == 9: - attributes = parse_gff_attributes( feature[8] ) - gene_id = attributes.get( 'gene_id', None ) - transcript_id = attributes.get( 'transcript_id', None ) - if gene_id and transcript_id and gene_id == last_gene_id and \ - transcript_id == last_transcript_id: - # Feature has same gene_id, transcript as last feature, so - # do not add. - offset += len(line) - continue - else: - # gene_id, transcript_id set and are different from last - # values. - last_gene_id = gene_id - last_transcript_id = transcript_id - - #print "%s %s %s %s %i %i %i" % (feature[2], last_gene_id, last_transcript_id, chrom, chrom_start, chrom_end, offset) - index.add( chrom, chrom_start, chrom_end, offset ) - offset += len(line) - - index.write( open(out_fname, "w") ) - -if __name__ == "__main__": - main() - \ No newline at end of file diff --git a/lib/galaxy/datatypes/converters/wig_to_bigwig_converter.xml b/lib/galaxy/datatypes/converters/wig_to_bigwig_converter.xml index e85c3a7b138..d90702efe78 100644 --- a/lib/galaxy/datatypes/converters/wig_to_bigwig_converter.xml +++ b/lib/galaxy/datatypes/converters/wig_to_bigwig_converter.xml @@ -1,6 +1,6 @@