mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Added track storage management and track indexing management. initial JSON controller is there along with viewport templates.
This commit is contained in:
+7
-1
@@ -1,6 +1,7 @@
|
||||
import sys, os, atexit
|
||||
|
||||
from galaxy import config, jobs, util, tools, web
|
||||
from galaxy.tracks import store
|
||||
from galaxy.web import security
|
||||
import galaxy.model
|
||||
import galaxy.model.mapping
|
||||
@@ -32,8 +33,10 @@ class UniverseApplication( object ):
|
||||
self.security = security.SecurityHelper( id_secret=self.config.id_secret )
|
||||
# Initialize the tools
|
||||
self.toolbox = tools.ToolBox( self.config.tool_config, self.config.tool_path, self )
|
||||
#Load datatype converters
|
||||
# Load datatype converters
|
||||
self.datatypes_registry.load_datatype_converters( self.toolbox )
|
||||
# Load datatype indexers
|
||||
self.datatypes_registry.load_datatype_indexers( self.toolbox )
|
||||
#Load security policy
|
||||
self.security_agent = self.model.security_agent
|
||||
# Heartbeat and memdump for thread / heap profiling
|
||||
@@ -60,6 +63,9 @@ class UniverseApplication( object ):
|
||||
# FIXME: These are exposed directly for backward compatibility
|
||||
self.job_queue = self.job_manager.job_queue
|
||||
self.job_stop_queue = self.job_manager.job_stop_queue
|
||||
# Track Store
|
||||
self.track_store = store.TrackStoreManager( self.config.track_store_path )
|
||||
|
||||
def shutdown( self ):
|
||||
self.job_manager.shutdown()
|
||||
if self.heartbeat:
|
||||
|
||||
@@ -31,6 +31,8 @@ class Configuration( object ):
|
||||
# Where dataset files are stored
|
||||
self.file_path = resolve_path( kwargs.get( "file_path", "database/files" ), self.root )
|
||||
self.new_file_path = resolve_path( kwargs.get( "new_file_path", "database/tmp" ), self.root )
|
||||
# dataset Track files
|
||||
self.track_store_path = kwargs.get( "track_store_path", "${extra_files_path}/tracks")
|
||||
self.tool_path = resolve_path( kwargs.get( "tool_path", "tools" ), self.root )
|
||||
self.tool_data_path = resolve_path( kwargs.get( "tool_data_path", "tool-data" ), os.getcwd() )
|
||||
self.test_conf = resolve_path( kwargs.get( "test_conf", "" ), self.root )
|
||||
|
||||
@@ -254,6 +254,10 @@ class Data( object ):
|
||||
"""This function is called on the dataset after metadata is edited."""
|
||||
dataset.clear_associated_files( metadata_safe = True )
|
||||
|
||||
@property
|
||||
def has_resolution(self):
|
||||
return False
|
||||
|
||||
class Text( Data ):
|
||||
|
||||
def write_from_stream(self, dataset, stream):
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
#!/usr/bin/env python
|
||||
"""
|
||||
Generate indices for track browsing of an interval file.
|
||||
|
||||
usage: %prog bed_file out_directory
|
||||
-1, --cols1=N,N,N,N: Columns for chrom, start, end, strand in interval file
|
||||
"""
|
||||
import sys
|
||||
from galaxy import eggs
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
from bx.intervals import io
|
||||
from bx.cookbook import doc_optparse
|
||||
import psyco_full
|
||||
import commands
|
||||
import os
|
||||
from os import environ
|
||||
import tempfile
|
||||
from bisect import bisect
|
||||
|
||||
def divide( intervals, out_path ):
|
||||
current_file = None
|
||||
lastchrom = ""
|
||||
for line in intervals:
|
||||
try:
|
||||
chrom = line.chrom
|
||||
except AttributeError, e:
|
||||
continue
|
||||
if not lastchrom == chrom:
|
||||
if current_file:
|
||||
current_file.flush()
|
||||
current_file.close()
|
||||
current_file = open( os.path.join( out_path, "%s" % chrom), "a" )
|
||||
print >> current_file, "\t".join(line)
|
||||
lastchrom = chrom
|
||||
current_file.flush()
|
||||
current_file.close()
|
||||
|
||||
if __name__ == "__main__":
|
||||
options, args = doc_optparse.parse( __doc__ )
|
||||
try:
|
||||
chr_col_1, start_col_1, end_col_1, strand_col_1 = [int(x)-1 for x in options.cols1.split(',')]
|
||||
in_fname, out_path = args
|
||||
except:
|
||||
doc_optparse.exception()
|
||||
|
||||
# Sort through a tempfile first
|
||||
temp_file = tempfile.NamedTemporaryFile(mode="r")
|
||||
environ['LC_ALL'] = 'POSIX'
|
||||
commandline = "sort -f -n -k %d -k %d -k %d -o %s %s" % (chr_col_1+1,start_col_1+1,end_col_1+1, temp_file.name, in_fname)
|
||||
errorcode, stdout = commands.getstatusoutput(commandline)
|
||||
|
||||
temp_file.seek(0)
|
||||
interval = io.NiceReaderWrapper( temp_file,
|
||||
chrom_col=chr_col_1,
|
||||
start_col=start_col_1,
|
||||
end_col=end_col_1,
|
||||
strand_col=strand_col_1,
|
||||
fix_strand=True )
|
||||
divide( interval, out_path )
|
||||
temp_file.close()
|
||||
@@ -0,0 +1,14 @@
|
||||
<tool id="INDEXER_Interval_0" name="Index Interval for Track Viewer">
|
||||
<!-- Used internally to generate track indexes -->
|
||||
<command interpreter="python">interval.py $input_dataset
|
||||
-1 ${input_dataset.metadata.chromCol},${input_dataset.metadata.startCol},${input_dataset.metadata.endCol},${input_dataset.metadata.strandCol}
|
||||
$store_path
|
||||
</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param format="interval" name="input_dataset" type="data" label="Choose intervals"/>
|
||||
</page>
|
||||
</inputs>
|
||||
<help>
|
||||
</help>
|
||||
</tool>
|
||||
@@ -287,6 +287,22 @@ class Interval( Tabular ):
|
||||
return True
|
||||
except:
|
||||
return False
|
||||
|
||||
def get_track_window(self, dataset, data, start, end):
|
||||
"""
|
||||
Assumes the incoming track data is sorted already.
|
||||
"""
|
||||
window = list()
|
||||
for record in data:
|
||||
fields = record.rstrip("\n\r").split("\t")
|
||||
record_start = int(fields[dataset.metadata.startCol-1])
|
||||
record_end = int(fields[dataset.metadata.endCol-1])
|
||||
if record_start < end and record_end > start:
|
||||
window.append( fields ) #Yes I did want to use a generator here, but it doesn't work downstream
|
||||
return window
|
||||
|
||||
def get_track_resolution( self, dataset, data, start, end):
|
||||
return None
|
||||
|
||||
class Bed( Interval ):
|
||||
"""Tab delimited data in BED format"""
|
||||
|
||||
@@ -16,7 +16,9 @@ class Registry( object ):
|
||||
self.datatypes_by_extension = {}
|
||||
self.mimetypes_by_extension = {}
|
||||
self.datatype_converters = odict()
|
||||
self.datatype_indexers = odict()
|
||||
self.converters = []
|
||||
self.indexers = []
|
||||
self.sniff_order = []
|
||||
self.upload_file_formats = []
|
||||
if root_dir and config:
|
||||
@@ -27,8 +29,11 @@ class Registry( object ):
|
||||
self.log.debug( 'Loading datatypes from %s' % config )
|
||||
registration = root.find( 'registration' )
|
||||
self.datatype_converters_path = os.path.join( root_dir, registration.get( 'converters_path', 'lib/galaxy/datatypes/converters' ) )
|
||||
self.datatype_indexers_path = os.path.join( root_dir, registration.get( 'indexers_path', 'lib/galaxy/datatypes/indexers' ) )
|
||||
if not os.path.isdir( self.datatype_converters_path ):
|
||||
raise ConfigurationError( "Directory does not exist: %s" % self.datatype_converters_path )
|
||||
if not os.path.isdir( self.datatype_indexers_path ):
|
||||
raise ConfigurationError( "Directory does not exist: %s" % self.datatype_indexers_path )
|
||||
for elem in registration.findall( 'datatype' ):
|
||||
try:
|
||||
extension = elem.get( 'extension', None )
|
||||
@@ -57,6 +62,11 @@ class Registry( object ):
|
||||
target_datatype = converter.get( 'target_datatype', None )
|
||||
if converter_config and target_datatype:
|
||||
self.converters.append( ( converter_config, extension, target_datatype ) )
|
||||
for indexer in elem.findall( 'indexer' ):
|
||||
# Build the list of datatype indexers for track building
|
||||
indexer_config = indexer.get( 'file', None )
|
||||
if indexer_config:
|
||||
self.indexers.append( (indexer_config, extension) )
|
||||
except Exception, e:
|
||||
self.log.warning( 'Error loading datatype "%s", problem: %s' % ( extension, str( e ) ) )
|
||||
# Load datatype sniffers from the config
|
||||
@@ -221,6 +231,16 @@ class Registry( object ):
|
||||
self.datatype_converters[source_datatype][target_datatype] = converter
|
||||
self.log.debug( "Loaded converter: %s", converter.id )
|
||||
|
||||
def load_datatype_indexers( self, toolbox ):
|
||||
"""Adds indexers from self.indexers to the toolbox from app"""
|
||||
for elem in self.indexers:
|
||||
tool_config = elem[0]
|
||||
datatype = elem[1]
|
||||
indexer = toolbox.load_tool( os.path.join( self.datatype_indexers_path, tool_config ) )
|
||||
toolbox.tools_by_id[indexer.id] = indexer
|
||||
self.datatype_indexers[datatype] = indexer
|
||||
self.log.debug( "Loaded indexer: %s", indexer.id )
|
||||
|
||||
def get_converters_by_datatype(self, ext):
|
||||
"""Returns available converters by source type"""
|
||||
converters = odict()
|
||||
@@ -233,12 +253,27 @@ class Registry( object ):
|
||||
if ext in self.datatype_converters.keys():
|
||||
converters.update(self.datatype_converters[ext])
|
||||
return converters
|
||||
|
||||
def get_indexers_by_datatype( self, ext ):
|
||||
"""Returns indexers based on datatype"""
|
||||
class_chain = list()
|
||||
source_datatype = type(self.get_datatype_by_extension(ext))
|
||||
for ext_spec in self.datatype_indexers.keys():
|
||||
datatype = type(self.get_datatype_by_extension(ext_spec))
|
||||
if issubclass( source_datatype, datatype ):
|
||||
class_chain.append( ext_spec )
|
||||
# Prioritize based on class chain
|
||||
ext2type = lambda x: self.get_datatype_by_extension(x)
|
||||
class_chain = sorted(class_chain, lambda x,y: issubclass(ext2type(x),ext2type(y)) and -1 or 1)
|
||||
return [self.datatype_indexers[x] for x in class_chain]
|
||||
|
||||
def get_converter_by_target_type(self, source_ext, target_ext):
|
||||
"""Returns a converter based on source and target datatypes"""
|
||||
converters = self.get_converters_by_datatype(source_ext)
|
||||
if target_ext in converters.keys():
|
||||
return converters[target_ext]
|
||||
return None
|
||||
|
||||
def find_conversion_destination_for_dataset_by_extensions( self, dataset, accepted_formats, converter_safe = True ):
|
||||
"""Returns ( target_ext, exisiting converted dataset )"""
|
||||
for convert_ext in self.get_converters_by_datatype( dataset.ext ):
|
||||
@@ -251,4 +286,4 @@ class Registry( object ):
|
||||
else:
|
||||
ret_data = None
|
||||
return ( convert_ext, ret_data )
|
||||
return ( None, None )
|
||||
return ( None, None )
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
import os
|
||||
from galaxy import config
|
||||
|
||||
# This silliness is needed for py2.4 compatibility
|
||||
try:
|
||||
from hashlib import sha1 as sha
|
||||
except:
|
||||
from sha import new as sha
|
||||
|
||||
class BaseCache( object ):
|
||||
"""
|
||||
Base/Abstract Cache for dataset indices.
|
||||
"""
|
||||
pass
|
||||
@@ -0,0 +1,29 @@
|
||||
import os
|
||||
from galaxy import config
|
||||
|
||||
|
||||
class BaseCache( object ):
|
||||
"""
|
||||
Base/Abstract Cache for dataset indices.
|
||||
"""
|
||||
pass
|
||||
|
||||
class FileCache( BaseCache ):
|
||||
def __init__(self, path=""):
|
||||
self.path = path
|
||||
|
||||
def get(self, key, default=None):
|
||||
cache_path = self._get_key_path( key )
|
||||
if os.path.exists( cache_path ):
|
||||
return open( cache_path, "rb" )
|
||||
else:
|
||||
return default
|
||||
|
||||
def set(self, key, value):
|
||||
cache_path = self._get_key_path( key )
|
||||
file = open( cache_path, "wb" )
|
||||
file.write( value )
|
||||
file.close()
|
||||
|
||||
def _get_key_path( self, key ):
|
||||
return os.path.join( self.path, key )
|
||||
@@ -0,0 +1,4 @@
|
||||
PENDING = "pending"
|
||||
NO_DATA = "no data"
|
||||
NO_CHROMOSOME = "no chromosome"
|
||||
DATA = "data"
|
||||
@@ -0,0 +1,61 @@
|
||||
import os
|
||||
from string import Template
|
||||
|
||||
class TemplateSubber( object ):
|
||||
def __init__(self, obj):
|
||||
self.obj = obj
|
||||
def get( self, key, default=None ):
|
||||
return getattr(self.obj, key, default)
|
||||
def __getitem__(self, key):
|
||||
return self.get(key)
|
||||
|
||||
class TrackStoreManager( object ):
|
||||
def __init__(self, path=""):
|
||||
self.path = path
|
||||
|
||||
def get( self, dataset ):
|
||||
s = Template(self.path)
|
||||
return TrackStore( path=s.substitute(TemplateSubber(dataset)) )
|
||||
|
||||
class TrackStore( object ):
|
||||
def __init__(self, path=""):
|
||||
self.path = path
|
||||
|
||||
def get(self, chrom="chr1", resolution=None, **kwargs):
|
||||
if not self.exists: raise self.DoesNotExist("TrackStore at %s does not exist." % self.path)
|
||||
object_path = self._get_object_path( chrom, resolution )
|
||||
if os.path.exists( object_path ):
|
||||
return open( object_path, "rb" )
|
||||
else:
|
||||
try:
|
||||
return kwargs['default']
|
||||
except KeyError:
|
||||
raise self.DoesNotExist("TrackStore object at %s does not exist." % object_path )
|
||||
|
||||
def set(self, chrom="chr1", resolution=None, data=None):
|
||||
if not self.exists: self._build_path( self.path )
|
||||
if not data: return
|
||||
object_path = self._get_object_path( chrom, resolution )
|
||||
fd = open( object_path, "wb" )
|
||||
fd.write( data )
|
||||
fd.close()
|
||||
|
||||
def _get_object_path( self, chrom, resolution ):
|
||||
object_name = chrom
|
||||
if resolution: object_name += "_%d" % resolution
|
||||
return os.path.join( self.path, object_name )
|
||||
|
||||
def _build_path( self, path ):
|
||||
try:
|
||||
print "building path %s " % path
|
||||
os.mkdir( path )
|
||||
except OSError:
|
||||
self._build_path( os.path.dirname( path ) )
|
||||
os.mkdir( path )
|
||||
|
||||
@property
|
||||
def exists(self):
|
||||
return os.path.exists( self.path )
|
||||
|
||||
class DoesNotExist( Exception ):
|
||||
pass
|
||||
@@ -0,0 +1,59 @@
|
||||
import os
|
||||
from string import Template
|
||||
|
||||
class TemplateSubber( object ):
|
||||
def __init__(self, obj):
|
||||
self.obj = obj
|
||||
def get( self, key, default=None ):
|
||||
return getattr(self.obj, key, default)
|
||||
def __getitem__(self, key):
|
||||
return self.get(key)
|
||||
|
||||
class TrackStoreManager( object ):
|
||||
def __init__(self, path=""):
|
||||
self.path = path
|
||||
|
||||
def get( self, dataset ):
|
||||
s = Template(self.path)
|
||||
return TrackStore( path=s.substitute(TemplateSubber(dataset)) )
|
||||
|
||||
class TrackStore( object ):
|
||||
def __init__(self, path=""):
|
||||
self.path = path
|
||||
|
||||
def get(self, chrom="chr1", resolution=None, **kwargs):
|
||||
if not self.exists: raise self.DoesNotExist("TrackStore at %s does not exist." % self.path)
|
||||
object_path = self._get_object_path( chrom, resolution )
|
||||
if os.path.exists( object_path ):
|
||||
return open( object_path, "rb" )
|
||||
else:
|
||||
try:
|
||||
return kwargs['default']
|
||||
except KeyError:
|
||||
raise self.DoesNotExist("TrackStore object at %s does not exist." % object_path )
|
||||
|
||||
def set(self, chrom="chr1", resolution=None, data=None):
|
||||
if not self.exists: self._build_path( self.path )
|
||||
object_path = self._get_object_path( chrom, resolution )
|
||||
fd = open( object_path, "wb" )
|
||||
fd.write( data )
|
||||
fd.close()
|
||||
|
||||
def _get_object_path( self, chrom, resolution ):
|
||||
object_name = chrom
|
||||
if resolution: object_name += "_%d" % resolution
|
||||
return os.path.join( self.path, object_name )
|
||||
|
||||
def _build_path( self, path ):
|
||||
try:
|
||||
os.mkdir( path )
|
||||
except OSError:
|
||||
self._build_path( os.path.dirname( path ) )
|
||||
os.mkdir( path )
|
||||
|
||||
@property
|
||||
def exists(self):
|
||||
return os.path.exists( self.path )
|
||||
|
||||
class DoesNotExist( Exception ):
|
||||
pass
|
||||
@@ -3,7 +3,18 @@ Utility functions used systemwide.
|
||||
|
||||
"""
|
||||
import logging
|
||||
import threading, sets, random, string, md5, re, binascii, pickle, time, datetime, math, re, os
|
||||
import threading, random, string, re, binascii, pickle, time, datetime, math, re, os
|
||||
|
||||
# Older py compatibility
|
||||
try:
|
||||
set()
|
||||
except:
|
||||
from sets import Set as set
|
||||
|
||||
try:
|
||||
from hashlib import md5
|
||||
except ImportError:
|
||||
from md5 import new as md5
|
||||
|
||||
import pkg_resources
|
||||
|
||||
@@ -57,11 +68,11 @@ def unique_id(KEY_SIZE=128):
|
||||
Genenerates a unique ids
|
||||
|
||||
>>> ids = [ unique_id() for i in range(1000) ]
|
||||
>>> len(sets.Set(ids))
|
||||
>>> len(set(ids))
|
||||
1000
|
||||
"""
|
||||
id = str( random.getrandbits( KEY_SIZE ) )
|
||||
return md5.new(id).hexdigest()
|
||||
return md5(id).hexdigest()
|
||||
|
||||
def parse_xml(fname):
|
||||
"""Returns an parsed xml tree"""
|
||||
@@ -74,7 +85,7 @@ def xml_to_string(elem):
|
||||
return text
|
||||
|
||||
# characters that are valid
|
||||
valid_chars = sets.Set(string.letters + string.digits + " -=_.()/+*^,:?!")
|
||||
valid_chars = set(string.letters + string.digits + " -=_.()/+*^,:?!")
|
||||
|
||||
# characters that are allowed but need to be escaped
|
||||
mapped_chars = { '>' :'__gt__',
|
||||
|
||||
@@ -117,7 +117,6 @@
|
||||
</div>
|
||||
%endif
|
||||
## Allow other body level elements
|
||||
${next.body()}
|
||||
</body>
|
||||
## Scripts can be loaded later since they progressively add features to
|
||||
## the panels, but do not change layout
|
||||
|
||||
Reference in New Issue
Block a user