Real Job(tm) support for the library upload tool. Does not include iframe upload for the library side yet.

This commit is contained in:
Nate Coraor
2009-09-25 14:36:12 -04:00
parent 732fbe63c8
commit 87db3952d3
20 changed files with 735 additions and 651 deletions
+13 -11
View File
@@ -357,13 +357,14 @@ class JobWrapper( object ):
# Restore input / output data lists
inp_data = dict( [ ( da.name, da.dataset ) for da in job.input_datasets ] )
out_data = dict( [ ( da.name, da.dataset ) for da in job.output_datasets ] )
out_data.update( [ ( da.name, da.dataset ) for da in job.output_library_datasets ] )
# These can be passed on the command line if wanted as $userId $userEmail
if job.history.user: # check for anonymous user!
userId = '%d' % job.history.user.id
userEmail = str(job.history.user.email)
if job.history and job.history.user: # check for anonymous user!
userId = '%d' % job.history.user.id
userEmail = str(job.history.user.email)
else:
userId = 'Anonymous'
userEmail = 'Anonymous'
userId = 'Anonymous'
userEmail = 'Anonymous'
incoming['userId'] = userId
incoming['userEmail'] = userEmail
# Build params, done before hook so hook can use
@@ -424,7 +425,7 @@ class JobWrapper( object ):
log.debug( "fail(): Moved %s to %s" % ( dataset_path.false_path, dataset_path.real_path ) )
except ( IOError, OSError ), e:
log.error( "fail(): Missing output file in working directory: %s" % e )
for dataset_assoc in job.output_datasets:
for dataset_assoc in job.output_datasets + job.output_library_datasets:
dataset = dataset_assoc.dataset
dataset.refresh()
dataset.state = dataset.states.ERROR
@@ -444,7 +445,7 @@ class JobWrapper( object ):
def change_state( self, state, info = False ):
job = model.Job.get( self.job_id )
job.refresh()
for dataset_assoc in job.output_datasets:
for dataset_assoc in job.output_datasets + job.output_library_datasets:
dataset = dataset_assoc.dataset
dataset.refresh()
dataset.state = state
@@ -504,10 +505,10 @@ class JobWrapper( object ):
self.fail( "Job %s's output dataset(s) could not be read" % job.id )
return
job_context = ExpressionContext( dict( stdout = stdout, stderr = stderr ) )
for dataset_assoc in job.output_datasets:
for dataset_assoc in job.output_datasets + job.output_library_datasets:
context = self.get_dataset_finish_context( job_context, dataset_assoc.dataset.dataset )
#should this also be checking library associations? - can a library item be added from a history before the job has ended? - lets not allow this to occur
for dataset in dataset_assoc.dataset.dataset.history_associations: #need to update all associated output hdas, i.e. history was shared with job running
for dataset in dataset_assoc.dataset.dataset.history_associations + dataset_assoc.dataset.dataset.library_associations: #need to update all associated output hdas, i.e. history was shared with job running
dataset.blurb = 'done'
dataset.peek = 'no peek'
dataset.info = context['stdout'] + context['stderr']
@@ -576,6 +577,7 @@ class JobWrapper( object ):
# custom post process setup
inp_data = dict( [ ( da.name, da.dataset ) for da in job.input_datasets ] )
out_data = dict( [ ( da.name, da.dataset ) for da in job.output_datasets ] )
out_data.update( [ ( da.name, da.dataset ) for da in job.output_library_datasets ] )
param_dict = dict( [ ( p.name, p.value ) for p in job.parameters ] ) # why not re-use self.param_dict here? ##dunno...probably should, this causes tools.parameters.basic.UnvalidatedValue to be used in following methods instead of validated and transformed values during i.e. running workflows
param_dict = self.tool.params_from_strings( param_dict, self.app )
# Check for and move associated_files
@@ -647,11 +649,11 @@ class JobWrapper( object ):
job = model.Job.get( self.job_id )
if self.app.config.outputs_to_working_directory:
self.output_paths = []
for name, data in [ ( da.name, da.dataset.dataset ) for da in job.output_datasets ]:
for name, data in [ ( da.name, da.dataset.dataset ) for da in job.output_datasets + job.output_library_datasets ]:
false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % data.id ) )
self.output_paths.append( DatasetPath( data.id, data.file_name, false_path ) )
else:
self.output_paths = [ DatasetPath( da.dataset.dataset.id, da.dataset.file_name ) for da in job.output_datasets ]
self.output_paths = [ DatasetPath( da.dataset.dataset.id, da.dataset.file_name ) for da in job.output_datasets + job.output_library_datasets ]
return self.output_paths
def get_output_file_id( self, file ):
+8
View File
@@ -74,6 +74,7 @@ class Job( object ):
self.parameters = []
self.input_datasets = []
self.output_datasets = []
self.output_library_datasets = []
self.state = Job.states.NEW
self.info = None
self.job_runner_name = None
@@ -84,6 +85,8 @@ class Job( object ):
self.input_datasets.append( JobToInputDatasetAssociation( name, dataset ) )
def add_output_dataset( self, name, dataset ):
self.output_datasets.append( JobToOutputDatasetAssociation( name, dataset ) )
def add_output_library_dataset( self, name, dataset ):
self.output_library_datasets.append( JobToOutputLibraryDatasetAssociation( name, dataset ) )
def set_state( self, state ):
self.state = state
# For historical reasons state propogates down to datasets
@@ -142,6 +145,11 @@ class JobToOutputDatasetAssociation( object ):
self.name = name
self.dataset = dataset
class JobToOutputLibraryDatasetAssociation( object ):
def __init__( self, name, dataset ):
self.name = name
self.dataset = dataset
class JobExternalOutputMetadata( object ):
def __init__( self, job = None, dataset = None ):
self.job = job
+13 -1
View File
@@ -107,7 +107,7 @@ Dataset.table = Table( "dataset", metadata,
Column( "id", Integer, primary_key=True ),
Column( "create_time", DateTime, default=now ),
Column( "update_time", DateTime, index=True, default=now, onupdate=now ),
Column( "state", TrimmedString( 64 ) ),
Column( "state", TrimmedString( 64 ), index=True ),
Column( "deleted", Boolean, index=True, default=False ),
Column( "purged", Boolean, index=True, default=False ),
Column( "purgable", Boolean, default=True ),
@@ -307,6 +307,7 @@ Job.table = Table( "job", metadata,
Column( "create_time", DateTime, default=now ),
Column( "update_time", DateTime, default=now, onupdate=now ),
Column( "history_id", Integer, ForeignKey( "history.id" ), index=True ),
Column( "library_folder_id", Integer, ForeignKey( "library_folder.id" ), index=True ),
Column( "tool_id", String( 255 ) ),
Column( "tool_version", TEXT, default="1.0.0" ),
Column( "state", String( 64 ), index=True ),
@@ -339,6 +340,12 @@ JobToOutputDatasetAssociation.table = Table( "job_to_output_dataset", metadata,
Column( "dataset_id", Integer, ForeignKey( "history_dataset_association.id" ), index=True ),
Column( "name", String(255) ) )
JobToOutputLibraryDatasetAssociation.table = Table( "job_to_output_library_dataset", metadata,
Column( "id", Integer, primary_key=True ),
Column( "job_id", Integer, ForeignKey( "job.id" ), index=True ),
Column( "ldda_id", Integer, ForeignKey( "library_dataset_dataset_association.id" ), index=True ),
Column( "name", String(255) ) )
JobExternalOutputMetadata.table = Table( "job_external_output_metadata", metadata,
Column( "id", Integer, primary_key=True ),
Column( "job_id", Integer, ForeignKey( "job.id" ), index=True ),
@@ -907,6 +914,9 @@ assign_mapper( context, JobToInputDatasetAssociation, JobToInputDatasetAssociati
assign_mapper( context, JobToOutputDatasetAssociation, JobToOutputDatasetAssociation.table,
properties=dict( job=relation( Job ), dataset=relation( HistoryDatasetAssociation, lazy=False ) ) )
assign_mapper( context, JobToOutputLibraryDatasetAssociation, JobToOutputLibraryDatasetAssociation.table,
properties=dict( job=relation( Job ), dataset=relation( LibraryDatasetDatasetAssociation, lazy=False ) ) )
assign_mapper( context, JobParameter, JobParameter.table )
assign_mapper( context, JobExternalOutputMetadata, JobExternalOutputMetadata.table,
@@ -917,9 +927,11 @@ assign_mapper( context, JobExternalOutputMetadata, JobExternalOutputMetadata.tab
assign_mapper( context, Job, Job.table,
properties=dict( galaxy_session=relation( GalaxySession ),
history=relation( History ),
library_folder=relation( LibraryFolder ),
parameters=relation( JobParameter, lazy=False ),
input_datasets=relation( JobToInputDatasetAssociation, lazy=False ),
output_datasets=relation( JobToOutputDatasetAssociation, lazy=False ),
output_library_datasets=relation( JobToOutputLibraryDatasetAssociation, lazy=False ),
external_output_metadata = relation( JobExternalOutputMetadata, lazy = False ) ) )
assign_mapper( context, Event, Event.table,
@@ -0,0 +1,121 @@
from sqlalchemy import *
from sqlalchemy.orm import *
from sqlalchemy.exceptions import *
from migrate import *
from migrate.changeset import *
import datetime
now = datetime.datetime.utcnow
import sys, logging
# Need our custom types, but don't import anything else from model
from galaxy.model.custom_types import *
log = logging.getLogger( __name__ )
log.setLevel(logging.DEBUG)
handler = logging.StreamHandler( sys.stdout )
format = "%(name)s %(levelname)s %(asctime)s %(message)s"
formatter = logging.Formatter( format )
handler.setFormatter( formatter )
log.addHandler( handler )
metadata = MetaData( migrate_engine )
db_session = scoped_session( sessionmaker( bind=migrate_engine, autoflush=False, transactional=False ) )
def display_migration_details():
print ""
print "========================================"
print """This script creates a job_to_output_library_dataset table for allowing library
uploads to run as regular jobs. To support this, a library_folder_id column is
added to the job table, and library_folder/output_library_datasets relations
are added to the Job object. An index is also added to the dataset.state
column."""
print "========================================"
JobToOutputLibraryDatasetAssociation_table = Table( "job_to_output_library_dataset", metadata,
Column( "id", Integer, primary_key=True ),
Column( "job_id", Integer, ForeignKey( "job.id" ), index=True ),
Column( "ldda_id", Integer, ForeignKey( "library_dataset_dataset_association.id" ), index=True ),
Column( "name", String(255) ) )
def upgrade():
display_migration_details()
# Load existing tables
metadata.reflect()
# Create the job_to_output_library_dataset table
try:
JobToOutputLibraryDatasetAssociation_table.create()
except Exception, e:
print "Creating job_to_output_library_dataset table failed: %s" % str( e )
log.debug( "Creating job_to_output_library_dataset table failed: %s" % str( e ) )
# Create the library_folder_id column
try:
Job_table = Table( "job", metadata, autoload=True )
except NoSuchTableError:
Job_table = None
log.debug( "Failed loading table job" )
if Job_table:
try:
col = Column( "library_folder_id", Integer, index=True )
col.create( Job_table )
assert col is Job_table.c.library_folder_id
except Exception, e:
log.debug( "Adding column 'library_folder_id' to job table failed: %s" % ( str( e ) ) )
try:
LibraryFolder_table = Table( "library_folder", metadata, autoload=True )
except NoSuchTableError:
LibraryFolder_table = None
log.debug( "Failed loading table library_folder" )
# Add 1 foreign key constraint to the job table
if Job_table and LibraryFolder_table:
try:
cons = ForeignKeyConstraint( [Job_table.c.library_folder_id],
[LibraryFolder_table.c.id],
name='job_library_folder_id_fk' )
# Create the constraint
cons.create()
except Exception, e:
log.debug( "Adding foreign key constraint 'job_library_folder_id_fk' to table 'library_folder' failed: %s" % ( str( e ) ) )
# Create the ix_dataset_state index
try:
Dataset_table = Table( "dataset", metadata, autoload=True )
except NoSuchTableError:
Dataset_table = None
log.debug( "Failed loading table dataset" )
i = Index( "ix_dataset_state", Dataset_table.c.state )
try:
i.create()
except Exception, e:
print str(e)
log.debug( "Adding index 'ix_dataset_state' to dataset table failed: %s" % str( e ) )
def downgrade():
metadata.reflect()
# Drop the library_folder_id column
try:
Job_table = Table( "job", metadata, autoload=True )
except NoSuchTableError:
Job_table = None
log.debug( "Failed loading table job" )
if Job_table:
try:
col = Job_table.c.library_folder_id
col.drop()
except Exception, e:
log.debug( "Dropping column 'library_folder_id' from job table failed: %s" % ( str( e ) ) )
# Drop the job_to_output_library_dataset table
try:
JobToOutputLibraryDatasetAssociation_table.drop()
except Exception, e:
print str(e)
log.debug( "Dropping job_to_output_library_dataset table failed: %s" % str( e ) )
# Drop the ix_dataset_state index
try:
Dataset_table = Table( "dataset", metadata, autoload=True )
except NoSuchTableError:
Dataset_table = None
log.debug( "Failed loading table dataset" )
i = Index( "ix_dataset_state", Dataset_table.c.state )
try:
i.drop()
except Exception, e:
print str(e)
log.debug( "Dropping index 'ix_dataset_state' from dataset table failed: %s" % str( e ) )
+7 -145
View File
@@ -1,126 +1,22 @@
import os, shutil, urllib, StringIO, re, gzip, tempfile, shutil, zipfile
from cgi import FieldStorage
import os
from __init__ import ToolAction
from galaxy import datatypes, jobs
from galaxy.datatypes import sniff
from galaxy import model, util
from galaxy.util.json import to_json_string
import sys, traceback
from galaxy.tools.actions import upload_common
import logging
log = logging.getLogger( __name__ )
class UploadToolAction( ToolAction ):
# Action for uploading files
def persist_uploads( self, incoming ):
if 'files' in incoming:
new_files = []
temp_files = []
for upload_dataset in incoming['files']:
f = upload_dataset['file_data']
if isinstance( f, FieldStorage ):
assert not isinstance( f.file, StringIO.StringIO )
assert f.file.name != '<fdopen>'
local_filename = util.mkstemp_ln( f.file.name, 'upload_file_data_' )
f.file.close()
upload_dataset['file_data'] = dict( filename = f.filename,
local_filename = local_filename )
if upload_dataset['url_paste'].strip() != '':
upload_dataset['url_paste'] = datatypes.sniff.stream_to_file( StringIO.StringIO( upload_dataset['url_paste'] ), prefix="strio_url_paste_" )[0]
else:
upload_dataset['url_paste'] = None
new_files.append( upload_dataset )
incoming['files'] = new_files
return incoming
def execute( self, tool, trans, incoming={}, set_output_hid = True ):
dataset_upload_inputs = []
for input_name, input in tool.inputs.iteritems():
if input.type == "upload_dataset":
dataset_upload_inputs.append( input )
assert dataset_upload_inputs, Exception( "No dataset upload groups were found." )
# Get any precreated datasets (when using asynchronous uploads)
async_datasets = []
self.precreated_datasets = []
if incoming.get( 'async_datasets', None ) not in ["None", "", None]:
async_datasets = incoming['async_datasets'].split(',')
for id in async_datasets:
try:
data = trans.app.model.HistoryDatasetAssociation.get( int( id ) )
except:
log.exception( 'Unable to load precreated dataset (%s) sent in upload form' % id )
continue
if trans.user is None and trans.galaxy_session.current_history != data.history:
log.error( 'Got a precreated dataset (%s) but it does not belong to anonymous user\'s current session (%s)' % ( data.id, trans.galaxy_session.id ) )
elif data.history.user != trans.user:
log.error( 'Got a precreated dataset (%s) but it does not belong to current user (%s)' % ( data.id, trans.user.id ) )
else:
self.precreated_datasets.append( data )
data_list = []
incoming = self.persist_uploads( incoming )
json_file = tempfile.mkstemp()
json_file_path = json_file[1]
json_file = os.fdopen( json_file[0], 'w' )
for dataset_upload_input in dataset_upload_inputs:
uploaded_datasets = dataset_upload_input.get_uploaded_datasets( trans, incoming )
for uploaded_dataset in uploaded_datasets:
data = self.get_precreated_dataset( uploaded_dataset.name )
if not data:
data = trans.app.model.HistoryDatasetAssociation( history = trans.history, create_dataset = True )
data.name = uploaded_dataset.name
data.state = data.states.QUEUED
data.extension = uploaded_dataset.file_type
data.dbkey = uploaded_dataset.dbkey
data.flush()
trans.history.add_dataset( data, genome_build = uploaded_dataset.dbkey )
permissions = trans.app.security_agent.history_get_default_permissions( trans.history )
trans.app.security_agent.set_all_dataset_permissions( data.dataset, permissions )
else:
data.extension = uploaded_dataset.file_type
data.dbkey = uploaded_dataset.dbkey
data.flush()
trans.history.genome_build = uploaded_dataset.dbkey
if uploaded_dataset.type == 'composite':
# we need to init metadata before the job is dispatched
data.init_meta()
for meta_name, meta_value in uploaded_dataset.metadata.iteritems():
setattr( data.metadata, meta_name, meta_value )
data.flush()
json = dict( file_type = uploaded_dataset.file_type,
dataset_id = data.dataset.id,
dbkey = uploaded_dataset.dbkey,
type = uploaded_dataset.type,
metadata = uploaded_dataset.metadata,
primary_file = uploaded_dataset.primary_file,
extra_files_path = data.extra_files_path,
composite_file_paths = uploaded_dataset.composite_files,
composite_files = dict( [ ( k, v.__dict__ ) for k, v in data.datatype.get_composite_files( data ).items() ] ) )
else:
try:
is_binary = uploaded_dataset.datatype.is_binary
except:
is_binary = None
json = dict( file_type = uploaded_dataset.file_type,
ext = uploaded_dataset.ext,
name = uploaded_dataset.name,
dataset_id = data.dataset.id,
dbkey = uploaded_dataset.dbkey,
type = uploaded_dataset.type,
is_binary = is_binary,
space_to_tab = uploaded_dataset.space_to_tab,
path = uploaded_dataset.path )
json_file.write( to_json_string( json ) + '\n' )
data_list.append( data )
json_file.close()
#cleanup unclaimed precreated datasets:
for data in self.precreated_datasets:
log.info( 'Cleaned up unclaimed precreated dataset (%s).' % ( data.id ) )
data.state = data.states.ERROR
data.info = 'No file contents were available.'
precreated_datasets = upload_common.get_precreated_datasets( trans, incoming, trans.app.model.HistoryDatasetAssociation )
incoming = upload_common.persist_uploads( incoming )
json_file_path, data_list = upload_common.create_paramfile( trans, incoming, precreated_datasets, dataset_upload_inputs )
upload_common.cleanup_unused_precreated_datasets( precreated_datasets )
if not data_list:
try:
@@ -129,38 +25,4 @@ class UploadToolAction( ToolAction ):
pass
return 'No data was entered in the upload form, please go back and choose data to upload.'
# Create the job object
job = trans.app.model.Job()
job.session_id = trans.get_galaxy_session().id
job.history_id = trans.history.id
job.tool_id = tool.id
job.tool_version = tool.version
job.state = trans.app.model.Job.states.UPLOAD
job.flush()
log.info( 'tool %s created job id %d' % ( tool.id, job.id ) )
trans.log_event( 'created job id %d' % job.id, tool_id=tool.id )
for name, value in tool.params_to_strings( incoming, trans.app ).iteritems():
job.add_parameter( name, value )
job.add_parameter( 'paramfile', to_json_string( json_file_path ) )
for i, dataset in enumerate( data_list ):
job.add_output_dataset( 'output%i' % i, dataset )
job.state = trans.app.model.Job.states.NEW
trans.app.model.flush()
# Queue the job for execution
trans.app.job_queue.put( job.id, tool )
trans.log_event( "Added job to the job queue, id: %s" % str(job.id), tool_id=job.tool_id )
return dict( [ ( i, v ) for i, v in enumerate( data_list ) ] )
def get_precreated_dataset( self, name ):
"""
Return a dataset matching a name from the list of precreated (via async
upload) datasets. If there's more than one upload with the exact same
name, we need to pop one (the first) so it isn't chosen next time.
"""
names = [ d.name for d in self.precreated_datasets ]
if names.count( name ) > 0:
return self.precreated_datasets.pop( names.index( name ) )
else:
return None
return upload_common.create_job( trans, incoming, tool, json_file_path, data_list )
+235
View File
@@ -0,0 +1,235 @@
import os, tempfile, StringIO
from cgi import FieldStorage
from galaxy import datatypes, util
from galaxy.datatypes import sniff
from galaxy.util.json import to_json_string
import logging
log = logging.getLogger( __name__ )
def persist_uploads( params ):
"""
Turn any uploads in the submitted form to persisted files.
"""
if 'files' in params:
new_files = []
temp_files = []
for upload_dataset in params['files']:
f = upload_dataset['file_data']
if isinstance( f, FieldStorage ):
assert not isinstance( f.file, StringIO.StringIO )
assert f.file.name != '<fdopen>'
local_filename = util.mkstemp_ln( f.file.name, 'upload_file_data_' )
f.file.close()
upload_dataset['file_data'] = dict( filename = f.filename,
local_filename = local_filename )
if upload_dataset['url_paste'].strip() != '':
upload_dataset['url_paste'] = datatypes.sniff.stream_to_file( StringIO.StringIO( upload_dataset['url_paste'] ), prefix="strio_url_paste_" )[0]
else:
upload_dataset['url_paste'] = None
new_files.append( upload_dataset )
params['files'] = new_files
return params
def get_precreated_datasets( trans, params, data_obj ):
"""
Get any precreated datasets (when using asynchronous uploads).
"""
rval = []
async_datasets = []
if params.get( 'async_datasets', None ) not in ["None", "", None]:
async_datasets = params['async_datasets'].split(',')
user, roles = trans.get_user_and_roles()
for id in async_datasets:
try:
data = data_obj.get( int( id ) )
except:
log.exception( 'Unable to load precreated dataset (%s) sent in upload form' % id )
continue
if data_obj is trans.app.model.HistoryDatasetAssociation:
if user is None and trans.galaxy_session.current_history != data.history:
log.error( 'Got a precreated dataset (%s) but it does not belong to anonymous user\'s current session (%s)' % ( data.id, trans.galaxy_session.id ) )
elif data.history.user != user:
log.error( 'Got a precreated dataset (%s) but it does not belong to current user (%s)' % ( data.id, user.id ) )
else:
rval.append( data )
elif data_obj is trans.app.model.LibraryDatasetDatasetAssociation:
if not trans.app.security_agent.can_add_library_item( user, roles, data.library_dataset.folder ):
log.error( 'Got a precreated dataset (%s) but this user (%s) is not allowed to write to it' % ( data.id, user.id ) )
else:
rval.append( data )
return rval
def get_precreated_dataset( precreated_datasets, name ):
"""
Return a dataset matching a name from the list of precreated (via async
upload) datasets. If there's more than one upload with the exact same
name, we need to pop one (the first) so it isn't chosen next time.
"""
names = [ d.name for d in precreated_datasets ]
if names.count( name ) > 0:
return precreated_datasets.pop( names.index( name ) )
else:
return None
def cleanup_unused_precreated_datasets( precreated_datasets ):
for data in precreated_datasets:
log.info( 'Cleaned up unclaimed precreated dataset (%s).' % ( data.id ) )
data.state = data.states.ERROR
data.info = 'No file contents were available.'
def new_history_upload( trans, uploaded_dataset ):
hda = trans.app.model.HistoryDatasetAssociation( name = uploaded_dataset.name,
extension = uploaded_dataset.file_type,
dbkey = uploaded_dataset.dbkey,
history = trans.history,
create_dataset = True )
hda.state = hda.states.QUEUED
hda.flush()
trans.history.add_dataset( hda, genome_build = uploaded_dataset.dbkey )
permissions = trans.app.security_agent.history_get_default_permissions( trans.history )
trans.app.security_agent.set_all_dataset_permissions( hda.dataset, permissions )
return hda
def new_library_upload( trans, uploaded_dataset, replace_dataset, folder,
template, template_field_contents, roles, message ):
if replace_dataset:
ld = replace_dataset
else:
ld = trans.app.model.LibraryDataset( folder=folder, name=uploaded_dataset.name )
ld.flush()
trans.app.security_agent.copy_library_permissions( folder, ld )
ldda = trans.app.model.LibraryDatasetDatasetAssociation( name = uploaded_dataset.name,
extension = uploaded_dataset.file_type,
dbkey = uploaded_dataset.dbkey,
library_dataset = ld,
user = trans.user,
create_dataset = True )
ldda.state = ldda.states.QUEUED
ldda.message = message
ldda.flush()
# Permissions must be the same on the LibraryDatasetDatasetAssociation and the associated LibraryDataset
trans.app.security_agent.copy_library_permissions( ld, ldda )
if replace_dataset:
# Copy the Dataset level permissions from replace_dataset to the new LibraryDatasetDatasetAssociation.dataset
trans.app.security_agent.copy_dataset_permissions( replace_dataset.library_dataset_dataset_association.dataset, ldda.dataset )
else:
# Copy the current user's DefaultUserPermissions to the new LibraryDatasetDatasetAssociation.dataset
trans.app.security_agent.set_all_dataset_permissions( ldda.dataset, trans.app.security_agent.user_get_default_permissions( trans.user ) )
folder.add_library_dataset( ld, genome_build=uploaded_dataset.dbkey )
folder.flush()
ld.library_dataset_dataset_association_id = ldda.id
ld.flush()
# Handle template included in the upload form, if any
if template and template_field_contents:
# Since information templates are inherited, the template fields can be displayed on the upload form.
# If the user has added field contents, we'll need to create a new form_values and info_association
# for the new library_dataset_dataset_association object.
# Create a new FormValues object, using the template we previously retrieved
form_values = trans.app.model.FormValues( template, template_field_contents )
form_values.flush()
# Create a new info_association between the current ldda and form_values
info_association = trans.app.model.LibraryDatasetDatasetInfoAssociation( ldda, template, form_values )
info_association.flush()
# If roles were selected upon upload, restrict access to the Dataset to those roles
if roles:
for role in roles:
dp = trans.app.model.DatasetPermissions( trans.app.security_agent.permitted_actions.DATASET_ACCESS.action, ldda.dataset, role )
dp.flush()
return ldda
def create_paramfile( trans, params, precreated_datasets, dataset_upload_inputs,
replace_dataset=None, folder=None, template=None,
template_field_contents=None, roles=None, message=None ):
"""
Create the upload tool's JSON "param" file.
"""
data_list = []
json_file = tempfile.mkstemp()
json_file_path = json_file[1]
json_file = os.fdopen( json_file[0], 'w' )
for dataset_upload_input in dataset_upload_inputs:
uploaded_datasets = dataset_upload_input.get_uploaded_datasets( trans, params )
for uploaded_dataset in uploaded_datasets:
data = get_precreated_dataset( precreated_datasets, uploaded_dataset.name )
if not data:
if folder:
data = new_library_upload( trans, uploaded_dataset, replace_dataset, folder, template, template_field_contents, roles, message )
else:
data = new_history_upload( trans, uploaded_dataset )
else:
data.extension = uploaded_dataset.file_type
data.dbkey = uploaded_dataset.dbkey
data.flush()
if folder:
folder.genome_build = uploaded_dataset.dbkey
folder.flush()
else:
trans.history.genome_build = uploaded_dataset.dbkey
if uploaded_dataset.type == 'composite':
# we need to init metadata before the job is dispatched
data.init_meta()
for meta_name, meta_value in uploaded_dataset.metadata.iteritems():
setattr( data.metadata, meta_name, meta_value )
data.flush()
json = dict( file_type = uploaded_dataset.file_type,
dataset_id = data.dataset.id,
dbkey = uploaded_dataset.dbkey,
type = uploaded_dataset.type,
metadata = uploaded_dataset.metadata,
primary_file = uploaded_dataset.primary_file,
extra_files_path = data.extra_files_path,
composite_file_paths = uploaded_dataset.composite_files,
composite_files = dict( [ ( k, v.__dict__ ) for k, v in data.datatype.get_composite_files( data ).items() ] ) )
else:
try:
is_binary = uploaded_dataset.datatype.is_binary
except:
is_binary = None
json = dict( file_type = uploaded_dataset.file_type,
ext = uploaded_dataset.ext,
name = uploaded_dataset.name,
dataset_id = data.dataset.id,
dbkey = uploaded_dataset.dbkey,
type = uploaded_dataset.type,
is_binary = is_binary,
space_to_tab = uploaded_dataset.space_to_tab,
path = uploaded_dataset.path )
json_file.write( to_json_string( json ) + '\n' )
data_list.append( data )
json_file.close()
return ( json_file_path, data_list )
def create_job( trans, params, tool, json_file_path, data_list, folder=None ):
"""
Create the upload job.
"""
job = trans.app.model.Job()
job.session_id = trans.get_galaxy_session().id
if folder:
job.library_folder_id = folder.id
else:
job.history_id = trans.history.id
job.tool_id = tool.id
job.tool_version = tool.version
job.state = job.states.UPLOAD
job.flush()
log.info( 'tool %s created job id %d' % ( tool.id, job.id ) )
trans.log_event( 'created job id %d' % job.id, tool_id=tool.id )
for name, value in tool.params_to_strings( params, trans.app ).iteritems():
job.add_parameter( name, value )
job.add_parameter( 'paramfile', to_json_string( json_file_path ) )
if folder:
for i, dataset in enumerate( data_list ):
job.add_output_library_dataset( 'output%i' % i, dataset )
else:
for i, dataset in enumerate( data_list ):
job.add_output_dataset( 'output%i' % i, dataset )
job.state = job.states.NEW
trans.app.model.flush()
# Queue the job for execution
trans.app.job_queue.put( job.id, tool )
trans.log_event( "Added job to the job queue, id: %s" % str(job.id), tool_id=job.tool_id )
return dict( [ ( 'output%i' % i, v ) for i, v in enumerate( data_list ) ] )
+13 -13
View File
@@ -726,17 +726,17 @@ class Library( BaseController ):
template_id = 'None'
widgets = []
upload_option = params.get( 'upload_option', 'upload_file' )
created_ldda_ids = trans.webapp.controllers[ 'library_dataset' ].upload_dataset( trans,
controller='library',
library_id=library_id,
folder_id=folder_id,
template_id=template_id,
widgets=widgets,
replace_dataset=replace_dataset,
**kwd )
if created_ldda_ids:
ldda_id_list = created_ldda_ids.split( ',' )
total_added = len( ldda_id_list )
created_outputs = trans.webapp.controllers[ 'library_dataset' ].upload_dataset( trans,
controller='library',
library_id=library_id,
folder_id=folder_id,
template_id=template_id,
widgets=widgets,
replace_dataset=replace_dataset,
**kwd )
if created_outputs:
ldda_id_list = [ str( v.id ) for v in created_outputs.values() ]
total_added = len( created_outputs.values() )
if replace_dataset:
msg = "Added %d dataset versions to the library dataset '%s' in the folder '%s'." % ( total_added, replace_dataset.name, folder.name )
else:
@@ -760,7 +760,7 @@ class Library( BaseController ):
action='browse_library',
id=library_id,
default_action=default_action,
created_ldda_ids=created_ldda_ids,
created_ldda_ids=",".join( ldda_id_list ),
msg=util.sanitize_text( msg ),
messagetype='done' ) )
@@ -769,7 +769,7 @@ class Library( BaseController ):
trans.response.send_redirect( web.url_for( controller='library',
action='browse_library',
id=library_id,
created_ldda_ids=created_ldda_ids,
created_ldda_ids=",".join( ldda_id_list ),
msg=util.sanitize_text( msg ),
messagetype='error' ) )
if not id or replace_dataset:
+11 -11
View File
@@ -438,16 +438,16 @@ class LibraryAdmin( BaseController ):
template_id = 'None'
widgets = []
upload_option = params.get( 'upload_option', 'upload_file' )
created_ldda_ids = trans.webapp.controllers[ 'library_dataset' ].upload_dataset( trans,
controller='library_admin',
library_id=library_id,
folder_id=folder_id,
template_id=template_id,
widgets=widgets,
replace_dataset=replace_dataset,
**kwd )
if created_ldda_ids:
total_added = len( created_ldda_ids.split( ',' ) )
created_outputs = trans.webapp.controllers[ 'library_dataset' ].upload_dataset( trans,
controller='library_admin',
library_id=library_id,
folder_id=folder_id,
template_id=template_id,
widgets=widgets,
replace_dataset=replace_dataset,
**kwd )
if created_outputs:
total_added = len( created_outputs.values() )
if replace_dataset:
msg = "Added %d dataset versions to the library dataset '%s' in the folder '%s'." % ( total_added, replace_dataset.name, folder.name )
else:
@@ -464,7 +464,7 @@ class LibraryAdmin( BaseController ):
trans.response.send_redirect( web.url_for( controller='library_admin',
action='browse_library',
id=library_id,
created_ldda_ids=created_ldda_ids,
created_ldda_ids=",".join( [ str( v.id ) for v in created_outputs.values() ] ),
msg=util.sanitize_text( msg ),
messagetype=messagetype ) )
elif not id or replace_dataset:
+107 -410
View File
@@ -3,196 +3,51 @@ from galaxy.web.base.controller import *
from galaxy import util, jobs
from galaxy.datatypes import sniff
from galaxy.security import RBACAgent
from galaxy.util.json import to_json_string
from galaxy.tools.actions import upload_common
log = logging.getLogger( __name__ )
class UploadLibraryDataset( BaseController ):
def remove_tempfile( self, filename ):
try:
os.unlink( filename )
except:
log.exception( 'failure removing temporary file: %s' % filename )
def add_file( self, trans, folder, file_obj, name, file_type, dbkey, roles,
info='no info', space_to_tab=False, replace_dataset=None,
template=None, template_field_contents=[], message=None ):
data_type = None
line_count = 0
temp_name, is_multi_byte = sniff.stream_to_file( file_obj )
# See if we have an empty file
if not os.path.getsize( temp_name ) > 0:
raise BadFileException( "you attempted to upload an empty file." )
if is_multi_byte:
ext = sniff.guess_ext( temp_name, is_multi_byte=True )
else:
if not data_type:
# See if we have a gzipped file, which, if it passes our restrictions, we'll uncompress on the fly.
is_gzipped, is_valid = self.check_gzip( temp_name )
if is_gzipped and not is_valid:
raise BadFileException( "you attempted to upload an inappropriate file." )
elif is_gzipped and is_valid:
# We need to uncompress the temp_name file
CHUNK_SIZE = 2**20 # 1Mb
fd, uncompressed = tempfile.mkstemp()
gzipped_file = gzip.GzipFile( temp_name )
while 1:
try:
chunk = gzipped_file.read( CHUNK_SIZE )
except IOError:
os.close( fd )
os.remove( uncompressed )
raise BadFileException( 'problem uncompressing gzipped data.' )
if not chunk:
break
os.write( fd, chunk )
os.close( fd )
gzipped_file.close()
# Replace the gzipped file with the decompressed file
shutil.move( uncompressed, temp_name )
name = name.rstrip( '.gz' )
data_type = 'gzip'
ext = ''
if not data_type:
# See if we have a zip archive
is_zipped, is_valid, test_ext = self.check_zip( temp_name )
if is_zipped and not is_valid:
raise BadFileException( "you attempted to upload an inappropriate file." )
elif is_zipped and is_valid:
# Currently, we force specific tools to handle this case. We also require the user
# to manually set the incoming file_type
if ( test_ext == 'ab1' or test_ext == 'scf' ) and file_type != 'binseq.zip':
raise BadFileException( "Invalid 'File Format' for archive consisting of binary files - use 'Binseq.zip'." )
elif test_ext == 'txt' and file_type != 'txtseq.zip':
raise BadFileException( "Invalid 'File Format' for archive consisting of text files - use 'Txtseq.zip'." )
if not ( file_type == 'binseq.zip' or file_type == 'txtseq.zip' ):
raise BadFileException( "you must manually set the 'File Format' to either 'Binseq.zip' or 'Txtseq.zip' when uploading zip files." )
data_type = 'zip'
ext = file_type
if not data_type:
if self.check_binary( temp_name ):
try:
ext = name.split( "." )[1].strip().lower()
except:
ext = ''
try:
is_pdf = open( temp_name ).read( len( '%PDF' ) ) == '%PDF'
except:
is_pdf = False #file failed to open or contents are smaller than pdf header
if is_pdf:
file_type = 'pdf' #allow the upload of PDFs to library via the admin interface.
else:
if not( ext == 'ab1' or ext == 'scf' ):
raise BadFileException( "you attempted to upload an inappropriate file." )
if ext == 'ab1' and file_type != 'ab1':
raise BadFileException( "you must manually set the 'File Format' to 'Ab1' when uploading ab1 files." )
elif ext == 'scf' and file_type != 'scf':
raise BadFileException( "you must manually set the 'File Format' to 'Scf' when uploading scf files." )
data_type = 'binary'
if not data_type:
# We must have a text file
if self.check_html( temp_name ):
raise BadFileException( "you attempted to upload an inappropriate file." )
if data_type != 'binary' and data_type != 'zip':
if space_to_tab:
line_count = sniff.convert_newlines_sep2tabs( temp_name )
elif os.stat( temp_name ).st_size < 262144000: # 250MB
line_count = sniff.convert_newlines( temp_name )
else:
if sniff.check_newlines( temp_name ):
line_count = sniff.convert_newlines( temp_name )
else:
line_count = None
if file_type == 'auto':
ext = sniff.guess_ext( temp_name, sniff_order=trans.app.datatypes_registry.sniff_order )
else:
ext = file_type
data_type = ext
if info is None:
info = 'uploaded %s file' % data_type
if file_type == 'auto':
data_type = sniff.guess_ext( temp_name, sniff_order=trans.app.datatypes_registry.sniff_order )
else:
data_type = file_type
if replace_dataset:
# The replace_dataset param ( when not None ) refers to a LibraryDataset that is being replaced with a new version.
library_dataset = replace_dataset
else:
# If replace_dataset is None, the Library level permissions will be taken from the folder and applied to the new
# LibraryDataset, and the current user's DefaultUserPermissions will be applied to the associated Dataset.
library_dataset = trans.app.model.LibraryDataset( folder=folder, name=name, info=info )
library_dataset.flush()
trans.app.security_agent.copy_library_permissions( folder, library_dataset )
ldda = trans.app.model.LibraryDatasetDatasetAssociation( name=name,
info=info,
extension=data_type,
dbkey=dbkey,
library_dataset=library_dataset,
user=trans.get_user(),
create_dataset=True )
ldda.message = message
ldda.flush()
# Permissions must be the same on the LibraryDatasetDatasetAssociation and the associated LibraryDataset
trans.app.security_agent.copy_library_permissions( library_dataset, ldda )
if replace_dataset:
# Copy the Dataset level permissions from replace_dataset to the new LibraryDatasetDatasetAssociation.dataset
trans.app.security_agent.copy_dataset_permissions( replace_dataset.library_dataset_dataset_association.dataset, ldda.dataset )
else:
# Copy the current user's DefaultUserPermissions to the new LibraryDatasetDatasetAssociation.dataset
trans.app.security_agent.set_all_dataset_permissions( ldda.dataset, trans.app.security_agent.user_get_default_permissions( trans.get_user() ) )
folder.add_library_dataset( library_dataset, genome_build=dbkey )
folder.flush()
library_dataset.library_dataset_dataset_association_id = ldda.id
library_dataset.flush()
# Handle template included in the upload form, if any
if template and template_field_contents:
# Since information templates are inherited, the template fields can be displayed on the upload form.
# If the user has added field contents, we'll need to create a new form_values and info_association
# for the new library_dataset_dataset_association object.
# Create a new FormValues object, using the template we previously retrieved
form_values = trans.app.model.FormValues( template, template_field_contents )
form_values.flush()
# Create a new info_association between the current ldda and form_values
info_association = trans.app.model.LibraryDatasetDatasetInfoAssociation( ldda, template, form_values )
info_association.flush()
# If roles were selected upon upload, restrict access to the Dataset to those roles
if roles:
for role in roles:
dp = trans.app.model.DatasetPermissions( RBACAgent.permitted_actions.DATASET_ACCESS.action, ldda.dataset, role )
dp.flush()
shutil.move( temp_name, ldda.dataset.file_name )
ldda.state = ldda.states.OK
ldda.init_meta()
if line_count:
try:
if is_multi_byte:
ldda.set_multi_byte_peek( line_count=line_count )
else:
ldda.set_peek( line_count=line_count )
except:
if is_multi_byte:
ldda.set_multi_byte_peek()
else:
ldda.set_peek()
else:
if is_multi_byte:
ldda.set_multi_byte_peek()
else:
ldda.set_peek()
ldda.set_size()
if ldda.missing_meta():
ldda.datatype.set_meta( ldda )
ldda.flush()
return ldda
@web.json
def library_item_updates( self, trans, ids=None, states=None ):
# Avoid caching
trans.response.headers['Pragma'] = 'no-cache'
trans.response.headers['Expires'] = '0'
# Create new HTML for any that have changed
rval = {}
if ids is not None and states is not None:
ids = map( int, ids.split( "," ) )
states = states.split( "," )
for id, state in zip( ids, states ):
data = self.app.model.LibraryDatasetDatasetAssociation.get( id )
if data.state != state:
job_ldda = data
while job_ldda.copied_from_library_dataset_dataset_association:
job_ldda = job_ldda.copied_from_library_dataset_dataset_association
force_history_refresh = False
rval[id] = {
"state": data.state,
"html": unicode( trans.fill_template( "library/library_item_info.mako", ldda=data ), 'utf-8' )
#"force_history_refresh": force_history_refresh
}
return rval
@web.expose
def upload_dataset( self, trans, controller, library_id, folder_id, replace_dataset=None, **kwd ):
# This method is called from both the admin and library controllers. The replace_dataset param ( when
# not None ) refers to a LibraryDataset that is being replaced with a new version.
params = util.Params( kwd )
# Set up the traditional tool state/params
tool_id = 'upload1'
tool = trans.app.toolbox.tools_by_id[ tool_id ]
state = tool.new_state( trans )
errors = tool.update_state( trans, tool.inputs_by_page[0], state.inputs, kwd, changed_dependencies={} )
tool_params = state.inputs
dataset_upload_inputs = []
for input_name, input in tool.inputs.iteritems():
if input.type == "upload_dataset":
dataset_upload_inputs.append( input )
# Library-specific params
params = util.Params( kwd ) # is this filetoolparam safe?
msg = util.restore_text( params.get( 'msg', '' ) )
messagetype = params.get( 'messagetype', 'done' )
dbkey = params.get( 'dbkey', '?' )
file_type = params.get( 'file_type', 'auto' )
data_file = params.get( 'files_0|file_data', '' )
url_paste = params.get( 'files_0|url_paste', '' )
server_dir = util.restore_text( params.get( 'server_dir', '' ) )
if replace_dataset not in [ None, 'None' ]:
replace_id = replace_dataset.id
@@ -217,24 +72,43 @@ class UploadLibraryDataset( BaseController ):
template_field_contents.append( field_value )
else:
template = None
if upload_option == 'upload_file' and data_file == '' and url_paste == '':
msg = 'Select a file, enter a URL or enter text'
err_redirect = True
elif upload_option == 'upload_directory':
if upload_option == 'upload_directory':
if server_dir in [ None, 'None', '' ]:
err_redirect = True
# See if our request is from the Admin view or the Libraries view
if trans.request.browser_url.find( 'admin' ) >= 0:
if controller == 'library_admin':
import_dir = trans.app.config.library_import_dir
import_dir_desc = 'library_import_dir'
full_dir = os.path.join( import_dir, server_dir )
else:
import_dir = trans.app.config.user_library_import_dir
import_dir_desc = 'user_library_import_dir'
if server_dir == trans.user.email:
full_dir = os.path.join( import_dir, server_dir )
else:
full_dir = os.path.join( import_dir, trans.user.email, server_dir )
if import_dir:
msg = 'Select a directory'
else:
msg = '"%s" is not defined in the Galaxy configuration file' % import_dir_desc
roles = []
for role_id in util.listify( params.get( 'roles', [] ) ):
roles.append( trans.app.model.Role.get( role_id ) )
# Proceed with (mostly) regular upload processing
precreated_datasets = upload_common.get_precreated_datasets( trans, tool_params, trans.app.model.HistoryDatasetAssociation )
if upload_option == 'upload_file':
tool_params = upload_common.persist_uploads( tool_params )
json_file_path, data_list = upload_common.create_paramfile( trans, tool_params, precreated_datasets, dataset_upload_inputs, replace_dataset, folder, template, template_field_contents, roles, message )
elif upload_option == 'upload_directory':
json_file_path, data_list = self.create_server_dir_paramfile( trans, params, full_dir, import_dir_desc, folder, template, template_field_contents, roles, message, err_redirect, msg )
upload_common.cleanup_unused_precreated_datasets( precreated_datasets )
if upload_option == 'upload_file' and not data_list:
msg = 'Select a file, enter a URL or enter text'
err_redirect = True
if err_redirect:
try:
os.remove( json_file_path )
except:
pass
trans.response.send_redirect( web.url_for( controller=controller,
action='library_dataset_dataset_association',
library_id=library_id,
@@ -243,226 +117,49 @@ class UploadLibraryDataset( BaseController ):
upload_option=upload_option,
msg=util.sanitize_text( msg ),
messagetype='error' ) )
space_to_tab = params.get( 'files_0|space_to_tab', False )
if space_to_tab and space_to_tab not in [ "None", None ]:
space_to_tab = True
roles = []
for role_id in util.listify( params.get( 'roles', [] ) ):
roles.append( trans.app.model.Role.get( role_id ) )
return upload_common.create_job( trans, tool_params, tool, json_file_path, data_list, folder=folder )
def create_server_dir_paramfile( self, trans, params, full_dir, import_dir_desc, folder, template,
template_field_contents, roles, message, err_redirect, msg ):
"""
Create JSON param file for the upload tool when using the server_dir upload.
"""
files = []
try:
for entry in os.listdir( full_dir ):
# Only import regular files
if os.path.isfile( os.path.join( full_dir, entry ) ):
files.append( entry )
except Exception, e:
msg = "Unable to get file list for configured %s, error: %s" % ( import_dir_desc, str( e ) )
err_redirect = True
return ( None, None )
if not files:
msg = "The directory '%s' contains no valid files" % full_dir
err_redirect = True
return ( None, None )
data_list = []
created_ldda_ids = ''
if 'filename' in dir( data_file ):
file_name = data_file.filename
file_name = file_name.split( '\\' )[-1]
file_name = file_name.split( '/' )[-1]
try:
created_ldda = self.add_file( trans,
folder,
data_file.file,
file_name,
file_type,
dbkey,
roles,
info="uploaded file",
space_to_tab=space_to_tab,
replace_dataset=replace_dataset,
template=template,
template_field_contents=template_field_contents,
message=message )
created_ldda_ids = str( created_ldda.id )
except Exception, e:
log.exception( 'exception in upload_dataset using file_name %s: %s' % ( str( file_name ), str( e ) ) )
return self.upload_empty( trans, controller, library_id, folder_id, "Error:", str( e ) )
elif url_paste not in [ None, "" ]:
if url_paste.lower().find( 'http://' ) >= 0 or url_paste.lower().find( 'ftp://' ) >= 0:
url_paste = url_paste.replace( '\r', '' ).split( '\n' )
# If we are setting the name from the line, it needs to be the line that creates that dataset
name_set_from_line = False
for line in url_paste:
line = line.rstrip( '\r\n' )
if line:
if not line or name_set_from_line:
name_set_from_line = True
try:
created_ldda = self.add_file( trans,
folder,
urllib.urlopen( line ),
line,
file_type,
dbkey,
roles,
info="uploaded url",
space_to_tab=space_to_tab,
replace_dataset=replace_dataset,
template=template,
template_field_contents=template_field_contents,
message=message )
created_ldda_ids = '%s,%s' % ( created_ldda_ids, str( created_ldda.id ) )
except Exception, e:
log.exception( 'exception in upload_dataset using url_paste %s' % str( e ) )
return self.upload_empty( trans, controller, library_id, folder_id, "Error:", str( e ) )
else:
is_valid = False
for line in url_paste:
line = line.rstrip( '\r\n' )
if line:
is_valid = True
break
if is_valid:
try:
created_ldda = self.add_file( trans,
folder,
StringIO.StringIO( url_paste ),
'Pasted Entry',
file_type,
dbkey,
roles,
info="pasted entry",
space_to_tab=space_to_tab,
replace_dataset=replace_dataset,
template=template,
template_field_contents=template_field_contents,
message=message )
created_ldda_ids = '%s,%s' % ( created_ldda_ids, str( created_ldda.id ) )
except Exception, e:
log.exception( 'exception in add_file using StringIO.StringIO( url_paste ) %s' % str( e ) )
return self.upload_empty( trans, controller, library_id, folder_id, "Error:", str( e ) )
elif server_dir not in [ None, "", "None" ]:
# See if our request is from the Admin view or the Libraries view
if trans.request.browser_url.find( 'admin' ) >= 0:
import_dir = trans.app.config.library_import_dir
import_dir_desc = 'library_import_dir'
full_dir = os.path.join( import_dir, server_dir )
else:
imrport_dir = trans.app.config.user_library_import_dir
import_dir_desc = 'user_library_import_dir'
# From the Libraries view, users are restricted to the directory named the same as
# their email within the configured user_library_import_dir. If this directory contains
# sub-directories, server_dir will be the name of the selected sub-directory. Otherwise
# server_dir will be the user's email address.
if server_dir == trans.user.email:
full_dir = os.path.join( import_dir, server_dir )
else:
full_dir = os.path.join( import_dir, trans.user.email, server_dir )
files = []
try:
for entry in os.listdir( full_dir ):
# Only import regular files
if os.path.isfile( os.path.join( full_dir, entry ) ):
files.append( entry )
except Exception, e:
msg = "Unable to get file list for configured %s, error: %s" % ( import_dir_desc, str( e ) )
return self.upload_empty( trans, controller, library_id, folder_id, "Error:", msg )
if not files:
msg = "The directory '%s' contains no valid files" % full_dir
return self.upload_empty( trans, controller, library_id, folder_id, "Error:", msg )
for file in files:
full_file = os.path.join( full_dir, file )
if not os.path.isfile( full_file ):
continue
try:
created_ldda = self.add_file( trans,
folder,
open( full_file, 'rb' ),
file,
file_type,
dbkey,
roles,
info="imported file",
space_to_tab=space_to_tab,
replace_dataset=replace_dataset,
template=template,
template_field_contents=template_field_contents,
message=message )
created_ldda_ids = '%s,%s' % ( created_ldda_ids, str( created_ldda.id ) )
except Exception, e:
log.exception( 'exception in add_file using server_dir %s' % str( e ) )
return self.upload_empty( trans, controller, library_id, folder_id, "Error:", str( e ) )
if created_ldda_ids:
created_ldda_ids = created_ldda_ids.lstrip( ',' )
return created_ldda_ids
else:
return ''
def check_gzip( self, temp_name ):
temp = open( temp_name, "U" )
magic_check = temp.read( 2 )
temp.close()
if magic_check != util.gzip_magic:
return ( False, False )
CHUNK_SIZE = 2**15 # 32Kb
gzipped_file = gzip.GzipFile( temp_name )
chunk = gzipped_file.read( CHUNK_SIZE )
gzipped_file.close()
if self.check_html( temp_name, chunk=chunk ) or self.check_binary( temp_name, chunk=chunk ):
return( True, False )
return ( True, True )
def check_zip( self, temp_name ):
if not zipfile.is_zipfile( temp_name ):
return ( False, False, None )
zip_file = zipfile.ZipFile( temp_name, "r" )
# Make sure the archive consists of valid files. The current rules are:
# 1. Archives can only include .ab1, .scf or .txt files
# 2. All file file_types within an archive must be the same
name = zip_file.namelist()[0]
test_ext = name.split( "." )[1].strip().lower()
if not ( test_ext == 'scf' or test_ext == 'ab1' or test_ext == 'txt' ):
return ( True, False, test_ext )
for name in zip_file.namelist():
ext = name.split( "." )[1].strip().lower()
if ext != test_ext:
return ( True, False, test_ext )
return ( True, True, test_ext )
def check_html( self, temp_name, chunk=None ):
if chunk is None:
temp = open(temp_name, "U")
else:
temp = chunk
regexp1 = re.compile( "<A\s+[^>]*HREF[^>]+>", re.I )
regexp2 = re.compile( "<IFRAME[^>]*>", re.I )
regexp3 = re.compile( "<FRAMESET[^>]*>", re.I )
regexp4 = re.compile( "<META[^>]*>", re.I )
lineno = 0
for line in temp:
lineno += 1
matches = regexp1.search( line ) or regexp2.search( line ) or regexp3.search( line ) or regexp4.search( line )
if matches:
if chunk is None:
temp.close()
return True
if lineno > 100:
break
if chunk is None:
temp.close()
return False
def check_binary( self, temp_name, chunk=None ):
if chunk is None:
temp = open( temp_name, "U" )
else:
temp = chunk
lineno = 0
for line in temp:
lineno += 1
line = line.strip()
if line:
if util.is_multi_byte( line ):
return False
for char in line:
if ord( char ) > 128:
if chunk is None:
temp.close()
return True
if lineno > 10:
break
if chunk is None:
temp.close()
return False
def upload_empty( self, trans, controller, library_id, folder_id, err_code, err_msg ):
msg = err_code + err_msg
return trans.response.send_redirect( web.url_for( controller=controller,
action='library_dataset_dataset_association',
library_id=library_id,
folder_id=folder_id,
msg=util.sanitize_text( msg ),
messagetype='error' ) )
class BadFileException( Exception ):
pass
json_file = tempfile.mkstemp()
json_file_path = json_file[1]
json_file = os.fdopen( json_file[0], 'w' )
for file in files:
full_file = os.path.join( full_dir, file )
if not os.path.isfile( full_file ):
continue
uploaded_dataset = util.bunch.Bunch()
uploaded_dataset.name = file
uploaded_dataset.file_type = params.file_type
uploaded_dataset.dbkey = params.dbkey
data = upload_common.new_library_upload( trans, uploaded_dataset, None, folder, template, template_field_contents, roles, message )
json = dict( file_type = uploaded_dataset.file_type,
ext = None,
name = uploaded_dataset.name,
dataset_id = data.dataset.id,
dbkey = uploaded_dataset.dbkey,
type = 'server_dir',
is_binary = None,
space_to_tab = params.space_to_tab,
path = full_file )
json_file.write( to_json_string( json ) + '\n' )
data_list.append( data )
json_file.close()
return ( json_file_path, data_list )
+4 -1
View File
@@ -1,7 +1,7 @@
.libraryRow{background-color:#ebd9b2;}
.datasetHighlighted{background-color:#C1C9E5;}
.libraryItemDeleted-True{font-style:italic;}
div.historyItemBody{padding:4px 4px 2px 4px;}
div.libraryItemBody{padding:4px 4px 2px 4px;}
li.folderRow,li.datasetRow{border-top:solid 1px #ddd;}
li.folderRow:hover,li.datasetRow:hover{background-color:#C1C9E5;}
img.expanderIcon{padding-right:4px;}
@@ -15,3 +15,6 @@ a.expandLink{text-decoration:none;}
span.expandLink{width:16px;height:16px;display:inline-block;vertical-align:middle;background:url(../images/silk/resultset_next.png);}
.folderRow.expanded span.expandLink{background:url(../images/silk/resultset_bottom.png);}
.folderRow span.rowIcon{width:16px;height:16px;display:inline-block;vertical-align:middle;background:url(../images/silk/folder.png);}
.libraryItem-error{margin-right:2px;padding:0 2px 0 2px;border:1px solid #AA6666;background:#FFCCCC;}
.libraryItem-queued{margin-right:2px;padding:0 2px 0 2px;border:1px solid #888888;background:#EEEEEE;}
.libraryItem-running{margin-right:2px;padding:0 2px 0 2px;border:1px solid #AAAA66;background:#FFFFCC;}
+22 -1
View File
@@ -10,7 +10,7 @@
font-style: italic;
}
div.historyItemBody {
div.libraryItemBody {
padding: 4px 4px 2px 4px;
}
@@ -88,3 +88,24 @@ span.expandLink {
background: url(../images/silk/folder.png);
}
.libraryItem-error {
margin-right: 2px;
padding: 0 2px 0 2px;
border: 1px solid $history_error_border;
background: $history_error_bg;
}
.libraryItem-queued {
margin-right: 2px;
padding: 0 2px 0 2px;
border: 1px solid $history_queued_border;
background: $history_queued_bg;
}
.libraryItem-running {
margin-right: 2px;
padding: 0 2px 0 2px;
border: 1px solid $history_running_border;
background: $history_running_bg;
}
+64 -26
View File
@@ -1,5 +1,6 @@
<%inherit file="/base.mako"/>
<%namespace file="/message.mako" import="render_msg" />
<%namespace file="/library/library_item_info.mako" import="render_library_item_info" />
<%
from time import strftime
from galaxy import util
@@ -11,6 +12,8 @@
<link href="${h.url_for('/static/style/library.css')}" rel="stylesheet" type="text/css" />
</%def>
<% tracked_datasets = {} %>
<script type="text/javascript">
$( document ).ready( function () {
// Hide all the folder contents
@@ -35,29 +38,6 @@
$(this).children().find("img.rowIcon").each( function() { this.src = icon_open; });
}
});
// Hide all dataset bodies
$("div.historyItemBody").hide();
// Handle the dataset body hide/show link.
$("div.historyItemWrapper").each( function() {
var id = this.id;
var li = $(this).parent();
var body = $(this).children( "div.historyItemBody" );
var peek = body.find( "pre.peek" )
$(this).children( ".historyItemTitleBar" ).find( ".historyItemTitle" ).wrap( "<a href='#'></a>" ).click( function() {
if ( body.is(":visible") ) {
if ( $.browser.mozilla ) { peek.css( "overflow", "hidden" ) }
body.slideUp( "fast" );
li.removeClass( "datasetHighlighted" );
}
else {
body.slideDown( "fast", function() {
if ( $.browser.mozilla ) { peek.css( "overflow", "auto" ); }
});
li.addClass( "datasetHighlighted" );
}
return false;
});
});
});
function checkForm() {
if ( $("select#action_on_datasets_select option:selected").text() == "delete" ) {
@@ -68,6 +48,54 @@
}
}
}
// Looks for changes in dataset state using an async request. Keeps
// calling itself (via setTimeout) until all datasets are in a terminal
// state.
var updater = function ( tracked_datasets ) {
// Check if there are any items left to track
var empty = true;
for ( i in tracked_datasets ) {
empty = false;
break;
}
if ( ! empty ) {
setTimeout( function() { updater_callback( tracked_datasets ) }, 3000 );
}
};
var updater_callback = function ( tracked_datasets ) {
// Build request data
var ids = []
var states = []
$.each( tracked_datasets, function ( id, state ) {
ids.push( id );
states.push( state );
});
// Make ajax call
$.ajax( {
type: "POST",
url: "${h.url_for( controller='library_dataset', action='library_item_updates' )}",
dataType: "json",
data: { ids: ids.join( "," ), states: states.join( "," ) },
success : function ( data ) {
$.each( data, function( id, val ) {
// Replace HTML
var cell = $("#libraryItem-" + id).find("#libraryItemInfo");
cell.html( val.html );
// If new state was terminal, stop tracking
if (( val.state == "ok") || ( val.state == "error") || ( val.state == "empty") || ( val.state == "deleted" ) || ( val.state == "discarded" )) {
delete tracked_datasets[ parseInt(id) ];
} else {
tracked_datasets[ parseInt(id) ] = val.state;
}
});
updater( tracked_datasets );
},
error: function() {
// Just retry, like the old method, should try to be smarter
updater( tracked_datasets );
}
});
};
</script>
<%def name="render_dataset( ldda, library_dataset, selected, library, folder, deleted, show_deleted )">
@@ -84,11 +112,13 @@
current_version = True
else:
current_version = False
if current_version and ldda.state not in ( 'ok', 'error', 'empty', 'deleted', 'discarded' ):
tracked_datasets[ldda.id] = ldda.state
%>
%if current_version:
<div class="historyItemWrapper historyItem historyItem-${ldda.state}" id="libraryItem-${ldda.id}">
<div class="libraryItemWrapper libraryItem" id="libraryItem-${ldda.id}">
## Header row for library items (name, state, action buttons)
<div class="historyItemTitleBar">
<div class="libraryItemTitleBar">
<table cellspacing="0" cellpadding="0" border="0" width="100%">
<tr>
<td width="*">
@@ -119,7 +149,7 @@
</div>
%endif
</td>
<td width="300">${ldda.message}</td>
<td width="300" id="libraryItemInfo">${render_library_item_info( ldda )}</td>
<td width="150">${uploaded_by}</td>
<td width="60">${ldda.create_time.strftime( "%Y-%m-%d" )}</td>
</tr>
@@ -287,3 +317,11 @@
</p>
%endif
</form>
%if tracked_datasets:
<script type="text/javascript">
// Updater
updater({${ ",".join( [ '"%s" : "%s"' % ( k, v ) for k, v in tracked_datasets.iteritems() ] ) }});
</script>
<!-- running: do not change this comment, used by TwillTestCase.library_wait -->
%endif
+3 -1
View File
@@ -29,7 +29,9 @@
</div>
<div style="clear: both"></div>
</div>
<input type="submit" name="create_library_button" value="Create"/>
<div class="form-row">
<input type="submit" name="create_library_button" value="Create"/>
</div>
</form>
</div>
</div>
+63 -2
View File
@@ -1,5 +1,6 @@
<%inherit file="/base.mako"/>
<%namespace file="/message.mako" import="render_msg" />
<%namespace file="/library/library_item_info.mako" import="render_library_item_info" />
<%
from galaxy import util
from galaxy.web.controllers.library import active_folders
@@ -13,6 +14,8 @@
<link href="${h.url_for('/static/style/library.css')}" rel="stylesheet" type="text/css" />
</%def>
<% tracked_datasets = {} %>
<%
class RowCounter( object ):
def __init__( self ):
@@ -77,6 +80,54 @@ class RowCounter( object ):
});
});
});
// Looks for changes in dataset state using an async request. Keeps
// calling itself (via setTimeout) until all datasets are in a terminal
// state.
var updater = function ( tracked_datasets ) {
// Check if there are any items left to track
var empty = true;
for ( i in tracked_datasets ) {
empty = false;
break;
}
if ( ! empty ) {
setTimeout( function() { updater_callback( tracked_datasets ) }, 3000 );
}
};
var updater_callback = function ( tracked_datasets ) {
// Build request data
var ids = []
var states = []
$.each( tracked_datasets, function ( id, state ) {
ids.push( id );
states.push( state );
});
// Make ajax call
$.ajax( {
type: "POST",
url: "${h.url_for( controller='library_dataset', action='library_item_updates' )}",
dataType: "json",
data: { ids: ids.join( "," ), states: states.join( "," ) },
success : function ( data ) {
$.each( data, function( id, val ) {
// Replace HTML
var cell = $("#libraryItem-" + id).find("#libraryItemInfo");
cell.html( val.html );
// If new state was terminal, stop tracking
if (( val.state == "ok") || ( val.state == "error") || ( val.state == "empty") || ( val.state == "deleted" ) || ( val.state == "discarded" )) {
delete tracked_datasets[ parseInt(id) ];
} else {
tracked_datasets[ parseInt(id) ] = val.state;
}
});
updater( tracked_datasets );
},
error: function() {
// Just retry, like the old method, should try to be smarter
updater( tracked_datasets );
}
});
};
</script>
<%def name="render_dataset( ldda, library_dataset, selected, library, folder, pad, parent, row_conter )">
@@ -95,6 +146,8 @@ class RowCounter( object ):
can_manage_library_dataset = trans.app.security_agent.can_manage_library_item( user, roles, library_dataset )
else:
current_version = False
if current_version and ldda.state not in ( 'ok', 'error', 'empty', 'deleted', 'discarded' ):
tracked_datasets[ldda.id] = ldda.state
%>
%if current_version:
<tr class="datasetRow"
@@ -102,7 +155,7 @@ class RowCounter( object ):
parent="${parent}"
style="display: none;"
%endif
>
id="libraryItem-${ldda.id}">
<td style="padding-left: ${pad+20}px;">
%if selected:
<input type="checkbox" name="ldda_ids" value="${ldda.id}" checked/>
@@ -129,7 +182,7 @@ class RowCounter( object ):
%endif
</div>
</td>
<td>${ldda.message}</td>
<td id="libraryItemInfo">${render_library_item_info( ldda )}</td>
<td>${uploaded_by}</td>
<td>${ldda.create_time.strftime( "%Y-%m-%d" )}</td>
</tr>
@@ -305,6 +358,14 @@ class RowCounter( object ):
</table>
</form>
%if tracked_datasets:
<script type="text/javascript">
// Updater
updater({${ ",".join( [ '"%s" : "%s"' % ( k, v ) for k, v in tracked_datasets.iteritems() ] ) }});
</script>
<!-- running: do not change this comment, used by TwillTestCase.library_wait -->
%endif
## Help about compression types
%if len( comptypes ) > 1:
+11 -5
View File
@@ -40,7 +40,8 @@
<div class="form-row">
<label>File:</label>
<div class="form-row-input">
<input type="file" name="files_0|file_data" galaxy-ajax-upload="true"/>
##<input type="file" name="files_0|file_data" galaxy-ajax-upload="true"/>
<input type="file" name="files_0|file_data"/>
</div>
<div style="clear: both"></div>
</div>
@@ -109,11 +110,16 @@
Convert spaces to tabs:
</label>
<div class="form-row-input">
<input type="checkbox" name="files_0|space_to_tab" value="Yes"/>Yes
## The files grouping only makes sense in the upload_file context.
%if upload_option == 'upload_file':
<input type="checkbox" name="files_0|space_to_tab" value="Yes"/>Yes
%else:
<input type="checkbox" name="space_to_tab" value="Yes"/>Yes
%endif
</div>
<div class="toolParamHelp" style="clear: both;">
Use this option if you are entering intervals by hand.
</div>
</div>
<div class="toolParamHelp" style="clear: both;">
Use this option if you are entering intervals by hand.
</div>
<div style="clear: both"></div>
<div class="form-row">
+13
View File
@@ -0,0 +1,13 @@
<%def name="render_library_item_info( ldda )">
%if ldda.state == 'error':
<div class="libraryItem-${ldda.state}">Job error <i>(click name for more info)</i></div>
%elif ldda.state == 'queued':
<div class="libraryItem-${ldda.state}">This job is queued</div>
%elif ldda.state == 'running':
<div class="libraryItem-${ldda.state}">This job is running</div>
%else:
${ldda.message}
%endif
</%def>
${render_library_item_info( ldda )}
@@ -8,4 +8,4 @@ tattatatgt agtaggttcg tctttaatct tcctttagca agtcttttac tgttttcgac
ctcaatgttc atgttcttag gttgttttgg ataatatgcg gtcagtttaa tcttcgttgt
ttcttcttaa aatatttatt catggtttaa tttttggttt gtacttgttc aggggccagt
tcattattta ctctgtttgt atacagcagt tcttttattt ttagtatgat tttaatttaa
aacaattcta atggtcaaaa a
aacaattcta atggtcaaaa a
+19 -19
View File
@@ -1274,6 +1274,7 @@ class TwillTestCase( unittest.TestCase ):
else:
check_str = "Added 1 datasets to the folder '%s' ( each is selected )." % folder_name
self.check_page_for_string( check_str )
self.library_wait( library_id )
self.home()
def set_library_dataset_permissions( self, library_id, folder_id, ldda_id, ldda_name, role_id, permissions_in, permissions_out ):
url = "library_admin/library_dataset_dataset_association?library_id=%s&folder_id=%s&&id=%s&permissions=True&update_roles_button=Save" % \
@@ -1359,25 +1360,7 @@ class TwillTestCase( unittest.TestCase ):
tc.submit( "runtool_btn" )
check_str = "Added 1 dataset versions to the library dataset '%s' in the folder '%s'." % ( ldda_name, folder_name )
self.check_page_for_string( check_str )
self.home()
def upload_new_dataset_versions( self, library_id, folder_id, folder_name, library_dataset_id, ldda_name, file_type='auto',
dbkey='hg18', message='', template_field_name1='', template_field_contents1='' ):
"""Upload new version(s) of a dataset using a directory of files"""
self.home()
self.visit_url( "%s/library_admin/library_dataset_dataset_association?upload_option=upload_directory&library_id=%s&folder_id=%s&replace_id=%s" \
% ( self.url, library_id, folder_id, library_dataset_id ) )
self.check_page_for_string( 'Upload a directory of files' )
self.check_page_for_string( 'You are currently selecting a new file to replace' )
tc.fv( "1", "file_type", file_type )
tc.fv( "1", "dbkey", dbkey )
tc.fv( "1", "message", message.replace( '+', ' ' ) )
tc.fv( "1", "server_dir", "library" )
# Add template field contents, if any...
if template_field_name1:
tc.fv( "1", template_field_name1, template_field_contents1 )
tc.submit( "runtool_btn" )
check_str = "Added 3 dataset versions to the library dataset '%s' in the folder '%s'." % ( ldda_name, folder_name )
self.check_page_for_string( check_str )
self.library_wait( library_id )
self.home()
def add_history_datasets_to_library( self, library_id, folder_id, folder_name, hda_id, root=False ):
"""Copy a dataset from the current history to a library folder"""
@@ -1410,6 +1393,7 @@ class TwillTestCase( unittest.TestCase ):
tc.submit( "runtool_btn" )
if check_str_after_submit:
self.check_page_for_string( check_str_after_submit )
self.library_wait( library_id )
self.home()
def add_dir_of_files_from_libraries_view( self, library_id, folder_id, selected_dir, file_type='auto', dbkey='hg18', roles_tuple=[],
message='', check_str_after_submit='', template_field_name1='', template_field_contents1='' ):
@@ -1432,6 +1416,7 @@ class TwillTestCase( unittest.TestCase ):
tc.submit( "runtool_btn" )
if check_str_after_submit:
self.check_page_for_string( check_str_after_submit )
self.library_wait( library_id, controller='library' )
self.home()
def delete_library_item( self, library_id, library_item_id, library_item_name, library_item_type='library_dataset' ):
"""Mark a library item as deleted"""
@@ -1464,3 +1449,18 @@ class TwillTestCase( unittest.TestCase ):
check_str = "Library '%s' and all of its contents have been purged" % library_name
self.check_page_for_string( check_str )
self.home()
def library_wait( self, library_id, controller='library_admin', maxiter=20 ):
"""Waits for the tools to finish"""
count = 0
sleep_amount = 1
self.home()
while count < maxiter:
count += 1
self.visit_url( "%s/%s/browse_library?id=%s" % ( self.url, controller, library_id ) )
page = tc.browser.get_html()
if page.find( '<!-- running: do not change this comment, used by TwillTestCase.library_wait -->' ) > -1:
time.sleep( sleep_amount )
sleep_amount += 1
else:
break
self.assertNotEqual(count, maxiter)
+2 -2
View File
@@ -79,8 +79,8 @@ def setup():
allow_user_creation = True,
allow_user_deletion = True,
admin_users = 'test@bx.psu.edu',
library_import_dir = galaxy_test_file_dir,
user_library_import_dir = os.path.join( galaxy_test_file_dir, 'users' ),
library_import_dir = os.path.join( os.getcwd(), galaxy_test_file_dir ),
user_library_import_dir = os.path.join( os.getcwd(), galaxy_test_file_dir, 'users' ),
global_conf = { "__file__": "universe_wsgi.ini.sample" } )
log.info( "Embedded Universe application started" )
+5 -2
View File
@@ -137,7 +137,7 @@ def add_file( dataset, json_file, output_path ):
# See if we have an empty file
if not os.path.exists( dataset.path ):
file_err( 'Uploaded temporary file (%s) does not exist. Please' % dataset.path, dataset, json_file )
file_err( 'Uploaded temporary file (%s) does not exist.' % dataset.path, dataset, json_file )
return
if not os.path.getsize( dataset.path ) > 0:
file_err( 'The uploaded file is empty', dataset, json_file )
@@ -237,7 +237,10 @@ def add_file( dataset, json_file, output_path ):
if ext == 'auto':
ext = 'data'
# Move the dataset to its "real" path
shutil.move( dataset.path, output_path )
if dataset.type == 'server_dir':
shutil.copy( dataset.path, output_path )
else:
shutil.move( dataset.path, output_path )
# Write the job info
info = dict( type = 'dataset',
dataset_id = dataset.dataset_id,