diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 1fe1885f4c9..eb2846328e0 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -280,6 +280,7 @@ class JobWrapper( object ): dataset.state = dataset.states.ERROR dataset.blurb = 'tool error' dataset.info = message + dataset.set_size() dataset.flush() job.state = model.Job.states.ERROR job.command_line = self.command_line @@ -341,6 +342,7 @@ class JobWrapper( object ): dataset.blurb = 'done' dataset.peek = 'no peek' dataset.info = stdout + stderr + dataset.set_size() if dataset.has_data(): # Call set_meta on each output dataset if metadata is missing if dataset.missing_meta(): diff --git a/lib/galaxy/model/__init__.py b/lib/galaxy/model/__init__.py index cdea79f4323..a5c81eadfb6 100644 --- a/lib/galaxy/model/__init__.py +++ b/lib/galaxy/model/__init__.py @@ -166,13 +166,12 @@ class Dataset( object ): RUNNING = 'running', OK = 'ok', EMPTY = 'empty', - ERROR = 'error', - FAKE = 'fake' ) + ERROR = 'error') file_path = "/tmp/" engine = None def __init__( self, id=None, hid=None, name=None, info=None, blurb=None, peek=None, extension=None, dbkey=None, state=None, metadata=None, history=None, parent_id=None, designation=None, - validation_errors=None, visible=True, filename_id = None ): + validation_errors=None, visible=True, filename_id = None, file_size=None ): self.name = name or "Unnamed dataset" self.id = id self.hid = hid @@ -189,6 +188,7 @@ class Dataset( object ): self.purged = False self.visible = visible self.filename_id = filename_id + self.file_size = file_size # Relationships self.history = history self.validation_errors = validation_errors @@ -234,7 +234,8 @@ class Dataset( object ): return datatypes_registry.get_datatype_by_extension( self.extension ) def get_metadata( self ): - if not self._metadata: self._metadata = dict() + if not self._metadata: + self._metadata = dict() return MetadataCollection( self, self.datatype.metadata_spec ) def set_metadata( self, bunch ): # Needs to accept a MetadataCollection, a bunch, or a dict @@ -253,22 +254,33 @@ class Dataset( object ): return dbkey[0] def set_dbkey( self, value ): if "dbkey" in self.datatype.metadata_spec: - if not isinstance(value, list): self.metadata.dbkey = [value] - else: self.metadata.dbkey = value - if isinstance(value, list): self.old_dbkey = value[0] - else: self.old_dbkey = value + if not isinstance(value, list): + self.metadata.dbkey = [value] + else: + self.metadata.dbkey = value + if isinstance(value, list): + self.old_dbkey = value[0] + else: + self.old_dbkey = value dbkey = property( get_dbkey, set_dbkey ) def change_datatype( self, new_ext ): datatypes_registry.change_datatype( self, new_ext ) def get_size( self ): - """ - Returns the size of the data on disk - """ + """Returns the size of the data on disk""" + if self.file_size: + return self.file_size + else: + try: + return os.path.getsize( self.file_name ) + except OSError: + return 0 + def set_size( self ): + """Returns the size of the data on disk""" try: - return os.path.getsize( self.file_name ) - except OSError, e: - return 0 + self.file_size = os.path.getsize( self.file_name ) + except OSError: + self.file_size = 0 def has_data( self ): """Detects whether there is any data""" return self.get_size() > 0 diff --git a/lib/galaxy/model/mapping.py b/lib/galaxy/model/mapping.py index 2655cf2ff4f..ae4e84300ef 100644 --- a/lib/galaxy/model/mapping.py +++ b/lib/galaxy/model/mapping.py @@ -81,6 +81,7 @@ Dataset.table = Table( "dataset", metadata, Column( "purged", Boolean ), Column( "visible", Boolean ), Column( "filename_id", Integer, ForeignKey( "dataset_filename.id" ), nullable=True ), + Column( 'file_size', Numeric( 15, 0 ) ), ForeignKeyConstraint(['parent_id'],['dataset.id'], ondelete="CASCADE") ) DatasetFileName.table = Table( "dataset_filename", metadata, diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index 9f5a2a0ace4..992adeb232f 100644 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -950,6 +950,7 @@ class Tool: child_data.state = child_data.states.OK child_data.init_meta() child_data.set_peek() + child_data.set_size() child_data.flush() # Add to child accociation table assoc = self.app.model.DatasetChildAssociation() @@ -986,6 +987,7 @@ class Tool: primary_data.state = primary_data.states.OK primary_data.init_meta(copy_from=outdata) primary_data.set_peek() + primary_data.set_size() primary_data.flush() # Add dataset to return dict primary_datasets[name][designation] = primary_data diff --git a/lib/galaxy/tools/actions/upload.py b/lib/galaxy/tools/actions/upload.py index 7d0a654fc01..63eb7c7a911 100644 --- a/lib/galaxy/tools/actions/upload.py +++ b/lib/galaxy/tools/actions/upload.py @@ -58,6 +58,7 @@ class UploadToolAction( object ): data.extension = "txt" data.dbkey = "?" data.info = err_msg + data.file_size = 0 data.flush() data.state = data.states.EMPTY trans.history.add_dataset( data ) @@ -154,6 +155,7 @@ class UploadToolAction( object ): data.state = data.states.OK data.init_meta() data.set_peek() + data.set_size() # validate incomming data """ diff --git a/lib/galaxy/web/controllers/root.py b/lib/galaxy/web/controllers/root.py index 88469bf9a6e..364d4ca8239 100644 --- a/lib/galaxy/web/controllers/root.py +++ b/lib/galaxy/web/controllers/root.py @@ -590,6 +590,7 @@ class Universe( BaseController ): history.add_dataset( data) history.flush() data.set_peek() + data.set_size() data.flush() trans.log_event("Added dataset %d to history %d" %(data.id, trans.history.id)) return trans.show_ok_message("Dataset "+str(data.hid)+" added to history "+str(history_id)+".") diff --git a/scripts/update_dataset_size.py b/scripts/update_dataset_size.py new file mode 100644 index 00000000000..d2db381a5d3 --- /dev/null +++ b/scripts/update_dataset_size.py @@ -0,0 +1,34 @@ +#!/usr/bin/env python2.4 +""" +Updates dataset.size column. +Remember to backup your database before running. +""" + +import sys, os, ConfigParser +import galaxy.app + + +def main(): + ini_file = sys.argv.pop(1) + conf_parser = ConfigParser.ConfigParser( {'here':os.getcwd()} ) + conf_parser.read( ini_file ) + configuration = {} + for key, value in conf_parser.items( "app:main" ): + configuration[key] = value + app = galaxy.app.UniverseApplication( global_conf = ini_file, **configuration ) + + #Step through Datasets, determining size on disk for each. + print "Determining the size of each dataset..." + for row in app.model.Dataset.table.select().execute(): + deleted = app.model.Dataset.get( row.id ).deleted + purged = app.model.Dataset.get( row.id ).purged + file_size = app.model.Dataset.get( row.id ).file_size + if file_size is None and not deleted and not purged: + size_on_disk = app.model.Dataset.get( row.id ).get_size() + print "Updating Dataset.%d with file_size: %d" %( row.id, size_on_disk ) + app.model.Dataset.table.update( app.model.Dataset.table.c.id == row.id ).execute( file_size=size_on_disk ) + app.shutdown() + sys.exit(0) + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/tools/data_source/biomart_filter.py b/tools/data_source/biomart_filter.py index 3e17b1f0be4..a1390a28717 100644 --- a/tools/data_source/biomart_filter.py +++ b/tools/data_source/biomart_filter.py @@ -107,4 +107,5 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No if data.missing_meta(): data.set_meta() data.set_peek() + data.set_size() data.flush() diff --git a/tools/data_source/encode_import_code.py b/tools/data_source/encode_import_code.py index ea5df9330af..b85e7ad8974 100644 --- a/tools/data_source/encode_import_code.py +++ b/tools/data_source/encode_import_code.py @@ -26,6 +26,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr data = app.datatypes_registry.change_datatype( data, file_type ) data.init_meta() data.set_peek() + data.set_size() app.model.flush() elif fields[0] == "#NewFile": description = fields[1] @@ -47,4 +48,5 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr newdata.dbkey = dbkey newdata.set_meta() newdata.set_peek() + new_data.set_size() app.model.flush() diff --git a/tools/data_source/encodedb_filter.py b/tools/data_source/encodedb_filter.py index 364719a8c5c..6fcc26c89d8 100644 --- a/tools/data_source/encodedb_filter.py +++ b/tools/data_source/encodedb_filter.py @@ -20,4 +20,5 @@ def exec_after_process( app, inp_data, out_data, param_dict, **kwd): if data: data.state = data.states.OK data.set_peek() + data.set_size() data.flush() diff --git a/tools/data_source/hbvar_filter.py b/tools/data_source/hbvar_filter.py index afe8c7d2917..70bff790aee 100644 --- a/tools/data_source/hbvar_filter.py +++ b/tools/data_source/hbvar_filter.py @@ -72,4 +72,5 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No else: data = app.datatypes_registry.change_datatype(data, 'tabular') data.set_peek() + data.set_size() data.flush() diff --git a/tools/data_source/microbial_import_code.py b/tools/data_source/microbial_import_code.py index f5a211dd9be..6706396f206 100644 --- a/tools/data_source/microbial_import_code.py +++ b/tools/data_source/microbial_import_code.py @@ -113,6 +113,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr data = app.datatypes_registry.change_datatype( data, file_type ) data.init_meta() data.set_peek() + data.set_size() app.model.flush() elif fields[0] == "#NewFile": description = fields[1] @@ -125,7 +126,6 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr newdata.name = basic_name + " (" + microbe_info[kingdom][org]['chrs'][chr]['data'][description]['feature'] +" for "+microbe_info[kingdom][org]['name']+":"+chr + ")" newdata.flush() history.add_dataset( newdata ) - newdata.flush() app.model.flush() try: copyfile(filepath,newdata.file_name) @@ -137,5 +137,5 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr newdata.dbkey = dbkey newdata.init_meta() newdata.set_peek() - # + newdata.set_size() app.model.flush() diff --git a/tools/data_source/ucsc_filter.py b/tools/data_source/ucsc_filter.py index 4b20080ebeb..9002e0e1b62 100644 --- a/tools/data_source/ucsc_filter.py +++ b/tools/data_source/ucsc_filter.py @@ -39,7 +39,6 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None): data = app.datatypes_registry.change_datatype(data, ext) out_data[name] = data - def exec_after_process( app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): """Verifies the data after the run""" items = out_data.items() @@ -61,10 +60,9 @@ def exec_after_process( app, inp_data, out_data, param_dict, tool=None, stdout=N if isinstance(data.datatype, datatypes.interval.Interval) and data.missing_meta(): data = app.datatypes_registry.change_datatype(data, 'tabular') out_data[name] = data - if err_flag: raise Exception(err_msg) - except Exception, exc: data.info = data.info + "\n" + str(exc) data.blurb = "error" + data.set_size() diff --git a/tools/data_source/ucsc_tablebrowser_code.py b/tools/data_source/ucsc_tablebrowser_code.py index 9e0bcd3712c..1e3910ea0d7 100644 --- a/tools/data_source/ucsc_tablebrowser_code.py +++ b/tools/data_source/ucsc_tablebrowser_code.py @@ -47,4 +47,5 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No if data.missing_meta(): data = app.datatypes_registry.change_datatype(data, 'tabular') data.set_peek() + data.set_size() data.flush() diff --git a/tools/filters/maf/maf_to_bed_code.py b/tools/filters/maf/maf_to_bed_code.py index 60f2a88fc58..2abc31bf1b4 100644 --- a/tools/filters/maf/maf_to_bed_code.py +++ b/tools/filters/maf/maf_to_bed_code.py @@ -46,6 +46,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr newdata.dbkey = dbkey newdata.init_meta() newdata.set_peek() + newdata.set_size() app.model.flush() output_data_list.append(newdata) else: