diff --git a/lib/galaxy/datatypes/binary.py b/lib/galaxy/datatypes/binary.py index 7b043d9aa53..c5f16d7098e 100644 --- a/lib/galaxy/datatypes/binary.py +++ b/lib/galaxy/datatypes/binary.py @@ -59,7 +59,8 @@ class Bam( Binary ): output = subprocess.Popen(params, stderr=subprocess.PIPE, stdout=subprocess.PIPE).communicate()[0] # find returns -1 if string is not found return output.find("SO:coordinate") != -1 or output.find("SO:sorted") != -1 - + def dataset_content_needs_grooming( self, file_name ): + return not self._is_coordinate_sorted( file_name ) def groom_dataset_content( self, file_name ): """ Ensures that the Bam file contents are sorted. This function is called @@ -72,11 +73,9 @@ class Bam( Binary ): ## This command may also create temporary files .%d.bam when the ## whole alignment cannot be fitted into memory ( controlled by option -m ). #do this in a unique temp directory, because of possible .%d.bam temp files - - if self._is_coordinate_sorted(file_name): + if not self.dataset_content_needs_grooming( file_name ): # Don't re-sort if already sorted return - tmp_dir = tempfile.mkdtemp() tmp_sorted_dataset_file_name_prefix = os.path.join( tmp_dir, 'sorted' ) stderr_name = tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "bam_sort_stderr" ).name @@ -84,7 +83,6 @@ class Bam( Binary ): command = "samtools sort %s %s" % ( file_name, tmp_sorted_dataset_file_name_prefix ) proc = subprocess.Popen( args=command, shell=True, cwd=tmp_dir, stderr=open( stderr_name, 'wb' ) ) exit_code = proc.wait() - #Did sort succeed? stderr = open( stderr_name ).read().strip() if stderr: @@ -93,10 +91,8 @@ class Bam( Binary ): raise Exception, "Error Grooming BAM file contents: %s" % stderr else: print stderr - # Move samtools_created_sorted_file_name to our output dataset location shutil.move( samtools_created_sorted_file_name, file_name ) - # Remove temp file and empty temporary directory os.unlink( stderr_name ) os.rmdir( tmp_dir ) @@ -124,9 +120,7 @@ class Bam( Binary ): raise Exception, "Error Setting BAM Metadata: %s" % stderr else: print stderr - dataset.metadata.bam_index = index_file - # Remove temp file os.unlink( stderr_name ) def sniff( self, filename ): diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index 8ba773f95ee..3905973373c 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -88,8 +88,11 @@ class Data( object ): except OSError, e: log.exception('%s reading a file that does not exist %s' % (self.__class__.__name__, dataset.file_name)) return '' - def groom_dataset_content( self, file_name ): + def dataset_content_needs_grooming( self, file_name ): """This function is called on an output dataset file after the content is initially generated.""" + return False + def groom_dataset_content( self, file_name ): + """This function is called on an output dataset file if dataset_content_needs_grooming returns True.""" pass def init_meta( self, dataset, copy_from=None ): # Metadata should be left mostly uninitialized. Dataset will diff --git a/test/base/twilltestcase.py b/test/base/twilltestcase.py index f864d1b789c..8147d5d9bd7 100644 --- a/test/base/twilltestcase.py +++ b/test/base/twilltestcase.py @@ -637,10 +637,10 @@ class TwillTestCase( unittest.TestCase ): try: if attributes is None: attributes = {} + compare = attributes.get( 'compare', 'diff' ) if attributes.get( 'ftype', None ) == 'bam': local_fh, temp_name = self._bam_to_sam( local_name, temp_name ) local_name = local_fh.name - compare = attributes.get( 'compare', 'diff' ) extra_files = attributes.get( 'extra_files', None ) if compare == 'diff': self.files_diff( local_name, temp_name, attributes=attributes ) diff --git a/tools/data_source/upload.py b/tools/data_source/upload.py index a7652d6dc2e..acfeb62b7a2 100644 --- a/tools/data_source/upload.py +++ b/tools/data_source/upload.py @@ -334,8 +334,7 @@ def add_file( dataset, registry, json_file, output_path ): file_err( 'The uploaded file contains inappropriate HTML content', dataset, json_file ) return if data_type != 'binary': - # don't convert newlines on data we're only going to symlink - if link_data_only == 'link_to_files': + if link_data_only == 'copy_files': in_place = True if dataset.type in ( 'server_dir', 'path_paste' ): in_place = False @@ -353,8 +352,16 @@ def add_file( dataset, registry, json_file, output_path ): ext = dataset.ext if ext == 'auto': ext = 'data' - # Move the dataset to its "real" path + datatype = registry.get_datatype_by_extension( ext ) + if dataset.type in ( 'server_dir', 'path_paste' ) and link_data_only == 'link_to_files': + # Never alter a file that will not be copied to Galaxy's local file store. + if datatype.dataset_content_needs_grooming( output_path ): + err_msg = 'The uploaded files need grooming, so change your Copy data into Galaxy? selection to be ' + \ + 'Copy files into Galaxy instead of Link to files without copying into Galaxy so grooming can be performed.' + file_err( err_msg, dataset, json_file ) + return if link_data_only == 'copy_files' and dataset.type in ( 'server_dir', 'path_paste' ): + # Move the dataset to its "real" path if converted_path is not None: shutil.copy( converted_path, output_path ) try: @@ -362,7 +369,7 @@ def add_file( dataset, registry, json_file, output_path ): except: pass else: - # this should not happen, but it's here just in case + # This should not happen, but it's here just in case shutil.copy( dataset.path, output_path ) elif link_data_only == 'copy_files': shutil.move( dataset.path, output_path ) @@ -375,9 +382,9 @@ def add_file( dataset, registry, json_file, output_path ): name = dataset.name, line_count = line_count ) json_file.write( to_json_string( info ) + "\n" ) - # Groom the dataset content if necessary - datatype = registry.get_datatype_by_extension( ext ) - datatype.groom_dataset_content( output_path ) + if datatype.dataset_content_needs_grooming( output_path ): + # Groom the dataset content if necessary + datatype.groom_dataset_content( output_path ) def add_composite_file( dataset, registry, json_file, output_path, files_path ): if dataset.composite_files: