Don't alter the contents of a file while uploading to a data library if using the filesystem_paths option. This partially resolves the issue where a supposedly sorted BAM file was being resorted upon upload to a data library when using this option. A better imlementation of determining whether a BAM file has been sorted (so that it does not get resorted) remains to be done.

This commit is contained in:
Greg Von Kuster
2011-03-15 09:38:29 -04:00
parent 5fc793c8f2
commit 7640af8e06
4 changed files with 22 additions and 18 deletions
+3 -9
View File
@@ -59,7 +59,8 @@ class Bam( Binary ):
output = subprocess.Popen(params, stderr=subprocess.PIPE, stdout=subprocess.PIPE).communicate()[0]
# find returns -1 if string is not found
return output.find("SO:coordinate") != -1 or output.find("SO:sorted") != -1
def dataset_content_needs_grooming( self, file_name ):
return not self._is_coordinate_sorted( file_name )
def groom_dataset_content( self, file_name ):
"""
Ensures that the Bam file contents are sorted. This function is called
@@ -72,11 +73,9 @@ class Bam( Binary ):
## This command may also create temporary files <out.prefix>.%d.bam when the
## whole alignment cannot be fitted into memory ( controlled by option -m ).
#do this in a unique temp directory, because of possible <out.prefix>.%d.bam temp files
if self._is_coordinate_sorted(file_name):
if not self.dataset_content_needs_grooming( file_name ):
# Don't re-sort if already sorted
return
tmp_dir = tempfile.mkdtemp()
tmp_sorted_dataset_file_name_prefix = os.path.join( tmp_dir, 'sorted' )
stderr_name = tempfile.NamedTemporaryFile( dir = tmp_dir, prefix = "bam_sort_stderr" ).name
@@ -84,7 +83,6 @@ class Bam( Binary ):
command = "samtools sort %s %s" % ( file_name, tmp_sorted_dataset_file_name_prefix )
proc = subprocess.Popen( args=command, shell=True, cwd=tmp_dir, stderr=open( stderr_name, 'wb' ) )
exit_code = proc.wait()
#Did sort succeed?
stderr = open( stderr_name ).read().strip()
if stderr:
@@ -93,10 +91,8 @@ class Bam( Binary ):
raise Exception, "Error Grooming BAM file contents: %s" % stderr
else:
print stderr
# Move samtools_created_sorted_file_name to our output dataset location
shutil.move( samtools_created_sorted_file_name, file_name )
# Remove temp file and empty temporary directory
os.unlink( stderr_name )
os.rmdir( tmp_dir )
@@ -124,9 +120,7 @@ class Bam( Binary ):
raise Exception, "Error Setting BAM Metadata: %s" % stderr
else:
print stderr
dataset.metadata.bam_index = index_file
# Remove temp file
os.unlink( stderr_name )
def sniff( self, filename ):
+4 -1
View File
@@ -88,8 +88,11 @@ class Data( object ):
except OSError, e:
log.exception('%s reading a file that does not exist %s' % (self.__class__.__name__, dataset.file_name))
return ''
def groom_dataset_content( self, file_name ):
def dataset_content_needs_grooming( self, file_name ):
"""This function is called on an output dataset file after the content is initially generated."""
return False
def groom_dataset_content( self, file_name ):
"""This function is called on an output dataset file if dataset_content_needs_grooming returns True."""
pass
def init_meta( self, dataset, copy_from=None ):
# Metadata should be left mostly uninitialized. Dataset will
+1 -1
View File
@@ -637,10 +637,10 @@ class TwillTestCase( unittest.TestCase ):
try:
if attributes is None:
attributes = {}
compare = attributes.get( 'compare', 'diff' )
if attributes.get( 'ftype', None ) == 'bam':
local_fh, temp_name = self._bam_to_sam( local_name, temp_name )
local_name = local_fh.name
compare = attributes.get( 'compare', 'diff' )
extra_files = attributes.get( 'extra_files', None )
if compare == 'diff':
self.files_diff( local_name, temp_name, attributes=attributes )
+14 -7
View File
@@ -334,8 +334,7 @@ def add_file( dataset, registry, json_file, output_path ):
file_err( 'The uploaded file contains inappropriate HTML content', dataset, json_file )
return
if data_type != 'binary':
# don't convert newlines on data we're only going to symlink
if link_data_only == 'link_to_files':
if link_data_only == 'copy_files':
in_place = True
if dataset.type in ( 'server_dir', 'path_paste' ):
in_place = False
@@ -353,8 +352,16 @@ def add_file( dataset, registry, json_file, output_path ):
ext = dataset.ext
if ext == 'auto':
ext = 'data'
# Move the dataset to its "real" path
datatype = registry.get_datatype_by_extension( ext )
if dataset.type in ( 'server_dir', 'path_paste' ) and link_data_only == 'link_to_files':
# Never alter a file that will not be copied to Galaxy's local file store.
if datatype.dataset_content_needs_grooming( output_path ):
err_msg = 'The uploaded files need grooming, so change your <b>Copy data into Galaxy?</b> selection to be ' + \
'<b>Copy files into Galaxy</b> instead of <b>Link to files without copying into Galaxy</b> so grooming can be performed.'
file_err( err_msg, dataset, json_file )
return
if link_data_only == 'copy_files' and dataset.type in ( 'server_dir', 'path_paste' ):
# Move the dataset to its "real" path
if converted_path is not None:
shutil.copy( converted_path, output_path )
try:
@@ -362,7 +369,7 @@ def add_file( dataset, registry, json_file, output_path ):
except:
pass
else:
# this should not happen, but it's here just in case
# This should not happen, but it's here just in case
shutil.copy( dataset.path, output_path )
elif link_data_only == 'copy_files':
shutil.move( dataset.path, output_path )
@@ -375,9 +382,9 @@ def add_file( dataset, registry, json_file, output_path ):
name = dataset.name,
line_count = line_count )
json_file.write( to_json_string( info ) + "\n" )
# Groom the dataset content if necessary
datatype = registry.get_datatype_by_extension( ext )
datatype.groom_dataset_content( output_path )
if datatype.dataset_content_needs_grooming( output_path ):
# Groom the dataset content if necessary
datatype.groom_dataset_content( output_path )
def add_composite_file( dataset, registry, json_file, output_path, files_path ):
if dataset.composite_files: