diff --git a/lib/galaxy/datatypes/data.py b/lib/galaxy/datatypes/data.py index f1d7cc3949c..2635bf0611c 100644 --- a/lib/galaxy/datatypes/data.py +++ b/lib/galaxy/datatypes/data.py @@ -719,7 +719,49 @@ class LineCount( Text ): pass class Newick( Text ): - pass + """New Hampshire/Newick Format""" + file_ext = "nhx" + + MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True ) + + def __init__(self, **kwd): + """Initialize foobar datatype""" + Text.__init__(self, **kwd) + + def init_meta( self, dataset, copy_from=None ): + Text.init_meta( self, dataset, copy_from=copy_from ) + + + def sniff( self, filename ): + """ Returning false as the newick format is too general and cannot be sniffed.""" + return False + + +class Nexus( Text ): + """Nexus format as used By Paup, Mr Bayes, etc""" + file_ext = "nex" + + MetadataElement( name="columns", default=3, desc="Number of columns", readonly=True ) + + def __init__(self, **kwd): + """Initialize foobar datatype""" + Text.__init__(self, **kwd) + + def init_meta( self, dataset, copy_from=None ): + Text.init_meta( self, dataset, copy_from=copy_from ) + + + def sniff( self, filename ): + """All Nexus Files Simply puts a '#NEXUS' in its first line""" + f = open(filename, "r") + firstline = f.readline().upper() + f.close() + + if "#NEXUS" in firstline: + return True + else: + return False + # ------------- Utility methods -------------- diff --git a/lib/galaxy/datatypes/sniff.py b/lib/galaxy/datatypes/sniff.py index d7a1f32ca05..4388b03a5b4 100644 --- a/lib/galaxy/datatypes/sniff.py +++ b/lib/galaxy/datatypes/sniff.py @@ -6,6 +6,7 @@ import registry from galaxy import util from galaxy.datatypes.checkers import * from galaxy.datatypes.binary import unsniffable_binary_formats +from encodings import search_function as encodings_search_function log = logging.getLogger(__name__) @@ -15,7 +16,7 @@ def get_test_fname(fname): full_path = os.path.join(path, 'test', fname) return full_path -def stream_to_open_named_file( stream, fd, filename ): +def stream_to_open_named_file( stream, fd, filename, source_encoding=None, source_error='strict', target_encoding=None, target_error='strict' ): """Writes a stream to the provided file descriptor, returns the file's name and bool( is_multi_byte ). Closes file descriptor""" #signature and behavor is somewhat odd, due to backwards compatibility, but this can/should be done better CHUNK_SIZE = 1048576 @@ -23,6 +24,10 @@ def stream_to_open_named_file( stream, fd, filename ): is_compressed = False is_binary = False is_multi_byte = False + if not target_encoding or not encodings_search_function( target_encoding ): + target_encoding = util.DEFAULT_ENCODING #utf-8 + if not source_encoding: + source_encoding = util.DEFAULT_ENCODING #sys.getdefaultencoding() would mimic old behavior (defaults to ascii) while 1: chunk = stream.read( CHUNK_SIZE ) if not chunk: @@ -42,13 +47,12 @@ def stream_to_open_named_file( stream, fd, filename ): chars = chunk[:100] is_multi_byte = util.is_multi_byte( chars ) if not is_multi_byte: - for char in chars: - if ord( char ) > 128: - is_binary = True - break + is_binary = util.is_binary( chunk ) data_checked = True if not is_compressed and not is_binary: - os.write( fd, chunk.encode( "utf-8" ) ) + if not isinstance( chunk, unicode ): + chunk = chunk.decode( source_encoding, source_error ) + os.write( fd, chunk.encode( target_encoding, target_error ) ) else: # Compressed files must be encoded after they are uncompressed in the upload utility, # while binary files should not be encoded at all. @@ -56,10 +60,10 @@ def stream_to_open_named_file( stream, fd, filename ): os.close( fd ) return filename, is_multi_byte -def stream_to_file( stream, suffix='', prefix='', dir=None, text=False ): +def stream_to_file( stream, suffix='', prefix='', dir=None, text=False, **kwd ): """Writes a stream to a temporary file, returns the temporary file's name""" fd, temp_name = tempfile.mkstemp( suffix=suffix, prefix=prefix, dir=dir, text=text ) - return stream_to_open_named_file( stream, fd, temp_name ) + return stream_to_open_named_file( stream, fd, temp_name, **kwd ) def check_newlines( fname, bytes_to_read=52428800 ): """ @@ -305,14 +309,9 @@ def guess_ext( fname, sniff_order=None, is_multi_byte=False ): else: for hdr in headers: for char in hdr: - if len( char ) > 1: - for c in char: - if ord( c ) > 128: - is_binary = True - break - elif ord( char ) > 128: - is_binary = True - break + #old behavior had 'char' possibly having length > 1, + #need to determine when/if this occurs + is_binary = util.is_binary( char ) if is_binary: break if is_binary: diff --git a/lib/galaxy/datatypes/xml.py b/lib/galaxy/datatypes/xml.py index 37f34f55169..cb217c65177 100644 --- a/lib/galaxy/datatypes/xml.py +++ b/lib/galaxy/datatypes/xml.py @@ -76,3 +76,24 @@ class CisML( GenericXml ): dataset.blurb = 'file purged from disk' def sniff( self, filename ): return False + +class Phyloxml( GenericXml ): + """Format for defining phyloxml data http://www.phyloxml.org/""" + file_ext = "phyloxml" + def set_peek( self, dataset, is_multi_byte=False ): + """Set the peek and blurb text""" + if not dataset.dataset.purged: + dataset.peek = data.get_file_peek( dataset.file_name, is_multi_byte=is_multi_byte ) + dataset.blurb = 'Phyloxml data' + else: + dataset.peek = 'file does not exist' + dataset.blurb = 'file purged from disk' + + def sniff( self, filename ): + """"Checking for keyword - 'phyloxml' always in lowercase in the first few lines""" + f = open(filename, "r") + firstlines = "".join(f.readlines(5)) + f.close() + if "phyloxml" in firstlines: + return True + return False \ No newline at end of file diff --git a/lib/galaxy/jobs/__init__.py b/lib/galaxy/jobs/__init__.py index 50c185834c2..99cdb7e5088 100644 --- a/lib/galaxy/jobs/__init__.py +++ b/lib/galaxy/jobs/__init__.py @@ -471,7 +471,7 @@ class JobWrapper( object ): job.user.total_disk_usage += bytes # fix permissions - for path in [ dp.real_path for dp in self.get_output_fnames() ]: + for path in [ dp.real_path for dp in self.get_mutable_output_fnames() ]: util.umask_fix_perms( path, self.app.config.umask, 0666, self.app.config.gid ) self.sa_session.flush() log.debug( 'job %d ended' % self.job_id ) @@ -679,6 +679,11 @@ class JobWrapper( object ): self.compute_outputs() return self.output_paths + def get_mutable_output_fnames( self ): + if self.output_paths is None: + self.compute_outputs() + return filter( lambda dsp: dsp.mutable, self.output_paths ) + def get_output_hdas_and_fnames( self ): if self.output_hdas_and_paths is None: self.compute_outputs() @@ -686,10 +691,11 @@ class JobWrapper( object ): def compute_outputs( self ) : class DatasetPath( object ): - def __init__( self, dataset_id, real_path, false_path = None ): + def __init__( self, dataset_id, real_path, false_path = None, mutable = True ): self.dataset_id = dataset_id self.real_path = real_path self.false_path = false_path + self.mutable = mutable def __str__( self ): if self.false_path is None: return self.real_path @@ -706,13 +712,13 @@ class JobWrapper( object ): self.output_hdas_and_paths = {} for name, hda in [ ( da.name, da.dataset ) for da in job.output_datasets + job.output_library_datasets ]: false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % hda.dataset.id ) ) - dsp = DatasetPath( hda.dataset.id, hda.dataset.file_name, false_path ) + dsp = DatasetPath( hda.dataset.id, hda.dataset.file_name, false_path, mutable = hda.dataset.external_filename is None ) self.output_paths.append( dsp ) self.output_hdas_and_paths[name] = hda, dsp if special: false_path = os.path.abspath( os.path.join( self.working_directory, "galaxy_dataset_%d.dat" % special.dataset.id ) ) else: - results = [ ( da.name, da.dataset, DatasetPath( da.dataset.dataset.id, da.dataset.file_name ) ) for da in job.output_datasets + job.output_library_datasets ] + results = [ ( da.name, da.dataset, DatasetPath( da.dataset.dataset.id, da.dataset.file_name, mutable = da.dataset.dataset.external_filename is None ) ) for da in job.output_datasets + job.output_library_datasets ] self.output_paths = [t[2] for t in results] self.output_hdas_and_paths = dict([(t[0], t[1:]) for t in results]) if special: diff --git a/lib/galaxy/jobs/deferred/genome_transfer.py b/lib/galaxy/jobs/deferred/genome_transfer.py index 2fb9a4719cf..9ea720ecea9 100644 --- a/lib/galaxy/jobs/deferred/genome_transfer.py +++ b/lib/galaxy/jobs/deferred/genome_transfer.py @@ -115,15 +115,16 @@ class GenomeTransferPlugin( DataTransfer ): files = tar.getmembers() for filename in files: z = tar.extractfile(filename) - try: - chunk = z.read( CHUNK_SIZE ) - except IOError: - os.close( fd ) - log.error( 'Problem decompressing compressed data' ) - exit() - if not chunk: - break - os.write( fd, chunk ) + while 1: + try: + chunk = z.read( CHUNK_SIZE ) + except IOError: + os.close( fd ) + log.error( 'Problem decompressing compressed data' ) + exit() + if not chunk: + break + os.write( fd, chunk ) os.write( fd, '\n' ) os.close( fd ) tar.close() diff --git a/lib/galaxy/model/item_attrs.py b/lib/galaxy/model/item_attrs.py index 5bd33bd3640..4adeb71e84b 100644 --- a/lib/galaxy/model/item_attrs.py +++ b/lib/galaxy/model/item_attrs.py @@ -95,7 +95,7 @@ class UsesAnnotations: """ Returns a user's annotation string for an item. """ annotation_obj = self.get_item_annotation_obj( db_session, user, item ) if annotation_obj: - return annotation_obj.annotation + return galaxy.util.unicodify( annotation_obj.annotation ) return None def get_item_annotation_obj( self, db_session, user, item ): diff --git a/lib/galaxy/tools/__init__.py b/lib/galaxy/tools/__init__.py index f686a9216a9..e3e64a69b3c 100755 --- a/lib/galaxy/tools/__init__.py +++ b/lib/galaxy/tools/__init__.py @@ -187,7 +187,9 @@ class ToolBox( object ): section.elems[ section_key ] = workflow log.debug( "Loaded workflow: %s %s" % ( workflow_id, workflow.name ) ) elif section_key.startswith( 'label_' ): - section.elems[ section_key ] = section_val + if section_val: + section.elems[ section_key ] = section_val + log.debug( "Loaded label: %s" % ( section_val.text ) ) self.tool_panel[ key ] = section def load_integrated_tool_panel_keys( self ): """ @@ -215,12 +217,12 @@ class ToolBox( object ): section.elems[ key ] = None elif section_elem.tag == 'label': key = 'label_%s' % section_elem.get( 'id' ) - section.elems[ key ] = ToolSectionLabel( section_elem ) + section.elems[ key ] = None key = 'section_%s' % elem.get( 'id' ) self.integrated_tool_panel[ key ] = section elif elem.tag == 'label': key = 'label_%s' % elem.get( 'id' ) - self.integrated_tool_panel[ key ] = ToolSectionLabel( elem ) + self.integrated_tool_panel[ key ] = None def write_integrated_tool_panel_config_file( self ): """ Write the current in-memory version of the integrated_tool_panel.xml file to disk. Since Galaxy administrators @@ -254,10 +256,11 @@ class ToolBox( object ): if section_item: os.write( fd, ' \n' % section_item.id ) elif section_key.startswith( 'label_' ): - label_id = section_item.id or '' - label_text = section_item.text or '' - label_version = section_item.version or '' - os.write( fd, '