diff --git a/lib/galaxy/datatypes/registry.py b/lib/galaxy/datatypes/registry.py index 0b4666175ab..31f793825b2 100644 --- a/lib/galaxy/datatypes/registry.py +++ b/lib/galaxy/datatypes/registry.py @@ -151,8 +151,8 @@ class Registry( object ): # call init_meta and copy metadata from itself. The datatype # being converted *to* will handle any metadata copying and # initialization. - data.init_meta( copy_from=data ) if data.has_data(): + data.init_meta( copy_from=data ) data.set_peek() return data diff --git a/tools/data_source/biomart_filter.py b/tools/data_source/biomart_filter.py index dddf416a970..3e17b1f0be4 100644 --- a/tools/data_source/biomart_filter.py +++ b/tools/data_source/biomart_filter.py @@ -1,7 +1,6 @@ # Greg Von Kuster import urllib from galaxy.datatypes import sniff - from galaxy import datatypes, config import tempfile, shutil @@ -18,27 +17,21 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None): def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): """Verifies the data after the run""" - URL = param_dict.get( 'URL', None ) if not URL: raise Exception('Datasource has not sent back a URL parameter') URL = URL + '&_export=1&GALAXY_URL=0' CHUNK_SIZE = 2**20 # 1Mb MAX_SIZE = CHUNK_SIZE * 100 - try: - # damn you stupid sanitizer! - URL is now in the NEVER_SANITIZE list in util.py - #URL = URL.replace('martX', 'mart&') - #URL = URL.replace('0X_', '0&_') page = urllib.urlopen(URL) except Exception, exc: raise Exception('Problems connecting to %s (%s)' % (URL, exc) ) - name, data = out_data.items()[0] - fp = open(data.file_name, 'wb') size = 0 max_size_exceeded = False + while 1: chunk = page.read(CHUNK_SIZE) if not chunk: @@ -48,15 +41,15 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No max_size_exceeded = True break fp.write(chunk) - fp.close() if max_size_exceeded: data.info = 'Maximum data size of 100 MB exceeded, incomplete data retrieval.' else: data.info = data.name - #Set meta data, format file to be valid interval type - if isinstance(data.datatype, datatypes.interval.Interval): + + if not isinstance(data.datatype, datatypes.interval.Bed) and isinstance(data.datatype, datatypes.interval.Interval): + #Set meta data, format file to be valid interval type data.set_meta(first_line_is_header=True) #check for missing meta data, if all there, comment first line and process file if not data.missing_meta(): @@ -69,17 +62,13 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No startCol = int(data.metadata.startCol) - 1 strandCol = int(data.metadata.strandCol) - 1 - for line in open(data.file_name, 'r'): line_ctr += 1 - #First line is a non-commented header line, lets comment it out here if line_ctr == 0: temp.write("#%s" % line) continue - fields = line.strip().split('\t') - #If chrom col is an int, make it chrInt try: int(fields[chromCol]) @@ -89,14 +78,12 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No if fields[chromCol].upper()== "X" or fields[chromCol].upper()== "Y": fields[chromCol] = "chr%s" % fields[chromCol].upper() except: - pass - + pass #change to BED coordinate system try: fields[startCol] = str(int(fields[startCol]) - 1) except: pass - #set strand to +/-, instead of +1/-1 try: if strandCol > 0: @@ -106,17 +93,18 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No fields[strandCol] = "-" except: pass - temp.write("%s\n" % '\t'.join(fields)) - temp.close() shutil.move(temp_filename,data.file_name) - else: data_type = sniff.guess_ext(data.file_name) data = app.datatypes_registry.change_datatype(data, data_type) + if data.missing_meta(): + data.set_meta() else: data_type = sniff.guess_ext(data.file_name) data = app.datatypes_registry.change_datatype(data, data_type) + if data.missing_meta(): + data.set_meta() data.set_peek() data.flush()