mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Bug fix in biomart - metadata will now be correctly set for biomart data.
This commit is contained in:
@@ -151,8 +151,8 @@ class Registry( object ):
|
||||
# call init_meta and copy metadata from itself. The datatype
|
||||
# being converted *to* will handle any metadata copying and
|
||||
# initialization.
|
||||
data.init_meta( copy_from=data )
|
||||
if data.has_data():
|
||||
data.init_meta( copy_from=data )
|
||||
data.set_peek()
|
||||
return data
|
||||
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
# Greg Von Kuster
|
||||
import urllib
|
||||
from galaxy.datatypes import sniff
|
||||
|
||||
from galaxy import datatypes, config
|
||||
import tempfile, shutil
|
||||
|
||||
@@ -18,27 +17,21 @@ def exec_before_job( app, inp_data, out_data, param_dict, tool=None):
|
||||
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None):
|
||||
"""Verifies the data after the run"""
|
||||
|
||||
URL = param_dict.get( 'URL', None )
|
||||
if not URL:
|
||||
raise Exception('Datasource has not sent back a URL parameter')
|
||||
URL = URL + '&_export=1&GALAXY_URL=0'
|
||||
CHUNK_SIZE = 2**20 # 1Mb
|
||||
MAX_SIZE = CHUNK_SIZE * 100
|
||||
|
||||
try:
|
||||
# damn you stupid sanitizer! - URL is now in the NEVER_SANITIZE list in util.py
|
||||
#URL = URL.replace('martX', 'mart&')
|
||||
#URL = URL.replace('0X_', '0&_')
|
||||
page = urllib.urlopen(URL)
|
||||
except Exception, exc:
|
||||
raise Exception('Problems connecting to %s (%s)' % (URL, exc) )
|
||||
|
||||
name, data = out_data.items()[0]
|
||||
|
||||
fp = open(data.file_name, 'wb')
|
||||
size = 0
|
||||
max_size_exceeded = False
|
||||
|
||||
while 1:
|
||||
chunk = page.read(CHUNK_SIZE)
|
||||
if not chunk:
|
||||
@@ -48,15 +41,15 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
|
||||
max_size_exceeded = True
|
||||
break
|
||||
fp.write(chunk)
|
||||
|
||||
fp.close()
|
||||
|
||||
if max_size_exceeded:
|
||||
data.info = 'Maximum data size of 100 MB exceeded, incomplete data retrieval.'
|
||||
else:
|
||||
data.info = data.name
|
||||
#Set meta data, format file to be valid interval type
|
||||
if isinstance(data.datatype, datatypes.interval.Interval):
|
||||
|
||||
if not isinstance(data.datatype, datatypes.interval.Bed) and isinstance(data.datatype, datatypes.interval.Interval):
|
||||
#Set meta data, format file to be valid interval type
|
||||
data.set_meta(first_line_is_header=True)
|
||||
#check for missing meta data, if all there, comment first line and process file
|
||||
if not data.missing_meta():
|
||||
@@ -69,17 +62,13 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
|
||||
startCol = int(data.metadata.startCol) - 1
|
||||
strandCol = int(data.metadata.strandCol) - 1
|
||||
|
||||
|
||||
for line in open(data.file_name, 'r'):
|
||||
line_ctr += 1
|
||||
|
||||
#First line is a non-commented header line, lets comment it out here
|
||||
if line_ctr == 0:
|
||||
temp.write("#%s" % line)
|
||||
continue
|
||||
|
||||
fields = line.strip().split('\t')
|
||||
|
||||
#If chrom col is an int, make it chrInt
|
||||
try:
|
||||
int(fields[chromCol])
|
||||
@@ -89,14 +78,12 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
|
||||
if fields[chromCol].upper()== "X" or fields[chromCol].upper()== "Y":
|
||||
fields[chromCol] = "chr%s" % fields[chromCol].upper()
|
||||
except:
|
||||
pass
|
||||
|
||||
pass
|
||||
#change to BED coordinate system
|
||||
try:
|
||||
fields[startCol] = str(int(fields[startCol]) - 1)
|
||||
except:
|
||||
pass
|
||||
|
||||
#set strand to +/-, instead of +1/-1
|
||||
try:
|
||||
if strandCol > 0:
|
||||
@@ -106,17 +93,18 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=No
|
||||
fields[strandCol] = "-"
|
||||
except:
|
||||
pass
|
||||
|
||||
temp.write("%s\n" % '\t'.join(fields))
|
||||
|
||||
temp.close()
|
||||
shutil.move(temp_filename,data.file_name)
|
||||
|
||||
else:
|
||||
data_type = sniff.guess_ext(data.file_name)
|
||||
data = app.datatypes_registry.change_datatype(data, data_type)
|
||||
if data.missing_meta():
|
||||
data.set_meta()
|
||||
else:
|
||||
data_type = sniff.guess_ext(data.file_name)
|
||||
data = app.datatypes_registry.change_datatype(data, data_type)
|
||||
if data.missing_meta():
|
||||
data.set_meta()
|
||||
data.set_peek()
|
||||
data.flush()
|
||||
|
||||
Reference in New Issue
Block a user