Merge pull request #5787 from phnmnl/feature/isa_data_type

Feature/isa data type
This commit is contained in:
Martin Cech
2018-10-19 17:26:52 -04:00
committed by GitHub
6 changed files with 417 additions and 3 deletions
+6
View File
@@ -155,6 +155,11 @@
<display file="rviewer/bed.xml" inherit="true"/>
<display file="igv/interval_as_bed.xml" inherit="true"/>
</datatype>
<!-- ISA data types -->
<datatype extension="isa-tab" type="galaxy.datatypes.isa:IsaTab" mimetype="application/isa-tools" display_in_upload="true" description="ISA-Tab data type." description_url="https://isa-tools.org"/>
<datatype extension="isa-json" type="galaxy.datatypes.isa:IsaJson" mimetype="application/isa-tools" display_in_upload="true" description="ISA-JSON data type." description_url="https://isa-tools.org"/>
<datatype extension="picard_interval_list" type="galaxy.datatypes.tabular:Tabular" subclass="true" display_in_upload="true">
<converter file="picard_interval_list_to_bed6_converter.xml" target_datatype="bed6"/>
</datatype>
@@ -837,6 +842,7 @@
<sniffer type="galaxy.datatypes.images:Xpm"/>
<sniffer type="galaxy.datatypes.images:Eps"/>
<sniffer type="galaxy.datatypes.images:Rast"/>
<!--
Keep this commented until the sniff method in the assembly.py
module is fixed to not read the entire file.
+383
View File
@@ -0,0 +1,383 @@
"""
ISA datatype
See https://github.com/ISA-tools
"""
from __future__ import print_function
import json
import logging
import os
import os.path
import re
import shutil
import sys
import tempfile
from cgi import escape
from json import dumps # noqa: F401
# Imports isatab after turning off warnings inside logger settings to avoid pandas warning making uploads fail.
logging.getLogger("isatools.isatab").setLevel(logging.ERROR)
from isatools import isajson
from isatools import isatab_meta
from galaxy import model
from galaxy import util
from galaxy.datatypes import data
from galaxy.util.compression_utils import CompressedFile
from galaxy.util.sanitize_html import sanitize_html
# CONSTANTS {{{1
################################################################
# Main files regex
JSON_FILE_REGEX = re.compile(r"^.*\.json$", flags=re.IGNORECASE)
INVESTIGATION_FILE_REGEX = re.compile(r"^i_\w+\.txt$", flags=re.IGNORECASE)
# The name of the ISA archive (compressed file) as saved inside Galaxy
ISA_ARCHIVE_NAME = "archive"
# Set max number of lines of the history peek
_MAX_LINES_HISTORY_PEEK = 11
# Configure logger {{{1
################################################################
logger = logging.getLogger(__name__)
# Function for opening correctly a CSV file for csv.reader() for both Python 2 and 3 {{{1
################################################################
def utf8_text_file_open(path):
if sys.version_info[0] < 3:
fp = open(path, 'rb')
else:
fp = open(path, 'r', newline='', encoding='utf8')
return fp
# ISA class {{{1
################################################################
class _Isa(data.Data):
""" Base class for implementing ISA datatypes """
composite_type = 'auto_primary_file'
allow_datatype_change = False
is_binary = True
_main_file_regex = None
# Make investigation instance {{{2
################################################################
def _make_investigation_instance(self, filename):
raise NotImplementedError()
# Constructor {{{2
################################################################
def __init__(self, main_file_regex, **kwd):
super(_Isa, self).__init__(**kwd)
self._main_file_regex = main_file_regex
# Add the archive file as the only composite file
self.add_composite_file(ISA_ARCHIVE_NAME, is_binary=True, optional=True)
# Get ISA folder path {{{2
################################################################
def _get_isa_folder_path(self, dataset):
isa_folder = None
if dataset:
if isinstance(dataset, model.Dataset):
isa_folder = dataset.extra_files_path
if isinstance(dataset, model.HistoryDatasetAssociation):
# XXX With this loop the dataset name is reset inside the history to the ISA archive ID. Why?
for attr, value in dataset.__dict__.iteritems():
if str(attr) == '_metadata_collection':
datatype = value.parent.datatype # noqa: F841
isa_folder = dataset.dataset.extra_files_path
if isa_folder is None:
raise Exception('Unvalid dataset object, or no extra files path found for this dataset.')
return isa_folder
# Get main file {{{2
################################################################
def _get_main_file(self, dataset):
"""Get the main file of the ISA archive. Either the investigation file i_*.txt for ISA-Tab, or the JSON file for ISA-JSON."""
main_file = None
isa_folder = self._get_isa_folder_path(dataset)
if os.path.exists(isa_folder):
# Get ISA archive older
isa_files = os.listdir(isa_folder)
# Try to find main file
main_file = self._find_main_file_in_archive(isa_files)
if main_file is None:
raise Exception('Invalid ISA archive. No main file found.')
# Make full path
main_file = os.path.join(isa_folder, main_file)
return main_file
# Get investigation {{{2
################################################################
def _get_investigation(self, dataset):
"""Create a contained instance specific to the exact ISA type (Tab or Json).
We will use it to parse and access information from the archive."""
investigation = None
main_file = self._get_main_file(dataset)
if main_file is not None:
investigation = self._make_investigation_instance(main_file)
return investigation
# Find main file in archive {{{2
################################################################
def _find_main_file_in_archive(self, files_list):
"""Find the main file inside the ISA archive."""
found_file = None
for f in files_list:
match = self._main_file_regex.match(f)
if match:
if found_file is None:
found_file = match.group()
else:
raise Exception('More than one file match the pattern "', str(self._main_file_regex), '" to identify the investigation file')
return found_file
# Set peek {{{2
################################################################
def set_peek(self, dataset, is_multi_byte=False):
"""Set the peek and blurb text. Get first lines of the main file and set it as the peek."""
main_file = self._get_main_file(dataset)
if main_file is None:
raise RuntimeError("Unable to find the main file within the 'files_path' folder")
# Read first lines of main file
with open(main_file, "r") as f:
data = []
for line in f:
if len(data) < _MAX_LINES_HISTORY_PEEK:
data.append(line)
else:
break
if not dataset.dataset.purged and data:
dataset.peek = json.dumps({"data": data})
dataset.blurb = 'data'
else:
dataset.peek = 'file does not exist'
dataset.blurb = 'file purged from disk'
# Display peek {{{2
################################################################
def display_peek(self, dataset):
"""Create the HTML table used for displaying peek, from the peek text found by set_peek() method."""
out = ['<table cellspacing="0" cellpadding="3">']
try:
if not dataset.peek:
dataset.set_peek()
json_data = json.loads(dataset.peek)
for line in json_data["data"]:
line = line.strip()
if not line:
continue
out.append('<tr><td>%s</td></tr>' % escape(util.unicodify(line, 'utf-8')))
out.append('</table>')
out = "".join(out)
except Exception as exc:
out = "Can't create peek %s" % str(exc)
return out
# Generate primary file {{{2
################################################################
def generate_primary_file(self, dataset=None):
"""Generate the primary file. It is an HTML file containing description of the composite dataset
as well as a list of the composite files that it contains."""
if dataset:
rval = ['<html><head><title>ISA Dataset </title></head><p/>']
if hasattr(dataset, "extra_files_path"):
rval.append('<div>ISA Dataset composed of the following files:<p/><ul>')
for cmp_file in os.listdir(dataset.extra_files_path):
rval.append('<li><a href="%s" type="text/plain">%s</a></li>' % (cmp_file, escape(cmp_file)))
rval.append('</ul></div></html>')
else:
rval.append('<div>ISA Dataset is empty!<p/><ul>')
return "\n".join(rval)
return "<div>No dataset available</div>"
# Dataset content needs grooming {{{2
################################################################
def dataset_content_needs_grooming(self, file_name):
"""This function is called on an output dataset file after the content is initially generated."""
return os.path.basename(file_name) == ISA_ARCHIVE_NAME
# Groom dataset content {{{2
################################################################
def groom_dataset_content(self, file_name):
"""This method is called by Galaxy to extract files contained in a composite data type."""
# XXX Is the right place to extract files? Should this step not be a cleaning step instead?
# Could extracting be done earlier and composite files declared as files contained inside the archive
# instead of the archive itself?
# extract basename and folder of the current file whose content has to be groomed
basename = os.path.basename(file_name)
output_path = os.path.dirname(file_name)
# extract archive if the file corresponds to the ISA archive
if basename == ISA_ARCHIVE_NAME:
# perform extraction
# For some ZIP files CompressedFile::extract() extract the file inside <output_folder>/<file_name> instead of outputing it inside <output_folder>. So we first create a temporary folder, extract inside it, and move content to final destination.
temp_folder = tempfile.mkdtemp()
CompressedFile(file_name).extract(temp_folder)
shutil.rmtree(output_path)
extracted_files = os.listdir(temp_folder)
logger.debug(' '.join(extracted_files))
if len(extracted_files) == 0:
os.makedirs(output_path)
shutil.rmtree(temp_folder)
elif len(extracted_files) == 1 and os.path.isdir(os.path.join(temp_folder, extracted_files[0])):
shutil.move(os.path.join(temp_folder, extracted_files[0]), output_path)
shutil.rmtree(temp_folder)
else:
shutil.move(temp_folder, output_path)
# Display data {{{2
################################################################
def display_data(self, trans, dataset, preview=False, filename=None, to_ext=None, offset=None, ck_size=None, **kwd):
"""Downloads the ISA dataset if `preview` is `False`;
if `preview` is `True`, it returns a preview of the ISA dataset as a HTML page.
The preview is triggered when user clicks on the eye icon of the composite dataset."""
# if it is not required a preview use the default behaviour of `display_data`
if not preview:
return super(_Isa, self).display_data(trans, dataset, preview, filename, to_ext, **kwd)
# prepare the preview of the ISA dataset
investigation = self._get_investigation(dataset)
if investigation is None:
html = """<html><header><title>Error while reading ISA archive.</title></header>
<body>
<h1>An error occured while reading content of ISA archive.</h1>
<p>If you have tried to load your archive with the uploader by selecting isa-tab as composite data type, then try to load it again with isa-json instead. Conversely, if you have tried to load your archive with the uploader by selecting isa-json as composite data type, then try isa-tab instead.</p>
<p>You may also try to look into your zip file in order to find out if this is a proper ISA archive. If you see a file i_Investigation.txt inside, then it is an ISA-Tab archive. If you see a file with extension .json inside, then it is an ISA-JSON archive. If you see nothing like that, then either your ISA archive is corrupted, or it is not an ISA archive.</p>
</body></html>"""
else:
html = '<html><body>'
html += '<h1>{0} {1}</h1>'.format(investigation.title, investigation.identifier)
# Loop on all studies
for study in investigation.studies:
html += '<h2>Study %s</h2>' % study.identifier
html += '<h3>%s</h3>' % study.title
html += '<p>%s</p>' % study.description
html += '<p>Submitted the %s</p>' % study.submission_date
html += '<p>Released on %s</p>' % study.public_release_date
html += '<p>Experimental factors used: %s</p>' % ', '.join([x.name for x in study.factors])
# Loop on all assays of this study
for assay in study.assays:
html += '<h3>Assay %s</h3>' % assay.filename
html += '<p>Measurement type: %s</p>' % assay.measurement_type.term # OntologyAnnotation
html += '<p>Technology type: %s</p>' % assay.technology_type.term # OntologyAnnotation
html += '<p>Technology platform: %s</p>' % assay.technology_platform
if assay.data_files is not None:
html += '<p>Data files:</p>'
html += '<ul>'
for data_file in assay.data_files:
if data_file.filename != '':
html += '<li>' + escape(util.unicodify(str(data_file.filename), 'utf-8')) + ' - ' + escape(util.unicodify(str(data_file.label), 'utf-8')) + '</li>'
html += '</ul>'
html += '</body></html>'
# Set mime type
mime = 'text/html'
self._clean_and_set_mime_type(trans, mime)
return sanitize_html(html).encode('utf-8')
# ISA-Tab class {{{1
################################################################
class IsaTab(_Isa):
file_ext = "isa-tab"
# Constructor {{{2
################################################################
def __init__(self, **kwd):
super(IsaTab, self).__init__(main_file_regex=INVESTIGATION_FILE_REGEX, **kwd)
# Make investigation instance {{{2
################################################################
def _make_investigation_instance(self, filename):
# Parse ISA-Tab investigation file
parser = isatab_meta.InvestigationParser()
isa_dir = os.path.dirname(filename)
fp = utf8_text_file_open(filename)
parser.parse(fp)
for study in parser.isa.studies:
s_parser = isatab_meta.LazyStudySampleTableParser(parser.isa)
s_parser.parse(os.path.join(isa_dir, study.filename))
for assay in study.assays:
a_parser = isatab_meta.LazyAssayTableParser(parser.isa)
a_parser.parse(os.path.join(isa_dir, assay.filename))
isa = parser.isa
return isa
# ISA-JSON class {{{1
################################################################
class IsaJson(_Isa):
file_ext = "isa-json"
# Constructor {{{2
################################################################
def __init__(self, **kwd):
super(IsaJson, self).__init__(main_file_regex=JSON_FILE_REGEX, **kwd)
# Make investigation instance {{{2
################################################################
def _make_investigation_instance(self, filename):
# Parse JSON file
fp = utf8_text_file_open(filename)
isa = isajson.load(fp)
return isa
@@ -60,6 +60,7 @@ h5py==2.8.0
html5lib==1.0.1
idna==2.7
ipaddress==1.0.22; python_version < '3.3'
isa-rwval==0.10.4
iso8601==0.1.12
isodate==0.6.0
jmespath==0.9.3
Binary file not shown.
+8
View File
@@ -240,6 +240,14 @@ class ToolsUploadTestCase(api.ApiTestCase):
roadmaps_content = self._get_roadmaps_content(history_id, dataset)
assert roadmaps_content.strip() == "roadmaps\ncontent", roadmaps_content
@skip_without_datatype("isa-tab")
def test_composite_datatype_isatab(self):
isatab_zip_path = TestDataResolver().get_filename("MTBLS6.zip")
details = self._upload_and_get_details(open(isatab_zip_path, "rb"), file_type="isa-tab")
assert details["state"] == "ok"
assert details["file_ext"] == "isa-tab", details
assert details["file_size"] == 85, details
def test_upload_dbkey(self):
with self.dataset_populator.test_history() as history_id:
payload = self.dataset_populator.upload_payload(history_id, "Test123", dbkey="hg19")
+19 -3
View File
@@ -171,7 +171,15 @@ def add_file(dataset, registry, output_path):
return info
def add_composite_file(dataset, output_path, files_path):
def add_composite_file(dataset, registry, output_path, files_path):
# Find data type
if dataset.file_type is not None:
try:
datatype = registry.get_datatype_by_extension(dataset.file_type)
except Exception as e:
print("Unable to instantiate the datatype object for the file type '%s'" % dataset.file_type)
if dataset.composite_files:
os.mkdir(files_path)
for name, value in dataset.composite_files.items():
@@ -195,9 +203,17 @@ def add_composite_file(dataset, output_path, files_path):
sniff.convert_newlines_sep2tabs(dp, tmp_dir=tmpdir, tmp_prefix=tmp_prefix)
else:
sniff.convert_newlines(dp, tmp_dir=tmpdir, tmp_prefix=tmp_prefix)
shutil.move(dp, os.path.join(files_path, name))
# move the file to its final destination
file_output_path = os.path.join(files_path, name)
shutil.move(dp, file_output_path)
# groom the dataset file content if required by the corresponding datatype definition
if datatype.dataset_content_needs_grooming(file_output_path):
datatype.groom_dataset_content(file_output_path)
# Move the dataset to its "real" path
shutil.move(dataset.primary_file, output_path)
# Write the job info
return dict(type='dataset',
dataset_id=dataset.dataset_id,
@@ -263,7 +279,7 @@ def __main__():
try:
if dataset.type == 'composite':
files_path = output_paths[int(dataset.dataset_id)][1]
metadata.append(add_composite_file(dataset, output_path, files_path))
metadata.append(add_composite_file(dataset, registry, output_path, files_path))
else:
metadata.append(add_file(dataset, registry, output_path))
except UploadProblemException as e: