Merge pull request #950 from fescudie/dev

Add BIOM v1 in default datatypes.
This commit is contained in:
John Chilton
2015-11-18 13:41:02 -05:00
2 changed files with 43 additions and 0 deletions
+3
View File
@@ -424,6 +424,8 @@
<datatype extension="plybinary" type="galaxy.datatypes.constructive_solid_geometry:PlyBinary" display_in_upload="true" />
<datatype extension="vtkascii" type="galaxy.datatypes.constructive_solid_geometry:VtkAscii" display_in_upload="true" />
<datatype extension="vtkbinary" type="galaxy.datatypes.constructive_solid_geometry:VtkBinary" display_in_upload="true" />
<!-- Metagenomic Datatype -->
<datatype extension="biom1" type="galaxy.datatypes.text:Biom1" display_in_upload="True" subclass="True" mimetype="application/json" />
</registration>
<sniffers>
<!--
@@ -485,6 +487,7 @@
<sniffer type="galaxy.datatypes.text:Obo"/>
<sniffer type="galaxy.datatypes.text:Arff"/>
<sniffer type="galaxy.datatypes.text:Ipynb"/>
<sniffer type="galaxy.datatypes.text:Biom1"/>
<sniffer type="galaxy.datatypes.text:Json"/>
<sniffer type="galaxy.datatypes.sequence:RNADotPlotMatrix"/>
<sniffer type="galaxy.datatypes.sequence:DotBracket"/>
+40
View File
@@ -122,6 +122,46 @@ class Ipynb( Json ):
pass
class Biom1(Json):
file_ext = "biom1"
def set_peek(self, dataset, is_multi_byte=False):
super(Biom1, self).set_peek(dataset, is_multi_byte)
if not dataset.dataset.purged:
dataset.blurb = "Biological Observation Matrix v1"
def sniff(self, filename):
is_biom = False
if self._looks_like_json( filename ):
is_biom = self._looks_like_biom(filename)
return is_biom
def _looks_like_biom(self, filepath, load_size=50000):
"""
@param filepath: [str] The path to the evaluated file.
@param load_size: [int] The size of the file block load in RAM (in
bytes).
"""
is_biom = False
segment_size = int(load_size / 2)
try:
with open(filepath, "r") as fh:
prev_str = ""
segment_str = fh.read(segment_size)
if segment_str.strip().startswith('{'):
while segment_str and not is_biom:
current_str = prev_str + segment_str
if '"format"' in current_str:
current_str = re.sub(r'\s', '', current_str)
if '"format":"BiologicalObservationMatrix' in current_str:
is_biom = True
prev_str = segment_str
segment_str = fh.read(segment_size)
except:
pass
return is_biom
class Obo( Text ):
"""
OBO file format description