mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merge pull request #5793 from jmchilton/bounded_memory_datatypes
Sniffing framework with constrained memory and I/O.
This commit is contained in:
@@ -3,11 +3,13 @@ import tarfile
|
||||
|
||||
from galaxy.datatypes.binary import CompressedArchive
|
||||
from galaxy.datatypes.data import get_file_peek, Text
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
from galaxy.util import nice_size
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class SnapHmm(Text):
|
||||
file_ext = "snaphmm"
|
||||
edam_data = "data_1364"
|
||||
@@ -26,13 +28,11 @@ class SnapHmm(Text):
|
||||
except Exception:
|
||||
return "SNAP HMM model (%s)" % (nice_size(dataset.get_size()))
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
SNAP model files start with zoeHMM
|
||||
"""
|
||||
with open(filename, 'r') as handle:
|
||||
return handle.read(6) == 'zoeHMM'
|
||||
return False
|
||||
return file_prefix.startswith('zoeHMM')
|
||||
|
||||
|
||||
class Augustus(CompressedArchive):
|
||||
|
||||
@@ -13,20 +13,20 @@ import sys
|
||||
from galaxy.datatypes import data
|
||||
from galaxy.datatypes import sequence
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
from galaxy.datatypes.text import Html
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Amos(data.Text):
|
||||
"""Class describing the AMOS assembly file """
|
||||
edam_data = "data_0925"
|
||||
edam_format = "format_3582"
|
||||
file_ext = 'afg'
|
||||
|
||||
def sniff(self, filename):
|
||||
# FIXME: this method will read the entire file.
|
||||
# It should call get_headers() like other sniff methods.
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is an amos assembly file format
|
||||
Example::
|
||||
@@ -50,25 +50,24 @@ class Amos(data.Text):
|
||||
}
|
||||
}
|
||||
"""
|
||||
with open(filename) as fh:
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if line.startswith('{'):
|
||||
if re.match(r'{(RED|CTG|TLE)$', line):
|
||||
return True
|
||||
for line in file_prefix.line_iterator():
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if line.startswith('{'):
|
||||
if re.match(r'{(RED|CTG|TLE)$', line):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Sequences(sequence.Fasta):
|
||||
"""Class describing the Sequences file generated by velveth """
|
||||
edam_data = "data_0925"
|
||||
file_ext = 'sequences'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a velveth produced fasta format
|
||||
The id line has 3 fields separated by tabs: sequence_name sequence_index category::
|
||||
@@ -78,33 +77,33 @@ class Sequences(sequence.Fasta):
|
||||
>SEQUENCE_1_length_35 2 1
|
||||
CGACGAATGACAGGTCACGAATTTGGCGGGGATTA
|
||||
"""
|
||||
|
||||
with open(filename) as fh:
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if line.startswith('>'):
|
||||
if not re.match(r'>[^\t]+\t\d+\t\d+$', line):
|
||||
break
|
||||
# The next line.strip() must not be '', nor startwith '>'
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a fasta header
|
||||
fh = file_prefix.string_io()
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if line.startswith('>'):
|
||||
if not re.match(r'>[^\t]+\t\d+\t\d+$', line):
|
||||
break
|
||||
# The next line.strip() must not be '', nor startwith '>'
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a fasta header
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Roadmaps(data.Text):
|
||||
"""Class describing the Sequences file generated by velveth """
|
||||
edam_format = "format_2561"
|
||||
file_ext = 'roadmaps'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a velveth produced RoadMap::
|
||||
142858 21 1
|
||||
@@ -113,22 +112,22 @@ class Roadmaps(data.Text):
|
||||
...
|
||||
"""
|
||||
|
||||
with open(filename) as fh:
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if not re.match(r'\d+\t\d+\t\d+$', line):
|
||||
break
|
||||
# The next line.strip() should be 'ROADMAP 1'
|
||||
line = fh.readline().strip()
|
||||
if not re.match(r'ROADMAP \d+$', line):
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a fasta header
|
||||
fh = file_prefix.string_io()
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if not re.match(r'\d+\t\d+\t\d+$', line):
|
||||
break
|
||||
# The next line.strip() should be 'ROADMAP 1'
|
||||
line = fh.readline().strip()
|
||||
if not re.match(r'ROADMAP \d+$', line):
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a fasta header
|
||||
return False
|
||||
|
||||
|
||||
|
||||
@@ -39,11 +39,13 @@ from .data import (
|
||||
get_file_peek,
|
||||
Text
|
||||
)
|
||||
from .sniff import build_sniff_from_prefix
|
||||
from .xml import GenericXml
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class BlastXml(GenericXml):
|
||||
"""NCBI Blast XML Output data"""
|
||||
file_ext = "blastxml"
|
||||
@@ -59,7 +61,7 @@ class BlastXml(GenericXml):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""Determines whether the file is blastxml
|
||||
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
@@ -73,17 +75,17 @@ class BlastXml(GenericXml):
|
||||
>>> BlastXml().sniff(fname)
|
||||
False
|
||||
"""
|
||||
with open(filename) as handle:
|
||||
line = handle.readline()
|
||||
if line.strip() != '<?xml version="1.0"?>':
|
||||
return False
|
||||
line = handle.readline()
|
||||
if line.strip() not in ['<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "http://www.ncbi.nlm.nih.gov/dtd/NCBI_BlastOutput.dtd">',
|
||||
'<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "NCBI_BlastOutput.dtd">']:
|
||||
return False
|
||||
line = handle.readline()
|
||||
if line.strip() != '<BlastOutput>':
|
||||
return False
|
||||
handle = file_prefix.string_io()
|
||||
line = handle.readline()
|
||||
if line.strip() != '<?xml version="1.0"?>':
|
||||
return False
|
||||
line = handle.readline()
|
||||
if line.strip() not in ['<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "http://www.ncbi.nlm.nih.gov/dtd/NCBI_BlastOutput.dtd">',
|
||||
'<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "NCBI_BlastOutput.dtd">']:
|
||||
return False
|
||||
line = handle.readline()
|
||||
if line.strip() != '<BlastOutput>':
|
||||
return False
|
||||
return True
|
||||
|
||||
def merge(split_files, output_file):
|
||||
|
||||
@@ -9,12 +9,14 @@ from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.data import get_file_peek
|
||||
from galaxy.datatypes.data import nice_size
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
|
||||
MAX_HEADER_LINES = 500
|
||||
MAX_LINE_LEN = 2000
|
||||
COLOR_OPTS = ['COLOR_SCALARS', 'red', 'green', 'blue']
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Ply(object):
|
||||
"""
|
||||
The PLY format describes an object as a collection of vertices,
|
||||
@@ -37,16 +39,14 @@ class Ply(object):
|
||||
def __init__(self, **kwd):
|
||||
raise NotImplementedError
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
The structure of a typical PLY file:
|
||||
Header, Vertex List, Face List, (lists of other elements)
|
||||
"""
|
||||
with open(filename, "r") as fh:
|
||||
if not self._is_ply_header(fh, self.subtype):
|
||||
return False
|
||||
return True
|
||||
return False
|
||||
if not self._is_ply_header(file_prefix.string_io(), self.subtype):
|
||||
return False
|
||||
return True
|
||||
|
||||
def _is_ply_header(self, fh, subtype):
|
||||
"""
|
||||
@@ -131,6 +131,7 @@ class PlyBinary(Ply, Binary):
|
||||
Binary.__init__(self, **kwd)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Vtk(object):
|
||||
r"""
|
||||
The Visualization Toolkit provides a number of source and writer objects to
|
||||
@@ -200,16 +201,14 @@ class Vtk(object):
|
||||
def __init__(self, **kwd):
|
||||
raise NotImplementedError
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
VTK files can be either ASCII or binary, with two different
|
||||
styles of file formats: legacy or XML. We'll assume if the
|
||||
file contains a valid VTK header, then it is a valid VTK file.
|
||||
"""
|
||||
with open(filename, "r") as fh:
|
||||
if self._is_vtk_header(fh, self.subtype):
|
||||
return True
|
||||
return False
|
||||
if self._is_vtk_header(file_prefix.string_io(), self.subtype):
|
||||
return True
|
||||
return False
|
||||
|
||||
def _is_vtk_header(self, fh, subtype):
|
||||
|
||||
@@ -16,6 +16,7 @@ import six
|
||||
|
||||
from galaxy import util
|
||||
from galaxy.datatypes.metadata import MetadataElement # import directly to maintain ease of use in Datatype class definitions
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
from galaxy.util import (
|
||||
compression_utils,
|
||||
FILENAME_VALID_CHARS,
|
||||
@@ -961,6 +962,7 @@ class Newick(Text):
|
||||
return ['phyloviz']
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Nexus(Text):
|
||||
"""Nexus format as used By Paup, Mr Bayes, etc"""
|
||||
edam_data = "data_0872"
|
||||
@@ -974,15 +976,9 @@ class Nexus(Text):
|
||||
def init_meta(self, dataset, copy_from=None):
|
||||
Text.init_meta(self, dataset, copy_from=copy_from)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""All Nexus Files Simply puts a '#NEXUS' in its first line"""
|
||||
with open(filename, "r") as f:
|
||||
firstline = f.readline().upper()
|
||||
|
||||
if "#NEXUS" in firstline:
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
return file_prefix.string_io().read(6).upper() == "#NEXUS"
|
||||
|
||||
def get_visualizations(self, dataset):
|
||||
"""
|
||||
|
||||
@@ -22,6 +22,7 @@ from six.moves.urllib.parse import quote_plus
|
||||
from galaxy.datatypes import metadata
|
||||
from galaxy.datatypes.data import Text
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
from galaxy.datatypes.tabular import Tabular
|
||||
from galaxy.datatypes.text import Html
|
||||
from galaxy.util import nice_size
|
||||
@@ -35,6 +36,7 @@ VALID_GENOME_GRAPH_MARKERS = re.compile('^(chr.*|RH.*|rs.*|SNP_.*|CN.*|A_.*)')
|
||||
VALID_GENOTYPES_LINE = re.compile('^([a-zA-Z0-9]+)(\\s([0-9]{2}|[A-Z]{2}|NC|\?\?))+\\s*$')
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class GenomeGraphs(Tabular):
|
||||
"""
|
||||
Tab delimited data containing a marker id and any number of numeric values
|
||||
@@ -166,7 +168,7 @@ class GenomeGraphs(Tabular):
|
||||
errors.append('row %d, %s' % (' '.join(badvals)))
|
||||
return errors
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in gg format
|
||||
|
||||
@@ -178,9 +180,7 @@ class GenomeGraphs(Tabular):
|
||||
>>> GenomeGraphs().sniff( fname )
|
||||
True
|
||||
"""
|
||||
with open(filename, 'r') as f:
|
||||
buf = f.read(1024)
|
||||
|
||||
buf = file_prefix.contents_header
|
||||
rows = [l.split() for l in buf.splitlines()[1:4]] # break on lines and drop header, small sample
|
||||
|
||||
if len(rows) < 1:
|
||||
@@ -247,14 +247,6 @@ class rgSampleList(rgTabList):
|
||||
self.column_names[1] = 'IID'
|
||||
# this is what Plink wants as at 2009
|
||||
|
||||
def sniff(self, filename):
|
||||
with open(filename, "r") as infile:
|
||||
header = next(infile) # header
|
||||
if header[0] == 'FID' and header[1] == 'IID':
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
|
||||
class rgFeatureList(rgTabList):
|
||||
"""
|
||||
@@ -909,6 +901,7 @@ class LinkageStudies(Text):
|
||||
self.max_lines = 10
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class GenotypeMatrix(LinkageStudies):
|
||||
"""
|
||||
Sample matrix of genotypes
|
||||
@@ -918,7 +911,6 @@ class GenotypeMatrix(LinkageStudies):
|
||||
|
||||
def __init__(self, **kwd):
|
||||
super(GenotypeMatrix, self).__init__(**kwd)
|
||||
self.num_cols = -1
|
||||
|
||||
def header_check(self, fio):
|
||||
header_elems = fio.readline().split('\t')
|
||||
@@ -933,7 +925,7 @@ class GenotypeMatrix(LinkageStudies):
|
||||
|
||||
return True
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
>>> classname = GenotypeMatrix
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
@@ -941,7 +933,6 @@ class GenotypeMatrix(LinkageStudies):
|
||||
>>> file_true = get_test_fname("linkstudies." + extn_true)
|
||||
>>> classname().sniff(file_true)
|
||||
True
|
||||
|
||||
>>> false_files = list(LinkageStudies.test_files)
|
||||
>>> false_files.remove("linkstudies." + extn_true)
|
||||
>>> result_true = []
|
||||
@@ -954,27 +945,29 @@ class GenotypeMatrix(LinkageStudies):
|
||||
>>> result_true
|
||||
[]
|
||||
"""
|
||||
with open(filename, "r") as fio:
|
||||
fio = file_prefix.string_io()
|
||||
num_cols = -1
|
||||
|
||||
if not self.header_check(fio):
|
||||
if not self.header_check(fio):
|
||||
return False
|
||||
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return True
|
||||
|
||||
tokens = line.split('\t')
|
||||
|
||||
if num_cols == -1:
|
||||
num_cols = len(tokens)
|
||||
elif num_cols != len(tokens):
|
||||
return False
|
||||
if not VALID_GENOTYPES_LINE.match(line):
|
||||
return False
|
||||
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return True
|
||||
|
||||
tokens = line.split('\t')
|
||||
|
||||
if self.num_cols == -1:
|
||||
self.num_cols = len(tokens)
|
||||
elif self.num_cols != len(tokens):
|
||||
return False
|
||||
if not VALID_GENOTYPES_LINE.match(line):
|
||||
return False
|
||||
|
||||
return True
|
||||
return True
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class MarkerMap(LinkageStudies):
|
||||
"""
|
||||
Map of genetic markers including physical and genetic distance
|
||||
@@ -992,7 +985,7 @@ class MarkerMap(LinkageStudies):
|
||||
|
||||
return False
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
>>> classname = MarkerMap
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
@@ -1000,7 +993,6 @@ class MarkerMap(LinkageStudies):
|
||||
>>> file_true = get_test_fname("linkstudies." + extn_true)
|
||||
>>> classname().sniff(file_true)
|
||||
True
|
||||
|
||||
>>> false_files = list(LinkageStudies.test_files)
|
||||
>>> false_files.remove("linkstudies." + extn_true)
|
||||
>>> result_true = []
|
||||
@@ -1013,32 +1005,32 @@ class MarkerMap(LinkageStudies):
|
||||
>>> result_true
|
||||
[]
|
||||
"""
|
||||
with open(filename, "r") as fio:
|
||||
fio = file_prefix.string_io()
|
||||
if not self.header_check(fio):
|
||||
return False
|
||||
|
||||
if not self.header_check(fio):
|
||||
return False
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return True
|
||||
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return True
|
||||
try:
|
||||
chrm, gpos, nam, bpos, row = line.split()
|
||||
float(gpos)
|
||||
int(bpos)
|
||||
|
||||
try:
|
||||
chrm, gpos, nam, bpos, row = line.split()
|
||||
float(gpos)
|
||||
int(bpos)
|
||||
|
||||
try:
|
||||
int(chrm)
|
||||
except ValueError:
|
||||
if not chrm.lower()[0] in ('x', 'y', 'm'):
|
||||
return False
|
||||
|
||||
int(chrm)
|
||||
except ValueError:
|
||||
return False
|
||||
if not chrm.lower()[0] in ('x', 'y', 'm'):
|
||||
return False
|
||||
|
||||
return True
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class DataIn(LinkageStudies):
|
||||
"""
|
||||
Common linkage input file for intermarker distances
|
||||
@@ -1048,13 +1040,8 @@ class DataIn(LinkageStudies):
|
||||
|
||||
def __init__(self, **kwd):
|
||||
super(DataIn, self).__init__(**kwd)
|
||||
self.num_markers = None
|
||||
self.intermarkers = 0
|
||||
|
||||
def eof_function(self):
|
||||
return self.intermarkers > 0
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
>>> classname = DataIn
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
@@ -1062,7 +1049,6 @@ class DataIn(LinkageStudies):
|
||||
>>> file_true = get_test_fname("linkstudies." + extn_true)
|
||||
>>> classname().sniff(file_true)
|
||||
True
|
||||
|
||||
>>> false_files = list(LinkageStudies.test_files)
|
||||
>>> false_files.remove("linkstudies." + extn_true)
|
||||
>>> result_true = []
|
||||
@@ -1075,41 +1061,47 @@ class DataIn(LinkageStudies):
|
||||
>>> result_true
|
||||
[]
|
||||
"""
|
||||
with open(filename, "r") as fio:
|
||||
intermarkers = 0
|
||||
num_markers = None
|
||||
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return self.eof_function()
|
||||
def eof_function():
|
||||
return intermarkers > 0
|
||||
|
||||
tokens = line.split()
|
||||
try:
|
||||
if lcount == 0:
|
||||
self.num_markers = int(tokens[0])
|
||||
map(int, tokens[1:])
|
||||
elif lcount == 1:
|
||||
map(float, tokens)
|
||||
fio = file_prefix.string_io()
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return eof_function()
|
||||
|
||||
if len(tokens) != 4:
|
||||
return False
|
||||
elif lcount == 2:
|
||||
map(int, tokens)
|
||||
last_token = int(tokens[-1])
|
||||
tokens = line.split()
|
||||
try:
|
||||
if lcount == 0:
|
||||
num_markers = int(tokens[0])
|
||||
map(int, tokens[1:])
|
||||
elif lcount == 1:
|
||||
map(float, tokens)
|
||||
|
||||
if self.num_markers is None:
|
||||
return False
|
||||
if len(tokens) != last_token:
|
||||
return False
|
||||
if self.num_markers != last_token:
|
||||
return False
|
||||
elif tokens[0] == "3" and tokens[1] == "2":
|
||||
self.intermarkers += 1
|
||||
if len(tokens) != 4:
|
||||
return False
|
||||
elif lcount == 2:
|
||||
map(int, tokens)
|
||||
last_token = int(tokens[-1])
|
||||
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
if num_markers is None:
|
||||
return False
|
||||
if len(tokens) != last_token:
|
||||
return False
|
||||
if num_markers != last_token:
|
||||
return False
|
||||
elif tokens[0] == "3" and tokens[1] == "2":
|
||||
intermarkers += 1
|
||||
|
||||
return self.eof_function()
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
|
||||
return eof_function()
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class AllegroLOD(LinkageStudies):
|
||||
"""
|
||||
Allegro output format for LOD scores
|
||||
@@ -1125,7 +1117,7 @@ class AllegroLOD(LinkageStudies):
|
||||
|
||||
return False
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
>>> classname = AllegroLOD
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
@@ -1133,7 +1125,6 @@ class AllegroLOD(LinkageStudies):
|
||||
>>> file_true = get_test_fname("linkstudies." + extn_true)
|
||||
>>> classname().sniff(file_true)
|
||||
True
|
||||
|
||||
>>> false_files = list(LinkageStudies.test_files)
|
||||
>>> false_files.remove("linkstudies." + extn_true)
|
||||
>>> result_true = []
|
||||
@@ -1146,28 +1137,28 @@ class AllegroLOD(LinkageStudies):
|
||||
>>> result_true
|
||||
[]
|
||||
"""
|
||||
with open(filename, "r") as fio:
|
||||
fio = file_prefix.string_io()
|
||||
|
||||
if not self.header_check(fio):
|
||||
if not self.header_check(fio):
|
||||
return False
|
||||
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return True
|
||||
|
||||
tokens = line.split()
|
||||
|
||||
try:
|
||||
int(tokens[0])
|
||||
float(tokens[1])
|
||||
|
||||
if tokens[2] != "-inf":
|
||||
float(tokens[2])
|
||||
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
|
||||
for lcount, line in enumerate(fio):
|
||||
if lcount > self.max_lines:
|
||||
return True
|
||||
|
||||
tokens = line.split()
|
||||
|
||||
try:
|
||||
int(tokens[0])
|
||||
float(tokens[1])
|
||||
|
||||
if tokens[2] != "-inf":
|
||||
float(tokens[2])
|
||||
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
|
||||
return True
|
||||
return True
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
@@ -13,6 +13,7 @@ from galaxy import util
|
||||
from galaxy.datatypes import metadata
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import (
|
||||
build_sniff_from_prefix,
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
@@ -50,6 +51,7 @@ VIEWPORT_MAX_READS_PER_LINE = 10
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
@build_sniff_from_prefix
|
||||
class Interval(Tabular):
|
||||
"""Tab delimited data containing interval information"""
|
||||
edam_data = "data_3002"
|
||||
@@ -297,7 +299,7 @@ class Interval(Tabular):
|
||||
"""Return options for removing errors along with a description"""
|
||||
return [("lines", "Remove erroneous lines")]
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Checks for 'intervalness'
|
||||
|
||||
@@ -312,26 +314,23 @@ class Interval(Tabular):
|
||||
>>> Interval().sniff( fname )
|
||||
True
|
||||
"""
|
||||
found_valid_lines = False
|
||||
try:
|
||||
"""
|
||||
If we got here, we already know the file is_column_based and is not bed,
|
||||
so we'll just look for some valid data.
|
||||
"""
|
||||
headers = iter_headers(filename, '\t', comment_designator='#')
|
||||
headers = iter_headers(file_prefix, '\t', comment_designator='#')
|
||||
# If we got here, we already know the file is_column_based and is not bed,
|
||||
# so we'll just look for some valid data.
|
||||
for hdr in headers:
|
||||
if hdr:
|
||||
if len(hdr) < 3:
|
||||
return False
|
||||
try:
|
||||
# Assume chrom start and end are in column positions 1 and 2
|
||||
# respectively ( for 0 based columns )
|
||||
int(hdr[1])
|
||||
int(hdr[2])
|
||||
except Exception:
|
||||
return False
|
||||
return True
|
||||
# Assume chrom start and end are in column positions 1 and 2
|
||||
# respectively ( for 0 based columns )
|
||||
int(hdr[1])
|
||||
int(hdr[2])
|
||||
found_valid_lines = True
|
||||
except Exception:
|
||||
return False
|
||||
return found_valid_lines
|
||||
|
||||
def get_track_resolution(self, dataset, start, end):
|
||||
return None
|
||||
@@ -464,7 +463,7 @@ class Bed(Interval):
|
||||
except Exception:
|
||||
return "This item contains no content"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Checks for 'bedness'
|
||||
|
||||
@@ -488,10 +487,10 @@ class Bed(Interval):
|
||||
>>> Bed().sniff( fname )
|
||||
True
|
||||
"""
|
||||
if not get_headers(filename, '\t', comment_designator='#', count=1):
|
||||
if not get_headers(file_prefix, '\t', comment_designator='#', count=1):
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(filename, '\t', comment_designator='#')
|
||||
headers = iter_headers(file_prefix, '\t', comment_designator='#')
|
||||
for hdr in headers:
|
||||
if hdr[0] == '':
|
||||
continue
|
||||
@@ -635,6 +634,7 @@ class _RemoteCallMixin(object):
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
@build_sniff_from_prefix
|
||||
class Gff(Tabular, _RemoteCallMixin):
|
||||
"""Tab delimited data in Gff format"""
|
||||
edam_data = "data_1255"
|
||||
@@ -822,7 +822,7 @@ class Gff(Tabular, _RemoteCallMixin):
|
||||
ret_val.append((site_name, link))
|
||||
return ret_val
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in gff format
|
||||
|
||||
@@ -831,17 +831,17 @@ class Gff(Tabular, _RemoteCallMixin):
|
||||
For complete details see http://genome.ucsc.edu/FAQ/FAQformat#format3
|
||||
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
>>> fname = get_test_fname( 'gff_version_3.gff' )
|
||||
>>> fname = get_test_fname('gff_version_3.gff')
|
||||
>>> Gff().sniff( fname )
|
||||
False
|
||||
>>> fname = get_test_fname( 'test.gff' )
|
||||
>>> fname = get_test_fname('test.gff')
|
||||
>>> Gff().sniff( fname )
|
||||
True
|
||||
"""
|
||||
if len(get_headers(filename, '\t', count=2)) < 2:
|
||||
if len(get_headers(file_prefix, '\t', count=2)) < 2:
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(filename, '\t')
|
||||
headers = iter_headers(file_prefix, '\t')
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
|
||||
return False
|
||||
@@ -937,7 +937,7 @@ class Gff3(Gff):
|
||||
break
|
||||
Tabular.set_meta(self, dataset, overwrite=overwrite, skip=i)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in GFF version 3 format
|
||||
|
||||
@@ -970,10 +970,10 @@ class Gff3(Gff):
|
||||
>>> Gff3().sniff( fname )
|
||||
True
|
||||
"""
|
||||
if len(get_headers(filename, '\t', count=2)) < 2:
|
||||
if len(get_headers(file_prefix, '\t', count=2)) < 2:
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(filename, '\t')
|
||||
headers = iter_headers(file_prefix, '\t')
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('3') >= 0:
|
||||
return True
|
||||
@@ -1020,7 +1020,7 @@ class Gtf(Gff):
|
||||
MetadataElement(name="column_types", default=['str', 'str', 'str', 'int', 'int', 'float', 'str', 'int', 'list'],
|
||||
param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in gtf format
|
||||
|
||||
@@ -1045,10 +1045,10 @@ class Gtf(Gff):
|
||||
>>> Gtf().sniff( fname )
|
||||
True
|
||||
"""
|
||||
if len(get_headers(filename, '\t', count=2)) < 2:
|
||||
if len(get_headers(file_prefix, '\t', count=2)) < 2:
|
||||
return False
|
||||
try:
|
||||
headers = iter_headers(filename, '\t')
|
||||
headers = iter_headers(file_prefix, '\t')
|
||||
for hdr in headers:
|
||||
if hdr and hdr[0].startswith('##gff-version') and hdr[0].find('2') < 0:
|
||||
return False
|
||||
@@ -1085,6 +1085,7 @@ class Gtf(Gff):
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
@build_sniff_from_prefix
|
||||
class Wiggle(Tabular, _RemoteCallMixin):
|
||||
"""Tab delimited data in wiggle format"""
|
||||
edam_format = "format_3005"
|
||||
@@ -1218,7 +1219,7 @@ class Wiggle(Tabular, _RemoteCallMixin):
|
||||
max_data_lines = 100
|
||||
Tabular.set_meta(self, dataset, overwrite=overwrite, skip=i, max_data_lines=max_data_lines)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines wether the file is in wiggle format
|
||||
|
||||
@@ -1242,7 +1243,7 @@ class Wiggle(Tabular, _RemoteCallMixin):
|
||||
True
|
||||
"""
|
||||
try:
|
||||
headers = iter_headers(filename, None)
|
||||
headers = iter_headers(file_prefix, None)
|
||||
for hdr in headers:
|
||||
if len(hdr) > 1 and hdr[0] == 'track' and hdr[1].startswith('type=wiggle'):
|
||||
return True
|
||||
@@ -1272,6 +1273,7 @@ class Wiggle(Tabular, _RemoteCallMixin):
|
||||
return dataproviders.dataset.WiggleDataProvider(dataset_source, **settings)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class CustomTrack(Tabular):
|
||||
"""UCSC CustomTrack"""
|
||||
edam_format = "format_3588"
|
||||
@@ -1360,7 +1362,7 @@ class CustomTrack(Tabular):
|
||||
ret_val.append((site_name, link))
|
||||
return ret_val
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in customtrack format.
|
||||
|
||||
@@ -1377,7 +1379,8 @@ class CustomTrack(Tabular):
|
||||
>>> CustomTrack().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = iter_headers(filename, None)
|
||||
headers = iter_headers(file_prefix, None)
|
||||
found_at_least_one_track = False
|
||||
first_line = True
|
||||
for hdr in headers:
|
||||
if first_line:
|
||||
@@ -1409,9 +1412,10 @@ class CustomTrack(Tabular):
|
||||
int(hdr[2])
|
||||
except Exception:
|
||||
return False
|
||||
found_at_least_one_track = True
|
||||
except Exception:
|
||||
return False
|
||||
return True
|
||||
return found_at_least_one_track
|
||||
|
||||
|
||||
class ENCODEPeak(Interval):
|
||||
@@ -1467,6 +1471,7 @@ class ChromatinInteractions(Interval):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class ScIdx(Tabular):
|
||||
"""
|
||||
ScIdx files are 1-based and consist of strand-specific coordinate counts.
|
||||
@@ -1492,55 +1497,55 @@ class ScIdx(Tabular):
|
||||
# line of the dataset displays them.
|
||||
self.column_names = ['chrom', 'index', 'forward', 'reverse', 'value']
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Checks for 'scidx-ness.'
|
||||
"""
|
||||
count = 0
|
||||
with open(filename, "r") as fh:
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
# EOF
|
||||
if count > 1:
|
||||
# The second line is always the labels:
|
||||
# chrom index forward reverse value
|
||||
# We need at least the column labels and a data line.
|
||||
return True
|
||||
return False
|
||||
line = line.strip()
|
||||
# The first line is always a comment like this:
|
||||
# 2015-11-23 20:18:56.51;input.bam;READ1
|
||||
if count == 0:
|
||||
if line.startswith('#'):
|
||||
count += 1
|
||||
continue
|
||||
else:
|
||||
return False
|
||||
# Skip first line.
|
||||
fh = file_prefix.string_io()
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
# EOF
|
||||
if count > 1:
|
||||
items = line.split('\t')
|
||||
if len(items) != 5:
|
||||
return False
|
||||
index = items[1]
|
||||
if not index.isdigit():
|
||||
return False
|
||||
forward = items[2]
|
||||
if not forward.isdigit():
|
||||
return False
|
||||
reverse = items[3]
|
||||
if not reverse.isdigit():
|
||||
return False
|
||||
value = items[4]
|
||||
if not value.isdigit():
|
||||
return False
|
||||
if int(forward) + int(reverse) != int(value):
|
||||
return False
|
||||
if count == 100:
|
||||
# The second line is always the labels:
|
||||
# chrom index forward reverse value
|
||||
# We need at least the column labels and a data line.
|
||||
return True
|
||||
count += 1
|
||||
if count < 100 and count > 0:
|
||||
return False
|
||||
line = line.strip()
|
||||
# The first line is always a comment like this:
|
||||
# 2015-11-23 20:18:56.51;input.bam;READ1
|
||||
if count == 0:
|
||||
if line.startswith('#'):
|
||||
count += 1
|
||||
continue
|
||||
else:
|
||||
return False
|
||||
# Skip first line.
|
||||
if count > 1:
|
||||
items = line.split('\t')
|
||||
if len(items) != 5:
|
||||
return False
|
||||
index = items[1]
|
||||
if not index.isdigit():
|
||||
return False
|
||||
forward = items[2]
|
||||
if not forward.isdigit():
|
||||
return False
|
||||
reverse = items[3]
|
||||
if not reverse.isdigit():
|
||||
return False
|
||||
value = items[4]
|
||||
if not value.isdigit():
|
||||
return False
|
||||
if int(forward) + int(reverse) != int(value):
|
||||
return False
|
||||
if count == 100:
|
||||
return True
|
||||
count += 1
|
||||
if count < 100 and count > 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.data import get_file_peek
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import (
|
||||
build_sniff_from_prefix,
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
@@ -84,10 +85,11 @@ class MOL(GenericMolFile):
|
||||
dataset.metadata.number_of_molecules = 1
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class SDF(GenericMolFile):
|
||||
file_ext = "sdf"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess if the file is a SDF2 file.
|
||||
|
||||
@@ -102,11 +104,9 @@ class SDF(GenericMolFile):
|
||||
>>> fname = get_test_fname('drugbank_drugs.sdf')
|
||||
>>> SDF().sniff(fname)
|
||||
True
|
||||
|
||||
>>> fname = get_test_fname('github88.v3k.sdf')
|
||||
>>> SDF().sniff(fname)
|
||||
True
|
||||
|
||||
>>> fname = get_test_fname('chebi_57262.v3k.mol')
|
||||
>>> SDF().sniff(fname)
|
||||
False
|
||||
@@ -114,23 +114,22 @@ class SDF(GenericMolFile):
|
||||
m_end_found = False
|
||||
limit = 10000
|
||||
idx = 0
|
||||
with open(filename) as in_file:
|
||||
for line in in_file:
|
||||
idx += 1
|
||||
line = line.rstrip('\n\r')
|
||||
if idx < 4:
|
||||
continue
|
||||
elif idx == 4:
|
||||
if len(line) != 39 or not(line.endswith(' V2000') or
|
||||
line.endswith(' V3000')):
|
||||
return False
|
||||
elif not m_end_found:
|
||||
if line == 'M END':
|
||||
m_end_found = True
|
||||
elif line == '$$$$':
|
||||
return True
|
||||
if idx == limit:
|
||||
break
|
||||
for line in file_prefix.line_iterator():
|
||||
idx += 1
|
||||
line = line.rstrip('\n\r')
|
||||
if idx < 4:
|
||||
continue
|
||||
elif idx == 4:
|
||||
if len(line) != 39 or not(line.endswith(' V2000') or
|
||||
line.endswith(' V3000')):
|
||||
return False
|
||||
elif not m_end_found:
|
||||
if line == 'M END':
|
||||
m_end_found = True
|
||||
elif line == '$$$$':
|
||||
return True
|
||||
if idx == limit:
|
||||
break
|
||||
return False
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
@@ -189,10 +188,11 @@ class SDF(GenericMolFile):
|
||||
split = classmethod(split)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class MOL2(GenericMolFile):
|
||||
file_ext = "mol2"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess if the file is a MOL2 file.
|
||||
|
||||
@@ -200,21 +200,19 @@ class MOL2(GenericMolFile):
|
||||
>>> fname = get_test_fname('drugbank_drugs.mol2')
|
||||
>>> MOL2().sniff(fname)
|
||||
True
|
||||
|
||||
>>> fname = get_test_fname('drugbank_drugs.cml')
|
||||
>>> MOL2().sniff(fname)
|
||||
False
|
||||
"""
|
||||
limit = 60
|
||||
idx = 0
|
||||
with open(filename) as in_file:
|
||||
for line in in_file:
|
||||
line = line.rstrip('\n\r')
|
||||
if line == '@<TRIPOS>MOLECULE':
|
||||
return True
|
||||
idx += 1
|
||||
if idx == limit:
|
||||
break
|
||||
for line in file_prefix.line_iterator():
|
||||
line = line.rstrip('\n\r')
|
||||
if line == '@<TRIPOS>MOLECULE':
|
||||
return True
|
||||
idx += 1
|
||||
if idx == limit:
|
||||
break
|
||||
return False
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
@@ -277,13 +275,14 @@ class MOL2(GenericMolFile):
|
||||
split = classmethod(split)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class FPS(GenericMolFile):
|
||||
"""
|
||||
chemfp fingerprint file: http://code.google.com/p/chem-fingerprints/wiki/FPS
|
||||
"""
|
||||
file_ext = "fps"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess if the file is a FPS file.
|
||||
|
||||
@@ -291,12 +290,11 @@ class FPS(GenericMolFile):
|
||||
>>> fname = get_test_fname('q.fps')
|
||||
>>> FPS().sniff(fname)
|
||||
True
|
||||
|
||||
>>> fname = get_test_fname('drugbank_drugs.cml')
|
||||
>>> FPS().sniff(fname)
|
||||
False
|
||||
"""
|
||||
header = get_headers(filename, sep='\t', count=1)
|
||||
header = get_headers(file_prefix, sep='\t', count=1)
|
||||
if header[0][0].strip() == '#FPS1':
|
||||
return True
|
||||
else:
|
||||
@@ -473,6 +471,7 @@ class PHAR(GenericMolFile):
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class PDB(GenericMolFile):
|
||||
"""
|
||||
Protein Databank format.
|
||||
@@ -480,7 +479,7 @@ class PDB(GenericMolFile):
|
||||
"""
|
||||
file_ext = "pdb"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess if the file is a PDB file.
|
||||
|
||||
@@ -488,12 +487,11 @@ class PDB(GenericMolFile):
|
||||
>>> fname = get_test_fname('5e5z.pdb')
|
||||
>>> PDB().sniff(fname)
|
||||
True
|
||||
|
||||
>>> fname = get_test_fname('drugbank_drugs.cml')
|
||||
>>> PDB().sniff(fname)
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep=' ', count=300)
|
||||
headers = iter_headers(file_prefix, sep=' ', count=300)
|
||||
h = t = c = s = k = e = False
|
||||
for line in headers:
|
||||
section_name = line[0].strip()
|
||||
@@ -526,6 +524,7 @@ class PDB(GenericMolFile):
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class PDBQT(GenericMolFile):
|
||||
"""
|
||||
PDBQT Autodock and Autodock Vina format
|
||||
@@ -533,7 +532,7 @@ class PDBQT(GenericMolFile):
|
||||
"""
|
||||
file_ext = "pdbqt"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess if the file is a PDBQT file.
|
||||
|
||||
@@ -541,12 +540,11 @@ class PDBQT(GenericMolFile):
|
||||
>>> fname = get_test_fname('NuBBE_1_obabel_3D.pdbqt')
|
||||
>>> PDBQT().sniff(fname)
|
||||
True
|
||||
|
||||
>>> fname = get_test_fname('drugbank_drugs.cml')
|
||||
>>> PDBQT().sniff(fname)
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep=' ', count=300)
|
||||
headers = iter_headers(file_prefix, sep=' ', count=300)
|
||||
h = t = c = s = k = False
|
||||
for line in headers:
|
||||
section_name = line[0].strip()
|
||||
@@ -601,6 +599,7 @@ class grdtgz(Binary):
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class InChI(Tabular):
|
||||
file_ext = "inchi"
|
||||
column_names = ['InChI']
|
||||
@@ -625,7 +624,7 @@ class InChI(Tabular):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess if the file is a InChI file.
|
||||
|
||||
@@ -633,16 +632,17 @@ class InChI(Tabular):
|
||||
>>> fname = get_test_fname('drugbank_drugs.inchi')
|
||||
>>> InChI().sniff(fname)
|
||||
True
|
||||
|
||||
>>> fname = get_test_fname('drugbank_drugs.cml')
|
||||
>>> InChI().sniff(fname)
|
||||
False
|
||||
"""
|
||||
inchi_lines = iter_headers(filename, sep=' ', count=10)
|
||||
inchi_lines = iter_headers(file_prefix, sep=' ', count=10)
|
||||
found_lines = False
|
||||
for inchi in inchi_lines:
|
||||
if not inchi[0].startswith('InChI='):
|
||||
return False
|
||||
return True
|
||||
found_lines = True
|
||||
return found_lines
|
||||
|
||||
|
||||
class SMILES(Tabular):
|
||||
@@ -704,6 +704,7 @@ class SMILES(Tabular):
|
||||
'''
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class CML(GenericXml):
|
||||
"""
|
||||
Chemical Markup Language
|
||||
@@ -729,7 +730,7 @@ class CML(GenericXml):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess if the file is a CML file.
|
||||
|
||||
@@ -737,18 +738,14 @@ class CML(GenericXml):
|
||||
>>> fname = get_test_fname('interval.interval')
|
||||
>>> CML().sniff(fname)
|
||||
False
|
||||
|
||||
>>> fname = get_test_fname('drugbank_drugs.cml')
|
||||
>>> CML().sniff(fname)
|
||||
True
|
||||
"""
|
||||
with open(filename) as handle:
|
||||
line = handle.readline()
|
||||
if line.strip() != '<?xml version="1.0"?>':
|
||||
return False
|
||||
line = handle.readline()
|
||||
if line.strip().find('http://www.xml-cml.org/schema') == -1:
|
||||
for expected_string in ['<?xml version="1.0"?>', 'http://www.xml-cml.org/schema']:
|
||||
if expected_string not in file_prefix.contents_header:
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def split(cls, input_datasets, subdir_generator_function, split_params):
|
||||
|
||||
@@ -8,6 +8,7 @@ import sys
|
||||
from galaxy.datatypes.data import Text
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import (
|
||||
build_sniff_from_prefix,
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
@@ -16,6 +17,7 @@ from galaxy.datatypes.tabular import Tabular
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Otu(Text):
|
||||
file_ext = 'mothur.otu'
|
||||
MetadataElement(name="columns", default=0, desc="Number of columns", readonly=True, visible=True, no_value=0)
|
||||
@@ -76,7 +78,7 @@ class Otu(Text):
|
||||
dataset.metadata.otulabels = list(otulabel_names)
|
||||
dataset.metadata.otulabels.sort()
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is otu (operational taxonomic unit) format
|
||||
|
||||
@@ -88,7 +90,7 @@ class Otu(Text):
|
||||
>>> Otu().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -120,7 +122,7 @@ class Sabund(Otu):
|
||||
def init_meta(self, dataset, copy_from=None):
|
||||
super(Sabund, self).init_meta(dataset, copy_from=copy_from)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is otu (operational taxonomic unit) format
|
||||
label<TAB>count[<TAB>value(1..n)]
|
||||
@@ -133,7 +135,7 @@ class Sabund(Otu):
|
||||
>>> Sabund().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -196,7 +198,7 @@ class GroupAbund(Otu):
|
||||
dataset.metadata.groups.sort()
|
||||
dataset.metadata.skip = skip
|
||||
|
||||
def sniff(self, filename, vals_are_int=False):
|
||||
def sniff_prefix(self, file_prefix, vals_are_int=False):
|
||||
"""
|
||||
Determines whether the file is a otu (operational taxonomic unit)
|
||||
Shared format
|
||||
@@ -211,7 +213,7 @@ class GroupAbund(Otu):
|
||||
>>> GroupAbund().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -235,6 +237,7 @@ class GroupAbund(Otu):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class SecondaryStructureMap(Tabular):
|
||||
file_ext = 'mothur.map'
|
||||
|
||||
@@ -243,7 +246,7 @@ class SecondaryStructureMap(Tabular):
|
||||
super(SecondaryStructureMap, self).__init__(**kwd)
|
||||
self.column_names = ['Map']
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a secondary structure map format
|
||||
A single column with an integer value which indicates the row that this
|
||||
@@ -258,7 +261,7 @@ class SecondaryStructureMap(Tabular):
|
||||
>>> SecondaryStructureMap().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
line_num = 0
|
||||
rowidxmap = {}
|
||||
for line in headers:
|
||||
@@ -337,6 +340,7 @@ class DistanceMatrix(Text):
|
||||
log.warning("DistanceMatrix set_meta %s" % e)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class LowerTriangleDistanceMatrix(DistanceMatrix):
|
||||
file_ext = 'mothur.lower.dist'
|
||||
|
||||
@@ -347,7 +351,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
|
||||
def init_meta(self, dataset, copy_from=None):
|
||||
super(LowerTriangleDistanceMatrix, self).init_meta(dataset, copy_from=copy_from)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a lower-triangle distance matrix (phylip) format
|
||||
The first line has the number of sequences in the matrix.
|
||||
@@ -368,7 +372,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
|
||||
False
|
||||
"""
|
||||
numlines = 300
|
||||
headers = iter_headers(filename, sep='\t', count=numlines)
|
||||
headers = iter_headers(file_prefix, sep='\t', count=numlines)
|
||||
line_num = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -400,6 +404,7 @@ class LowerTriangleDistanceMatrix(DistanceMatrix):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class SquareDistanceMatrix(DistanceMatrix):
|
||||
file_ext = 'mothur.square.dist'
|
||||
|
||||
@@ -409,7 +414,7 @@ class SquareDistanceMatrix(DistanceMatrix):
|
||||
def init_meta(self, dataset, copy_from=None):
|
||||
super(SquareDistanceMatrix, self).init_meta(dataset, copy_from=copy_from)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a square distance matrix (Column-formatted distance matrix) format
|
||||
The first line has the number of sequences in the matrix.
|
||||
@@ -429,7 +434,7 @@ class SquareDistanceMatrix(DistanceMatrix):
|
||||
False
|
||||
"""
|
||||
numlines = 300
|
||||
headers = iter_headers(filename, sep='\t', count=numlines)
|
||||
headers = iter_headers(file_prefix, sep='\t', count=numlines)
|
||||
line_num = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -460,6 +465,7 @@ class SquareDistanceMatrix(DistanceMatrix):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
|
||||
file_ext = 'mothur.pair.dist'
|
||||
|
||||
@@ -472,7 +478,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
|
||||
def set_meta(self, dataset, overwrite=True, skip=None, **kwd):
|
||||
super(PairwiseDistanceMatrix, self).set_meta(dataset, overwrite=overwrite, skip=skip, **kwd)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a pairwise distance matrix (Column-formatted distance matrix) format
|
||||
The first and second columns have the sequence names and the third column is the distance between those sequences.
|
||||
@@ -485,7 +491,7 @@ class PairwiseDistanceMatrix(DistanceMatrix, Tabular):
|
||||
>>> PairwiseDistanceMatrix().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -566,10 +572,11 @@ class AccNos(Tabular):
|
||||
self.columns = 1
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Oligos(Text):
|
||||
file_ext = 'mothur.oligos'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
http://www.mothur.org/wiki/Oligos_File
|
||||
Determines whether the file is a otu (operational taxonomic unit) format
|
||||
@@ -582,7 +589,7 @@ class Oligos(Text):
|
||||
>>> Oligos().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@') and not line[0].startswith('#'):
|
||||
@@ -600,6 +607,7 @@ class Oligos(Text):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Frequency(Tabular):
|
||||
file_ext = 'mothur.freq'
|
||||
|
||||
@@ -609,7 +617,7 @@ class Frequency(Tabular):
|
||||
self.column_names = ['position', 'frequency']
|
||||
self.column_types = ['int', 'float']
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a frequency tabular format for chimera analysis
|
||||
#1.14.0
|
||||
@@ -625,13 +633,12 @@ class Frequency(Tabular):
|
||||
>>> fname = get_test_fname( 'mothur_datatypetest_false.mothur.freq' )
|
||||
>>> Frequency().sniff( fname )
|
||||
False
|
||||
|
||||
# Expression count matrix (EdgeR wrapper)
|
||||
>>> # Expression count matrix (EdgeR wrapper)
|
||||
>>> fname = get_test_fname( 'mothur_datatypetest_false_2.mothur.freq' )
|
||||
>>> Frequency().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@'):
|
||||
@@ -660,6 +667,7 @@ class Frequency(Tabular):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Quantile(Tabular):
|
||||
file_ext = 'mothur.quan'
|
||||
MetadataElement(name="filtered", default=False, no_value=False, optional=True, desc="Quantiles calculated using a mask", readonly=True)
|
||||
@@ -671,7 +679,7 @@ class Quantile(Tabular):
|
||||
self.column_names = ['num', 'ten', 'twentyfive', 'fifty', 'seventyfive', 'ninetyfive', 'ninetynine']
|
||||
self.column_types = ['int', 'float', 'float', 'float', 'float', 'float', 'float']
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a quantiles tabular format for chimera analysis
|
||||
1 0 0 0 0 0 0
|
||||
@@ -687,7 +695,7 @@ class Quantile(Tabular):
|
||||
>>> Quantile().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
for line in headers:
|
||||
if not line[0].startswith('@') and not line[0].startswith('#'):
|
||||
@@ -710,10 +718,11 @@ class Quantile(Tabular):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class LaneMask(Text):
|
||||
file_ext = 'mothur.filter'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a lane mask filter: 1 line consisting of zeros and ones.
|
||||
|
||||
@@ -725,7 +734,7 @@ class LaneMask(Text):
|
||||
>>> LaneMask().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, sep='\t', count=2)
|
||||
headers = get_headers(file_prefix, sep='\t', count=2)
|
||||
if len(headers) != 1 or len(headers[0]) != 1:
|
||||
return False
|
||||
|
||||
@@ -775,6 +784,7 @@ class CountTable(Tabular):
|
||||
dataset.metadata.data_lines -= 1
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class RefTaxonomy(Tabular):
|
||||
file_ext = 'mothur.ref.taxonomy'
|
||||
|
||||
@@ -782,7 +792,7 @@ class RefTaxonomy(Tabular):
|
||||
super(RefTaxonomy, self).__init__(**kwd)
|
||||
self.column_names = ['name', 'taxonomy']
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is a Reference Taxonomy
|
||||
|
||||
@@ -808,7 +818,7 @@ class RefTaxonomy(Tabular):
|
||||
>>> RefTaxonomy().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t', count=300)
|
||||
headers = iter_headers(file_prefix, sep='\t', count=300)
|
||||
count = 0
|
||||
pat_prog = re.compile('^([^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?(;[^ \t\n\r\x0c\x0b;]+([(]\\d+[)])?)*(;)?)$')
|
||||
found_semicolons = False
|
||||
@@ -852,6 +862,7 @@ class TaxonomySummary(Tabular):
|
||||
self.column_names = ['taxlevel', 'rankID', 'taxon', 'daughterlevels', 'total']
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Axes(Tabular):
|
||||
file_ext = 'mothur.axes'
|
||||
|
||||
@@ -859,7 +870,7 @@ class Axes(Tabular):
|
||||
"""Initialize axes datatype"""
|
||||
super(Axes, self).__init__(**kwd)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is an axes format
|
||||
The first line may have column headings.
|
||||
@@ -883,7 +894,7 @@ class Axes(Tabular):
|
||||
>>> Axes().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, sep='\t')
|
||||
headers = iter_headers(file_prefix, sep='\t')
|
||||
count = 0
|
||||
col_cnt = None
|
||||
all_integers = True
|
||||
|
||||
+20
-24
@@ -1,16 +1,21 @@
|
||||
import abc
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
|
||||
from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.data import get_file_peek, Text
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
from galaxy.datatypes.util import generic_util
|
||||
from galaxy.util import nice_size
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
STOCKHOLM_SEARCH_PATTERN = re.compile(r'#\s+STOCKHOLM\s+1\.0')
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class InfernalCM(Text):
|
||||
file_ext = "cm"
|
||||
|
||||
@@ -32,20 +37,17 @@ class InfernalCM(Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
>>> fname = get_test_fname( 'infernal_model.cm' )
|
||||
>>> InfernalCM().sniff( fname )
|
||||
True
|
||||
>>> fname = get_test_fname( 'test.mz5' )
|
||||
>>> fname = get_test_fname( '2.txt' )
|
||||
>>> InfernalCM().sniff( fname )
|
||||
False
|
||||
"""
|
||||
with open(filename, 'r') as f:
|
||||
first_line = f.readline()
|
||||
|
||||
return first_line.startswith("INFERNAL")
|
||||
return file_prefix.startswith("INFERNAL")
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
"""
|
||||
@@ -58,6 +60,7 @@ class InfernalCM(Text):
|
||||
dataset.metadata.cm_version = (first_line.split()[0]).replace('INFERNAL', '')
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Hmmer(Text):
|
||||
edam_data = "data_1364"
|
||||
edam_format = "format_1370"
|
||||
@@ -77,7 +80,7 @@ class Hmmer(Text):
|
||||
return "HMMER database (%s)" % (nice_size(dataset.get_size()))
|
||||
|
||||
@abc.abstractmethod
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, filename):
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
@@ -85,24 +88,20 @@ class Hmmer2(Hmmer):
|
||||
edam_format = "format_3328"
|
||||
file_ext = "hmm2"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""HMMER2 files start with HMMER2.0
|
||||
"""
|
||||
with open(filename, 'r') as handle:
|
||||
return handle.read(8) == 'HMMER2.0'
|
||||
return False
|
||||
return file_prefix.startswith('HMMER2.0')
|
||||
|
||||
|
||||
class Hmmer3(Hmmer):
|
||||
edam_format = "format_3329"
|
||||
file_ext = "hmm3"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""HMMER3 files start with HMMER3/f
|
||||
"""
|
||||
with open(filename, 'r') as handle:
|
||||
return handle.read(8) == 'HMMER3/f'
|
||||
return False
|
||||
return file_prefix.startswith('HMMER3/f')
|
||||
|
||||
|
||||
class HmmerPress(Binary):
|
||||
@@ -139,6 +138,7 @@ class HmmerPress(Binary):
|
||||
self.add_composite_file('model.hmm.h3p', is_binary=True)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Stockholm_1_0(Text):
|
||||
edam_data = "data_0863"
|
||||
edam_format = "format_1961"
|
||||
@@ -157,11 +157,8 @@ class Stockholm_1_0(Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
if generic_util.count_special_lines('^#[[:space:]+]STOCKHOLM[[:space:]+]1.0', filename) > 0:
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
def sniff_prefix(self, file_prefix):
|
||||
return file_prefix.search(STOCKHOLM_SEARCH_PATTERN)
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
"""
|
||||
@@ -222,6 +219,7 @@ class Stockholm_1_0(Text):
|
||||
split = classmethod(split)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class MauveXmfa(Text):
|
||||
file_ext = "xmfa"
|
||||
|
||||
@@ -238,10 +236,8 @@ class MauveXmfa(Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
with open(filename, 'r') as handle:
|
||||
return handle.read(21) == '#FormatVersion Mauve1'
|
||||
return False
|
||||
def sniff_prefix(self, file_prefix):
|
||||
return file_prefix.startswith('#FormatVersion Mauve1')
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
dataset.metadata.number_of_models = generic_util.count_special_lines('^#Sequence([[:digit:]]+)Entry', dataset.file_name)
|
||||
|
||||
@@ -9,10 +9,12 @@ Phylip datatype sniffer
|
||||
"""
|
||||
from galaxy import util
|
||||
from galaxy.datatypes.data import get_file_peek, Text
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
from galaxy.util import nice_size
|
||||
from .metadata import MetadataElement
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Phylip(Text):
|
||||
"""Phylip format stores a multiple sequence alignment"""
|
||||
edam_data = "data_0863"
|
||||
@@ -44,7 +46,7 @@ class Phylip(Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
All Phylip files starts with the number of sequences so we can use this
|
||||
to count the following number of sequences in the first 'stack'
|
||||
@@ -54,15 +56,15 @@ class Phylip(Text):
|
||||
>>> Phylip().sniff(fname)
|
||||
True
|
||||
"""
|
||||
with open(filename, "r") as f:
|
||||
# Get number of sequence from first line
|
||||
nb_seq = int(f.readline().split()[0])
|
||||
# counts number of sequence from first stack
|
||||
count = 0
|
||||
for line in f:
|
||||
if not line.split():
|
||||
break
|
||||
count += 1
|
||||
if count > nb_seq:
|
||||
return False
|
||||
f = file_prefix.string_io()
|
||||
# Get number of sequence from first line
|
||||
nb_seq = int(f.readline().split()[0])
|
||||
# counts number of sequence from first stack
|
||||
count = 0
|
||||
for line in f:
|
||||
if not line.split():
|
||||
break
|
||||
count += 1
|
||||
if count > nb_seq:
|
||||
return False
|
||||
return count == nb_seq
|
||||
|
||||
@@ -3,13 +3,14 @@ import re
|
||||
|
||||
from galaxy.datatypes.data import get_file_peek, Text
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import get_headers
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix, get_headers
|
||||
from galaxy.datatypes.tabular import Tabular
|
||||
from galaxy.util import nice_size
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Smat(Text):
|
||||
file_ext = "smat"
|
||||
|
||||
@@ -27,7 +28,7 @@ class Smat(Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
The use of ESTScan implies the creation of scores matrices which
|
||||
reflect the codons preferences in the studied organisms. The
|
||||
@@ -51,26 +52,27 @@ class Smat(Text):
|
||||
True
|
||||
"""
|
||||
line_no = 0
|
||||
with open(filename, "r") as fh:
|
||||
for line in fh:
|
||||
line_no += 1
|
||||
if line_no > 10000:
|
||||
return True
|
||||
if line_no == 1 and not line.startswith('FORMAT'):
|
||||
# The first line is always the start of a format section.
|
||||
fh = file_prefix.string_io()
|
||||
for line in fh:
|
||||
line_no += 1
|
||||
if line_no > 10000:
|
||||
return True
|
||||
if line_no == 1 and not line.startswith('FORMAT'):
|
||||
# The first line is always the start of a format section.
|
||||
return False
|
||||
if not line.startswith('FORMAT'):
|
||||
if line.find('\t') >= 0:
|
||||
# Smat files are not tabular.
|
||||
return False
|
||||
if not line.startswith('FORMAT'):
|
||||
if line.find('\t') >= 0:
|
||||
# Smat files are not tabular.
|
||||
items = line.split()
|
||||
if len(items) != 4:
|
||||
return False
|
||||
for item in items:
|
||||
# Make sure each item is an integer.
|
||||
if re.match(r"[-+]?\d+$", item) is None:
|
||||
return False
|
||||
items = line.split()
|
||||
if len(items) != 4:
|
||||
return False
|
||||
for item in items:
|
||||
# Make sure each item is an integer.
|
||||
if re.match(r"[-+]?\d+$", item) is None:
|
||||
return False
|
||||
return True
|
||||
# Ensure at least a few matching lines are found.
|
||||
return line_no > 2
|
||||
|
||||
|
||||
# These commented classes are required by versions 1.0.0, 1.0.1 and 1.0.2 of the
|
||||
|
||||
@@ -7,6 +7,7 @@ import re
|
||||
from galaxy.datatypes import data
|
||||
from galaxy.datatypes.binary import Binary
|
||||
from galaxy.datatypes.data import Text
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix
|
||||
from galaxy.datatypes.tabular import Tabular
|
||||
from galaxy.datatypes.xml import GenericXml
|
||||
from galaxy.util import nice_size
|
||||
@@ -97,16 +98,16 @@ class ProteomicsXml(GenericXml):
|
||||
edam_data = "data_2536"
|
||||
edam_format = "format_2032"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
""" Determines whether the file is the correct XML type. """
|
||||
with open(filename, 'r') as contents:
|
||||
while True:
|
||||
line = contents.readline()
|
||||
if line is None or not line.startswith('<?'):
|
||||
break
|
||||
# pattern match <root or <ns:root for any ns string
|
||||
pattern = '^<(\w*:)?%s' % self.root
|
||||
return line is not None and re.match(pattern, line) is not None
|
||||
contents = file_prefix.string_io()
|
||||
while True:
|
||||
line = contents.readline()
|
||||
if line is None or not line.startswith('<?'):
|
||||
break
|
||||
# pattern match <root or <ns:root for any ns string
|
||||
pattern = '^<(\w*:)?%s' % self.root
|
||||
return line is not None and re.match(pattern, line) is not None
|
||||
|
||||
def set_peek(self, dataset, is_multi_byte=False):
|
||||
"""Set the peek and blurb text"""
|
||||
@@ -300,6 +301,7 @@ class ThermoRAW(Binary):
|
||||
return "Thermo Finnigan RAW file (%s)" % (nice_size(dataset.get_size()))
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Msp(Text):
|
||||
""" Output of NIST MS Search Program chemdata.nist.gov/mass-spc/ftp/mass-spc/PepLib.pdf """
|
||||
file_ext = "msp"
|
||||
@@ -309,16 +311,15 @@ class Msp(Text):
|
||||
next_line = contents.readline()
|
||||
return next_line is not None and next_line.startswith(prefix)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
""" Determines whether the file is a NIST MSP output file."""
|
||||
with open(filename, 'r') as f:
|
||||
begin_contents = f.read(1024)
|
||||
if "\n" not in begin_contents:
|
||||
return False
|
||||
lines = begin_contents.splitlines()
|
||||
if len(lines) < 2:
|
||||
return False
|
||||
return lines[0].startswith("Name:") and lines[1].startswith("MW:")
|
||||
begin_contents = file_prefix.contents_header
|
||||
if "\n" not in begin_contents:
|
||||
return False
|
||||
lines = begin_contents.splitlines()
|
||||
if len(lines) < 2:
|
||||
return False
|
||||
return lines[0].startswith("Name:") and lines[1].startswith("MW:")
|
||||
|
||||
|
||||
class SPLibNoIndex(Text):
|
||||
@@ -335,6 +336,7 @@ class SPLibNoIndex(Text):
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class SPLib(Msp):
|
||||
"""SpectraST Spectral Library. Closely related to msp format"""
|
||||
file_ext = "splib"
|
||||
@@ -374,29 +376,29 @@ class SPLib(Msp):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
""" Determines whether the file is a SpectraST generated file.
|
||||
"""
|
||||
with open(filename, 'r') as contents:
|
||||
return Msp.next_line_starts_with(contents, "Name:") and Msp.next_line_starts_with(contents, "LibID:")
|
||||
contents = file_prefix.string_io()
|
||||
return Msp.next_line_starts_with(contents, "Name:") and Msp.next_line_starts_with(contents, "LibID:")
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Ms2(Text):
|
||||
file_ext = "ms2"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
""" Determines whether the file is a valid ms2 file."""
|
||||
|
||||
with open(filename, 'r') as contents:
|
||||
header_lines = []
|
||||
while True:
|
||||
line = contents.readline()
|
||||
if line is None or len(line) == 0:
|
||||
pass
|
||||
elif line.startswith('H\t'):
|
||||
header_lines.append(line)
|
||||
else:
|
||||
break
|
||||
contents = file_prefix.string_io()
|
||||
header_lines = []
|
||||
while True:
|
||||
line = contents.readline()
|
||||
if line is None or len(line) == 0:
|
||||
pass
|
||||
elif line.startswith('H\t'):
|
||||
header_lines.append(line)
|
||||
else:
|
||||
break
|
||||
|
||||
for header_field in ['CreationDate', 'Extractor', 'ExtractorVersion', 'ExtractorOptions']:
|
||||
found_header = False
|
||||
|
||||
@@ -3,7 +3,10 @@ Qualityscore class
|
||||
"""
|
||||
import logging
|
||||
|
||||
from . import data
|
||||
from . import (
|
||||
data,
|
||||
sniff
|
||||
)
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -17,6 +20,7 @@ class QualityScore(data.Text):
|
||||
file_ext = "qual"
|
||||
|
||||
|
||||
@sniff.build_sniff_from_prefix
|
||||
class QualityScoreSOLiD(QualityScore):
|
||||
"""
|
||||
until we know more about quality score formats
|
||||
@@ -24,7 +28,7 @@ class QualityScoreSOLiD(QualityScore):
|
||||
edam_format = "format_3610"
|
||||
file_ext = "qualsolid"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
>>> fname = get_test_fname( 'sequence.fasta' )
|
||||
@@ -34,34 +38,34 @@ class QualityScoreSOLiD(QualityScore):
|
||||
>>> QualityScoreSOLiD().sniff( fname )
|
||||
True
|
||||
"""
|
||||
with open(filename) as fh:
|
||||
readlen = None
|
||||
goodblock = 0
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
if goodblock > 0:
|
||||
fh = file_prefix.string_io()
|
||||
readlen = None
|
||||
goodblock = 0
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
if goodblock > 0:
|
||||
return True
|
||||
else:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'): # first non-empty non-comment line
|
||||
if line.startswith('>'):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
try:
|
||||
[int(x) for x in line.split()]
|
||||
if not(readlen):
|
||||
readlen = len(line.split())
|
||||
assert len(line.split()) == readlen # SOLiD reads should be of the same length
|
||||
except Exception:
|
||||
break
|
||||
goodblock += 1
|
||||
if goodblock > 10:
|
||||
return True
|
||||
else:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'): # first non-empty non-comment line
|
||||
if line.startswith('>'):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
try:
|
||||
[int(x) for x in line.split()]
|
||||
if not(readlen):
|
||||
readlen = len(line.split())
|
||||
assert len(line.split()) == readlen # SOLiD reads should be of the same length
|
||||
except Exception:
|
||||
break
|
||||
goodblock += 1
|
||||
if goodblock > 10:
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a header
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a header
|
||||
return False
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
@@ -71,6 +75,7 @@ class QualityScoreSOLiD(QualityScore):
|
||||
return QualityScore.set_meta(self, dataset, **kwd)
|
||||
|
||||
|
||||
@sniff.build_sniff_from_prefix
|
||||
class QualityScore454(QualityScore):
|
||||
"""
|
||||
until we know more about quality score formats
|
||||
@@ -78,7 +83,7 @@ class QualityScore454(QualityScore):
|
||||
edam_format = "format_3611"
|
||||
file_ext = "qual454"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
>>> fname = get_test_fname( 'sequence.fasta' )
|
||||
@@ -88,24 +93,24 @@ class QualityScore454(QualityScore):
|
||||
>>> QualityScore454().sniff( fname )
|
||||
True
|
||||
"""
|
||||
with open(filename) as fh:
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'): # first non-empty non-comment line
|
||||
if line.startswith('>'):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
try:
|
||||
[int(x) for x in line.split()]
|
||||
except Exception:
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a header
|
||||
fh = file_prefix.string_io()
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'): # first non-empty non-comment line
|
||||
if line.startswith('>'):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
try:
|
||||
[int(x) for x in line.split()]
|
||||
except Exception:
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a header
|
||||
return False
|
||||
|
||||
|
||||
|
||||
+120
-116
@@ -22,8 +22,9 @@ from galaxy.datatypes.binary import (
|
||||
)
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import (
|
||||
build_sniff_from_prefix,
|
||||
get_headers,
|
||||
iter_headers
|
||||
iter_headers,
|
||||
)
|
||||
from galaxy.util import (
|
||||
compression_utils,
|
||||
@@ -46,6 +47,7 @@ SNIFF_COMPRESSED_FASTAS = os.environ.get("GALAXY_ENABLE_BETA_COMPRESSED_FASTA_SN
|
||||
SNIFF_COMPRESSED_GENBANKS = os.environ.get("GALAXY_ENABLE_BETA_COMPRESSED_GENBANK_SNIFFING", "0") == "1"
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class SequenceSplitLocations(data.Text):
|
||||
"""
|
||||
Class storing information about a sequence file composed of multiple gzip files concatenated as
|
||||
@@ -75,10 +77,10 @@ class SequenceSplitLocations(data.Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
if os.path.getsize(filename) < 50000:
|
||||
def sniff_prefix(self, file_prefix):
|
||||
if file_prefix.file_size < 50000 and not file_prefix.truncated:
|
||||
try:
|
||||
data = json.load(open(filename))
|
||||
data = json.loads(file_prefix.contents_header)
|
||||
sections = data['sections']
|
||||
for section in sections:
|
||||
if 'start' not in section or 'end' not in section or 'sequences' not in section:
|
||||
@@ -330,12 +332,13 @@ class FastaGz(Sequence, CompressedArchive):
|
||||
return Sequence.sniff(self, filename)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Fasta(Sequence):
|
||||
"""Class representing a FASTA sequence"""
|
||||
edam_format = "format_1929"
|
||||
file_ext = "fasta"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in fasta format
|
||||
|
||||
@@ -369,26 +372,26 @@ class Fasta(Sequence):
|
||||
>>> Fasta().sniff( fname )
|
||||
True
|
||||
"""
|
||||
with open(filename) as fh:
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if line.startswith('>'):
|
||||
# The next line.strip() must not be '', nor startwith '>'
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
fh = file_prefix.string_io()
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line: # first non-empty line
|
||||
if line.startswith('>'):
|
||||
# The next line.strip() must not be '', nor startwith '>'
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
|
||||
# If there is a third line, and it isn't a header line, it may not contain chars like '()[].' otherwise it's most likely a DotBracket file
|
||||
line = fh.readline()
|
||||
if not line.startswith('>') and re.search("[\(\)\[\]\.]", line):
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a fasta header
|
||||
# If there is a third line, and it isn't a header line, it may not contain chars like '()[].' otherwise it's most likely a DotBracket file
|
||||
line = fh.readline()
|
||||
if not line.startswith('>') and re.search("[\(\)\[\]\.]", line):
|
||||
break
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a fasta header
|
||||
return False
|
||||
|
||||
def split(cls, input_datasets, subdir_generator_function, split_params):
|
||||
@@ -514,12 +517,13 @@ class Fasta(Sequence):
|
||||
_count_split = classmethod(_count_split)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class csFasta(Sequence):
|
||||
""" Class representing the SOLID Color-Space sequence ( csfasta ) """
|
||||
edam_format = "format_3589"
|
||||
file_ext = "csfasta"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Color-space sequence:
|
||||
>2_15_85_F3
|
||||
@@ -533,24 +537,24 @@ class csFasta(Sequence):
|
||||
>>> csFasta().sniff( fname )
|
||||
True
|
||||
"""
|
||||
with open(filename) as fh:
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'): # first non-empty non-comment line
|
||||
if line.startswith('>'):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
elif line[0] not in string.ascii_uppercase:
|
||||
return False
|
||||
elif len(line) > 1 and not re.search('^[\d.]+$', line[1:]):
|
||||
return False
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a header
|
||||
fh = file_prefix.string_io()
|
||||
while True:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
break # EOF
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'): # first non-empty non-comment line
|
||||
if line.startswith('>'):
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
break
|
||||
elif line[0] not in string.ascii_uppercase:
|
||||
return False
|
||||
elif len(line) > 1 and not re.search('^[\d.]+$', line[1:]):
|
||||
return False
|
||||
return True
|
||||
else:
|
||||
break # we found a non-empty line, but it's not a header
|
||||
return False
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
@@ -561,6 +565,7 @@ class csFasta(Sequence):
|
||||
return Sequence.set_meta(self, dataset, **kwd)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class BaseFastq(Sequence):
|
||||
"""Base class for FastQ sequences"""
|
||||
edam_format = "format_1930"
|
||||
@@ -599,7 +604,7 @@ class BaseFastq(Sequence):
|
||||
dataset.metadata.data_lines = data_lines
|
||||
dataset.metadata.sequences = sequences
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in generic fastq format
|
||||
For details, see http://maq.sourceforge.net/fastq.shtml
|
||||
@@ -620,11 +625,10 @@ class BaseFastq(Sequence):
|
||||
>>> FastqSanger().sniff( fname )
|
||||
False
|
||||
"""
|
||||
compressed = is_gzip(filename) or is_bz2(filename)
|
||||
compressed = file_prefix.compressed_format is not None
|
||||
if compressed and not isinstance(self, Binary):
|
||||
return False
|
||||
headers = iter_headers(filename, None, count=1000)
|
||||
|
||||
headers = iter_headers(file_prefix, None, count=1000)
|
||||
# If this is a FastqSanger-derived class, then check to see if the base qualities match
|
||||
if isinstance(self, FastqSanger) or isinstance(self, FastqSangerGz) or isinstance(self, FastqSangerBz2):
|
||||
if not self.sangerQualities(headers):
|
||||
@@ -633,7 +637,7 @@ class BaseFastq(Sequence):
|
||||
bases_regexp = re.compile("^[NGTAC]*")
|
||||
# check that first block looks like a fastq block
|
||||
try:
|
||||
headers = get_headers(filename, None, count=4)
|
||||
headers = get_headers(file_prefix, None, count=4)
|
||||
if len(headers) == 4 and headers[0][0] and headers[0][0][0] == "@" and headers[2][0] and headers[2][0][0] == "+" and headers[1][0]:
|
||||
# Check the sequence line, make sure it contains only G/C/A/T/N
|
||||
if not bases_regexp.match(headers[1][0]):
|
||||
@@ -818,6 +822,7 @@ class FastqCSSangerBz2(FastqBz2):
|
||||
file_ext = "fastqcssanger.bz2"
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Maf(Alignment):
|
||||
"""Class describing a Maf alignment"""
|
||||
edam_format = "format_3008"
|
||||
@@ -900,7 +905,7 @@ class Maf(Alignment):
|
||||
out = "Can't create peek %s" % exc
|
||||
return out
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines wether the file is in maf format
|
||||
|
||||
@@ -923,7 +928,7 @@ class Maf(Alignment):
|
||||
>>> Maf().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, None)
|
||||
headers = get_headers(file_prefix, None)
|
||||
try:
|
||||
if len(headers) > 1 and headers[0][0] and headers[0][0] == "##maf":
|
||||
return True
|
||||
@@ -971,6 +976,7 @@ class MafCustomTrack(data.Text):
|
||||
pass
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Axt(data.Text):
|
||||
"""Class describing an axt alignment"""
|
||||
# gvk- 11/19/09 - This is really an alignment, but we no longer have tools that use this data type, and it is
|
||||
@@ -981,7 +987,7 @@ class Axt(data.Text):
|
||||
edam_format = "format_3013"
|
||||
file_ext = "axt"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in axt format
|
||||
|
||||
@@ -1007,7 +1013,7 @@ class Axt(data.Text):
|
||||
>>> Axt().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, None)
|
||||
headers = get_headers(file_prefix, None)
|
||||
if len(headers) < 4:
|
||||
return False
|
||||
for hdr in headers:
|
||||
@@ -1026,6 +1032,7 @@ class Axt(data.Text):
|
||||
return True
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Lav(data.Text):
|
||||
"""Class describing a LAV alignment"""
|
||||
# gvk- 11/19/09 - This is really an alignment, but we no longer have tools that use this data type, and it is
|
||||
@@ -1036,7 +1043,7 @@ class Lav(data.Text):
|
||||
edam_format = "format_3014"
|
||||
file_ext = "lav"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in lav format
|
||||
|
||||
@@ -1053,7 +1060,7 @@ class Lav(data.Text):
|
||||
>>> Lav().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = get_headers(filename, None)
|
||||
headers = get_headers(file_prefix, None)
|
||||
try:
|
||||
if len(headers) > 1 and headers[0][0] and headers[0][0].startswith('#:lav'):
|
||||
return True
|
||||
@@ -1096,6 +1103,7 @@ class RNADotPlotMatrix(data.Data):
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class DotBracket(Sequence):
|
||||
edam_data = "data_0880"
|
||||
edam_format = "format_1457"
|
||||
@@ -1128,7 +1136,7 @@ class DotBracket(Sequence):
|
||||
dataset.metadata.data_lines = data_lines
|
||||
dataset.metadata.sequences = sequences
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Galaxy Dbn (Dot-Bracket notation) rules:
|
||||
|
||||
@@ -1160,47 +1168,47 @@ class DotBracket(Sequence):
|
||||
|
||||
state = 0
|
||||
|
||||
with open(filename, "r") as handle:
|
||||
for line in handle:
|
||||
line = line.strip()
|
||||
for line in file_prefix.line_iterator():
|
||||
line = line.strip()
|
||||
|
||||
if line:
|
||||
# header line
|
||||
if state == 0:
|
||||
if(line[0] != '>'):
|
||||
return False
|
||||
else:
|
||||
state = 1
|
||||
if line:
|
||||
# header line
|
||||
if state == 0:
|
||||
if(line[0] != '>'):
|
||||
return False
|
||||
else:
|
||||
state = 1
|
||||
|
||||
# sequence line
|
||||
elif state == 1:
|
||||
if not self.sequence_regexp.match(line):
|
||||
return False
|
||||
else:
|
||||
sequence_size = len(line)
|
||||
state = 2
|
||||
# sequence line
|
||||
elif state == 1:
|
||||
if not self.sequence_regexp.match(line):
|
||||
return False
|
||||
else:
|
||||
sequence_size = len(line)
|
||||
state = 2
|
||||
|
||||
# dot-bracket structure line
|
||||
elif state == 2:
|
||||
if sequence_size != len(line) or not self.structure_regexp.match(line) or \
|
||||
line.count('(') != line.count(')') or \
|
||||
line.count('[') != line.count(']') or \
|
||||
line.count('{') != line.count('}'):
|
||||
return False
|
||||
else:
|
||||
return True
|
||||
# dot-bracket structure line
|
||||
elif state == 2:
|
||||
if sequence_size != len(line) or not self.structure_regexp.match(line) or \
|
||||
line.count('(') != line.count(')') or \
|
||||
line.count('[') != line.count(']') or \
|
||||
line.count('{') != line.count('}'):
|
||||
return False
|
||||
else:
|
||||
return True
|
||||
|
||||
# Number of lines is less than 3
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Genbank(data.Text):
|
||||
"""Class representing a Genbank sequence"""
|
||||
edam_format = "format_1936"
|
||||
edam_data = "data_0849"
|
||||
file_ext = "genbank"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determine whether the file is in genbank format.
|
||||
Works for compressed files.
|
||||
@@ -1210,15 +1218,10 @@ class Genbank(data.Text):
|
||||
>>> Genbank().sniff( fname )
|
||||
True
|
||||
"""
|
||||
compressed = is_gzip(filename)
|
||||
compressed = file_prefix.compressed_format
|
||||
if compressed and not isinstance(self, Binary):
|
||||
return False
|
||||
try:
|
||||
with compression_utils.get_fileobj(filename) as file:
|
||||
return 'LOCUS ' == file.read(6)
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
return 'LOCUS ' == file_prefix.contents_header[0:6]
|
||||
|
||||
|
||||
class GenbankGz(Genbank, CompressedArchive):
|
||||
@@ -1244,11 +1247,12 @@ class GenbankGz(Genbank, CompressedArchive):
|
||||
return Genbank.sniff(self, filename)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class MemePsp(Sequence):
|
||||
"""Class representing MEME Position Specific Priors"""
|
||||
file_ext = "memepsp"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
The format of an entry in a PSP file is:
|
||||
|
||||
@@ -1274,34 +1278,34 @@ class MemePsp(Sequence):
|
||||
return True
|
||||
try:
|
||||
num_lines = 0
|
||||
with open(filename) as fh:
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
# EOF.
|
||||
return False
|
||||
num_lines += 1
|
||||
if num_lines > 100:
|
||||
return True
|
||||
line = line.strip()
|
||||
if line:
|
||||
if line.startswith('>'):
|
||||
# The line must not be blank, nor start with '>'
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
return False
|
||||
# All items within the line must be floats.
|
||||
fh = file_prefix.string_io()
|
||||
line = fh.readline()
|
||||
if not line:
|
||||
# EOF.
|
||||
return False
|
||||
num_lines += 1
|
||||
if num_lines > 100:
|
||||
return True
|
||||
line = line.strip()
|
||||
if line:
|
||||
if line.startswith('>'):
|
||||
# The line must not be blank, nor start with '>'
|
||||
line = fh.readline().strip()
|
||||
if line == '' or line.startswith('>'):
|
||||
return False
|
||||
# All items within the line must be floats.
|
||||
if not floats_verified(line):
|
||||
return False
|
||||
# If there is a second line within the ID section,
|
||||
# all items within the line must be floats.
|
||||
line = fh.readline().strip()
|
||||
if line:
|
||||
if not floats_verified(line):
|
||||
return False
|
||||
# If there is a second line within the ID section,
|
||||
# all items within the line must be floats.
|
||||
line = fh.readline().strip()
|
||||
if line:
|
||||
if not floats_verified(line):
|
||||
return False
|
||||
else:
|
||||
# We found a non-empty line,
|
||||
# but it's not a psp id width.
|
||||
return False
|
||||
else:
|
||||
# We found a non-empty line,
|
||||
# but it's not a psp id width.
|
||||
return False
|
||||
except Exception:
|
||||
return False
|
||||
# We've reached EOF in less than 100 lines.
|
||||
|
||||
+235
-41
@@ -13,7 +13,7 @@ import sys
|
||||
import tempfile
|
||||
import zipfile
|
||||
|
||||
from six import text_type
|
||||
from six import StringIO, text_type
|
||||
from six.moves import filter
|
||||
from six.moves.urllib.request import urlopen
|
||||
|
||||
@@ -35,6 +35,8 @@ else:
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
SNIFF_PREFIX_BYTES = int(os.environ.get("GALAXY_SNIFF_PREFIX_BYTES", None) or 2 ** 20)
|
||||
|
||||
|
||||
def get_test_fname(fname):
|
||||
"""Returns test data filename"""
|
||||
@@ -189,10 +191,10 @@ def convert_newlines_sep2tabs(fname, in_place=True, patt="\\s+", tmp_dir=None, t
|
||||
return (i + 1, temp_name)
|
||||
|
||||
|
||||
def iter_headers(fname, sep, count=60, comment_designator=None):
|
||||
with compression_utils.get_fileobj(fname) as in_file:
|
||||
def iter_headers(fname_or_file_prefix, sep, count=60, comment_designator=None):
|
||||
if isinstance(fname_or_file_prefix, FilePrefix):
|
||||
idx = 0
|
||||
for line in in_file:
|
||||
for line in fname_or_file_prefix.line_iterator():
|
||||
line = line.rstrip('\n\r')
|
||||
if comment_designator is not None and comment_designator != '' and line.startswith(comment_designator):
|
||||
continue
|
||||
@@ -200,9 +202,20 @@ def iter_headers(fname, sep, count=60, comment_designator=None):
|
||||
idx += 1
|
||||
if idx == count:
|
||||
break
|
||||
else:
|
||||
with compression_utils.get_fileobj(fname_or_file_prefix) as in_file:
|
||||
idx = 0
|
||||
for line in in_file:
|
||||
line = line.rstrip('\n\r')
|
||||
if comment_designator is not None and comment_designator != '' and line.startswith(comment_designator):
|
||||
continue
|
||||
yield line.split(sep)
|
||||
idx += 1
|
||||
if idx == count:
|
||||
break
|
||||
|
||||
|
||||
def get_headers(fname, sep, count=60, comment_designator=None):
|
||||
def get_headers(fname_or_file_prefix, sep, count=60, comment_designator=None):
|
||||
"""
|
||||
Returns a list with the first 'count' lines split by 'sep', ignoring lines
|
||||
starting with 'comment_designator'
|
||||
@@ -214,10 +227,10 @@ def get_headers(fname, sep, count=60, comment_designator=None):
|
||||
>>> get_headers(fname, '\\t', count=5, comment_designator='#') == [[''], ['chr7', 'bed2gff', 'AR', '26731313', '26731437', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731491', '26731536', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731541', '26731649', '.', '+', '.', 'score'], ['chr7', 'bed2gff', 'AR', '26731659', '26731841', '.', '+', '.', 'score']]
|
||||
True
|
||||
"""
|
||||
return list(iter_headers(fname=fname, sep=sep, count=count, comment_designator=comment_designator))
|
||||
return list(iter_headers(fname_or_file_prefix=fname_or_file_prefix, sep=sep, count=count, comment_designator=comment_designator))
|
||||
|
||||
|
||||
def is_column_based(fname, sep='\t', skip=0):
|
||||
def is_column_based(fname_or_file_prefix, sep='\t', skip=0):
|
||||
"""
|
||||
Checks whether the file is column based with respect to a separator
|
||||
(defaults to tab separator).
|
||||
@@ -245,8 +258,11 @@ def is_column_based(fname, sep='\t', skip=0):
|
||||
>>> is_column_based(fname)
|
||||
True
|
||||
"""
|
||||
if getattr(fname_or_file_prefix, "binary", None) is True:
|
||||
return False
|
||||
|
||||
try:
|
||||
headers = get_headers(fname, sep)
|
||||
headers = get_headers(fname_or_file_prefix, sep)
|
||||
except UnicodeDecodeError:
|
||||
return False
|
||||
count = 0
|
||||
@@ -306,17 +322,14 @@ def guess_ext(fname, sniff_order, is_binary=False):
|
||||
>>> fname = get_test_fname('gff_version_3.gff')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'gff3'
|
||||
>>> fname = get_test_fname('temp.txt')
|
||||
>>> open(fname, 'wt').write("a\\t2")
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
>>> fname = get_test_fname('2.txt')
|
||||
>>> guess_ext(fname, sniff_order) # 2.txt
|
||||
'txt'
|
||||
>>> fname = get_test_fname('temp.txt')
|
||||
>>> open(fname, 'wt').write("a\\t2\\nc\\t1\\nd\\t0")
|
||||
>>> fname = get_test_fname('2.tabular')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'tabular'
|
||||
>>> fname = get_test_fname('temp.txt')
|
||||
>>> open(fname, 'wt').write("a 1 2 x\\nb 3 4 y\\nc 5 6 z")
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
>>> fname = get_test_fname('3.txt')
|
||||
>>> guess_ext(fname, sniff_order) # 3.txt
|
||||
'txt'
|
||||
>>> fname = get_test_fname('test_tab1.tabular')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
@@ -363,6 +376,29 @@ def guess_ext(fname, sniff_order, is_binary=False):
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.otu')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.otu'
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.lower.dist')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.lower.dist'
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.square.dist')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.square.dist'
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.pair.dist')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.pair.dist'
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.freq')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.freq'
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.quan')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.quan'
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.ref.taxonomy')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.ref.taxonomy'
|
||||
>>> fname = get_test_fname('mothur_datatypetest_true.mothur.axes')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'mothur.axes'
|
||||
>>> guess_ext(get_test_fname('infernal_model.cm'), sniff_order)
|
||||
'cm'
|
||||
>>> fname = get_test_fname('1.gg')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'gg'
|
||||
@@ -378,7 +414,83 @@ def guess_ext(fname, sniff_order, is_binary=False):
|
||||
>>> fname = get_test_fname('454Score.pdf')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'pdf'
|
||||
>>> fname = get_test_fname('1.obo')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'obo'
|
||||
>>> fname = get_test_fname('1.arff')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'arff'
|
||||
>>> fname = get_test_fname('1.afg')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'afg'
|
||||
>>> fname = get_test_fname('1.owl')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'owl'
|
||||
>>> fname = get_test_fname('Acanium.hmm')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'snaphmm'
|
||||
>>> fname = get_test_fname('wiggle.wig')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'wig'
|
||||
>>> fname = get_test_fname('example.iqtree')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'iqtree'
|
||||
>>> fname = get_test_fname('1.stockholm')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'stockholm'
|
||||
>>> fname = get_test_fname('1.xmfa')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'xmfa'
|
||||
>>> fname = get_test_fname('test.phylip')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'phylip'
|
||||
>>> fname = get_test_fname('1.smat')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'smat'
|
||||
>>> fname = get_test_fname('1.ttl')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'ttl'
|
||||
>>> fname = get_test_fname('1.hdt')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'hdt'
|
||||
>>> fname = get_test_fname('1.phyloxml')
|
||||
>>> guess_ext(fname, sniff_order)
|
||||
'phyloxml'
|
||||
"""
|
||||
file_prefix = FilePrefix(fname)
|
||||
file_ext = run_sniffers_raw(file_prefix, sniff_order, is_binary)
|
||||
|
||||
# Ugly hack for tsv vs tabular sniffing, we want to prefer tabular
|
||||
# to tsv but it doesn't have a sniffer - is TSV was sniffed just check
|
||||
# if it is an okay tabular and use that instead.
|
||||
if file_ext == 'tsv':
|
||||
if is_column_based(file_prefix, '\t', 1):
|
||||
file_ext = 'tabular'
|
||||
if file_ext is not None:
|
||||
return file_ext
|
||||
|
||||
# skip header check if data is already known to be binary
|
||||
if is_binary:
|
||||
return file_ext or 'binary'
|
||||
try:
|
||||
get_headers(file_prefix, None)
|
||||
except UnicodeDecodeError:
|
||||
return 'data' # default data type file extension
|
||||
if is_column_based(file_prefix, '\t', 1):
|
||||
return 'tabular' # default tabular data type file extension
|
||||
return 'txt' # default text data type file extension
|
||||
|
||||
|
||||
def run_sniffers_raw(filename_or_file_prefix, sniff_order, is_binary=False):
|
||||
"""Run through sniffers specified by sniff_order, return None of None match.
|
||||
"""
|
||||
if isinstance(filename_or_file_prefix, FilePrefix):
|
||||
fname = filename_or_file_prefix.filename
|
||||
file_prefix = filename_or_file_prefix
|
||||
else:
|
||||
fname = filename_or_file_prefix
|
||||
file_prefix = FilePrefix(filename_or_file_prefix)
|
||||
|
||||
file_ext = None
|
||||
for datatype in sniff_order:
|
||||
"""
|
||||
@@ -390,31 +502,29 @@ def guess_ext(fname, sniff_order, is_binary=False):
|
||||
successfully discovered.
|
||||
"""
|
||||
try:
|
||||
if ((is_binary and datatype.is_binary) or
|
||||
(not is_binary)) and datatype.sniff(fname):
|
||||
if hasattr(datatype, "sniff_prefix"):
|
||||
datatype_compressed = getattr(datatype, "compressed", False)
|
||||
if datatype_compressed and not file_prefix.compressed_format:
|
||||
continue
|
||||
if not datatype_compressed and file_prefix.compressed_format:
|
||||
continue
|
||||
if file_prefix.compressed_format and getattr(datatype, "compressed_format"):
|
||||
# In this case go a step further and compare the compressed format detected
|
||||
# to the expected.
|
||||
if file_prefix.compressed_format != datatype.compressed_format:
|
||||
continue
|
||||
if datatype.sniff_prefix(file_prefix):
|
||||
file_ext = datatype.file_ext
|
||||
break
|
||||
elif is_binary and not datatype.is_binary:
|
||||
continue
|
||||
elif datatype.sniff(fname):
|
||||
file_ext = datatype.file_ext
|
||||
break
|
||||
except Exception:
|
||||
pass
|
||||
# Ugly hack for tsv vs tabular sniffing, we want to prefer tabular
|
||||
# to tsv but it doesn't have a sniffer - is TSV was sniffed just check
|
||||
# if it is an okay tabular and use that instead.
|
||||
if file_ext == 'tsv':
|
||||
if is_column_based(fname, '\t', 1):
|
||||
file_ext = 'tabular'
|
||||
if file_ext is not None:
|
||||
return file_ext
|
||||
|
||||
# skip header check if data is already known to be binary
|
||||
if is_binary:
|
||||
return file_ext or 'binary'
|
||||
try:
|
||||
get_headers(fname, None)
|
||||
except UnicodeDecodeError:
|
||||
return 'data' # default data type file extension
|
||||
if is_column_based(fname, '\t', 1):
|
||||
return 'tabular' # default tabular data type file extension
|
||||
return 'txt' # default text data type file extension
|
||||
return file_ext
|
||||
|
||||
|
||||
def zip_single_fileobj(path):
|
||||
@@ -424,6 +534,91 @@ def zip_single_fileobj(path):
|
||||
return z.open(name)
|
||||
|
||||
|
||||
class FilePrefix(object):
|
||||
|
||||
def __init__(self, filename):
|
||||
binary = False
|
||||
compressed_format = None
|
||||
contents_header = None # First MAX_BYTES of the file.
|
||||
truncated = False
|
||||
# A future direction to optimize sniffing even more for sniffers at the top of the list
|
||||
# is to lazy load contents_header based on what interface is requested. For instance instead
|
||||
# of returning a StringIO directly in string_io() return an object that reads the contents and
|
||||
# populates contents_header while providing a StringIO-like interface until the file is read
|
||||
# but then would fallback to native string_io()
|
||||
try:
|
||||
compressed_format, f = compression_utils.get_fileobj_raw(filename)
|
||||
try:
|
||||
contents_header = f.read(SNIFF_PREFIX_BYTES)
|
||||
truncated = len(contents_header) == SNIFF_PREFIX_BYTES
|
||||
finally:
|
||||
f.close()
|
||||
except UnicodeDecodeError:
|
||||
binary = True
|
||||
|
||||
self.truncated = truncated
|
||||
self.filename = filename
|
||||
self.binary = binary
|
||||
self.compressed_format = compressed_format
|
||||
self.contents_header = contents_header
|
||||
self._file_size = None
|
||||
|
||||
@property
|
||||
def file_size(self):
|
||||
if self._file_size is None:
|
||||
self._file_size = os.path.getsize(self.filename)
|
||||
return self._file_size
|
||||
|
||||
def string_io(self):
|
||||
if self.binary:
|
||||
raise Exception("Attempting to create a StringIO object for binary data.")
|
||||
rval = StringIO(self.contents_header)
|
||||
return rval
|
||||
|
||||
def startswith(self, prefix):
|
||||
return self.string_io().read(len(prefix)) == prefix
|
||||
|
||||
def line_iterator(self):
|
||||
s = self.string_io()
|
||||
for line in s:
|
||||
if line.endswith("\n") or line.endswith("\r"):
|
||||
yield line
|
||||
elif s.pos == s.len and not self.truncated:
|
||||
# At the end, return the last line if it wasn't truncated when reading it in.
|
||||
yield line
|
||||
|
||||
# Convenience wrappers around contents_header, shielding contents_header means we can
|
||||
# potentially do a better job lazy loading this data later on.
|
||||
def search(self, pattern):
|
||||
return pattern.search(self.contents_header)
|
||||
|
||||
def search_str(self, query_str):
|
||||
return query_str in self.contents_header
|
||||
|
||||
|
||||
def build_sniff_from_prefix(klass):
|
||||
def auto_sniff(self, filename):
|
||||
file_prefix = FilePrefix(filename)
|
||||
datatype_compressed = getattr(self, "compressed", False)
|
||||
if file_prefix.compressed_format and not datatype_compressed:
|
||||
return False
|
||||
if datatype_compressed and not file_prefix.compressed_format:
|
||||
return False
|
||||
if hasattr(self, "compressed_format"):
|
||||
if self.compressed_format != file_prefix.compressed_format:
|
||||
return False
|
||||
return self.sniff_prefix(file_prefix)
|
||||
|
||||
klass.sniff = auto_sniff
|
||||
return klass
|
||||
|
||||
|
||||
def disable_parent_class_sniffing(klass):
|
||||
klass.sniff = lambda self, filename: False
|
||||
klass.sniff_prefix = lambda self, file_prefix: False
|
||||
return klass
|
||||
|
||||
|
||||
def handle_compressed_file(
|
||||
filename,
|
||||
datatypes_registry,
|
||||
@@ -464,11 +659,10 @@ def handle_compressed_file(
|
||||
if ext in AUTO_DETECT_EXTENSIONS:
|
||||
# attempt to sniff for a keep-compressed datatype (observing the sniff order)
|
||||
sniff_datatypes = filter(lambda d: getattr(d, 'compressed', False), datatypes_registry.sniff_order)
|
||||
for datatype in sniff_datatypes:
|
||||
if datatype.sniff(filename):
|
||||
ext = datatype.file_ext
|
||||
keep_compressed = True
|
||||
break
|
||||
sniffed_ext = run_sniffers_raw(filename, sniff_datatypes)
|
||||
if sniffed_ext:
|
||||
ext = sniffed_ext
|
||||
keep_compressed = True
|
||||
else:
|
||||
datatype = datatypes_registry.get_datatype_by_extension(ext)
|
||||
keep_compressed = getattr(datatype, 'compressed', False)
|
||||
|
||||
+106
-106
@@ -21,11 +21,11 @@ from galaxy import util
|
||||
from galaxy.datatypes import binary, data, metadata
|
||||
from galaxy.datatypes.metadata import MetadataElement
|
||||
from galaxy.datatypes.sniff import (
|
||||
build_sniff_from_prefix,
|
||||
get_headers,
|
||||
iter_headers
|
||||
)
|
||||
from galaxy.util import compression_utils
|
||||
from galaxy.util.checkers import is_gzip
|
||||
from . import dataproviders
|
||||
|
||||
if sys.version_info > (3,):
|
||||
@@ -416,6 +416,7 @@ class Taxonomy(Tabular):
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
@build_sniff_from_prefix
|
||||
class Sam(Tabular):
|
||||
edam_format = "format_2573"
|
||||
edam_data = "data_0863"
|
||||
@@ -434,7 +435,7 @@ class Sam(Tabular):
|
||||
"""Returns formated html of peek"""
|
||||
return self.make_html_table(dataset, column_names=self.column_names)
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in SAM format
|
||||
|
||||
@@ -463,31 +464,31 @@ class Sam(Tabular):
|
||||
>>> Sam().sniff( fname )
|
||||
True
|
||||
"""
|
||||
with open(filename) as fh:
|
||||
count = 0
|
||||
while True:
|
||||
line = fh.readline()
|
||||
line = line.strip()
|
||||
if not line:
|
||||
break # EOF
|
||||
if line:
|
||||
if line[0] != '@':
|
||||
line_pieces = line.split('\t')
|
||||
if len(line_pieces) < 11:
|
||||
return False
|
||||
try:
|
||||
int(line_pieces[1])
|
||||
int(line_pieces[3])
|
||||
int(line_pieces[4])
|
||||
int(line_pieces[7])
|
||||
int(line_pieces[8])
|
||||
except ValueError:
|
||||
return False
|
||||
count += 1
|
||||
if count == 5:
|
||||
return True
|
||||
if count < 5 and count > 0:
|
||||
return True
|
||||
fh = file_prefix.string_io()
|
||||
count = 0
|
||||
while True:
|
||||
line = fh.readline()
|
||||
line = line.strip()
|
||||
if not line:
|
||||
break # EOF
|
||||
if line:
|
||||
if line[0] != '@':
|
||||
line_pieces = line.split('\t')
|
||||
if len(line_pieces) < 11:
|
||||
return False
|
||||
try:
|
||||
int(line_pieces[1])
|
||||
int(line_pieces[3])
|
||||
int(line_pieces[4])
|
||||
int(line_pieces[7])
|
||||
int(line_pieces[8])
|
||||
except ValueError:
|
||||
return False
|
||||
count += 1
|
||||
if count == 5:
|
||||
return True
|
||||
if count < 5 and count > 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
def set_meta(self, dataset, overwrite=True, skip=None, max_data_lines=5, **kwd):
|
||||
@@ -592,6 +593,7 @@ class Sam(Tabular):
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
@build_sniff_from_prefix
|
||||
class Pileup(Tabular):
|
||||
"""Tab delimited data in pileup (6- or 10-column) format"""
|
||||
edam_format = "format_3015"
|
||||
@@ -616,7 +618,7 @@ class Pileup(Tabular):
|
||||
"""Return options for removing errors along with a description"""
|
||||
return [("lines", "Remove erroneous lines")]
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Checks for 'pileup-ness'
|
||||
|
||||
@@ -634,24 +636,32 @@ class Pileup(Tabular):
|
||||
>>> fname = get_test_fname( '10col.pileup' )
|
||||
>>> Pileup().sniff( fname )
|
||||
True
|
||||
>>> fname = get_test_fname( '1.xls' )
|
||||
>>> Pileup().sniff( fname )
|
||||
False
|
||||
>>> fname = get_test_fname( '2.txt' )
|
||||
>>> Pileup().sniff( fname ) # 2.txt
|
||||
False
|
||||
>>> fname = get_test_fname( '2.tabular' )
|
||||
>>> Pileup().sniff( fname )
|
||||
False
|
||||
"""
|
||||
headers = iter_headers(filename, '\t')
|
||||
found_non_comment_lines = False
|
||||
try:
|
||||
headers = iter_headers(file_prefix, '\t')
|
||||
for hdr in headers:
|
||||
if hdr and not hdr[0].startswith('#'):
|
||||
if len(hdr) < 5:
|
||||
return False
|
||||
try:
|
||||
# chrom start in column 1 (with 0-based columns)
|
||||
# and reference base is in column 2
|
||||
chrom = int(hdr[1])
|
||||
assert chrom >= 0
|
||||
assert hdr[2] in ['A', 'C', 'G', 'T', 'N', 'a', 'c', 'g', 't', 'n']
|
||||
except Exception:
|
||||
return False
|
||||
return True
|
||||
# chrom start in column 1 (with 0-based columns)
|
||||
# and reference base is in column 2
|
||||
chrom = int(hdr[1])
|
||||
assert chrom >= 0
|
||||
assert hdr[2] in ['A', 'C', 'G', 'T', 'N', 'a', 'c', 'g', 't', 'n']
|
||||
found_non_comment_lines = True
|
||||
except Exception:
|
||||
return False
|
||||
return found_non_comment_lines
|
||||
|
||||
# Dataproviders
|
||||
@dataproviders.decorators.dataprovider_factory('genomic-region',
|
||||
@@ -667,6 +677,7 @@ class Pileup(Tabular):
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
@build_sniff_from_prefix
|
||||
class BaseVcf(Tabular):
|
||||
""" Variant Call Format for describing SNPs and other simple genome variations. """
|
||||
edam_format = "format_3016"
|
||||
@@ -680,15 +691,12 @@ class BaseVcf(Tabular):
|
||||
MetadataElement(name="viz_filter_cols", desc="Score column for visualization", default=[5], param=metadata.ColumnParameter, optional=True, multiple=True, visible=False)
|
||||
MetadataElement(name="sample_names", default=[], desc="Sample names", readonly=True, visible=False, optional=True, no_value=[])
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
# Because this sniffer is run on compressed files that might be BGZF (due to the VcfGz subclass), we should
|
||||
# handle unicode decode errors. This should ultimately be done in get_headers(), but guess_ext() currently
|
||||
# relies on get_headers() raising this exception.
|
||||
try:
|
||||
headers = get_headers(filename, '\n', count=1)
|
||||
return headers[0][0].startswith("##fileformat=VCF")
|
||||
except UnicodeDecodeError:
|
||||
return False
|
||||
headers = get_headers(file_prefix, '\n', count=1)
|
||||
return headers[0][0].startswith("##fileformat=VCF")
|
||||
|
||||
def display_peek(self, dataset):
|
||||
"""Returns formated html of peek"""
|
||||
@@ -737,23 +745,14 @@ class BaseVcf(Tabular):
|
||||
class Vcf(BaseVcf):
|
||||
file_ext = 'vcf'
|
||||
|
||||
def sniff(self, filename):
|
||||
if is_gzip(filename):
|
||||
return False
|
||||
return super(Vcf, self).sniff(filename)
|
||||
|
||||
|
||||
class VcfGz(BaseVcf, binary.Binary):
|
||||
file_ext = 'vcf_bgzip'
|
||||
compressed = True
|
||||
compressed_format = "gzip"
|
||||
|
||||
MetadataElement(name="tabix_index", desc="Vcf Index File", param=metadata.FileParameter, file_ext="tbi", readonly=True, no_value=None, visible=False, optional=True)
|
||||
|
||||
def sniff(self, filename):
|
||||
if not is_gzip(filename):
|
||||
return False
|
||||
return super(VcfGz, self).sniff(filename)
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
super(BaseVcf, self).set_meta(dataset, **kwd)
|
||||
""" Creates the index for the VCF file. """
|
||||
@@ -769,8 +768,11 @@ class VcfGz(BaseVcf, binary.Binary):
|
||||
dataset.metadata.tabix_index = index_file
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Eland(Tabular):
|
||||
"""Support for the export.txt.gz file used by Illumina's ELANDv2e aligner"""
|
||||
compressed = True
|
||||
compressed_format = "gzip"
|
||||
file_ext = '_export.txt.gz'
|
||||
MetadataElement(name="columns", default=0, desc="Number of columns", readonly=True, visible=False)
|
||||
MetadataElement(name="column_types", default=[], param=metadata.ColumnTypesParameter, desc="Column types", readonly=True, visible=False, no_value=[])
|
||||
@@ -811,7 +813,7 @@ class Eland(Tabular):
|
||||
out = "Can't create peek %s" % exc
|
||||
return out
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in ELAND export format
|
||||
|
||||
@@ -824,32 +826,31 @@ class Eland(Tabular):
|
||||
- LANE, TILEm X, Y, INDEX, READ_NO, SEQ, QUAL, POSITION, *STRAND, FILT must be correct
|
||||
- We will only check that up to the first 5 alignments are correctly formatted.
|
||||
"""
|
||||
with compression_utils.get_fileobj(filename, compressed_formats=['gzip']) as fh:
|
||||
count = 0
|
||||
while True:
|
||||
line = fh.readline()
|
||||
line = line.strip()
|
||||
if not line:
|
||||
break # EOF
|
||||
if line:
|
||||
line_pieces = line.split('\t')
|
||||
if len(line_pieces) != 22:
|
||||
return False
|
||||
if long(line_pieces[1]) < 0:
|
||||
raise Exception('Out of range')
|
||||
if long(line_pieces[2]) < 0:
|
||||
raise Exception('Out of range')
|
||||
if long(line_pieces[3]) < 0:
|
||||
raise Exception('Out of range')
|
||||
int(line_pieces[4])
|
||||
int(line_pieces[5])
|
||||
# can get a lot more specific
|
||||
count += 1
|
||||
if count == 5:
|
||||
break
|
||||
if count > 0:
|
||||
return True
|
||||
return False
|
||||
fh = file_prefix.string_io()
|
||||
count = 0
|
||||
while True:
|
||||
line = fh.readline()
|
||||
line = line.strip()
|
||||
if not line:
|
||||
break # EOF
|
||||
if line:
|
||||
line_pieces = line.split('\t')
|
||||
if len(line_pieces) != 22:
|
||||
return False
|
||||
if long(line_pieces[1]) < 0:
|
||||
raise Exception('Out of range')
|
||||
if long(line_pieces[2]) < 0:
|
||||
raise Exception('Out of range')
|
||||
if long(line_pieces[3]) < 0:
|
||||
raise Exception('Out of range')
|
||||
int(line_pieces[4])
|
||||
int(line_pieces[5])
|
||||
# can get a lot more specific
|
||||
count += 1
|
||||
if count == 5:
|
||||
break
|
||||
if count > 0:
|
||||
return True
|
||||
|
||||
def set_meta(self, dataset, overwrite=True, skip=None, max_data_lines=5, **kwd):
|
||||
if dataset.has_data():
|
||||
@@ -884,10 +885,11 @@ class Eland(Tabular):
|
||||
dataset.metadata.reads = list(reads.keys())
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class ElandMulti(Tabular):
|
||||
file_ext = 'elandmulti'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
return False
|
||||
|
||||
|
||||
@@ -1056,6 +1058,7 @@ class TSV(BaseCSV):
|
||||
strict_width = True # Leave files with different width to tabular
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class ConnectivityTable(Tabular):
|
||||
edam_format = "format_3309"
|
||||
file_ext = "ct"
|
||||
@@ -1077,7 +1080,7 @@ class ConnectivityTable(Tabular):
|
||||
|
||||
dataset.metadata.data_lines = data_lines
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
The ConnectivityTable (CT) is a file format used for describing
|
||||
RNA 2D structures by tools including MFOLD, UNAFOLD and
|
||||
@@ -1112,31 +1115,28 @@ class ConnectivityTable(Tabular):
|
||||
i = 0
|
||||
j = 1
|
||||
|
||||
try:
|
||||
with open(filename) as handle:
|
||||
for line in handle:
|
||||
line = line.strip()
|
||||
handle = file_prefix.string_io()
|
||||
for line in handle:
|
||||
line = line.strip()
|
||||
|
||||
if len(line) > 0:
|
||||
if i == 0:
|
||||
if not self.header_regexp.match(line):
|
||||
return False
|
||||
else:
|
||||
length = int(re.split('\W+', line, 1)[0])
|
||||
if len(line) > 0:
|
||||
if i == 0:
|
||||
if not self.header_regexp.match(line):
|
||||
return False
|
||||
else:
|
||||
length = int(re.split('\W+', line, 1)[0])
|
||||
else:
|
||||
if not self.structure_regexp.match(line.upper()):
|
||||
return False
|
||||
else:
|
||||
if j != int(re.split('\W+', line, 1)[0]):
|
||||
return False
|
||||
elif j == length: # Last line of first sequence has been recheached
|
||||
return True
|
||||
else:
|
||||
if not self.structure_regexp.match(line.upper()):
|
||||
return False
|
||||
else:
|
||||
if j != int(re.split('\W+', line, 1)[0]):
|
||||
return False
|
||||
elif j == length: # Last line of first sequence has been recheached
|
||||
return True
|
||||
else:
|
||||
j += 1
|
||||
i += 1
|
||||
return False
|
||||
except Exception:
|
||||
return False
|
||||
j += 1
|
||||
i += 1
|
||||
return False
|
||||
|
||||
def get_chunk(self, trans, dataset, chunk):
|
||||
ck_index = int(chunk)
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
{UNV
|
||||
iid:1
|
||||
com:
|
||||
Generated by dsommer with tarchive2amos on Wed Aug 30 13:10:59 2006
|
||||
.
|
||||
}
|
||||
{RED
|
||||
iid:1
|
||||
eid:zfishG-a2661d04.q1c
|
||||
seq:
|
||||
TAAAATAAATGTTATGTTATCATGTTGACAGATCAATGATAAAATAAAGCCTGGTGATTA
|
||||
AAAACCTGCAATACCTTGACAAGAACTTTCATGTAAACTAAAGTACTAACTAAAAAAGTG
|
||||
TCCTGAGAAATCTCGACAGTTTTTTGAGTTTGATAGCCCTGGGCTCAATCAGAAAACTAG
|
||||
CCAGTCAGAAACACTCTTCATTTCACTCGTTCGGGTCTGCTGACACTGACTTTGCTGACA
|
||||
AGTCTTTGGAGGTTGAGTTTTGGAGAGAGATGGCGTTAGCAAAAATGGCTGAAGTTAGCA
|
||||
AAATGGCTGCAGTCGCCTACAGCATTCGAATTCATACCTTGTTTCTGAGACCATGTGTCA
|
||||
CTCACCTGGGCGCTGACTTTGCTCTTGCTTGCCACGGCTTGTTTGAGAGCCTGTCTGATA
|
||||
ATGACATGCAGTCGGAGCACCACAGCCTCGATGTCCTGTTCCCCCTTCATGGCAGCGGCG
|
||||
AGTGACCCCAGTTCATCCACATATGCTCTCCAATGAGGCTGTACCTCCGTGGTTCCCCCC
|
||||
AAACATGCTCCCATGGTGCTGACACACAGTGGACGGCACGGCCGGGCCTTCAGCAGGCTT
|
||||
TGACAGTGCGGGCAGTACCACAGTTTCAGAAGTGAGCGCCCACAATCTCTACCTGCCCGC
|
||||
AAATGCTGAGT
|
||||
.
|
||||
qlt:
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXX
|
||||
.
|
||||
clr:0,671
|
||||
}
|
||||
{RED
|
||||
iid:2
|
||||
eid:zfishI-a72c06.p1c
|
||||
seq:
|
||||
CACCCAAAAGCAATGGCAAAAGATCTGGCGGACGCATTGCGGGCTGGGCGGGTGCTCAGC
|
||||
TTGGCACTTGCCGAAGGGTCAGAGGTGATGAACATTACGGAAACAGCAGGGTTATCAAAA
|
||||
GAGTGCACACGGGCTCTGGTACGCATGCATTACTGCTCTCACTGCCGTGGACTCACCCTG
|
||||
ATCCATGCGTGCAGCAACTACTGTCTTAATGTCATGCGCGGGTGCCTGGCGAGCTACTCC
|
||||
GAGCTCCACCAGCCCTGGAGACAGTATGTCACCATATTGCAGGACCTCACGCAAATGGTT
|
||||
GCCGGAGCTCACAATTTAGAGCTGGCCTTACTGGGGATCAGAGGTCAGGTCGAGGAGGCC
|
||||
ATACTCTACGCTCAGCTTCACGGGCCCAGGCTAACTGCCACAGTGAGTACTAGCATTTTT
|
||||
ACGCTTTTACAGCTAGCATTAGCTTGTATTGTAGCATGAAAAAAGGTCTACAGAGTTATG
|
||||
AAGCACAAAGACCTCTTCTGCTATAAGCGGGTTTCTGAA
|
||||
.
|
||||
qlt:
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX
|
||||
.
|
||||
clr:0,519
|
||||
}
|
||||
5B
|
||||
@@ -0,0 +1,85 @@
|
||||
% 1. Title: Database for fitting contact lenses
|
||||
%
|
||||
% 2. Sources:
|
||||
% (a) Cendrowska, J. "PRISM: An algorithm for inducing modular rules",
|
||||
% International Journal of Man-Machine Studies, 1987, 27, 349-370
|
||||
% (b) Donor: Benoit Julien (Julien@ce.cmu.edu)
|
||||
% (c) Date: 1 August 1990
|
||||
%
|
||||
% 3. Past Usage:
|
||||
% 1. See above.
|
||||
% 2. Witten, I. H. & MacDonald, B. A. (1988). Using concept
|
||||
% learning for knowledge acquisition. International Journal of
|
||||
% Man-Machine Studies, 27, (pp. 349-370).
|
||||
%
|
||||
% Notes: This database is complete (all possible combinations of
|
||||
% attribute-value pairs are represented).
|
||||
%
|
||||
% Each instance is complete and correct.
|
||||
%
|
||||
% 9 rules cover the training set.
|
||||
%
|
||||
% 4. Relevant Information Paragraph:
|
||||
% The examples are complete and noise free.
|
||||
% The examples highly simplified the problem. The attributes do not
|
||||
% fully describe all the factors affecting the decision as to which type,
|
||||
% if any, to fit.
|
||||
%
|
||||
% 5. Number of Instances: 24
|
||||
%
|
||||
% 6. Number of Attributes: 4 (all nominal)
|
||||
%
|
||||
% 7. Attribute Information:
|
||||
% -- 3 Classes
|
||||
% 1 : the patient should be fitted with hard contact lenses,
|
||||
% 2 : the patient should be fitted with soft contact lenses,
|
||||
% 1 : the patient should not be fitted with contact lenses.
|
||||
%
|
||||
% 1. age of the patient: (1) young, (2) pre-presbyopic, (3) presbyopic
|
||||
% 2. spectacle prescription: (1) myope, (2) hypermetrope
|
||||
% 3. astigmatic: (1) no, (2) yes
|
||||
% 4. tear production rate: (1) reduced, (2) normal
|
||||
%
|
||||
% 8. Number of Missing Attribute Values: 0
|
||||
%
|
||||
% 9. Class Distribution:
|
||||
% 1. hard contact lenses: 4
|
||||
% 2. soft contact lenses: 5
|
||||
% 3. no contact lenses: 15
|
||||
|
||||
@relation contact-lenses
|
||||
|
||||
@attribute age {young, pre-presbyopic, presbyopic}
|
||||
@attribute spectacle-prescrip {myope, hypermetrope}
|
||||
@attribute astigmatism {no, yes}
|
||||
@attribute tear-prod-rate {reduced, normal}
|
||||
@attribute contact-lenses {soft, hard, none}
|
||||
|
||||
@data
|
||||
%
|
||||
% 24 instances
|
||||
%
|
||||
young,myope,no,reduced,none
|
||||
young,myope,no,normal,soft
|
||||
young,myope,yes,reduced,none
|
||||
young,myope,yes,normal,hard
|
||||
young,hypermetrope,no,reduced,none
|
||||
young,hypermetrope,no,normal,soft
|
||||
young,hypermetrope,yes,reduced,none
|
||||
young,hypermetrope,yes,normal,hard
|
||||
pre-presbyopic,myope,no,reduced,none
|
||||
pre-presbyopic,myope,no,normal,soft
|
||||
pre-presbyopic,myope,yes,reduced,none
|
||||
pre-presbyopic,myope,yes,normal,hard
|
||||
pre-presbyopic,hypermetrope,no,reduced,none
|
||||
pre-presbyopic,hypermetrope,no,normal,soft
|
||||
pre-presbyopic,hypermetrope,yes,reduced,none
|
||||
pre-presbyopic,hypermetrope,yes,normal,none
|
||||
presbyopic,myope,no,reduced,none
|
||||
presbyopic,myope,no,normal,none
|
||||
presbyopic,myope,yes,reduced,none
|
||||
presbyopic,myope,yes,normal,hard
|
||||
presbyopic,hypermetrope,no,reduced,none
|
||||
presbyopic,hypermetrope,no,normal,soft
|
||||
presbyopic,hypermetrope,yes,reduced,none
|
||||
presbyopic,hypermetrope,yes,normal,none
|
||||
Binary file not shown.
@@ -0,0 +1,31 @@
|
||||
format-version: GO_1.0
|
||||
!any comment here
|
||||
typeref: relationship.types
|
||||
subsetdef: goslim "Generic GO Slim"
|
||||
version: $Revision: 1.18 $
|
||||
date: April 18th, 2003
|
||||
saved-by: jrichter
|
||||
remark: Example file
|
||||
|
||||
[Term]
|
||||
id: GO:0003674
|
||||
name: molecular_function
|
||||
def: "The action characteristic of a gene product." [GO:curators]
|
||||
subset: goslim
|
||||
|
||||
[Term]
|
||||
id: GO:0016209
|
||||
name: antioxidant activity
|
||||
is_a: GO:0003674
|
||||
def: "Inhibition of the reactions brought about by dioxygen or peroxides. Usually the antioxidant is effective because it can itself be more easily oxidized than the substance protected. The term is often applied to components that can trap free radicals, thereby breaking the chain reaction that normally leads to extensive biological damage." [ISBN:0198506732]
|
||||
|
||||
[Term]
|
||||
id: GO:0045174
|
||||
name: glutathione dehydrogenase (ascorbate) activity
|
||||
xref_analog: EC:1.8.5.1 ""
|
||||
def: "Catalysis of the reaction: 2 glutathione + dehydroascorbate = glutathione disulfide + ascorbate." [EC:1.8.5.1]
|
||||
synonym: dehydroascorbate reductase []
|
||||
is_a: GO:0009055
|
||||
is_a: GO:0015038
|
||||
is_a: GO:0016672
|
||||
|
||||
@@ -0,0 +1,221 @@
|
||||
<?xml version="1.0"?>
|
||||
<rdf:RDF
|
||||
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
|
||||
xmlns:xsd="http://www.w3.org/2001/XMLSchema#"
|
||||
xmlns:rdfs="http://www.w3.org/2000/01/rdf-schema#"
|
||||
xmlns:mfg="http://www.workingontologist.org/Examples/Chapter3/Product.owl#"
|
||||
xmlns:owl="http://www.w3.org/2002/07/owl#"
|
||||
xmlns:daml="http://www.daml.org/2001/03/daml+oil#"
|
||||
xml:base="http://www.workingontologist.org/Examples/Chapter3/Product.owl">
|
||||
<owl:Ontology rdf:about="">
|
||||
<owl:versionInfo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Created with TopBraid Spreadsheet converter</owl:versionInfo>
|
||||
</owl:Ontology>
|
||||
<owl:Class rdf:ID="Product"/>
|
||||
<owl:DatatypeProperty rdf:ID="Product_SKU">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>SKU</rdfs:label>
|
||||
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
|
||||
<rdfs:domain rdf:resource="#Product"/>
|
||||
</owl:DatatypeProperty>
|
||||
<owl:DatatypeProperty rdf:ID="Product_Product_Line">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product Line</rdfs:label>
|
||||
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
|
||||
<rdfs:domain rdf:resource="#Product"/>
|
||||
</owl:DatatypeProperty>
|
||||
<owl:DatatypeProperty rdf:ID="Product_Manufacture_Location">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Manufacture Location</rdfs:label>
|
||||
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
|
||||
<rdfs:domain rdf:resource="#Product"/>
|
||||
</owl:DatatypeProperty>
|
||||
<owl:DatatypeProperty rdf:ID="Product_Available">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Available</rdfs:label>
|
||||
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
|
||||
<rdfs:domain rdf:resource="#Product"/>
|
||||
</owl:DatatypeProperty>
|
||||
<owl:DatatypeProperty rdf:ID="Product_Division">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Division</rdfs:label>
|
||||
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
|
||||
<rdfs:domain rdf:resource="#Product"/>
|
||||
</owl:DatatypeProperty>
|
||||
<owl:DatatypeProperty rdf:ID="Product_ModelNo">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>ModelNo</rdfs:label>
|
||||
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
|
||||
<rdfs:domain rdf:resource="#Product"/>
|
||||
</owl:DatatypeProperty>
|
||||
<owl:DatatypeProperty rdf:ID="Product_ID">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>ID</rdfs:label>
|
||||
<rdfs:range rdf:resource="http://www.w3.org/2001/XMLSchema#string"/>
|
||||
<rdfs:domain rdf:resource="#Product"/>
|
||||
</owl:DatatypeProperty>
|
||||
<mfg:Product rdf:ID="Product2">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 2</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>2</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>ZX-3P</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Manufacturing support</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Paper machine</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Sacramento</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>KD5243</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>4</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product6">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 6</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>6</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>B-1431</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Control Engineering</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Active sensor</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Seoul</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>KK3945</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>0</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product1">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 1</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>1</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>ZX-3</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Manufacturing support</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Papermachine</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Sacramento</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>FB3524</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>23</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product7">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 7</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>7</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>DBB-12</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Accessories</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Monitor</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Hong Kong</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>ND5520</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>100</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product9">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 9</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>9</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>SPX-1234</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Safety</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Safety</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>valve</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Cleveland</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>OP5333</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product4">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 4</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>4</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>B-1430</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Control Engineering</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Feedback line</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Elizabeth</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>KS4520</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>23</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product8">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 8</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>8</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>SP-1234</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Safety</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Safety</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>valve</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Cleveland</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>HI4554</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product5">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 5</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>5</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>B-1430X</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Control Engineering</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Feedback line</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Elizabeth</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>CL5934</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>14</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
<mfg:Product rdf:ID="Product3">
|
||||
<rdfs:label rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Product 3</rdfs:label>
|
||||
<mfg:Product_ID rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>3</mfg:Product_ID>
|
||||
<mfg:Product_ModelNo rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>ZX-3S</mfg:Product_ModelNo>
|
||||
<mfg:Product_Division rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Manufacturing support</mfg:Product_Division>
|
||||
<mfg:Product_Product_Line rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Paper machine</mfg:Product_Product_Line>
|
||||
<mfg:Product_Manufacture_Location rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>Sacramento</mfg:Product_Manufacture_Location>
|
||||
<mfg:Product_SKU rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>IL4028</mfg:Product_SKU>
|
||||
<mfg:Product_Available rdf:datatype="http://www.w3.org/2001/XMLSchema#string"
|
||||
>34</mfg:Product_Available>
|
||||
</mfg:Product>
|
||||
</rdf:RDF>
|
||||
|
||||
<!-- Created with TopBraid Composer -->
|
||||
@@ -0,0 +1,48 @@
|
||||
<phyloxml xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns="http://www.phyloxml.org" xsi:schemaLocation="http://www.phyloxml.org http://www.phyloxml.org/1.10/phyloxml.xsd">
|
||||
<phylogeny rooted="true" rerootable="false">
|
||||
<name>Alcohol dehydrogenases</name>
|
||||
<description>contains examples of commonly used elements</description>
|
||||
<clade>
|
||||
<events>
|
||||
<speciations>1</speciations>
|
||||
</events>
|
||||
<clade>
|
||||
<taxonomy>
|
||||
<id provider="ncbi">6645</id>
|
||||
<scientific_name>Octopus vulgaris</scientific_name>
|
||||
</taxonomy>
|
||||
<sequence>
|
||||
<accession source="UniProtKB">P81431</accession>
|
||||
<name>Alcohol dehydrogenase class-3</name>
|
||||
</sequence>
|
||||
</clade>
|
||||
<clade>
|
||||
<confidence type="bootstrap">100</confidence>
|
||||
<events>
|
||||
<speciations>1</speciations>
|
||||
</events>
|
||||
<clade>
|
||||
<taxonomy>
|
||||
<id provider="ncbi">1423</id>
|
||||
<scientific_name>Bacillus subtilis</scientific_name>
|
||||
</taxonomy>
|
||||
<sequence>
|
||||
<accession source="UniProtKB">P71017</accession>
|
||||
<name>Alcohol dehydrogenase</name>
|
||||
</sequence>
|
||||
</clade>
|
||||
<clade>
|
||||
<taxonomy>
|
||||
<id provider="ncbi">562</id>
|
||||
<scientific_name>Escherichia coli</scientific_name>
|
||||
</taxonomy>
|
||||
<sequence>
|
||||
<accession source="UniProtKB">Q46856</accession>
|
||||
<name>Alcohol dehydrogenase</name>
|
||||
</sequence>
|
||||
</clade>
|
||||
</clade>
|
||||
</clade>
|
||||
</phylogeny>
|
||||
</phyloxml>
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
# STOCKHOLM 1.0
|
||||
#=GF ID UPSK
|
||||
#=GF SE Predicted; Infernal
|
||||
#=GF SS Published; PMID 9223489
|
||||
#=GF RN [1]
|
||||
#=GF RM 9223489
|
||||
#=GF RT The role of the pseudoknot at the 3' end of turnip yellow mosaic
|
||||
#=GF RT virus RNA in minus-strand synthesis by the viral RNA-dependent RNA
|
||||
#=GF RT polymerase.
|
||||
#=GF RA Deiman BA, Kortlever RM, Pleij CW;
|
||||
#=GF RL J Virol 1997;71:5990-5996.
|
||||
|
||||
AF035635.1/619-641 UGAGUUCUCGAUCUCUAAAAUCG
|
||||
M24804.1/82-104 UGAGUUCUCUAUCUCUAAAAUCG
|
||||
J04373.1/6212-6234 UAAGUUCUCGAUCUUUAAAAUCG
|
||||
M24803.1/1-23 UAAGUUCUCGAUCUCUAAAAUCG
|
||||
#=GC SS_cons .AAA....<<<<aaa....>>>>
|
||||
//
|
||||
@@ -0,0 +1,10 @@
|
||||
@prefix rdf: <http://www.w3.org/1999/02/22-rdf-syntax-ns#> .
|
||||
@prefix dc: <http://purl.org/dc/elements/1.1/> .
|
||||
@prefix ex: <http://example.org/stuff/1.0/> .
|
||||
|
||||
<http://www.w3.org/TR/rdf-syntax-grammar>
|
||||
dc:title "RDF/XML Syntax Specification (Revised)" ;
|
||||
ex:editor [
|
||||
ex:fullname "Dave Beckett";
|
||||
ex:homePage <http://purl.org/net/dajobe/>
|
||||
] .
|
||||
@@ -0,0 +1,30 @@
|
||||
#FormatVersion Mauve1
|
||||
#Sequence1File a.fa
|
||||
#Sequence1Entry 1
|
||||
#Sequence1Format FastA
|
||||
#Sequence2File b.fa
|
||||
#Sequence2Entry 2
|
||||
#Sequence2Format FastA
|
||||
#Sequence3File c.fa
|
||||
#Sequence3Entry 3
|
||||
#Sequence3Format FastA
|
||||
#BackboneFile three.xmfa.bbcols
|
||||
> 1:0-0 + a.fa
|
||||
--------------------------------------------------------------------------------
|
||||
--------------------------------------------------------------------------------
|
||||
--------------------------------------------------------------------------------
|
||||
> 2:5417-5968 + b.fa
|
||||
TTTAAACATCCCTCGGCCCGTCGCCCTTTTATAATAGCAGTACGTGAGAGGAGCGCCCTAAGCTTTGGGAAATTCAAGC-
|
||||
--------------------------------------------------------------------------------
|
||||
CTGGAACGTACTTGCTGGTTTCGCTACTATTTCAAACAAGTTAGAGGCCGTTACCTCGGGCGAACGTATAAACCATTCTG
|
||||
> 3:9476-10076 - c.fa
|
||||
TTTAAACACCTTTTTGGATG--GCCCAGTTCGTTCAGTTGTG-GGGAGGAGATCGCCCCAAACGTATGGTGAGTCGGGCG
|
||||
TTTCCTATAGCTATAGGACCAATCCACTTACCATACGCCCGGCGTCGCCCAGTCCGGTTCGGTACCCTCCATGACCCACG
|
||||
---------------------------------------------------------AAATGAGGGCCCAGGGTATGCTT
|
||||
=
|
||||
> 2:5969-6015 + b.fa
|
||||
-----------------------
|
||||
GGGCGAACGTATAAACCATTCTG
|
||||
> 3:9429-9476 - c.fa
|
||||
TTCGGTACCCTCCATGACCCACG
|
||||
AAATGAGGGCCCAGGGTATGCTT
|
||||
@@ -0,0 +1,3 @@
|
||||
a 2
|
||||
c 1
|
||||
d 0
|
||||
@@ -0,0 +1 @@
|
||||
a 2
|
||||
@@ -0,0 +1,3 @@
|
||||
a 1 2 x
|
||||
b 3 4 y
|
||||
c 5 6 z
|
||||
@@ -0,0 +1,21 @@
|
||||
zoeHMM Acanium.hmm 6 8 6 7
|
||||
|
||||
<STATES>
|
||||
|
||||
Einit 0 0 3 -1 explicit
|
||||
Esngl 0 0 150 -1 explicit
|
||||
Eterm 0 0 3 -1 explicit
|
||||
Exon 0 0 6 -1 explicit
|
||||
Inter 0.9 0.9 0 0 geometric
|
||||
Intron 0.1 0.1 0 0 geometric
|
||||
|
||||
<STATE_TRANSITIONS>
|
||||
|
||||
Einit Intron 1
|
||||
Esngl Inter 1
|
||||
Eterm Inter 1
|
||||
Exon Intron 1
|
||||
Inter Einit 0.852754
|
||||
Inter Esngl 0.147246
|
||||
Intron Eterm 0.129065
|
||||
Intron Exon 0.870935
|
||||
@@ -14,12 +14,13 @@ from six.moves import shlex_quote
|
||||
|
||||
from galaxy.datatypes.data import get_file_peek, Text
|
||||
from galaxy.datatypes.metadata import MetadataElement, MetadataParameter
|
||||
from galaxy.datatypes.sniff import iter_headers
|
||||
from galaxy.datatypes.sniff import build_sniff_from_prefix, iter_headers
|
||||
from galaxy.util import nice_size, string_as_bool
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Html(Text):
|
||||
"""Class describing an html file"""
|
||||
edam_format = "format_2331"
|
||||
@@ -37,7 +38,7 @@ class Html(Text):
|
||||
"""Returns the mime type of the datatype"""
|
||||
return 'text/html'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is in html format
|
||||
|
||||
@@ -49,13 +50,14 @@ class Html(Text):
|
||||
>>> Html().sniff( fname )
|
||||
True
|
||||
"""
|
||||
headers = iter_headers(filename, None)
|
||||
headers = iter_headers(file_prefix, None)
|
||||
for i, hdr in enumerate(headers):
|
||||
if hdr and hdr[0].lower().find('<html>') >= 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Json(Text):
|
||||
edam_format = "format_3464"
|
||||
file_ext = "json"
|
||||
@@ -72,30 +74,27 @@ class Json(Text):
|
||||
"""Returns the mime type of the datatype"""
|
||||
return 'application/json'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to load the string with the json module. If successful it's a json file.
|
||||
"""
|
||||
return self._looks_like_json(filename)
|
||||
return self._looks_like_json(file_prefix)
|
||||
|
||||
def _looks_like_json(self, filename):
|
||||
def _looks_like_json(self, file_prefix):
|
||||
# Pattern used by SequenceSplitLocations
|
||||
if os.path.getsize(filename) < 50000:
|
||||
if file_prefix.file_size < 50000 and not file_prefix.truncated:
|
||||
# If the file is small enough - don't guess just check.
|
||||
try:
|
||||
json.load(open(filename, "r"))
|
||||
json.loads(file_prefix.contents_header)
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
else:
|
||||
with open(filename, "r") as fh:
|
||||
while True:
|
||||
# Grab first chunk of file and see if it looks like json.
|
||||
start = fh.read(100).strip()
|
||||
if start:
|
||||
# simple types are valid JSON as well - but would such a file
|
||||
# be interesting as JSON in Galaxy?
|
||||
return start.startswith("[") or start.startswith("{")
|
||||
start = file_prefix.string_io().read(100).strip()
|
||||
if start:
|
||||
# simple types are valid JSON as well - but would such a file
|
||||
# be interesting as JSON in Galaxy?
|
||||
return start.startswith("[") or start.startswith("{")
|
||||
return False
|
||||
|
||||
def display_peek(self, dataset):
|
||||
@@ -105,6 +104,7 @@ class Json(Text):
|
||||
return "JSON file (%s)" % (nice_size(dataset.get_size()))
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Ipynb(Json):
|
||||
file_ext = "ipynb"
|
||||
|
||||
@@ -116,13 +116,14 @@ class Ipynb(Json):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to load the string with the json module. If successful it's a json file.
|
||||
"""
|
||||
if self._looks_like_json(filename):
|
||||
if self._looks_like_json(file_prefix):
|
||||
try:
|
||||
ipynb = json.load(open(filename))
|
||||
with open(file_prefix.filename) as f:
|
||||
ipynb = json.load(f)
|
||||
if ipynb.get('nbformat', False) is not False and ipynb.get('metadata', False):
|
||||
return True
|
||||
else:
|
||||
@@ -161,6 +162,7 @@ class Ipynb(Json):
|
||||
pass
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Biom1(Json):
|
||||
"""
|
||||
BIOM version 1.0 file format description
|
||||
@@ -186,13 +188,13 @@ class Biom1(Json):
|
||||
if not dataset.dataset.purged:
|
||||
dataset.blurb = "Biological Observation Matrix v1"
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
is_biom = False
|
||||
if self._looks_like_json(filename):
|
||||
is_biom = self._looks_like_biom(filename)
|
||||
if self._looks_like_json(file_prefix):
|
||||
is_biom = self._looks_like_biom(file_prefix)
|
||||
return is_biom
|
||||
|
||||
def _looks_like_biom(self, filepath, load_size=50000):
|
||||
def _looks_like_biom(self, file_prefix, load_size=50000):
|
||||
"""
|
||||
@param filepath: [str] The path to the evaluated file.
|
||||
@param load_size: [int] The size of the file block load in RAM (in
|
||||
@@ -201,7 +203,7 @@ class Biom1(Json):
|
||||
is_biom = False
|
||||
segment_size = int(load_size / 2)
|
||||
try:
|
||||
with open(filepath, "r") as fh:
|
||||
with open(file_prefix.filename, "r") as fh:
|
||||
prev_str = ""
|
||||
segment_str = fh.read(segment_size)
|
||||
if segment_str.strip().startswith('{'):
|
||||
@@ -255,6 +257,7 @@ class Biom1(Json):
|
||||
pass
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Obo(Text):
|
||||
"""
|
||||
OBO file format description
|
||||
@@ -272,25 +275,26 @@ class Obo(Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess the Obo filetype.
|
||||
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
|
||||
"""
|
||||
stanza = re.compile(r'^\[.*\]$')
|
||||
with open(filename) as handle:
|
||||
first_line = handle.readline()
|
||||
if not first_line.startswith('format-version:'):
|
||||
return False
|
||||
handle = file_prefix.string_io()
|
||||
first_line = handle.readline()
|
||||
if not first_line.startswith('format-version:'):
|
||||
return False
|
||||
|
||||
for line in handle:
|
||||
if stanza.match(line.strip()):
|
||||
# a stanza needs to begin with an ID tag
|
||||
if handle.next().startswith('id:'):
|
||||
return True
|
||||
for line in handle:
|
||||
if stanza.match(line.strip()):
|
||||
# a stanza needs to begin with an ID tag
|
||||
if handle.next().startswith('id:'):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Arff(Text):
|
||||
"""
|
||||
An ARFF (Attribute-Relation File Format) file is an ASCII text file that describes a list of instances sharing a set of attributes.
|
||||
@@ -312,31 +316,31 @@ class Arff(Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Try to guess the Arff filetype.
|
||||
It usually starts with a "format-version:" string and has several stanzas which starts with "id:".
|
||||
"""
|
||||
with open(filename) as handle:
|
||||
relation_found = False
|
||||
attribute_found = False
|
||||
for line_count, line in enumerate(handle):
|
||||
if line_count > 1000:
|
||||
# only investigate the first 1000 lines
|
||||
return False
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
handle = file_prefix.string_io()
|
||||
relation_found = False
|
||||
attribute_found = False
|
||||
for line_count, line in enumerate(handle):
|
||||
if line_count > 1000:
|
||||
# only investigate the first 1000 lines
|
||||
return False
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
|
||||
start_string = line[:20].upper()
|
||||
if start_string.startswith("@RELATION"):
|
||||
relation_found = True
|
||||
elif start_string.startswith("@ATTRIBUTE"):
|
||||
attribute_found = True
|
||||
elif start_string.startswith("@DATA"):
|
||||
# @DATA should be the last data block
|
||||
if relation_found and attribute_found:
|
||||
return True
|
||||
start_string = line[:20].upper()
|
||||
if start_string.startswith("@RELATION"):
|
||||
relation_found = True
|
||||
elif start_string.startswith("@ATTRIBUTE"):
|
||||
attribute_found = True
|
||||
elif start_string.startswith("@DATA"):
|
||||
# @DATA should be the last data block
|
||||
if relation_found and attribute_found:
|
||||
return True
|
||||
return False
|
||||
|
||||
def set_meta(self, dataset, **kwd):
|
||||
@@ -543,11 +547,12 @@ class SnpSiftDbNSFP(Text):
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class IQTree(Text):
|
||||
"""IQ-TREE format"""
|
||||
file_ext = 'iqtree'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Detect the IQTree file
|
||||
|
||||
@@ -567,7 +572,4 @@ class IQTree(Text):
|
||||
>>> IQTree().sniff(fname)
|
||||
False
|
||||
"""
|
||||
with open(filename, 'r') as fio:
|
||||
return fio.read(7) == "IQ-TREE"
|
||||
|
||||
return False
|
||||
return file_prefix.startswith("IQ-TREE")
|
||||
|
||||
@@ -4,6 +4,9 @@ Triple format classes
|
||||
import logging
|
||||
import re
|
||||
|
||||
from galaxy.datatypes.sniff import (
|
||||
build_sniff_from_prefix,
|
||||
)
|
||||
from . import (
|
||||
binary,
|
||||
data,
|
||||
@@ -13,6 +16,9 @@ from . import (
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
TURTLE_PREFIX_PATTERN = re.compile(r'@prefix\s+[^:]*:\s+<[^>]*>\s\.')
|
||||
TURTLE_BASE_PATTERN = re.compile(r'@base\s+<[^>]*>\s\.')
|
||||
|
||||
|
||||
class Triples(data.Data):
|
||||
"""
|
||||
@@ -38,6 +44,7 @@ class Triples(data.Data):
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class NTriples(data.Text, Triples):
|
||||
"""
|
||||
The N-Triples triple data format
|
||||
@@ -45,11 +52,10 @@ class NTriples(data.Text, Triples):
|
||||
edam_format = "format_3256"
|
||||
file_ext = "nt"
|
||||
|
||||
def sniff(self, filename):
|
||||
with open(filename, "r") as f:
|
||||
# <http://example.org/dir/relfile> <http://www.w3.org/1999/02/22-rdf-syntax-ns#type> <http://example.org/type> .
|
||||
if re.compile(r'<[^>]*>\s<[^>]*>\s<[^>]*>\s\.').search(f.readline(1024)):
|
||||
return True
|
||||
def sniff_prefix(self, file_prefix):
|
||||
# <http://example.org/dir/relfile> <http://www.w3.org/1999/02/22-rdf-syntax-ns#type> <http://example.org/type> .
|
||||
if re.compile(r'<[^>]*>\s<[^>]*>\s<[^>]*>\s\.').search(file_prefix.contents_header):
|
||||
return True
|
||||
return False
|
||||
|
||||
def set_peek(self, dataset, is_multi_byte=False):
|
||||
@@ -85,6 +91,7 @@ class N3(data.Text, Triples):
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Turtle(data.Text, Triples):
|
||||
"""
|
||||
The Turtle triple data format
|
||||
@@ -92,14 +99,13 @@ class Turtle(data.Text, Triples):
|
||||
edam_format = "format_3255"
|
||||
file_ext = "ttl"
|
||||
|
||||
def sniff(self, filename):
|
||||
with open(filename, "r") as f:
|
||||
# @prefix rdf: <http://www.w3.org/1999/02/22-rdf-syntax-ns#> .
|
||||
line = f.readline(1024)
|
||||
if re.compile(r'@prefix\s+[^:]*:\s+<[^>]*>\s\.').search(line):
|
||||
return True
|
||||
if re.compile(r'@base\s+<[^>]*>\s\.').search(line):
|
||||
return True
|
||||
def sniff_prefix(self, file_prefix):
|
||||
# @prefix rdf: <http://www.w3.org/1999/02/22-rdf-syntax-ns#> .
|
||||
if file_prefix.search(TURTLE_PREFIX_PATTERN):
|
||||
return True
|
||||
|
||||
if file_prefix.search(TURTLE_BASE_PATTERN):
|
||||
return True
|
||||
return False
|
||||
|
||||
def set_peek(self, dataset, is_multi_byte=False):
|
||||
@@ -113,6 +119,7 @@ class Turtle(data.Text, Triples):
|
||||
|
||||
|
||||
# TODO: we might want to look at rdflib or a similar, larger lib/egg
|
||||
@build_sniff_from_prefix
|
||||
class Rdf(xml.GenericXml, Triples):
|
||||
"""
|
||||
Resource Description Framework format (http://www.w3.org/RDF/).
|
||||
@@ -120,13 +127,11 @@ class Rdf(xml.GenericXml, Triples):
|
||||
edam_format = "format_3261"
|
||||
file_ext = "rdf"
|
||||
|
||||
def sniff(self, filename):
|
||||
with open(filename, "r") as f:
|
||||
firstlines = "".join(f.readlines(5000))
|
||||
# <rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" ...
|
||||
match = re.compile(r'xmlns:([^=]*)="http://www.w3.org/1999/02/22-rdf-syntax-ns#"').search(firstlines)
|
||||
if not match and (match.group(1) + ":RDF") in firstlines:
|
||||
return True
|
||||
def sniff_prefix(self, file_prefix):
|
||||
# <rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" ...
|
||||
match = re.compile(r'xmlns:([^=]*)="http://www.w3.org/1999/02/22-rdf-syntax-ns#"').search(file_prefix.contents_header)
|
||||
if not match and (match.group(1) + ":RDF") in file_prefix.contents_header:
|
||||
return True
|
||||
return False
|
||||
|
||||
def set_peek(self, dataset, is_multi_byte=False):
|
||||
@@ -139,6 +144,7 @@ class Rdf(xml.GenericXml, Triples):
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
|
||||
@build_sniff_from_prefix
|
||||
class Jsonld(text.Json, Triples):
|
||||
"""
|
||||
The JSON-LD data format
|
||||
@@ -147,12 +153,10 @@ class Jsonld(text.Json, Triples):
|
||||
edam_format = "format_3464"
|
||||
file_ext = "jsonld"
|
||||
|
||||
def sniff(self, filename):
|
||||
if self._looks_like_json(filename):
|
||||
with open(filename, "r") as f:
|
||||
firstlines = "".join(f.readlines(5000))
|
||||
if "\"@id\"" in firstlines or "\"@context\"" in firstlines:
|
||||
return True
|
||||
def sniff_prefix(self, file_prefix):
|
||||
if self._looks_like_json(file_prefix):
|
||||
if "\"@id\"" in file_prefix.contents_header or "\"@context\"" in file_prefix.contents_header:
|
||||
return True
|
||||
return False
|
||||
|
||||
def set_peek(self, dataset, is_multi_byte=False):
|
||||
@@ -176,7 +180,6 @@ class HDT(binary.Binary, Triples):
|
||||
with open(filename, "rb") as f:
|
||||
if f.read(4) == "$HDT":
|
||||
return True
|
||||
return False
|
||||
|
||||
def set_peek(self, dataset, is_multi_byte=False):
|
||||
"""Set the peek and blurb text"""
|
||||
|
||||
+35
-31
@@ -6,13 +6,17 @@ import re
|
||||
|
||||
from . import (
|
||||
data,
|
||||
dataproviders
|
||||
dataproviders,
|
||||
sniff
|
||||
)
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
OWL_MARKER = re.compile(r'\<owl:')
|
||||
|
||||
|
||||
@dataproviders.decorators.has_dataproviders
|
||||
@sniff.build_sniff_from_prefix
|
||||
class GenericXml(data.Text):
|
||||
"""Base format class for any XML file."""
|
||||
edam_format = "format_2332"
|
||||
@@ -27,7 +31,17 @@ class GenericXml(data.Text):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
def _has_root_element_in_prefix(self, file_prefix, root):
|
||||
contents = file_prefix.string_io()
|
||||
while True:
|
||||
line = contents.readline()
|
||||
if line is None or not line.startswith('<?'):
|
||||
break
|
||||
# pattern match <root or <ns:root for any ns string
|
||||
pattern = '^<(\w*:)?%s' % root
|
||||
return line is not None and re.match(pattern, line) is not None
|
||||
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Determines whether the file is XML or not
|
||||
|
||||
@@ -39,12 +53,7 @@ class GenericXml(data.Text):
|
||||
>>> GenericXml().sniff( fname )
|
||||
False
|
||||
"""
|
||||
# TODO - Use a context manager on Python 2.5+ to close handle
|
||||
with open(filename) as handle:
|
||||
line = handle.readline()
|
||||
|
||||
# TODO - Is there a more robust way to do this?
|
||||
return line.startswith('<?xml ')
|
||||
return file_prefix.startswith('<?xml ')
|
||||
|
||||
def merge(split_files, output_file):
|
||||
"""Merging multiple XML files is non-trivial and must be done in subclasses."""
|
||||
@@ -60,6 +69,7 @@ class GenericXml(data.Text):
|
||||
return dataproviders.hierarchy.XMLDataProvider(dataset_source, **settings)
|
||||
|
||||
|
||||
@sniff.disable_parent_class_sniffing
|
||||
class MEMEXml(GenericXml):
|
||||
"""MEME XML Output data"""
|
||||
file_ext = "memexml"
|
||||
@@ -73,10 +83,8 @@ class MEMEXml(GenericXml):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
return False
|
||||
|
||||
|
||||
@sniff.disable_parent_class_sniffing
|
||||
class CisML(GenericXml):
|
||||
"""CisML XML data""" # see: http://www.ncbi.nlm.nih.gov/pubmed/15001475
|
||||
file_ext = "cisml"
|
||||
@@ -90,9 +98,6 @@ class CisML(GenericXml):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
return False
|
||||
|
||||
|
||||
class Phyloxml(GenericXml):
|
||||
"""Format for defining phyloxml data http://www.phyloxml.org/"""
|
||||
@@ -109,15 +114,21 @@ class Phyloxml(GenericXml):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disk'
|
||||
|
||||
def sniff(self, filename):
|
||||
""""Checking for keyword - 'phyloxml' always in lowercase in the first few lines"""
|
||||
def sniff_prefix(self, file_prefix):
|
||||
""""Checking for keyword - 'phyloxml' always in lowercase in the first few lines.
|
||||
|
||||
with open(filename, "r") as f:
|
||||
firstlines = "".join(f.readlines(5))
|
||||
|
||||
if "phyloxml" in firstlines:
|
||||
return True
|
||||
return False
|
||||
>>> from galaxy.datatypes.sniff import get_test_fname
|
||||
>>> fname = get_test_fname( '1.phyloxml' )
|
||||
>>> Phyloxml().sniff( fname )
|
||||
True
|
||||
>>> fname = get_test_fname( 'interval.interval' )
|
||||
>>> Phyloxml().sniff( fname )
|
||||
False
|
||||
>>> fname = get_test_fname( 'megablast_xml_parser_test1.blastxml' )
|
||||
>>> Phyloxml().sniff( fname )
|
||||
False
|
||||
"""
|
||||
return self._has_root_element_in_prefix(file_prefix, "phyloxml")
|
||||
|
||||
def get_visualizations(self, dataset):
|
||||
"""
|
||||
@@ -143,15 +154,8 @@ class Owl(GenericXml):
|
||||
dataset.peek = 'file does not exist'
|
||||
dataset.blurb = 'file purged from disc'
|
||||
|
||||
def sniff(self, filename):
|
||||
def sniff_prefix(self, file_prefix):
|
||||
"""
|
||||
Checking for keyword - '<owl' in the first 200 lines.
|
||||
"""
|
||||
owl_marker = re.compile(r'\<owl:')
|
||||
with open(filename) as handle:
|
||||
# Check first 200 lines for the string "<owl:"
|
||||
first_lines = handle.readlines(200)
|
||||
for line in first_lines:
|
||||
if owl_marker.search(line):
|
||||
return True
|
||||
return False
|
||||
return file_prefix.search(OWL_MARKER)
|
||||
|
||||
@@ -26,6 +26,10 @@ def get_fileobj(filename, mode="r", compressed_formats=None):
|
||||
:param compressed_formats: list of allowed compressed file formats among
|
||||
'bz2', 'gzip' and 'zip'. If left to None, all 3 formats are allowed
|
||||
"""
|
||||
return get_fileobj_raw(filename, mode, compressed_formats)[1]
|
||||
|
||||
|
||||
def get_fileobj_raw(filename, mode="r", compressed_formats=None):
|
||||
if compressed_formats is None:
|
||||
compressed_formats = ['bz2', 'gzip', 'zip']
|
||||
# Remove 't' from mode, which may cause an error for compressed files
|
||||
@@ -36,22 +40,26 @@ def get_fileobj(filename, mode="r", compressed_formats=None):
|
||||
cmode = 'r'
|
||||
else:
|
||||
cmode = mode
|
||||
compressed_format = None
|
||||
if 'gzip' in compressed_formats and is_gzip(filename):
|
||||
fh = gzip.GzipFile(filename, cmode)
|
||||
compressed_format = 'gzip'
|
||||
elif 'bz2' in compressed_formats and is_bz2(filename):
|
||||
fh = bz2.BZ2File(filename, cmode)
|
||||
compressed_format = 'bz2'
|
||||
elif 'zip' in compressed_formats and zipfile.is_zipfile(filename):
|
||||
# Return fileobj for the first file in a zip file.
|
||||
with zipfile.ZipFile(filename, cmode) as zh:
|
||||
fh = zh.open(zh.namelist()[0], cmode)
|
||||
compressed_format = 'zip'
|
||||
elif 'b' in mode:
|
||||
return open(filename, mode)
|
||||
return compressed_format, open(filename, mode)
|
||||
else:
|
||||
return io.open(filename, mode, encoding='utf-8')
|
||||
return compressed_format, io.open(filename, mode, encoding='utf-8')
|
||||
if 'b' not in mode:
|
||||
return io.TextIOWrapper(fh, encoding='utf-8')
|
||||
return compressed_format, io.TextIOWrapper(fh, encoding='utf-8')
|
||||
else:
|
||||
return fh
|
||||
return compressed_format, fh
|
||||
|
||||
|
||||
class CompressedFile(object):
|
||||
|
||||
Reference in New Issue
Block a user