Merge pull request #9336 from mvdbeek/legacy_tool_fixes

[20.01] Add requirements, fix legacy tools and move them out of GALAXY_LIB_TOOLS_UNVERSIONED
This commit is contained in:
John Chilton
2020-02-18 08:52:04 -05:00
committed by GitHub
16 changed files with 117 additions and 144 deletions
@@ -20,30 +20,26 @@ def main():
"""
input_fname = sys.argv[1]
if is_gzip(input_fname):
print('Conversion is only possible for uncompressed files')
sys.exit(1)
out_file = open(sys.argv[2], 'w')
sys.exit('Conversion is only possible for uncompressed files')
current_line = 0
sequences = 1000000
lines_per_chunk = 4 * sequences
chunk_begin = 0
in_file = open(input_fname)
with open(input_fname) as in_file, open(sys.argv[2], 'w') as out_file:
out_file.write('{"sections" : [')
out_file.write('{"sections" : [')
for line in in_file:
current_line += 1
if 0 == current_line % lines_per_chunk:
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"},' % (chunk_begin, chunk_end, sequences))
chunk_begin = chunk_end
for line in in_file:
current_line += 1
if 0 == current_line % lines_per_chunk:
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"},' % (chunk_begin, chunk_end, sequences))
chunk_begin = chunk_end
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"}' % (chunk_begin, chunk_end, (current_line % lines_per_chunk) / 4))
out_file.write(']}\n')
chunk_end = in_file.tell()
out_file.write('{"start":"%s","end":"%s","sequences":"%s"}' % (chunk_begin, chunk_end, (current_line % lines_per_chunk) / 4))
out_file.write(']}\n')
if __name__ == "__main__":
@@ -1,4 +1,7 @@
<tool id="CONVERTER_fastq_to_fqtoc0" name="Convert FASTQ files to seek locations" version="1.0.0" hidden="true">
<tool id="CONVERTER_fastq_to_fqtoc0" name="Convert FASTQ files to seek locations" version="1.0.1" hidden="true" profile="16.04">
<requirements>
<requirement type="package" version="19.09">galaxy-util</requirement>
</requirements>
<command>python '$__tool_directory__/fastq_to_fqtoc.py' '$input1' '$output1'</command>
<inputs>
<param format="fastq" name="input1" type="data" label="Choose FASTQ file"/>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_tar_to_directory" name="Convert tar to directory" version="1.0.0" profile="17.05">
<tool id="CONVERTER_tar_to_directory" name="Convert tar to directory" version="1.0.1" profile="17.05">
<!-- Don't use tar directly so we can verify safety of results - tar -xzf '$input1'; -->
<requirements>
<requirement type="package" version="19.09">galay-util</requirement>
</requirements>
<command>
mkdir '$output1.files_path';
cd '$output1.files_path';
@@ -11,40 +11,20 @@ import sys
import bx.wiggle
from galaxy.util import unicodify
from galaxy.util.ucsc import (
UCSCLimitException,
UCSCOutWrapper
)
def stop_err(msg):
sys.stderr.write(msg)
sys.exit(1)
def main():
if len(sys.argv) > 1:
in_file = open(sys.argv[1])
else:
in_file = open(sys.stdin)
if len(sys.argv) > 2:
out_file = open(sys.argv[2], "w")
else:
out_file = sys.stdout
try:
for fields in bx.wiggle.IntervalReader(UCSCOutWrapper(in_file)):
out_file.write("%s\n" % "\t".join(map(str, fields)))
except UCSCLimitException:
# Wiggle data was truncated, at the very least need to warn the user.
print('Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.')
except ValueError as e:
stop_err(unicodify(e))
finally:
in_file.close()
out_file.close()
with open(sys.argv[1]) as in_file, open(sys.argv[2], "w") as out_file:
try:
for fields in bx.wiggle.IntervalReader(UCSCOutWrapper(in_file)):
out_file.write("%s\n" % "\t".join(map(str, fields)))
except UCSCLimitException:
# Wiggle data was truncated, at the very least need to warn the user.
sys.stderr.write('Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.')
if __name__ == "__main__":
@@ -1,6 +1,10 @@
<tool id="CONVERTER_wiggle_to_interval_0" name="Wiggle to Interval" version="1.0.0">
<tool id="CONVERTER_wiggle_to_interval_0" name="Wiggle to Interval" version="1.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="19.09">galaxy-util</requirement>
<requirement type="package" version="0.8.6">bx-python</requirement>
</requirements>
<command>python '$__tool_directory__/wiggle_to_simple_converter.py' '$input' '$out_file1'</command>
<inputs>
<param format="wig" name="input" type="data" label="Convert"/>
+3 -4
View File
@@ -58,8 +58,7 @@ def get_species_in_block(block):
def tool_fail(msg="Unknown Error"):
print("Fatal Error: %s" % msg, file=sys.stderr)
sys.exit()
sys.exit("Fatal Error: %s" % msg)
class TempFileHandler(object):
@@ -103,11 +102,11 @@ class TempFileHandler(object):
else:
raise e
tmp_file.close()
self.files.append(open(filename, 'w+b'))
self.files.append(open(filename, 'w'))
else:
while True:
try:
self.files[index] = open(self.files[index].name, 'r+b')
self.files[index] = open(self.files[index].name, 'r')
break
except OSError as e:
if self.open_file_indexes and e.errno == EMFILE:
+8 -8
View File
@@ -127,13 +127,9 @@ GALAXY_LIB_TOOLS_UNVERSIONED = [
"send_to_cloud",
"__DATA_FETCH__",
# Legacy tools bundled with Galaxy.
"vcf_to_maf_customtrack1",
"laj_1",
"secure_hash_message_digest",
"join1",
"gff2bed1",
"gff_filter_by_feature_count",
"aggregate_scores_in_intervals2",
"Interval_Maf_Merged_Fasta2",
"GeneBed_Maf_Fasta2",
"maf_stats1",
@@ -146,16 +142,12 @@ GALAXY_LIB_TOOLS_UNVERSIONED = [
"MAF_split_blocks_by_species1",
"MAF_Limit_To_Species1",
"maf_by_block_number1",
"wiggle2simple1",
# Converters
"CONVERTER_bed_to_fli_0",
"CONVERTER_fastq_to_fqtoc0",
"CONVERTER_gff_to_fli_0",
"CONVERTER_gff_to_interval_index_0",
"CONVERTER_maf_to_fasta_0",
"CONVERTER_maf_to_interval_0",
"CONVERTER_wiggle_to_interval_0",
"CONVERTER_tar_to_directory",
# Tools improperly migrated to the tool shed (devteam)
"qualityFilter",
"winSplitter",
@@ -189,6 +181,14 @@ GALAXY_LIB_TOOLS_VERSIONED = {
"PEsortedSAM2readprofile": packaging.version.parse("1.1.1"),
"sam_to_bam": packaging.version.parse("1.1.3"),
"sam_pileup": packaging.version.parse("1.1.3"),
"vcf_to_maf_customtrack1": packaging.version.parse("1.0.1"),
"secure_hash_message_digest": packaging.version.parse("0.0.2"),
"join1": packaging.version.parse("2.1.3"),
"wiggle2simple1": packaging.version.parse("1.0.1"),
"CONVERTER_wiggle_to_interval_0": packaging.version.parse("1.0.1"),
"aggregate_scores_in_intervals2": packaging.version.parse("1.1.4"),
"CONVERTER_fastq_to_fqtoc0": packaging.version.parse("1.0.1"),
"CONVERTER_tar_to_directory": packaging.version.parse("1.0.1"),
}
+8 -8
View File
@@ -1,15 +1,15 @@
<tool id="join1" name="Join two Datasets" version="2.1.2">
<tool id="join1" name="Join two Datasets" version="2.1.3">
<description>side by side on a specified field</description>
<requirements>
<requirement type="package" version="2.7.13">python</requirement>
<requirement type="package" version="19.9.0">galaxy-util</requirement>
</requirements>
<command detect_errors="aggressive">
python '$__tool_directory__/join.py' '$input1' '$input2' $field1 $field2 '$out_file1' $unmatched $partial --index_depth=3 --buffer=50000000 --fill_options_file=$fill_options_file $header
</command>
<configfiles>
<configfile name="fill_options_file">&lt;%
<configfile name="fill_options_file"><![CDATA[<%
import json
%&gt;
%>
#set $__fill_options = {}
#if $fill_empty_columns['fill_empty_columns_switch'] == 'fill_empty':
#set $__fill_options['fill_unjoined_only'] = $fill_empty_columns['fill_columns_by'].value == 'fill_unjoined_only'
@@ -30,7 +30,7 @@ import json
#end if
#end if
${json.dumps( __fill_options )}
</configfile>
]]></configfile>
</configfiles>
<inputs>
<param format="tabular" name="input1" type="data" label="Join"/>
@@ -174,7 +174,7 @@ ${json.dumps( __fill_options )}
<output name="out_file1" file="joiner_out7.tab"/>
</test>
</tests>
<help>
<help><![CDATA
.. class:: warningmark
@@ -182,7 +182,7 @@ ${json.dumps( __fill_options )}
.. class:: infomark
**TIP:** If your data is not TAB delimited, use *Text Manipulation-&gt;Convert*
**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert*
-----
@@ -221,6 +221,6 @@ Joining the 4th column of Dataset1 with the 1st column of Dataset2, while keepin
chr1 50 80 geneB geneB Foxp2
chr5 10 40 geneL
</help>
]]></help>
<citations></citations>
</tool>
+6 -2
View File
@@ -1,6 +1,10 @@
<tool id="secure_hash_message_digest" name="Secure Hash / Message Digest" version="0.0.1">
<tool id="secure_hash_message_digest" name="Secure Hash / Message Digest" version="0.0.2" profile="18.01">
<description>on a dataset</description>
<command interpreter="python">secure_hash_message_digest.py --input "${input1}" --output "${out_file1}"
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>
python '$__tool_directory__/secure_hash_message_digest.py' --input '${input1}' --output '${out_file1}'
#if $algorithms.value:
#for $algorithm in str( $algorithms ).split( "," ):
--algorithm "${algorithm}"
+11 -29
View File
@@ -10,38 +10,20 @@ import sys
import bx.wiggle
from galaxy.util.ucsc import UCSCLimitException, UCSCOutWrapper
def stop_err(msg):
sys.stderr.write(msg)
sys.exit()
from galaxy.util.ucsc import (
UCSCLimitException,
UCSCOutWrapper
)
def main():
if len(sys.argv) > 1:
in_file = open(sys.argv[1])
else:
in_file = open(sys.stdin)
if len(sys.argv) > 2:
out_file = open(sys.argv[2], "w")
else:
out_file = sys.stdout
try:
for fields in bx.wiggle.IntervalReader(UCSCOutWrapper(in_file)):
out_file.write("%s\n" % "\t".join(map(str, fields)))
except UCSCLimitException:
# Wiggle data was truncated, at the very least need to warn the user.
print('Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.')
except ValueError as e:
in_file.close()
out_file.close()
stop_err(str(e))
in_file.close()
out_file.close()
with open(sys.argv[1]) as in_file, open(sys.argv[2], "w") as out_file:
try:
for fields in bx.wiggle.IntervalReader(UCSCOutWrapper(in_file)):
out_file.write("%s\n" % "\t".join(map(str, fields)))
except UCSCLimitException:
# Wiggle data was truncated, at the very least need to warn the user.
sys.stderr.write('Encountered message from UCSC: "Reached output limit of 100000 data values", so be aware your data was truncated.')
if __name__ == "__main__":
+6 -2
View File
@@ -1,6 +1,10 @@
<tool id="wiggle2simple1" name="Wiggle-to-Interval" version="1.0.0">
<tool id="wiggle2simple1" name="Wiggle-to-Interval" version="1.0.1" profile="16.04">
<description>converter</description>
<command interpreter="python">wiggle_to_simple.py $input $out_file1 </command>
<requirements>
<requirement type="package" version="19.09">galaxy-util</requirement>
<requirement type="package" version="0.8.6">bx-python</requirement>
</requirements>
<command>python '$__tool_directory__/wiggle_to_simple.py' '$input' '$out_file1'</command>
<inputs>
<param format="wig" name="input" type="data" label="Convert"/>
</inputs>
+1 -2
View File
@@ -35,8 +35,7 @@ from galaxy.tools.util import maf_utilities
def stop_err(msg):
sys.stderr.write(msg)
sys.exit()
sys.exit(msg)
def __main__():
+27 -35
View File
@@ -13,51 +13,43 @@ UNKNOWN_NUCLEOTIDE = '*'
class PopulationVCFParser(Iterator):
def __init__(self, reader, name):
self.reader = reader
self.reader = iter(reader)
self.name = name
self.counter = 0
def __next__(self):
rval = []
vc = next(self.reader)
for i, allele in enumerate(vc.alt):
rval.append(('%s_%i.%i' % (self.name, i + 1, self.counter + 1), allele))
self.counter += 1
return (vc, rval)
def __iter__(self):
while True:
yield next(self)
for vc in self.reader:
rval = []
for i, allele in enumerate(vc.alt):
rval.append(('%s_%i.%i' % (self.name, i + 1, self.counter + 1), allele))
self.counter += 1
yield (vc, rval)
class SampleVCFParser(Iterator):
def __init__(self, reader):
self.reader = reader
self.reader = iter(reader)
self.counter = 0
def __next__(self):
rval = []
vc = next(self.reader)
alleles = [vc.ref] + vc.alt
if 'GT' in vc.format:
gt_index = vc.format.index('GT')
for sample_name, sample_value in zip(vc.sample_names, vc.sample_values):
gt_indexes = []
for i in sample_value[gt_index].replace('|', '/').replace('\\', '/').split('/'): # Do we need to consider phase here?
try:
gt_indexes.append(int(i))
except Exception:
gt_indexes.append(None)
for i, allele_i in enumerate(gt_indexes):
if allele_i is not None:
rval.append(('%s_%i.%i' % (sample_name, i + 1, self.counter + 1), alleles[allele_i]))
self.counter += 1
return (vc, rval)
def __iter__(self):
while True:
yield next(self)
for vc in self.reader:
rval = []
alleles = [vc.ref] + vc.alt
if 'GT' in vc.format:
gt_index = vc.format.index('GT')
for sample_name, sample_value in zip(vc.sample_names, vc.sample_values):
gt_indexes = []
for i in sample_value[gt_index].replace('|', '/').replace('\\', '/').split('/'): # Do we need to consider phase here?
try:
gt_indexes.append(int(i))
except Exception:
gt_indexes.append(None)
for i, allele_i in enumerate(gt_indexes):
if allele_i is not None:
rval.append(('%s_%i.%i' % (sample_name, i + 1, self.counter + 1), alleles[allele_i]))
self.counter += 1
yield (vc, rval)
def main():
@@ -77,7 +69,7 @@ def main():
if not (options.population ^ options.sample):
parser.error('You must specify either a per population conversion or a per sample conversion, but not both')
out = open(args.pop(0), 'wb')
out = open(args.pop(0), 'w')
out.write('track name="%s" visibility=pack\n' % options.name.replace("\"", "'"))
maf_writer = bx.align.maf.Writer(out)
+9 -5
View File
@@ -1,5 +1,9 @@
<tool id="vcf_to_maf_customtrack1" name="VCF to MAF Custom Track" version="1.0.0">
<tool id="vcf_to_maf_customtrack1" name="VCF to MAF Custom Track" version="1.0.1" profile="18.01">
<description>for display at UCSC</description>
<requirements>
<requirement type="package" version="1.1.4">galaxy_sequence_utils</requirement>
<requirement type="package" version="0.8.6">bx-python</requirement>
</requirements>
<macros>
<import>macros.xml</import>
</macros>
@@ -43,21 +47,21 @@ ${vcf_source_type.vcf_source} -n '$track_name'
<outputs>
<data format="mafcustomtrack" name="out_file1" />
</outputs>
<!-- <tests>
<tests>
<test>
<param name="track_name" value="Galaxy Custom Track"/>
<param name="vcf_source" value="Per Population"/>
<param name="vcf_source" value="-p"/>
<param name="vcf_input" value="vcf_to_maf_in.vcf" ftype="tabular"/>
<param name="population_name" value=""/>
<output name="out_file1" file="vcf_to_maf_population_out.mafcustomtrack"/>
</test>
<test>
<param name="track_name" value="Galaxy Custom Track"/>
<param name="vcf_source" value="Per Sample"/>
<param name="vcf_source" value="-s"/>
<param name="vcf_input" value="vcf_to_maf_in.vcf" ftype="tabular"/>
<output name="out_file1" file="vcf_to_maf_sample_out.mafcustomtrack"/>
</test>
</tests> -->
</tests>
<help>
**What it does**
@@ -1,5 +1,9 @@
<tool id="aggregate_scores_in_intervals2" name="Aggregate datapoints" version="1.1.3">
<tool id="aggregate_scores_in_intervals2" name="Aggregate datapoints" version="1.1.4" profile="16.04">
<description>Appends the average, min, max of datapoints per interval</description>
<requirements>
<requirement type="package" version="19.09">galaxy-util</requirement>
<requirement type="package" version="0.8.6">bx-python</requirement>
</requirements>
<command>
python '$__tool_directory__/aggregate_scores_in_intervals.py'
#if $score_source_type.score_source == "user"
+1 -2
View File
@@ -99,8 +99,7 @@ class FileBinnedArrayDir(Mapping):
def stop_err(msg):
sys.stderr.write(msg)
sys.exit()
sys.exit(msg)
def load_scores_wiggle(fname, chrom_buffer_size=3):