Merge pull request #9345 from mvdbeek/converter_requirements

[20.01] Add requirements to (almost) all converters, profile and minor fixes
This commit is contained in:
John Chilton
2020-02-14 14:24:29 -05:00
committed by GitHub
97 changed files with 729 additions and 632 deletions
+29
View File
@@ -0,0 +1,29 @@
name: Converter tests
on: [push, pull_request]
jobs:
test:
name: Test
runs-on: ubuntu-18.04
strategy:
matrix:
python-version: [3.7]
steps:
- uses: actions/checkout@v1
with:
fetch-depth: 1
- uses: actions/setup-python@v1
with:
python-version: ${{ matrix.python-version }}
- name: Cache venv dir
uses: actions/cache@v1
id: pip-cache
with:
path: ~/.cache/pip
key: pip-cache-${{ matrix.python-version }}-${{ hashFiles('requirements.txt') }}
- name: fetch test data
run: git clone https://github.com/galaxyproject/galaxy-test-data && cp -R galaxy-test-data/* test-data
- name: Install planemo
run: pip install planemo
- name: Run tests
run: 'planemo test --galaxy_python_version ${{ matrix.python-version }} --galaxy_root . lib/galaxy/datatypes/converters/*xml'
@@ -142,6 +142,7 @@
<converter file="gff_to_interval_index_converter.xml" target_datatype="interval_index"/>
<converter file="bed_gff_or_vcf_to_bigwig_converter.xml" target_datatype="bigwig"/>
<converter file="gff_to_fli_converter.xml" target_datatype="fli"/>
<converter file="gff_to_bgzip_converter.xml" target_datatype="bgzip"/>
<display file="ensembl/ensembl_gff.xml" inherit="true"/>
<display file="igv/gff.xml" inherit="true"/>
<!-- <display file="gbrowse/gbrowse_gff.xml" inherit="true" /> -->
@@ -1,14 +1,20 @@
<tool id="CONVERTER_Bam_Bai_0" name="Bam to Bai" version="1.0.0" hidden="true">
<tool id="CONVERTER_Bam_Bai_0" name="Bam to Bai" version="1.0.0" hidden="true" profile="16.04">
<requirements>
<requirement type="package">samtools</requirement>
<requirement type="package" version="1.10">samtools</requirement>
</requirements>
<command>samtools index '$input1' '$output1'</command>
<command>samtools index -@ \${GALAXY_SLOTS:-1} '$input1' '$output1'</command>
<inputs>
<param format="bam" name="input1" type="data" label="Choose BAM"/>
</inputs>
<outputs>
<data format="bai" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="bam" value="srma_out2.bam"/>
<output name="output1" format="bai" value="srma_out2.bai"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,8 +1,9 @@
<tool id="CONVERTER_bam_to_bigwig_0" name="Convert BAM to BigWig" version="1.0.2" hidden="true">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="357">ucsc-bedgraphtobigwig</requirement>
<requirement type="package" version="2.27.1">bedtools</requirement>
<requirement type="package" version="377">ucsc-bedgraphtobigwig</requirement>
<requirement type="package" version="2.29.2">bedtools</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command detect_errors="aggressive"><![CDATA[
bedtools genomecov -bg -split -ibam '$input'
@@ -21,6 +22,12 @@ bedtools genomecov -bg -split -ibam '$input'
<outputs>
<data name="output" format="bigwig"/>
</outputs>
<tests>
<test>
<param name="input" format="bam" value="srma_out2.bam" dbkey="hg17"/>
<output name="output" format="bigwig" value="srma_out2.bigwig"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -12,6 +12,12 @@ bcftools view -o '$output1' -O u '$input1'
<outputs>
<data name="output1" format="bcf_uncompressed" />
</outputs>
<tests>
<test>
<param name="input1" format="bcf" value="bcf_index_metadata_test.bcf"/>
<output name="output1" format="bcf_uncompressed" value="bcf_uncompressed_index_metadata_test.bcf_uncompressed" compare="sim_size"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -12,6 +12,12 @@ bcftools view -o '$output1' -O b '$input1'
<outputs>
<data name="output1" format="bcf" />
</outputs>
<tests>
<test>
<param name="input1" format="bcf_uncompressed" value="bcf_uncompressed_index_metadata_test.bcf_uncompressed"/>
<output name="output1" format="bcf" value="bcf_index_metadata_test.bcf" compare="sim_size"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -38,6 +38,12 @@
<outputs>
<data format="bigwig" name="output"/>
</outputs>
<tests>
<test>
<param name="input" format="bed" value="droPer1.bed" dbkey="hg17"/>
<output name="output" format="bigwig" value="droPer1.bigwig"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,9 @@
<tool id="CONVERTER_bed_to_bgzip_0" name="Convert BED to BGZIP" version="1.0.0" hidden="true">
<tool id="CONVERTER_bed_to_bgzip_0" name="Convert BED to BGZIP" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command>python '$__tool_directory__/bgzip.py' -P bed '$input1' '$output1'</command>
<inputs>
<param format="bed" name="input1" type="data" label="Choose BED file"/>
@@ -7,6 +11,12 @@
<outputs>
<data format="bgzip" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="bed" value="droPer1.bed"/>
<output name="output1" format="bgzip" value="droPer1.bgzip"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -8,6 +8,12 @@
<outputs>
<data format="fli" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="bed" value="droPer1.bed"/>
<output name="output1" format="fli" value="droPer1.fli"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,6 +1,9 @@
<tool id="CONVERTER_bed_to_gff_0" name="Convert BED to GFF" version="2.0.0">
<tool id="CONVERTER_bed_to_gff_0" name="Convert BED to GFF" version="2.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/bed_to_gff_converter.py' '$input1' '$output1'</command>
<inputs>
<param format="bed" name="input1" type="data" label="Choose BED file"/>
@@ -8,6 +11,12 @@
<outputs>
<data format="gff" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="bed" value="droPer1.bed"/>
<output name="output1" format="gff" value="droPer1.gff"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_bed_to_interval_index_0" name="Convert BED to Interval Index" version="1.0.0" hidden="true">
<tool id="CONVERTER_bed_to_interval_index_0" name="Convert BED to Interval Index" version="1.0.0" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.8.8">bx-python</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_interval_index_converter.py' '$input1' '$output1'</command>
<inputs>
<param format="bed" name="input1" type="data" label="Choose BED file"/>
@@ -7,6 +10,12 @@
<outputs>
<data format="interval_index" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="bed" value="droPer1.bed"/>
<output name="output1" format="interval_index" value="droPer1.interval_index"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_bed_to_tabix_0" name="Convert BED to tabix" version="1.0.0" hidden="true">
<tool id="CONVERTER_bed_to_tabix_0" name="Convert BED to tabix" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_tabix_converter.py' -P bed '$input1' '$bgzip' '$output1'</command>
<inputs>
<param format="bed" name="input1" type="data" label="Choose BED file"/>
@@ -8,6 +11,12 @@
<outputs>
<data format="tabix" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="bed" value="droPer1.bed"/>
<output name="output1" format="tabix" value="droPer1.tabix"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,57 +0,0 @@
#!/usr/bin/env python
from __future__ import division
import sys
from bx.arrays.array_tree import array_tree_dict_from_reader, FileArrayTreeDict
from six import Iterator
BLOCK_SIZE = 100
class BedGraphReader(Iterator):
def __init__(self, f):
self.f = f
def __iter__(self):
return self
def __next__(self):
while True:
line = self.f.readline()
if not line:
raise StopIteration()
if line.isspace():
continue
if line[0] == "#":
continue
if line[0].isalpha():
if line.startswith("track") or line.startswith("browser"):
continue
feature = line.strip().split()
chrom = feature[0]
chrom_start = int(feature[1])
chrom_end = int(feature[2])
score = float(feature[3])
return chrom, chrom_start, chrom_end, None, score
def main():
input_fname = sys.argv[1]
out_fname = sys.argv[2]
reader = BedGraphReader(open(input_fname))
# Fill array from reader
d = array_tree_dict_from_reader(reader, {}, block_size=BLOCK_SIZE)
for array_tree in d.values():
array_tree.root.build_summary()
FileArrayTreeDict.dict_to_file(d, open(out_fname, "w"))
if __name__ == "__main__":
main()
@@ -1,12 +0,0 @@
<tool id="CONVERTER_BedGraph_0" name="Index BedGraph for Track Viewer" version="1.0.0" hidden="true">
<!-- Used internally to generate track indexes -->
<command>python '$__tool_directory__/bedgraph_to_array_tree_converter.py' '$input' '$output'</command>
<inputs>
<param format="bedgraph" name="input" type="data" label="Choose BedGraph"/>
</inputs>
<outputs>
<data format="array_tree" name="output"/>
</outputs>
<help>
</help>
</tool>
@@ -10,6 +10,12 @@
<outputs>
<data format="bigwig" name="output"/>
</outputs>
<tests>
<test>
<param name="input" format="bedgraph" value="1.bedgraph" dbkey="hg17"/>
<output name="output" format="bigwig" value="1.bedgrpah_to_bigwig.bigwig"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -9,6 +9,12 @@
<outputs>
<data name="output" format="biom2"/>
</outputs>
<tests>
<test>
<param name="input" format="biom1" value="input_taxonomy.biom1"/>
<output name="output" format="biom2" value="input_taxonomy.biom2" compare="sim_size"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -9,6 +9,16 @@
<outputs>
<data name="output" format="biom1"/>
</outputs>
<tests>
<test>
<param name="input" ftype="biom2" value="input_taxonomy.biom2"/>
<output name="output">
<assert_contents>
<has_text text="Biological Observation Matrix 1.0.0"/>
</assert_contents>
</output>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,4 +1,7 @@
<tool id="CONVERTER_bz2_to_uncompressed" name="Convert compressed file to uncompressed." version="1.0.0" hidden="true" profile="17.09">
<tool id="CONVERTER_bz2_to_uncompressed" name="Convert compressed file to uncompressed." version="1.0.1" hidden="true" profile="17.09">
<requirements>
<requirement type="package" version="1.0.8">bzip2</requirement>
</requirements>
<command><![CDATA[
cp '$ext_config' 'galaxy.json' && bzip2 -dcf '$input1' > '$output1'
]]></command>
@@ -1,4 +1,7 @@
<tool id="CONVERTER_cram_to_bam_0" name="Convert CRAM to BAM" version="1.0.0" hidden="true">
<tool id="CONVERTER_cram_to_bam_0" name="Convert CRAM to BAM" version="1.0.0" hidden="true" profile="16.04">
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
</requirements>
<command><![CDATA[python $__tool_directory__/cram_to_bam.py '$input' '$output']]></command>
<inputs>
<param format="cram" name="input" type="data" label="Choose CRAM file"/>
@@ -1,5 +1,8 @@
<tool id="csv_to_tabular" name="Convert CSV to tabular" version="1.0.0">
<tool id="csv_to_tabular" name="Convert CSV to tabular" version="1.0.0" profile="16.04">
<description></description>
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/tabular_csv.py' -o '$tabular' -i '$csv'</command>
<inputs>
<param format="csv" name="csv" type="data" label="Choose file with comma-separated values"/>
@@ -7,6 +10,12 @@
<outputs>
<data format="tabular" name="tabular"/>
</outputs>
<tests>
<test>
<param name="csv" format="csv" value="1.csv"/>
<output name="tabular" format="tabular" value="1.csv_to_tabular.tabular"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -0,0 +1,2 @@
#<dbkey> <display_name> <len_file_path>
hg17 hg17 ${__HERE__}/hg17.len
@@ -1,5 +1,9 @@
<tool id="CONVERTER_encodepeak_to_bgzip_0" name="Convert ENCODEPeak to BGZIP" version="1.0.0" hidden="true">
<tool id="CONVERTER_encodepeak_to_bgzip_0" name="Convert ENCODEPeak to BGZIP" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command>
python '$__tool_directory__/bgzip.py'
-c ${input1.metadata.chromCol}
@@ -1,5 +1,8 @@
<tool id="CONVERTER_encodepeak_to_tabix_0" name="Convert ENCODEPeak to tabix" version="1.0.0" hidden="true">
<tool id="CONVERTER_encodepeak_to_tabix_0" name="Convert ENCODEPeak to tabix" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
</requirements>
<command>
python '$__tool_directory__/interval_to_tabix_converter.py'
-c ${input1.metadata.chromCol}
@@ -11,6 +11,12 @@
<outputs>
<data name="output" format="twobit"/>
</outputs>
<tests>
<test>
<param name="input" format="fasta" value="chr_m.fasta"/>
<output name="output" format="twobit" value="chr_m.twobit"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -18,6 +18,12 @@
<outputs>
<data name="output" format="bowtie_base_index"/>
</outputs>
<tests>
<test>
<param name="input" format="fasta" value="chr_m.fasta"/>
<output name="output" format="bowtie_base_index" value="chr_m.bowtie_base_index"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -19,6 +19,12 @@
<outputs>
<data name="output" format="bowtie_color_index"/>
</outputs>
<tests>
<test>
<param name="input" format="fasta" value="chr_m.fasta"/>
<output name="output" format="bowtie_color_index" value="chr_m.bowtie_color_index"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -12,6 +12,12 @@ samtools faidx temp.fasta
<outputs>
<data name="output" format="fai" from_work_dir="temp.fasta.fai" />
</outputs>
<tests>
<test>
<param name="input" format="fasta" value="chr_m.fasta"/>
<output name="output" format="fai" value="chr_m.fai"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,6 +1,9 @@
<tool id="CONVERTER_fasta_to_len" name="Convert FASTA to len file" version="1.0.0">
<tool id="CONVERTER_fasta_to_len" name="Convert FASTA to len file" version="1.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/fasta_to_len.py' '$input' '$output' 0</command>
<inputs>
<param name="input" type="data" format="fasta" label="Fasta file"/>
@@ -8,6 +11,12 @@
<outputs>
<data name="output" format="len"/>
</outputs>
<tests>
<test>
<param name="input" format="fasta" value="chr_m.fasta"/>
<output name="output" format="len" value="chr_m.len"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,6 +1,9 @@
<tool id="CONVERTER_fasta_to_tabular" name="Convert FASTA to Tabular" version="1.0.1">
<tool id="CONVERTER_fasta_to_tabular" name="Convert FASTA to Tabular" version="1.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/fasta_to_tabular_converter.py' '$input' '$output'</command>
<inputs>
<param name="input" type="data" format="fasta" label="Fasta file"/>
@@ -8,6 +11,12 @@
<outputs>
<data name="output" format="tabular"/>
</outputs>
<tests>
<test>
<param name="input" format="fasta" value="chr_m.fasta"/>
<output name="output" format="tabular" value="chr_m.tabular"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -6,6 +6,12 @@
<outputs>
<data format="fqtoc" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="fastq" value="shrimp_wrapper_test1.fastq"/>
<output name="output1" format="fqtoc" value="shrimp_wrapper_test1.fqtoc"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -19,8 +19,7 @@ assert sys.version_info[:2] >= (2, 4)
def stop_err(msg):
sys.stderr.write("%s" % msg)
sys.exit()
sys.exit("%s" % msg)
def __main__():
@@ -1,5 +1,8 @@
<tool id="CONVERTER_fastqsolexa_to_fasta_0" name="Convert Fastqsolexa to Fasta" version="1.0.0">
<tool id="CONVERTER_fastqsolexa_to_fasta_0" name="Convert Fastqsolexa to Fasta" version="1.0.0" profile="16.04">
<description>converts Fastqsolexa file to Fasta format</description>
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/fastqsolexa_to_fasta_converter.py' '$input' '$output'</command>
<inputs>
<param name="input" type="data" format="fastqsolexa" label="Choose Fastqsolexa file"/>
@@ -7,6 +10,12 @@
<outputs>
<data name="output" format="fasta"/>
</outputs>
<tests>
<test>
<param name="input" format="fastqsolexa" value="2.fastqsolexa"/>
<output name="output" format="fasta" value="2.fastqsolexa.fasta"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -19,8 +19,7 @@ assert sys.version_info[:2] >= (2, 4)
def stop_err(msg):
sys.stderr.write("%s" % msg)
sys.exit()
sys.exit("%s" % msg)
def __main__():
@@ -1,4 +1,7 @@
<tool id="CONVERTER_fastqsolexa_to_qual_0" name="Convert Fastqsolexa to Qual" version="1.0.0">
<tool id="CONVERTER_fastqsolexa_to_qual_0" name="Convert Fastqsolexa to Qual" version="1.0.1" profile="16.04">
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/fastqsolexa_to_qual_converter.py' '$input1' '$output1' ${input1.extension}</command>
<inputs>
<param format="fastqsolexa" name="input1" type="data" label="Choose Fastqsolexa file"/>
@@ -6,6 +9,12 @@
<outputs>
<data format="qualsolexa" name="output1" />
</outputs>
<tests>
<test>
<param name="input1" format="fastqsolexa" value="2.fastqsolexa"/>
<output name="output1" format="qualsolexa" value="2.qualsolexa"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,4 +1,4 @@
<tool id="CONVERTER_gff_to_bed_0" name="Convert GFF to BED" version="1.0.0">
<tool id="CONVERTER_gff_to_bed_0" name="Convert GFF to BED" version="1.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command>python '$__tool_directory__/gff_to_bed_converter.py' '$input1' '$output1'</command>
@@ -8,6 +8,12 @@
<outputs>
<data format="bed" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="gff" value="gff_filter_by_feature_count_out2.gff"/>
<output name="output1" format="bed" value="bed_filter_by_feature_count_out2.bed"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,9 @@
<tool id="CONVERTER_gff_to_bgzip_0" name="Convert GFF to BGZIP" version="1.0.0" hidden="true">
<tool id="CONVERTER_gff_to_bgzip_0" name="Convert GFF to BGZIP" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command>python '$__tool_directory__/bgzip.py' -P gff '$input1' '$output1'</command>
<inputs>
<param format="gff" name="input1" type="data" label="Choose GFF file"/>
@@ -7,6 +11,12 @@
<outputs>
<data format="bgzip" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="gff" value="gff_filter_by_feature_count_out2.gff"/>
<output name="output1" format="bgzip" value="bgzip_filter_by_feature_count_out2.bgzip"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,4 +1,4 @@
<tool id="CONVERTER_gff_to_fli_0" name="Convert GFF to Feature Location Index" version="1.0.0">
<tool id="CONVERTER_gff_to_fli_0" name="Convert GFF to Feature Location Index" version="1.0.0" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<command>python '$__tool_directory__/interval_to_fli.py' -F $input1.extension '$input1' '$output1'</command>
@@ -8,6 +8,12 @@
<outputs>
<data format="fli" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="gff" value="gff_filter_by_feature_count_out2.gff"/>
<output name="output1" format="fli" value="fli_filter_by_feature_count_out2.fli"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -34,7 +34,7 @@ def main():
# not included in the index.
offset += feature.raw_size
index.write(open(out_fname, "w"))
index.write(open(out_fname, "wb"))
if __name__ == "__main__":
@@ -1,4 +1,4 @@
<tool id="CONVERTER_gff_to_interval_index_0" name="Convert GFF to Interval Index" version="1.0.0" hidden="true">
<tool id="CONVERTER_gff_to_interval_index_0" name="Convert GFF to Interval Index" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command>python '$__tool_directory__/gff_to_interval_index_converter.py' '$input1' '$output1'</command>
<inputs>
@@ -7,6 +7,12 @@
<outputs>
<data format="interval_index" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="gff" value="gff_filter_by_feature_count_out2.gff"/>
<output name="output1" format="interval_index" value="interval_index_filter_by_feature_count_out2.interval_index"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_gff_to_tabix_0" name="Convert GFF to tabix" version="1.0.0" hidden="true">
<tool id="CONVERTER_gff_to_tabix_0" name="Convert GFF to tabix" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_tabix_converter.py' -P gff '$input1' '$bgzip' '$output1'</command>
<inputs>
<param format="gff" name="input1" type="data" label="Choose GFF file"/>
@@ -8,6 +11,13 @@
<outputs>
<data format="tabix" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="gff" value="gff_filter_by_feature_count_out2.gff"/>
<param name="bgzip" format="gff.gz" value="gff_filter_by_feature_count_out2.gff"/>
<output name="output1" format="tabix" value="tabix_filter_by_feature_count_out2.tabix"/>
</test>
</tests>
<help>
</help>
</tool>
+25
View File
@@ -0,0 +1,25 @@
dummy_chr 100000000
super_1 100000000
chr1 100000000
chr7 200000000
chrX 200000000
phiX174 100000000
random_phiX_region_1 100000000
random_phiX_region_2 100000000
random_phiX_region_3 100000000
random_phiX_region_4 100000000
random_phiX_region_5 100000000
random_phiX_region_6 100000000
random_phiX_region_7 100000000
random_phiX_region_8 100000000
random_phiX_region_9 100000000
random_phiX_region_10 100000000
random_phiX_region_11 100000000
random_phiX_region_12 100000000
random_phiX_region_13 100000000
random_phiX_region_14 100000000
random_phiX_region_15 100000000
random_phiX_region_16 100000000
random_phiX_region_17 100000000
random_phiX_region_18 100000000
random_phiX_region_19 100000000
@@ -1,5 +1,8 @@
<tool id="CONVERTER_interval_to_bed12_0" name="Convert Genomic Intervals To Strict BED12" version="1.0.0">
<tool id="CONVERTER_interval_to_bed12_0" name="Convert Genomic Intervals To Strict BED12" version="1.0.0" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.8.8">bx-python</requirement>
</requirements>
<command>
python '$__tool_directory__/interval_to_bedstrict_converter.py'
'$output1' '$input1' ${input1.metadata.chromCol}
@@ -13,6 +16,12 @@
<outputs>
<data format="bed12" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval"/>
<output name="output1" format="bed12" value="2.bed12"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_interval_to_bed6_0" name="Convert Genomic Intervals To Strict BED6" version="1.0.0">
<tool id="CONVERTER_interval_to_bed6_0" name="Convert Genomic Intervals To Strict BED6" version="1.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.8.8">bx-python</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_bedstrict_converter.py' '$output1' '$input1' ${input1.metadata.chromCol} ${input1.metadata.startCol} ${input1.metadata.endCol} ${input1.metadata.strandCol} ${input1.metadata.nameCol} ${input1.extension} 6</command>
<inputs>
<param format="interval" name="input1" type="data" label="Choose intervals"/>
@@ -7,6 +10,12 @@
<outputs>
<data format="bed6" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval"/>
<output name="output1" format="bed6" value="2.bed6"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -10,8 +10,7 @@ assert sys.version_info[:2] >= (2, 6)
def stop_err(msg):
sys.stderr.write(msg)
sys.exit()
sys.exit(msg)
def __main__():
@@ -1,8 +1,7 @@
<tool id="CONVERTER_interval_to_bed_0" name="Convert Genomic Intervals To BED" version="1.0.0">
<tool id="CONVERTER_interval_to_bed_0" name="Convert Genomic Intervals To BED" version="1.0.0" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="2.7">python</requirement>
<requirement type="package" version="0.8.6">bx-python</requirement>
<requirement type="package" version="0.8.8">bx-python</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_bed_converter.py' '$output1' '$input1' ${input1.metadata.chromCol} ${input1.metadata.startCol} ${input1.metadata.endCol} ${input1.metadata.strandCol} ${input1.metadata.nameCol}</command>
<inputs>
@@ -11,6 +10,12 @@
<outputs>
<data format="bed" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval"/>
<output name="output1" format="bed" value="2.interval_to_bed.bed"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -10,8 +10,7 @@ assert sys.version_info[:2] >= (2, 6)
def stop_err(msg):
sys.stderr.write(msg)
sys.exit()
sys.exit(msg)
def force_bed_field_count(fields, region_count, force_num_columns):
@@ -1,5 +1,8 @@
<tool id="CONVERTER_interval_to_bedstrict_0" name="Convert Genomic Intervals To Strict BED" version="1.0.0">
<tool id="CONVERTER_interval_to_bedstrict_0" name="Convert Genomic Intervals To Strict BED" version="1.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.8.8">bx-python</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_bedstrict_converter.py' '$output1' '$input1' ${input1.metadata.chromCol} ${input1.metadata.startCol} ${input1.metadata.endCol} ${input1.metadata.strandCol} ${input1.metadata.nameCol} ${input1.extension}</command>
<inputs>
<param format="interval" name="input1" type="data" label="Choose intervals"/>
@@ -7,6 +10,12 @@
<outputs>
<data format="bedstrict" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval"/>
<output name="output1" format="bedstrict" value="2.bedstrict"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,9 @@
<tool id="CONVERTER_interval_to_bgzip_0" name="Convert Interval to BGZIP" version="1.0.0" hidden="true">
<tool id="CONVERTER_interval_to_bgzip_0" name="Convert Interval to BGZIP" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command>
python '$__tool_directory__/bgzip.py'
-c ${input1.metadata.chromCol}
@@ -13,6 +17,12 @@
<outputs>
<data format="bgzip" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval"/>
<output name="output1" format="bgzip" value="2.bgzip"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,8 +1,9 @@
<tool id="CONVERTER_interval_to_bigwig_0" name="Convert Genomic Intervals To Coverage" version="1.0.1">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="357">ucsc-bedgraphtobigwig</requirement>
<requirement type="package" version="2.27.1">bedtools</requirement>
<requirement type="package" version="377">ucsc-bedgraphtobigwig</requirement>
<requirement type="package" version="2.29.2">bedtools</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command>
<![CDATA[
@@ -29,6 +30,12 @@
<outputs>
<data format="bigwig" name="output"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval" dbkey="hg17"/>
<output name="output" format="bigwig" value="2.bigwig" compare="sim_size"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,151 +0,0 @@
#!/usr/bin/env python
"""
Converter to generate 3 (or 4) column base-pair coverage from an interval file.
usage: %prog bed_file out_file
-1, --cols1=N,N,N,N: Columns for chrom, start, end, strand in interval file
-2, --cols2=N,N,N,N: Columns for chrom, start, end, strand in coverage file
"""
import subprocess
import tempfile
from bisect import bisect
from os import environ
from bx.cookbook import doc_optparse
from bx.intervals import io
INTERVAL_METADATA = ('chromCol',
'startCol',
'endCol',
'strandCol',)
COVERAGE_METADATA = ('chromCol',
'positionCol',
'forwardCol',
'reverseCol',)
def main(interval, coverage):
"""
Uses a sliding window of partitions to count coverages.
Every interval record adds its start and end to the partitions. The result
is a list of partitions, or every position that has a (maybe) different
number of basepairs covered. We don't worry about merging because we pop
as the sorted intervals are read in. As the input start positions exceed
the partition positions in partitions, coverages are kicked out in bulk.
"""
partitions = []
forward_covs = []
reverse_covs = []
chrom = None
lastchrom = None
for record in interval:
chrom = record.chrom
if lastchrom and not lastchrom == chrom and partitions:
for partition in range(0, len(partitions) - 1):
forward = forward_covs[partition]
reverse = reverse_covs[partition]
if forward + reverse > 0:
coverage.write(chrom=chrom, position=range(partitions[partition], partitions[partition + 1]),
forward=forward, reverse=reverse)
partitions = []
forward_covs = []
reverse_covs = []
start_index = bisect(partitions, record.start)
forward = int(record.strand == "+")
reverse = int(record.strand == "-")
forward_base = 0
reverse_base = 0
if start_index > 0:
forward_base = forward_covs[start_index - 1]
reverse_base = reverse_covs[start_index - 1]
partitions.insert(start_index, record.start)
forward_covs.insert(start_index, forward_base)
reverse_covs.insert(start_index, reverse_base)
end_index = bisect(partitions, record.end)
for index in range(start_index, end_index):
forward_covs[index] += forward
reverse_covs[index] += reverse
partitions.insert(end_index, record.end)
forward_covs.insert(end_index, forward_covs[end_index - 1] - forward)
reverse_covs.insert(end_index, reverse_covs[end_index - 1] - reverse)
if partitions:
for partition in range(0, start_index):
forward = forward_covs[partition]
reverse = reverse_covs[partition]
if forward + reverse > 0:
coverage.write(chrom=chrom, position=range(partitions[partition], partitions[partition + 1]),
forward=forward, reverse=reverse)
partitions = partitions[start_index:]
forward_covs = forward_covs[start_index:]
reverse_covs = reverse_covs[start_index:]
lastchrom = chrom
# Finish the last chromosome
if partitions:
for partition in range(0, len(partitions) - 1):
forward = forward_covs[partition]
reverse = reverse_covs[partition]
if forward + reverse > 0:
coverage.write(chrom=chrom, position=range(partitions[partition], partitions[partition + 1]),
forward=forward, reverse=reverse)
class CoverageWriter(object):
def __init__(self, out_stream=None, chromCol=0, positionCol=1, forwardCol=2, reverseCol=3):
self.out_stream = out_stream
self.reverseCol = reverseCol
self.nlines = 0
positions = {str(chromCol): '%(chrom)s',
str(positionCol): '%(position)d',
str(forwardCol): '%(forward)d',
str(reverseCol): '%(reverse)d'}
if reverseCol < 0:
self.template = "%(0)s\t%(1)s\t%(2)s\n" % positions
else:
self.template = "%(0)s\t%(1)s\t%(2)s\t%(3)s\n" % positions
def write(self, **kwargs):
if self.reverseCol < 0:
kwargs['forward'] += kwargs['reverse']
posgen = kwargs['position']
for position in posgen:
kwargs['position'] = position
self.out_stream.write(self.template % kwargs)
def close(self):
self.out_stream.flush()
self.out_stream.close()
if __name__ == "__main__":
options, args = doc_optparse.parse(__doc__)
try:
chr_col_1, start_col_1, end_col_1, strand_col_1 = [int(x) - 1 for x in options.cols1.split(',')]
chr_col_2, position_col_2, forward_col_2, reverse_col_2 = [int(x) - 1 for x in options.cols2.split(',')]
in_fname, out_fname = args
except Exception:
doc_optparse.exception()
# Sort through a tempfile first
with tempfile.NamedTemporaryFile(mode="r") as temp_file:
environ['LC_ALL'] = 'POSIX'
subprocess.check_call([
'sort', '-f', '-n', '-k', chr_col_1 + 1, '-k', start_col_1 + 1, '-k', end_col_1 + 1, '-o', temp_file.name, in_fname
])
coverage = CoverageWriter(out_stream=open(out_fname, "a"),
chromCol=chr_col_2, positionCol=position_col_2,
forwardCol=forward_col_2, reverseCol=reverse_col_2, )
temp_file.seek(0)
interval = io.NiceReaderWrapper(temp_file,
chrom_col=chr_col_1,
start_col=start_col_1,
end_col=end_col_1,
strand_col=strand_col_1,
fix_strand=True)
main(interval, coverage)
coverage.close()
@@ -1,16 +0,0 @@
<tool id="CONVERTER_interval_to_coverage_0" name="Convert Genomic Intervals To COVERAGE" version="1.0.0">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command>
python '$__tool_directory__/interval_to_coverage.py' '$input1' '$output1'
-1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol}
-2 ${output1.metadata.chromCol},${output1.metadata.positionCol},${output1.metadata.forwardCol},${output1.metadata.reverseCol}
</command>
<inputs>
<param format="interval" name="input1" type="data" label="Choose intervals"/>
</inputs>
<outputs>
<data format="coverage" name="output1"/>
</outputs>
<help>
</help>
</tool>
@@ -33,18 +33,20 @@ def main():
# Do conversion.
index = Indexes()
offset = 0
for line in open(input_fname, "r"):
feature = line.strip().split()
if not feature or feature[0].startswith("track") or feature[0].startswith("#"):
with open(input_fname) as in_fh:
for line in in_fh:
feature = line.strip().split()
if not feature or feature[0].startswith("track") or feature[0].startswith("#"):
offset += len(line)
continue
chrom = feature[options.chrom_col]
chrom_start = int(feature[options.start_col])
chrom_end = int(feature[options.end_col])
index.add(chrom, chrom_start, chrom_end, offset)
offset += len(line)
continue
chrom = feature[options.chrom_col]
chrom_start = int(feature[options.start_col])
chrom_end = int(feature[options.end_col])
index.add(chrom, chrom_start, chrom_end, offset)
offset += len(line)
index.write(open(output_fname, "w"))
with open(output_fname, 'wb') as out:
index.write(out)
if __name__ == "__main__":
@@ -1,5 +1,8 @@
<tool id="CONVERTER_interval_to_interval_index_0" name="Convert Interval to Interval Index" version="1.0.0" hidden="true">
<tool id="CONVERTER_interval_to_interval_index_0" name="Convert Interval to Interval Index" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.8.8">bx-python</requirement>
</requirements>
<command>
python '$__tool_directory__/interval_to_interval_index_converter.py'
-c ${input1.metadata.chromCol}
@@ -13,6 +16,12 @@
<outputs>
<data format="interval_index" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval"/>
<output name="output1" format="interval_index" value="2.interval_index"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -42,7 +42,7 @@ def to_tabix(bgzip_fname, out_fname, preset=None, chrom_col=None, start_col=None
start_col=(start_col - 1), end_col=(end_col - 1),
keep_original=True, index=out_fname, force=True)
if os.path.getsize(out_fname) == 0:
sys.stderr.write("The converted tabix index file is empty, meaning the input data is invalid.")
sys.exit("The converted tabix index file is empty, meaning the input data is invalid.")
return bgzip_fname
@@ -1,4 +1,7 @@
<tool id="CONVERTER_interval_to_tabix_0" name="Convert Interval to tabix" version="1.0.0" hidden="true">
<tool id="CONVERTER_interval_to_tabix_0" name="Convert Interval to tabix" version="1.0.1" hidden="true" profile="16.04">
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
</requirements>
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command>
python '$__tool_directory__/interval_to_tabix_converter.py'
@@ -14,6 +17,12 @@
<outputs>
<data format="tabix" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="interval" value="2.interval"/>
<output name="output1" format="tabix" value="2.tabix"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,6 +1,9 @@
<tool id="CONVERTER_len_to_linecount" name="Convert Len file to Linecount" version="1.0.0">
<tool id="CONVERTER_len_to_linecount" name="Convert Len file to Linecount" version="1.0.1" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="5.0.1">gawk</requirement>
</requirements>
<command>
<![CDATA[
wc -l '$input' | awk '{print $1}' > '$output'
@@ -12,6 +15,12 @@
<outputs>
<data name="output" format="linecount"/>
</outputs>
<tests>
<test>
<param name="input" format="len" value="chr_m.len"/>
<output name="output" format="linecount" value="chr_m.linecount"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -42,13 +42,11 @@ def rgConv(inpedfilepath, outhtmlname, outfilepath):
try:
mf = open(inmap, 'r')
except Exception:
sys.stderr.write('%s cannot open inmap file %s - do you have permission?\n' % (prog, inmap))
sys.exit(1)
sys.exit('%s cannot open inmap file %s - do you have permission?\n' % (prog, inmap))
try:
rsl = [x.split()[1] for x in mf]
except Exception:
sys.stderr.write('## cannot parse %s' % inmap)
sys.exit(1)
sys.exit('## cannot parse %s' % inmap)
try:
os.makedirs(outfilepath)
except Exception:
@@ -85,8 +83,7 @@ def main():
"""
nparm = 3
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (prog, sys.argv, nparm))
sys.exit(1)
sys.exit('## %s called with %s - needs %d parameters \n' % (prog, sys.argv, nparm))
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
@@ -1,6 +1,9 @@
<tool id="lped2fpedconvert" name="Convert lped to fped" version="0.01">
<tool id="lped2fpedconvert" name="Convert lped to fped" version="0.02" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>
python '$__tool_directory__/lped_to_fped_converter.py' '$input1.extra_files_path/$input1.metadata.base_name' '$output1' '$output1.files_path'
</command>
@@ -87,8 +87,7 @@ def main():
"""
nparm = 4
if len(sys.argv) < nparm:
sys.stderr.write('## %s called with %s - needs %d parameters \n' % (prog, sys.argv, nparm))
sys.exit(1)
sys.exit('## %s called with %s - needs %d parameters \n' % (prog, sys.argv, nparm))
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
@@ -1,6 +1,9 @@
<tool id="lped2pbedconvert" name="Convert lped to plink pbed" version="0.01">
<tool id="lped2pbedconvert" name="Convert lped to plink pbed" version="0.02" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>
python '$__tool_directory__/lped_to_pbed_converter.py' '$input1.extra_files_path/$input1.metadata.base_name'
'$output1' '$output1.files_path' 'plink'
@@ -1,4 +1,4 @@
<tool id="CONVERTER_maf_to_fasta_0" name="Convert MAF to Fasta" version="1.0.1">
<tool id="CONVERTER_maf_to_fasta_0" name="Convert MAF to Fasta" version="1.0.2" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command>python '$__tool_directory__/maf_to_fasta_converter.py' '$output1' '$input1'</command>
<inputs>
@@ -7,6 +7,12 @@
<outputs>
<data format="fasta" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="maf" value="interval2maf_3from28way.maf"/>
<output name="output1" format="fasta" value="interval2fasta_3from28way.fasta"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,4 +1,4 @@
<tool id="CONVERTER_maf_to_interval_0" name="Convert MAF to Genomic Intervals" version="1.0.2">
<tool id="CONVERTER_maf_to_interval_0" name="Convert MAF to Genomic Intervals" version="1.0.3" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<command>python '$__tool_directory__/maf_to_interval_converter.py' '$output1' '$input1' '${input1.metadata.dbkey}'</command>
<inputs>
@@ -7,6 +7,12 @@
<outputs>
<data format="interval" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="maf" value="interval2maf_3from28way.maf"/>
<output name="output1" format="interval" value="interval2interval_3from28way.interval"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,4 +1,7 @@
<tool id="CONVERTER_neostorezip_to_neostore" name="Convert neostore.zip files to neostore" version="1.0.0" hidden="true">
<tool id="CONVERTER_neostorezip_to_neostore" name="Convert neostore.zip files to neostore" version="1.0.0" hidden="true" profile="16.04">
<requirements>
<requirement type="package" version="6.0">unzip</requirement>
</requirements>
<command><![CDATA[
unzip '${input1.extra_files_path}/neostore_file.zip' -d '${output1.files_path}' > $output1
]]>
@@ -1,5 +1,8 @@
<tool id="pbed2ldindepconvert" name="Convert plink pbed to ld reduced format" version="0.01">
<tool id="pbed2ldindepconvert" name="Convert plink pbed to ld reduced format" version="0.02" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>
python '$__tool_directory__/pbed_ldreduced_converter.py' '$input1.extra_files_path/$input1.metadata.base_name' '60' '55' '0.1' '$output1' '$output1.files_path' 'plink'
</command>
@@ -53,8 +53,7 @@ def main():
"""
nparm = 4
if len(sys.argv) < nparm:
sys.stderr.write('PBED to LPED converter called with %s - needs %d parameters \n' % (sys.argv, nparm))
sys.exit(1)
sys.exit('PBED to LPED converter called with %s - needs %d parameters \n' % (sys.argv, nparm))
inpedfilepath = sys.argv[1]
outhtmlname = sys.argv[2]
outfilepath = sys.argv[3]
@@ -1,6 +1,9 @@
<tool id="pbed2lpedconvert" name="Convert plink pbed to linkage lped" version="0.01">
<tool id="pbed2lpedconvert" name="Convert plink pbed to linkage lped" version="0.02" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>
python '$__tool_directory__/pbed_to_lped_converter.py' '$input1.extra_files_path/$input1.metadata.base_name'
'$output1' '$output1.files_path' plink
@@ -1,5 +1,8 @@
<tool id="CONVERTER_picard_interval_list_to_bed6" name="Convert Picard Interval List to BED6" version="1.0.0">
<tool id="CONVERTER_picard_interval_list_to_bed6" name="Convert Picard Interval List to BED6" version="1.0.1" profile="16.04">
<description>converter</description>
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/picard_interval_list_to_bed6_converter.py' '$input' '$output'</command>
<inputs>
<param name="input" type="data" format="picard_interval_list" label="Picard Interval List file"/>
@@ -22,14 +22,16 @@ def main():
# Do conversion.
index = Indexes()
offset = 0
for line in open(input_fname, "r"):
chrom, start = line.split()[0:2]
# Pileup format is 1-based.
start = int(start) - 1
index.add(chrom, start, start + 1, offset)
offset += len(line)
with open(input_fname) as in_fh:
for line in in_fh:
chrom, start = line.split()[0:2]
# Pileup format is 1-based.
start = int(start) - 1
index.add(chrom, start, start + 1, offset)
offset += len(line)
index.write(open(output_fname, "w"))
with open(output_fname, 'wb') as out:
index.write(out)
if __name__ == "__main__":
@@ -1,5 +1,8 @@
<tool id="CONVERTER_pileup_to_interval_index_0" name="Convert Pileup to Interval Index" version="1.0.0" hidden="true">
<tool id="CONVERTER_pileup_to_interval_index_0" name="Convert Pileup to Interval Index" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.8.8">bx-python</requirement>
</requirements>
<command>
python '$__tool_directory__/pileup_to_interval_index_converter.py'
'$input' '$output'
@@ -10,6 +13,12 @@
<outputs>
<data format="interval_index" name="output"/>
</outputs>
<tests>
<test>
<param name="input" format="pileup" value="1.pileup"/>
<output name="output" format="interval_index" value="1.interval_index"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_ref_to_seq_taxomony" name="Convert Ref taxonomy to Seq Taxonomy" version="1.0.0">
<tool id="CONVERTER_ref_to_seq_taxomony" name="Convert Ref taxonomy to Seq Taxonomy" version="1.0.1" profile="16.04">
<description>converts 2 or 3 column sequence taxonomy file to a 2 column mothur taxonomy_outline format</description>
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command> python '$__tool_directory__/ref_to_seq_taxonomy_converter.py' '$input' '$output'</command>
<inputs>
<param name="input" type="data" format="mothur.ref.taxonomy" label="a Sequence Taxomony file"/>
@@ -9,4 +12,4 @@
</outputs>
<help>
</help>
</tool>
</tool>
@@ -1,117 +0,0 @@
#!/usr/bin/env python
# Dan Blankenberg
"""
A wrapper script for converting SAM to BAM, with sorting.
%prog input_filename.sam output_filename.bam
"""
import optparse
import os
import shutil
import subprocess
import sys
import tempfile
import packaging.version
CHUNK_SIZE = 2 ** 20 # 1mb
def cleanup_before_exit(tmp_dir):
if tmp_dir and os.path.exists(tmp_dir):
shutil.rmtree(tmp_dir)
def cmd_exists(cmd):
# http://stackoverflow.com/questions/5226958/which-equivalent-function-in-python
for path in os.environ["PATH"].split(":"):
if os.path.exists(os.path.join(path, cmd)):
return True
return False
def _get_samtools_version():
version = '0.0.0'
if not cmd_exists('samtools'):
raise Exception('This tool needs samtools, but it is not on PATH.')
# Get the version of samtools via --version-only, if available
p = subprocess.Popen(['samtools', '--version-only'], stdout=subprocess.PIPE, stderr=subprocess.PIPE)
output, error = p.communicate()
# --version-only is available
# Format is <version x.y.z>+htslib-<a.b.c>
if p.returncode == 0:
version = output.split('+')[0]
return version
output = subprocess.Popen(['samtools'], stderr=subprocess.PIPE, stdout=subprocess.PIPE).communicate()[1]
lines = output.split('\n')
for line in lines:
if line.lower().startswith('version'):
# Assuming line looks something like: version: 0.1.12a (r862)
version = line.split()[1]
break
return version
def __main__():
# Parse Command Line
parser = optparse.OptionParser()
(options, args) = parser.parse_args()
assert len(args) == 2, 'You must specify the input and output filenames'
input_filename, output_filename = args
tmp_dir = tempfile.mkdtemp(prefix='tmp-sam_to_bam_converter-')
# convert to SAM
unsorted_bam_filename = os.path.join(tmp_dir, 'unsorted.bam')
unsorted_stderr_filename = os.path.join(tmp_dir, 'unsorted.stderr')
proc = subprocess.Popen(['samtools', 'view', '-bS', input_filename],
stdout=open(unsorted_bam_filename, 'wb'),
stderr=open(unsorted_stderr_filename, 'wb'),
cwd=tmp_dir)
return_code = proc.wait()
if return_code:
stderr_target = sys.stderr
else:
stderr_target = sys.stdout
with open(unsorted_stderr_filename) as stderr:
while True:
chunk = stderr.read(CHUNK_SIZE)
if chunk:
stderr_target.write(chunk)
else:
break
# sort sam, so indexing will not fail
sorted_stderr_filename = os.path.join(tmp_dir, 'sorted.stderr')
sorting_prefix = os.path.join(tmp_dir, 'sorted_bam')
# samtools changed sort command arguments (starting from version 1.3)
samtools_version = packaging.version.parse(_get_samtools_version())
if samtools_version < packaging.version.parse('1.0'):
sort_args = ['-o', unsorted_bam_filename, sorting_prefix]
else:
sort_args = ['-T', sorting_prefix, unsorted_bam_filename]
proc = subprocess.Popen(['samtools', 'sort'] + sort_args,
stdout=open(output_filename, 'wb'),
stderr=open(sorted_stderr_filename, 'wb'),
cwd=tmp_dir)
return_code = proc.wait()
if return_code:
stderr_target = sys.stderr
else:
stderr_target = sys.stdout
with open(sorted_stderr_filename) as stderr:
while True:
chunk = stderr.read(CHUNK_SIZE)
if chunk:
stderr_target.write(chunk)
else:
break
cleanup_before_exit(tmp_dir)
if __name__ == "__main__":
__main__()
@@ -1,20 +0,0 @@
<tool id="CONVERTER_sam_to_bam" name="Convert SAM to BAM" version="2.0.0">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<!-- Used on the metadata edit page. -->
<!-- FIXME: conversion will only work if headers for reference sequences are in input file.
To fix this: (a) merge sam_to_bam tool in tools with this conversion (like fasta_to_len
conversion); and (b) define a datatype-specific way to set converter parameters.
-->
<requirements>
<requirement type="package">samtools</requirement>
</requirements>
<command>python '$__tool_directory__/sam_to_bam.py' '$input1' '$output'</command>
<inputs>
<param name="input1" type="data" format="sam" label="SAM file"/>
</inputs>
<outputs>
<data name="output" format="bam"/>
</outputs>
<help>
</help>
</tool>
@@ -20,6 +20,12 @@ samtools view -bh '$input' | bedtools genomecov -bg -split -ibam stdin
<outputs>
<data name="output" format="bigwig"/>
</outputs>
<tests>
<test>
<param name="input" format="sam" value="bfast_out1.sam" dbkey="hg17"/>
<output name="output" format="bigwig" value="bfast_out1.bigwig"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -17,6 +17,12 @@
<outputs>
<data name="output" format="unsorted.bam"/>
</outputs>
<tests>
<test>
<param name="input" format="sam" value="bfast_out1.sam"/>
<output name="output" format="unsorted.bam" value="bfast_out1.unsorted.bam"/>
</test>
</tests>
<help>
</help>
</tool>
+8 -10
View File
@@ -25,19 +25,17 @@ def main():
def convert_to_tsv(input_fname, output_fname):
with open(input_fname, 'rb') as csvfile:
with open(output_fname, 'wb') as ofh:
reader = csv.reader(csvfile)
for line in reader:
ofh.write('\t'.join(line) + '\n')
with open(input_fname, newline="") as csvfile, open(output_fname, 'w') as ofh:
reader = csv.reader(csvfile)
for line in reader:
ofh.write('\t'.join(line) + '\n')
def convert_to_csv(input_fname, output_fname):
with open(input_fname, 'rb') as tabfile:
with open(output_fname, 'wb') as ofh:
writer = csv.writer(ofh, delimiter=',')
for line in tabfile.readlines():
writer.writerow(line.strip().split('\t'))
with open(input_fname) as tabfile, open(output_fname, 'w', newline='') as ofh:
writer = csv.writer(ofh, delimiter=',')
for line in tabfile.readlines():
writer.writerow(line.strip().split('\t'))
if __name__ == "__main__":
@@ -1,5 +1,8 @@
<tool id="tabular_to_csv" name="Convert tabular to CSV" version="1.0.0">
<tool id="tabular_to_csv" name="Convert tabular to CSV" version="1.0.0" profile="16.04">
<description></description>
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/tabular_csv.py' --from-tabular -i '$tabular' -o '$csv'</command>
<inputs>
<param format="tsv,tabular" name="tabular" type="data" label="Choose file with tab-separated values"/>
@@ -7,6 +10,12 @@
<outputs>
<data format="csv" name="csv"/>
</outputs>
<tests>
<test>
<param name="tabular" format="tabular" value="gtf_filter_by_attribute_values_list_in3.tabular"/>
<output name="csv" format="csv" value="gtf_filter_by_attribute_values_list_in3.csv"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,8 @@
<tool id="tabular_to_dbnsfp" name="Convert tabular to dbnsfp" version="1.0.0">
<tool id="tabular_to_dbnsfp" name="Convert tabular to dbnsfp" version="1.0.1" profile="16.04">
<description></description>
<requirements>
<requirement type="package" version="3.7">python</requirement>
</requirements>
<command>python '$__tool_directory__/tabular_to_dbnsfp.py' '$input' '$dbnsfp.extra_files_path/dbNSFP.gz'</command>
<inputs>
<param format="tabular" name="input" type="data" label="Choose a dbnsfp tabular file"/>
@@ -7,6 +10,12 @@
<outputs>
<data format="snpsiftdbnsfp" name="dbnsfp"/>
</outputs>
<tests>
<test>
<param name="input" format="tabular" value="gtf_filter_by_attribute_values_list_in3.tabular"/>
<output name="dbnsfp" format="snpsiftdbnsfp" value="gtf_filter_by_attribute_values_list_in3.snpsiftdbnsfp"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -11,6 +11,12 @@
<outputs>
<data format="directory" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="tar" value="testdir1.tar"/>
<output name="output1" format="directory" value="testdir1.tar.directory"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -17,6 +17,12 @@
<outputs>
<data format="bam" name="output"/>
</outputs>
<tests>
<test>
<param name="input" format="sam" value="bfast_out1.sam"/>
<output name="output" format="bam" value="bfast_out1.bam"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -18,6 +18,12 @@
<outputs>
<data format="qname_sorted.bam" name="output"/>
</outputs>
<tests>
<test>
<param name="input" format="sam" value="bfast_out1.sam"/>
<output name="output" format="qname_sorted.bam" value="bfast_out1.qname_sorted.bam"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -0,0 +1,7 @@
<tables>
<!-- Locations of dbkeys and len files under genome directory -->
<table name="__dbkeys__" comment_char="#" allow_duplicate_entries="False">
<columns>value, name, len_path</columns>
<file path="${__HERE__}/dbkeys.loc.test" />
</table>
</tables>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_vcf_bgzip_to_tabix_0" name="Convert BGZ VCF to tabix" version="1.0.0" hidden="true">
<tool id="CONVERTER_vcf_bgzip_to_tabix_0" name="Convert BGZ VCF to tabix" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_tabix_converter.py' -P 'vcf' '' '$input1' '$output1'</command>
<inputs>
<param format="vcf_bgzip" name="input1" type="data" label="Choose BGZIP'd VCF file"/>
@@ -7,6 +10,12 @@
<outputs>
<data format="tabix" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="vcf_bgzip" value="vcf_bgzip_to_maf_in.vcf_bgzip"/>
<output name="output1" format="tabix" value="tabix_to_maf_in_to_tabix.tabix"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,9 @@
<tool id="CONVERTER_vcf_to_bgzip_0" name="Convert VCF to BGZIP" version="1.0.0" hidden="true">
<tool id="CONVERTER_vcf_to_bgzip_0" name="Convert VCF to BGZIP" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command>python '$__tool_directory__/bgzip.py' -P vcf '$input1' '$output1'</command>
<inputs>
<param format="vcf" name="input1" type="data" label="Choose Vcf file"/>
@@ -7,6 +11,12 @@
<outputs>
<data format="bgzip" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="vcf" value="vcf_to_maf_in.vcf"/>
<output name="output1" format="bgzip" value="bgzip_to_maf_in.bgzip"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -18,15 +18,17 @@ def main():
# Do conversion.
index = Indexes()
reader = galaxy_utils.sequence.vcf.Reader(open(in_file))
offset = reader.metadata_len
for vcf_line in reader:
# VCF format provides a chrom and 1-based position for each variant.
# IntervalIndex expects 0-based coordinates.
index.add(vcf_line.chrom, vcf_line.pos - 1, vcf_line.pos, offset)
offset += len(vcf_line.raw_line)
with open(in_file) as in_fh:
reader = galaxy_utils.sequence.vcf.Reader(in_fh)
offset = reader.metadata_len
for vcf_line in reader:
# VCF format provides a chrom and 1-based position for each variant.
# IntervalIndex expects 0-based coordinates.
index.add(vcf_line.chrom, vcf_line.pos - 1, vcf_line.pos, offset)
offset += len(vcf_line.raw_line)
index.write(open(out_file, "w"))
with open(out_file, "wb") as out_fh:
index.write(out_fh)
if __name__ == "__main__":
@@ -1,5 +1,9 @@
<tool id="CONVERTER_vcf_to_interval_index_0" name="Convert VCF to Interval Index" version="1.0.0" hidden="true">
<tool id="CONVERTER_vcf_to_interval_index_0" name="Convert VCF to Interval Index" version="1.0.1" hidden="true" profile="16.04">
<description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description>
<requirements>
<requirement type="package" version="0.8.8">bx-python</requirement>
<requirement type="package" version="1.1.4">galaxy_sequence_utils</requirement>
</requirements>
<command>python '$__tool_directory__/vcf_to_interval_index_converter.py' '$input1' '$output1'</command>
<inputs>
<param format="vcf" name="input1" type="data" label="Choose VCF file"/>
@@ -7,6 +11,12 @@
<outputs>
<data format="interval_index" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="vcf" value="vcf_to_maf_in.vcf"/>
<output name="output1" format="interval_index" value="interval_index_to_maf_in.interval_index"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,8 @@
<tool id="CONVERTER_vcf_to_tabix_0" name="Convert Vcf to tabix" version="1.0.0" hidden="true">
<tool id="CONVERTER_vcf_to_tabix_0" name="Convert Vcf to tabix" version="1.0.1" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
</requirements>
<command>python '$__tool_directory__/interval_to_tabix_converter.py' -P vcf '$input1' '$bgzip' '$output1'</command>
<inputs>
<param format="vcf" name="input1" type="data" label="Choose Vcf file"/>
@@ -8,6 +11,12 @@
<outputs>
<data format="tabix" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="vcf" value="vcf_to_maf_in.vcf"/>
<output name="output1" format="tabix" value="tabix_to_maf_in.tabix"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,5 +1,9 @@
<tool id="CONVERTER_vcf_to_vcf_bgzip_0" name="Convert VCF to VCF_BGZIP" version="1.0.1" hidden="true">
<tool id="CONVERTER_vcf_to_vcf_bgzip_0" name="Convert VCF to VCF_BGZIP" version="1.0.2" hidden="true" profile="16.04">
<!-- <description>__NOT_USED_CURRENTLY_FOR_CONVERTERS__</description> -->
<requirements>
<requirement type="package" version="0.15.4">pysam</requirement>
<requirement type="package" version="8.25">coreutils</requirement>
</requirements>
<command>python '$__tool_directory__/vcf_to_vcf_bgzip.py' '$input1' '$output1'</command>
<inputs>
<param format="vcf" name="input1" type="data" label="Choose Vcf file"/>
@@ -7,6 +11,12 @@
<outputs>
<data format="vcf_bgzip" name="output1"/>
</outputs>
<tests>
<test>
<param name="input1" format="vcf" value="vcf_to_maf_in.vcf"/>
<output name="output1" format="vcf_bgzip" value="vcf_bgzip_to_maf_in.vcf_bgzip"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -15,6 +15,12 @@
<outputs>
<data format="bigwig" name="output"/>
</outputs>
<tests>
<test>
<param name="input" format="wig" value="aggregate_binned_scores_3.wig" dbkey="hg17"/>
<output name="output" format="bigwig" value="aggregate_binned_scores_3.bigwig"/>
</test>
</tests>
<help>
</help>
</tool>
@@ -1,30 +0,0 @@
#!/usr/bin/env python
from __future__ import division
import sys
from bx.arrays.array_tree import array_tree_dict_from_reader, FileArrayTreeDict
from bx.arrays.wiggle import WiggleReader
BLOCK_SIZE = 100
def main():
input_fname = sys.argv[1]
out_fname = sys.argv[2]
reader = WiggleReader(open(input_fname))
# Fill array from reader
d = array_tree_dict_from_reader(reader, {}, block_size=BLOCK_SIZE)
for array_tree in d.values():
array_tree.root.build_summary()
FileArrayTreeDict.dict_to_file(d, open(out_fname, "w"))
if __name__ == "__main__":
main()
@@ -1,12 +0,0 @@
<tool id="CONVERTER_Wiggle_0" name="Index Wiggle for Track Viewer" version="1.0.0" hidden="true">
<!-- Used internally to generate track indexes -->
<command>python '$__tool_directory__/wiggle_to_array_tree_converter.py' '$input' '$output'</command>
<inputs>
<param format="wiggle" name="input" type="data" label="Choose wiggle"/>
</inputs>
<outputs>
<data format="array_tree" name="output"/>
</outputs>
<help>
</help>
</tool>
@@ -8,4 +8,10 @@
<outputs>
<data format="interval" name="out_file1" />
</outputs>
<tests>
<test>
<param name="input" format="wig" value="aggregate_binned_scores_3.wig"/>
<output name="out_file1" format="interval" value="aggregate_binned_scores_3.interval"/>
</test>
</tests>
</tool>
+2
View File
@@ -45,6 +45,7 @@ class Registry(object):
self.datatype_converters = OrderedDict()
# Converters defined in local datatypes_conf.xml
self.converters = []
self.converter_tools = set()
# Converters defined in datatypes_conf.xml included in installed tool shed repositories.
self.proprietary_converters = []
self.converter_deps = {}
@@ -612,6 +613,7 @@ class Registry(object):
try:
config_path = os.path.join(converter_path, tool_config)
converter = toolbox.load_tool(config_path, use_cached=use_cached)
self.converter_tools.add(converter)
if installed_repository_dict:
# If the converter is included in an installed tool shed repository, set the tool
# shed related tool attributes.
@@ -28,7 +28,7 @@ boltons==19.3.0
boto3==1.9.114
boto==2.49.0
botocore==1.12.253
bx-python==0.8.6
bx-python==0.8.8
bz2file==0.98 ; python_version < '3.3'
cachecontrol==0.11.7
cachetools==3.1.1
+4
View File
@@ -537,6 +537,10 @@ class Tool(Dictifiable):
tool_versions = self.tool_versions
return not tool_versions or self.version == self.tool_versions[-1]
@property
def is_datatype_converter(self):
return self in self.app.datatypes_registry.converter_tools
@property
def tool_shed_repository(self):
# If this tool is included in an installed tool shed repository, return it.
+10 -9
View File
@@ -172,15 +172,16 @@ class ToolsController(BaseAPIController, UsesVisualizationMixin):
"""
test_counts_by_tool = {}
for id, tool in self.app.toolbox.tools():
tests = tool.tests
if tests:
if tool.id not in test_counts_by_tool:
test_counts_by_tool[tool.id] = {}
available_versions = test_counts_by_tool[tool.id]
available_versions[tool.version] = {
"tool_name": tool.name,
"count": len(tests),
}
if not tool.is_datatype_converter:
tests = tool.tests
if tests:
if tool.id not in test_counts_by_tool:
test_counts_by_tool[tool.id] = {}
available_versions = test_counts_by_tool[tool.id]
available_versions[tool.version] = {
"tool_name": tool.name,
"count": len(tests),
}
return test_counts_by_tool
@expose_api_raw_anonymous_and_sessionless
+46 -47
View File
@@ -12,60 +12,59 @@ def __main__():
output_name = sys.argv[2]
skipped_lines = 0
first_skipped_line = 0
out = open(output_name, 'w')
i = 0
for i, line in enumerate(open(input_name)):
complete_bed = False
line = line.rstrip('\r\n')
if line and not line.startswith('#') and not line.startswith('track') and not line.startswith('browser'):
try:
elems = line.split('\t')
if len(elems) == 12:
complete_bed = True
chrom = elems[0]
if complete_bed:
feature = "mRNA"
else:
with open(output_name, 'w') as out, open(input_name) as fh_in:
for i, line in enumerate(fh_in):
complete_bed = False
line = line.rstrip('\r\n')
if line and not line.startswith('#') and not line.startswith('track') and not line.startswith('browser'):
try:
elems = line.split('\t')
if len(elems) == 12:
complete_bed = True
chrom = elems[0]
if complete_bed:
feature = "mRNA"
else:
try:
feature = elems[3]
except Exception:
feature = 'feature%d' % (i + 1)
start = int(elems[1]) + 1
end = int(elems[2])
try:
feature = elems[3]
score = elems[4]
except Exception:
feature = 'feature%d' % (i + 1)
start = int(elems[1]) + 1
end = int(elems[2])
try:
score = elems[4]
score = '0'
try:
strand = elems[5]
except Exception:
strand = '+'
try:
group = elems[3]
except Exception:
group = 'group%d' % (i + 1)
if complete_bed:
out.write('%s\tbed2gff\t%s\t%d\t%d\t%s\t%s\t.\t%s %s;\n' % (chrom, feature, start, end, score, strand, feature, group))
else:
out.write('%s\tbed2gff\t%s\t%d\t%d\t%s\t%s\t.\t%s;\n' % (chrom, feature, start, end, score, strand, group))
if complete_bed:
# We have all the info necessary to annotate exons for genes and mRNAs
block_count = int(elems[9])
block_sizes = elems[10].split(',')
block_starts = elems[11].split(',')
for j in range(block_count):
exon_start = int(start) + int(block_starts[j])
exon_end = exon_start + int(block_sizes[j]) - 1
out.write('%s\tbed2gff\texon\t%d\t%d\t%s\t%s\t.\texon %s;\n' % (chrom, exon_start, exon_end, score, strand, group))
except Exception:
score = '0'
try:
strand = elems[5]
except Exception:
strand = '+'
try:
group = elems[3]
except Exception:
group = 'group%d' % (i + 1)
if complete_bed:
out.write('%s\tbed2gff\t%s\t%d\t%d\t%s\t%s\t.\t%s %s;\n' % (chrom, feature, start, end, score, strand, feature, group))
else:
out.write('%s\tbed2gff\t%s\t%d\t%d\t%s\t%s\t.\t%s;\n' % (chrom, feature, start, end, score, strand, group))
if complete_bed:
# We have all the info necessary to annotate exons for genes and mRNAs
block_count = int(elems[9])
block_sizes = elems[10].split(',')
block_starts = elems[11].split(',')
for j in range(block_count):
exon_start = int(start) + int(block_starts[j])
exon_end = exon_start + int(block_sizes[j]) - 1
out.write('%s\tbed2gff\texon\t%d\t%d\t%s\t%s\t.\texon %s;\n' % (chrom, exon_start, exon_end, score, strand, group))
except Exception:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
out.close()
info_msg = "%i lines converted to GFF version 2. " % (i + 1 - skipped_lines)
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % (skipped_lines, first_skipped_line)
+51 -52
View File
@@ -55,29 +55,50 @@ def __main__():
output_name = sys.argv[2]
skipped_lines = 0
first_skipped_line = 0
out = open(output_name, 'w')
i = 0
cur_transcript_chrome = None
cur_transcript_id = None
cur_transcript_strand = None
cur_transcripts_blocks = [] # (start, end) for each block.
for i, line in enumerate(open(input_name)):
line = line.rstrip('\r\n')
if line and not line.startswith('#'):
try:
# GFF format: chrom source, name, chromStart, chromEnd, score, strand, attributes
elems = line.split('\t')
start = str(int(elems[3]) - 1)
coords = [int(start), int(elems[4])]
strand = elems[6]
if strand not in ['+', '-']:
strand = '+'
attributes = parse_gff_attributes(elems[8])
t_id = attributes.get("transcript_id", None)
with open(output_name, 'w') as out, open(input_name) as in_fh:
for i, line in enumerate(in_fh):
line = line.rstrip('\r\n')
if line and not line.startswith('#'):
try:
# GFF format: chrom source, name, chromStart, chromEnd, score, strand, attributes
elems = line.split('\t')
start = str(int(elems[3]) - 1)
coords = [int(start), int(elems[4])]
strand = elems[6]
if strand not in ['+', '-']:
strand = '+'
attributes = parse_gff_attributes(elems[8])
t_id = attributes.get("transcript_id", None)
if not t_id:
#
# No transcript ID, so write last transcript and write current line as its own line.
#
# Write previous transcript.
if cur_transcript_id:
# Write BED entry.
out.write(get_bed_line(cur_transcript_chrome, cur_transcript_id, cur_transcript_strand, cur_transcripts_blocks))
# Replace any spaces in the name with underscores so UCSC will not complain.
name = elems[2].replace(" ", "_")
out.write(get_bed_line(elems[0], name, strand, [coords]))
continue
# There is a transcript ID, so process line at transcript level.
if t_id == cur_transcript_id:
# Line is element of transcript and will be a block in the BED entry.
cur_transcripts_blocks.append(coords)
continue
if not t_id:
#
# No transcript ID, so write last transcript and write current line as its own line.
# Line is part of new transcript; write previous transcript and start
# new transcript.
#
# Write previous transcript.
@@ -85,47 +106,25 @@ def __main__():
# Write BED entry.
out.write(get_bed_line(cur_transcript_chrome, cur_transcript_id, cur_transcript_strand, cur_transcripts_blocks))
# Replace any spaces in the name with underscores so UCSC will not complain.
name = elems[2].replace(" ", "_")
out.write(get_bed_line(elems[0], name, strand, [coords]))
continue
# There is a transcript ID, so process line at transcript level.
if t_id == cur_transcript_id:
# Line is element of transcript and will be a block in the BED entry.
# Start new transcript.
cur_transcript_chrome = elems[0]
cur_transcript_id = t_id
cur_transcript_strand = strand
cur_transcripts_blocks = []
cur_transcripts_blocks.append(coords)
continue
#
# Line is part of new transcript; write previous transcript and start
# new transcript.
#
# Write previous transcript.
if cur_transcript_id:
# Write BED entry.
out.write(get_bed_line(cur_transcript_chrome, cur_transcript_id, cur_transcript_strand, cur_transcripts_blocks))
# Start new transcript.
cur_transcript_chrome = elems[0]
cur_transcript_id = t_id
cur_transcript_strand = strand
cur_transcripts_blocks = []
cur_transcripts_blocks.append(coords)
except Exception:
except Exception:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
else:
skipped_lines += 1
if not first_skipped_line:
first_skipped_line = i + 1
# Write last transcript.
if cur_transcript_id:
# Write BED entry.
out.write(get_bed_line(cur_transcript_chrome, cur_transcript_id, cur_transcript_strand, cur_transcripts_blocks))
out.close()
# Write last transcript.
if cur_transcript_id:
# Write BED entry.
out.write(get_bed_line(cur_transcript_chrome, cur_transcript_id, cur_transcript_strand, cur_transcripts_blocks))
info_msg = "%i lines converted to BED. " % (i + 1 - skipped_lines)
if skipped_lines > 0:
info_msg += "Skipped %d blank/comment/invalid lines starting with line #%d." % (skipped_lines, first_skipped_line)