mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
MAF tools are no longer duplicated for different sources, i.e. locally cached and user histories.
Tests will not pass until testing framework is updated to be compatible with new tool options. Several files can be removed, but have been left for the time being: tools/filters/maf/maf_to_fasta_multiple_sets.xml tools/filters/maf/maf_to_fasta_concat.xml tools/extract/user_interval2maf.xml tools/extract/interval_maf_to_merged_fasta_user.xml tools/extract/interval_maf_to_merged_fasta_user_code.py tools/extract/genebed_maf_to_fasta_user.xml tools/extract/genebed_maf_to_fasta_user_code.py
This commit is contained in:
+1
-4
@@ -52,8 +52,7 @@
|
||||
<tool file="stats/grouping.xml" />
|
||||
</section>
|
||||
<section name="Convert Formats" id="convert">
|
||||
<tool file="filters/maf/maf_to_fasta_multiple_sets.xml" />
|
||||
<tool file="filters/maf/maf_to_fasta_concat.xml" />
|
||||
<tool file="filters/maf/maf_to_fasta.xml" />
|
||||
<tool file="filters/maf/maf_to_bed.xml" />
|
||||
<tool file="filters/gff2bed.xml" />
|
||||
<tool file="filters/bed2gff.xml" />
|
||||
@@ -76,9 +75,7 @@
|
||||
<section name="Fetch Alignments" id="fetchAlign">
|
||||
<tool file="extract/interval2maf_pairwise.xml" />
|
||||
<tool file="extract/interval2maf.xml" />
|
||||
<tool file="extract/user_interval2maf.xml" />
|
||||
<tool file="extract/interval_maf_to_merged_fasta.xml" />
|
||||
<tool file="extract/interval_maf_to_merged_fasta_user.xml" />
|
||||
<!-- <tool file="extract/genebed_maf_to_fasta.xml"/>
|
||||
<tool file="extract/genebed_maf_to_fasta_user.xml"/>
|
||||
<tool file="filters/maf/maf_stats.xml"/> -->
|
||||
|
||||
@@ -52,8 +52,7 @@
|
||||
<tool file="stats/grouping.xml" />
|
||||
</section>
|
||||
<section name="Convert Formats" id="convert">
|
||||
<tool file="filters/maf/maf_to_fasta_multiple_sets.xml" />
|
||||
<tool file="filters/maf/maf_to_fasta_concat.xml" />
|
||||
<tool file="filters/maf/maf_to_fasta.xml" />
|
||||
<tool file="filters/maf/maf_to_bed.xml" />
|
||||
<tool file="filters/gff2bed.xml" />
|
||||
<tool file="filters/bed2gff.xml" />
|
||||
@@ -76,11 +75,8 @@
|
||||
<section name="Fetch Alignments" id="fetchAlign">
|
||||
<tool file="extract/interval2maf_pairwise.xml" />
|
||||
<tool file="extract/interval2maf.xml" />
|
||||
<tool file="extract/user_interval2maf.xml" />
|
||||
<tool file="extract/interval_maf_to_merged_fasta.xml" />
|
||||
<tool file="extract/interval_maf_to_merged_fasta_user.xml" />
|
||||
<tool file="extract/genebed_maf_to_fasta.xml"/>
|
||||
<tool file="extract/genebed_maf_to_fasta_user.xml"/>
|
||||
<tool file="filters/maf/maf_stats.xml"/>
|
||||
<tool file="filters/maf/maf_thread_for_species.xml"/>
|
||||
<tool file="filters/maf/maf_limit_to_species.xml"/>
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
<tool id="GeneBed_Maf_Fasta1" name="Extract Gene blocks">
|
||||
<description>from locally cached alignments</description>
|
||||
<command interpreter="python2.4">genebed_maf_to_fasta.py $dbkey $species $mafSource $input1 $out_file1 -</command>
|
||||
<tool id="GeneBed_Maf_Fasta1" name="Stitch Gene blocks">
|
||||
<description>given a set of coding exon intervals</description>
|
||||
<command interpreter="python2.4">#if $maf_source_type.maf_source == "user":#genebed_maf_to_fasta_user.py $dbkey $species $maf_source_type.input2 $input1 $out_file1 -
|
||||
#else:#genebed_maf_to_fasta.py $dbkey $species $maf_source_type.maf_uid $input1 $out_file1 -
|
||||
#end if
|
||||
</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param format="interval" name="input1" type="data" label="Interval File"/>
|
||||
<param format="bed" name="input1" type="data" label="Gene BED File"/>
|
||||
<!-- <param name="unkown_gap_char" type="select">
|
||||
<label>Character for missing data</label>
|
||||
<option value="-">-</option>
|
||||
@@ -11,10 +14,21 @@
|
||||
</param> -->
|
||||
</page>
|
||||
<page>
|
||||
<param name="mafSource" label="Choose MAF source" type="select" dynamic_options="get_available_data( input1.dbkey )"/>
|
||||
<conditional name="maf_source_type">
|
||||
<param name="maf_source" type="select" label="MAF Source">
|
||||
<option value="cached" selected="true">Locally Cached Alignments</option>
|
||||
<option value="user">Alignments in Your History</option>
|
||||
</param>
|
||||
<when value="user">
|
||||
<param format="maf" name="input2" label="MAF File" type="data"/>
|
||||
</when>
|
||||
<when value="cached">
|
||||
<param name="maf_uid" label="MAF Type" type="select" dynamic_options="get_available_data( input1.dbkey )"/>
|
||||
</when>
|
||||
</conditional>
|
||||
</page>
|
||||
<page>
|
||||
<param name="species" label="Choose desired species" type="select" display="checkboxes" multiple="True" dynamic_options="get_available_species( mafSource )"/>
|
||||
<param name="species" label="Choose species" type="select" display="checkboxes" multiple="True" dynamic_options="get_available_species( maf_source_type )" help="select species to be included in the final alignment" />
|
||||
</page>
|
||||
</inputs>
|
||||
<outputs>
|
||||
@@ -23,13 +37,31 @@
|
||||
<!-- <tests>
|
||||
<test>
|
||||
<param name="input1" value="gene_bed.dat"/>
|
||||
<param name="mafSource" value="8_WAY_MULTIZ_hg17"/>
|
||||
<param name="maf_source" value="cached"/>
|
||||
<param name="maf_uid" value="8_WAY_MULTIZ_hg17"/>
|
||||
<param name="species" value="canFam1,hg17,mm5,panTro1,rn3"/>
|
||||
<output name="out_file1" file="gene_bed_maf_to_fasta_out.dat" />
|
||||
</test>
|
||||
<test>
|
||||
<param name="input1" value="gene_bed.dat"/>
|
||||
<param name="input2" value="gene_bed_maf_to_fasta_user_maf_in.dat"/>
|
||||
<param name="maf_source" value="user"/>
|
||||
<param name="species" value="canFam1,hg17,mm5,panTro1,rn3"/>
|
||||
<output name="out_file1" file="gene_bed_maf_to_fasta_out.dat" />
|
||||
</test>
|
||||
</tests> -->
|
||||
<help>
|
||||
This tool takes a Gene BED file and creates a FASTA file which contains alignments of user specified species which correspond to the coding exons of the supplied Gene using locally cached alignments.
|
||||
|
||||
**What it does**
|
||||
|
||||
The coding sequence of genes are usually composed of several coding exons. Each of these coding exons is an individual genomic region, which when concatenated with each other constitutes the coding sequence. A single genomic region can be covered by multiple alignment blocks. In many cases it is desirable to stitch these alignment blocks together. This tool accepts a list of gene-based intervals, in the Gene BED format. For every interval it performs the following:
|
||||
|
||||
* finds all MAF blocks that overlap the coding regions;
|
||||
* sorts MAF blocks by alignment score;
|
||||
* stitches blocks together and resolves overlaps based on alignment score;
|
||||
* outputs alignments in FASTA format.
|
||||
|
||||
|
||||
</help>
|
||||
<code file="genebed_maf_to_fasta_code.py"/>
|
||||
</tool>
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
#build list of available data
|
||||
import os, sys
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
import bx.align.maf
|
||||
maf_sets = {}
|
||||
|
||||
try:
|
||||
@@ -37,14 +39,52 @@ def get_available_data( build ):
|
||||
available_sets.append(('No data available for this build','None',True))
|
||||
return available_sets
|
||||
|
||||
def get_available_species( maf_uid ):
|
||||
available_sets = []
|
||||
for key in maf_sets[maf_uid]['builds']:
|
||||
available_sets.append((key,key,True))
|
||||
if len(available_sets) < 1:
|
||||
available_sets.append(('No data available for this configuration','None',True))
|
||||
return available_sets
|
||||
def get_available_species( maf_source_type ):
|
||||
if maf_source_type['maf_source'] == 'cached':
|
||||
maf_uid = maf_source_type['maf_uid']
|
||||
available_sets = []
|
||||
for key in maf_sets[maf_uid]['builds']:
|
||||
available_sets.append((key,key,True))
|
||||
if len(available_sets) < 1:
|
||||
available_sets.append(('No data available for this configuration','None',True))
|
||||
return available_sets
|
||||
else:
|
||||
try:
|
||||
rval = []
|
||||
species={}
|
||||
input_filename = maf_source_type['input2'].file_name
|
||||
try:
|
||||
file_in = open(input_filename, 'r')
|
||||
maf_reader = bx.align.maf.Reader( file_in )
|
||||
|
||||
for i, m in enumerate( maf_reader ):
|
||||
l = m.components
|
||||
for c in l:
|
||||
spec,chrom = bx.align.maf.src_split( c.src )
|
||||
if not spec or not chrom:
|
||||
spec = chrom = c.src
|
||||
if spec not in species:
|
||||
species[spec]={"bases":0,"nongaps":0}
|
||||
species[spec]["bases"] = species[spec]["bases"] + c.size + c.text.count("-")
|
||||
species[spec]["nongaps"] = species[spec]["nongaps"] + c.size
|
||||
|
||||
file_in.close()
|
||||
except Exception:
|
||||
return [("There is a problem with your MAF file",'None',True)]
|
||||
species_names = species.keys()
|
||||
species_names.sort()
|
||||
|
||||
for spec in species_names:
|
||||
#species_sequence[spec] = "".join(species_sequence[spec])
|
||||
|
||||
display = "%s: %i nongap, %i total bases" % (spec, species[spec]["nongaps"], species[spec]["bases"] )
|
||||
rval.append( ( display,spec,True) )
|
||||
|
||||
return rval
|
||||
except:
|
||||
return [("<B>You must wait for the MAF file to be created before you can use this tool.</B>",'None',True)]
|
||||
|
||||
def exec_before_job(app,inp_data, out_data, param_dict, tool):
|
||||
for name, data in out_data.items():
|
||||
data.name = data.name + " [" + maf_sets[param_dict['mafSource']]['description'] + "]"
|
||||
def exec_before_job(app, inp_data, out_data, param_dict, tool):
|
||||
if param_dict['maf_source_type']['maf_source'] == "cached":
|
||||
for name, data in out_data.items():
|
||||
data.name = data.name + " [" + maf_sets[str(param_dict['maf_source_type']['maf_uid'])]['description'] + "]"
|
||||
|
||||
@@ -1,38 +1,49 @@
|
||||
<tool id="Interval2Maf1" name="Extract MAF blocks">
|
||||
<description>from locally cached alignments</description>
|
||||
<command interpreter="python2.4">interval2maf.py --dbkey=$dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafType=$mafType --interval_file=$input1 --output_file=$out_file1</command>
|
||||
<description>given a set of genomic intervals</description>
|
||||
<command interpreter="python2.4">#if $maf_source_type.maf_source == "user":#user_interval2maf.py --dbkey=$input1_dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafFile=$maf_source_type.mafFile --interval_file=$input1 --output_file=$out_file1
|
||||
#else:#interval2maf.py --dbkey=$input1_dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafType=$maf_source_type.mafType --interval_file=$input1 --output_file=$out_file1
|
||||
#end if
|
||||
</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param format="interval" name="input1" type="data" label="Choose intervals"/>
|
||||
</page>
|
||||
<page>
|
||||
<param name="mafType" label="Choose alignments" type="select" dynamic_options="get_available_data( input1.dbkey )"/>
|
||||
<conditional name="maf_source_type">
|
||||
<param name="maf_source" type="select" label="MAF Source">
|
||||
<option value="cached" selected="true">Locally Cached Alignments</option>
|
||||
<option value="user">Alignments in Your History</option>
|
||||
</param>
|
||||
<when value="user">
|
||||
<param format="maf" name="mafFile" label="Choose alignments" type="data"/>
|
||||
</when>
|
||||
<when value="cached">
|
||||
<param name="mafType" label="Choose alignments" type="select" dynamic_options="get_available_data( input1.dbkey )"/>
|
||||
</when>
|
||||
</conditional>
|
||||
</page>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="maf" name="out_file1" />
|
||||
<data format="maf" name="out_file1" metadata_source="input1"/>
|
||||
</outputs>
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input1" value="1.bed"/>
|
||||
<param name="maf_source" value="cached"/>
|
||||
<param name="mafType" value="ENCODE_TBA_hg17"/>
|
||||
<output name="out_file1" file="fsa_interval2maf.dat" />
|
||||
</test>
|
||||
<test>
|
||||
<param name="input1" value="1.bed"/>
|
||||
<param name="maf_source" value="user"/>
|
||||
<param name="mafFile" value="fsa_interval2maf.dat"/>
|
||||
<output name="out_file1" file="fsa_interval2maf.dat" />
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
What is the differences between these two tools: **Extract MAF blocks from locally cached alignments** and **Extract MAF blocks from user supplied alignments**?
|
||||
|
||||
* **Extract MAF blocks from locally cached alignments** uses alignments stored at Galaxy installation at Penn State and is appropriate for most situations. It takes only one dataset as the input - a list of genomic intervals.
|
||||
* **Extract MAF blocks from user supplied alignments** allows the user to work with his/her own alignments instead of those cached at Galaxy site. This tool takes two datasets as inputs - (1) genomic intervals and (2) alignments.
|
||||
|
||||
-----
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool takes genomic coordinates, superimposes them on multiple alignments (in MAF format) stored on Galaxy site, and excises alignment blocks corresponding to each set of coordinates. Alignment blocks that extend past START and/or END positions of an interval are trimmed. Note that a single genomic interval may correspond to two or more alignment blocks.
|
||||
This tool takes genomic coordinates, superimposes them on multiple alignments (in MAF format) stored on the Galaxy site or from your history, and excises alignment blocks corresponding to each set of coordinates. Alignment blocks that extend past START and/or END positions of an interval are trimmed. Note that a single genomic interval may correspond to two or more alignment blocks.
|
||||
|
||||
-----
|
||||
|
||||
|
||||
@@ -38,6 +38,7 @@ def get_available_data( build ):
|
||||
return available_sets
|
||||
|
||||
|
||||
def exec_before_job(app,inp_data, out_data, param_dict, tool):
|
||||
for name, data in out_data.items():
|
||||
data.name = data.name + " [" + maf_sets[param_dict['mafType']]['description'] + "]"
|
||||
def exec_before_job(app, inp_data, out_data, param_dict, tool):
|
||||
if param_dict['maf_source_type']['maf_source'] == "cached":
|
||||
for name, data in out_data.items():
|
||||
data.name = data.name + " [" + maf_sets[str(param_dict['maf_source_type']['mafType'])]['description'] + "]"
|
||||
|
||||
@@ -13,47 +13,17 @@
|
||||
<data format="maf" name="out_file1" />
|
||||
</outputs>
|
||||
<help>
|
||||
**What it does**
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** If your query does not work, you need to ensure that you have specified a database build for your data. Alignments may also not be available for your build or intervals at this time.
|
||||
|
||||
-----
|
||||
|
||||
**Syntax**
|
||||
|
||||
This tool uses coordinate and strand information to fetch sequence in MAF format.
|
||||
|
||||
- **MAF format** multiple alignment format file. This format stores multiple alignments at the DNA level between entire genomes.
|
||||
|
||||
- The .maf format is line-oriented. Each multiple alignment ends with a blank line.
|
||||
- Each sequence in an alignment is on a single line.
|
||||
- Lines starting with # are considered to be comments.
|
||||
- Each multiple alignment is in a separate paragraph that begins with an "a" line and contains an "s" line for each sequence in the multiple alignment.
|
||||
- Some MAF files may contain two optional line types:
|
||||
|
||||
- An "i" line containing information about what is in the aligned species DNA before and after the immediately preceding "s" line;
|
||||
- An "e" line containing information about the size of the gap between the alignments that span the current block.
|
||||
This tool takes genomic coordinates, superimposes them on pairwise alignments (in MAF format) stored on the Galaxy site, and excises alignment blocks corresponding to each set of coordinates. Alignment blocks that extend past START and/or END positions of an interval are trimmed. Note that a single genomic interval may correspond to two or more alignment blocks.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
- Input file::
|
||||
Here a single interval is superimposed on three MAF blocks. Blocks 1 and 3 are trimmed because they extend beyond boundaries of the interval:
|
||||
|
||||
chr7 127475281 127475310 NM_000230 0 +
|
||||
chr7 127486011 127486166 D49487 0 +
|
||||
|
||||
- Extract 8-way multiZ alignments in MAF format from above file::
|
||||
|
||||
##maf version=1
|
||||
a score=301244.0
|
||||
s hg17.chr7 127475281 29 + 158628139 GTAGGAATCGCAGCGCCAGCGGTTGCAAG
|
||||
s panTro1.chr6 129889135 29 + 161576975 GTAGGAATCGCAGCGCCAGCGGTTGCAAG
|
||||
|
||||
a score=407957.0
|
||||
s hg17.chr7 127486011 155 + 158628139 TGGGAAGGAAAATGCATTGGGGAACCCTGTGCGGATTCTTGTGGCTTTGGCCCTATCTTTTCTATGTCCAAGCTGTGCCCATCCAAAAAGTCCAAGATGACACCAAAACCCTCATCAAGACAATTGTCACCAGGATCAATGACATTTCACACACG
|
||||
s panTro1.chr6 129900247 155 + 161576975 TGGGAAGGAAAATGCATTGGGGAACCCTGTGCGGATTCTTGTGGCTTTGGCCCTATCTTTTCTATGTCCAAGCTGTGCCCATCCAAAAAGTCCAAGATGACACCAAAACCCTCATCAAGACAATTGTCACCAGGATCAATGACATTTCACACACG
|
||||
.. image:: ../static/images/maf_icons/interval2maf.png
|
||||
|
||||
</help>
|
||||
<code file="interval2maf_pairwise_code.py"/>
|
||||
|
||||
@@ -1,20 +1,29 @@
|
||||
<tool id="Interval_Maf_Merged_Fasta1" name="Stitch MAF blocks for intervals">
|
||||
<description>using locally cached alignments</description>
|
||||
<command interpreter="python2.4">interval_maf_to_merged_fasta.py $dbkey $species $mafSource $input1 $out_file1 $input1_chromCol $input1_startCol $input1_endCol $input1_strandCol -</command>
|
||||
<tool id="Interval_Maf_Merged_Fasta1" name="Stitch MAF blocks">
|
||||
<description>given a set of genomic intervals</description>
|
||||
<command interpreter="python2.4">#if $maf_source_type.maf_source == "user":#interval_maf_to_merged_fasta_user.py $dbkey $species $maf_source_type.input2 $input1 $out_file1 $input1_chromCol $input1_startCol $input1_endCol $input1_strandCol -
|
||||
#else:#interval_maf_to_merged_fasta.py $dbkey $species $maf_source_type.maf_uid $input1 $out_file1 $input1_chromCol $input1_startCol $input1_endCol $input1_strandCol -
|
||||
#end if
|
||||
</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param format="interval" name="input1" type="data" label="Choose intervals"/>
|
||||
<!-- <param name="unkown_gap_char" type="select">
|
||||
<label>Character for missing data</label>
|
||||
<option value="-">-</option>
|
||||
<option value="?">?</option>
|
||||
</param> -->
|
||||
</page>
|
||||
<page>
|
||||
<param name="mafSource" label="Choose alignments" type="select" dynamic_options="get_available_data( input1.dbkey )"/>
|
||||
<conditional name="maf_source_type">
|
||||
<param name="maf_source" type="select" label="MAF Source">
|
||||
<option value="cached" selected="true">Locally Cached Alignments</option>
|
||||
<option value="user">Alignments in Your History</option>
|
||||
</param>
|
||||
<when value="user">
|
||||
<param format="maf" name="input2" type="data" label="Choose alignments"/>
|
||||
</when>
|
||||
<when value="cached">
|
||||
<param name="maf_uid" label="Choose alignments" type="select" dynamic_options="get_available_data( input1.dbkey )"/>
|
||||
</when>
|
||||
</conditional>
|
||||
</page>
|
||||
<page>
|
||||
<param name="species" label="Choose species" type="select" display="checkboxes" multiple="True" dynamic_options="get_available_species( mafSource )" help="select species to be included in the final alignment" />
|
||||
<param name="species" label="Choose species" type="select" display="checkboxes" multiple="True" dynamic_options="get_available_species( maf_source_type )" help="select species to be included in the final alignment" />
|
||||
</page>
|
||||
</inputs>
|
||||
<outputs>
|
||||
@@ -23,25 +32,21 @@
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input1" value="1.bed" dbkey="hg17" format="bed"/>
|
||||
<param name="mafSource" value="8_WAY_MULTIZ_hg17"/>
|
||||
<param name="maf_source" value="cached"/>
|
||||
<param name="maf_uid" value="8_WAY_MULTIZ_hg17"/>
|
||||
<param name="species" value="canFam1,hg17,mm5,panTro1,rn3"/>
|
||||
<output name="out_file1" file="gene_bed_maf_to_fasta_out.dat" />
|
||||
<output name="out_file1" file="interval_maf_to_merged_fasta_out.dat" />
|
||||
</test>
|
||||
<test>
|
||||
<param name="input1" value="1.bed" dbkey="hg17" format="bed"/>
|
||||
<param name="maf_source" value="user"/>
|
||||
<param name="input2" value="5.maf"/>
|
||||
<param name="species" value="canFam1,hg17,mm5,panTro1,rn3"/>
|
||||
<output name="out_file1" file="interval_maf_to_merged_fasta_user_out.dat" />
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
What is the differences between these two tools: **Stitch MAF blocks for intervals using locally cached alignments** and **Stitch MAF blocks for intervals from user supplied alignments**?
|
||||
|
||||
* **Stitch MAF blocks for intervals using locally cached alignments** uses alignments stored at Galaxy installation at Penn State and is appropriate for most situations. It takes only one dataset as the input - a list of genomic intervals.
|
||||
* **Stitch MAF blocks for intervals from user supplied alignments** allows the user to work with his/her own alignments instead of those cached at Galaxy site. This tool takes two datasets as inputs - (1) genomic intervals and (2) alignments.
|
||||
|
||||
In the future these two tools will be merged.
|
||||
|
||||
------
|
||||
|
||||
**What It Does**
|
||||
**What it does**
|
||||
|
||||
A single genomic region can be covered by multiple alignment blocks. In many cases it is desirable to stitch these alignment blocks together. This tool accepts a list of genomic intervals. For every interval it performs the following:
|
||||
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
#build list of available data
|
||||
import os, sys
|
||||
import bx.align.maf
|
||||
maf_sets = {}
|
||||
|
||||
try:
|
||||
@@ -37,14 +38,53 @@ def get_available_data( build ):
|
||||
available_sets.append(('No data available for this build','None',True))
|
||||
return available_sets
|
||||
|
||||
def get_available_species( maf_uid ):
|
||||
available_sets = []
|
||||
for key in maf_sets[maf_uid]['builds']:
|
||||
available_sets.append((key,key,True))
|
||||
if len(available_sets) < 1:
|
||||
available_sets.append(('No data available for this configuration','None',True))
|
||||
return available_sets
|
||||
def get_available_species( maf_source_type ):
|
||||
if maf_source_type['maf_source'] == 'cached':
|
||||
maf_uid = maf_source_type['maf_uid']
|
||||
available_sets = []
|
||||
for key in maf_sets[maf_uid]['builds']:
|
||||
available_sets.append((key,key,True))
|
||||
if len(available_sets) < 1:
|
||||
available_sets.append(('No data available for this configuration','None',True))
|
||||
return available_sets
|
||||
else:
|
||||
try:
|
||||
rval = []
|
||||
species={}
|
||||
input_filename = maf_source_type['input2'].file_name
|
||||
try:
|
||||
file_in = open(input_filename, 'r')
|
||||
maf_reader = bx.align.maf.Reader( file_in )
|
||||
|
||||
for i, m in enumerate( maf_reader ):
|
||||
l = m.components
|
||||
for c in l:
|
||||
spec,chrom = bx.align.maf.src_split( c.src )
|
||||
if not spec or not chrom:
|
||||
spec = chrom = c.src
|
||||
if spec not in species:
|
||||
species[spec]={"bases":0,"nongaps":0}
|
||||
species[spec]["bases"] = species[spec]["bases"] + c.size + c.text.count("-")
|
||||
species[spec]["nongaps"] = species[spec]["nongaps"] + c.size
|
||||
|
||||
file_in.close()
|
||||
except Exception:
|
||||
return [("There is a problem with your MAF file",'None',True)]
|
||||
species_names = species.keys()
|
||||
species_names.sort()
|
||||
|
||||
for spec in species_names:
|
||||
#species_sequence[spec] = "".join(species_sequence[spec])
|
||||
|
||||
display = "%s: %i nongap, %i total bases" % (spec, species[spec]["nongaps"], species[spec]["bases"] )
|
||||
rval.append( ( display,spec,True) )
|
||||
|
||||
return rval
|
||||
except:
|
||||
return [("<B>You must wait for the MAF file to be created before you can use this tool.</B>",'None',True)]
|
||||
|
||||
def exec_before_job(app,inp_data, out_data, param_dict, tool):
|
||||
for name, data in out_data.items():
|
||||
data.name = data.name + " [" + maf_sets[param_dict['mafSource']]['description'] + "]"
|
||||
|
||||
def exec_before_job(app, inp_data, out_data, param_dict, tool):
|
||||
if param_dict['maf_source_type']['maf_source'] == "cached":
|
||||
for name, data in out_data.items():
|
||||
data.name = data.name + " [" + maf_sets[str(param_dict['maf_source_type']['maf_uid'])]['description'] + "]"
|
||||
|
||||
@@ -20,6 +20,9 @@
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool takes a list of block numbers, one per line, and extracts the corresponding MAF blocks from the provided file. Block numbers start at 0.
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
@@ -20,6 +20,9 @@
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool takes a MAF file and a size range and extracts the MAF blocks which fall within the specified range.
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
@@ -39,6 +39,9 @@ $maf_source_type.maf_source $maf_source_type.mafType $input1 $out_file1 $dbkey $
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool takes a MAF file and an interval file and relates coverage information by interval for each species.
|
||||
If a column does not exist in the reference genome, it is not included in the output.
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
<tool id="MAF_Thread_For_Species1" name="Join MAF Blocks">
|
||||
<tool id="MAF_Thread_For_Species1" name="Join MAF blocks">
|
||||
<description>by Species</description>
|
||||
<command interpreter="python2.4">maf_thread_for_species.py $input1 $out_file1 $species</command>
|
||||
<inputs>
|
||||
@@ -21,7 +21,7 @@
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**What It Does**
|
||||
**What it does**
|
||||
|
||||
This tool allows the user to merge MAF blocks which are adjoining in each specified species from a MAF file. Columns which contain only gaps are removed. Species which are not desired are removed from the output.
|
||||
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
<tool id="MAF_To_Fasta1" name="Maf to FASTA">
|
||||
<description>Converts a MAF formated file to FASTA format</description>
|
||||
<command interpreter="python2.4">#if $fasta_target_type.fasta_type == "multiple":#maf_to_fasta_multiple_sets.py $input1 $out_file1 $fasta_target_type.species $fasta_target_type.complete_blocks
|
||||
#else:#maf_to_fasta_concat.py $fasta_target_type.species $input1 $out_file1
|
||||
#end if
|
||||
</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param format="maf" name="input1" type="data" label="MAF file to convert"/>
|
||||
</page>
|
||||
<page>
|
||||
<conditional name="fasta_target_type">
|
||||
<param name="fasta_type" type="select" label="Type of FASTA Output">
|
||||
<option value="multiple" selected="true">Multiple Blocks</option>
|
||||
<option value="concatenated">One Sequence per Species</option>
|
||||
</param>
|
||||
<when value="multiple">
|
||||
<param name="species" type="select" label="Select species" dynamic_options="get_available_species( input1.file_name )" display="checkboxes" multiple="true" help="checked taxa will be included in the output"/>
|
||||
<param name="complete_blocks" type="select">
|
||||
<label>Choose to</label>
|
||||
<option value="partial_allowed">include blocks with missing species</option>
|
||||
<option value="partial_disallowed">exclude blocks with missing species</option>
|
||||
</param>
|
||||
</when>
|
||||
<when value="concatenated">
|
||||
<param name="species" type="select" label="Species to extract" dynamic_options="get_available_species( input1.file_name )" display="checkboxes" multiple="true"/>
|
||||
</when>
|
||||
</conditional>
|
||||
</page>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="fasta" name="out_file1" />
|
||||
</outputs>
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input1" value="3.maf"/>
|
||||
<param name="species" values="canFam1,hg17,mm5,panTro1,rn3"/>
|
||||
<output name="out_file1" file="cf_maf2fasta_concat.dat"/>
|
||||
</test>
|
||||
<test>
|
||||
<param name="input1" value="4.maf"/>
|
||||
<param name="species" values="bosTau2,canFam2,dasNov1,hg17,mm7,panTro1,rheMac2,rn3"/>
|
||||
<param name="complete_blocks" value="partial_allowed"/>
|
||||
<output name="out_file1" file="cf_maf2fasta.dat"/>
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
**Types of MAF to FASTA conversion**
|
||||
|
||||
* **Multiple Blocks** converts a single MAF block to a single FASTA block. For example, if you have 6 MAF blocks, they will be converted to 6 FASTA blocks.
|
||||
* **One Sequence per Species** converts MAF blocks to a single aggregated FASTA block. For example, if you have 6 MAF blocks, they will be converted and concatenated into a single FASTA block.
|
||||
|
||||
-------
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool converts MAF blocks to FASTA format and concatenates them into a single FASTA block or outputs multiple FASTA blocks separated by empty lines.
|
||||
|
||||
The interface for this tool contains two pages (steps):
|
||||
|
||||
* **Step 1 of 2**. Choose multiple alignments from history to be converted to FASTA format.
|
||||
* **Step 2 of 2**. Choose the type of output as well as the species from the alignment to be included in the output.
|
||||
|
||||
Multiple Block output has additional options:
|
||||
|
||||
* **Choose species** - the tool reads the alignment provided during Step 1 and generates a list of species contained within that alignment. Using checkboxes you can specify taxa to be included in the output (all species are selected by default).
|
||||
* **Choose to include/exclude blocks with missing species** - if an alignment block does not contain any one of the species you selected within **Choose species** menu and this option is set to **exclude blocks with missing species**, then such a block **will not** be included in the output (see **Example 2** below). For example, if you want to extact human, mouse, and rat from a series of alignments and one of the blocks does not contain mouse sequence, then this block will not be converted to FASTA and will not be returned.
|
||||
|
||||
|
||||
-----
|
||||
|
||||
**Example 1**:
|
||||
|
||||
In the concatenated approach, the following alignment::
|
||||
|
||||
##maf version=1
|
||||
a score=68686.000000
|
||||
s hg18.chr20 56827368 75 + 62435964 GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
s panTro2.chr20 56528685 75 + 62293572 GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
s rheMac2.chr10 89144112 69 - 94855758 GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
|
||||
s mm8.chr2 173910832 61 + 181976762 AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
|
||||
s canFam2.chr24 46551822 67 + 50763139 CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
|
||||
|
||||
a score=10289.000000
|
||||
s hg18.chr20 56827443 37 + 62435964 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
s panTro2.chr20 56528760 37 + 62293572 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
s rheMac2.chr10 89144181 37 - 94855758 ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
|
||||
will be converted to (**note** that because mm8 (mouse) and canFam2 (dog) are absent from the second block, they are replaced with gaps after concatenation)::
|
||||
|
||||
>canFam2
|
||||
CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C-------------------------------------
|
||||
>hg18
|
||||
GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
>mm8
|
||||
AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC--------------------------------------------
|
||||
>panTro2
|
||||
GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
>rheMac2
|
||||
GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
|
||||
------
|
||||
|
||||
**Example 2a**: Multiple Block Approach **Include all species** and **include blocks with missing species**:
|
||||
|
||||
The following alignment::
|
||||
|
||||
##maf version=1
|
||||
a score=68686.000000
|
||||
s hg18.chr20 56827368 75 + 62435964 GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
s panTro2.chr20 56528685 75 + 62293572 GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
s rheMac2.chr10 89144112 69 - 94855758 GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
|
||||
s mm8.chr2 173910832 61 + 181976762 AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
|
||||
s canFam2.chr24 46551822 67 + 50763139 CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
|
||||
|
||||
a score=10289.000000
|
||||
s hg18.chr20 56827443 37 + 62435964 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
s panTro2.chr20 56528760 37 + 62293572 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
s rheMac2.chr10 89144181 37 - 94855758 ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
|
||||
will be converted to::
|
||||
|
||||
>hg18.chr20(+):56827368-56827443|hg18_0
|
||||
GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
>panTro2.chr20(+):56528685-56528760|panTro2_0
|
||||
GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
>rheMac2.chr10(-):89144112-89144181|rheMac2_0
|
||||
GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
|
||||
>mm8.chr2(+):173910832-173910893|mm8_0
|
||||
AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
|
||||
>canFam2.chr24(+):46551822-46551889|canFam2_0
|
||||
CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
|
||||
|
||||
>hg18.chr20(+):56827443-56827480|hg18_1
|
||||
ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
>panTro2.chr20(+):56528760-56528797|panTro2_1
|
||||
ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
>rheMac2.chr10(-):89144181-89144218|rheMac2_1
|
||||
ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
|
||||
-----
|
||||
|
||||
**Example 2b**: Multiple Block Approach **Include hg18 and mm8** and **exclude blocks with missing species**:
|
||||
|
||||
The following alignment::
|
||||
|
||||
##maf version=1
|
||||
a score=68686.000000
|
||||
s hg18.chr20 56827368 75 + 62435964 GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
s panTro2.chr20 56528685 75 + 62293572 GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
|
||||
s rheMac2.chr10 89144112 69 - 94855758 GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
|
||||
s mm8.chr2 173910832 61 + 181976762 AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
|
||||
s canFam2.chr24 46551822 67 + 50763139 CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
|
||||
|
||||
a score=10289.000000
|
||||
s hg18.chr20 56827443 37 + 62435964 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
s panTro2.chr20 56528760 37 + 62293572 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
s rheMac2.chr10 89144181 37 - 94855758 ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
|
||||
|
||||
will be converted to (**note** that the second MAF block, which does not have mm8, is not included in the output)::
|
||||
|
||||
>hg18.chr20(+):56827368-56827443|hg18_0
|
||||
GACAGGGTGCATCTGGGAGGGCCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC
|
||||
>mm8.chr2(+):173910832-173910893|mm8_0
|
||||
AGAAGGATCCACCT---------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC------
|
||||
|
||||
------
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**About formats**
|
||||
|
||||
**MAF format** multiple alignment format file. This format stores multiple alignments at the DNA level between entire genomes.
|
||||
|
||||
- The .maf format is line-oriented. Each multiple alignment ends with a blank line.
|
||||
- Each sequence in an alignment is on a single line.
|
||||
- Lines starting with # are considered to be comments.
|
||||
- Each multiple alignment is in a separate paragraph that begins with an "a" line and contains an "s" line for each sequence in the multiple alignment.
|
||||
- Some MAF files may contain two optional line types:
|
||||
|
||||
- An "i" line containing information about what is in the aligned species DNA before and after the immediately preceding "s" line;
|
||||
- An "e" line containing information about the size of the gap between the alignments that span the current block.
|
||||
|
||||
|
||||
</help>
|
||||
<code file="maf_to_fasta_code.py"/>
|
||||
</tool>
|
||||
Reference in New Issue
Block a user