Removing old/unused files for maf tools (mostly config files for user only versions).

This commit is contained in:
Daniel Blankenberg
2007-10-30 19:47:33 +00:00
parent fad429f4d2
commit 96afd6bf8d
7 changed files with 0 additions and 393 deletions
-1
View File
@@ -77,7 +77,6 @@
<tool file="extract/interval2maf.xml" />
<tool file="extract/interval_maf_to_merged_fasta.xml" />
<!-- <tool file="extract/genebed_maf_to_fasta.xml"/>
<tool file="extract/genebed_maf_to_fasta_user.xml"/>
<tool file="filters/maf/maf_stats.xml"/> -->
<tool file="filters/maf/maf_limit_to_species.xml"/>
<tool file="filters/maf/maf_limit_size.xml"/>
@@ -1,27 +0,0 @@
<tool id="GeneBed_Maf_Fasta_User1" name="Extract Gene User blocks">
<description>from a MAF File</description>
<command interpreter="python2.4">genebed_maf_to_fasta_user.py $dbkey $species $input2 $input1 $out_file1 -</command>
<inputs>
<param format="interval" name="input1" type="data" label="Interval File"/>
<param format="maf" name="input2" type="data" label="MAF File"/>
<param name="species" label="Choose desired species" type="select" display="checkboxes" multiple="True" >
<select_options data_ref="input2" func="get_options_for_species" />
</param>
</inputs>
<outputs>
<data format="fasta" name="out_file1" />
</outputs>
<!--
<tests>
<test>
<param name="input1" value="gene_bed.dat"/>
<param name="input2" value="gene_bed_maf_to_fasta_user_maf_in.dat"/>
<param name="species" value="canFam1,hg17,mm5,panTro1,rn3"/>
<output name="out_file1" file="gene_bed_maf_to_fasta_out.dat" />
</test>
</tests>
-->
<help>
This tool takes a MAF file and a Gene BED file and creates a FASTA file which contains alignments of user specified species which correspond to the coding exons of the supplied Gene.
</help>
</tool>
@@ -1,61 +0,0 @@
<tool id="Interval_Maf_Merged_Fasta_User1" name="Stitch MAF blocks for intervals">
<description>using used-defined alignments</description>
<command interpreter="python2.4">interval_maf_to_merged_fasta_user.py $dbkey $species $input2 $input1 $out_file1 $input1_chromCol $input1_startCol $input1_endCol $input1_strandCol -</command>
<inputs>
<page>
<param format="interval" name="input1" type="data" label="Choose intervals"/>
<param format="maf" name="input2" type="data" label="Choose alignments"/>
<!-- <param name="unkown_gap_char" type="select">
<label>Character for missing data</label>
<option value="-">-</option>
<option value="?">?</option>
</param> -->
</page>
<page>
<param name="species" label="Choose species" type="select" display="checkboxes" multiple="True" dynamic_options="get_available_species( input2.file_name )"/>
</page>
</inputs>
<outputs>
<data format="fasta" name="out_file1" />
</outputs>
<tests>
<test>
<param name="input1" value="1.bed" dbkey="hg17" format="bed"/>
<param name="input2" value="5.maf"/>
<param name="species" value="canFam1,hg17,mm5,panTro1,rn3"/>
<output name="out_file1" file="gene_bed_maf_to_fasta_user_out.dat" />
</test>
</tests>
<help>
.. class:: infomark
What is the differences between these two tools: **Stitch MAF blocks for intervals using locally cached alignments** and **Stitch MAF blocks for intervals from user supplied alignments**?
* **Stitch MAF blocks for intervals using locally cached alignments** uses alignments stored at Galaxy installation at Penn State and is appropriate for most situations. It takes only one dataset as the input - a list of genomic intervals.
* **Stitch MAF blocks for intervals from user supplied alignments** allows the user to work with his/her own alignemnts instead of those cached at Galaxy site. This tool takes two datasets as inputs - (1) genomic intervals and (2) alignemnts.
In the future these two tools will be merged.
------
**What It Does**
A single genomic region can be covered by multiple alignment blocks. In many cases it is desirable to stitch these alignment blocks together. This tool accepts a list of genomic intervals. For every interval it performs the following:
* finds all MAF blocks that overlap the interval;
* sorts MAF blocks by alignment score;
* stitches blocks together and resolves overlaps based on alignment score;
* outputs alignments in FASTA format.
------
**Example**
Here three MAF blocks overlapping a single interval are stitched together. Space between blocks 2 and 3 is filled with gaps:
.. image:: ../static/images/maf_icons/stitchMaf.png
</help>
<code file="interval_maf_to_merged_fasta_user_code.py"/>
</tool>
@@ -1,39 +0,0 @@
import pkg_resources; pkg_resources.require( "bx-python" )
from bx.align import maf
# No initialization required.
#return lists of species available, showing gapped and ungapped base counts
def get_available_species( input_filename ):
try:
rval = []
species={}
try:
file_in = open(input_filename, 'r')
maf_reader = maf.Reader( file_in )
for i, m in enumerate( maf_reader ):
l = m.components
for c in l:
spec,chrom = maf.src_split( c.src )
if not spec or not chrom:
spec = chrom = c.src
if spec not in species:
species[spec]={"bases":0,"nongaps":0}
species[spec]["bases"] = species[spec]["bases"] + c.size + c.text.count("-")
species[spec]["nongaps"] = species[spec]["nongaps"] + c.size
file_in.close()
except:
return [("There is a problem with your MAF file",'None',True)]
species_names = species.keys()
species_names.sort()
for spec in species_names:
#species_sequence[spec] = "".join(species_sequence[spec])
display = "%s: %i nongap, %i total bases" % (spec, species[spec]["nongaps"], species[spec]["bases"] )
rval.append( ( display,spec,True) )
return rval
except:
return [("<B>You must wait for the MAF file to be created before you can use this tool.</B>",'None',True)]
-37
View File
@@ -1,37 +0,0 @@
<tool id="User_Interval2Maf1" name="Extract MAF blocks">
<description>from user supplied alignments</description>
<command interpreter="python2.4">user_interval2maf.py --dbkey=$dbkey --chromCol=$input1_chromCol --startCol=$input1_startCol --endCol=$input1_endCol --strandCol=$input1_strandCol --mafFile=$mafFile --interval_file=$input1 --output_file=$out_file1</command>
<inputs>
<page>
<param format="interval" name="input1" type="data" label="Choose intervals"/>
<param format="maf" name="mafFile" label="Choose alignments" type="data"/>
</page>
</inputs>
<outputs>
<data format="maf" name="out_file1" />
</outputs>
<help>
.. class:: infomark
What is the differences between these two tools: **Extract MAF blocks from locally cached alignments** and **Extract MAF blocks from user supplied alignments**?
* **Extract MAF blocks from locally cached alignments** uses alignments stored at Galaxy installation at Penn State and is appropriate for most situations. It takes only one dataset as the input - a list of genomic intervals.
* **Extract MAF blocks from user supplied alignments** allows the user to work with his/her own alignments instead of those cached at Galaxy site. This tool takes two datasets as inputs - (1) genomic intervals and (2) alignments.
-----
**What it does**
This tool takes genomic coordinates, superimposes them on multiple alignments (in MAF format) uploaded by the user, and excises alignment blocks corresponding to each set of coordinates. Alignment blocks that extend past START and/or END positions of an interval are trimmed. Note that a single genomic interval may correspond to two or more alignment blocks.
-----
**Example**
Here a single interval is superimposed on three MAF blocks. Blocks 1 and 3 are trimmed because they extend beyond boundaries of the interval:
.. image:: ../static/images/maf_icons/interval2maf.png
</help>
</tool>
-93
View File
@@ -1,93 +0,0 @@
<tool id="MAF_To_Fasta_Concat1" name="Maf to concatenated FASTA">
<description>Converts a MAF formated file to FASTA format</description>
<command interpreter="python2.4">maf_to_fasta_concat.py $species $input1 $out_file1</command>
<inputs>
<param format="maf" name="input1" type="data" label="MAF file"/>
<param name="species" type="select" label="Species to extract" display="checkboxes" multiple="true">
<select_options data_ref="input1" func="get_options_for_species" />
</param>
</inputs>
<outputs>
<data format="fasta" name="out_file1" />
</outputs>
<tests>
<test>
<param name="input1" value="3.maf"/>
<param name="species" values="canFam1,hg17,mm5,panTro1,rn3"/>
<output name="out_file1" file="cf_maf2fasta_concat.dat"/>
</test>
</tests>
<help>
.. class:: infomark
What is the differences between these two tools: **MAF to FASTA** and **MAF to concatenated FASTA** (scroll down for the description of MAF format)?
* **MAF to FASTA** converts a single MAF block to a single FASTA block. For example, if you have 6 MAF blocks, they will be converted to 6 FASTA blocks.
* **MAF to concatenated FASTA** converts MAF blocks to a single aggregated FASTA block. For example, if you have 6 MAF blocks, they will be converted and concatenated into a single FASTA block.
In the future these two tools will be merged.
-------
**What it does**
This tool converts MAF blocks to FASTA format and concatenates them into a single FASTA block.
The interface for this tool contains two pages (steps):
* **Step 1 of 2**. Choose multiple alignments from history to be converted to FASTA format.
* **Step 2 of 2**. Choose species from the alignment to be included in the output.
-----
**Example**:
The following alignment::
##maf version=1
a score=68686.000000
s hg18.chr20 56827368 75 + 62435964 GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
s panTro2.chr20 56528685 75 + 62293572 GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
s rheMac2.chr10 89144112 69 - 94855758 GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
s mm8.chr2 173910832 61 + 181976762 AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
s canFam2.chr24 46551822 67 + 50763139 CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
a score=10289.000000
s hg18.chr20 56827443 37 + 62435964 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
s panTro2.chr20 56528760 37 + 62293572 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
s rheMac2.chr10 89144181 37 - 94855758 ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
will be converted to (**note** that because mm8 (mouse) and canFam2 (dog) are absent from the second block, they are replaced with gaps after concatenation)::
&gt;canFam2
CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C-------------------------------------
&gt;hg18
GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
&gt;mm8
AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC--------------------------------------------
&gt;panTro2
GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
&gt;rheMac2
GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
------
.. class:: infomark
**About formats**
**MAF format** multiple alignment format file. This format stores multiple alignments at the DNA level between entire genomes.
- The .maf format is line-oriented. Each multiple alignment ends with a blank line.
- Each sequence in an alignment is on a single line.
- Lines starting with # are considered to be comments.
- Each multiple alignment is in a separate paragraph that begins with an "a" line and contains an "s" line for each sequence in the multiple alignment.
- Some MAF files may contain two optional line types:
- An "i" line containing information about what is in the aligned species DNA before and after the immediately preceding "s" line;
- An "e" line containing information about the size of the gap between the alignments that span the current block.
</help>
</tool>
@@ -1,135 +0,0 @@
<tool id="MAF_To_Fasta1" name="Maf to FASTA">
<description>Converts a MAF formated file to FASTA format</description>
<command interpreter="python2.4">maf_to_fasta_multiple_sets.py $input1 $out_file1 $species $complete_blocks</command>
<inputs>
<param format="maf" name="input1" type="data" label="MAF file to convert"/>
<param name="species" type="select" label="Select species" display="checkboxes" multiple="true" help="checked taxa will be included in the output">
<select_options data_ref="input1" func="get_options_for_species" />
</param>
<param name="complete_blocks" type="select" label="Choose to">
<option value="partial_allowed">include blocks with missing species</option>
<option value="partial_disallowed">exclude blocks with missing species</option>
</param>
</inputs>
<outputs>
<data format="fasta" name="out_file1" />
</outputs>
<tests>
<test>
<param name="input1" value="4.maf"/>
<param name="species" values="bosTau2,canFam2,dasNov1,hg17,mm7,panTro1,rheMac2,rn3"/>
<param name="complete_blocks" value="partial_allowed"/>
<output name="out_file1" file="cf_maf2fasta.dat"/>
</test>
</tests>
<help>
.. class:: infomark
What is the differences between these two tools: **MAF to FASTA** and **MAF to concatenated FASTA** (scroll down for the description of MAF format)?
* **MAF to FASTA** converts a single MAF block to a single FASTA block. For example, if you have 6 MAF blocks, they will be converted to 6 FASTA blocks.
* **MAF to concatenated FASTA** converts MAF blocks to a single aggregated FASTA block. For example, if you have 6 MAF blocks, they will be converted and concatenated into a single FASTA block.
In the future these two tools will be merged.
-------
**What it does**
This tool converts MAF blocks to FASTA blocks separated by empty lines.
The interface for this tool contains two pages (steps):
* **Step 1 of 2**. Choose multiple alignments from history to be converted to FASTA format.
* **Step 2 of 2**. Choose species from the alignment to be included in the output and specify how to deal with alignment blocks that lack one or more species:
* **Choose species** - the tool reads the alignment provided during Step 1 and generates a list of species contained within that alignment. Using checkboxes you can specify taxa to be included in the output (all species are selected by default).
* **Choose to include/exclude blocks with missing species** - if an alignment block does not contain any one of the species you selected within **Choose species** menu and this option is set to **exclude blocks with missing species**, then such a block **will not** be included in the output (see **Example 2** below). For example, if you want to extact human, mouse, and rat from a series of alignments and one of the blocks does not contain mouse sequence, then this block will not be converted to FASTA and will not be returned.
-----
**Example 1**: **Include all species** and **include blocks with missing species**:
The following alignment::
##maf version=1
a score=68686.000000
s hg18.chr20 56827368 75 + 62435964 GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
s panTro2.chr20 56528685 75 + 62293572 GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
s rheMac2.chr10 89144112 69 - 94855758 GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
s mm8.chr2 173910832 61 + 181976762 AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
s canFam2.chr24 46551822 67 + 50763139 CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
a score=10289.000000
s hg18.chr20 56827443 37 + 62435964 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
s panTro2.chr20 56528760 37 + 62293572 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
s rheMac2.chr10 89144181 37 - 94855758 ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
will be converted to::
&gt;hg18.chr20(+):56827368-56827443|hg18_0
GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
&gt;panTro2.chr20(+):56528685-56528760|panTro2_0
GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
&gt;rheMac2.chr10(-):89144112-89144181|rheMac2_0
GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
&gt;mm8.chr2(+):173910832-173910893|mm8_0
AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
&gt;canFam2.chr24(+):46551822-46551889|canFam2_0
CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
&gt;hg18.chr20(+):56827443-56827480|hg18_1
ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
&gt;panTro2.chr20(+):56528760-56528797|panTro2_1
ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
&gt;rheMac2.chr10(-):89144181-89144218|rheMac2_1
ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
-----
**Example 2**: **Include hg18 and mm8** and **exclude blocks with missing species**:
The following alignment::
##maf version=1
a score=68686.000000
s hg18.chr20 56827368 75 + 62435964 GACAGGGTGCATCTGGGAGGG---CCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
s panTro2.chr20 56528685 75 + 62293572 GACAGGGTGCATCTGAGAGGG---CCTGCCAGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC-
s rheMac2.chr10 89144112 69 - 94855758 GACAGGGTGCATCTGAGAGGG---CCTGCTGGGCCTTTG-TTCAAAACTAGATATGCCCCAACTCCAATTCTA-------
s mm8.chr2 173910832 61 + 181976762 AGAAGGATCCACCT------------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC-------
s canFam2.chr24 46551822 67 + 50763139 CG------GCGTCTGTAAGGGGCCACCGCCCGGCCTGTG-CTCAAAGCTACAAATGACTCAACTCCCAACCGA------C
a score=10289.000000
s hg18.chr20 56827443 37 + 62435964 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
s panTro2.chr20 56528760 37 + 62293572 ATGTGCAGAAAATGTGATACAGAAACCTGCAGAGCAG
s rheMac2.chr10 89144181 37 - 94855758 ATGTGCGGAAAATGTGATACAGAAACCTGCAGAGCAG
will be converted to (**note** that the second MAF block, which does not have mm8, is not included in the output)::
&gt;hg18.chr20(+):56827368-56827443|hg18_0
GACAGGGTGCATCTGGGAGGGCCTGCCGGGCCTTTA-TTCAACACTAGATACGCCCCATCTCCAATTCTAATGGAC
&gt;mm8.chr2(+):173910832-173910893|mm8_0
AGAAGGATCCACCT---------TGCTGGGCCTCTGCTCCAGCAAGACCCACCTCCCAACTCAAATGCCC------
-----
.. class:: infomark
**About formats**
**MAF format** multiple alignment format file. This format stores multiple alignments at the DNA level between entire genomes.
- The .maf format is line-oriented. Each multiple alignment ends with a blank line.
- Each sequence in an alignment is on a single line.
- Lines starting with # are considered to be comments.
- Each multiple alignment is in a separate paragraph that begins with an "a" line and contains an "s" line for each sequence in the multiple alignment.
- Some MAF files may contain two optional line types:
- An "i" line containing information about what is in the aligned species DNA before and after the immediately preceding "s" line;
- An "e" line containing information about the size of the gap between the alignments that span the current block.
</help>
</tool>