A simple tool for merging columns. Surprisingly useful for some analyses and also requested by a user lest Friday

This commit is contained in:
Anton Nekrutenko
2009-02-09 23:57:41 -05:00
parent 63ff7980d0
commit 19bb2c924c
5 changed files with 91 additions and 18 deletions
+8 -8
View File
@@ -32,7 +32,7 @@
<div class="welcomeBlue" id="screencasts" align="center">
<strong>Introducing Galactic Quickies</strong>
<hr>
Galactic quickies are <i>super-short</i> screencasts that are <i>always</i> under 5 minutes. We thought it may be a good way to spread the word about Galaxy's functionality while keeping the &quot;annoyance factor&quot; to the minimum. The quickies will be updated weekly.
Galactic quickies are <i>super-short</i> screencasts that are <i>always</i> under 5 minutes. We thought it may be a good way to spread the word about Galaxy's functionality while keeping the &quot;annoyance factor&quot; to the minimum. The quickies are updated weekly. For previous quickies <a target="_blank" class="reference" href="http://galaxy.psu.edu/screencasts.html">click here</a>.
<hr>
<table width="100%" border="0" height="100%">
<tr>
@@ -40,8 +40,8 @@
<br>
<script type="text/javascript"><!--
QT_WritePoster_XHTML('click to play', 'http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq-poster.jpg',
'http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.mov',
QT_WritePoster_XHTML('click to play', 'http://screencast.g2.bx.psu.edu/galaxy/quickie2_Grouping/quickie2_Grouping-poster.jpg',
'http://screencast.g2.bx.psu.edu/galaxy/quickie2_Grouping/quickie2_Grouping.mov',
'640', '480', '',
'controller', 'true',
'autoplay', 'true',
@@ -52,15 +52,15 @@
</script>
<noscript>
<object width="640" height="480" classid="clsid:02BF25D5-8C17-4B23-BC80-D3488ABDDC6B" codebase="http://www.apple.com/qtactivex/qtplugin.cab">
<param name="src" value="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.jpg" />
<param name="href" value="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.mov" />
<param name="src" value="http://screencast.g2.bx.psu.edu/galaxy/quickie2_Grouping/quickie2_Grouping.jpg" />
<param name="href" value="http://screencast.g2.bx.psu.edu/galaxy/quickie2_Grouping/quickie2_Grouping.mov" />
<param name="target" value="myself" />
<param name="controller" value="false" />
<param name="autoplay" value="false" />
<param name="scale" value="aspect" />
<embed width="640" height="480" type="video/quicktime" pluginspage="http://www.apple.com/quicktime/download/"
src="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.jpg"
href="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/quickie1_TabSeq.mov"
src="http://screencast.g2.bx.psu.edu/galaxy/quickie2_Grouping/quickie2_Grouping.jpg"
href="http://screencast.g2.bx.psu.edu/galaxy/quickie2_Grouping/quickie2_Grouping.mov"
target="myself"
controller="false"
autoplay="false"
@@ -73,7 +73,7 @@
</tr>
</table>
<br>
For a high resolution version <a target="_blank" class="reference" href="http://screencast.g2.bx.psu.edu/galaxy/quickie1_TabSeq/">click here</a>.
For a high resolution version <a target="_blank" class="reference" href="http://screencast.g2.bx.psu.edu/galaxy/quickie2_Grouping/">click here</a>.
<br>
<hr>
</div>
+2 -1
View File
@@ -41,7 +41,8 @@
<tool file="filters/fixedValueColumn.xml" />
<tool file="stats/column_maker.xml" />
<tool file="filters/catWrapper.xml" />
<tool file="filters/condense_characters.xml" />
<tool file="filters/cutWrapper.xml" />
<tool file="filters/mergeCols.xml" />
<tool file="filters/convert_characters.xml" />
<tool file="filters/CreateInterval.xml" />
<tool file="filters/cutWrapper.xml" />
+25
View File
@@ -0,0 +1,25 @@
import sys, re
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
def __main__():
infile = open ( sys.argv[1], 'r')
cols = ( re.sub( '\s*','',sys.argv[2] ) ).split( ',' )
outfile = open ( sys.argv[3], 'w')
for line in infile:
line = line.rstrip( '\r\n' )
if line and not line.startswith( '#' ):
fields = line.split( '\t' )
line += '\t'
for col in cols:
try:
line += fields[ int( col.lstrip( 'c' ) ) -1 ]
except:
stop_err( 'Column %s does not appear in the input file' % str( col ) )
print >>outfile, line
if __name__ == "__main__" : __main__()
+47
View File
@@ -0,0 +1,47 @@
<tool id="mergeCols1" name="Merge columns">
<description>together</description>
<command interpreter="python">mergeCols.py $input "$columnList" $out_file1</command>
<inputs>
<param name="columnList" size="10" type="text" value="c1,c2" label="Merge columns"/>
<param format="tabular" name="input" type="data" label="In"/>
</inputs>
<outputs>
<data format="tabular" name="out_file1" />
</outputs>
<tests>
<test>
<param name="columnList" value="c4, c1"/>
<param name="input" value="1.bed"/>
<output name="out_file1" file="mergeCols.dat"/>
</test>
</tests>
<help>
**What it does**
This tool merges columns together
- Columns are specified as **c1**, **c2**, and so on. Column count begins with **1**
- Columns can be specified in any order (e.g., **c2,c1,c6**)
-----
**Example**
Input dataset (five columns: c1, c2, c3, c4, and c5)::
1 10 1000 gene1 chr
2 100 1500 gene2 chr
merging columns "**c5,c1**" will return::
1 10 1000 gene1 chr chr1
2 100 1500 gene2 chr chr2
.. class:: infomark
Note that all original columns are preserved and the result of merge is added as the rightmost column.
</help>
</tool>
+9 -9
View File
@@ -1,14 +1,14 @@
<tool id="lastz_wrapper_1" name="Lastz" version="1.0.0">
<description> map short reads against reference sequence</description>
<command>
#if ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#if ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
#end if
</command>
<inputs>
@@ -16,7 +16,7 @@
<param name="input1" format="fasta" type="data" label="against reference" help="must be a single sequence"/>
<param name="out_format" type="select" label="Select output format">
<option value="diffs">Polymorphisms</option>
<option value="maf">Alignments in MAF format</option>
<option value="maf-">Alignments in MAF format</option>
</param>
<conditional name="params">