mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
push changesets to security from central
This commit is contained in:
@@ -38,3 +38,4 @@ The EpiGRAPH_ web service enables biologists to uncover hidden associations in v
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
<?xml version="1.0"?>
|
||||
<tool name="Access Libraries" id="library_access1">
|
||||
<description>stored locally</description>
|
||||
<inputs action="/library/index" method="get" target="_parent">
|
||||
<param name="default_action" type="hidden" value="add" />
|
||||
</inputs>
|
||||
<uihints minwidth="800"/>
|
||||
</tool>
|
||||
@@ -5,7 +5,8 @@ from shutil import copyfile
|
||||
|
||||
#post processing, set build for data and add additional data to history
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
|
||||
history = out_data.items()[0][1].history
|
||||
base_dataset = out_data.items()[0][1]
|
||||
history = base_dataset.history
|
||||
if history == None:
|
||||
print "unknown history!"
|
||||
return
|
||||
@@ -38,6 +39,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
|
||||
newdata.extension = file_type
|
||||
newdata.name = basic_name + " (" + description + ")"
|
||||
history.add_dataset( newdata )
|
||||
app.security_agent.copy_dataset_permissions( base_dataset.dataset, newdata.dataset )
|
||||
app.model.flush()
|
||||
try:
|
||||
copyfile(filepath,newdata.file_name)
|
||||
|
||||
@@ -84,7 +84,8 @@ from galaxy import datatypes, config, jobs, tools
|
||||
from shutil import copyfile
|
||||
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr):
|
||||
history = out_data.items()[0][1].history
|
||||
base_dataset = out_data.items()[0][1]
|
||||
history = base_dataset.history
|
||||
if history == None:
|
||||
print "unknown history!"
|
||||
return
|
||||
@@ -134,6 +135,7 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
|
||||
newdata.extension = file_type
|
||||
newdata.name = basic_name + " (" + microbe_info[kingdom][org]['chrs'][chr]['data'][description]['feature'] +" for "+microbe_info[kingdom][org]['name']+":"+chr + ")"
|
||||
newdata.flush()
|
||||
app.security_agent.copy_dataset_permissions( base_dataset.dataset, newdata.dataset )
|
||||
history.add_dataset( newdata )
|
||||
app.model.flush()
|
||||
try:
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
<tool id="cshl_fasta_clipping_histogram" name="Length Distribution">
|
||||
<description>chart</description>
|
||||
<command>fasta_clipping_histogram.pl $input $outfile</command>
|
||||
|
||||
<inputs>
|
||||
<param format="fasta" name="input" type="data" label="Library to analyze" />
|
||||
</inputs>
|
||||
|
||||
<outputs>
|
||||
<data format="png" name="outfile" metadata_source="input" />
|
||||
</outputs>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool creates a histogram image of sequence lengths distribution in a given fasta data set file.
|
||||
|
||||
**TIP:** Use this tool after clipping your library (with **FASTX Clipper tool**), to visualize the clipping results.
|
||||
|
||||
-----
|
||||
|
||||
**Output Examples**
|
||||
|
||||
|
||||
In the following library, most sequences are 24-mers to 27-mers.
|
||||
This could indicate an abundance of endo-siRNAs (depending of course of what you've tried to sequence in the first place).
|
||||
|
||||
.. image:: ../static/fastx_icons/fasta_clipping_histogram_1.png
|
||||
|
||||
|
||||
In the following library, most sequences are 19,22 or 23-mers.
|
||||
This could indicate an abundance of miRNAs (depending of course of what you've tried to sequence in the first place).
|
||||
|
||||
.. image:: ../static/fastx_icons/fasta_clipping_histogram_2.png
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTA-Clipping-Histogram is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,75 @@
|
||||
<tool id="cshl_fasta_collapser" name="Collapse">
|
||||
<description>sequences</description>
|
||||
<command>fasta_collapser.pl $input $output</command>
|
||||
|
||||
<inputs>
|
||||
<param format="fasta" name="input" type="data" label="Library to collapse" />
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input" value="fasta_collapser1.fasta" />
|
||||
<output name="output" file="fasta_collapser1.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="fasta" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool collapses identical sequences in a FASTA file into a single sequence.
|
||||
|
||||
--------
|
||||
|
||||
**Example**
|
||||
|
||||
Example Input File (Sequence "ATAT" appears multiple times)::
|
||||
|
||||
>CSHL_2_FC0042AGLLOO_1_1_605_414
|
||||
TGCG
|
||||
>CSHL_2_FC0042AGLLOO_1_1_537_759
|
||||
ATAT
|
||||
>CSHL_2_FC0042AGLLOO_1_1_774_520
|
||||
TGGC
|
||||
>CSHL_2_FC0042AGLLOO_1_1_742_502
|
||||
ATAT
|
||||
>CSHL_2_FC0042AGLLOO_1_1_781_514
|
||||
TGAG
|
||||
>CSHL_2_FC0042AGLLOO_1_1_757_487
|
||||
TTCA
|
||||
>CSHL_2_FC0042AGLLOO_1_1_903_769
|
||||
ATAT
|
||||
>CSHL_2_FC0042AGLLOO_1_1_724_499
|
||||
ATAT
|
||||
|
||||
Example Output file::
|
||||
|
||||
>1-1
|
||||
TGCG
|
||||
>2-4
|
||||
ATAT
|
||||
>3-1
|
||||
TGGC
|
||||
>4-1
|
||||
TGAG
|
||||
>5-1
|
||||
TTCA
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
Original Sequence Names / Lane descriptions (e.g. "CSHL_2_FC0042AGLLOO_1_1_742_502") are discarded.
|
||||
|
||||
The output seqeunce name is composed of two numbers: the first is the sequence's number, the second is the multiplicity value.
|
||||
|
||||
The following output::
|
||||
|
||||
>2-4
|
||||
ATAT
|
||||
|
||||
means that the sequence "ATAT" is the second sequence in the file, and it appeared 4 times in the input FASTA file.
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
@@ -0,0 +1,66 @@
|
||||
<tool id="cshl_fastq_nucleotides_distribution" name="Nucleotides Distribution">
|
||||
<description>chart</description>
|
||||
<command>fastq_nucleotide_distribution_graph.sh -t '$input.name' -i $input -o $output</command>
|
||||
|
||||
<inputs>
|
||||
<param format="txt" name="input" type="data" label="Statistics Text File (output of 'FASTQ Statistics' tool)" />
|
||||
</inputs>
|
||||
|
||||
<outputs>
|
||||
<data format="png" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
Creates a stacked-histogram graph for the nucleotide distribution in the Solexa library.
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** Use the **FASTQ Statistics** tool to generate the report file needed for this tool.
|
||||
|
||||
-----
|
||||
|
||||
**Output Examples**
|
||||
|
||||
|
||||
|
||||
The following chart clearly shows the barcode used at the 5'-end of the library: **GATCT**
|
||||
|
||||
.. image:: ../static/fastx_icons/fastq_nucleotides_distribution_1.png
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
In the following chart, one can almost 'read' the most abundant sequence by looking at the dominant values: **TGATA TCGTA TTGAT GACTG AA...**
|
||||
|
||||
.. image:: ../static/fastx_icons/fastq_nucleotides_distribution_2.png
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
The following chart shows a growing number of unknown (N) nucleotides towards later cycles (which might indicate a sequencing problem):
|
||||
|
||||
.. image:: ../static/fastx_icons/fastq_nucleotides_distribution_3.png
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
But most of the time, the chart will look rather random:
|
||||
|
||||
.. image:: ../static/fastx_icons/fastq_nucleotides_distribution_4.png
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTQ-Nucleotides-Distribution is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,82 @@
|
||||
<tool id="cshl_fastq_qual_conv" name="Quality format converter">
|
||||
<description>(ASCII-Numeric)</description>
|
||||
<command>zcat -f $input | fastq_quality_converter $QUAL_FORMAT -o $output</command>
|
||||
<inputs>
|
||||
<param format="fastqsolexa" name="input" type="data" label="Library to convert" />
|
||||
|
||||
<param name="QUAL_FORMAT" type="select" label="Desired output format">
|
||||
<option value="-a">ASCII (letters) quality scores</option>
|
||||
<option value="-n">Numeric quality scores</option>
|
||||
</param>
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- ASCII to NUMERIC -->
|
||||
<param name="input" value="fastq_qual_conv1.fastq" />
|
||||
<param name="QUAL_FORMAT" value="Numeric quality scores" />
|
||||
<output name="output" file="fastq_qual_conv1.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- ASCII to ASCII (basically, a no-op, but it should still produce a valid output -->
|
||||
<param name="input" value="fastq_qual_conv1.fastq" />
|
||||
<param name="QUAL_FORMAT" value="ASCII (letters) quality scores" />
|
||||
<output name="output" file="fastq_qual_conv1a.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- NUMERIC to ASCII -->
|
||||
<param name="input" value="fastq_qual_conv2.fastq" />
|
||||
<param name="QUAL_FORMAT" value="ASCII (letters) quality scores" />
|
||||
<output name="output" file="fastq_qual_conv2.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- NUMERIC to NUMERIC (basically, a no-op, but it should still produce a valid output -->
|
||||
<param name="input" value="fastq_qual_conv2.fastq" />
|
||||
<param name="QUAL_FORMAT" value="Numeric quality scores" />
|
||||
<output name="output" file="fastq_qual_conv2n.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="fastqsolexa" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
Converts a solexa FASTQ file to/from numeric or ASCII quality format.
|
||||
|
||||
.. class:: warningmark
|
||||
|
||||
Re-scaling is **not** performed. (e.g. conversion from Phred scale to Solexa scale).
|
||||
|
||||
|
||||
-----
|
||||
|
||||
FASTQ with Numeric quality scores::
|
||||
|
||||
@CSHL__2_FC042AGWWWXX:8:1:120:202
|
||||
ACGATAGATCGGAAGAGCTAGTATGCCGTTTTCTGC
|
||||
+CSHL__2_FC042AGWWWXX:8:1:120:202
|
||||
40 40 40 40 20 40 40 40 40 6 40 40 28 40 40 25 40 20 40 -1 30 40 14 27 40 8 1 3 7 -1 11 10 -1 21 10 8
|
||||
@CSHL__2_FC042AGWWWXX:8:1:103:1185
|
||||
ATCACGATAGATCGGCAGAGCTCGTTTACCGTCTTC
|
||||
+CSHL__2_FC042AGWWWXX:8:1:103:1185
|
||||
40 40 40 40 40 35 33 31 40 40 40 32 30 22 40 -0 9 22 17 14 8 36 15 34 22 12 23 3 10 -0 8 2 4 25 30 2
|
||||
|
||||
|
||||
FASTQ with ASCII quality scores::
|
||||
|
||||
@CSHL__2_FC042AGWWWXX:8:1:120:202
|
||||
ACGATAGATCGGAAGAGCTAGTATGCCGTTTTCTGC
|
||||
+CSHL__2_FC042AGWWWXX:8:1:120:202
|
||||
hhhhThhhhFhh\hhYhTh?^hN[hHACG?KJ?UJH
|
||||
@CSHL__2_FC042AGWWWXX:8:1:103:1185
|
||||
ATCACGATAGATCGGCAGAGCTCGTTTACCGTCTTC
|
||||
+CSHL__2_FC042AGWWWXX:8:1:103:1185
|
||||
hhhhhca_hhh`^Vh@IVQNHdObVLWCJ@HBDY^B
|
||||
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTQ-Quality-Converter is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,100 @@
|
||||
<tool id="cshl_fastq_qual_stat" name="Quality Statistics">
|
||||
<description></description>
|
||||
<command>zcat -f $input | fastq_quality_stats -o $output</command>
|
||||
|
||||
<inputs>
|
||||
<param format="fastqsolexa" name="input" type="data" label="Library to analyse" />
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input" value="fastq_stats1.fastq" />
|
||||
<output name="output" file="fastq_stats1.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="txt" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
Creates quality statistics report for the given Solexa/FASTQ library.
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** This statistics report can be used as input for **Quality Score** and **Nucleotides Distribution** tools.
|
||||
|
||||
-----
|
||||
|
||||
**The output file will contain the following fields:**
|
||||
|
||||
* column = column number (1 to 36 for a 36-cycles read solexa file)
|
||||
* count = number of bases found in this column.
|
||||
* min = Lowest quality score value found in this column.
|
||||
* max = Highest quality score value found in this column.
|
||||
* sum = Sum of quality score values for this column.
|
||||
* mean = Mean quality score value for this column.
|
||||
* Q1 = 1st quartile quality score.
|
||||
* med = Median quality score.
|
||||
* Q3 = 3rd quartile quality score.
|
||||
* IQR = Inter-Quartile range (Q3-Q1).
|
||||
* lW = 'Left-Whisker' value (for boxplotting).
|
||||
* rW = 'Right-Whisker' value (for boxplotting).
|
||||
* A_Count = Count of 'A' nucleotides found in this column.
|
||||
* C_Count = Count of 'C' nucleotides found in this column.
|
||||
* G_Count = Count of 'G' nucleotides found in this column.
|
||||
* T_Count = Count of 'T' nucleotides found in this column.
|
||||
* N_Count = Count of 'N' nucleotides found in this column.
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
**Output Example**::
|
||||
|
||||
column count min max sum mean Q1 med Q3 IQR lW rW A_Count C_Count G_Count T_Count N_Count
|
||||
1 6362991 -4 40 250734117 39.41 40 40 40 0 40 40 1396976 1329101 678730 2958184 0
|
||||
2 6362991 -5 40 250531036 39.37 40 40 40 0 40 40 1786786 1055766 1738025 1782414 0
|
||||
3 6362991 -5 40 248722469 39.09 40 40 40 0 40 40 2296384 984875 1443989 1637743 0
|
||||
4 6362991 -5 40 247654797 38.92 40 40 40 0 40 40 1683197 1410855 1722633 1546306 0
|
||||
5 6362991 -4 40 248214827 39.01 40 40 40 0 40 40 2536861 1167423 1248968 1409739 0
|
||||
6 6362991 -5 40 248499903 39.05 40 40 40 0 40 40 1598956 1236081 1568608 1959346 0
|
||||
7 6362991 -4 40 247719760 38.93 40 40 40 0 40 40 1692667 1822140 1496741 1351443 0
|
||||
8 6362991 -5 40 245745205 38.62 40 40 40 0 40 40 2230936 1343260 1529928 1258867 0
|
||||
9 6362991 -5 40 245766735 38.62 40 40 40 0 40 40 1702064 1306257 1336511 2018159 0
|
||||
10 6362991 -5 40 245089706 38.52 40 40 40 0 40 40 1519917 1446370 1450995 1945709 0
|
||||
11 6362991 -5 40 242641359 38.13 40 40 40 0 40 40 1717434 1282975 1387804 1974778 0
|
||||
12 6362991 -5 40 242026113 38.04 40 40 40 0 40 40 1662872 1202041 1519721 1978357 0
|
||||
13 6362991 -5 40 238704245 37.51 40 40 40 0 40 40 1549965 1271411 1973291 1566681 1643
|
||||
14 6362991 -5 40 235622401 37.03 40 40 40 0 40 40 2101301 1141451 1603990 1515774 475
|
||||
15 6362991 -5 40 230766669 36.27 40 40 40 0 40 40 2344003 1058571 1440466 1519865 86
|
||||
16 6362991 -5 40 224466237 35.28 38 40 40 2 35 40 2203515 1026017 1474060 1651582 7817
|
||||
17 6362991 -5 40 219990002 34.57 34 40 40 6 25 40 1522515 1125455 2159183 1555765 73
|
||||
18 6362991 -5 40 214104778 33.65 30 40 40 10 15 40 1479795 2068113 1558400 1249337 7346
|
||||
19 6362991 -5 40 212934712 33.46 30 40 40 10 15 40 1432749 1231352 1769799 1920093 8998
|
||||
20 6362991 -5 40 212787944 33.44 29 40 40 11 13 40 1311657 1411663 2126316 1513282 73
|
||||
21 6362991 -5 40 211369187 33.22 28 40 40 12 10 40 1887985 1846300 1300326 1318380 10000
|
||||
22 6362991 -5 40 213371720 33.53 30 40 40 10 15 40 542299 3446249 516615 1848190 9638
|
||||
23 6362991 -5 40 221975899 34.89 36 40 40 4 30 40 347679 1233267 926621 3855355 69
|
||||
24 6362991 -5 40 194378421 30.55 21 40 40 19 -5 40 433560 674358 3262764 1992242 67
|
||||
25 6362991 -5 40 199773985 31.40 23 40 40 17 -2 40 944760 325595 1322800 3769641 195
|
||||
26 6362991 -5 40 179404759 28.20 17 34 40 23 -5 40 3457922 156013 1494664 1254293 99
|
||||
27 6362991 -5 40 163386668 25.68 13 28 40 27 -5 40 1392177 281250 3867895 821491 178
|
||||
28 6362991 -5 40 156230534 24.55 12 25 40 28 -5 40 907189 981249 4174945 299437 171
|
||||
29 6362991 -5 40 163236046 25.65 13 28 40 27 -5 40 1097171 3418678 1567013 280008 121
|
||||
30 6362991 -5 40 151309826 23.78 12 23 40 28 -5 40 3514775 2036194 566277 245613 132
|
||||
31 6362991 -5 40 141392520 22.22 10 21 40 30 -5 40 1569000 4571357 124732 97721 181
|
||||
32 6362991 -5 40 143436943 22.54 10 21 40 30 -5 40 1453607 4519441 38176 351107 660
|
||||
33 6362991 -5 40 114269843 17.96 6 14 30 24 -5 40 3311001 2161254 155505 734297 934
|
||||
34 6362991 -5 40 140638447 22.10 10 20 40 30 -5 40 1501615 1637357 18113 3205237 669
|
||||
35 6362991 -5 40 138910532 21.83 10 20 40 30 -5 40 1532519 3495057 23229 1311834 352
|
||||
36 6362991 -5 40 117158566 18.41 7 15 30 23 -5 40 4074444 1402980 63287 822035 245
|
||||
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTQ-Statistics is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,47 @@
|
||||
<tool id="cshl_fastq_quality_boxplot" name="Quality Score">
|
||||
<description>chart</description>
|
||||
|
||||
<command>fastq_quality_boxplot_graph.sh -t '$input.name' -i $input -o $output</command>
|
||||
|
||||
<inputs>
|
||||
<param format="txt" name="input" type="data" label="Statistics report file (output of 'FASTQ Statistics' tool)" />
|
||||
</inputs>
|
||||
|
||||
<outputs>
|
||||
<data format="png" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
Creates a boxplot graph for the quality scores in the library.
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** Use the **FASTQ Statistics** tool to generate the report file needed for this tool.
|
||||
|
||||
-----
|
||||
|
||||
**Output Examples**
|
||||
|
||||
* Black horizontal lines are medians
|
||||
* Rectangular red boxes show the Inter-quartile Range (IQR) (top value is Q3, bottom value is Q1)
|
||||
* Whiskers show outlier at max. 1.5*IQR
|
||||
|
||||
|
||||
An excellent quality library (median quality is 40 for almost all 36 cycles):
|
||||
|
||||
.. image:: ../static/fastx_icons/fastq_quality_boxplot_1.png
|
||||
|
||||
|
||||
A relatively good quality library (median quality degrades towards later cycles):
|
||||
|
||||
.. image:: ../static/fastx_icons/fastq_quality_boxplot_2.png
|
||||
|
||||
A low quality library (median drops quickly):
|
||||
|
||||
.. image:: ../static/fastx_icons/fastq_quality_boxplot_3.png
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTQ-Quality-Boxplot is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,73 @@
|
||||
<tool id="cshl_fastq_quality_filter" name="Quality Filter">
|
||||
<description></description>
|
||||
|
||||
<command>zcat -f '$input' | fastq_quality_filter -q $quality -p $percent -v -o $output</command>
|
||||
|
||||
<inputs>
|
||||
<param format="fastqsolexa" name="input" type="data" label="Library to filter" />
|
||||
|
||||
<param name="quality" size="4" type="integer" value="20">
|
||||
<label>Quality cut-off value</label>
|
||||
</param>
|
||||
|
||||
<param name="percent" size="4" type="integer" value="90">
|
||||
<label>Percent of bases in sequence that must have quality equal to / higher than cut-off value</label>
|
||||
</param>
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- Test1: 100% of bases with quality 33 or higher (pretty steep requirement...) -->
|
||||
<param name="input" value="fastq_qual_filter1.fastq" />
|
||||
<param name="quality" value="33"/>
|
||||
<param name="percent" value="100"/>
|
||||
<output name="output" file="fastq_qual_filter1a.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- Test2: 80% of bases with quality 20 or higher -->
|
||||
<param name="input" value="fastq_qual_filter1.fastq" />
|
||||
<param name="quality" value="20"/>
|
||||
<param name="percent" value="80"/>
|
||||
<output name="output" file="fastq_qual_filter1b.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="input" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
|
||||
<help>
|
||||
**What it does**
|
||||
|
||||
This tool filters reads based on quality scores.
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
Using **percent = 100** requires all cycles of all reads to be at least the quality cut-off value.
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
Using **percent = 50** requires the median quality of the cycles (in each read) to be at least the quality cut-off value.
|
||||
|
||||
--------
|
||||
|
||||
Quality score distribution (of all cycles) is calculated for each read. If it is lower than the quality cut-off value - the read is discarded.
|
||||
|
||||
|
||||
**Example**::
|
||||
|
||||
@CSHL_4_FC042AGOOII:1:2:214:584
|
||||
GACAATAAAC
|
||||
+CSHL_4_FC042AGOOII:1:2:214:584
|
||||
30 30 30 30 30 30 30 30 20 10
|
||||
|
||||
Using **percent = 50** and **cut-off = 30** - This read will not be discarded (the median quality is higher than 30).
|
||||
|
||||
Using **percent = 90** and **cut-off = 30** - This read will be discarded (90% of the cycles do no have quality equal to / higher than 30).
|
||||
|
||||
Using **percent = 100** and **cut-off = 20** - This read will be discarded (not all cycles have quality equal to / higher than 20).
|
||||
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTQ-Quality-Filter is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,70 @@
|
||||
<tool id="cshl_fastq_to_fasta" name="FASTQ to FASTA">
|
||||
<description>converter</description>
|
||||
<command>gunzip -cf $input | fastq_to_fasta $SKIPN $RENAMESEQ -o $output -v </command>
|
||||
|
||||
<inputs>
|
||||
<param format="fastqsolexa" name="input" type="data" label="FASTQ Library to convert" />
|
||||
|
||||
<param name="SKIPN" type="select" label="Discard sequences with unknown (N) bases ">
|
||||
<option value="">yes</option>
|
||||
<option value="-n">no</option>
|
||||
</param>
|
||||
|
||||
<param name="RENAMESEQ" type="select" label="Rename sequence names in output file (reduces file size)">
|
||||
<option value="-r">yes</option>
|
||||
<option value="">no</option>
|
||||
</param>
|
||||
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- FASTQ-To-FASTA, keep N, don't rename -->
|
||||
<param name="input" value="fastq_to_fasta1.fastq" />
|
||||
<param name="SKIPN" value=""/>
|
||||
<param name="RENAMESEQ" value=""/>
|
||||
<output name="output" file="fastq_to_fasta1a.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- FASTQ-To-FASTA, discard N, rename -->
|
||||
<param name="input" value="fastq_to_fasta1.fastq" />
|
||||
<param name="SKIPN" value="no"/>
|
||||
<param name="RENAMESEQ" value="yes"/>
|
||||
<output name="output" file="fastq_to_fasta1b.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="fasta" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool converts data from Solexa format to FASTA format (scroll down for format description).
|
||||
|
||||
--------
|
||||
|
||||
**Example**
|
||||
|
||||
The following data in Solexa-FASTQ format::
|
||||
|
||||
@CSHL_4_FC042GAMMII_2_1_517_596
|
||||
GGTCAATGATGAGTTGGCACTGTAGGCACCATCAAT
|
||||
+CSHL_4_FC042GAMMII_2_1_517_596
|
||||
40 40 40 40 40 40 40 40 40 40 38 40 40 40 40 40 14 40 40 40 40 40 36 40 13 14 24 24 9 24 9 40 10 10 15 40
|
||||
|
||||
Will be converted to FASTA (with 'rename sequence names' = NO)::
|
||||
|
||||
>CSHL_4_FC042GAMMII_2_1_517_596
|
||||
GGTCAATGATGAGTTGGCACTGTAGGCACCATCAAT
|
||||
|
||||
Will be converted to FASTA (with 'rename sequence names' = YES)::
|
||||
|
||||
>1
|
||||
GGTCAATGATGAGTTGGCACTGTAGGCACCATCAAT
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTQ-to-FASTA is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,80 @@
|
||||
<tool id="cshl_fastx_artifacts_filter" name="Artifacts Filter">
|
||||
<description></description>
|
||||
<command>zcat -f '$input' | fastx_artifacts_filter -v -o "$output"</command>
|
||||
|
||||
<inputs>
|
||||
<param format="fasta,fastqsolexa" name="input" type="data" label="Library to filter" />
|
||||
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- Filter FASTA file -->
|
||||
<param name="input" value="fastx_artifacts1.fasta" />
|
||||
<output name="output" file="fastx_artifacts1.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- Filter FASTQ file -->
|
||||
<param name="input" value="fastx_artifacts2.fastq" />
|
||||
<output name="output" file="fastx_artifacts2.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="input" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
<help>
|
||||
**What it does**
|
||||
|
||||
This tool filters sequencing artifacts (reads with all but 3 identical bases).
|
||||
|
||||
--------
|
||||
|
||||
**The following is an example of sequences which will be filtered out**::
|
||||
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAACACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAACACAAAAAAAAAAAAAAAAAAAAAAAAAAAAACACAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
CCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCCC
|
||||
AAAAACACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAACACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAACACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAA
|
||||
AAAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAA
|
||||
AAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAAAAAA
|
||||
AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAACAAAAAAAAAAAA
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTX-Artifacts-filter is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,68 @@
|
||||
<tool id="cshl_fastx_barcode_splitter" name="Barcode Splitter">
|
||||
<description></description>
|
||||
<command>fastx_barcode_splitter_galaxy_wrapper.sh $BARCODE $input "$input.name" --mismatches $mismatches --partial $partial $EOL > $output </command>
|
||||
|
||||
<inputs>
|
||||
<param format="txt" name="BARCODE" type="data" label="Barcodes to use" />
|
||||
<param format="fasta,fastqsolexa" name="input" type="data" label="Library to split" />
|
||||
|
||||
<param name="EOL" type="select" label="Barcodes found at">
|
||||
<option value="--bol">Start of sequence (5' end)</option>
|
||||
<option value="--eol">End of sequence (3' end)</option>
|
||||
</param>
|
||||
|
||||
<param name="mismatches" type="integer" size="3" value="2" label="Number of allowed mismatches" />
|
||||
|
||||
<param name="partial" type="integer" size="3" value="0" label="Number of allowed barcodes nucleotide deletions" />
|
||||
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- Split a FASTQ file -->
|
||||
<param name="BARCODE" value="fastx_barcode_splitter1.txt" />
|
||||
<param name="input" value="fastx_barcode_splitter1.fastq" />
|
||||
<param name="EOL" value="Start of sequence (5' end)" />
|
||||
<param name="mismatches" value="2" />
|
||||
<param name="partial" value="0" />
|
||||
<output name="output" file="fastx_barcode_splitter1.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="html" name="output" />
|
||||
</outputs>
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool splits a solexa library (FASTQ file) or a regular FASTA file to several files, using barcodes as the split criteria.
|
||||
|
||||
--------
|
||||
|
||||
**Barcode file Format**
|
||||
|
||||
Barcode files are simple text files.
|
||||
Each line should contain an identifier (descriptive name for the barcode), and the barcode itself (A/C/G/T), separated by a TAB character.
|
||||
Example::
|
||||
|
||||
#This line is a comment (starts with a 'number' sign)
|
||||
BC1 GATCT
|
||||
BC2 ATCGT
|
||||
BC3 GTGAT
|
||||
BC4 TGTCT
|
||||
|
||||
For each barcode, a new FASTQ file will be created (with the barcode's identifier as part of the file name).
|
||||
Sequences matching the barcode will be stored in the appropriate file.
|
||||
|
||||
One additional FASTQ file will be created (the 'unmatched' file), where sequences not matching any barcode will be stored.
|
||||
|
||||
The output of this tool is an HTML file, displaying the split counts and the file locations.
|
||||
|
||||
**Output Example**
|
||||
|
||||
.. image:: ../static/fastx_icons/barcode_splitter_output_example.png
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTX-barcode-splitter is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,109 @@
|
||||
<tool id="cshl_fastx_clipper" name="Clip" version="1.0.1" >
|
||||
<description>adapter sequences</description>
|
||||
<command>
|
||||
zcat -f $input | fastx_clipper -s $maxmismatches -l $minlength -a $clip_source.clip_sequence -d $keepdelta -o $output -v $KEEP_N $DISCARD_OPTIONS
|
||||
</command>
|
||||
|
||||
<inputs>
|
||||
<param format="fasta,fastqsolexa" name="input" type="data" label="Library to clip" />
|
||||
|
||||
<param name="maxmismatches" size="4" type="integer" value="2">
|
||||
<label>Maximum number of mismatches allowed (when matching the adapter sequence)</label>
|
||||
</param>
|
||||
|
||||
<param name="minlength" size="4" type="integer" value="15">
|
||||
<label>Minimum sequence length (after clipping, sequences shorter than this length will be discarded)</label>
|
||||
</param>
|
||||
|
||||
<conditional name="clip_source">
|
||||
<param name="clip_source_list" type="select" label="Source">
|
||||
<option value="prebuilt" selected="true">Standard (select from the list below)</option>
|
||||
<option value="user">Enter custom sequence</option>
|
||||
</param>
|
||||
|
||||
<when value="user">
|
||||
<param name="clip_sequence" size="30" label="Enter custom clipping sequence" type="text" value="AATTGGCC" />
|
||||
</when>
|
||||
|
||||
<when value="prebuilt">
|
||||
<param name="clip_sequence" type="select" label="Choose Adapter">
|
||||
<options from_file="fastx_clipper_sequences.txt">
|
||||
<column name="name" index="1"/>
|
||||
<column name="value" index="0"/>
|
||||
</options>
|
||||
</param>
|
||||
</when>
|
||||
</conditional>
|
||||
|
||||
<param name="keepdelta" size="2" type="integer" value="0">
|
||||
<label>enter non-zero value to keep the adapter sequence and x bases that follow it</label>
|
||||
<help>use this for hairpin barcoding. keep at 0 unless you know what you're doing.</help>
|
||||
</param>
|
||||
|
||||
<param name="KEEP_N" type="select" label="Discard sequences with unknown (N) bases">
|
||||
<option value="">Yes</option>
|
||||
<option value="-n">No</option>
|
||||
</param>
|
||||
|
||||
<param name="DISCARD_OPTIONS" type="select" label="Output options">
|
||||
<option value="-c">Output only clipped seqeunces (i.e. sequences which contained the adapter)</option>
|
||||
<option value="-C">Output only non-clipped seqeunces (i.e. sequences which did not contained the adapter)</option>
|
||||
<option value="">Output both clipped and non-clipped sequences</option>
|
||||
</param>
|
||||
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- Clip a FASTQ file -->
|
||||
<param name="input" value="fastx_clipper1.fastq" />
|
||||
<param name="maxmismatches" value="2" />
|
||||
<param name="minlength" value="15" />
|
||||
<param name="clip_source.clip_source_list" value="user" />
|
||||
<param name="clip_source.clip_sequence" value="CAATTGGTTAATCCCCCTATATA" />
|
||||
<param name="keepdelta" value="0" />
|
||||
<param name="KEEP_N" value="-n" />
|
||||
<param name="DISCARD_OPTIONS" value="-c" />
|
||||
<output name="output" file="fastx_clipper1a.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="input" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
|
||||
<help>
|
||||
**What it does**
|
||||
|
||||
This tool clips adapters from the 3'-end of the sequences in a FASTA/FASTQ file.
|
||||
|
||||
--------
|
||||
|
||||
|
||||
**Clipping Illustration:**
|
||||
|
||||
.. image:: ../static/fastx_icons/fastx_clipper_illustration.png
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
**Clipping Example:**
|
||||
|
||||
.. image:: ../static/fastx_icons/fastx_clipper_example.png
|
||||
|
||||
|
||||
|
||||
**In the above example:**
|
||||
|
||||
* Sequence no. 1 was discarded since it wasn't clipped (i.e. didn't contain the adapter sequence). (**Output** parameter).
|
||||
* Sequence no. 5 was discarded --- it's length (after clipping) was shorter than 15 nt (**Minimum Sequence Length** parameter).
|
||||
|
||||
|
||||
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
@@ -0,0 +1,52 @@
|
||||
<tool id="cshl_fastx_reverse_complement" name="Reverse-Complement">
|
||||
<description>sequences</description>
|
||||
<command>zcat -f '$input' | fastx_reverse_complement -v -o $output</command>
|
||||
<inputs>
|
||||
<param format="fasta,fastqsolexa" name="input" type="data" label="Library to reverse-complement" />
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- Reverse-complement a FASTA file -->
|
||||
<param name="input" value="fastx_rev_comp1.fasta" />
|
||||
<output name="output" file="fastx_reverse_complement1.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- Reverse-complement a FASTQ file -->
|
||||
<param name="input" value="fastx_rev_comp2.fastq" />
|
||||
<output name="output" file="fastx_reverse_complement2.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
|
||||
<outputs>
|
||||
<data format="input" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
|
||||
<help>
|
||||
**What it does**
|
||||
|
||||
This tool reverse-complements each sequence in a library.
|
||||
If the library is a FASTQ, the quality-scores are also reversed.
|
||||
|
||||
--------
|
||||
|
||||
**Example**
|
||||
|
||||
Input FASTQ file::
|
||||
|
||||
@CSHL_1_FC42AGWWWXX:8:1:3:740
|
||||
TGTCTGTAGCCTCNTCCTTGTAATTCAAAGNNGGTA
|
||||
+CSHL_1_FC42AGWWWXX:8:1:3:740
|
||||
33 33 33 34 33 33 33 33 33 33 33 33 27 5 27 33 33 33 33 33 33 27 21 27 33 32 31 29 26 24 5 5 15 17 27 26
|
||||
|
||||
|
||||
Output FASTQ file::
|
||||
|
||||
@CSHL_1_FC42AGWWWXX:8:1:3:740
|
||||
TACCNNCTTTGAATTACAAGGANGAGGCTACAGACA
|
||||
+CSHL_1_FC42AGWWWXX:8:1:3:740
|
||||
26 27 17 15 5 5 24 26 29 31 32 33 27 21 27 33 33 33 33 33 33 27 5 27 33 33 33 33 33 33 33 33 34 33 33 33
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
@@ -0,0 +1,70 @@
|
||||
<tool id="cshl_fastx_trimmer" name="Trim">
|
||||
<description>sequences</description>
|
||||
<command>zcat -f '$input' | fastx_trimmer -v -f $first -l $last -o $output</command>
|
||||
|
||||
<inputs>
|
||||
<param format="fasta,fastqsolexa" name="input" type="data" label="Library to clip" />
|
||||
|
||||
<param name="first" size="4" type="integer" value="1">
|
||||
<label>First base to keep</label>
|
||||
</param>
|
||||
|
||||
<param name="last" size="4" type="integer" value="21">
|
||||
<label>Last base to keep</label>
|
||||
</param>
|
||||
</inputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<!-- Trim a FASTA file - remove first four bases (e.g. a barcode) -->
|
||||
<param name="input" value="fastx_trimmer1.fasta" />
|
||||
<param name="first" value="5"/>
|
||||
<param name="last" value="36"/>
|
||||
<output name="output" file="fastx_trimmer1.out" />
|
||||
</test>
|
||||
<test>
|
||||
<!-- Trim a FASTQ file - remove last 9 bases (e.g. keep only miRNA length sequences) -->
|
||||
<param name="input" value="fastx_trimmer2.fastq" />
|
||||
<param name="first" value="1"/>
|
||||
<param name="last" value="27"/>
|
||||
<output name="output" file="fastx_trimmer2.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<outputs>
|
||||
<data format="input" name="output" metadata_source="input" />
|
||||
</outputs>
|
||||
<help>
|
||||
**What it does**
|
||||
|
||||
This tool trims (cut bases from) sequences in a FASTA/Q file.
|
||||
|
||||
--------
|
||||
|
||||
**Example**
|
||||
|
||||
Input Fasta file (with 36 bases in each sequences)::
|
||||
|
||||
>1-1
|
||||
TATGGTCAGAAACCATATGCAGAGCCTGTAGGCACC
|
||||
>2-1
|
||||
CAGCGAGGCTTTAATGCCATTTGGCTGTAGGCACCA
|
||||
|
||||
|
||||
Trimming with First=1 and Last=21, we get a FASTA file with 21 bases in each sequences (starting from the first base)::
|
||||
|
||||
>1-1
|
||||
TATGGTCAGAAACCATATGCA
|
||||
>2-1
|
||||
CAGCGAGGCTTTAATGCCATT
|
||||
|
||||
Trimming with First=6 and Last=10, will generate a FASTA file with 5 bases (bases 6,7,8,9,10) in each sequences::
|
||||
|
||||
>1-1
|
||||
TCAGA
|
||||
>2-1
|
||||
AGGCT
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
<!-- FASTX-Trimmer is part of the FASTX-toolkit, by A.Gordon (gordon@cshl.edu) -->
|
||||
@@ -0,0 +1,37 @@
|
||||
import sys, re
|
||||
|
||||
def stop_err( msg ):
|
||||
sys.stderr.write( msg )
|
||||
sys.exit()
|
||||
|
||||
def __main__():
|
||||
try:
|
||||
infile = open ( sys.argv[1], 'r')
|
||||
outfile = open ( sys.argv[2], 'w')
|
||||
except:
|
||||
stop_err( 'Cannot open or create a file\n' )
|
||||
|
||||
if len( sys.argv ) < 4:
|
||||
stop_err( 'No columns to merge' )
|
||||
else:
|
||||
cols = sys.argv[3:]
|
||||
|
||||
skipped_lines = 0
|
||||
|
||||
for line in infile:
|
||||
line = line.rstrip( '\r\n' )
|
||||
if line and not line.startswith( '#' ):
|
||||
fields = line.split( '\t' )
|
||||
line += '\t'
|
||||
for col in cols:
|
||||
try:
|
||||
line += fields[ int( col ) -1 ]
|
||||
except:
|
||||
skipped_lines += 1
|
||||
|
||||
print >>outfile, line
|
||||
|
||||
if skipped_lines > 0:
|
||||
print 'Skipped %d invalid lines' % skipped_lines
|
||||
|
||||
if __name__ == "__main__" : __main__()
|
||||
@@ -0,0 +1,63 @@
|
||||
<tool id="mergeCols1" name="Merge Columns" version="1.0.1">
|
||||
<description>together</description>
|
||||
<command interpreter="python">
|
||||
mergeCols.py
|
||||
$input1
|
||||
$out_file1
|
||||
$col1
|
||||
$col2
|
||||
#for $col in $columns
|
||||
${col.datacol}
|
||||
#end for
|
||||
</command>
|
||||
<inputs>
|
||||
<param format="tabular" name="input1" type="data" label="Select data" help="Query missing? See TIP below."/>
|
||||
<param name="col1" label="Merge column" type="data_column" data_ref="input1" />
|
||||
<param name="col2" label="with column" type="data_column" data_ref="input1" help="Need to add more columns? Use controls below."/>
|
||||
<repeat name="columns" title="Columns">
|
||||
<param name="datacol" label="Add column" type="data_column" data_ref="input1" />
|
||||
</repeat>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="tabular" name="out_file1" />
|
||||
</outputs>
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input1" value="1.bed"/>
|
||||
<param name="col1" value="4" />
|
||||
<param name="col2" value="1" />
|
||||
<param name="datacol" value="6" />
|
||||
<output name="out_file1" file="mergeCols.dat"/>
|
||||
</test>
|
||||
</tests>
|
||||
<help>
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
**TIP:** If your data is not TAB delimited, use *Text Manipulation->Convert*
|
||||
|
||||
-----
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool merges columns together. Any number of valid columns can be merged in any order.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
Input dataset (five columns: c1, c2, c3, c4, and c5)::
|
||||
|
||||
1 10 1000 gene1 chr
|
||||
2 100 1500 gene2 chr
|
||||
|
||||
merging columns "**c5,c1**" will return::
|
||||
|
||||
1 10 1000 gene1 chr chr1
|
||||
2 100 1500 gene2 chr chr2
|
||||
|
||||
.. class:: warningmark
|
||||
|
||||
Note that all original columns are preserved and the result of merge is added as the rightmost column.
|
||||
</help>
|
||||
</tool>
|
||||
@@ -48,10 +48,7 @@ def __main__():
|
||||
print >>sys.stderr, "You must specify a proper build in order to extract alignments. You can specify your genome build by clicking on the pencil icon associated with your interval file."
|
||||
sys.exit()
|
||||
|
||||
species = None
|
||||
if options.species:
|
||||
species = options.species.split( ',' )
|
||||
if "None" in species: species = None
|
||||
species = maf_utilities.parse_species_option( options.species )
|
||||
|
||||
if options.chromCol: chromCol = int( options.chromCol ) - 1
|
||||
else:
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
<tool id="Interval2Maf1" name="Extract MAF blocks">
|
||||
<tool id="Interval2Maf1" name="Extract MAF blocks" version="1.0.1">
|
||||
<description>given a set of genomic intervals</description>
|
||||
<command interpreter="python">
|
||||
#if $maf_source_type.maf_source == "user":#interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafFile=$maf_source_type.mafFile --mafIndex=$maf_source_type.mafFile.metadata.maf_index --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc
|
||||
#else:#interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafType=$maf_source_type.mafType --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc
|
||||
#if $maf_source_type.maf_source == "user":#interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafFile=$maf_source_type.mafFile --mafIndex=$maf_source_type.mafFile.metadata.maf_index --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc --species=$maf_source_type.species
|
||||
#else:#interval2maf.py --dbkey=${input1.dbkey} --chromCol=${input1.metadata.chromCol} --startCol=${input1.metadata.startCol} --endCol=${input1.metadata.endCol} --strandCol=${input1.metadata.strandCol} --mafType=$maf_source_type.mafType --interval_file=$input1 --output_file=$out_file1 --mafIndexFile=${GALAXY_DATA_INDEX_DIR}/maf_index.loc --species=$maf_source_type.species
|
||||
#end if
|
||||
</command>
|
||||
<inputs>
|
||||
@@ -21,6 +21,11 @@
|
||||
</options>
|
||||
<validator type="dataset_ok_validator" />
|
||||
</param>
|
||||
<param name="species" type="select" display="checkboxes" multiple="true" label="Choose species" help="Select species to be included in the final alignment">
|
||||
<options>
|
||||
<filter type="data_meta" ref="mafFile" key="species" />
|
||||
</options>
|
||||
</param>
|
||||
</when>
|
||||
<when value="cached">
|
||||
<param name="mafType" type="select" label="Choose alignments">
|
||||
@@ -32,7 +37,16 @@
|
||||
<filter type="data_meta" ref="input1" key="dbkey" column="2" multiple="True" separator=","/>
|
||||
<validator type="no_options" message="No alignments are available for the build associated with the selected interval file"/>
|
||||
</options>
|
||||
</param>
|
||||
</param>
|
||||
<param name="species" type="select" display="checkboxes" multiple="true" label="Choose species" help="Select species to be included in the final alignment">
|
||||
<options from_file="maf_index.loc">
|
||||
<column name="uid" index="1"/>
|
||||
<column name="value" index="3"/>
|
||||
<column name="name" index="3"/>
|
||||
<filter type="param_value" ref="mafType" name="uid" column="1"/>
|
||||
<filter type="multiple_splitter" column="3" separator=","/>
|
||||
</options>
|
||||
</param>
|
||||
</when>
|
||||
</conditional>
|
||||
</inputs>
|
||||
@@ -44,12 +58,14 @@
|
||||
<param name="input1" value="1.bed"/>
|
||||
<param name="maf_source" value="cached"/>
|
||||
<param name="mafType" value="ENCODE_TBA_hg17"/>
|
||||
<param name="species" value="hg17,panTro1,baboon,marmoset,galago,rn3,mm6,rabbit,cow,canFam1,rfbat,shrew,armadillo,tenrec,monDom1,tetNig1,fr1,rheMac1,galGal2,xenTro1,danRer2,elephant,platypus,hedgehog,colobus_monkey,dusky_titi,owl_monkey,mouse_lemur"/>
|
||||
<output name="out_file1" file="fsa_interval2maf.dat" />
|
||||
</test>
|
||||
<test>
|
||||
<param name="input1" value="1.bed"/>
|
||||
<param name="maf_source" value="user"/>
|
||||
<param name="mafFile" value="fsa_interval2maf.dat"/>
|
||||
<param name="species" value="hg17,panTro1,baboon,marmoset,galago,rn3,mm6,rabbit,cow,canFam1,rfbat,shrew,armadillo,tenrec,monDom1,tetNig1,fr1,rheMac1,galGal2,xenTro1,danRer2,elephant,platypus,hedgehog,colobus_monkey,dusky_titi,owl_monkey,mouse_lemur"/>
|
||||
<output name="out_file1" file="fsa_interval2maf.dat" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
@@ -44,7 +44,7 @@ def __main__():
|
||||
options, args = doc_optparse.parse( __doc__ )
|
||||
mincols = 0
|
||||
strand_col = -1
|
||||
|
||||
|
||||
if options.dbkey:
|
||||
primary_species = options.dbkey
|
||||
else:
|
||||
@@ -53,17 +53,15 @@ def __main__():
|
||||
stop_err( "You must specify a proper build in order to extract alignments. You can specify your genome build by clicking on the pencil icon associated with your interval file." )
|
||||
|
||||
include_primary = True
|
||||
if options.species:
|
||||
secondary_species = options.species.split( ',' )
|
||||
if "None" in secondary_species:
|
||||
secondary_species = None
|
||||
species = None
|
||||
secondary_species = maf_utilities.parse_species_option( options.species )
|
||||
if secondary_species:
|
||||
species = list( secondary_species ) # make copy of species list
|
||||
if primary_species in secondary_species:
|
||||
secondary_species.remove( primary_species )
|
||||
else:
|
||||
try:
|
||||
secondary_species.remove( primary_species )
|
||||
except:
|
||||
include_primary = False
|
||||
species = [primary_species] + secondary_species
|
||||
include_primary = False
|
||||
else:
|
||||
species = None
|
||||
|
||||
if options.interval_file:
|
||||
interval_file = options.interval_file
|
||||
@@ -90,10 +88,10 @@ def __main__():
|
||||
end_col = int( options.endCol ) - 1
|
||||
else:
|
||||
stop_err( "End column not set, click the pencil icon in the history item to set the metadata attributes." )
|
||||
|
||||
|
||||
if options.strandCol:
|
||||
strand_col = int( options.strandCol ) - 1
|
||||
|
||||
|
||||
mafIndexFile = "%s/maf_index.loc" % options.mafIndexFileDir
|
||||
#Finish parsing command line
|
||||
|
||||
|
||||
@@ -8,12 +8,12 @@ blocks specified by number.
|
||||
import sys
|
||||
from galaxy import eggs
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
from galaxy.tools.util import maf_utilities
|
||||
import bx.align.maf
|
||||
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
|
||||
def __main__():
|
||||
|
||||
input_block_filename = sys.argv[1].strip()
|
||||
input_maf_filename = sys.argv[2].strip()
|
||||
output_filename1 = sys.argv[3].strip()
|
||||
@@ -21,6 +21,7 @@ def __main__():
|
||||
if block_col < 0:
|
||||
print >> sys.stderr, "Invalid column specified"
|
||||
sys.exit(0)
|
||||
species = maf_utilities.parse_species_option( sys.argv[5].strip() )
|
||||
|
||||
maf_writer = bx.align.maf.Writer( open( output_filename1, 'w' ) )
|
||||
#we want to maintain order of block file and write blocks as many times as they are listed
|
||||
@@ -34,6 +35,8 @@ def __main__():
|
||||
try:
|
||||
for count, block in enumerate( bx.align.maf.Reader( open( input_maf_filename, 'r' ) ) ):
|
||||
if count == block_wanted:
|
||||
if species:
|
||||
block = block.limit_to_species( species )
|
||||
maf_writer.write( block )
|
||||
break
|
||||
except:
|
||||
|
||||
@@ -1,10 +1,15 @@
|
||||
<tool id="maf_by_block_number1" name="Extract MAF by block number">
|
||||
<tool id="maf_by_block_number1" name="Extract MAF by block number" version="1.0.1">
|
||||
<description>given a set of block numbers and a MAF file</description>
|
||||
<command interpreter="python">maf_by_block_number.py $input1 $input2 $out_file1 $block_col</command>
|
||||
<command interpreter="python">maf_by_block_number.py $input1 $input2 $out_file1 $block_col $species</command>
|
||||
<inputs>
|
||||
<param format="txt" name="input1" type="data" label="Block Numbers"/>
|
||||
<param format="maf" name="input2" label="MAF File" type="data"/>
|
||||
<param name="block_col" type="data_column" label="Column containing Block number" data_ref="input1" accept_default="True" />
|
||||
<param name="species" type="select" display="checkboxes" multiple="true" label="Choose species" help="Select species to be included in the final alignment">
|
||||
<options>
|
||||
<filter type="data_meta" ref="input2" key="species" />
|
||||
</options>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="maf" name="out_file1" />
|
||||
@@ -14,6 +19,7 @@
|
||||
<param name="input1" value="maf_by_block_numbers.dat"/>
|
||||
<param name="input2" value="3.maf"/>
|
||||
<param name="block_col" value="1"/>
|
||||
<param name="species" value="hg17,panTro1,mm5,rn3,canFam1"/>
|
||||
<output name="out_file1" file="maf_by_block_number_out.dat" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
+65
-48
@@ -1,48 +1,65 @@
|
||||
#Dan Blankenberg
|
||||
#Filters a MAF file according to the provided code file, which is generated in maf_filter.xml <configfiles>
|
||||
import sys, os, shutil
|
||||
from galaxy import eggs
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
import bx.align.maf
|
||||
|
||||
def main():
|
||||
#Read command line arguments
|
||||
try:
|
||||
script_file = sys.argv.pop( 1 )
|
||||
maf_file = sys.argv.pop( 1 )
|
||||
out_file = sys.argv.pop( 1 )
|
||||
additional_files_path = sys.argv.pop( 1 )
|
||||
except:
|
||||
print >>sys.stderr, "One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug"
|
||||
sys.exit()
|
||||
|
||||
#Open input and output MAF files
|
||||
try:
|
||||
maf_reader = bx.align.maf.Reader( open( maf_file,'r' ) )
|
||||
maf_writer = bx.align.maf.Writer( open( out_file,'w' ) )
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
sys.exit()
|
||||
|
||||
#Save script file for debuging/verification info later
|
||||
os.mkdir( additional_files_path )
|
||||
shutil.copy( script_file, os.path.join( additional_files_path, 'debug.txt' ) )
|
||||
|
||||
#Loop through blocks, running filter on each
|
||||
#'maf_block' and 'ret_val' are used/shared in the provided code file
|
||||
#'ret_val' should be set to True if the block is to be kept
|
||||
i = 0
|
||||
blocks_kept = 0
|
||||
for i, maf_block in enumerate( maf_reader ):
|
||||
local = {'maf_block':maf_block, 'ret_val':False}
|
||||
execfile( script_file, {}, local )
|
||||
if local['ret_val']:
|
||||
maf_writer.write( maf_block )
|
||||
blocks_kept += 1
|
||||
maf_writer.close()
|
||||
maf_reader.close()
|
||||
if i == 0: print "Your file contains no valid maf_blocks."
|
||||
else: print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
#Dan Blankenberg
|
||||
#Filters a MAF file according to the provided code file, which is generated in maf_filter.xml <configfiles>
|
||||
#Also allows filtering by number of columns in a block, and limiting output species
|
||||
import sys, os, shutil
|
||||
from galaxy import eggs
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
import bx.align.maf
|
||||
from galaxy.tools.util import maf_utilities
|
||||
|
||||
def main():
|
||||
#Read command line arguments
|
||||
try:
|
||||
script_file = sys.argv.pop( 1 )
|
||||
maf_file = sys.argv.pop( 1 )
|
||||
out_file = sys.argv.pop( 1 )
|
||||
additional_files_path = sys.argv.pop( 1 )
|
||||
species = maf_utilities.parse_species_option( sys.argv.pop( 1 ) )
|
||||
min_size = int( sys.argv.pop( 1 ) )
|
||||
max_size = int( sys.argv.pop( 1 ) )
|
||||
if max_size < 1: max_size = sys.maxint
|
||||
min_species_per_block = int( sys.argv.pop( 1 ) )
|
||||
exclude_incomplete_blocks = int( sys.argv.pop( 1 ) )
|
||||
if species:
|
||||
num_species = len( species )
|
||||
else:
|
||||
num_species = len( sys.argv.pop( 1 ).split( ',') )
|
||||
except:
|
||||
print >>sys.stderr, "One or more arguments is missing.\nUsage: maf_filter.py maf_filter_file input_maf output_maf path_to_save_debug species_to_keep"
|
||||
sys.exit()
|
||||
|
||||
#Open input and output MAF files
|
||||
try:
|
||||
maf_reader = bx.align.maf.Reader( open( maf_file,'r' ) )
|
||||
maf_writer = bx.align.maf.Writer( open( out_file,'w' ) )
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
sys.exit()
|
||||
|
||||
#Save script file for debuging/verification info later
|
||||
os.mkdir( additional_files_path )
|
||||
shutil.copy( script_file, os.path.join( additional_files_path, 'debug.txt' ) )
|
||||
|
||||
#Loop through blocks, running filter on each
|
||||
#'maf_block' and 'ret_val' are used/shared in the provided code file
|
||||
#'ret_val' should be set to True if the block is to be kept
|
||||
i = 0
|
||||
blocks_kept = 0
|
||||
for i, maf_block in enumerate( maf_reader ):
|
||||
if min_size <= maf_block.components[0].size <= max_size:
|
||||
local = {'maf_block':maf_block, 'ret_val':False}
|
||||
execfile( script_file, {}, local )
|
||||
if local['ret_val']:
|
||||
#Species limiting must be done after filters as filters could be run on non-requested output species
|
||||
if species:
|
||||
maf_block = maf_block.limit_to_species( species )
|
||||
if len( maf_block.components ) >= min_species_per_block and ( not exclude_incomplete_blocks or len( maf_block.components ) >= num_species ):
|
||||
maf_writer.write( maf_block )
|
||||
blocks_kept += 1
|
||||
maf_writer.close()
|
||||
maf_reader.close()
|
||||
if i == 0: print "Your file contains no valid maf_blocks."
|
||||
else: print 'Kept %s of %s blocks (%.2f%%).' % ( blocks_kept, i + 1, float( blocks_kept ) / float( i + 1 ) * 100.0 )
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -1,9 +1,24 @@
|
||||
<tool id="MAF_filter" name="Filter MAF">
|
||||
<tool id="MAF_filter" name="Filter MAF" version="1.0.1">
|
||||
<description>by specified attributes</description>
|
||||
<command interpreter="python">maf_filter.py $maf_filter_file $input1 $out_file1 $out_file1.files_path</command>
|
||||
<command interpreter="python">maf_filter.py $maf_filter_file $input1 $out_file1 $out_file1.files_path $species $min_size $max_size $min_species_per_block $exclude_incomplete_blocks ${input1.metadata.species}</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param name="input1" type="data" format="maf" label="MAF File"/>
|
||||
<param name="min_size" label="Minimum Size" value="0" type="integer"/>
|
||||
<param name="max_size" label="Maximum Size" value="0" type="integer" help="A maximum size less than 1 indicates no limit"/>
|
||||
<param name="species" type="select" display="checkboxes" multiple="true" label="Choose species" help="Select species to be included in the final alignment">
|
||||
<options>
|
||||
<filter type="data_meta" ref="input1" key="species" />
|
||||
</options>
|
||||
</param>
|
||||
<param name="min_species_per_block" type="select" label="Exclude blocks which have only one species" >
|
||||
<option value="2">Yes</option>
|
||||
<option value="1" selected="True">No</option>
|
||||
</param>
|
||||
<param name="exclude_incomplete_blocks" type="select" label="Exclude blocks which have missing species" >
|
||||
<option value="1">Yes</option>
|
||||
<option value="0" selected="True">No</option>
|
||||
</param>
|
||||
<repeat name="maf_filters" title="Filter">
|
||||
<param name="species1" type="select" label="When Species" multiple="false">
|
||||
<options>
|
||||
@@ -131,6 +146,26 @@ ret_val = maf_block_pass_filter( maf_block )
|
||||
<outputs>
|
||||
<data format="maf" name="out_file1" />
|
||||
</outputs>
|
||||
<!--
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input1" value="4.maf"/>
|
||||
<param name="species" value="bosTau2,canFam2,hg17,panTro1,rheMac2,rn3"/>
|
||||
<param name="exclude_incomplete_blocks" value="0"/>
|
||||
<param name="min_species_per_block" value="1"/>
|
||||
<param name="min_size" value="0"/>
|
||||
<param name="max_size" value="0"/>
|
||||
<param name="species1" value="hg17"/>
|
||||
<param name="species2" value="hg17"/>
|
||||
<param name="species1_attribute_type" value="attribute_chr"/>
|
||||
<param name="species1_is_isnot" value="in"/>
|
||||
<param name="species1_attribute" value="chr1"/>
|
||||
<param name="filter_condition"/> Test will ERROR when this is set or when it is not set.
|
||||
|
||||
<output name="out_file1" file="cf_maf_limit_to_species.dat"/>
|
||||
</test>
|
||||
</tests>
|
||||
-->
|
||||
<help>
|
||||
This tool allows you to build complex filters to be applied to each alignment block of a MAF file. You can define restraints on species based upon chromosome and strand. You can specify comma separated lists of chromosomes where appropriate.
|
||||
|
||||
@@ -144,5 +179,17 @@ For example, this tool is useful to restrict a set of alignments to only those b
|
||||
|
||||
If a species is not found in a particular block, all filters on that species are ignored.
|
||||
|
||||
-----
|
||||
|
||||
This tool allows the user to remove any undesired species from a MAF file. If no species are specified then all species will be kept. If species are specified, columns which contain only gaps are removed. The options for this are:
|
||||
|
||||
* **Exclude blocks which have missing species** - suppose you want to restrict an 8-way alignment to human, mouse, and rat. The tool will first remove all other species. Next, if this option is set to **YES** the tool WILL NOT return MAF blocks, which do not include human, mouse, or rat. This means that all alignment blocks returned by the tool will have exactly three sequences in this example.
|
||||
|
||||
* **Exclude blocks which have only one species** - if this option is set to **YES** all single sequence alignment blocks WILL NOT be returned.
|
||||
|
||||
-----
|
||||
|
||||
You can also provide a size range and limit your output to the MAF blocks which fall within the specified range.
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
|
||||
@@ -10,6 +10,7 @@ usage: %prog input_maf_file output_maf_file
|
||||
from galaxy import eggs
|
||||
import pkg_resources; pkg_resources.require( "bx-python" )
|
||||
import bx.align.maf
|
||||
from galaxy.tools.util import maf_utilities
|
||||
import sys
|
||||
|
||||
assert sys.version_info[:2] >= ( 2, 4 )
|
||||
@@ -18,6 +19,7 @@ def __main__():
|
||||
#Parse Command Line
|
||||
input_file = sys.argv.pop( 1 )
|
||||
output_file = sys.argv.pop( 1 )
|
||||
species = maf_utilities.parse_species_option( sys.argv.pop( 1 ) )
|
||||
|
||||
try:
|
||||
maf_writer = bx.align.maf.Writer( open( output_file, 'w' ) )
|
||||
@@ -27,7 +29,10 @@ def __main__():
|
||||
try:
|
||||
count = 0
|
||||
for count, maf in enumerate( bx.align.maf.Reader( open( input_file ) ) ):
|
||||
maf_writer.write( maf.reverse_complement() )
|
||||
maf = maf.reverse_complement()
|
||||
if species:
|
||||
maf = maf.limit_to_species( species )
|
||||
maf_writer.write( maf )
|
||||
except:
|
||||
print >>sys.stderr, "Your MAF file appears to be malformed."
|
||||
sys.exit()
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
<tool id="MAF_Reverse_Complement_1" name="Reverse Compliment">
|
||||
<tool id="MAF_Reverse_Complement_1" name="Reverse Compliment" version="1.0.1">
|
||||
<description>a MAF file</description>
|
||||
<command interpreter="python">maf_reverse_complement.py $input1 $out_file1</command>
|
||||
<command interpreter="python">maf_reverse_complement.py $input1 $out_file1 $species</command>
|
||||
<inputs>
|
||||
<page>
|
||||
<param format="maf" name="input1" label="Alignment File" type="data"/>
|
||||
<param name="species" type="select" display="checkboxes" multiple="true" label="Choose species" help="Select species to be included in the final alignment">
|
||||
<options>
|
||||
<filter type="data_meta" ref="input1" key="species" />
|
||||
</options>
|
||||
</param>
|
||||
</page>
|
||||
</inputs>
|
||||
<outputs>
|
||||
@@ -12,6 +17,7 @@
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input1" value="3.maf" dbkey="hg17" format="maf"/>
|
||||
<param name="species" value="hg17,panTro1,mm5,rn3,canFam1"/>
|
||||
<output name="out_file1" file="maf_reverse_complement_out.dat"/>
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
@@ -27,13 +27,13 @@ def exec_after_process(app, inp_data, out_data, param_dict, tool, stdout, stderr
|
||||
fields = line.split("\t")
|
||||
dbkey = fields[1]
|
||||
filepath = fields[2]
|
||||
|
||||
newdata = app.model.HistoryDatasetAssociation( create_dataset = True )
|
||||
newdata.set_size()
|
||||
newdata.extension = "bed"
|
||||
newdata.name = basic_name + " (" + dbkey + ")"
|
||||
newdata.flush()
|
||||
history.add_dataset( newdata )
|
||||
app.security_agent.copy_dataset_permissions( output_data.dataset, newdata.dataset )
|
||||
newdata.flush()
|
||||
history.flush()
|
||||
app.model.flush()
|
||||
|
||||
@@ -14,7 +14,7 @@ a generic histogram builder based on gnuplot backend
|
||||
yrange_max - maximal value at the y_axis (integer)
|
||||
to set yrange to autoscaling assign 0 to yrange_min and yrange_max
|
||||
graph_file - file to write histogram image to
|
||||
pdf_size - as X,Y pair in inches (e.g., 11,8 or 8,11 etc.)
|
||||
img_size - as X,Y pair in pixels (e.g., 800,600 or 600,800 etc.)
|
||||
|
||||
|
||||
This tool required gnuplot and gnuplot.py
|
||||
@@ -48,7 +48,7 @@ def main(tmpFileName):
|
||||
ymin = sys.argv[6]
|
||||
ymax = sys.argv[7]
|
||||
img_file = sys.argv[8]
|
||||
pdf_size = sys.argv[9]
|
||||
img_size = sys.argv[9]
|
||||
except:
|
||||
stop_err("Check arguments\n")
|
||||
|
||||
@@ -119,7 +119,7 @@ def main(tmpFileName):
|
||||
if xtic == 0: g('unset xtics')
|
||||
g(title)
|
||||
g(ylabel)
|
||||
g_term = 'set terminal pdf size ' + pdf_size
|
||||
g_term = 'set terminal png tiny size ' + img_size
|
||||
g(g_term)
|
||||
g_out = 'set output "' + img_file + '"'
|
||||
if ymin != ymax:
|
||||
|
||||
@@ -23,13 +23,13 @@
|
||||
<param name="ylabel" type="text" size="30" value="V1" label="Label for Y axis"/>
|
||||
<param name="ymin" type="integer" size="4" value="0" label="Minimal value on Y axis" help="set to 0 for autoscaling"/>
|
||||
<param name="ymax" type="integer" size="4" value="0" label="Maximal value on Y axis" help="set to 0 for autoscaling"/>
|
||||
<param name="pdf_size" type="select" label="Choose chart size (inches)" help="inch = 2.54 cm">
|
||||
<option value="11,8">Normal: 11 by 8</option>
|
||||
<option value="5,3">Small: 5 by 3</option>
|
||||
<option value="17,11">Large: 17 by 11</option>
|
||||
<option value="8,11">Normal Flipped: 8 by 11</option>
|
||||
<option value="3,5">Small Flipped: 3 by 5</option>
|
||||
<option value="11,17">Large Flipped: 11 by 17</option>
|
||||
<param name="pdf_size" type="select" label="Choose chart size (pixels)">
|
||||
<option value="800,600">Normal: 800 by 600</option>
|
||||
<option value="640,480">Small: 640 by 480</option>
|
||||
<option value="1480,800">Large: 1480 by 800</option>
|
||||
<option value="600,800">Normal Flipped: 600 by 800</option>
|
||||
<option value="480,640">Small Flipped: 480 by 640</option>
|
||||
<option value="800,1480">Large Flipped: 800 by 1480</option>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
|
||||
@@ -1,14 +1,14 @@
|
||||
<tool id="lastz_wrapper_1" name="Lastz" version="1.0.0">
|
||||
<description> map short reads against reference sequence</description>
|
||||
<command>
|
||||
#if ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#if ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] --${params.pre_set_options} --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="diffs"):#lastz $input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="diffs"):#lastz $seq_name.ref_name::$input1 ${input2}[fullnames] $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --nolaj --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="pre_set" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} --${params.pre_set_options} --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="No" and $out_format.value=="maf"):#lastz $input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#elif ($params.source_select=="full" and $seq_name.how_to_name=="Yes" and $out_format.value=="maf"):#lastz $seq_name.ref_name::$input1 read::${input2} $params.strand $params.seed $params.transition O=$params.O E=$params.E X=$params.X Y=$params.Y K=$params.K L=$params.L $params.entropy --ambiguousn --identity=${min_ident}..${max_ident} --census32=$output2 --coverage=$min_cvrg --format=$out_format > $output1
|
||||
#end if
|
||||
</command>
|
||||
<inputs>
|
||||
@@ -16,7 +16,7 @@
|
||||
<param name="input1" format="fasta" type="data" label="against reference" help="must be a single sequence"/>
|
||||
<param name="out_format" type="select" label="Select output format">
|
||||
<option value="diffs">Polymorphisms</option>
|
||||
<option value="maf">Alignments in MAF format</option>
|
||||
<option value="maf-">Alignments in MAF format</option>
|
||||
</param>
|
||||
|
||||
<conditional name="params">
|
||||
@@ -82,7 +82,7 @@
|
||||
<when input="out_format" value="maf" format="maf" />
|
||||
</change_format>
|
||||
</data>
|
||||
<data format="tabular" name="output2" />
|
||||
<data format="coverage" name="output2" />
|
||||
</outputs>
|
||||
<requirements>
|
||||
<requirement type="binary">lastz</requirement>
|
||||
|
||||
@@ -71,7 +71,8 @@ taxRank = {
|
||||
'genus' :20,
|
||||
'subgenus' :21,
|
||||
'species' :22,
|
||||
'subspecies' :23
|
||||
'subspecies' :23,
|
||||
'order' :13
|
||||
}
|
||||
|
||||
|
||||
@@ -157,12 +158,16 @@ try:
|
||||
for item in cur.fetchall():
|
||||
out_string = '%s\t%s\t%d\t' % ( item[0], item[1], item[2] )
|
||||
out_string += rankName
|
||||
out_string += '\t'
|
||||
out_string += str(taxRank[rankName])
|
||||
print >>out_file, out_string
|
||||
else:
|
||||
cur.execute('select rank, count(*) from %s_count where N = 1 and length(rank)>1 group by rank' % rank)
|
||||
for item in cur.fetchall():
|
||||
out_string = '%s\t%s\t' % ( item[0], item[1] )
|
||||
out_string += rankName
|
||||
out_string += '\t'
|
||||
out_string += str(taxRank[rankName])
|
||||
print >>out_file, out_string
|
||||
except Exception, e:
|
||||
stop_err("%s\n" % e)
|
||||
|
||||
+56
-23
@@ -1,10 +1,9 @@
|
||||
#!/usr/bin/env python
|
||||
#Guruprasad Ananda
|
||||
"""
|
||||
This tool provides the SQL "group by" functionality.
|
||||
Least Common Ancestor tool.
|
||||
"""
|
||||
import sys, string, re, commands, tempfile, random
|
||||
#from rpy import *
|
||||
|
||||
def stop_err(msg):
|
||||
sys.stderr.write(msg)
|
||||
@@ -14,10 +13,36 @@ def main():
|
||||
try:
|
||||
inputfile = sys.argv[1]
|
||||
outfile = sys.argv[2]
|
||||
rank_bound = int( sys.argv[3] )
|
||||
"""
|
||||
Mapping of ranks:
|
||||
root :2,
|
||||
superkingdom:3,
|
||||
kingdom :4,
|
||||
subkingdom :5,
|
||||
superphylum :6,
|
||||
phylum :7,
|
||||
subphylum :8,
|
||||
superclass :9,
|
||||
class :10,
|
||||
subclass :11,
|
||||
superorder :12,
|
||||
order :13,
|
||||
suborder :14,
|
||||
superfamily :15,
|
||||
family :16,
|
||||
subfamily :17,
|
||||
tribe :18,
|
||||
subtribe :19,
|
||||
genus :20,
|
||||
subgenus :21,
|
||||
species :22,
|
||||
subspecies :23,
|
||||
"""
|
||||
except:
|
||||
stop_err("Syntax error: Use correct syntax: program infile outfile")
|
||||
|
||||
group_col = 0
|
||||
|
||||
tmpfile = tempfile.NamedTemporaryFile()
|
||||
|
||||
try:
|
||||
@@ -42,10 +67,6 @@ def main():
|
||||
prev_vals = []
|
||||
remaining_vals = []
|
||||
skipped_lines = 0
|
||||
first_invalid_line = 0
|
||||
invalid_line = ''
|
||||
invalid_value = ''
|
||||
invalid_column = 0
|
||||
fout = open(outfile, "w")
|
||||
cols = range(1,25)
|
||||
block_valid = False
|
||||
@@ -78,9 +99,10 @@ def main():
|
||||
out_list[0] = str(prev_item)
|
||||
out_list[1] = str(prev_vals[0][0])
|
||||
out_list[2] = str(prev_vals[1][0])
|
||||
out_list[24] = str(prev_vals[23][0])
|
||||
#print >> fout, prev_vals
|
||||
#sys.exit()
|
||||
try:
|
||||
out_list[24] = str(prev_vals[23][0])
|
||||
except:
|
||||
pass
|
||||
for k, col in enumerate(cols):
|
||||
if col >= 3 and col < 24:
|
||||
if len(set(prev_vals[k])) == 1:
|
||||
@@ -90,8 +112,14 @@ def main():
|
||||
while k < 23:
|
||||
out_list[k+1] = 'n'
|
||||
k += 1
|
||||
|
||||
print >>fout, '\t'.join(out_list)
|
||||
|
||||
if rank_bound == 0:
|
||||
print >>fout, '\t'.join(out_list).strip()
|
||||
#print 'n'*( 24 - rank_bound )
|
||||
else:
|
||||
#print '\t'.join(out_list[rank_bound:24])
|
||||
if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ):
|
||||
print >>fout, '\t'.join(out_list).strip()
|
||||
|
||||
block_valid = True
|
||||
prev_item = item
|
||||
@@ -110,21 +138,21 @@ def main():
|
||||
val_list.append(fields[col].strip())
|
||||
prev_vals.append(val_list)
|
||||
|
||||
except Exception, exc:
|
||||
except:
|
||||
skipped_lines += 1
|
||||
if not first_invalid_line:
|
||||
first_invalid_line = ii+1
|
||||
else:
|
||||
skipped_lines += 1
|
||||
if not first_invalid_line:
|
||||
first_invalid_line = ii+1
|
||||
|
||||
|
||||
# Handle the last grouped value
|
||||
out_list = ['']*25
|
||||
out_list[0] = str(prev_item)
|
||||
out_list[1] = str(prev_vals[0][0])
|
||||
out_list[2] = str(prev_vals[1][0])
|
||||
out_list[24] = str(prev_vals[23][0])
|
||||
try:
|
||||
out_list[24] = str(prev_vals[23][0])
|
||||
except:
|
||||
pass
|
||||
|
||||
for k, col in enumerate(cols):
|
||||
if col >= 3 and col < 24:
|
||||
if len(set(prev_vals[k])) == 1:
|
||||
@@ -134,12 +162,17 @@ def main():
|
||||
while k < 23:
|
||||
out_list[k+1] = 'n'
|
||||
k += 1
|
||||
|
||||
print >>fout, '\t'.join(out_list)
|
||||
|
||||
if rank_bound == 0:
|
||||
print >>fout, '\t'.join(out_list).strip()
|
||||
else:
|
||||
#print ''.join(out_list[rank_bound:24])
|
||||
#print 'n'*( 24 - rank_bound )
|
||||
if ''.join(out_list[rank_bound:24]) != 'n'*( 24 - rank_bound ):
|
||||
print >>fout, '\t'.join(out_list).strip()
|
||||
|
||||
if skipped_lines > 0:
|
||||
msg= "Skipped %d invalid lines starting with line %d. Value '%s' in column %d is not numeric." % ( skipped_lines, first_invalid_line, invalid_value, invalid_column )
|
||||
print msg
|
||||
print "Skipped %d invalid lines." % ( skipped_lines )
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+78
-5
@@ -1,12 +1,85 @@
|
||||
<tool id="lca1" name="Least Common Ancestor" version="1.0.0">
|
||||
<tool id="lca1" name="Find lowest diagnostic rank" version="1.0.0">
|
||||
<description></description>
|
||||
<command interpreter="python">
|
||||
lca.py $input1 $out_file1
|
||||
lca.py $input1 $out_file1 $rank_bound
|
||||
</command>
|
||||
<inputs>
|
||||
<param format="taxonomy" name="input1" type="data" label="Select taxonomy dataset"/>
|
||||
<param format="taxonomy" name="input1" type="data" label="for taxonomy dataset"/>
|
||||
<param name="rank_bound" label="require the lowest rank to be at least" type="select">
|
||||
<option value="0">No restriction</option>
|
||||
<option value="3">Superkingdom</option>
|
||||
<option value="4">Kingdom</option>
|
||||
<option value="5">Subkingdom</option>
|
||||
<option value="6">Superphylum</option>
|
||||
<option value="7">Phylum</option>
|
||||
<option value="8">Subphylum</option>
|
||||
<option value="9">Superclass</option>
|
||||
<option value="10">Class</option>
|
||||
<option value="11">Subclass</option>
|
||||
<option value="12">Superorder</option>
|
||||
<option value="13">Order</option>
|
||||
<option value="14">Suborder</option>
|
||||
<option value="15">Superfamily</option>
|
||||
<option value="16">Family</option>
|
||||
<option value="17">Subfamily</option>
|
||||
<option value="18">Tribe</option>
|
||||
<option value="19">Subtribe</option>
|
||||
<option value="20">Genus</option>
|
||||
<option value="21">Subgenus</option>
|
||||
<option value="22">Species</option>
|
||||
<option value="23">Subspecies</option>
|
||||
</param>
|
||||
</inputs>
|
||||
<outputs>
|
||||
<data format="taxonomy" name="out_file1" metadata_source="input1" />
|
||||
</outputs>
|
||||
</tool>
|
||||
</outputs>
|
||||
<tests>
|
||||
<test>
|
||||
<param name="input1" value="lca_input.taxonomy" ftype="taxonomy"/>
|
||||
<param name="rank_bound" value="0" />
|
||||
<output name="out_file1" file="lca_output.taxonomy" ftype="taxonomy"/>
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<help>
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool identifies the lowest taxonomic rank for which a mategenomic sequencing read is diagnostic. It takes datasets produced by *Fetch Taxonomic Ranks* tool (aka Taxonomy format) as the input.
|
||||
|
||||
-------
|
||||
|
||||
**Example**
|
||||
|
||||
Suppose you have two reads, **read_1** and **read_2**, with the following taxonomic profiles (scroll sideways to see the entire dataset)::
|
||||
|
||||
read_1 1 root superkingdom1 kingdom1 subkingdom1 superphylum1 phylum1 subphylum1 superclass1 class1 subclass1 superorder1 order1 suborder1 superfamily1 family1 subfamily1 tribe1 subtribe1 genus1 subgenus1 species1 subspecies1
|
||||
read_1 2 root superkingdom1 kingdom1 subkingdom1 superphylum1 phylum1 subphylum1 superclass1 class1 subclass1 superorder1 order1 suborder1 superfamily1 family1 subfamily1 tribe1 subtribe1 genus2 subgenus2 species2 subspecies2
|
||||
read_2 3 root superkingdom1 kingdom1 subkingdom1 superphylum1 phylum3 subphylum3 superclass3 class3 subclass3 superorder3 order3 suborder3 superfamily3 family3 subfamily3 tribe3 subtribe3 genus3 subgenus3 species3 subspecies3
|
||||
read_2 4 root superkingdom1 kingdom1 subkingdom1 superphylum1 phylum4 subphylum4 superclass4 class4 subclass4 superorder4 order4 suborder4 superfamily4 family4 subfamily4 tribe4 subtribe4 genus4 subgenus4 species4 subspecies4
|
||||
|
||||
For **read_1** taxonomic labels are consistent until the genus level, where the taxonomy splits into two branches, one ending with *subspecies1* and the other with *subspecies2*. This implies **that the lowest taxomomic rank read_1 can identify is SUBTRIBE**. Similarly, read_2 is diagnostic up until the **superphylum** level. As a results the output of this tool will be::
|
||||
|
||||
read_1 2 root superkingdom1 kingdom1 subkingdom1 superphylum1 phylum1 subphylum1 superclass1 class1 subclass1 superorder1 order1 suborder1 superfamily1 family1 subfamily1 tribe1 subtribe1 n n n n
|
||||
read_2 3 root superkingdom1 kingdom1 subkingdom1 superphylum1 n n n n n n n n n n n n n n n n n
|
||||
|
||||
where, **n** means *EMPTY*.
|
||||
|
||||
--------
|
||||
|
||||
**What's up with the drop down?**
|
||||
|
||||
Why do we need the *require the lowest rank to be at least* dropdown? Let's look at the above example again. Suppose you need to find only those reads that are diagnostic on at least phylum level. To do this you need to set the *require the lowest rank to be at least* to **phylum**. As a result your output will look like this::
|
||||
|
||||
read_1 2 root superkingdom1 kingdom1 subkingdom1 superphylum1 phylum1 subphylum1 superclass1 class1 subclass1 superorder1 order1 suborder1 superfamily1 family1 subfamily1 tribe1 subtribe1 n n n n
|
||||
|
||||
.. class:: infomark
|
||||
|
||||
Note, that **read_2** is now omitted as it matches two phyla (**phylum3** and **phylum4**) and therefore is not diagnostic (but rather cosmopolitan) on *phylum* level.
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
@@ -0,0 +1,189 @@
|
||||
#!/usr/bin/env python
|
||||
"""
|
||||
Run GeneTrack(atlas) with a faked conf file to generate GeneTrack data files.
|
||||
|
||||
usage: %prog
|
||||
-l, --label=N: Data label for fit curve/peak plot
|
||||
-1, --fits=N/N/N/N/N,...: Data files (interval format) for fit curve/peak plot
|
||||
-2, --feats=N:M/N/N/N/N/N,...: Data files (interval format) for features.
|
||||
-d, --data=N: Output path for hdf5 and sqlite databases.
|
||||
-o, --output=N: Output path for export file.
|
||||
"""
|
||||
from galaxy import eggs
|
||||
import pkg_resources
|
||||
pkg_resources.require("GeneTrack")
|
||||
pkg_resources.require("bx-python")
|
||||
|
||||
import commands as oscommands
|
||||
from atlas import commands
|
||||
from atlas import sql
|
||||
from bx.cookbook import doc_optparse
|
||||
from bx.intervals import io
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
from functools import partial
|
||||
|
||||
SIGMA = 20
|
||||
WIDTH = 5 * SIGMA
|
||||
EXCLUSION_ZONE = 147
|
||||
|
||||
def main(label, fit, feats, data_dir, output):
|
||||
os.mkdir(data_dir)
|
||||
conf = DummyConf(
|
||||
__name__=label,
|
||||
CLOBBER = True,
|
||||
DATA_SIZE = 3*10**6,
|
||||
MINIMUM_PEAK_SIZE = 0.1,
|
||||
LOADER_ENABLED = False,
|
||||
FITTER_ENABLED = False,
|
||||
PREDICTOR_ENABLED = False,
|
||||
EXPORTER_ENABLED = False,
|
||||
LOADER = loader,
|
||||
FITTER = fitter,
|
||||
PREDICTOR = predictor,
|
||||
EXPORTER = partial( commands.exporter, formatter=commands.bed_formatter),
|
||||
HDF_DATABASE = os.path.join( data_dir, "data.hdf" ),
|
||||
SQL_URI = "sqlite:///%s" % os.path.join( data_dir, "features.sqlite" ),
|
||||
SIGMA = SIGMA,
|
||||
WIDTH = WIDTH,
|
||||
DATA_LABEL = label,
|
||||
FIT_LABEL = "%s-SIGMA-%d" % ( label,SIGMA ),
|
||||
PEAK_LABEL = "PRED-%s-SIGMA-%d" % ( label,SIGMA ),
|
||||
EXCLUSION_ZONE = EXCLUSION_ZONE,
|
||||
LEFT_SHIFT = EXCLUSION_ZONE / 2,
|
||||
RIGHT_SHIFT = EXCLUSION_ZONE / 2,
|
||||
EXPORT_LABELS = [ "PRED-%s-SIGMA-%d" % ( label,SIGMA ) ],
|
||||
EXPORT_DIR = os.path.join( data_dir ),
|
||||
DATA_FILE=fit and fit[1] or None,
|
||||
fit=fit,
|
||||
feats=feats,
|
||||
)
|
||||
if fit:
|
||||
# Turn on fit processing.
|
||||
conf.LOADER_ENABLED = True,
|
||||
conf.FITTER_ENABLED = True,
|
||||
conf.PREDICTOR_ENABLED = True,
|
||||
conf.EXPORTER_ENABLED = True,
|
||||
for feat in feats:
|
||||
load_feature_files(conf, feats)
|
||||
commands.execute(conf)
|
||||
outname = "%s.%s.txt" % (conf.__name__, conf.EXPORT_LABELS[0] )
|
||||
if os.path.exists( os.path.join(data_dir, outname) ):
|
||||
os.rename( os.path.join(data_dir, outname), output)
|
||||
|
||||
# mod454 seems to be a module without a package. The necessary funcitons are
|
||||
# stubbed out here until I'm sure of their final home. INS
|
||||
|
||||
def loader( conf ):
|
||||
from atlas import hdf
|
||||
from mod454.schema import Mod454Schema as Schema
|
||||
last_chrom = table = None
|
||||
db = hdf.hdf_open( conf.HDF_DATABASE, mode='a', title='HDF database')
|
||||
gp = hdf.create_group( db=db, name=conf.DATA_LABEL, desc='data group', clobber=conf.CLOBBER )
|
||||
fit_meta = conf.fit[2]
|
||||
# iterate over the file and insert into table
|
||||
for line in open( conf.fit[1], "r" ):
|
||||
if line.startswith("chrom"): continue #Skip possible header
|
||||
if line.startswith("#"): continue
|
||||
fields = line.rstrip('\r\n').split('\t')
|
||||
chrom = fields[fit_meta.chromCol]
|
||||
if chrom != last_chrom:
|
||||
if table: table.flush()
|
||||
table = hdf.create_table( db=db, name=chrom, where=gp, schema=Schema, clobber=False )
|
||||
last_chrom = chrom
|
||||
try:
|
||||
position = int(fields[fit_meta.positionCol])
|
||||
forward = float(fields[fit_meta.forwardCol])
|
||||
reverse = fit_meta.reverseCol > -1 and float(fields[fit_meta.reverseCol]) or 0.0
|
||||
row = ( position, forward, reverse, forward+reverse, )
|
||||
table.append( [ row ] )
|
||||
except ValueError:
|
||||
# Ignore bad lines
|
||||
pass
|
||||
table.flush()
|
||||
db.close()
|
||||
|
||||
def fitter( conf ):
|
||||
from mod454.fitter import fitter as mod454_fitter
|
||||
return mod454_fitter( conf )
|
||||
|
||||
def predictor( conf ):
|
||||
from mod454.predictor import predictor as mod454_predictor
|
||||
return mod454_predictor( conf )
|
||||
|
||||
def load_feature_files( conf, feats):
|
||||
"""
|
||||
Loads features from file names
|
||||
"""
|
||||
engine = sql.get_engine( conf.SQL_URI )
|
||||
sql.drop_indices(engine)
|
||||
conn = engine.connect()
|
||||
for label, fname, col_spec in feats:
|
||||
label_id = sql.make_label(engine, name=label, clobber=False)
|
||||
reader = io.NiceReaderWrapper( open(fname,"r"),
|
||||
chrom_col=col_spec.chromCol,
|
||||
start_col=col_spec.startCol,
|
||||
end_col=col_spec.endCol,
|
||||
strand_col=col_spec.strandCol,
|
||||
fix_strand=False )
|
||||
values = list()
|
||||
for interval in reader:
|
||||
print interval
|
||||
if not type( interval ) is io.GenomicInterval: continue
|
||||
row = {'label_id':label_id,
|
||||
'name':col_spec.nameCol == -1 and "%s-%s" % (str(interval.start), str(interval.end)) or interval.fields[col_spec.nameCol],
|
||||
'altname':"",
|
||||
'chrom':interval.chrom,
|
||||
'start':interval.start,
|
||||
'end':interval.end,
|
||||
'strand':interval.strand,
|
||||
'value':0,
|
||||
'freetext':""}
|
||||
values.append(row)
|
||||
insert = sql.feature_table.insert()
|
||||
conn.execute( insert, values)
|
||||
conn.close()
|
||||
sql.create_indices(engine)
|
||||
|
||||
|
||||
class Bunch( object ):
|
||||
def __init__(self, **kwargs):
|
||||
for key,value in kwargs.items():
|
||||
setattr( self, key, value )
|
||||
|
||||
class DummyConf( Bunch ):
|
||||
"""
|
||||
Fake conf module for genetrack/atlas.
|
||||
"""
|
||||
pass
|
||||
|
||||
if __name__ == "__main__":
|
||||
options, args = doc_optparse.parse( __doc__ )
|
||||
try:
|
||||
label = options.label
|
||||
if options.fits:
|
||||
fit_name, fit_meta = options.fits.split(':')[0], [int(x)-1 for x in options.fits.split(':')[1:]]
|
||||
fit_meta = Bunch(chromCol=fit_meta[0], positionCol=fit_meta[1], forwardCol=fit_meta[2], reverseCol=fit_meta[3])
|
||||
fit = ( label, fit_name, fit_meta, )
|
||||
else:
|
||||
fit = []
|
||||
# split apart the string into nested lists, preserves order
|
||||
if options.feats:
|
||||
feats = [ (
|
||||
feat_label,
|
||||
fname,
|
||||
Bunch(chromCol=int(chromCol)-1, startCol=int(startCol)-1, endCol=int(endCol)-1,
|
||||
strandCol=int(strandCol)-1, nameCol=int(nameCol)-1),
|
||||
)
|
||||
for feat_label, fname, chromCol, startCol, endCol, strandCol, nameCol
|
||||
in ( feat.split(':') for feat in options.feats.split(',') if len(feat) > 0 )]
|
||||
else:
|
||||
feats = []
|
||||
data_dir = options.data
|
||||
output = options.output
|
||||
except:
|
||||
doc_optparse.exception()
|
||||
|
||||
main(label, fit, feats, data_dir, output)
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
<tool id="genetrack1" name="GeneTrack">
|
||||
|
||||
<description>Track creator/viewer</description>
|
||||
|
||||
<code file="genetrack_code.py">
|
||||
<hook exec_after_process="exec_after_process" />
|
||||
</code>
|
||||
|
||||
<command interpreter="python">
|
||||
genetrack.py -l $data_label
|
||||
#if not str($fit_data) == "None"
|
||||
-1
|
||||
${fit_data}:${fit_data.metadata.chromCol}:${fit_data.metadata.positionCol}:${fit_data.metadata.forwardCol}:${fit_data.metadata.reverseCol}
|
||||
#end if
|
||||
#if $feature_data
|
||||
-2
|
||||
#end if
|
||||
#for $data in $feature_data
|
||||
${data.name}:${data.input}:${data.input.metadata.chromCol}:${data.input.metadata.startCol}:${data.input.metadata.endCol}:${data.input.metadata.strandCol}:${data.input.metadata.nameCol},
|
||||
#end for
|
||||
-d ${genetrack.files_path}
|
||||
-o ${bed_out}
|
||||
</command>
|
||||
|
||||
<inputs>
|
||||
<param name="data_label" type="text" label="Track Label" size="50">
|
||||
<validator type="regex" message="Please name the track with only alphanumeric characters.">[a-zA-Z0-9]{1,25}</validator>
|
||||
</param>
|
||||
<param name="fit_data" type="data" format="coverage" label="Coverage Dataset (optional)" optional="true" />
|
||||
<repeat name="feature_data" title="Features">
|
||||
<param name="input" type="data" format="interval" label="Dataset" />
|
||||
<param name="name" type="text" label="Feature Type (mRNA, ESTs, ORFs, etc.)" size="25">
|
||||
<validator type="regex" message="Please name the feature with only alphanumeric characters.">[a-zA-Z0-9]{1,25}</validator>
|
||||
</param>
|
||||
</repeat>
|
||||
</inputs>
|
||||
|
||||
<outputs>
|
||||
<data format="genetrack" name="genetrack" />
|
||||
<data format="bed" name="bed_out" />
|
||||
</outputs>
|
||||
|
||||
<requirements>
|
||||
<requirement type="python-module">tables</requirement>
|
||||
<requirement type="python-module">atlas</requirement>
|
||||
<requirement type="python-module">pychartdir</requirement>
|
||||
<requirement type="python-module">numpy</requirement>
|
||||
</requirements>
|
||||
<help>
|
||||
This tool takes the input Fit Data and creates a peak and curve plot
|
||||
showing the reads and fitness on each basepair. Features can be
|
||||
plotted below as tracks. Fit data is coverage output from tools like
|
||||
the Lastz tool. Features are simply interval datasets that may be
|
||||
plotted as tracks below the optional fit data. Both the fit data and
|
||||
feature datasets are optional, but at least one of either is required
|
||||
to generate a track.
|
||||
|
||||
-----
|
||||
|
||||
**Syntax**
|
||||
|
||||
- **Track Label** is the name of the generated track.
|
||||
|
||||
- **Fit Data** is the dataset to calculate coverage/reads across
|
||||
basepairs and generate a curve. This is optional, and tracks may
|
||||
be created simply showing features.
|
||||
|
||||
- **Features** are datasets (interval format) to be plotted as tracks.
|
||||
These are also optional, but at least 1 feature track or 1 fit
|
||||
data is required to generate a track.
|
||||
|
||||
</help>
|
||||
</tool>
|
||||
@@ -0,0 +1,12 @@
|
||||
import sets, os
|
||||
from galaxy import eggs
|
||||
from galaxy import jobs
|
||||
from galaxy.tools.parameters import DataToolParameter
|
||||
|
||||
def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None):
|
||||
"""
|
||||
Copy data_label to genetrack.metadata.label
|
||||
"""
|
||||
out_data['genetrack'].metadata.label = param_dict['data_label']
|
||||
out_data['genetrack'].info = "Use the link below to view the custom track."
|
||||
out_data['bed_out'].info = ""
|
||||
Reference in New Issue
Block a user