From 83af05b1f2541f3c78d5259f45c495c5bff3e33e Mon Sep 17 00:00:00 2001 From: Dave Bouvier Date: Tue, 1 Apr 2014 11:04:32 -0400 Subject: [PATCH] Migrate 46 tools from the distribution to the tool shed: gatk, gops, regional variation. --- .../migrate/versions/0010_tools.py | 110 +++ scripts/migrate_tools/0010_tools.sh | 4 + scripts/migrate_tools/0010_tools.xml | 141 ++++ test-data/gops_concat_out1.bed | 133 ---- tool-data/gatk_annotations.txt.sample | 30 - tool-data/gatk_sorted_picard_index.loc.sample | 26 - tool_conf.xml.sample | 66 -- tool_data_table_conf.xml.sample | 10 - tools/gatk/analyze_covariates.xml | 101 --- tools/gatk/count_covariates.xml | 292 ------- tools/gatk/depth_of_coverage.xml | 743 ------------------ tools/gatk/gatk_macros.xml | 305 ------- tools/gatk/gatk_wrapper.py | 126 --- tools/gatk/indel_realigner.xml | 209 ----- tools/gatk/print_reads.xml | 150 ---- tools/gatk/realigner_target_creator.xml | 165 ---- tools/gatk/table_recalibration.xml | 232 ------ tools/gatk/unified_genotyper.xml | 327 -------- tools/gatk/variant_annotator.xml | 266 ------- tools/gatk/variant_apply_recalibration.xml | 139 ---- tools/gatk/variant_combine.xml | 171 ---- tools/gatk/variant_eval.xml | 288 ------- tools/gatk/variant_filtration.xml | 182 ----- tools/gatk/variant_recalibrator.xml | 431 ---------- tools/gatk/variant_select.xml | 289 ------- tools/gatk/variants_validate.xml | 122 --- tools/new_operations/basecoverage.xml | 45 -- tools/new_operations/cluster.xml | 96 --- tools/new_operations/complement.xml | 61 -- tools/new_operations/concat.xml | 59 -- tools/new_operations/coverage.xml | 91 --- tools/new_operations/flanking_features.py | 214 ----- tools/new_operations/flanking_features.xml | 127 --- tools/new_operations/get_flanks.py | 191 ----- tools/new_operations/get_flanks.xml | 78 -- tools/new_operations/gops_basecoverage.py | 50 -- tools/new_operations/gops_cluster.py | 132 ---- tools/new_operations/gops_complement.py | 98 --- tools/new_operations/gops_concat.py | 76 -- tools/new_operations/gops_coverage.py | 68 -- tools/new_operations/gops_intersect.py | 98 --- tools/new_operations/gops_join.py | 82 -- tools/new_operations/gops_merge.py | 71 -- tools/new_operations/gops_subtract.py | 99 --- tools/new_operations/intersect.xml | 143 ---- tools/new_operations/join.xml | 117 --- tools/new_operations/merge.xml | 58 -- tools/new_operations/operation_filter.py | 99 --- tools/new_operations/subtract.xml | 124 --- tools/new_operations/subtract_query.py | 113 --- tools/new_operations/subtract_query.xml | 126 --- .../tables_arithmetic_operations.pl | 117 --- .../tables_arithmetic_operations.xml | 105 --- tools/regVariation/WeightedAverage.py | 94 --- tools/regVariation/WeightedAverage.xml | 71 -- tools/regVariation/best_regression_subsets.py | 91 --- .../regVariation/best_regression_subsets.xml | 66 -- tools/regVariation/compute_q_values.pl | 95 --- tools/regVariation/compute_q_values.xml | 155 ---- tools/regVariation/featureCounter.py | 149 ---- tools/regVariation/featureCounter.xml | 75 -- tools/regVariation/getIndelRates_3way.py | 248 ------ tools/regVariation/getIndelRates_3way.xml | 61 -- tools/regVariation/getIndels.py | 123 --- tools/regVariation/getIndels_2way.xml | 59 -- tools/regVariation/linear_regression.py | 147 ---- tools/regVariation/linear_regression.xml | 71 -- tools/regVariation/logistic_regression_vif.py | 168 ---- .../regVariation/logistic_regression_vif.xml | 74 -- tools/regVariation/maf_cpg_filter.py | 60 -- tools/regVariation/maf_cpg_filter.xml | 87 -- .../regVariation/microsats_alignment_level.py | 318 -------- .../microsats_alignment_level.xml | 61 -- tools/regVariation/microsats_mutability.py | 495 ------------ tools/regVariation/microsats_mutability.xml | 121 --- tools/regVariation/partialR_square.py | 147 ---- tools/regVariation/partialR_square.xml | 68 -- tools/regVariation/quality_filter.py | 242 ------ tools/regVariation/quality_filter.xml | 115 --- tools/regVariation/qv_to_bqv.py | 89 --- tools/regVariation/qv_to_bqv.xml | 17 - tools/regVariation/rcve.py | 144 ---- tools/regVariation/rcve.xml | 70 -- tools/regVariation/substitution_rates.py | 123 --- tools/regVariation/substitution_rates.xml | 61 -- tools/regVariation/substitutions.py | 85 -- tools/regVariation/substitutions.xml | 38 - tools/regVariation/windowSplitter.py | 84 -- tools/regVariation/windowSplitter.xml | 104 --- 89 files changed, 255 insertions(+), 11817 deletions(-) create mode 100644 lib/tool_shed/galaxy_install/migrate/versions/0010_tools.py create mode 100644 scripts/migrate_tools/0010_tools.sh create mode 100644 scripts/migrate_tools/0010_tools.xml delete mode 100644 test-data/gops_concat_out1.bed delete mode 100644 tool-data/gatk_annotations.txt.sample delete mode 100644 tool-data/gatk_sorted_picard_index.loc.sample delete mode 100644 tools/gatk/analyze_covariates.xml delete mode 100644 tools/gatk/count_covariates.xml delete mode 100644 tools/gatk/depth_of_coverage.xml delete mode 100644 tools/gatk/gatk_macros.xml delete mode 100644 tools/gatk/gatk_wrapper.py delete mode 100644 tools/gatk/indel_realigner.xml delete mode 100644 tools/gatk/print_reads.xml delete mode 100644 tools/gatk/realigner_target_creator.xml delete mode 100644 tools/gatk/table_recalibration.xml delete mode 100644 tools/gatk/unified_genotyper.xml delete mode 100644 tools/gatk/variant_annotator.xml delete mode 100644 tools/gatk/variant_apply_recalibration.xml delete mode 100644 tools/gatk/variant_combine.xml delete mode 100644 tools/gatk/variant_eval.xml delete mode 100644 tools/gatk/variant_filtration.xml delete mode 100644 tools/gatk/variant_recalibrator.xml delete mode 100644 tools/gatk/variant_select.xml delete mode 100644 tools/gatk/variants_validate.xml delete mode 100644 tools/new_operations/basecoverage.xml delete mode 100644 tools/new_operations/cluster.xml delete mode 100644 tools/new_operations/complement.xml delete mode 100644 tools/new_operations/concat.xml delete mode 100644 tools/new_operations/coverage.xml delete mode 100644 tools/new_operations/flanking_features.py delete mode 100644 tools/new_operations/flanking_features.xml delete mode 100644 tools/new_operations/get_flanks.py delete mode 100644 tools/new_operations/get_flanks.xml delete mode 100644 tools/new_operations/gops_basecoverage.py delete mode 100644 tools/new_operations/gops_cluster.py delete mode 100644 tools/new_operations/gops_complement.py delete mode 100644 tools/new_operations/gops_concat.py delete mode 100644 tools/new_operations/gops_coverage.py delete mode 100755 tools/new_operations/gops_intersect.py delete mode 100644 tools/new_operations/gops_join.py delete mode 100644 tools/new_operations/gops_merge.py delete mode 100644 tools/new_operations/gops_subtract.py delete mode 100644 tools/new_operations/intersect.xml delete mode 100644 tools/new_operations/join.xml delete mode 100644 tools/new_operations/merge.xml delete mode 100644 tools/new_operations/operation_filter.py delete mode 100644 tools/new_operations/subtract.xml delete mode 100644 tools/new_operations/subtract_query.py delete mode 100644 tools/new_operations/subtract_query.xml delete mode 100644 tools/new_operations/tables_arithmetic_operations.pl delete mode 100644 tools/new_operations/tables_arithmetic_operations.xml delete mode 100755 tools/regVariation/WeightedAverage.py delete mode 100755 tools/regVariation/WeightedAverage.xml delete mode 100644 tools/regVariation/best_regression_subsets.py delete mode 100644 tools/regVariation/best_regression_subsets.xml delete mode 100644 tools/regVariation/compute_q_values.pl delete mode 100644 tools/regVariation/compute_q_values.xml delete mode 100644 tools/regVariation/featureCounter.py delete mode 100644 tools/regVariation/featureCounter.xml delete mode 100755 tools/regVariation/getIndelRates_3way.py delete mode 100644 tools/regVariation/getIndelRates_3way.xml delete mode 100644 tools/regVariation/getIndels.py delete mode 100644 tools/regVariation/getIndels_2way.xml delete mode 100644 tools/regVariation/linear_regression.py delete mode 100644 tools/regVariation/linear_regression.xml delete mode 100755 tools/regVariation/logistic_regression_vif.py delete mode 100755 tools/regVariation/logistic_regression_vif.xml delete mode 100644 tools/regVariation/maf_cpg_filter.py delete mode 100644 tools/regVariation/maf_cpg_filter.xml delete mode 100644 tools/regVariation/microsats_alignment_level.py delete mode 100644 tools/regVariation/microsats_alignment_level.xml delete mode 100644 tools/regVariation/microsats_mutability.py delete mode 100644 tools/regVariation/microsats_mutability.xml delete mode 100755 tools/regVariation/partialR_square.py delete mode 100755 tools/regVariation/partialR_square.xml delete mode 100644 tools/regVariation/quality_filter.py delete mode 100644 tools/regVariation/quality_filter.xml delete mode 100755 tools/regVariation/qv_to_bqv.py delete mode 100644 tools/regVariation/qv_to_bqv.xml delete mode 100644 tools/regVariation/rcve.py delete mode 100644 tools/regVariation/rcve.xml delete mode 100644 tools/regVariation/substitution_rates.py delete mode 100644 tools/regVariation/substitution_rates.xml delete mode 100644 tools/regVariation/substitutions.py delete mode 100644 tools/regVariation/substitutions.xml delete mode 100644 tools/regVariation/windowSplitter.py delete mode 100644 tools/regVariation/windowSplitter.xml diff --git a/lib/tool_shed/galaxy_install/migrate/versions/0010_tools.py b/lib/tool_shed/galaxy_install/migrate/versions/0010_tools.py new file mode 100644 index 00000000000..754d36e48d5 --- /dev/null +++ b/lib/tool_shed/galaxy_install/migrate/versions/0010_tools.py @@ -0,0 +1,110 @@ +""" +The following tools have been eliminated from the distribution: + +1. Analyze Covariates +2. Base Coverage of all intervals +3. Perform Best-subsets Regression +4. Cluster +5. Complement intervals of a dataset +6. Compute q-values based on multiple simultaneous tests p-values +7. Concatenate two datasets into one dataset +8. Count Covariates on BAM files +9. Coverage of a set of intervals on second set of intervals +10. Depth of Coverage on BAM files +11. Feature coverage +12. Fetch closest non-overlapping feature for every interval +13. Get flanks - returns flanking region/s for every gene +14. Estimate Indel Rates for 3-way alignments +15. Fetch Indels from pairwise alignments +16. Indel Realigner - perform local realignment +17. Intersect the intervals of two datasets +18. Join the intervals of two datasets side-by-side +19. Perform Linear Regression +20. Perform Logistic Regression with vif +21. Mask CpG/non-CpG sites from MAF file +22. Merge the overlapping intervals of a dataset +23. Extract Orthologous Microsatellites from pair-wise alignments +24. Estimate microsatellite mutability by specified attributes +25. Compute partial R square +26. Print Reads from BAM files +27. Filter nucleotides based on quality scores +28. Compute RCVE +29. Realigner Target Creator for use in local realignment +30. Estimate substitution rates for non-coding regions +31. Fetch substitutions from pairwise alignments +32. Subtract the intervals of two datasets +33. Subtract Whole Dataset from another dataset +34. Table Recalibration on BAM files +35. Arithmetic Operations on tables +36. Unified Genotyper SNP and indel caller +37. Variant Annotator +38. Apply Variant Recalibration +39. Combine Variants +40. Eval Variants +41. Variant Filtration on VCF files +42. Variant Recalibrator +43. Select Variants from VCF files +44. Validate Variants +45. Assign weighted-average of the values of features overlapping an interval +46. Make windows + +The tools are now available in the repositories respectively: + +1. analyze_covariates +2. basecoverage +3. best_regression_subsets +4. cluster +5. complement +6. compute_q_values +7. concat +8. count_covariates +9. coverage +10. depth_of_coverage +11. featurecounter +12. flanking_features +13. get_flanks +14. getindelrates_3way +15. getindels_2way +16. indel_realigner +17. intersect +18. join +19. linear_regression +20. logistic_regression_vif +21. maf_cpg_filter +22. merge +23. microsats_alignment_level +24. microsats_mutability +25. partialr_square +26. print_reads +27. quality_filter +28. rcve +29. realigner_target_creator +30. substitution_rates +31. substitutions +32. subtract +33. subtract_query +34. table_recalibration +35. tables_arithmetic_operations +36. unified_genotyper +37. variant_annotator +38. variant_apply_recalibration +39. variant_combine +40. variant_eval +41. variant_filtration +42. variant_recalibrator +43. variant_select +44. variants_validate +45. weightedaverage +46. windowsplitter + +from the main Galaxy tool shed at http://toolshed.g2.bx.psu.edu +and will be installed into your local Galaxy instance at the +location discussed above by running the following command. + +""" + +def upgrade( migrate_engine ): + print __doc__ + +def downgrade( migrate_engine ): + pass diff --git a/scripts/migrate_tools/0010_tools.sh b/scripts/migrate_tools/0010_tools.sh new file mode 100644 index 00000000000..fde17704dd3 --- /dev/null +++ b/scripts/migrate_tools/0010_tools.sh @@ -0,0 +1,4 @@ +#!/bin/sh + +cd `dirname $0`/../.. +python ./scripts/migrate_tools/migrate_tools.py 0010_tools.xml $@ diff --git a/scripts/migrate_tools/0010_tools.xml b/scripts/migrate_tools/0010_tools.xml new file mode 100644 index 00000000000..449701fe650 --- /dev/null +++ b/scripts/migrate_tools/0010_tools.xml @@ -0,0 +1,141 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + \ No newline at end of file diff --git a/test-data/gops_concat_out1.bed b/test-data/gops_concat_out1.bed deleted file mode 100644 index 8ebd30a2657..00000000000 --- a/test-data/gops_concat_out1.bed +++ /dev/null @@ -1,133 +0,0 @@ -chr1 147962192 147962580 CCDS989.1_cds_0_0_chr1_147962193_r 0 - -chr1 147984545 147984630 CCDS990.1_cds_0_0_chr1_147984546_f 0 + -chr1 148078400 148078582 CCDS993.1_cds_0_0_chr1_148078401_r 0 - -chr1 148185136 148185276 CCDS996.1_cds_0_0_chr1_148185137_f 0 + -chr10 55251623 55253124 CCDS7248.1_cds_0_0_chr10_55251624_r 0 - -chr11 116124407 116124501 CCDS8374.1_cds_0_0_chr11_116124408_r 0 - -chr11 116206508 116206563 CCDS8377.1_cds_0_0_chr11_116206509_f 0 + -chr11 116211733 116212337 CCDS8378.1_cds_0_0_chr11_116211734_r 0 - -chr11 1812377 1812407 CCDS7726.1_cds_0_0_chr11_1812378_f 0 + -chr12 38440094 38440321 CCDS8736.1_cds_0_0_chr12_38440095_r 0 - -chr13 112381694 112381953 CCDS9526.1_cds_0_0_chr13_112381695_f 0 + -chr14 98710240 98712285 CCDS9949.1_cds_0_0_chr14_98710241_r 0 - -chr15 41486872 41487060 CCDS10096.1_cds_0_0_chr15_41486873_r 0 - -chr15 41673708 41673857 CCDS10097.1_cds_0_0_chr15_41673709_f 0 + -chr15 41679161 41679250 CCDS10098.1_cds_0_0_chr15_41679162_r 0 - -chr15 41826029 41826196 CCDS10101.1_cds_0_0_chr15_41826030_f 0 + -chr16 142908 143003 CCDS10397.1_cds_0_0_chr16_142909_f 0 + -chr16 179963 180135 CCDS10401.1_cds_0_0_chr16_179964_r 0 - -chr16 244413 244681 CCDS10402.1_cds_0_0_chr16_244414_f 0 + -chr16 259268 259383 CCDS10403.1_cds_0_0_chr16_259269_r 0 - -chr18 23786114 23786321 CCDS11891.1_cds_0_0_chr18_23786115_r 0 - -chr18 59406881 59407046 CCDS11985.1_cds_0_0_chr18_59406882_f 0 + -chr18 59455932 59456337 CCDS11986.1_cds_0_0_chr18_59455933_r 0 - -chr18 59600586 59600754 CCDS11988.1_cds_0_0_chr18_59600587_f 0 + -chr19 59068595 59069564 CCDS12866.1_cds_0_0_chr19_59068596_f 0 + -chr19 59236026 59236146 CCDS12872.1_cds_0_0_chr19_59236027_r 0 - -chr19 59297998 59298008 CCDS12877.1_cds_0_0_chr19_59297999_f 0 + -chr19 59302168 59302288 CCDS12878.1_cds_0_0_chr19_59302169_r 0 - -chr2 118288583 118288668 CCDS2120.1_cds_0_0_chr2_118288584_f 0 + -chr2 118394148 118394202 CCDS2121.1_cds_0_0_chr2_118394149_r 0 - -chr2 220190202 220190242 CCDS2441.1_cds_0_0_chr2_220190203_f 0 + -chr2 220229609 220230869 CCDS2443.1_cds_0_0_chr2_220229610_r 0 - -chr20 33330413 33330423 CCDS13249.1_cds_0_0_chr20_33330414_r 0 - -chr20 33513606 33513792 CCDS13255.1_cds_0_0_chr20_33513607_f 0 + -chr20 33579500 33579527 CCDS13256.1_cds_0_0_chr20_33579501_r 0 - -chr20 33593260 33593348 CCDS13257.1_cds_0_0_chr20_33593261_f 0 + -chr21 32707032 32707192 CCDS13614.1_cds_0_0_chr21_32707033_f 0 + -chr21 32869641 32870022 CCDS13615.1_cds_0_0_chr21_32869642_r 0 - -chr21 33321040 33322012 CCDS13620.1_cds_0_0_chr21_33321041_f 0 + -chr21 33744994 33745040 CCDS13625.1_cds_0_0_chr21_33744995_r 0 - -chr22 30120223 30120265 CCDS13897.1_cds_0_0_chr22_30120224_f 0 + -chr22 30160419 30160661 CCDS13898.1_cds_0_0_chr22_30160420_r 0 - -chr22 30665273 30665360 CCDS13901.1_cds_0_0_chr22_30665274_f 0 + -chr22 30939054 30939266 CCDS13903.1_cds_0_0_chr22_30939055_r 0 - -chr5 131424298 131424460 CCDS4149.1_cds_0_0_chr5_131424299_f 0 + -chr5 131556601 131556672 CCDS4151.1_cds_0_0_chr5_131556602_r 0 - -chr5 131621326 131621419 CCDS4152.1_cds_0_0_chr5_131621327_f 0 + -chr5 131847541 131847666 CCDS4155.1_cds_0_0_chr5_131847542_r 0 - -chr6 108299600 108299744 CCDS5061.1_cds_0_0_chr6_108299601_r 0 - -chr6 108594662 108594687 CCDS5063.1_cds_0_0_chr6_108594663_f 0 + -chr6 108640045 108640151 CCDS5064.1_cds_0_0_chr6_108640046_r 0 - -chr6 108722976 108723115 CCDS5067.1_cds_0_0_chr6_108722977_f 0 + -chr7 113660517 113660685 CCDS5760.1_cds_0_0_chr7_113660518_f 0 + -chr7 116512159 116512389 CCDS5771.1_cds_0_0_chr7_116512160_r 0 - -chr7 116714099 116714152 CCDS5773.1_cds_0_0_chr7_116714100_f 0 + -chr7 116945541 116945787 CCDS5774.1_cds_0_0_chr7_116945542_r 0 - -chr8 118881131 118881317 CCDS6324.1_cds_0_0_chr8_118881132_r 0 - -chr9 128764156 128764189 CCDS6914.1_cds_0_0_chr9_128764157_f 0 + -chr9 128787519 128789136 CCDS6915.1_cds_0_0_chr9_128787520_r 0 - -chr9 128882427 128882523 CCDS6917.1_cds_0_0_chr9_128882428_f 0 + -chr9 128937229 128937445 CCDS6919.1_cds_0_0_chr9_128937230_r 0 - -chrX 122745047 122745924 CCDS14606.1_cds_0_0_chrX_122745048_f 0 + -chrX 152648964 152649196 CCDS14733.1_cds_0_0_chrX_152648965_r 0 - -chrX 152691446 152691471 CCDS14735.1_cds_0_0_chrX_152691447_f 0 + -chrX 152694029 152694263 CCDS14736.1_cds_0_0_chrX_152694030_r 0 - -chr1 147962192 147962580 NM_005997_cds_0_0_chr1_147962193_r 0 - -chr1 147984545 147984630 BC007833_cds_0_0_chr1_147984546_f 0 + -chr1 148078400 148078582 AJ011123_cds_0_0_chr1_148078401_r 0 - -chr1 148185136 148185276 NM_002796_cds_0_0_chr1_148185137_f 0 + -chr10 55251623 55253124 AY029205_cds_0_0_chr10_55251624_r 0 - -chr11 116124407 116124501 AK057832_cds_0_0_chr11_116124408_r 0 - -chr11 116206508 116206563 NM_000040_cds_1_0_chr11_116206509_f 0 + -chr11 116211733 116212337 BC005380_cds_0_0_chr11_116211734_r 0 - -chr11 130745911 130745993 AY358331_cds_0_0_chr11_130745912_f 0 + -chr12 38440094 38440321 NM_052885_cds_0_0_chr12_38440095_r 0 - -chr12 38905200 38905351 AY792511_cds_0_0_chr12_38905201_f 0 + -chr13 112381694 112381953 NM_207440_cds_1_0_chr13_112381695_f 0 + -chr13 29680676 29680875 NM_032116_cds_0_0_chr13_29680677_r 0 - -chr14 98521864 98521922 U88895_cds_0_0_chr14_98521865_f 0 + -chr14 98710240 98712285 NM_022898_cds_0_0_chr14_98710241_r 0 - -chr15 41486872 41487060 BX537418_cds_0_0_chr15_41486873_r 0 - -chr15 41673708 41673857 AK223365_cds_0_0_chr15_41673709_f 0 + -chr15 41679161 41679250 NM_153700_cds_0_0_chr15_41679162_r 0 - -chr15 41773540 41773689 AK223365_cds_0_0_chr15_41773541_f 0 + -chr16 142908 143003 NM_005332_cds_0_0_chr16_142909_f 0 + -chr16 179197 179339 BC065198_cds_0_0_chr16_179198_r 0 - -chr16 244413 244681 AK057165_cds_2_0_chr16_244414_f 0 + -chr16 259268 259383 AB016929_cds_0_0_chr16_259269_r 0 - -chr18 23786114 23786321 NM_001792_cds_0_0_chr18_23786115_r 0 - -chr18 59406881 59407046 NM_012397_cds_1_0_chr18_59406882_f 0 + -chr18 59455932 59456337 AB046400_cds_0_0_chr18_59455933_r 0 - -chr18 59528407 59528575 AY792326_cds_0_0_chr18_59528408_f 0 + -chr19 59068595 59069564 BC013995_cds_1_0_chr19_59068596_f 0 + -chr19 59236026 59236146 NM_198481_cds_0_0_chr19_59236027_r 0 - -chr19 59297998 59298008 NM_004542_cds_0_0_chr19_59297999_f 0 + -chr19 59318205 59318718 AK128544_cds_3_0_chr19_59318206_r 0 - -chr2 118288583 118288668 NM_006773_cds_0_0_chr2_118288584_f 0 + -chr2 118390395 118390500 BC005078_cds_0_0_chr2_118390396_r 0 - -chr2 220108689 220109267 AY125465_cds_0_0_chr2_220108690_f 0 + -chr2 220229609 220230869 NM_024536_cds_0_0_chr2_220229610_r 0 - -chr20 33330413 33330423 NM_181466_cds_0_0_chr20_33330414_r 0 - -chr20 33485370 33486123 BC085019_cds_1_0_chr20_33485371_f 0 + -chr20 33488491 33489122 NM_000557_cds_1_0_chr20_33488492_r 0 - -chr20 33513606 33513792 AF022655_cds_1_0_chr20_33513607_f 0 + -chr21 32687402 32687588 NM_032910_cds_0_0_chr21_32687403_f 0 + -chr21 32869641 32870022 NM_018277_cds_3_0_chr21_32869642_r 0 - -chr21 33321040 33322012 NM_005806_cds_1_0_chr21_33321041_f 0 + -chr21 33728358 33728724 AK129657_cds_0_0_chr21_33728359_r 0 - -chr22 30120223 30120265 NM_004147_cds_0_0_chr22_30120224_f 0 + -chr22 30160419 30160661 BC032941_cds_0_0_chr22_30160420_r 0 - -chr22 30228824 30228916 NM_001007467_cds_1_0_chr22_30228825_f 0 + -chr22 30340151 30340376 CR456540_cds_0_0_chr22_30340152_r 0 - -chr5 131311206 131311254 AF099740_cds_11_0_chr5_131311207_r 0 - -chr5 131424298 131424460 NM_000588_cds_0_0_chr5_131424299_f 0 + -chr5 131556601 131556672 BC035813_cds_0_0_chr5_131556602_r 0 - -chr5 131621326 131621419 BC003096_cds_0_0_chr5_131621327_f 0 + -chr6 108299600 108299744 NM_007214_cds_0_0_chr6_108299601_r 0 - -chr6 108594662 108594687 NM_003269_cds_0_0_chr6_108594663_f 0 + -chr6 108640045 108640151 NM_003795_cds_0_0_chr6_108640046_r 0 - -chr6 108722976 108723115 NM_145315_cds_0_0_chr6_108722977_f 0 + -chr7 113660517 113660685 AF467257_cds_1_0_chr7_113660518_f 0 + -chr7 116512159 116512389 NM_003391_cds_0_0_chr7_116512160_r 0 - -chr7 116714099 116714152 NM_000492_cds_0_0_chr7_116714100_f 0 + -chr7 116945541 116945787 AF377960_cds_0_0_chr7_116945542_r 0 - -chr8 118881131 118881317 NM_000127_cds_0_0_chr8_118881132_r 0 - -chr9 128764156 128764189 BC051300_cds_0_0_chr9_128764157_f 0 + -chr9 128787519 128789136 NM_014908_cds_0_0_chr9_128787520_r 0 - -chr9 128789552 128789584 NM_015354_cds_0_0_chr9_128789553_f 0 + -chr9 128850516 128850624 AB058751_cds_0_0_chr9_128850517_r 0 - -chrX 122745047 122745924 NM_001167_cds_1_0_chrX_122745048_f 0 + -chrX 152648964 152649196 NM_000425_cds_0_0_chrX_152648965_r 0 - -chrX 152691446 152691471 AF101728_cds_0_0_chrX_152691447_f 0 + -chrX 152694029 152694263 BC052303_cds_0_0_chrX_152694030_r 0 - diff --git a/tool-data/gatk_annotations.txt.sample b/tool-data/gatk_annotations.txt.sample deleted file mode 100644 index d089f6f3288..00000000000 --- a/tool-data/gatk_annotations.txt.sample +++ /dev/null @@ -1,30 +0,0 @@ -#unique_id name gatk_value tools_valid_for -AlleleBalance AlleleBalance AlleleBalance UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -AlleleBalanceBySample AlleleBalanceBySample AlleleBalanceBySample UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -BaseCounts BaseCounts BaseCounts UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -BaseQualityRankSumTest BaseQualityRankSumTest BaseQualityRankSumTest UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -ChromosomeCounts ChromosomeCounts ChromosomeCounts UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -DepthOfCoverage DepthOfCoverage DepthOfCoverage UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -DepthPerAlleleBySample DepthPerAlleleBySample DepthPerAlleleBySample UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -FisherStrand FisherStrand FisherStrand UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -GCContent GCContent GCContent UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -HaplotypeScore HaplotypeScore HaplotypeScore UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -HardyWeinberg HardyWeinberg HardyWeinberg UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -HomopolymerRun HomopolymerRun HomopolymerRun UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -InbreedingCoeff InbreedingCoeff InbreedingCoeff UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -IndelType IndelType IndelType UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -LowMQ LowMQ LowMQ UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -MVLikelihoodRatio MVLikelihoodRatio MVLikelihoodRatio UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -MappingQualityRankSumTest MappingQualityRankSumTest MappingQualityRankSumTest UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -MappingQualityZero MappingQualityZero MappingQualityZero UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -MappingQualityZeroBySample MappingQualityZeroBySample MappingQualityZeroBySample UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -MappingQualityZeroFraction MappingQualityZeroFraction MappingQualityZeroFraction UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -NBaseCount NBaseCount NBaseCount UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -QualByDepth QualByDepth QualByDepth UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -RMSMappingQuality RMSMappingQuality RMSMappingQuality UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -ReadDepthAndAllelicFractionBySample ReadDepthAndAllelicFractionBySample ReadDepthAndAllelicFractionBySample UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -ReadPosRankSumTest ReadPosRankSumTest ReadPosRankSumTest UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -SampleList SampleList SampleList UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -SnpEff SnpEff SnpEff VariantAnnotator,VariantRecalibrator -SpanningDeletions SpanningDeletions SpanningDeletions UnifiedGenotyper,VariantAnnotator,VariantRecalibrator -TechnologyComposition TechnologyComposition TechnologyComposition UnifiedGenotyper,VariantAnnotator,VariantRecalibrator diff --git a/tool-data/gatk_sorted_picard_index.loc.sample b/tool-data/gatk_sorted_picard_index.loc.sample deleted file mode 100644 index 8598ffd8953..00000000000 --- a/tool-data/gatk_sorted_picard_index.loc.sample +++ /dev/null @@ -1,26 +0,0 @@ -#This is a sample file distributed with Galaxy that enables tools -#to use a directory of Picard dict and associated files. You will need -#to create these data files and then create a picard_index.loc file -#similar to this one (store it in this directory) that points to -#the directories in which those files are stored. The picard_index.loc -#file has this format (longer white space is the TAB character): -# -# -# -#So, for example, if you had hg18 indexed and stored in -#/depot/data2/galaxy/srma/hg18/, -#then the srma_index.loc entry would look like this: -# -#hg18 hg18 hg18 Pretty /depot/data2/galaxy/picard/hg18/hg18.fa -# -#and your /depot/data2/galaxy/srma/hg18/ directory -#would contain the following three files: -#hg18.fa -#hg18.dict -#hg18.fa.fai -# -#The dictionary file for each reference (ex. hg18.dict) must be -#created via Picard (http://picard.sourceforge.net). Note that -#the dict file does not have the .fa extension although the -#path list in the loc file does include it. -# diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index 7efff0d3a3c..7527f450b7f 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -46,7 +46,6 @@ -
@@ -62,7 +61,6 @@
-
@@ -108,17 +106,6 @@
- - - - - - - - - - -
@@ -143,26 +130,6 @@
-
- - - - - - - - - - - -
-
- - - - - -
@@ -231,39 +198,6 @@
-
-
diff --git a/tool_data_table_conf.xml.sample b/tool_data_table_conf.xml.sample index 688a01202d2..a7a7da46829 100644 --- a/tool_data_table_conf.xml.sample +++ b/tool_data_table_conf.xml.sample @@ -55,16 +55,6 @@ value, dbkey, name, path - - - value, dbkey, name, path - -
- - - value, name, gatk_value, tools_valid_for - -
value, dbkey, name, path diff --git a/tools/gatk/analyze_covariates.xml b/tools/gatk/analyze_covariates.xml deleted file mode 100644 index 84f33b6849c..00000000000 --- a/tools/gatk/analyze_covariates.xml +++ /dev/null @@ -1,101 +0,0 @@ - - - draw plots - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - --html_report_from_directory "${output_html}" "${output_html.files_path}" - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/AnalyzeCovariates.jar" - -recalFile "${input_recal}" - -outputDir "${output_html.files_path}" - ##--num_threads 4 ##hard coded, for now - ##-log "${output_log}" - ##-Rscript,--path_to_Rscript path_to_Rscript; on path is good enough - #if $analysis_param_type.analysis_param_type_selector == "advanced": - --ignoreQ "${analysis_param_type.ignore_q}" - --numRG "${analysis_param_type.num_read_groups}" - --max_quality_score "${analysis_param_type.max_quality_score}" - --max_histogram_value "${analysis_param_type.max_histogram_value}" - ${analysis_param_type.do_indel_quality} - #end if - ' - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Create collapsed versions of the recal csv file and call R scripts to plot residual error versus the various covariates. - -For more information on base quality score recalibration using the GATK, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Base_quality_score_recalibration>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: AnalyzeCovariates accepts an recal CSV file. - - -**Outputs** - -The output is in CSV and HTML files with links to PDF graphs and a data files. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - recal_file The input recal csv file to analyze - output_dir The directory in which to output all the plots and intermediate data files - path_to_Rscript The path to your implementation of Rscript. For Broad users this is maybe /broad/tools/apps/R-2.6.0/bin/Rscript - path_to_resources Path to resources folder holding the Sting R scripts. - ignoreQ Ignore bases with reported quality less than this number. - numRG Only process N read groups. Default value: -1 (process all read groups) - max_quality_score The integer value at which to cap the quality scores, default is 50 - max_histogram_value If supplied, this value will be the max value of the histogram plots - do_indel_quality If supplied, this value will be the max value of the histogram plots - -@CITATION_SECTION@ - - diff --git a/tools/gatk/count_covariates.xml b/tools/gatk/count_covariates.xml deleted file mode 100644 index 658502adf59..00000000000 --- a/tools/gatk/count_covariates.xml +++ /dev/null @@ -1,292 +0,0 @@ - - on BAM files - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - -d "-I" "${reference_source.input_bam}" "${reference_source.input_bam.ext}" "gatk_input" - #if str( $reference_source.input_bam.metadata.bam_index ) != "None": - -d "" "${reference_source.input_bam.metadata.bam_index}" "bam_index" "gatk_input" ##hardcode galaxy ext type as bam_index - #end if - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "CountCovariates" - --num_threads \${GALAXY_SLOTS:-4} - -et "NO_ET" ##ET no phone home - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - --recal_file "${output_recal}" - ${standard_covs} - #if str( $covariates ) != "None": - #for $cov in str( $covariates ).split( ',' ): - -cov "${cov}" - #end for - #end if - ' - - #set $snp_dataset_provided = False - #set $rod_binding_names = dict() - #for $rod_binding in $rod_bind: - #if str( $rod_binding.rod_bind_type.rod_bind_type_selector ) == 'custom': - #set $rod_bind_name = $rod_binding.rod_bind_type.custom_rod_name - #else - #set $rod_bind_name = $rod_binding.rod_bind_type.rod_bind_type_selector - #end if - #if str( $rod_binding.rod_bind_type.rod_bind_type_selector ) == 'dbsnp': - #set $snp_dataset_provided = True - #end if - #set $rod_binding_names[$rod_bind_name] = $rod_binding_names.get( $rod_bind_name, -1 ) + 1 - -d "--knownSites:${rod_bind_name},%(file_type)s" "${rod_binding.rod_bind_type.input_rod}" "${rod_binding.rod_bind_type.input_rod.ext}" "input_${rod_bind_name}_${rod_binding_names[$rod_bind_name]}" - #end for - - #include source=$standard_gatk_options# - - ##start analysis specific options - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - #if $analysis_param_type.default_read_group_type.default_read_group_type_selector == "set": - --default_read_group "${analysis_param_type.default_read_group_type.default_read_group}" - #end if - #if str( $analysis_param_type.default_platform ) != "default": - --default_platform "${analysis_param_type.default_platform}" - #end if - #if str( $analysis_param_type.force_read_group_type.force_read_group_type_selector ) == "set": - --force_read_group "${analysis_param_type.force_read_group_type.force_read_group}" - #end if - #if str( $analysis_param_type.force_platform ) != "default": - --force_platform "${analysis_param_type.force_platform}" - #end if - ${analysis_param_type.exception_if_no_tile} - #if str( $analysis_param_type.solid_options_type.solid_options_type_selector ) == "set": - #if str( $analysis_param_type.solid_options_type.solid_recal_mode ) != "default": - --solid_recal_mode "${analysis_param_type.solid_options_type.solid_recal_mode}" - #end if - #if str( $analysis_param_type.solid_options_type.solid_nocall_strategy ) != "default": - --solid_nocall_strategy "${analysis_param_type.solid_options_type.solid_nocall_strategy}" - #end if - #end if - --window_size_nqs "${analysis_param_type.window_size_nqs}" - --homopolymer_nback "${analysis_param_type.homopolymer_nback}" - ' - #end if - #if not $snp_dataset_provided: - -p '--run_without_dbsnp_potentially_ruining_quality' - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: warningmark - -"This calculation is critically dependent on being able to skip over known variant sites. Please provide a dbSNP ROD or a VCF file containing known sites of genetic variation." -However, if you do not provide this file, the '--run_without_dbsnp_potentially_ruining_quality' flag will be automatically used, and the command will be allowed to run. - -**What it does** - -This walker is designed to work as the first pass in a two-pass processing step. It does a by-locus traversal operating only at sites that are not in dbSNP. We assume that all reference mismatches we see are therefore errors and indicative of poor base quality. This walker generates tables based on various user-specified covariates (such as read group, reported quality score, cycle, and dinucleotide) Since there is a large amount of data one can then calculate an empirical probability of error given the particular covariates seen at this site, where p(error) = num mismatches / num observations The output file is a CSV list of (the several covariate values, num observations, num mismatches, empirical quality score) The first non-comment line of the output file gives the name of the covariates that were used for this calculation. Note: ReadGroupCovariate and QualityScoreCovariate are required covariates and will be added for the user regardless of whether or not they were specified Note: This walker is designed to be used in conjunction with TableRecalibrationWalker. - -For more information on base quality score recalibration using the GATK, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Base_quality_score_recalibration>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: CountCovariates accepts an aligned BAM input file. - - -**Outputs** - -The output is in CSV format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - - default_read_group If a read has no read group then default to the provided String. - default_platform If a read has no platform then default to the provided String. Valid options are illumina, 454, and solid. - force_read_group If provided, the read group ID of EVERY read will be forced to be the provided String. This is useful to collapse all data into a single read group. - force_platform If provided, the platform of EVERY read will be forced to be the provided String. Valid options are illumina, 454, and solid. - window_size_nqs The window size used by MinimumNQSCovariate for its calculation - homopolymer_nback The number of previous bases to look at in HomopolymerCovariate - exception_if_no_tile If provided, TileCovariate will throw an exception when no tile can be found. The default behavior is to use tile = -1 - solid_recal_mode How should we recalibrate solid bases in whichthe reference was inserted? Options = DO_NOTHING, SET_Q_ZERO, SET_Q_ZERO_BASE_N, or REMOVE_REF_BIAS (DO_NOTHING|SET_Q_ZERO|SET_Q_ZERO_BASE_N|REMOVE_REF_BIAS) - solid_nocall_strategy Defines the behavior of the recalibrator when it encounters no calls in the color space. Options = THROW_EXCEPTION, LEAVE_READ_UNRECALIBRATED, or PURGE_READ (THROW_EXCEPTION|LEAVE_READ_UNRECALIBRATED|PURGE_READ) - recal_file Filename for the input covariates table recalibration .csv file - out The output CSV file - recal_file Filename for the outputted covariates table recalibration file - standard_covs Use the standard set of covariates in addition to the ones listed using the -cov argument - run_without_dbsnp_potentially_ruining_quality If specified, allows the recalibrator to be used without a dbsnp rod. Very unsafe and for expert users only. - -@CITATION_SECTION@ - - diff --git a/tools/gatk/depth_of_coverage.xml b/tools/gatk/depth_of_coverage.xml deleted file mode 100644 index 20b1800d6d7..00000000000 --- a/tools/gatk/depth_of_coverage.xml +++ /dev/null @@ -1,743 +0,0 @@ - - on BAM files - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - #for $i, $input_bam in enumerate( $reference_source.input_bams ): - -d "-I" "${input_bam.input_bam}" "${input_bam.input_bam.ext}" "gatk_input_${i}" - #if str( $input_bam.input_bam.metadata.bam_index ) != "None": - -d "" "${input_bam.input_bam.metadata.bam_index}" "bam_index" "gatk_input_${i}" ##hardcode galaxy ext type as bam_index - #end if - #end for - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "DepthOfCoverage" - ##--num_threads 4 ##hard coded, for now - - -et "NO_ET" ##ET no phone home - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - #if str( $input_calculate_coverage_over_genes ) != "None": - --calculateCoverageOverGenes "${input_calculate_coverage_over_genes}" - #end if - #if str( $partition_type ) != "None": - #for $pt in str( $partition_type ).split( ',' ): - --partitionType "${pt}" - #end for - #end if - --out "${output_per_locus_coverage}" - - #for $ct_group in $summary_coverage_threshold_group: - --summaryCoverageThreshold "${ct_group.summary_coverage_threshold}" - #end for - --outputFormat "${output_format}" - ' - - #include source=$standard_gatk_options# - ##start analysis specific options - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - ${analysis_param_type.ignore_deletion_sites} - ${analysis_param_type.include_deletions} - --maxBaseQuality "${analysis_param_type.max_base_quality}" - --maxMappingQuality "${analysis_param_type.max_mapping_quality}" - --minBaseQuality "${analysis_param_type.min_base_quality}" - --minMappingQuality "${analysis_param_type.min_mapping_quality}" - --nBins "${analysis_param_type.n_bins}" - ${analysis_param_type.omit_depth_output_at_each_base} - ${analysis_param_type.omit_interval_statistics} - ${analysis_param_type.omit_locus_table} - ${analysis_param_type.omit_per_sample_stats} - ${analysis_param_type.print_base_counts} - ${analysis_param_type.print_bin_endpoints_and_exit} - --start "${analysis_param_type.start}" - --stop "${analysis_param_type.stop}" - ' - #end if - ##Move additional files to final location - #if str( $partition_type ) != "None": - #set $partition_types = str( $partition_type ).split( ',' ) - #else: - #set $partition_types = [ 'sample' ] - #end if - #if 'sample' in $partition_types and ( str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.print_bin_endpoints_and_exit ) == "" ): - #if str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_per_sample_stats ) == "": - && mv ${output_per_locus_coverage}.sample_summary ${output_summary_sample} - && mv ${output_per_locus_coverage}.sample_statistics ${output_statistics_sample} - #end if - #if $gatk_param_type.gatk_param_type_selector == "advanced" and len( $gatk_param_type.input_interval_repeat ) and ( str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_interval_statistics ) == "" ): - && mv ${output_per_locus_coverage}.sample_interval_summary ${output_interval_summary_sample} - && mv ${output_per_locus_coverage}.sample_interval_statistics ${output_interval_statistics_sample} - #end if - #if str( $input_calculate_coverage_over_genes ) != "None": - && mv ${output_per_locus_coverage}.sample_gene_summary ${output_gene_summary_sample} - && mv ${output_per_locus_coverage}.sample_gene_statistics ${output_gene_statistics_sample} - #end if - #if str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_depth_output_at_each_base ) == "": - && mv ${output_per_locus_coverage}.sample_cumulative_coverage_counts ${output_cumulative_coverage_counts_sample} - && mv ${output_per_locus_coverage}.sample_cumulative_coverage_proportions ${output_cumulative_coverage_proportions_sample} - #end if - #end if - - #if 'readgroup' in $partition_types and ( str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.print_bin_endpoints_and_exit ) == "" ): - #if str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_per_sample_stats ) == "": - && mv ${output_per_locus_coverage}.read_group_summary ${output_summary_readgroup} - && mv ${output_per_locus_coverage}.read_group_statistics ${output_statistics_readgroup} - #end if - #if $gatk_param_type.gatk_param_type_selector == "advanced" and len( $gatk_param_type.input_interval_repeat ) and ( str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_interval_statistics ) == "" ): - && mv ${output_per_locus_coverage}.read_group_interval_summary ${output_interval_summary_readgroup} - && mv ${output_per_locus_coverage}.read_group_interval_statistics ${output_interval_statistics_readgroup} - #end if - #if str( $input_calculate_coverage_over_genes ) != "None": - && mv ${output_per_locus_coverage}.read_group_gene_summary ${output_gene_summary_readgroup} - && mv ${output_per_locus_coverage}.read_group_gene_statistics ${output_gene_statistics_readgroup} - #end if - #if str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_depth_output_at_each_base ) == "": - && mv ${output_per_locus_coverage}.read_group_cumulative_coverage_counts ${output_cumulative_coverage_counts_readgroup} - && mv ${output_per_locus_coverage}.read_group_cumulative_coverage_proportions ${output_cumulative_coverage_proportions_readgroup} - #end if - #end if - - #if 'library' in $partition_types and ( str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.print_bin_endpoints_and_exit ) == "" ): - #if str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_per_sample_stats ) == "": - && mv ${output_per_locus_coverage}.library_summary ${output_summary_library} - && mv ${output_per_locus_coverage}.library_statistics ${output_statistics_library} - #end if - #if $gatk_param_type.gatk_param_type_selector == "advanced" and len( $gatk_param_type.input_interval_repeat ) and ( str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_interval_statistics ) == "" ): - && mv ${output_per_locus_coverage}.library_interval_summary ${output_interval_summary_library} - && mv ${output_per_locus_coverage}.library_interval_statistics ${output_interval_statistics_library} - #end if - #if str( $input_calculate_coverage_over_genes ) != "None": - && mv ${output_per_locus_coverage}.library_gene_summary ${output_gene_summary_library} - && mv ${output_per_locus_coverage}.library_gene_statistics ${output_gene_statistics_library} - #end if - #if str( $analysis_param_type.analysis_param_type_selector ) == "basic" or str( $analysis_param_type.omit_depth_output_at_each_base ) == "": - && mv ${output_per_locus_coverage}.library_cumulative_coverage_counts ${output_cumulative_coverage_counts_library} - && mv ${output_per_locus_coverage}.library_cumulative_coverage_proportions ${output_cumulative_coverage_proportions_library} - #end if - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_per_sample_stats'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'sample' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_per_sample_stats'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'sample' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'sample' in partition_type or not partition_type - gatk_param_type['gatk_param_type_selector'] == "advanced" and len( gatk_param_type['input_interval_repeat'] ) - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_interval_statistics'] == False - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'sample' in partition_type or not partition_type - gatk_param_type['gatk_param_type_selector'] == "advanced" and len( gatk_param_type['input_interval_repeat'] ) - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_interval_statistics'] == False - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - input_calculate_coverage_over_genes is not None and 'sample' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - input_calculate_coverage_over_genes is not None and 'sample' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_depth_output_at_each_base'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'sample' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_depth_output_at_each_base'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'sample' in partition_type or not partition_type - - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_per_sample_stats'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'readgroup' in partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_per_sample_stats'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'readgroup' in partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'readgroup' in partition_type - gatk_param_type['gatk_param_type_selector'] == "advanced" and len( gatk_param_type['input_interval_repeat'] ) - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_interval_statistics'] == False - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'readgroup' in partition_type - gatk_param_type['gatk_param_type_selector'] == "advanced" and len( gatk_param_type['input_interval_repeat'] ) - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_interval_statistics'] == False - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - input_calculate_coverage_over_genes is not None and 'readgroup' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - input_calculate_coverage_over_genes is not None and 'readgroup' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_depth_output_at_each_base'] == False - 'readgroup' in partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_depth_output_at_each_base'] == False - 'readgroup' in partition_type - - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_per_sample_stats'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'library' in partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_per_sample_stats'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'library' in partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'library' in partition_type - gatk_param_type['gatk_param_type_selector'] == "advanced" and len( gatk_param_type['input_interval_repeat'] ) - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_interval_statistics'] == False - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'library' in partition_type - gatk_param_type['gatk_param_type_selector'] == "advanced" and len( gatk_param_type['input_interval_repeat'] ) - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_interval_statistics'] == False - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - input_calculate_coverage_over_genes is not None and 'library' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - input_calculate_coverage_over_genes is not None and 'library' in partition_type or not partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_depth_output_at_each_base'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'library' in partition_type - - - - - - - - - - - - - - - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['omit_depth_output_at_each_base'] == False - analysis_param_type['analysis_param_type_selector'] == "basic" or analysis_param_type['print_bin_endpoints_and_exit'] == False - 'library' in partition_type - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -DepthOfCoverage processes a set of bam files to determine coverage at different levels of partitioning and aggregation. Coverage can be analyzed per locus, per interval, per gene, or in total; can be partitioned by sample, by read group, by technology, by center, or by library; and can be summarized by mean, median, quartiles, and/or percentage of bases covered to or beyond a threshold. Additionally, reads and bases can be filtered by mapping or base quality score. - -For more information on the GATK Depth of Coverage, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Depth_of_Coverage>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: DepthOfCoverage accepts aligned BAM input files. - - -**Outputs** - -The output is in various table formats. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - calculateCoverageOverGenes File NA Calculate the coverage statistics over this list of genes. Currently accepts RefSeq. - ignoreDeletionSites boolean false Ignore sites consisting only of deletions - includeDeletions boolean false Include information on deletions - maxBaseQuality byte 127 Maximum quality of bases to count towards depth. Defaults to 127 (Byte.MAX_VALUE). - maxMappingQuality int 2147483647 Maximum mapping quality of reads to count towards depth. Defaults to 2^31-1 (Integer.MAX_VALUE). - minBaseQuality byte -1 Minimum quality of bases to count towards depth. Defaults to -1. - minMappingQuality int -1 Minimum mapping quality of reads to count towards depth. Defaults to -1. - nBins int 499 Number of bins to use for granular binning - omitDepthOutputAtEachBase boolean false Will omit the output of the depth of coverage at each base, which should result in speedup - omitIntervalStatistics boolean false Will omit the per-interval statistics section, which should result in speedup - omitLocusTable boolean false Will not calculate the per-sample per-depth counts of loci, which should result in speedup - omitPerSampleStats boolean false Omits the summary files per-sample. These statistics are still calculated, so this argument will not improve runtime. - outputFormat String rtable the format of the output file (e.g. csv, table, rtable); defaults to r-readable table - partitionType Set[Partition] [sample] Partition type for depth of coverage. Defaults to sample. Can be any combination of sample, readgroup, library. - printBaseCounts boolean false Will add base counts to per-locus output. - printBinEndpointsAndExit boolean false Prints the bin values and exits immediately. Use to calibrate what bins you want before running on data. - start int 1 Starting (left endpoint) for granular binning - stop int 500 Ending (right endpoint) for granular binning - summaryCoverageThreshold int[] [15] for summary file outputs, report the % of bases coverd to >= this number. Defaults to 15; can take multiple arguments. - -@CITATION_SECTION@ - - diff --git a/tools/gatk/gatk_macros.xml b/tools/gatk/gatk_macros.xml deleted file mode 100644 index 7bd2bc07010..00000000000 --- a/tools/gatk/gatk_macros.xml +++ /dev/null @@ -1,305 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - ------ - -**Citation** - -For the underlying tool, please cite `DePristo MA, Banks E, Poplin R, Garimella KV, Maguire JR, Hartl C, Philippakis AA, del Angel G, Rivas MA, Hanna M, McKenna A, Fennell TJ, Kernytsky AM, Sivachenko AY, Cibulskis K, Gabriel SB, Altshuler D, Daly MJ. A framework for variation discovery and genotyping using next-generation DNA sequencing data. Nat Genet. 2011 May;43(5):491-8. <http://www.ncbi.nlm.nih.gov/pubmed/21478889>`_ - -If you use this tool in Galaxy, please cite Blankenberg D, et al. *In preparation.* - - - \ No newline at end of file diff --git a/tools/gatk/gatk_wrapper.py b/tools/gatk/gatk_wrapper.py deleted file mode 100644 index 17e8379811d..00000000000 --- a/tools/gatk/gatk_wrapper.py +++ /dev/null @@ -1,126 +0,0 @@ -#!/usr/bin/env python -#Dan Blankenberg - -""" -A wrapper script for running the GenomeAnalysisTK.jar commands. -""" - -import sys, optparse, os, tempfile, subprocess, shutil -from binascii import unhexlify -from string import Template - -GALAXY_EXT_TO_GATK_EXT = { 'gatk_interval':'intervals', 'bam_index':'bam.bai', 'gatk_dbsnp':'dbSNP', 'picard_interval_list':'interval_list' } #items not listed here will use the galaxy extension as-is -GALAXY_EXT_TO_GATK_FILE_TYPE = GALAXY_EXT_TO_GATK_EXT #for now, these are the same, but could be different if needed -DEFAULT_GATK_PREFIX = "gatk_file" -CHUNK_SIZE = 2**20 #1mb - - -def cleanup_before_exit( tmp_dir ): - if tmp_dir and os.path.exists( tmp_dir ): - shutil.rmtree( tmp_dir ) - -def gatk_filename_from_galaxy( galaxy_filename, galaxy_ext, target_dir = None, prefix = None ): - suffix = GALAXY_EXT_TO_GATK_EXT.get( galaxy_ext, galaxy_ext ) - if prefix is None: - prefix = DEFAULT_GATK_PREFIX - if target_dir is None: - target_dir = os.getcwd() - gatk_filename = os.path.join( target_dir, "%s.%s" % ( prefix, suffix ) ) - os.symlink( galaxy_filename, gatk_filename ) - return gatk_filename - -def gatk_filetype_argument_substitution( argument, galaxy_ext ): - return argument % dict( file_type = GALAXY_EXT_TO_GATK_FILE_TYPE.get( galaxy_ext, galaxy_ext ) ) - -def open_file_from_option( filename, mode = 'rb' ): - if filename: - return open( filename, mode = mode ) - return None - -def html_report_from_directory( html_out, dir ): - html_out.write( '\n\nGalaxy - GATK Output\n\n\n

\n

    \n' ) - for fname in sorted( os.listdir( dir ) ): - html_out.write( '
  • %s
  • \n' % ( fname, fname ) ) - html_out.write( '
\n\n\n' ) - -def index_bam_files( bam_filenames, tmp_dir ): - for bam_filename in bam_filenames: - bam_index_filename = "%s.bai" % bam_filename - if not os.path.exists( bam_index_filename ): - #need to index this bam file - stderr_name = tempfile.NamedTemporaryFile( prefix = "bam_index_stderr" ).name - command = 'samtools index %s %s' % ( bam_filename, bam_index_filename ) - proc = subprocess.Popen( args=command, shell=True, stderr=open( stderr_name, 'wb' ) ) - return_code = proc.wait() - if return_code: - for line in open( stderr_name ): - print >> sys.stderr, line - os.unlink( stderr_name ) #clean up - cleanup_before_exit( tmp_dir ) - raise Exception( "Error indexing BAM file" ) - os.unlink( stderr_name ) #clean up - -def __main__(): - #Parse Command Line - parser = optparse.OptionParser() - parser.add_option( '-p', '--pass_through', dest='pass_through_options', action='append', type="string", help='These options are passed through directly to GATK, without any modification.' ) - parser.add_option( '-o', '--pass_through_options', dest='pass_through_options_encoded', action='append', type="string", help='These options are passed through directly to GATK, with decoding from binascii.unhexlify.' ) - parser.add_option( '-d', '--dataset', dest='datasets', action='append', type="string", nargs=4, help='"-argument" "original_filename" "galaxy_filetype" "name_prefix"' ) - parser.add_option( '', '--max_jvm_heap', dest='max_jvm_heap', action='store', type="string", default=None, help='If specified, the maximum java virtual machine heap size will be set to the provide value.' ) - parser.add_option( '', '--max_jvm_heap_fraction', dest='max_jvm_heap_fraction', action='store', type="int", default=None, help='If specified, the maximum java virtual machine heap size will be set to the provide value as a fraction of total physical memory.' ) - parser.add_option( '', '--stdout', dest='stdout', action='store', type="string", default=None, help='If specified, the output of stdout will be written to this file.' ) - parser.add_option( '', '--stderr', dest='stderr', action='store', type="string", default=None, help='If specified, the output of stderr will be written to this file.' ) - parser.add_option( '', '--html_report_from_directory', dest='html_report_from_directory', action='append', type="string", nargs=2, help='"Target HTML File" "Directory"') - (options, args) = parser.parse_args() - - tmp_dir = tempfile.mkdtemp( prefix='tmp-gatk-' ) - if options.pass_through_options: - cmd = ' '.join( options.pass_through_options ) - else: - cmd = '' - if options.pass_through_options_encoded: - cmd = '%s %s' % ( cmd, ' '.join( map( unhexlify, options.pass_through_options_encoded ) ) ) - if options.max_jvm_heap is not None: - cmd = cmd.replace( 'java ', 'java -Xmx%s ' % ( options.max_jvm_heap ), 1 ) - elif options.max_jvm_heap_fraction is not None: - cmd = cmd.replace( 'java ', 'java -XX:DefaultMaxRAMFraction=%s -XX:+UseParallelGC ' % ( options.max_jvm_heap_fraction ), 1 ) - bam_filenames = [] - if options.datasets: - for ( dataset_arg, filename, galaxy_ext, prefix ) in options.datasets: - gatk_filename = gatk_filename_from_galaxy( filename, galaxy_ext, target_dir = tmp_dir, prefix = prefix ) - if dataset_arg: - cmd = '%s %s "%s"' % ( cmd, gatk_filetype_argument_substitution( dataset_arg, galaxy_ext ), gatk_filename ) - if galaxy_ext == "bam": - bam_filenames.append( gatk_filename ) - index_bam_files( bam_filenames, tmp_dir ) - #set up stdout and stderr output options - stdout = open_file_from_option( options.stdout, mode = 'wb' ) - stderr = open_file_from_option( options.stderr, mode = 'wb' ) - #if no stderr file is specified, we'll use our own - if stderr is None: - stderr = tempfile.NamedTemporaryFile( prefix="gatk-stderr-", dir=tmp_dir ) - - proc = subprocess.Popen( args=cmd, stdout=stdout, stderr=stderr, shell=True, cwd=tmp_dir ) - return_code = proc.wait() - - if return_code: - stderr_target = sys.stderr - else: - stderr_target = sys.stdout - stderr.flush() - stderr.seek(0) - while True: - chunk = stderr.read( CHUNK_SIZE ) - if chunk: - stderr_target.write( chunk ) - else: - break - stderr.close() - #generate html reports - if options.html_report_from_directory: - for ( html_filename, html_dir ) in options.html_report_from_directory: - html_report_from_directory( open( html_filename, 'wb' ), html_dir ) - - cleanup_before_exit( tmp_dir ) - -if __name__=="__main__": __main__() diff --git a/tools/gatk/indel_realigner.xml b/tools/gatk/indel_realigner.xml deleted file mode 100644 index 88bdb878305..00000000000 --- a/tools/gatk/indel_realigner.xml +++ /dev/null @@ -1,209 +0,0 @@ - - - perform local realignment - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - -d "-I" "${reference_source.input_bam}" "${reference_source.input_bam.ext}" "gatk_input" - #if str( $reference_source.input_bam.metadata.bam_index ) != "None": - -d "" "${reference_source.input_bam.metadata.bam_index}" "bam_index" "gatk_input" ##hardcode galaxy ext type as bam_index - #end if - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "IndelRealigner" - -o "${output_bam}" - -et "NO_ET" ##ET no phone home - ##--num_threads 4 ##hard coded, for now - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - -LOD "${lod_threshold}" - ${knowns_only} - ' - - #set $rod_binding_names = dict() - #for $rod_binding in $rod_bind: - #if str( $rod_binding.rod_bind_type.rod_bind_type_selector ) == 'custom': - #set $rod_bind_name = $rod_binding.rod_bind_type.custom_rod_name - #else - #set $rod_bind_name = $rod_binding.rod_bind_type.rod_bind_type_selector - #end if - #set $rod_binding_names[$rod_bind_name] = $rod_binding_names.get( $rod_bind_name, -1 ) + 1 - -d "-known:${rod_bind_name},%(file_type)s" "${rod_binding.rod_bind_type.input_rod}" "${rod_binding.rod_bind_type.input_rod.ext}" "input_${rod_bind_name}_${rod_binding_names[$rod_bind_name]}" - #end for - - #include source=$standard_gatk_options# - ##start analysis specific options - -d "-targetIntervals" "${target_intervals}" "${target_intervals.ext}" "gatk_target_intervals" - -p ' - --disable_bam_indexing - ' - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - --entropyThreshold "${analysis_param_type.entropy_threshold}" - ${analysis_param_type.simplify_bam} - --consensusDeterminationModel "${analysis_param_type.consensus_determination_model}" - --maxIsizeForMovement "${analysis_param_type.max_insert_size_for_movement}" - --maxPositionalMoveAllowed "${analysis_param_type.max_positional_move_allowed}" - --maxConsensuses "${analysis_param_type.max_consensuses}" - --maxReadsForConsensuses "${analysis_param_type.max_reads_for_consensuses}" - --maxReadsForRealignment "${analysis_param_type.max_reads_for_realignment}" - ${analysis_param_type.no_original_alignment_tags} - ' - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Performs local realignment of reads based on misalignments due to the presence of indels. Unlike most mappers, this walker uses the full alignment context to determine whether an appropriate alternate reference (i.e. indel) exists and updates SAMRecords accordingly. - -For more information on local realignment around indels using the GATK, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Local_realignment_around_indels>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: IndelRealigner accepts an aligned BAM and a list of intervals to realign as input files. - - -**Outputs** - -The output is in the BAM format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - targetIntervals intervals file output from RealignerTargetCreator - LODThresholdForCleaning LOD threshold above which the cleaner will clean - entropyThreshold percentage of mismatches at a locus to be considered having high entropy - out Output bam - bam_compression Compression level to use for writing BAM files - disable_bam_indexing Turn off on-the-fly creation of indices for output BAM files. - simplifyBAM If provided, output BAM files will be simplified to include just key reads for downstream variation discovery analyses (removing duplicates, PF-, non-primary reads), as well stripping all extended tags from the kept reads except the read group identifier - useOnlyKnownIndels Don't run 'Smith-Waterman' to generate alternate consenses; use only known indels provided as RODs for constructing the alternate references. - maxReadsInMemory max reads allowed to be kept in memory at a time by the SAMFileWriter. Keep it low to minimize memory consumption (but the tool may skip realignment on regions with too much coverage. If it is too low, it may generate errors during realignment); keep it high to maximize realignment (but make sure to give Java enough memory). - maxIsizeForMovement maximum insert size of read pairs that we attempt to realign - maxPositionalMoveAllowed maximum positional move in basepairs that a read can be adjusted during realignment - maxConsensuses max alternate consensuses to try (necessary to improve performance in deep coverage) - maxReadsForConsensuses max reads used for finding the alternate consensuses (necessary to improve performance in deep coverage) - maxReadsForRealignment max reads allowed at an interval for realignment; if this value is exceeded, realignment is not attempted and the reads are passed to the output file(s) as-is - noOriginalAlignmentTags Don't output the original cigar or alignment start tags for each realigned read in the output bam. - -@CITATION_SECTION@ - - diff --git a/tools/gatk/print_reads.xml b/tools/gatk/print_reads.xml deleted file mode 100644 index 68667b1434b..00000000000 --- a/tools/gatk/print_reads.xml +++ /dev/null @@ -1,150 +0,0 @@ - - from BAM files - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - #for $i, $input_bam in enumerate( $reference_source.input_bams ): - -d "-I" "${input_bam.input_bam}" "${input_bam.input_bam.ext}" "gatk_input_${i}" - #if str( $input_bam.input_bam.metadata.bam_index ) != "None": - -d "" "${input_bam.input_bam.metadata.bam_index}" "bam_index" "gatk_input_${i}" ##hardcode galaxy ext type as bam_index - #end if - #end for - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "PrintReads" - ##--num_threads 4 ##hard coded, for now - --out "${output_bam}" - -et "NO_ET" ##ET no phone home - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - --number "${number}" - #if $platform: - --platform "${platform}" - #end if - #if $read_group: - --readGroup "${read_group}" - #end if - #for $sample_file in $sample_file_repeat: - --sample_file "${sample_file.input_sample_file}" - #end for - #for $sample_name in $sample_name_repeat: - --sample_name "${sample_name.sample_name}" - #end for - ' - - #include source=$standard_gatk_options# - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -PrintReads can dynamically merge the contents of multiple input BAM files, resulting in merged output sorted in coordinate order. - -For more information on the GATK Print Reads Walker, see this `tool specific page <http://www.broadinstitute.org/gsa/gatkdocs/release/org_broadinstitute_sting_gatk_walkers_PrintReadsWalker.html>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: PrintReads accepts one or more BAM or SAM input files. - - -**Outputs** - -The output is in BAM format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - number int -1 Print the first n reads from the file, discarding the rest - platform String NA Exclude all reads with this platform from the output - readGroup String NA Exclude all reads with this read group from the output - sample_file Set[File] [] File containing a list of samples (one per line). Can be specified multiple times - sample_name Set[String] [] Sample name to be included in the analysis. Can be specified multiple times. - -@CITATION_SECTION@ - - diff --git a/tools/gatk/realigner_target_creator.xml b/tools/gatk/realigner_target_creator.xml deleted file mode 100644 index 58b6c86dc05..00000000000 --- a/tools/gatk/realigner_target_creator.xml +++ /dev/null @@ -1,165 +0,0 @@ - - for use in local realignment - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - -d "-I" "${reference_source.input_bam}" "${reference_source.input_bam.ext}" "gatk_input" - #if str( $reference_source.input_bam.metadata.bam_index ) != "None": - -d "" "${reference_source.input_bam.metadata.bam_index}" "bam_index" "gatk_input" ##hardcode galaxy ext type as bam_index - #end if - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "RealignerTargetCreator" - -o "${output_interval}" - -et "NO_ET" ##ET no phone home - --num_threads \${GALAXY_SLOTS:-4} - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - ' - #set $rod_binding_names = dict() - #for $rod_binding in $rod_bind: - #if str( $rod_binding.rod_bind_type.rod_bind_type_selector ) == 'custom': - #set $rod_bind_name = $rod_binding.rod_bind_type.custom_rod_name - #else - #set $rod_bind_name = $rod_binding.rod_bind_type.rod_bind_type_selector - #end if - #set $rod_binding_names[$rod_bind_name] = $rod_binding_names.get( $rod_bind_name, -1 ) + 1 - -d "-known:${rod_bind_name},%(file_type)s" "${rod_binding.rod_bind_type.input_rod}" "${rod_binding.rod_bind_type.input_rod.ext}" "input_${rod_bind_name}_${rod_binding_names[$rod_bind_name]}" - #end for - - #include source=$standard_gatk_options# - ##start analysis specific options - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - --minReadsAtLocus "${analysis_param_type.minReadsAtLocus}" - --windowSize "${analysis_param_type.windowSize}" - --mismatchFraction "${analysis_param_type.mismatchFraction}" - --maxIntervalSize "${analysis_param_type.maxIntervalSize}" - ' - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Emits intervals for the Local Indel Realigner to target for cleaning. Ignores 454 reads, MQ0 reads, and reads with consecutive indel operators in the CIGAR string. - -For more information on local realignment around indels using the GATK, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Local_realignment_around_indels>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: RealignerTargetCreator accepts an aligned BAM input file. - - -**Outputs** - -The output is in GATK Interval format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - windowSize window size for calculating entropy or SNP clusters - mismatchFraction fraction of base qualities needing to mismatch for a position to have high entropy; to disable set to <= 0 or > 1 - minReadsAtLocus minimum reads at a locus to enable using the entropy calculation - maxIntervalSize maximum interval size - -@CITATION_SECTION@ - - diff --git a/tools/gatk/table_recalibration.xml b/tools/gatk/table_recalibration.xml deleted file mode 100644 index caf2480cee4..00000000000 --- a/tools/gatk/table_recalibration.xml +++ /dev/null @@ -1,232 +0,0 @@ - - on BAM files - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - -d "-I" "${reference_source.input_bam}" "${reference_source.input_bam.ext}" "gatk_input" - #if str( $reference_source.input_bam.metadata.bam_index ) != "None": - -d "" "${reference_source.input_bam.metadata.bam_index}" "bam_index" "gatk_input" ##hardcode galaxy ext type as bam_index - #end if - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "TableRecalibration" - -o "${output_bam}" - -et "NO_ET" ##ET no phone home - ##--num_threads 4 ##hard coded, for now - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - --recal_file "${input_recal}" - --disable_bam_indexing - ' - #include source=$standard_gatk_options# - - ##start analysis specific options - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - #if $analysis_param_type.default_read_group_type.default_read_group_type_selector == "set": - --default_read_group "${analysis_param_type.default_read_group_type.default_read_group}" - #end if - #if str( $analysis_param_type.default_platform ) != "default": - --default_platform "${analysis_param_type.default_platform}" - #end if - #if str( $analysis_param_type.force_read_group_type.force_read_group_type_selector ) == "set": - --force_read_group "${analysis_param_type.force_read_group_type.force_read_group}" - #end if - #if str( $analysis_param_type.force_platform ) != "default": - --force_platform "${analysis_param_type.force_platform}" - #end if - ${analysis_param_type.exception_if_no_tile} - #if str( $analysis_param_type.solid_options_type.solid_options_type_selector ) == "set": - #if str( $analysis_param_type.solid_options_type.solid_recal_mode ) != "default": - --solid_recal_mode "${analysis_param_type.solid_options_type.solid_recal_mode}" - #end if - #if str( $analysis_param_type.solid_options_type.solid_nocall_strategy ) != "default": - --solid_nocall_strategy "${analysis_param_type.solid_options_type.solid_nocall_strategy}" - #end if - #end if - ${analysis_param_type.simplify_bam} - --preserve_qscores_less_than "${analysis_param_type.preserve_qscores_less_than}" - --smoothing "${analysis_param_type.smoothing}" - --max_quality_score "${analysis_param_type.max_quality_score}" - --window_size_nqs "${analysis_param_type.window_size_nqs}" - --homopolymer_nback "${analysis_param_type.homopolymer_nback}" - ${analysis_param_type.do_not_write_original_quals} - ' - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -This walker is designed to work as the second pass in a two-pass processing step, doing a by-read traversal. For each base in each read this walker calculates various user-specified covariates (such as read group, reported quality score, cycle, and dinuc) Using these values as a key in a large hashmap the walker calculates an empirical base quality score and overwrites the quality score currently in the read. This walker then outputs a new bam file with these updated (recalibrated) reads. Note: This walker expects as input the recalibration table file generated previously by CovariateCounterWalker. Note: This walker is designed to be used in conjunction with CovariateCounterWalker. - -For more information on base quality score recalibration using the GATK, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Base_quality_score_recalibration>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: TableRecalibration accepts an aligned BAM and a recalibration CSV input files. - - -**Outputs** - -The output is in BAM format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - default_read_group If a read has no read group then default to the provided String. - default_platform If a read has no platform then default to the provided String. Valid options are illumina, 454, and solid. - force_read_group If provided, the read group ID of EVERY read will be forced to be the provided String. This is useful to collapse all data into a single read group. - force_platform If provided, the platform of EVERY read will be forced to be the provided String. Valid options are illumina, 454, and solid. - window_size_nqs The window size used by MinimumNQSCovariate for its calculation - homopolymer_nback The number of previous bases to look at in HomopolymerCovariate - exception_if_no_tile If provided, TileCovariate will throw an exception when no tile can be found. The default behavior is to use tile = -1 - solid_recal_mode How should we recalibrate solid bases in whichthe reference was inserted? Options = DO_NOTHING, SET_Q_ZERO, SET_Q_ZERO_BASE_N, or REMOVE_REF_BIAS (DO_NOTHING|SET_Q_ZERO|SET_Q_ZERO_BASE_N|REMOVE_REF_BIAS) - solid_nocall_strategy Defines the behavior of the recalibrator when it encounters no calls in the color space. Options = THROW_EXCEPTION, LEAVE_READ_UNRECALIBRATED, or PURGE_READ (THROW_EXCEPTION|LEAVE_READ_UNRECALIBRATED|PURGE_READ) - recal_file Filename for the input covariates table recalibration .csv file - out The output BAM file - bam_compression Compression level to use for writing BAM files - disable_bam_indexing Turn off on-the-fly creation of indices for output BAM files. - simplifyBAM If provided, output BAM files will be simplified to include just key reads for downstream variation discovery analyses (removing duplicates, PF-, non-primary reads), as well stripping all extended tags from the kept reads except the read group identifier - preserve_qscores_less_than Bases with quality scores less than this threshold won't be recalibrated, default=5. In general it's unsafe to change qualities scores below < 5, since base callers use these values to indicate random or bad bases - smoothing Number of imaginary counts to add to each bin bin order to smooth out bins with few data points, default=1 - max_quality_score The integer value at which to cap the quality scores, default=50 - doNotWriteOriginalQuals If true, we will not write the original quality (OQ) tag for each read - -@CITATION_SECTION@ - - diff --git a/tools/gatk/unified_genotyper.xml b/tools/gatk/unified_genotyper.xml deleted file mode 100644 index c54e7b1b08b..00000000000 --- a/tools/gatk/unified_genotyper.xml +++ /dev/null @@ -1,327 +0,0 @@ - - SNP and indel caller - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - #for $i, $input_bam in enumerate( $reference_source.input_bams ): - -d "-I" "${input_bam.input_bam}" "${input_bam.input_bam.ext}" "gatk_input_${i}" - #if str( $input_bam.input_bam.metadata.bam_index ) != "None": - -d "" "${input_bam.input_bam.metadata.bam_index}" "bam_index" "gatk_input_${i}" ##hardcode galaxy ext type as bam_index - #end if - #end for - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "UnifiedGenotyper" - --num_threads \${GALAXY_SLOTS:-4} - --out "${output_vcf}" - --metrics_file "${output_metrics}" - -et "NO_ET" ##ET no phone home - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - --genotype_likelihoods_model "${genotype_likelihoods_model}" - --standard_min_confidence_threshold_for_calling "${standard_min_confidence_threshold_for_calling}" - --standard_min_confidence_threshold_for_emitting "${standard_min_confidence_threshold_for_emitting}" - ' - #set $rod_binding_names = dict() - #for $rod_binding in $rod_bind: - #if str( $rod_binding.rod_bind_type.rod_bind_type_selector ) == 'custom': - #set $rod_bind_name = $rod_binding.rod_bind_type.custom_rod_name - #else - #set $rod_bind_name = $rod_binding.rod_bind_type.rod_bind_type_selector - #end if - #set $rod_binding_names[$rod_bind_name] = $rod_binding_names.get( $rod_bind_name, -1 ) + 1 - -d "--dbsnp:${rod_bind_name},%(file_type)s" "${rod_binding.rod_bind_type.input_rod}" "${rod_binding.rod_bind_type.input_rod.ext}" "input_${rod_bind_name}_${rod_binding_names[$rod_bind_name]}" - #end for - - #include source=$standard_gatk_options# - ##start analysis specific options - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - --p_nonref_model "${analysis_param_type.p_nonref_model}" - --heterozygosity "${analysis_param_type.heterozygosity}" - --pcr_error_rate "${analysis_param_type.pcr_error_rate}" - --genotyping_mode "${analysis_param_type.genotyping_mode_type.genotyping_mode}" - #if str( $analysis_param_type.genotyping_mode_type.genotyping_mode ) == 'GENOTYPE_GIVEN_ALLELES': - --alleles "${analysis_param_type.genotyping_mode_type.input_alleles_rod}" - #end if - --output_mode "${analysis_param_type.output_mode}" - ${analysis_param_type.compute_SLOD} - --min_base_quality_score "${analysis_param_type.min_base_quality_score}" - --max_deletion_fraction "${analysis_param_type.max_deletion_fraction}" - --max_alternate_alleles "${analysis_param_type.max_alternate_alleles}" - --min_indel_count_for_genotyping "${analysis_param_type.min_indel_count_for_genotyping}" - --indel_heterozygosity "${analysis_param_type.indel_heterozygosity}" - --indelGapContinuationPenalty "${analysis_param_type.indelGapContinuationPenalty}" - --indelGapOpenPenalty "${analysis_param_type.indelGapOpenPenalty}" - --indelHaplotypeSize "${analysis_param_type.indelHaplotypeSize}" - ${analysis_param_type.doContextDependentGapPenalties} - #if str( $analysis_param_type.annotation ) != "None": - #for $annotation in str( $analysis_param_type.annotation.fields.gatk_value ).split( ','): - --annotation "${annotation}" - #end for - #end if - #for $additional_annotation in $analysis_param_type.additional_annotations: - --annotation "${additional_annotation.additional_annotation_name}" - #end for - #if str( $analysis_param_type.group ) != "None": - #for $group in str( $analysis_param_type.group ).split( ','): - --group "${group}" - #end for - #end if - #if str( $analysis_param_type.exclude_annotations ) != "None": - #for $annotation in str( $analysis_param_type.exclude_annotations.fields.gatk_value ).split( ','): - --excludeAnnotation "${annotation}" - #end for - #end if - ${analysis_param_type.multiallelic} - ' -## #if str( $analysis_param_type.snpEff_rod_bind_type.snpEff_rod_bind_type_selector ) == 'set_snpEff': -## -p '--annotation "SnpEff"' -## -d "--snpEffFile:${analysis_param_type.snpEff_rod_bind_type.snpEff_rod_name},%(file_type)s" "${analysis_param_type.snpEff_rod_bind_type.snpEff_input_rod}" "${analysis_param_type.snpEff_rod_bind_type.snpEff_input_rod.ext}" "input_snpEff_${analysis_param_type.snpEff_rod_bind_type.snpEff_rod_name}" -## #else: -## -p '--excludeAnnotation "SnpEff"' -## #end if - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -A variant caller which unifies the approaches of several disparate callers. Works for single-sample and multi-sample data. The user can choose from several different incorporated calculation models. - -For more information on the GATK Unified Genotyper, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Unified_genotyper>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: UnifiedGenotyper accepts an aligned BAM input file. - - -**Outputs** - -The output is in VCF format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - genotype_likelihoods_model Genotype likelihoods calculation model to employ -- BOTH is the default option, while INDEL is also available for calling indels and SNP is available for calling SNPs only (SNP|INDEL|BOTH) - p_nonref_model Non-reference probability calculation model to employ -- EXACT is the default option, while GRID_SEARCH is also available. (EXACT|GRID_SEARCH) - heterozygosity Heterozygosity value used to compute prior likelihoods for any locus - pcr_error_rate The PCR error rate to be used for computing fragment-based likelihoods - genotyping_mode Should we output confident genotypes (i.e. including ref calls) or just the variants? (DISCOVERY|GENOTYPE_GIVEN_ALLELES) - output_mode Should we output confident genotypes (i.e. including ref calls) or just the variants? (EMIT_VARIANTS_ONLY|EMIT_ALL_CONFIDENT_SITES|EMIT_ALL_SITES) - standard_min_confidence_threshold_for_calling The minimum phred-scaled confidence threshold at which variants not at 'trigger' track sites should be called - standard_min_confidence_threshold_for_emitting The minimum phred-scaled confidence threshold at which variants not at 'trigger' track sites should be emitted (and filtered if less than the calling threshold) - noSLOD If provided, we will not calculate the SLOD - min_base_quality_score Minimum base quality required to consider a base for calling - max_deletion_fraction Maximum fraction of reads with deletions spanning this locus for it to be callable [to disable, set to < 0 or > 1; default:0.05] - min_indel_count_for_genotyping Minimum number of consensus indels required to trigger genotyping run - indel_heterozygosity Heterozygosity for indel calling - indelGapContinuationPenalty Indel gap continuation penalty - indelGapOpenPenalty Indel gap open penalty - indelHaplotypeSize Indel haplotype size - doContextDependentGapPenalties Vary gap penalties by context - indel_recal_file Filename for the input covariates table recalibration .csv file - EXPERIMENTAL, DO NO USE - indelDebug Output indel debug info - out File to which variants should be written - annotation One or more specific annotations to apply to variant calls - group One or more classes/groups of annotations to apply to variant calls - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variant_annotator.xml b/tools/gatk/variant_annotator.xml deleted file mode 100644 index cb41296ac0d..00000000000 --- a/tools/gatk/variant_annotator.xml +++ /dev/null @@ -1,266 +0,0 @@ - - - - gatk - samtools - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - #if str( $reference_source.input_bam ) != "None": - -d "-I" "${reference_source.input_bam}" "${reference_source.input_bam.ext}" "gatk_input" - #if str( $reference_source.input_bam.metadata.bam_index ) != "None": - -d "" "${reference_source.input_bam.metadata.bam_index}" "bam_index" "gatk_input" ##hardcode galaxy ext type as bam_index - #end if - #end if - -d "--variant" "${reference_source.input_variant}" "${reference_source.input_variant.ext}" "input_variant" - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - ##--list - -T "VariantAnnotator" - ##--num_threads 4 ##hard coded, for now - -et "NO_ET" ##ET no phone home - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - -o "${output_vcf}" - #if str( $annotations_type.annotations_type_selector ) == "use_all_annotations": - --useAllAnnotations - #else: - #if $annotations_type.annotations: - #for $annotation in str( $annotations_type.annotations.fields.gatk_value ).split( ',' ): - --annotation "${annotation}" - #end for - #end if - #end if - #if $exclude_annotations: - #for $annotation in str( $exclude_annotations.fields.gatk_value ).split( ',' ): - --excludeAnnotation "${annotation}" - #end for - #end if - #for $additional_annotation in $additional_annotations: - --annotation "${additional_annotation.additional_annotation_name}" - #end for - ' - #if $reference_source.input_variant_bti: - -d "--intervals" "${reference_source.input_variant}" "${reference_source.input_variant.ext}" "input_variant_bti" - #end if - - #for $rod_binding in $comp_rod_bind: - -d "--comp:${rod_binding.comp_rod_name},%(file_type)s" "${rod_binding.comp_input_rod}" "${rod_binding.comp_input_rod.ext}" "input_comp_${rod_binding.comp_rod_name}" - #end for - - #if str( $dbsnp_rod_bind_type.dbsnp_rod_bind_type_selector ) == 'set_dbsnp': - -d "--dbsnp:${dbsnp_rod_bind_type.dbsnp_rod_name},%(file_type)s" "${dbsnp_rod_bind_type.dbsnp_input_rod}" "${dbsnp_rod_bind_type.dbsnp_input_rod.ext}" "input_dbsnp_${dbsnp_rod_bind_type.dbsnp_rod_name}" - #end if - - - #for $rod_binding in $resource_rod_bind: - -d "--resource:${rod_binding.resource_rod_name},%(file_type)s" "${rod_binding.resource_input_rod}" "${rod_binding.resource_input_rod.ext}" "input_resource_${rod_binding.resource_rod_name}" - #end for - - #if str( $snpEff_rod_bind_type.snpEff_rod_bind_type_selector ) == 'set_snpEff': - -p '--annotation "SnpEff"' - -d "--snpEffFile:${snpEff_rod_bind_type.snpEff_rod_name},%(file_type)s" "${snpEff_rod_bind_type.snpEff_input_rod}" "${snpEff_rod_bind_type.snpEff_input_rod.ext}" "input_snpEff_${snpEff_rod_bind_type.snpEff_rod_name}" - #else: - -p '--excludeAnnotation "SnpEff"' - #end if - - #for $expression in $expressions: - -p '--expression "${expression.expression}"' - #end for - - #include source=$standard_gatk_options# - - -p ' - #if str( $annotation_group ) != "None": - #for $group in str( $annotation_group ).split( ',' ): - --group "${group}" - #end for - #end if - #if str( $family_string ) != "": - --family_string "${family_string}" - #end if - --MendelViolationGenotypeQualityThreshold "${mendel_violation_genotype_quality_threshold}" - ' - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Annotates variant calls with context information. Users can specify which of the available annotations to use. - -For more information on using the VariantAnnotator, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/VariantAnnotator>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - - -**Inputs** - -GenomeAnalysisTK: VariantAnnotator accepts a variant input file. - - -**Outputs** - -The output is in VCF format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - - sampleName The sample (NA-ID) corresponding to the variant input (for non-VCF input only) - annotation One or more specific annotations to apply to variant calls - group One or more classes/groups of annotations to apply to variant calls - expression One or more specific expressions to apply to variant calls; see documentation for more details - useAllAnnotations Use all possible annotations (not for the faint of heart) - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variant_apply_recalibration.xml b/tools/gatk/variant_apply_recalibration.xml deleted file mode 100644 index 0fb4d82e1a6..00000000000 --- a/tools/gatk/variant_apply_recalibration.xml +++ /dev/null @@ -1,139 +0,0 @@ - - - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - #for $var_count, $variant in enumerate( $reference_source.variants ): - -d "--input:input_${var_count},%(file_type)s" "${variant.input_variants}" "${variant.input_variants.ext}" "input_variants_${var_count}" - #end for - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "ApplyRecalibration" - ##--num_threads 4 ##hard coded, for now - -et "NO_ET" ##ET no phone home - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - --recal_file "${reference_source.input_recal}" - --tranches_file "${reference_source.input_tranches}" - --out "${output_variants}" - ' - - #include source=$standard_gatk_options# - - ##start analysis specific options - -p ' - --mode "${mode}" - - #for $ignore_filter in $ignore_filters: - #set $ignore_filter_name = str( $ignore_filter.ignore_filter_type.ignore_filter_type_selector ) - #if $ignore_filter_name == "custom": - #set $ignore_filter_name = str( $ignore_filter.ignore_filter_type.filter_name ) - #end if - --ignore_filter "${ignore_filter_name}" - #end for - --ts_filter_level "${ts_filter_level}" - ' - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Applies cuts to the input vcf file (by adding filter lines) to achieve the desired novel FDR levels which were specified during VariantRecalibration - -For more information on using the ApplyRecalibration module, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Variant_quality_score_recalibration>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: ApplyRecalibration accepts a variant input file, a recalibration file and a tranches file. - - -**Outputs** - -The output is in VCF format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - - recal_file The output recal file used by ApplyRecalibration - tranches_file The input tranches file describing where to cut the data - out The output filtered, recalibrated VCF file - ts_filter_level The truth sensitivity level at which to start filtering - ignore_filter If specified the optimizer will use variants even if the specified filter name is marked in the input VCF file - mode Recalibration mode to employ: 1.) SNP for recalibrating only SNPs (emitting indels untouched in the output VCF); 2.) INDEL for indels; and 3.) BOTH for recalibrating both SNPs and indels simultaneously. (SNP|INDEL|BOTH) - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variant_combine.xml b/tools/gatk/variant_combine.xml deleted file mode 100644 index a18e5023c47..00000000000 --- a/tools/gatk/variant_combine.xml +++ /dev/null @@ -1,171 +0,0 @@ - - - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - - #set $priority_order = [] - #for $input_variant in $reference_source.input_variants: - -d "--variant:${input_variant.input_variant_name},%(file_type)s" "${input_variant.input_variant}" "${input_variant.input_variant.ext}" "input_variant_${input_variant.input_variant_name}" - #set $input_variant_name = str( $input_variant.input_variant_name ) - #assert $input_variant_name not in $priority_order, "Variant Names must be unique" ##this should be handled by a validator - #silent $priority_order.append( $input_variant_name ) - #end for - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "CombineVariants" - --out "${output_variants}" - ##--num_threads 4 ##hard coded, for now - -et "NO_ET" ##ET no phone home - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - --genotypemergeoption "${genotype_merge_option}" - --rod_priority_list "${ ','.join( $priority_order ) }" - ' - - #include source=$standard_gatk_options# - - - ##start analysis specific options - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - --filteredrecordsmergetype "${analysis_param_type.filtered_records_merge_type}" - ${analysis_param_type.print_complex_merges} - ${analysis_param_type.filtered_are_uncalled} - ${analysis_param_type.minimal_vcf} - ${analysis_param_type.assume_identical_samples} - - #if str( $analysis_param_type.set_key ): - --setKey "${analysis_param_type.set_key}" - #end if - - --minimumN "${analysis_param_type.minimum_n}" - ' - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Combines VCF records from different sources; supports both full merges and set unions. Merge: combines multiple records into a single one; if sample names overlap then they are uniquified. Union: assumes each rod represents the same set of samples (although this is not enforced); using the priority list (if provided), emits a single record instance at every position represented in the rods. - -For more information on using the CombineVariants module, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/CombineVariants>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: CombineVariants accepts variant files as input. - ------- - -**Outputs** - -The output is a combined vcf file. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - out File to which variants should be written - genotypemergeoption How should we merge genotype records for samples shared across the ROD files? (UNIQUIFY|PRIORITIZE|UNSORTED|REQUIRE_UNIQUE) - filteredrecordsmergetype How should we deal with records seen at the same site in the VCF, but with different FILTER fields? KEEP_IF_ANY_UNFILTERED PASSes the record if any record is unfiltered, KEEP_IF_ALL_UNFILTERED requires all records to be unfiltered (KEEP_IF_ANY_UNFILTERED|KEEP_IF_ALL_UNFILTERED) - rod_priority_list When taking the union of variants containing genotypes: a comma-separated string describing the priority ordering for the genotypes as far as which record gets emitted; a complete priority list MUST be provided - printComplexMerges Print out interesting sites requiring complex compatibility merging - filteredAreUncalled If true, then filtered VCFs are treated as uncalled, so that filtered set annotation don't appear in the combined VCF - minimalVCF If true, then the output VCF will contain no INFO or genotype INFO field - setKey Key, by default set, in the INFO key=value tag emitted describing which set the combined VCF record came from. Set to null if you don't want the set field emitted. - assumeIdenticalSamples If true, assume input VCFs have identical sample sets and disjoint calls so that one can simply perform a merge sort to combine the VCFs into one, drastically reducing the runtime. - minimumN Combine variants and output site only if variant is present in at least N input files. - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variant_eval.xml b/tools/gatk/variant_eval.xml deleted file mode 100644 index bd036254a13..00000000000 --- a/tools/gatk/variant_eval.xml +++ /dev/null @@ -1,288 +0,0 @@ - - - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - #from binascii import hexlify - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - #for $var_count, $variant in enumerate( $reference_source.variants ): - -d "--eval:input_${var_count},%(file_type)s" "${variant.input_variant}" "${variant.input_variant.ext}" "input_variants_${var_count}" - #end for - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "VariantEval" - --out "${output_report}" - --num_threads \${GALAXY_SLOTS:-4} - -et "NO_ET" ##ET no phone home - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - ' - - #for $rod_binding in $comp_rod_bind: - -d "--comp:${rod_binding.comp_rod_name},%(file_type)s" "${rod_binding.comp_input_rod}" "${rod_binding.comp_input_rod.ext}" "input_comp_${rod_binding.comp_rod_name}" - #if str( $rod_binding.comp_known_names ): - -p '--known_names "${rod_binding.comp_rod_name}"' - #end if - #end for - - #if str( $dbsnp_rod_bind_type.dbsnp_rod_bind_type_selector ) == 'set_dbsnp': - -d "--dbsnp:${dbsnp_rod_bind_type.dbsnp_rod_name},%(file_type)s" "${dbsnp_rod_bind_type.dbsnp_input_rod}" "${dbsnp_rod_bind_type.dbsnp_input_rod.ext}" "input_dbsnp_${dbsnp_rod_bind_type.dbsnp_rod_name}" - #if str( $dbsnp_rod_bind_type.dbsnp_known_names ): - -p '--known_names "${dbsnp_rod_bind_type.dbsnp_rod_name}"' - #end if - #end if - - #include source=$standard_gatk_options# - - - ##start analysis specific options - #if $analysis_param_type.analysis_param_type_selector == "advanced": - #for $stratification in $analysis_param_type.stratifications: - #set $select_string = "--select_exps '%s' --select_names '%s'" % ( str( $stratification.select_exps ), str( $stratification.select_name ) ) - -o '${ hexlify( $select_string ) }' - #end for - -p ' - - #for $sample in $analysis_param_type.samples: - --sample "${sample.sample}" - #end for - - #if str( $analysis_param_type.stratification_modules ) != "None": - #for $stratification_module in str( $analysis_param_type.stratification_modules).split( ',' ): - --stratificationModule "${stratification_module}" - #end for - #end if - - ${analysis_param_type.do_not_use_all_standard_stratifications} - - #for $variant_type in $analysis_param_type.only_variants_of_type: - --onlyVariantsOfType "${variant_type.variant_type}" - #end for - - #if str( $analysis_param_type.eval_modules ) != "None": - #for $eval_module in str( $analysis_param_type.eval_modules).split( ',' ): - --evalModule "${eval_module}" - #end for - #end if - - ${analysis_param_type.do_not_use_all_standard_modules} - - #if str( $analysis_param_type.num_samples ) != "0": - --numSamples "${analysis_param_type.num_samples}" - #end if - - --minPhaseQuality "${analysis_param_type.min_phase_quality}" - - #if str( $analysis_param_type.family ): - --family_structure "${analysis_param_type.family}" - #end if - - --mendelianViolationQualThreshold "${analysis_param_type.mendelian_violation_qual_threshold}" - - #if str( $analysis_param_type.ancestral_alignments ) != "None": - --ancestralAlignments "${analysis_param_type.ancestral_alignments}" - #end if - ' - #if str( $analysis_param_type.known_cnvs ) != "None": - -d "--knownCNVs" "${analysis_param_type.known_cnvs}" "${analysis_param_type.known_cnvs.ext}" "input_known_cnvs" - #end if - - #if str( $analysis_param_type.strat_intervals ) != "None": - -d "--stratIntervals" "${analysis_param_type.strat_intervals}" "${analysis_param_type.strat_intervals.ext}" "input_strat_intervals" - #end if - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -General-purpose tool for variant evaluation (% in dbSNP, genotype concordance, Ti/Tv ratios, and a lot more) - -For more information on using the VariantEval module, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/VariantEval>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: VariantEval accepts variant files as input. - - -**Outputs** - -The output is a table of variant evaluation. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - - -------- - -**Settings**:: - - out An output file presented to the walker. Will overwrite contents if file exists. - list List the available eval modules and exit - select_exps One or more stratifications to use when evaluating the data - select_names Names to use for the list of stratifications (must be a 1-to-1 mapping) - sample Derive eval and comp contexts using only these sample genotypes, when genotypes are available in the original context - known_names Name of ROD bindings containing variant sites that should be treated as known when splitting eval rods into known and novel subsets - stratificationModule One or more specific stratification modules to apply to the eval track(s) (in addition to the standard stratifications, unless -noS is specified) - doNotUseAllStandardStratifications Do not use the standard stratification modules by default (instead, only those that are specified with the -S option) - onlyVariantsOfType If provided, only variants of these types will be considered during the evaluation, in - evalModule One or more specific eval modules to apply to the eval track(s) (in addition to the standard modules, unless -noE is specified) - doNotUseAllStandardModules Do not use the standard modules by default (instead, only those that are specified with the -E option) - numSamples Number of samples (used if no samples are available in the VCF file - minPhaseQuality Minimum phasing quality - family_structure If provided, genotypes in will be examined for mendelian violations: this argument is a string formatted as dad+mom=child where these parameters determine which sample names are examined - mendelianViolationQualThreshold Minimum genotype QUAL score for each trio member required to accept a site as a violation - ancestralAlignments Fasta file with ancestral alleles - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variant_filtration.xml b/tools/gatk/variant_filtration.xml deleted file mode 100644 index ae63e762145..00000000000 --- a/tools/gatk/variant_filtration.xml +++ /dev/null @@ -1,182 +0,0 @@ - - on VCF files - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - #from binascii import hexlify - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - -d "--variant:variant,%(file_type)s" "${reference_source.input_variant}" "${reference_source.input_variant.ext}" "input_variant" - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "VariantFiltration" - ##--num_threads 4 ##hard coded, for now - -et "NO_ET" ##ET no phone home - -o "${output_vcf}" - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - ' - #for $variant_filter in $variant_filters: - #set $variant_filter = "--%sExpression '%s' --%sName '%s'" % ( str( $variant_filter.is_genotype_filter ), str( $variant_filter.filter_expression ), str( $variant_filter.is_genotype_filter ), str( $variant_filter.filter_name ) ) - -o '${ hexlify( $variant_filter ) }' - #end for - - #if str( $mask_rod_bind_type.mask_rod_bind_type_selector ) == 'set_mask': - -d "--mask:${mask_rod_bind_type.mask_rod_name},%(file_type)s" "${mask_rod_bind_type.input_mask_rod}" "${mask_rod_bind_type.input_mask_rod.ext}" "input_mask_${mask_rod_bind_type.mask_rod_name}" - -p ' - --maskExtension "${mask_rod_bind_type.mask_extension}" - --maskName "${mask_rod_bind_type.mask_rod_name}" - ' - #end if - - #include source=$standard_gatk_options# - - ##start analysis specific options - #if $cluster_snp_type.cluster_snp_type_selector == "cluster_snp": - -p ' - --clusterSize "${cluster_snp_type.cluster_size}" - --clusterWindowSize "${cluster_snp_type.cluster_window_size}" - ' - #end if - -p '${missing_values_in_expressions_should_evaluate_as_failing}' - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Filters variant calls using a number of user-selectable, parameterizable criteria. - -For more information on using the VariantFiltration module, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/VariantFiltrationWalker>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: VariantFiltration accepts a VCF input file. - - -**Outputs** - -The output is in VCF format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - - filterExpression One or more expression used with INFO fields to filter (see wiki docs for more info) - filterName Names to use for the list of filters (must be a 1-to-1 mapping); this name is put in the FILTER field for variants that get filtered - genotypeFilterExpression One or more expression used with FORMAT (sample/genotype-level) fields to filter (see wiki docs for more info) - genotypeFilterName Names to use for the list of sample/genotype filters (must be a 1-to-1 mapping); this name is put in the FILTER field for variants that get filtered - clusterSize The number of SNPs which make up a cluster (see also --clusterWindowSize); [default:3] - clusterWindowSize The window size (in bases) in which to evaluate clustered SNPs (to disable the clustered SNP filter, set this value to less than 1); [default:0] - maskName The text to put in the FILTER field if a 'mask' rod is provided and overlaps with a variant call; [default:'Mask'] - missingValuesInExpressionsShouldEvaluateAsFailing When evaluating the JEXL expressions, should missing values be considered failing the expression (by default they are considered passing)? - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variant_recalibrator.xml b/tools/gatk/variant_recalibrator.xml deleted file mode 100644 index e1d2513ff7c..00000000000 --- a/tools/gatk/variant_recalibrator.xml +++ /dev/null @@ -1,431 +0,0 @@ - - - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - #for $var_count, $variant in enumerate( $reference_source.variants ): - -d "--input:input_${var_count},%(file_type)s" "${variant.input_variants}" "${variant.input_variants.ext}" "input_variants_${var_count}" - #end for - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "VariantRecalibrator" - --num_threads \${GALAXY_SLOTS:-4} - -et "NO_ET" ##ET no phone home - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - --recal_file "${output_recal}" - --tranches_file "${output_tranches}" - --rscript_file "${output_rscript}" - ' - - #set $rod_binding_names = dict() - #for $rod_binding in $rod_bind: - #if str( $rod_binding.rod_bind_type.rod_bind_type_selector ) == 'custom': - #set $rod_bind_name = $rod_binding.rod_bind_type.custom_rod_name - #elif str( $rod_binding.rod_bind_type.rod_bind_type_selector ) == 'comp': - #set $rod_bind_name = "comp" + $rod_binding.rod_bind_type.custom_rod_name - #else - #set $rod_bind_name = $rod_binding.rod_bind_type.rod_bind_type_selector - #end if - #set $rod_binding_names[$rod_bind_name] = $rod_binding_names.get( $rod_bind_name, -1 ) + 1 - #if $rod_binding.rod_bind_type.rod_training_type.rod_training_type_selector == "not_training_truth_known": - -d "--resource:${rod_bind_name},%(file_type)s" "${rod_binding.rod_bind_type.input_rod}" "${rod_binding.rod_bind_type.input_rod.ext}" "input_${rod_bind_name}_${rod_binding_names[$rod_bind_name]}" - #else: - -d "--resource:${rod_bind_name},%(file_type)s,known=${rod_binding.rod_bind_type.rod_training_type.known},training=${rod_binding.rod_bind_type.rod_training_type.training},truth=${rod_binding.rod_bind_type.rod_training_type.truth},bad=${rod_binding.rod_bind_type.rod_training_type.bad},prior=${rod_binding.rod_bind_type.rod_training_type.prior}" "${rod_binding.rod_bind_type.input_rod}" "${rod_binding.rod_bind_type.input_rod.ext}" "input_${rod_bind_name}_${rod_binding_names[$rod_bind_name]}" - #end if - #end for - - #include source=$standard_gatk_options# - - ##start analysis specific options - -p ' - #if str( $annotations ) != "None": - #for $annotation in str( $annotations.fields.gatk_value ).split( ',' ): - --use_annotation "${annotation}" - #end for - #end if - #for $additional_annotation in $additional_annotations: - --use_annotation "${additional_annotation.additional_annotation_name}" - #end for - --mode "${mode}" - ' - - #if $analysis_param_type.analysis_param_type_selector == "advanced": - -p ' - --maxGaussians "${analysis_param_type.max_gaussians}" - --maxIterations "${analysis_param_type.max_iterations}" - --numKMeans "${analysis_param_type.num_k_means}" - --stdThreshold "${analysis_param_type.std_threshold}" - --qualThreshold "${analysis_param_type.qual_threshold}" - --shrinkage "${analysis_param_type.shrinkage}" - --dirichlet "${analysis_param_type.dirichlet}" - --priorCounts "${analysis_param_type.prior_counts}" - #if str( $analysis_param_type.bad_variant_selector.bad_variant_selector_type ) == 'percent': - --percentBadVariants "${analysis_param_type.bad_variant_selector.percent_bad_variants}" - #else: - --minNumBadVariants "${analysis_param_type.bad_variant_selector.min_num_bad_variants}" - #end if - --target_titv "${analysis_param_type.target_titv}" - #for $tranche in [ $tranche.strip() for $tranche in str( $analysis_param_type.ts_tranche ).split( ',' ) if $tranche.strip() ] - --TStranche "${tranche}" - #end for - #for $ignore_filter in $analysis_param_type.ignore_filters: - #set $ignore_filter_name = str( $ignore_filter.ignore_filter_type.ignore_filter_type_selector ) - #if $ignore_filter_name == "custom": - #set $ignore_filter_name = str( $ignore_filter.ignore_filter_type.filter_name ) - #end if - --ignore_filter "${ignore_filter_name}" - #end for - --ts_filter_level "${analysis_param_type.ts_filter_level}" - ' - #end if - - - && - mv "${output_rscript}.pdf" "${output_tranches_pdf}" - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Takes variant calls as .vcf files, learns a Gaussian mixture model over the variant annotations and evaluates the variant -- assigning an informative lod score - -For more information on using the VariantRecalibrator module, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/Variant_quality_score_recalibration>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: VariantRecalibrator accepts a variant input file. - - -**Outputs** - -The output is in VCF format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - - tranches_file The output tranches file used by ApplyRecalibration - use_annotation The names of the annotations which should used for calculations - mode Recalibration mode to employ: 1.) SNP for recalibrating only snps (emitting indels untouched in the output VCF); 2.) INDEL for indels; and 3.) BOTH for recalibrating both snps and indels simultaneously. (SNP|INDEL|BOTH) - maxGaussians The maximum number of Gaussians to try during variational Bayes algorithm - maxIterations The maximum number of VBEM iterations to be performed in variational Bayes algorithm. Procedure will normally end when convergence is detected. - numKMeans The number of k-means iterations to perform in order to initialize the means of the Gaussians in the Gaussian mixture model. - stdThreshold If a variant has annotations more than -std standard deviations away from mean then don't use it for building the Gaussian mixture model. - qualThreshold If a known variant has raw QUAL value less than -qual then don't use it for building the Gaussian mixture model. - shrinkage The shrinkage parameter in variational Bayes algorithm. - dirichlet The dirichlet parameter in variational Bayes algorithm. - priorCounts The number of prior counts to use in variational Bayes algorithm. - percentBadVariants What percentage of the worst scoring variants to use when building the Gaussian mixture model of bad variants. 0.07 means bottom 7 percent. - minNumBadVariants The minimum amount of worst scoring variants to use when building the Gaussian mixture model of bad variants. Will override -percentBad arugment if necessary. - recal_file The output recal file used by ApplyRecalibration - target_titv The expected novel Ti/Tv ratio to use when calculating FDR tranches and for display on optimization curve output figures. (approx 2.15 for whole genome experiments). ONLY USED FOR PLOTTING PURPOSES! - TStranche The levels of novel false discovery rate (FDR, implied by ti/tv) at which to slice the data. (in percent, that is 1.0 for 1 percent) - ignore_filter If specified the optimizer will use variants even if the specified filter name is marked in the input VCF file - path_to_Rscript The path to your implementation of Rscript. For Broad users this is maybe /broad/tools/apps/R-2.6.0/bin/Rscript - rscript_file The output rscript file generated by the VQSR to aid in visualization of the input data and learned model - path_to_resources Path to resources folder holding the Sting R scripts. - ts_filter_level The truth sensitivity level at which to start filtering, used here to indicate filtered variants in plots - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variant_select.xml b/tools/gatk/variant_select.xml deleted file mode 100644 index 4c14793c657..00000000000 --- a/tools/gatk/variant_select.xml +++ /dev/null @@ -1,289 +0,0 @@ - - from VCF files - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - #from binascii import hexlify - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - -d "--variant:variant,%(file_type)s" "${reference_source.input_variant}" "${reference_source.input_variant.ext}" "input_variant" - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "SelectVariants" - --num_threads \${GALAXY_SLOTS:-4} - -et "NO_ET" ##ET no phone home - -o "${output_vcf}" - - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - ' - -p ' - #if $input_concordance: - --concordance "${input_concordance}" - #end if - #if $input_discordance: - --discordance "${input_discordance}" - #end if - - #for $exclude_sample_name in $exclude_sample_name_repeat: - --exclude_sample_name "${exclude_sample_name.exclude_sample_name}" - #end for - - ${exclude_filtered} - - #for $sample_name in $sample_name_repeat: - --sample_name "${sample_name.sample_name}" - #end for - - ' - - #for $select_expressions in $select_expressions_repeat: - #set $select_expression = "--select_expressions '%s'" % ( str( $select_expressions.select_expressions ) ) - -o '${ hexlify( $select_expression ) }' - #end for - - ##start tool specific options - #if str( $analysis_param_type.analysis_param_type_selector ) == 'advanced': - -p ' - #for $exclude_sample_file in $analysis_param_type.exclude_sample_file_repeat: - --exclude_sample_file "${exclude_sample_file.exclude_sample_file}" - #end for - - #for $sample_file in $analysis_param_type.sample_file_repeat: - --sample_file "${ample_file.sample_file}" - #end for - - #if $analysis_param_type.input_keep_ids: - --keepIDs "${analysis_param_type.input_keep_ids}" - #end if - - ${analysis_param_type.keep_original_AC} - - ${analysis_param_type.mendelian_violation} - - --mendelianViolationQualThreshold "${analysis_param_type.mendelian_violation_qual_threshold}" - - --remove_fraction_genotypes "${analysis_param_type.remove_fraction_genotypes}" - - --restrictAllelesTo "${analysis_param_type.restrict_alleles_to}" - - #if str( $analysis_param_type.select_random_type.select_random_type_selector ) == 'select_random_fraction': - --select_random_fraction "${analysis_param_type.select_random_type.select_random_fraction}" - #elif str( $analysis_param_type.select_random_type.select_random_type_selector ) == 'select_random_number': - --select_random_number "${analysis_param_type.select_random_type.select_random_number}" - #end if - - #if $analysis_param_type.select_type_to_include: - #for $type_to_include in str( $analysis_param_type.select_type_to_include ).split( ',' ): - --selectTypeToInclude "${type_to_include}" - #end for - #end if - - ${analysis_param_type.exclude_non_variants} - ' - - #for $sample_expressions in $analysis_param_type.sample_expressions_repeat: - #set $sample_expression = "--sample_expressions '%s'" % ( str( $sample_expressions.sample_expressions ) ) - -o '${ hexlify( $sample_expression ) }' - #end for - - #end if - ##end tool specific options - - #include source=$standard_gatk_options# - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Often, a VCF containing many samples and/or variants will need to be subset in order to facilitate certain analyses (e.g. comparing and contrasting cases vs. controls; extracting variant or non-variant loci that meet certain requirements, displaying just a few samples in a browser like IGV, etc.). SelectVariants can be used for this purpose. Given a single VCF file, one or more samples can be extracted from the file (based on a complete sample name or a pattern match). Variants can be further selected by specifying criteria for inclusion, i.e. "DP > 1000" (depth of coverage greater than 1000x), "AF < 0.25" (sites with allele frequency less than 0.25). These JEXL expressions are documented in the Using JEXL expressions section (http://www.broadinstitute.org/gsa/wiki/index.php/Using_JEXL_expressions). One can optionally include concordance or discordance tracks for use in selecting overlapping variants. - -For more information on using the SelectVariants module, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/SelectVariants>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: SelectVariants accepts a VCF input file. - - -**Outputs** - -The output is in VCF format. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - - out VCFWriter stdout File to which variants should be written - variant RodBinding[VariantContext] NA Input VCF file - concordance RodBinding[VariantContext] none Output variants that were also called in this comparison track - discordance RodBinding[VariantContext] none Output variants that were not called in this comparison track - exclude_sample_file Set[File] [] File containing a list of samples (one per line) to exclude. Can be specified multiple times - exclude_sample_name Set[String] [] Exclude genotypes from this sample. Can be specified multiple times - excludeFiltered boolean false Don't include filtered loci in the analysis - excludeNonVariants boolean false Don't include loci found to be non-variant after the subsetting procedure - keepIDs File NA Only emit sites whose ID is found in this file (one ID per line) - keepOriginalAC boolean false Don't update the AC, AF, or AN values in the INFO field after selecting - mendelianViolation Boolean false output mendelian violation sites only - mvq double 0.0 Minimum genotype QUAL score for each trio member required to accept a site as a violation - remove_fraction_genotypes double 0.0 Selects a fraction (a number between 0 and 1) of the total genotypes at random from the variant track and sets them to nocall - restrictAllelesTo NumberAlleleRestriction ALL Select only variants of a particular allelicity. Valid options are ALL (default), MULTIALLELIC or BIALLELIC - sample_expressions Set[String] NA Regular expression to select many samples from the ROD tracks provided. Can be specified multiple times - sample_file Set[File] NA File containing a list of samples (one per line) to include. Can be specified multiple times - sample_name Set[String] [] Include genotypes from this sample. Can be specified multiple times - select_expressions ArrayList[String] [] One or more criteria to use when selecting the data - select_random_fraction double 0.0 Selects a fraction (a number between 0 and 1) of the total variants at random from the variant track - select_random_number int 0 Selects a number of variants at random from the variant track - selectTypeToInclude List[Type] [] Select only a certain type of variants from the input file. Valid types are INDEL, SNP, MIXED, MNP, SYMBOLIC, NO_VARIATION. Can be specified multiple times - -@CITATION_SECTION@ - - diff --git a/tools/gatk/variants_validate.xml b/tools/gatk/variants_validate.xml deleted file mode 100644 index 89520274099..00000000000 --- a/tools/gatk/variants_validate.xml +++ /dev/null @@ -1,122 +0,0 @@ - - - - gatk - - - gatk_macros.xml - - gatk_wrapper.py - --max_jvm_heap_fraction "1" - --stdout "${output_log}" - -d "--variant:variant,%(file_type)s" "${reference_source.input_variant}" "${reference_source.input_variant.ext}" "input_variant" - -p 'java - -jar "${GALAXY_DATA_INDEX_DIR}/shared/jars/gatk/GenomeAnalysisTK.jar" - -T "ValidateVariants" - - -et "NO_ET" ##ET no phone home - ##--num_threads 4 ##hard coded, for now - ##-log "${output_log}" ##don't use this to log to file, instead directly capture stdout - #if $reference_source.reference_source_selector != "history": - -R "${reference_source.ref_file.fields.path}" - #end if - ${warn_on_errors} - ${do_not_validate_filtered_records} - ' - - #if str( $dbsnp_rod_bind_type.dbsnp_rod_bind_type_selector ) == 'set_dbsnp': - -d "--dbsnp:${dbsnp_rod_bind_type.dbsnp_rod_name},%(file_type)s" "${dbsnp_rod_bind_type.dbsnp_input_rod}" "${dbsnp_rod_bind_type.dbsnp_input_rod.ext}" "input_dbsnp_${dbsnp_rod_bind_type.dbsnp_rod_name}" - #end if - - #include source=$standard_gatk_options# - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -**What it does** - -Validates a variants file. - -For more information on using the ValidateVariants module, see this `tool specific page <http://www.broadinstitute.org/gsa/wiki/index.php/VariantValidator>`_. - -To learn about best practices for variant detection using GATK, see this `overview <http://www.broadinstitute.org/gsa/wiki/index.php/Best_Practice_Variant_Detection_with_the_GATK_v3>`_. - -If you encounter errors, please view the `GATK FAQ <http://www.broadinstitute.org/gsa/wiki/index.php/Frequently_Asked_Questions>`_. - ------- - -**Inputs** - -GenomeAnalysisTK: ValidateVariants accepts variant files as input. - - -**Outputs** - -The output is a log of variant validation. - - -Go `here <http://www.broadinstitute.org/gsa/wiki/index.php/Input_files_for_the_GATK>`_ for details on GATK file formats. - -------- - -**Settings**:: - - doNotValidateFilteredRecords should we skip validation on filtered records? - warnOnErrors should we just emit warnings on errors instead of terminating the run? - -@CITATION_SECTION@ - - diff --git a/tools/new_operations/basecoverage.xml b/tools/new_operations/basecoverage.xml deleted file mode 100644 index 19161a5f423..00000000000 --- a/tools/new_operations/basecoverage.xml +++ /dev/null @@ -1,45 +0,0 @@ - - of all intervals - gops_basecoverage.py $input1 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu, it means that it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - -This operation counts the total bases covered by a set of intervals. Bases that are covered by more than one interval are **not** counted more than once towards the total. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - -**Example** - -.. image:: ${static_path}/operation_icons/gops_baseCoverage.gif - - - - diff --git a/tools/new_operations/cluster.xml b/tools/new_operations/cluster.xml deleted file mode 100644 index 24bedd46f25..00000000000 --- a/tools/new_operations/cluster.xml +++ /dev/null @@ -1,96 +0,0 @@ - - the intervals of a dataset - - gops_cluster.py $input1 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} -d $distance -m $minregions -o $returntype - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu, it means that it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -**Syntax** - -- **Maximum distance** is greatest distance in base pairs allowed between intervals that will be considered "clustered". **Negative** values for distance are allowed, and are useful for clustering intervals that overlap. -- **Minimum intervals per cluster** allow a threshold to be set on the minimum number of intervals to be considered a cluster. Any area with less than this minimum will not be included in the output. -- **Merge clusters into single intervals** outputs intervals that span the entire cluster. -- **Find cluster intervals; preserve comments and order** filters out non-cluster intervals while maintaining the original ordering and comments in the file. -- **Find cluster intervals; output grouped by clusters** filters out non-cluster intervals, but outputs the cluster intervals so that they are grouped together. Comments and original ordering in the file are lost. - ------ - -**Examples** - -Find Clusters: - -.. image:: ${static_path}/operation_icons/gops_clusterFind.gif - -Merge Clusters: - -.. image:: ${static_path}/operation_icons/gops_clusterMerge.gif - - - diff --git a/tools/new_operations/complement.xml b/tools/new_operations/complement.xml deleted file mode 100644 index 664aa30a288..00000000000 --- a/tools/new_operations/complement.xml +++ /dev/null @@ -1,61 +0,0 @@ - - intervals of a dataset - gops_complement.py $input1 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} -l ${chromInfo} $allchroms - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu, it means that it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - -This operation complements the regions of a set of intervals. Regions are returned that represent the empty space in the input interval. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -**Syntax** - -- **Genome-wide complement** will complement all chromosomes of the genome. Leaving this option unchecked will only complement chromosomes present in the dataset. - ------ - -**Example** - -.. image:: ${static_path}/operation_icons/gops_complement.gif - - - diff --git a/tools/new_operations/concat.xml b/tools/new_operations/concat.xml deleted file mode 100644 index dc33affd1da..00000000000 --- a/tools/new_operations/concat.xml +++ /dev/null @@ -1,59 +0,0 @@ - - two datasets into one dataset - gops_concat.py $input1 $input2 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} -2 ${input2.metadata.chromCol},${input2.metadata.startCol},${input2.metadata.endCol},${input2.metadata.strandCol} $sameformat - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu -> it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -**Syntax** - -- **Both datasets are exactly the same filetype** will preserve all extra fields in both files. Leaving this unchecked will force the second dataset to use the same column assignments for chrom, start, end and strand, but will fill extra fields with a period(.). In both cases, the output fields are truncated or padded with fields of periods to maintain a truly tabular output. - ------ - -**Example** - -.. image:: ${static_path}/operation_icons/gops_concatenate.gif - - - \ No newline at end of file diff --git a/tools/new_operations/coverage.xml b/tools/new_operations/coverage.xml deleted file mode 100644 index 8dea61b3f49..00000000000 --- a/tools/new_operations/coverage.xml +++ /dev/null @@ -1,91 +0,0 @@ - - of a set of intervals on second set of intervals - gops_coverage.py $input1 $input2 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} -2 ${input2.metadata.chromCol},${input2.metadata.startCol},${input2.metadata.endCol},${input2.metadata.strandCol} - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu -> it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - -Find the coverage of intervals in the first dataset on intervals in the second dataset. The coverage is added as two columns, the first being bases covered, and the second being the fraction of bases covered by that interval. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -**Example** - - - if **First dataset** are genes :: - - chr11 5203271 5204877 NM_000518 0 - - chr11 5210634 5212434 NM_000519 0 - - chr11 5226077 5227663 NM_000559 0 - - chr11 5226079 5232587 BC020719 0 - - chr11 5230996 5232587 NM_000184 0 - - - and **Second dataset** are repeats:: - - chr11 5203895 5203991 L1MA6 500 + - chr11 5204163 5204239 A-rich 219 + - chr11 5211034 5211167 (CATATA)n 245 + - chr11 5211642 5211673 AT_rich 24 + - chr11 5226551 5226606 (CA)n 303 + - chr11 5228782 5228825 (TTTTTG)n 208 + - chr11 5229045 5229121 L1PA11 440 + - chr11 5229133 5229319 MER41A 1106 + - chr11 5229374 5229485 L2 244 - - chr11 5229751 5230083 MLT1A 913 - - chr11 5231469 5231526 (CA)n 330 + - - the Result is the coverage density of repeats in the genes:: - - chr11 5203271 5204877 NM_000518 0 - 172 0.107098 - chr11 5210634 5212434 NM_000519 0 - 164 0.091111 - chr11 5226077 5227663 NM_000559 0 - 55 0.034678 - chr11 5226079 5232587 BC020719 0 - 860 0.132145 - chr11 5230996 5232587 NM_000184 0 - 57 0.035827 - - For example, the following line of output:: - - chr11 5203271 5204877 NM_000518 0 - 172 0.107098 - - implies that 172 nucleotides accounting for 10.7% of the this interval (chr11:5203271-5204877) overlap with repetitive elements. - - - \ No newline at end of file diff --git a/tools/new_operations/flanking_features.py b/tools/new_operations/flanking_features.py deleted file mode 100644 index 5d266552ed3..00000000000 --- a/tools/new_operations/flanking_features.py +++ /dev/null @@ -1,214 +0,0 @@ -#!/usr/bin/env python -#By: Guruprasad Ananda -""" -Fetch closest up/downstream interval from features corresponding to every interval in primary - -usage: %prog primary_file features_file out_file direction - -1, --cols1=N,N,N,N: Columns for start, end, strand in first file - -2, --cols2=N,N,N,N: Columns for start, end, strand in second file - -G, --gff1: input 1 is GFF format, meaning start and end coordinates are 1-based, closed interval - -H, --gff2: input 2 is GFF format, meaning start and end coordinates are 1-based, closed interval -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * -from bx.intervals.io import * -from bx.intervals.operations import quicksect -from galaxy.datatypes.util.gff_util import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def get_closest_feature (node, direction, threshold_up, threshold_down, report_func_up, report_func_down): - #direction=1 for +ve strand upstream and -ve strand downstream cases; and it is 0 for +ve strand downstream and -ve strand upstream cases - #threhold_Up is equal to the interval start for +ve strand, and interval end for -ve strand - #threhold_down is equal to the interval end for +ve strand, and interval start for -ve strand - if direction == 1: - if node.maxend <= threshold_up: - if node.end == node.maxend: - report_func_up(node) - elif node.right and node.left: - if node.right.maxend == node.maxend: - get_closest_feature(node.right, direction, threshold_up, threshold_down, report_func_up, report_func_down) - elif node.left.maxend == node.maxend: - get_closest_feature(node.left, direction, threshold_up, threshold_down, report_func_up, report_func_down) - elif node.right and node.right.maxend == node.maxend: - get_closest_feature(node.right, direction, threshold_up, threshold_down, report_func_up, report_func_down) - elif node.left and node.left.maxend == node.maxend: - get_closest_feature(node.left, direction, threshold_up, threshold_down, report_func_up, report_func_down) - elif node.minend <= threshold_up: - if node.end <= threshold_up: - report_func_up(node) - if node.left and node.right: - if node.right.minend <= threshold_up: - get_closest_feature(node.right, direction, threshold_up, threshold_down, report_func_up, report_func_down) - if node.left.minend <= threshold_up: - get_closest_feature(node.left, direction, threshold_up, threshold_down, report_func_up, report_func_down) - elif node.left: - if node.left.minend <= threshold_up: - get_closest_feature(node.left, direction, threshold_up, threshold_down, report_func_up, report_func_down) - elif node.right: - if node.right.minend <= threshold_up: - get_closest_feature(node.right, direction, threshold_up, threshold_down, report_func_up, report_func_down) - elif direction == 0: - if node.start > threshold_down: - report_func_down(node) - if node.left: - get_closest_feature(node.left, direction, threshold_up, threshold_down, report_func_up, report_func_down) - else: - if node.right: - get_closest_feature(node.right, direction, threshold_up, threshold_down, report_func_up, report_func_down) - -def proximal_region_finder(readers, region, comments=True): - """ - Returns an iterator that yields elements of the form [ , ]. - Intervals are GenomicInterval objects. - """ - primary = readers[0] - features = readers[1] - either = False - if region == 'Upstream': - up, down = True, False - elif region == 'Downstream': - up, down = False, True - else: - up, down = True, True - if region == 'Either': - either = True - - # Read features into memory: - rightTree = quicksect.IntervalTree() - for item in features: - if type( item ) is GenomicInterval: - rightTree.insert( item, features.linenum, item ) - - for interval in primary: - if type( interval ) is Header: - yield interval - if type( interval ) is Comment and comments: - yield interval - elif type( interval ) == GenomicInterval: - chrom = interval.chrom - start = int(interval.start) - end = int(interval.end) - strand = interval.strand - if chrom not in rightTree.chroms: - continue - else: - root = rightTree.chroms[chrom] #root node for the chrom tree - result_up = [] - result_down = [] - if (strand == '+' and up) or (strand == '-' and down): - #upstream +ve strand and downstream -ve strand cases - get_closest_feature (root, 1, start, None, lambda node: result_up.append( node ), None) - - if (strand == '+' and down) or (strand == '-' and up): - #downstream +ve strand and upstream -ve strand case - get_closest_feature (root, 0, None, end-1, None, lambda node: result_down.append( node )) - - if result_up: - if len(result_up) > 1: #The results_up list has a list of intervals upstream to the given interval. - ends = [] - for n in result_up: - ends.append(n.end) - res_ind = ends.index(max(ends)) #fetch the index of the closest interval i.e. the interval with the max end from the results_up list - else: - res_ind = 0 - if not(either): - yield [ interval, result_up[res_ind].other ] - - if result_down: - if not(either): - #The last element of result_down will be the closest element to the given interval - yield [ interval, result_down[-1].other ] - - if either and (result_up or result_down): - iter_val = [] - if result_up and result_down: - if abs(start - int(result_up[res_ind].end)) <= abs(end - int(result_down[-1].start)): - iter_val = [ interval, result_up[res_ind].other ] - else: - #The last element of result_down will be the closest element to the given interval - iter_val = [ interval, result_down[-1].other ] - elif result_up: - iter_val = [ interval, result_up[res_ind].other ] - elif result_down: - #The last element of result_down will be the closest element to the given interval - iter_val = [ interval, result_down[-1].other ] - yield iter_val - -def main(): - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col_2 = parse_cols_arg( options.cols2 ) - in1_gff_format = bool( options.gff1 ) - in2_gff_format = bool( options.gff2 ) - in_fname, in2_fname, out_fname, direction = args - except: - doc_optparse.exception() - - # Set readers to handle either GFF or default format. - if in1_gff_format: - in1_reader_wrapper = GFFIntervalToBEDReaderWrapper - else: - in1_reader_wrapper = NiceReaderWrapper - if in2_gff_format: - in2_reader_wrapper = GFFIntervalToBEDReaderWrapper - else: - in2_reader_wrapper = NiceReaderWrapper - - g1 = in1_reader_wrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - g2 = in2_reader_wrapper( fileinput.FileInput( in2_fname ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - - # Find flanking features. - out_file = open( out_fname, "w" ) - try: - for result in proximal_region_finder([g1,g2], direction): - if type( result ) is list: - line, closest_feature = result - # Need to join outputs differently depending on file types. - if in1_gff_format: - # Output is GFF with added attribute 'closest feature.' - - # Invervals are in BED coordinates; need to convert to GFF. - line = convert_bed_coords_to_gff( line ) - closest_feature = convert_bed_coords_to_gff( closest_feature ) - - # Replace double quotes with single quotes in closest feature's attributes. - out_file.write( "%s closest_feature \"%s\" \n" % - ( "\t".join( line.fields ), \ - "\t".join( closest_feature.fields ).replace( "\"", "\\\"" ) - ) ) - else: - # Output is BED + closest feature fields. - output_line_fields = [] - output_line_fields.extend( line.fields ) - output_line_fields.extend( closest_feature.fields ) - out_file.write( "%s\n" % ( "\t".join( output_line_fields ) ) ) - else: - out_file.write( "%s\n" % result ) - except ParseError, exc: - fail( "Invalid file format: %s" % str( exc ) ) - - print "Direction: %s" %(direction) - if g1.skipped > 0: - print skipped( g1, filedesc=" of 1st dataset" ) - if g2.skipped > 0: - print skipped( g2, filedesc=" of 2nd dataset" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/flanking_features.xml b/tools/new_operations/flanking_features.xml deleted file mode 100644 index e8018b83677..00000000000 --- a/tools/new_operations/flanking_features.xml +++ /dev/null @@ -1,127 +0,0 @@ - - for every interval - - flanking_features.py $input1 $input2 $out_file1 $direction - - #if isinstance( $input1.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gff').__class__): - -1 1,4,5,7 --gff1 - #else: - -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} - #end if - - #if isinstance( $input2.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gff').__class__): - -2 1,4,5,7 --gff2 - #else: - -2 ${input2.metadata.chromCol},${input2.metadata.startCol},${input2.metadata.endCol},${input2.metadata.strandCol} - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -For every interval in the **interval** dataset, this tool fetches the **closest non-overlapping** upstream and / or downstream features from the **features** dataset. - ------ - -.. class:: warningmark - -**Note:** - -Every line should contain at least 3 columns: chromosome number, start and stop coordinates. If any of these columns is missing or if start and stop coordinates are not numerical, the lines will be treated as invalid and skipped. The number of skipped lines is documented in the resulting history item as a "data issue". - -If the strand column is missing from your input interval dataset, the intervals will be considered to be on positive strand. You can add a strand column to your input dataset by using the *Text Manipulation->Add column* tool. - -For GFF files, features are added as a GTF-style attribute at the end of the line. - ------ - -**Example** - -If the **intervals** are:: - - chr1 10 100 Query1.1 - chr1 500 1000 Query1.2 - chr1 1100 1250 Query1.3 - -and the **features** are:: - - chr1 120 180 Query2.1 - chr1 140 200 Query2.2 - chr1 580 1050 Query2.3 - chr1 2000 2204 Query2.4 - chr1 2500 3000 Query2.5 - -Running this tool for **Both Upstream and Downstream** will return:: - - chr1 10 100 Query1.1 chr1 120 180 Query2.1 - chr1 500 1000 Query1.2 chr1 140 200 Query2.2 - chr1 500 1000 Query1.2 chr1 2000 2204 Query2.4 - chr1 1100 1250 Query1.3 chr1 580 1050 Query2.3 - chr1 1100 1250 Query1.3 chr1 2000 2204 Query2.4 - - - - - \ No newline at end of file diff --git a/tools/new_operations/get_flanks.py b/tools/new_operations/get_flanks.py deleted file mode 100644 index 05d6ef42605..00000000000 --- a/tools/new_operations/get_flanks.py +++ /dev/null @@ -1,191 +0,0 @@ -#!/usr/bin/env python -#Done by: Guru - -""" -Get Flanking regions. - -usage: %prog input out_file size direction region - -l, --cols=N,N,N,N: Columns for chrom, start, end, strand in file - -o, --off=N: Offset -""" - -import sys, re, os -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -def stop_err( msg ): - sys.stderr.write( msg ) - sys.exit() - -def main(): - try: - if int( sys.argv[3] ) < 0: - raise Exception - except: - stop_err( "Length of flanking region(s) must be a non-negative integer." ) - - # Parsing Command Line here - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols ) - inp_file, out_file, size, direction, region = args - if strand_col_1 <= 0: - strand = "+" #if strand is not defined, default it to + - except: - stop_err( "Metadata issue, correct the metadata attributes by clicking on the pencil icon in the history item." ) - try: - offset = int(options.off) - size = int(size) - except: - stop_err( "Invalid offset or length entered. Try again by entering valid integer values." ) - - fo = open(out_file,'w') - - skipped_lines = 0 - first_invalid_line = 0 - invalid_line = None - elems = [] - j=0 - for i, line in enumerate( file( inp_file ) ): - line = line.strip() - if line and (not line.startswith( '#' )) and line != '': - j+=1 - try: - elems = line.split('\t') - #if the start and/or end columns are not numbers, skip that line. - assert int(elems[start_col_1]) - assert int(elems[end_col_1]) - if strand_col_1 != -1: - strand = elems[strand_col_1] - #if the stand value is not + or -, skip that line. - assert strand in ['+', '-'] - if direction == 'Upstream': - if strand == '+': - if region == 'end': - elems[end_col_1] = str(int(elems[end_col_1]) + offset) - elems[start_col_1] = str( int(elems[end_col_1]) - size ) - else: - elems[end_col_1] = str(int(elems[start_col_1]) + offset) - elems[start_col_1] = str( int(elems[end_col_1]) - size ) - elif strand == '-': - if region == 'end': - elems[start_col_1] = str(int(elems[start_col_1]) - offset) - elems[end_col_1] = str(int(elems[start_col_1]) + size) - else: - elems[start_col_1] = str(int(elems[end_col_1]) - offset) - elems[end_col_1] = str(int(elems[start_col_1]) + size) - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - - elif direction == 'Downstream': - if strand == '-': - if region == 'start': - elems[end_col_1] = str(int(elems[end_col_1]) - offset) - elems[start_col_1] = str( int(elems[end_col_1]) - size ) - else: - elems[end_col_1] = str(int(elems[start_col_1]) - offset) - elems[start_col_1] = str( int(elems[end_col_1]) - size ) - elif strand == '+': - if region == 'start': - elems[start_col_1] = str(int(elems[start_col_1]) + offset) - elems[end_col_1] = str(int(elems[start_col_1]) + size) - else: - elems[start_col_1] = str(int(elems[end_col_1]) + offset) - elems[end_col_1] = str(int(elems[start_col_1]) + size) - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - - elif direction == 'Both': - if strand == '-': - if region == 'start': - start = str(int(elems[end_col_1]) - offset) - end1 = str(int(start) + size) - end2 = str(int(start) - size) - elems[start_col_1]=start - elems[end_col_1]=end1 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elems[start_col_1]=end2 - elems[end_col_1]=start - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elif region == 'end': - start = str(int(elems[start_col_1]) - offset) - end1 = str(int(start) + size) - end2 = str(int(start) - size) - elems[start_col_1]=start - elems[end_col_1]=end1 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elems[start_col_1]=end2 - elems[end_col_1]=start - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - else: - start1 = str(int(elems[end_col_1]) - offset) - end1 = str(int(start1) + size) - start2 = str(int(elems[start_col_1]) - offset) - end2 = str(int(start2) - size) - elems[start_col_1]=start1 - elems[end_col_1]=end1 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elems[start_col_1]=end2 - elems[end_col_1]=start2 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elif strand == '+': - if region == 'start': - start = str(int(elems[start_col_1]) + offset) - end1 = str(int(start) - size) - end2 = str(int(start) + size) - elems[start_col_1]=end1 - elems[end_col_1]=start - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elems[start_col_1]=start - elems[end_col_1]=end2 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elif region == 'end': - start = str(int(elems[end_col_1]) + offset) - end1 = str(int(start) - size) - end2 = str(int(start) + size) - elems[start_col_1]=end1 - elems[end_col_1]=start - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elems[start_col_1]=start - elems[end_col_1]=end2 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - else: - start1 = str(int(elems[start_col_1]) + offset) - end1 = str(int(start1) - size) - start2 = str(int(elems[end_col_1]) + offset) - end2 = str(int(start2) + size) - elems[start_col_1]=end1 - elems[end_col_1]=start1 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - elems[start_col_1]=start2 - elems[end_col_1]=end2 - assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0 - fo.write( "%s\n" % '\t'.join( elems ) ) - except: - skipped_lines += 1 - if not invalid_line: - first_invalid_line = i + 1 - invalid_line = line - fo.close() - - if skipped_lines == j: - stop_err( "Data issue: click the pencil icon in the history item to correct the metadata attributes." ) - if skipped_lines > 0: - print 'Skipped %d invalid lines starting with #%dL "%s"' % ( skipped_lines, first_invalid_line, invalid_line ) - print 'Location: %s, Region: %s, Flank-length: %d, Offset: %d ' %( direction, region, size, offset ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/get_flanks.xml b/tools/new_operations/get_flanks.xml deleted file mode 100644 index 2c4a0e5b852..00000000000 --- a/tools/new_operations/get_flanks.xml +++ /dev/null @@ -1,78 +0,0 @@ - - returns flanking region/s for every gene - get_flanks.py $input $out_file1 $size $direction $region -o $offset -l ${input.metadata.chromCol},${input.metadata.startCol},${input.metadata.endCol},${input.metadata.strandCol} - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -This tool finds the upstream and/or downstream flanking region(s) of all the selected regions in the input file. - -**Note:** Every line should contain at least 3 columns: Chromosome number, Start and Stop co-ordinates. If any of these columns is missing or if start and stop co-ordinates are not numerical, the tool may encounter exceptions and such lines are skipped as invalid. The number of invalid skipped lines is documented in the resulting history item as a "Data issue". - ------ - - -**Example 1** - -- For the following dataset:: - - chr22 1000 7000 NM_174568 0 + - -- running get flanks with Region: Around start, Offset: -200, Flank-length: 300 and Location: Upstream will return **(Red: Dataset positive strand; Blue: Flanks output)**:: - - chr22 500 800 NM_174568 0 + - -.. image:: ${static_path}/operation_icons/flanks_ex1.gif - -**Example 2** - -- For the following dataset:: - - chr22 1000 7000 NM_028946 0 - - -- running get flanks with Region: Whole, Offset: 200, Flank-length: 300 and Location: Downstream will return **(Orange: Dataset negative strand; Magenta: Flanks output)**:: - - chr22 500 800 NM_028946 0 - - -.. image:: ${static_path}/operation_icons/flanks_ex2.gif - - - - - diff --git a/tools/new_operations/gops_basecoverage.py b/tools/new_operations/gops_basecoverage.py deleted file mode 100644 index da78701afa9..00000000000 --- a/tools/new_operations/gops_basecoverage.py +++ /dev/null @@ -1,50 +0,0 @@ -#!/usr/bin/env python -""" -Count total base coverage. - -usage: %prog in_file out_file - -1, --cols1=N,N,N,N: Columns for start, end, strand in first file -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.base_coverage import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - in_fname, out_fname = args - except: - doc_optparse.exception() - - g1 = NiceReaderWrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col = strand_col_1, - fix_strand=True ) - - try: - bases = base_coverage(g1) - except ParseError, exc: - fail( "Invalid file format: %s" % str( exc ) ) - out_file = open( out_fname, "w" ) - out_file.write( "%s\n" % str( bases ) ) - out_file.close() - if g1.skipped > 0: - print skipped( g1, filedesc="" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_cluster.py b/tools/new_operations/gops_cluster.py deleted file mode 100644 index ac2f4f75d0a..00000000000 --- a/tools/new_operations/gops_cluster.py +++ /dev/null @@ -1,132 +0,0 @@ -#!/usr/bin/env python -""" -Cluster regions of intervals. - -usage: %prog in_file out_file - -1, --cols1=N,N,N,N: Columns for start, end, strand in file - -d, --distance=N: Maximum distance between clustered intervals - -v, --overlap=N: Minimum overlap require (negative distance) - -m, --minregions=N: Minimum regions per cluster - -o, --output=N: 1)merged 2)filtered 3)clustered 4) minimum 5) maximum -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.find_clusters import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - distance = 0 - minregions = 2 - output = 1 - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - if options.distance: distance = int( options.distance ) - if options.overlap: distance = -1 * int( options.overlap ) - if options.output: output = int( options.output ) - if options.minregions: minregions = int( options.minregions ) - in_fname, out_fname = args - except: - doc_optparse.exception() - - g1 = NiceReaderWrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - - # Get the cluster tree - try: - clusters, extra = find_clusters( g1, mincols=distance, minregions=minregions) - except ParseError, exc: - fail( "Invalid file format: %s" % str( exc ) ) - - f1 = open( in_fname, "r" ) - out_file = open( out_fname, "w" ) - - # If "merge" - if output == 1: - fields = ["." for x in range(max(g1.chrom_col, g1.start_col, g1.end_col)+1)] - for chrom, tree in clusters.items(): - for start, end, lines in tree.getregions(): - fields[g1.chrom_col] = chrom - fields[g1.start_col] = str(start) - fields[g1.end_col] = str(end) - out_file.write( "%s\n" % "\t".join( fields ) ) - - # If "filtered" we preserve order of file and comments, etc. - if output == 2: - linenums = dict() - for chrom, tree in clusters.items(): - for linenum in tree.getlines(): - linenums[linenum] = 0 - linenum = -1 - f1.seek(0) - for line in f1.readlines(): - linenum += 1 - if linenum in linenums or linenum in extra: - out_file.write( "%s\n" % line.rstrip( "\n\r" ) ) - - # If "clustered" we output original intervals, but near each other (i.e. clustered) - if output == 3: - linenums = list() - f1.seek(0) - fileLines = f1.readlines() - for chrom, tree in clusters.items(): - for linenum in tree.getlines(): - out_file.write( "%s\n" % fileLines[linenum].rstrip( "\n\r" ) ) - - # If "minimum" we output the smallest interval in each cluster - if output == 4 or output == 5: - linenums = list() - f1.seek(0) - fileLines = f1.readlines() - for chrom, tree in clusters.items(): - regions = tree.getregions() - for start, end, lines in tree.getregions(): - outsize = -1 - outinterval = None - for line in lines: - # three nested for loops? - # should only execute this code once per line - fileline = fileLines[line].rstrip("\n\r") - try: - cluster_interval = GenomicInterval( g1, fileline.split("\t"), - g1.chrom_col, - g1.start_col, - g1.end_col, - g1.strand_col, - g1.default_strand, - g1.fix_strand ) - except Exception, exc: - print >> sys.stderr, str( exc ) - f1.close() - sys.exit() - interval_size = cluster_interval.end - cluster_interval.start - if outsize == -1 or \ - ( outsize > interval_size and output == 4 ) or \ - ( outsize < interval_size and output == 5 ) : - outinterval = cluster_interval - outsize = interval_size - out_file.write( "%s\n" % outinterval ) - - f1.close() - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc="" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_complement.py b/tools/new_operations/gops_complement.py deleted file mode 100644 index 7615f83f4e2..00000000000 --- a/tools/new_operations/gops_complement.py +++ /dev/null @@ -1,98 +0,0 @@ -#!/usr/bin/env python -""" -Complement regions. - -usage: %prog in_file out_file - -1, --cols1=N,N,N,N: Columns for chrom, start, end, strand in file - -l, --lengths=N: Filename of .len file for species (chromosome lengths) - -a, --all: Complement all chromosomes (Genome-wide complement) -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.complement import complement -from bx.intervals.operations.subtract import subtract -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - allchroms = False - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - lengths = options.lengths - if options.all: allchroms = True - in_fname, out_fname = args - except: - doc_optparse.exception() - - g1 = NiceReaderWrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - - lens = dict() - chroms = list() - # dbfile is used to determine the length of each chromosome. The lengths - # are added to the lens dict and passed copmlement operation code in bx. - dbfile = fileinput.FileInput( lengths ) - - if dbfile: - if not allchroms: - try: - for line in dbfile: - fields = line.split("\t") - lens[fields[0]] = int(fields[1]) - except: - # assume LEN doesn't exist or is corrupt somehow - pass - elif allchroms: - try: - for line in dbfile: - fields = line.split("\t") - end = int(fields[1]) - chroms.append("\t".join([fields[0],"0",str(end)])) - except: - pass - - # Safety...if the dbfile didn't exist and we're on allchroms, then - # default to generic complement - if allchroms and len(chroms) == 0: - allchroms = False - - if allchroms: - chromReader = GenomicIntervalReader(chroms) - generator = subtract([chromReader, g1]) - else: - generator = complement(g1, lens) - - out_file = open( out_fname, "w" ) - - try: - for interval in generator: - if type( interval ) is GenomicInterval: - out_file.write( "%s\n" % "\t".join( interval ) ) - else: - out_file.write( "%s\n" % interval ) - except ParseError, exc: - out_file.close() - fail( "Invalid file format: %s" % str( exc ) ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc="" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_concat.py b/tools/new_operations/gops_concat.py deleted file mode 100644 index 9490c41401d..00000000000 --- a/tools/new_operations/gops_concat.py +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env python -""" -Concatenate two bed files. The concatenated files are returned in the -same format as the first. If --sameformat is specified, then all -columns will be treated as the same, and all fields will be saved, -although the output will be trimmed to match the primary input. In -addition, if --sameformat is specified, missing fields will be padded -with a period(.). - -usage: %prog in_file_1 in_file_2 out_file - -1, --cols1=N,N,N,N: Columns for chrom, start, end, strand in first file - -2, --cols2=N,N,N,N: Columns for chrom, start, end, strand in second file - -s, --sameformat: All files are precisely the same format. -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.concat import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - sameformat=False - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col_2 = parse_cols_arg( options.cols2 ) - if options.sameformat: sameformat = True - in_file_1, in_file_2, out_fname = args - except: - doc_optparse.exception() - - g1 = NiceReaderWrapper( fileinput.FileInput( in_file_1 ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - - g2 = NiceReaderWrapper( fileinput.FileInput( in_file_2 ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - - out_file = open( out_fname, "w" ) - - try: - for line in concat( [g1, g2], sameformat=sameformat ): - if type( line ) is GenomicInterval: - out_file.write( "%s\n" % "\t".join( line.fields ) ) - else: - out_file.write( "%s\n" % line ) - except ParseError, exc: - out_file.close() - fail( "Invalid file format: %s" % str( exc ) ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc=" of 1st dataset" ) - if g2.skipped > 0: - print skipped( g2, filedesc=" of 2nd dataset" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_coverage.py b/tools/new_operations/gops_coverage.py deleted file mode 100644 index 91b3ad19803..00000000000 --- a/tools/new_operations/gops_coverage.py +++ /dev/null @@ -1,68 +0,0 @@ -#!/usr/bin/env python -""" -Calculate coverage of one query on another, and append the coverage to -the last two columns as bases covered and percent coverage. - -usage: %prog bed_file_1 bed_file_2 out_file - -1, --cols1=N,N,N,N: Columns for start, end, strand in first file - -2, --cols2=N,N,N,N: Columns for start, end, strand in second file -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.coverage import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col_2 = parse_cols_arg( options.cols2 ) - in_fname, in2_fname, out_fname = args - except: - doc_optparse.exception() - - g1 = NiceReaderWrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - g2 = NiceReaderWrapper( fileinput.FileInput( in2_fname ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - - out_file = open( out_fname, "w" ) - - try: - for line in coverage( [g1,g2] ): - if type( line ) is GenomicInterval: - out_file.write( "%s\n" % "\t".join( line.fields ) ) - else: - out_file.write( "%s\n" % line ) - except ParseError, exc: - out_file.close() - fail( "Invalid file format: %s" % str( exc ) ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc=" of 1st dataset" ) - if g2.skipped > 0: - print skipped( g2, filedesc=" of 2nd dataset" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_intersect.py b/tools/new_operations/gops_intersect.py deleted file mode 100755 index 11a0c80f902..00000000000 --- a/tools/new_operations/gops_intersect.py +++ /dev/null @@ -1,98 +0,0 @@ -#!/usr/bin/env python -""" -Find regions of first interval file that overlap regions in a second interval file. -Interval files can either be BED or GFF format. - -usage: %prog interval_file_1 interval_file_2 out_file - -1, --cols1=N,N,N,N: Columns for start, end, strand in first file - -2, --cols2=N,N,N,N: Columns for start, end, strand in second file - -m, --mincols=N: Require this much overlap (default 1bp) - -p, --pieces: just print pieces of second set (after padding) - -G, --gff1: input 1 is GFF format, meaning start and end coordinates are 1-based, closed interval - -H, --gff2: input 2 is GFF format, meaning start and end coordinates are 1-based, closed interval -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.intersect import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * -from galaxy.datatypes.util.gff_util import GFFFeature, GFFReaderWrapper, convert_bed_coords_to_gff - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - mincols = 1 - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col_2 = parse_cols_arg( options.cols2 ) - if options.mincols: mincols = int( options.mincols ) - pieces = bool( options.pieces ) - in1_gff_format = bool( options.gff1 ) - in2_gff_format = bool( options.gff2 ) - in_fname, in2_fname, out_fname = args - except: - doc_optparse.exception() - - # Set readers to handle either GFF or default format. - if in1_gff_format: - in1_reader_wrapper = GFFReaderWrapper - else: - in1_reader_wrapper = NiceReaderWrapper - if in2_gff_format: - in2_reader_wrapper = GFFReaderWrapper - else: - in2_reader_wrapper = NiceReaderWrapper - - g1 = in1_reader_wrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - if in1_gff_format: - # Intersect requires coordinates in BED format. - g1.convert_to_bed_coord=True - g2 = in2_reader_wrapper( fileinput.FileInput( in2_fname ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - if in2_gff_format: - # Intersect requires coordinates in BED format. - g2.convert_to_bed_coord=True - - out_file = open( out_fname, "w" ) - try: - for feature in intersect( [g1,g2], pieces=pieces, mincols=mincols ): - if isinstance( feature, GFFFeature ): - # Convert back to GFF coordinates since reader converted automatically. - convert_bed_coords_to_gff( feature ) - for interval in feature.intervals: - out_file.write( "%s\n" % "\t".join( interval.fields ) ) - elif isinstance( feature, GenomicInterval ): - out_file.write( "%s\n" % "\t".join( feature.fields ) ) - else: - out_file.write( "%s\n" % feature ) - except ParseError, e: - out_file.close() - fail( "Invalid file format: %s" % str( e ) ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc=" of 1st dataset" ) - if g2.skipped > 0: - print skipped( g2, filedesc=" of 2nd dataset" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_join.py b/tools/new_operations/gops_join.py deleted file mode 100644 index daab6cb18e0..00000000000 --- a/tools/new_operations/gops_join.py +++ /dev/null @@ -1,82 +0,0 @@ -#!/usr/bin/env python -""" -Join two sets of intervals using their overlap as the key. - -usage: %prog bed_file_1 bed_file_2 out_file - -1, --cols1=N,N,N,N: Columns for start, end, strand in first file - -2, --cols2=N,N,N,N: Columns for start, end, strand in second file - -m, --mincols=N: Require this much overlap (default 1bp) - -f, --fill=N: none, right, left, both -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.join import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - mincols = 1 - upstream_pad = 0 - downstream_pad = 0 - leftfill = False - rightfill = False - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col_2 = parse_cols_arg( options.cols2 ) - if options.mincols: mincols = int( options.mincols ) - if options.fill: - if options.fill == "both": - rightfill = leftfill = True - else: - rightfill = options.fill == "right" - leftfill = options.fill == "left" - in_fname, in2_fname, out_fname = args - except: - doc_optparse.exception() - - g1 = NiceReaderWrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - g2 = NiceReaderWrapper( fileinput.FileInput( in2_fname ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - - out_file = open( out_fname, "w" ) - - try: - for outfields in join(g1, g2, mincols=mincols, rightfill=rightfill, leftfill=leftfill): - if type( outfields ) is list: - out_file.write( "%s\n" % "\t".join( outfields ) ) - else: - out_file.write( "%s\n" % outfields ) - except ParseError, exc: - out_file.close() - fail( "Invalid file format: %s" % str( exc ) ) - except MemoryError: - out_file.close() - fail( "Input datasets were too large to complete the join operation." ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc=" of 1st dataset" ) - if g2.skipped > 0: - print skipped( g2, filedesc=" of 2nd dataset" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_merge.py b/tools/new_operations/gops_merge.py deleted file mode 100644 index 85d215f83f4..00000000000 --- a/tools/new_operations/gops_merge.py +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env python -""" -Merge overlaping regions. - -usage: %prog in_file out_file - -1, --cols1=N,N,N,N: Columns for start, end, strand in first file - -m, --mincols=N: Require this much overlap (default 1bp) - -3, --threecol: Output 3 column bed -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.merge import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - mincols = 1 - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - if options.mincols: mincols = int( options.mincols ) - in_fname, out_fname = args - except: - doc_optparse.exception() - - g1 = NiceReaderWrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col = strand_col_1, - fix_strand=True ) - - out_file = open( out_fname, "w" ) - - try: - for line in merge(g1,mincols=mincols): - if options.threecol: - if type( line ) is GenomicInterval: - out_file.write( "%s\t%s\t%s\n" % ( line.chrom, str( line.startCol ), str( line.endCol ) ) ) - elif type( line ) is list: - out_file.write( "%s\t%s\t%s\n" % ( line[chr_col_1], str( line[start_col_1] ), str( line[end_col_1] ) ) ) - else: - out_file.write( "%s\n" % line ) - else: - if type( line ) is GenomicInterval: - out_file.write( "%s\n" % "\t".join( line.fields ) ) - elif type( line ) is list: - out_file.write( "%s\n" % "\t".join( line ) ) - else: - out_file.write( "%s\n" % line ) - except ParseError, exc: - out_file.close() - fail( "Invalid file format: %s" % str( exc ) ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc=" of 1st dataset" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/gops_subtract.py b/tools/new_operations/gops_subtract.py deleted file mode 100644 index 9a8aa0d66ad..00000000000 --- a/tools/new_operations/gops_subtract.py +++ /dev/null @@ -1,99 +0,0 @@ -#!/usr/bin/env python -""" -Find regions of first interval file that do not overlap regions in a second -interval file. Interval files can either be BED or GFF format. - -usage: %prog interval_file_1 interval_file_2 out_file - -1, --cols1=N,N,N,N: Columns for start, end, strand in first file - -2, --cols2=N,N,N,N: Columns for start, end, strand in second file - -m, --mincols=N: Require this much overlap (default 1bp) - -p, --pieces: just print pieces of second set (after padding) - -G, --gff1: input 1 is GFF format, meaning start and end coordinates are 1-based, closed interval - -H, --gff2: input 2 is GFF format, meaning start and end coordinates are 1-based, closed interval -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, traceback, fileinput -from warnings import warn -from bx.intervals import * -from bx.intervals.io import * -from bx.intervals.operations.subtract import * -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * -from galaxy.datatypes.util.gff_util import GFFFeature, GFFReaderWrapper, convert_bed_coords_to_gff - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - mincols = 1 - upstream_pad = 0 - downstream_pad = 0 - - options, args = doc_optparse.parse( __doc__ ) - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col_2 = parse_cols_arg( options.cols2 ) - if options.mincols: mincols = int( options.mincols ) - pieces = bool( options.pieces ) - in1_gff_format = bool( options.gff1 ) - in2_gff_format = bool( options.gff2 ) - in_fname, in2_fname, out_fname = args - except: - doc_optparse.exception() - - # Set readers to handle either GFF or default format. - if in1_gff_format: - in1_reader_wrapper = GFFReaderWrapper - else: - in1_reader_wrapper = NiceReaderWrapper - if in2_gff_format: - in2_reader_wrapper = GFFReaderWrapper - else: - in2_reader_wrapper = NiceReaderWrapper - - g1 = in1_reader_wrapper( fileinput.FileInput( in_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - if in1_gff_format: - # Subtract requires coordinates in BED format. - g1.convert_to_bed_coord=True - - g2 = in2_reader_wrapper( fileinput.FileInput( in2_fname ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - if in2_gff_format: - # Subtract requires coordinates in BED format. - g2.convert_to_bed_coord=True - - out_file = open( out_fname, "w" ) - try: - for feature in subtract( [g1,g2], pieces=pieces, mincols=mincols ): - if isinstance( feature, GFFFeature ): - # Convert back to GFF coordinates since reader converted automatically. - convert_bed_coords_to_gff( feature ) - for interval in feature.intervals: - out_file.write( "%s\n" % "\t".join( interval.fields ) ) - elif isinstance( feature, GenomicInterval ): - out_file.write( "%s\n" % "\t".join( feature.fields ) ) - else: - out_file.write( "%s\n" % feature ) - except ParseError, exc: - out_file.close() - fail( "Invalid file format: %s" % str( exc ) ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc=" of 2nd dataset" ) - if g2.skipped > 0: - print skipped( g2, filedesc=" of 1st dataset" ) - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/intersect.xml b/tools/new_operations/intersect.xml deleted file mode 100644 index 642a94d34cd..00000000000 --- a/tools/new_operations/intersect.xml +++ /dev/null @@ -1,143 +0,0 @@ - - the intervals of two datasets - gops_intersect.py - $input1 $input2 $output - - #if isinstance( $input1.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gff').__class__): - -1 1,4,5,7 --gff1 - #else: - -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} - #end if - - #if isinstance( $input2.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gff').__class__): - -2 1,4,5,7 --gff2 - #else: - -2 ${input2.metadata.chromCol},${input2.metadata.startCol},${input2.metadata.endCol},${input2.metadata.strandCol} - #end if - - -m $min $returntype - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu, it means that it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -**Syntax** - -- **Where overlap is at least** sets the minimum length (in base pairs) of overlap between elements of the two datasets -- **Overlapping Intervals** returns entire intervals from the first dataset that overlap the second dataset. The returned intervals are completely unchanged, and this option only filters out intervals that do not overlap with the second dataset. -- **Overlapping pieces of Intervals** returns intervals that indicate the exact base pair overlap between the first dataset and the second dataset. The intervals returned are from the first dataset, and all fields besides start and end are guaranteed to remain unchanged. - ------ - -**Examples** - -Overlapping Intervals: - -.. image:: ${static_path}/operation_icons/gops_intersectOverlappingIntervals.gif - -Overlapping Pieces of Intervals: - -.. image:: ${static_path}/operation_icons/gops_intersectOverlappingPieces.gif - - - diff --git a/tools/new_operations/join.xml b/tools/new_operations/join.xml deleted file mode 100644 index 4d5331ec995..00000000000 --- a/tools/new_operations/join.xml +++ /dev/null @@ -1,117 +0,0 @@ - - the intervals of two datasets side-by-side - gops_join.py $input1 $input2 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} -2 ${input2.metadata.chromCol},${input2.metadata.startCol},${input2.metadata.endCol},${input2.metadata.strandCol} -m $min -f $fill - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu, it means that it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -**Syntax** - -- **Where overlap** specifies the minimum overlap between intervals that allows them to be joined. -- **Return only records that are joined** returns only the records of the first dataset that join to a record in the second dataset. This is analogous to an INNER JOIN. -- **Return all records of first dataset (fill null with ".")** returns all intervals of the first dataset, and any intervals that do not join an interval from the second dataset are filled in with a period(.). This is analogous to a LEFT JOIN. -- **Return all records of second dataset (fill null with ".")** returns all intervals of the second dataset, and any intervals that do not join an interval from the first dataset are filled in with a period(.). **Note that this may produce an invalid interval file, since a period(.) is not a valid chrom, start, end or strand.** -- **Return all records of both datasets (fill nulls with ".")** returns all records from both datasets, and fills on either the right or left with periods. **Note that this may produce an invalid interval file, since a period(.) is not a valid chrom, start, end or strand.** - ------ - -**Examples** - -.. image:: ${static_path}/operation_icons/gops_joinRecordsList.gif - -Only records that are joined (inner join): - -.. image:: ${static_path}/operation_icons/gops_joinInner.gif - -All records of first dataset: - -.. image:: ${static_path}/operation_icons/gops_joinLeftOuter.gif - -All records of second dataset: - -.. image:: ${static_path}/operation_icons/gops_joinRightOuter.gif - -All records of both datasets: - -.. image:: ${static_path}/operation_icons/gops_joinFullOuter.gif - - - - diff --git a/tools/new_operations/merge.xml b/tools/new_operations/merge.xml deleted file mode 100644 index 33c2abc096f..00000000000 --- a/tools/new_operations/merge.xml +++ /dev/null @@ -1,58 +0,0 @@ - - the overlapping intervals of a dataset - gops_merge.py $input1 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} $returntype - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu, it means that it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -This operation merges all overlapping intervals into single intervals. - -**Example** - -.. image:: ${static_path}/operation_icons/gops_merge.gif - - - \ No newline at end of file diff --git a/tools/new_operations/operation_filter.py b/tools/new_operations/operation_filter.py deleted file mode 100644 index a44a043cbd6..00000000000 --- a/tools/new_operations/operation_filter.py +++ /dev/null @@ -1,99 +0,0 @@ -# runs after the job (and after the default post-filter) -import os -from galaxy import eggs -from galaxy import jobs -from galaxy.tools.parameters import DataToolParameter - -from galaxy.jobs.handler import JOB_ERROR - -# Older py compatibility -try: - set() -except: - from sets import Set as set - -#def exec_before_process(app, inp_data, out_data, param_dict, tool=None): -# """Sets the name of the data""" -# dbkeys = sets.Set( [data.dbkey for data in inp_data.values() ] ) -# if len(dbkeys) != 1: -# raise Exception, '

Both Queries must be from the same genome build

' - -def validate_input( trans, error_map, param_values, page_param_map ): - dbkeys = set() - data_param_names = set() - data_params = 0 - for name, param in page_param_map.iteritems(): - if isinstance( param, DataToolParameter ): - # for each dataset parameter - if param_values.get(name, None) != None: - dbkeys.add( param_values[name].dbkey ) - data_params += 1 - # check meta data - try: - param = param_values[name] - if isinstance( param.datatype, trans.app.datatypes_registry.get_datatype_by_extension( 'gff' ).__class__ ): - # TODO: currently cannot validate GFF inputs b/c they are not derived from interval. - pass - else: # Validate interval datatype. - startCol = int( param.metadata.startCol ) - endCol = int( param.metadata.endCol ) - chromCol = int( param.metadata.chromCol ) - if param.metadata.strandCol is not None: - strandCol = int ( param.metadata.strandCol ) - else: - strandCol = 0 - except: - error_msg = "The attributes of this dataset are not properly set. " + \ - "Click the pencil icon in the history item to set the chrom, start, end and strand columns." - error_map[name] = error_msg - data_param_names.add( name ) - if len( dbkeys ) > 1: - for name in data_param_names: - error_map[name] = "All datasets must belong to same genomic build, " \ - "this dataset is linked to build '%s'" % param_values[name].dbkey - if data_params != len(data_param_names): - for name in data_param_names: - error_map[name] = "A dataset of the appropriate type is required" - -# Commented out by INS, 5/30/2007. What is the PURPOSE of this? -def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): - """Verify the output data after each run""" - items = out_data.items() - - for name, data in items: - try: - if stderr and len( stderr ) > 0: - raise Exception( stderr ) - - except Exception, exc: - data.blurb = JOB_ERROR - data.state = JOB_ERROR - -## def exec_after_process(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): -## pass - - -def exec_after_merge(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): - exec_after_process( - app, inp_data, out_data, param_dict, tool=tool, stdout=stdout, stderr=stderr) - - # strip strand column if clusters were merged - items = out_data.items() - for name, data in items: - if param_dict['returntype'] == True: - data.metadata.chromCol = 1 - data.metadata.startCol = 2 - data.metadata.endCol = 3 - # merge always clobbers strand - data.metadata.strandCol = None - - -def exec_after_cluster(app, inp_data, out_data, param_dict, tool=None, stdout=None, stderr=None): - exec_after_process( - app, inp_data, out_data, param_dict, tool=tool, stdout=stdout, stderr=stderr) - - # strip strand column if clusters were merged - if param_dict["returntype"] == '1': - items = out_data.items() - for name, data in items: - data.metadata.strandCol = None diff --git a/tools/new_operations/subtract.xml b/tools/new_operations/subtract.xml deleted file mode 100644 index 9ab69748252..00000000000 --- a/tools/new_operations/subtract.xml +++ /dev/null @@ -1,124 +0,0 @@ - - the intervals of two datasets - gops_subtract.py - $input1 $input2 $output - - #if isinstance( $input1.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gff').__class__): - -1 1,4,5,7 --gff1 - #else: - -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} - #end if - - #if isinstance( $input2.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gff').__class__): - -2 1,4,5,7 --gff2 - #else: - -2 ${input2.metadata.chromCol},${input2.metadata.startCol},${input2.metadata.endCol},${input2.metadata.strandCol} - #end if - - -m $min $returntype - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your dataset does not appear in the pulldown menu, it means that it is not in interval format. Use "edit attributes" to set chromosome, start, end, and strand columns. - ------ - -**Screencasts!** - -See Galaxy Interval Operation Screencasts_ (right click to open this link in another window). - -.. _Screencasts: http://wiki.g2.bx.psu.edu/Learn/Interval%20Operations - ------ - -**Syntax** - -- **Where overlap is at least** sets the minimum length (in base pairs) of overlap between elements of the two datasets. -- **Intervals with no overlap** returns entire intervals from the first dataset that do not overlap the second dataset. The returned intervals are completely unchanged, and this option only filters out intervals that overlap with the second dataset. -- **Non-overlapping pieces of intervals** returns intervals from the first dataset that have the intervals from the second dataset removed. Any overlapping base pairs are removed from the range of the interval. All fields besides start and end are guaranteed to remain unchanged. - ------ - -**Example** - -Intervals with no overlap: - -.. image:: ${static_path}/operation_icons/gops_subtractOverlappingIntervals.gif - -Non-overlapping pieces of intervals: - -.. image:: ${static_path}/operation_icons/gops_subtractOverlappingPieces.gif - - - diff --git a/tools/new_operations/subtract_query.py b/tools/new_operations/subtract_query.py deleted file mode 100644 index b06440dd0ea..00000000000 --- a/tools/new_operations/subtract_query.py +++ /dev/null @@ -1,113 +0,0 @@ -#!/usr/bin/env python -# Greg Von Kuster - -""" -Subtract an entire query from another query -usage: %prog in_file_1 in_file_2 begin_col end_col output - --ignore-empty-end-cols: ignore empty end columns when subtracting -""" -import sys, re -from galaxy import eggs -import pkg_resources; pkg_resources.require( "bx-python" ) -from bx.cookbook import doc_optparse - -# Older py compatibility -try: - set() -except: - from sets import Set as set - -assert sys.version_info[:2] >= ( 2, 4 ) - -def get_lines(fname, begin_col='', end_col='', ignore_empty_end_cols=False): - lines = set([]) - i = 0 - for i, line in enumerate(file(fname)): - line = line.rstrip('\r\n') - if line and not line.startswith('#'): - if begin_col and end_col: - """Both begin_col and end_col must be integers at this point.""" - try: - line = line.split('\t') - line = '\t'.join([line[j] for j in range(begin_col-1, end_col)]) - if ignore_empty_end_cols: - # removing empty fields, we do not compare empty fields at the end of a line. - line = line.rstrip() - lines.add( line ) - except: pass - else: - if ignore_empty_end_cols: - # removing empty fields, we do not compare empty fields at the end of a line. - line = line.rstrip() - lines.add( line ) - if i: return (i+1, lines) - else: return (i, lines) - -def main(): - - # Parsing Command Line here - options, args = doc_optparse.parse( __doc__ ) - - try: - inp1_file, inp2_file, begin_col, end_col, out_file = args - except: - doc_optparse.exception() - - begin_col = begin_col.strip() - end_col = end_col.strip() - - if begin_col != 'None' or end_col != 'None': - """ - The user selected columns for restriction. We'll allow default - values for both begin_col and end_col as long as the user selected - at least one of them for restriction. - """ - if begin_col == 'None': - begin_col = end_col - elif end_col == 'None': - end_col = begin_col - begin_col = int(begin_col) - end_col = int(end_col) - """Make sure that begin_col <= end_col (switch if not)""" - if begin_col > end_col: - tmp_col = end_col - end_col = begin_col - begin_col = tmp_col - else: - begin_col = end_col = '' - - try: - fo = open(out_file,'w') - except: - print >> sys.stderr, "Unable to open output file" - sys.exit() - - """ - len1 is the number of lines in inp1_file - lines1 is the set of unique lines in inp1_file - diff1 is the number of duplicate lines removed from inp1_file - """ - len1, lines1 = get_lines(inp1_file, begin_col, end_col, options.ignore_empty_end_cols) - diff1 = len1 - len(lines1) - len2, lines2 = get_lines(inp2_file, begin_col, end_col, options.ignore_empty_end_cols) - - lines1.difference_update(lines2) - """lines1 is now the set of unique lines in inp1_file - the set of unique lines in inp2_file""" - - for line in lines1: - print >> fo, line - - fo.close() - - info_msg = 'Subtracted %d lines. ' %((len1 - diff1) - len(lines1)) - - if begin_col and end_col: - info_msg += 'Restricted to columns c' + str(begin_col) + ' thru c' + str(end_col) + '. ' - - if diff1 > 0: - info_msg += 'Eliminated %d duplicate/blank/comment/invalid lines from first query.' %diff1 - - print info_msg - -if __name__ == "__main__": - main() diff --git a/tools/new_operations/subtract_query.xml b/tools/new_operations/subtract_query.xml deleted file mode 100644 index a603ce144be..00000000000 --- a/tools/new_operations/subtract_query.xml +++ /dev/null @@ -1,126 +0,0 @@ - - from another dataset - - subtract_query.py $input1 $input2 $begin_col $end_col $output - #if str($ignore_empty_end_cols) == 'true': - --ignore-empty-end-cols - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** This tool complements the tool in the **Operate on Genomic Intervals** tool set which subtracts the intervals of two datasets. - - ------ - -**Syntax** - -This tool subtracts an entire dataset from another dataset. - -- Any text format is valid. -- If both dataset formats are tabular, you may restrict the subtraction to specific columns **contained in both datasets** and the resulting dataset will include only the columns specified. -- The begin column must be less than or equal to the end column. If it is not, begin column is switched with end column. -- If begin column is specified but end column is not, end column will default to begin_column (and vice versa). -- All blank and comment lines are skipped and not included in the resulting dataset (comment lines are lines beginning with a # character). -- Duplicate lines are eliminated from both dataset prior to subtraction. If any duplicate lines were eliminated from the first dataset, the number is displayed in the resulting history item. - ------ - -**Example** - -If this is the **First dataset**:: - - chr1 4225 19670 - chr10 6 8 - chr1 24417 24420 - chr6_hla_hap2 0 150 - chr2 1 5 - chr10 2 10 - chr1 30 55 - chrY 1 20 - chr1 1225979 42287290 - chr10 7 8 - -and this is the **Second dataset**:: - - chr1 4225 19670 - chr10 6 8 - chr1 24417 24420 - chr6_hla_hap2 0 150 - chr2 1 5 - chr1 30 55 - chrY 1 20 - chr1 1225979 42287290 - -Subtracting the **Second dataset** from the **First dataset** (including all columns) will yield:: - - chr10 7 8 - chr10 2 10 - -Conversely, subtracting the **First dataset** from the **Second dataset** (including all columns) will result in an empty dataset. - -Subtracting the **Second dataset** from the **First dataset** (restricting to columns c1 and c2) will yield:: - - chr10 7 - chr10 2 - - - \ No newline at end of file diff --git a/tools/new_operations/tables_arithmetic_operations.pl b/tools/new_operations/tables_arithmetic_operations.pl deleted file mode 100644 index e5b0ce2e3f8..00000000000 --- a/tools/new_operations/tables_arithmetic_operations.pl +++ /dev/null @@ -1,117 +0,0 @@ -# A program to implement arithmetic operations on tabular files data. The program takes three inputs: -# The first input is a TABULAR format file containing numbers only. -# The second input is a TABULAR format file containing numbers only. -# The two files must have the same number of columns and the same number of rows -# The third input is an arithmetic operation: +, -, *, or / for addition, subtraction, multiplication, or division, respectively -# The output file is a TABULAR format file containing the result of implementing the arithmetic operation on both input files. -# The output file has the same number of columns and the same number of rows as each of the two input files. -# Note: in case of division, none of the values in the second input file could be 0. - -use strict; -use warnings; - -#variables to handle information of the first input tabular file -my $lineData1 = ""; -my @lineDataArray1 = (); -my $lineArraySize = 0; -my $lineCounter1 = 0; - -#variables to handle information of the second input tabular file -my $lineData2= ""; -my @lineDataArray2 = (); -my $lineCounter2 = 0; - -my $result = 0; - -# check to make sure having the correct number of arguments -my $usage = "usage: tables_arithmetic_operations.pl [TABULAR.in] [TABULAR.in] [ArithmeticOperation] [TABULAR.out] \n"; -die $usage unless @ARGV == 4; - -#variables to store the names of input and output files -my $inputTabularFile1 = $ARGV[0]; -my $inputTabularFile2 = $ARGV[1]; -my $arithmeticOperation = $ARGV[2]; -my $outputTabularFile = $ARGV[3]; - -#open the input and output files -open (INPUT1, "<", $inputTabularFile1) || die("Could not open file $inputTabularFile1 \n"); -open (INPUT2, "<", $inputTabularFile2) || die("Could not open file $inputTabularFile2 \n"); -open (OUTPUT, ">", $outputTabularFile) || die("Could not open file $outputTabularFile \n"); - -#store the first input file in the array @motifsFrequencyData1 -my @tabularData1 = ; - -#store the second input file in the array @motifsFrequencyData2 -my @tabularData2 = ; - -#reset the $lineCounter1 to 0 -$lineCounter1 = 0; - -#iterated through the lines of the first input file -INDEL1: -foreach $lineData1 (@tabularData1){ - chomp ($lineData1); - $lineCounter1++; - - #reset the $lineCounter2 to 0 - $lineCounter2 = 0; - - #iterated through the lines of the second input file - foreach $lineData2 (@tabularData2){ - chomp ($lineData2); - $lineCounter2++; - - #check if the two motifs are the same in the two input files - if ($lineCounter1 == $lineCounter2){ - - @lineDataArray1 = split(/\t/, $lineData1); - @lineDataArray2 = split(/\t/, $lineData2); - - $lineArraySize = @lineDataArray1; - - for (my $index = 0; $index < $lineArraySize; $index++){ - - if ($arithmeticOperation eq "Addition"){ - #compute the additin of both values - $result = $lineDataArray1[$index] + $lineDataArray2[$index]; - } - - if ($arithmeticOperation eq "Subtraction"){ - #compute the subtraction of both values - $result = $lineDataArray1[$index] - $lineDataArray2[$index]; - } - - if ($arithmeticOperation eq "Multiplication"){ - #compute the multiplication of both values - $result = $lineDataArray1[$index] * $lineDataArray2[$index]; - } - - if ($arithmeticOperation eq "Division"){ - - #check if the denominator is 0 - if ($lineDataArray2[$index] != 0){ - #compute the division of both values - $result = $lineDataArray1[$index] / $lineDataArray2[$index]; - } - else{ - die("A denominator could not be zero \n"); - } - } - - #store the result in the output file - if ($index < $lineArraySize - 1){ - print OUTPUT $result . "\t"; - } - else{ - print OUTPUT $result . "\n"; - } - } - next INDEL1; - } - } -} - -#close the input and output files -close(OUTPUT); -close(INPUT2); -close(INPUT1); \ No newline at end of file diff --git a/tools/new_operations/tables_arithmetic_operations.xml b/tools/new_operations/tables_arithmetic_operations.xml deleted file mode 100644 index 0e2f891de95..00000000000 --- a/tools/new_operations/tables_arithmetic_operations.xml +++ /dev/null @@ -1,105 +0,0 @@ - - on tables - - - tables_arithmetic_operations.pl $inputFile1 $inputFile2 $inputArithmeticOperation3 $outputFile1 - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This program implements arithmetic operations on tabular files data. The program takes three inputs: - -- The first input is a TABULAR format file containing numbers only. -- The second input is a TABULAR format file containing numbers only. -- The third input is an arithmetic operation: +, -, x, or / for addition, subtraction, multiplication, or division, respectively. -- The output file is a TABULAR format file containing the result of implementing the arithmetic operation on both input files. - - -Notes: - -- The two files must have the same number of columns and the same number of rows. -- The output file has the same number of columns and the same number of rows as each of the two input files. -- In case of division, none of the values in the second input file could be 0, otherwise the program will stop and report an error. - -**Example** - -Let us have the first input file as follows:: - - 5 4 0 - 10 11 12 - 1 3 1 - 1 2 1 - 2 0 4 - -And the second input file as follows:: - - 5 4 4 - 2 5 8 - 1 2 1 - 3 2 5 - 2 4 4 - -Running the program and choosing "Addition" as an arithmetic operation will give the following output:: - - 10 8 4 - 12 16 20 - 2 5 2 - 4 4 6 - 4 4 8 - - - - - diff --git a/tools/regVariation/WeightedAverage.py b/tools/regVariation/WeightedAverage.py deleted file mode 100755 index 8c4c934ebc5..00000000000 --- a/tools/regVariation/WeightedAverage.py +++ /dev/null @@ -1,94 +0,0 @@ -#!/usr/bin/env python -""" -usage: %prog bed_file_1 bed_file_2 out_file - -1, --cols1=N,N,N,N: Columns for chr, start, end, strand in first file - -2, --cols2=N,N,N,N,N: Columns for chr, start, end, strand, name/value in second file -""" - -import collections -import sys -#import numpy -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -from galaxy.tools.util.galaxyops import * -from bx.cookbook import doc_optparse - - -#export PYTHONPATH=~/galaxy/lib/ -#running command python WeightedAverage.py interval_interpolate.bed value_interpolate.bed interpolate_result.bed - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -def FindRate(chromosome, start_stop, dictType): - OverlapList = [] - for tempO in dictType[chromosome]: - DatabaseInterval = [tempO[0], tempO[1]] - Overlap = GetOverlap( start_stop, DatabaseInterval ) - if Overlap > 0: - OverlapList.append([Overlap, tempO[2]]) - - if len(OverlapList) > 0: - SumRecomb = 0 - SumOverlap = 0 - for member in OverlapList: - SumRecomb += member[0]*member[1] - SumOverlap += member[0] - averageRate = SumRecomb/SumOverlap - return averageRate - else: - return 'NA' - - -def GetOverlap(a, b): - return min(a[1], b[1])-max(a[0], b[0]) - - -options, args = doc_optparse.parse( __doc__ ) - -try: - chr_col_1, start_col_1, end_col_1, strand_col1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col2, name_col_2 = parse_cols_arg( options.cols2 ) - input1, input2, input3 = args -except Exception, eee: - print eee - stop_err( "Data issue: click the pencil icon in the history item to correct the metadata attributes." ) - -fd2 = open(input2) -lines2 = fd2.readlines() -RecombChrDict = collections.defaultdict(list) - -skipped = 0 -for line in lines2: - temp = line.strip().split() - try: - assert float(temp[int(name_col_2)]) - except: - skipped += 1 - continue - tempIndex = [int(temp[int(start_col_2)]), int(temp[int(end_col_2)]), float(temp[int(name_col_2)])] - RecombChrDict[temp[int(chr_col_2)]].append(tempIndex) - -print "Skipped %d features with invalid values" % (skipped) - -fd1 = open(input1) -lines = fd1.readlines() -finalProduct = '' -for line in lines: - temp = line.strip().split('\t') - chromosome = temp[int(chr_col_1)] - start = int(temp[int(start_col_1)]) - stop = int(temp[int(end_col_1)]) - start_stop = [start, stop] - RecombRate = FindRate( chromosome, start_stop, RecombChrDict ) - try: - RecombRate = "%.4f" % (float(RecombRate)) - except: - RecombRate = RecombRate - finalProduct += line.strip()+'\t'+str(RecombRate)+'\n' -fdd = open(input3, 'w') -fdd.writelines(finalProduct) -fdd.close() diff --git a/tools/regVariation/WeightedAverage.xml b/tools/regVariation/WeightedAverage.xml deleted file mode 100755 index 9e9406985d2..00000000000 --- a/tools/regVariation/WeightedAverage.xml +++ /dev/null @@ -1,71 +0,0 @@ - - of the values of features overlapping an interval - WeightedAverage.py $genomic_interval $genomic_feature $out_file1 -1 ${genomic_interval.metadata.chromCol},${genomic_interval.metadata.startCol},${genomic_interval.metadata.endCol},${genomic_interval.metadata.strandCol} -2 ${genomic_feature.metadata.chromCol},${genomic_feature.metadata.startCol},${genomic_feature.metadata.endCol},${genomic_feature.metadata.strandCol},${genomic_feature.metadata.nameCol} - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -For each interval in your first dataset, this tool calculates the weighted average value of the overlapping features in your second dataset. - -- When a genomic interval partially or totally overlaps a single genomic feature, the value of that genomic feature is assigned to the genomic interval. -- When a genomic interval partially or totally overlaps with more than one genomic features, the average of the values of the overlapping genomic features weighted by the corresponding number of overlapping bases is assigned to the genomic interval. -- When a genomic interval does not overlap with any genomic feature, 'NA' will be assigned as it's value. - ------ - -.. class:: warningmark - -**Note** - -The input datasets should be in **bed** or **interval** format. Please use "edit attributes"/pencil icon to specify the column containing the values for the features in the second dataset as **name/identifier** column. - -The output will contain all the columns in the first input plus a new column containing the assigned value for each interval. - ------ - -**Example** - -- Suppose our first dataset contains the following **genomic intervals**:: - - chr start stop - chr1 1000 2000 - chr1 3000 5000 - chr1 8000 9000 - -- and our second dataset contains the following **genomic features** each having an associated value (in fourth column) :: - - chr start stop name - chr1 900 1200 0.5 - chr1 2900 3100 0.2 - chr1 4800 5100 0.8 - -- For each **genomic interval** in our first dataset, this tool calculates the weighted average value of the overlapping **genomic features** in our second dataset :: - - chr1 1000 2000 0.5 - chr1 3000 5000 0.6 - chr1 8000 9000 NA - - - - - \ No newline at end of file diff --git a/tools/regVariation/best_regression_subsets.py b/tools/regVariation/best_regression_subsets.py deleted file mode 100644 index a4883e33c3d..00000000000 --- a/tools/regVariation/best_regression_subsets.py +++ /dev/null @@ -1,91 +0,0 @@ -#!/usr/bin/env python - -from galaxy import eggs - -import sys -from rpy import * -import numpy - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -infile = sys.argv[1] -y_col = int(sys.argv[2])-1 -x_cols = sys.argv[3].split(',') -outfile = sys.argv[4] -outfile2 = sys.argv[5] -print "Predictor columns: %s; Response column: %d" % ( x_cols, y_col+1 ) -fout = open(outfile,'w') - -for i, line in enumerate( file ( infile )): - line = line.rstrip('\r\n') - if len( line )>0 and not line.startswith( '#' ): - elems = line.split( '\t' ) - break - if i == 30: - break # Hopefully we'll never get here... - -if len( elems )<1: - stop_err( "The data in your input dataset is either missing or not formatted properly." ) - -y_vals = [] -x_vals = [] - -for k, col in enumerate(x_cols): - x_cols[k] = int(col)-1 - x_vals.append([]) - -NA = 'NA' -for ind, line in enumerate( file( infile ) ): - if line and not line.startswith( '#' ): - try: - fields = line.split("\t") - try: - yval = float(fields[y_col]) - except Exception, ey: - yval = r('NA') - y_vals.append(yval) - for k, col in enumerate(x_cols): - try: - xval = float(fields[col]) - except Exception, ex: - xval = r('NA') - x_vals[k].append(xval) - except: - pass - -response_term = "" - -x_vals1 = numpy.asarray(x_vals).transpose() - -dat = r.list(x=array(x_vals1), y=y_vals) - -r.library("leaps") - -set_default_mode(NO_CONVERSION) -try: - leaps = r.regsubsets(r("y ~ x"), data= r.na_exclude(dat)) -except RException, rex: - stop_err("Error performing linear regression on the input data.\nEither the response column or one of the predictor columns contain no numeric values.") -set_default_mode(BASIC_CONVERSION) - -summary = r.summary(leaps) -tot = len(x_vals) -pattern = "[" -for i in range(tot): - pattern = pattern + 'c' + str(int(x_cols[int(i)]) + 1) + ' ' -pattern = pattern.strip() + ']' -print >> fout, "#Vars\t%s\tR-sq\tAdj. R-sq\tC-p\tbic" % (pattern) -for ind, item in enumerate(summary['outmat']): - print >> fout, "%s\t%s\t%s\t%s\t%s\t%s" % (str(item).count('*'), item, summary['rsq'][ind], summary['adjr2'][ind], summary['cp'][ind], summary['bic'][ind]) - - -r.pdf( outfile2, 8, 8 ) -r.plot(leaps, scale="Cp", main="Best subsets using Cp Criterion") -r.plot(leaps, scale="r2", main="Best subsets using R-sq Criterion") -r.plot(leaps, scale="adjr2", main="Best subsets using Adjusted R-sq Criterion") -r.plot(leaps, scale="bic", main="Best subsets using bic Criterion") - -r.dev_off() diff --git a/tools/regVariation/best_regression_subsets.xml b/tools/regVariation/best_regression_subsets.xml deleted file mode 100644 index b93ae03ac15..00000000000 --- a/tools/regVariation/best_regression_subsets.xml +++ /dev/null @@ -1,66 +0,0 @@ - - - - best_regression_subsets.py - $input1 - $response_col - $predictor_cols - $out_file1 - $out_file2 - 1>/dev/null - 2>/dev/null - - - - - - - - - - - - - - rpy - - - - - - -.. class:: infomark - -**TIP:** If your data is not TAB delimited, use *Edit Datasets->Convert characters* - ------ - -.. class:: infomark - -**What it does** - -This tool uses the 'regsubsets' function from R statistical package for regression subset selection. It outputs two files, one containing a table with the best subsets and the corresponding summary statistics, and the other containing the graphical representation of the results. - ------ - -.. class:: warningmark - -**Note** - -- This tool currently treats all predictor and response variables as continuous variables. - -- Rows containing non-numeric (or missing) data in any of the chosen columns will be skipped from the analysis. - -- The 6 columns in the output are described below: - - - Column 1 (Vars): denotes the number of variables in the model - - Column 2 ([c2 c3 c4...]): represents a list of the user-selected predictor variables (full model). An asterix denotes the presence of the corresponding predictor variable in the selected model. - - Column 3 (R-sq): the fraction of variance explained by the model - - Column 4 (Adj. R-sq): the above R-squared statistic adjusted, penalizing for higher number of predictors (p) - - Column 5 (Cp): Mallow's Cp statistics - - Column 6 (bic): Bayesian Information Criterion. - - - - diff --git a/tools/regVariation/compute_q_values.pl b/tools/regVariation/compute_q_values.pl deleted file mode 100644 index 6ff3e8f4881..00000000000 --- a/tools/regVariation/compute_q_values.pl +++ /dev/null @@ -1,95 +0,0 @@ -# A program to compute the q-values based on the p-values of multiple simultaneous tests. -# The q-valules are computed using a specific R package created by John Storey called "qvalue". -# The input is a TABULAR format file consisting of one column only that represents the p-values -# of multiple simultaneous tests, one line for every p-value. -# The first output is a TABULAR format file consisting of one column only that represents the q-values -# corresponding to p-values, one line for every q-value. -# the second output is a TABULAR format file consisting of three pages: the first page represents -# the p-values histogram, the second page represents the q-values histogram, and the third page represents -# the four Q-plots as introduced in the "qvalue" package manual. - -use strict; -use warnings; -use IO::Handle; -use File::Temp qw/ tempfile tempdir /; -my $tdir = tempdir( CLEANUP => 0 ); - -# check to make sure having correct input and output files -my $usage = "usage: compute_q_values.pl [TABULAR.in] [lambda] [pi0_method] [fdr_level] [robust] [TABULAR.out] [PDF.out] \n"; -die $usage unless @ARGV == 7; - -#get the input arguments -my $p_valuesInputFile = $ARGV[0]; -my $lambdaValue = $ARGV[1]; -my $pi0_method = $ARGV[2]; -my $fdr_level = $ARGV[3]; -my $robustValue = $ARGV[4]; -my $q_valuesOutputFile = $ARGV[5]; -my $p_q_values_histograms_QPlotsFile = $ARGV[6]; - -if($lambdaValue =~ /sequence/){ - $lambdaValue = "seq(0, 0.95, 0.05)"; -} - -#open the input files -open (INPUT, "<", $p_valuesInputFile) || die("Could not open file $p_valuesInputFile \n"); -open (OUTPUT1, ">", $q_valuesOutputFile) || die("Could not open file $q_valuesOutputFile \n"); -open (OUTPUT2, ">", $p_q_values_histograms_QPlotsFile) || die("Could not open file $p_q_values_histograms_QPlotsFile \n"); -#open (ERROR, ">", "error.txt") or die ("Could not open file error.txt \n"); - -#save all error messages into the error file $errorFile using the error file handle ERROR -#STDERR -> fdopen( \*ERROR, "w" ) or die ("Could not direct errors to the error file error.txt \n"); - -#warn "Hello Error File \n"; - -#variable to store the name of the R script file -my $r_script; - -# R script to implement the calcualtion of q-values based on multiple simultaneous tests p-values -# construct an R script file and save it in a temp directory -chdir $tdir; -$r_script = "q_values_computation.r"; - -open(Rcmd,">", $r_script) or die "Cannot open $r_script \n\n"; -print Rcmd " - #options(show.error.messages = FALSE); - - #load necessary packages - suppressPackageStartupMessages(library(tcltk)); - library(qvalue); - - #read the p-values of the multiple simultaneous tests from the input file $p_valuesInputFile - p <- scan(\"$p_valuesInputFile\", quiet = TRUE); - - #compute the q-values that correspond to the p-values of the multiple simultaneous tests - qobj <- qvalue(p, pi0.meth = \"$pi0_method\", lambda = $lambdaValue, fdr.level = $fdr_level, robust = $robustValue); - #qobj <- qvalue(p, pi0.meth = \"smoother\", lambda = seq(0, 0.95, 0.05), fdr.level = 0.05); - #qobj <- qvalue(p, pi0.meth = \"bootstrap\", fdr.level = 0.05); - - #draw the p-values histogram, the q-values histogram, and the four Q-plots - # and save them on multiple pages of the output file $p_q_values_histograms_QPlotsFile - pdf(file = \"$p_q_values_histograms_QPlotsFile\", width = 6.25, height = 6, family = \"Times\", pointsize = 12, onefile = TRUE) - hist(qobj\$pvalues); - #dev.off(); - - hist(qobj\$qvalues); - #dev.off(); - - qplot(qobj); - dev.off(); - - #save the q-values in the output file $q_valuesOutputFile - qobj\$pi0 <- signif(qobj\$pi0,digits=6) - qwrite(qobj, filename=\"$q_valuesOutputFile\"); - - #options(show.error.messages = TRUE); - #eof\n"; -close Rcmd; - -system("R --no-restore --no-save --no-readline < $r_script > $r_script.out"); - -#close the input and output and error files -#close(ERROR); -close(OUTPUT2); -close(OUTPUT1); -close(INPUT); diff --git a/tools/regVariation/compute_q_values.xml b/tools/regVariation/compute_q_values.xml deleted file mode 100644 index 075b1f090ad..00000000000 --- a/tools/regVariation/compute_q_values.xml +++ /dev/null @@ -1,155 +0,0 @@ - - based on multiple simultaneous tests p-values - - - compute_q_values.pl $inputFile1 $inputLambda2 $inputPI0_method3 $inputFDR_level4 $inputRobust5 $outputFile1 $outputFile2 - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This program computes the q-values based on the p-values of multiple simultaneous tests. The q-values are computed using a specific R package, created by John Storey and Alan Dabney, called "qvalue". The program takes five inputs: - -- The first input is a TABULAR format file consisting of one column only that represents the p-values of multiple simultaneous tests, one line for every p-value. -- The second input is the lambda parameter. The user can choose either the default: seq(0, 0.95, 0.05) or a decimal number between 0.0 and 1.0. -- The third input is PI method which is either "smoother" or "bootstrap". -- The fourth input is the FDR (false discovery rate) level which is a decimal number between 0.0 and 1.0. -- The fifth input is either TRUE or FALSE for the estimate robustness. - -The program gives two outputs: - -- The first output is a TABULAR format file consisting of three columns: - - - the left column represents the p-values of multiple simultaneous tests, one line for every p-value - - the middle column represents the q-values corresponding to the p-values - - the third column represent the significance values, either 1 for significant or 0 for non-significant - -- The second output is a PDF format file consisting of three pages: - - - the first page represents the p-values histogram - - the second page represents the q-values histogram - - the third page represents the four Q-plots as introduced in the "qvalue" package manual. - - -**Example** - -Let us have the first input file of p-values as follows:: - - 0.140627492 - 0.432249886 - 0.122120877 - 0.142010182 - 0.012909858 - 0.000142807 - 0.039841941 - 0.035173303 - 0.011340057 - 1.01E-05 - 0.212738282 - 0.091256284 - 0.547375415 - 0.189589833 - 6.18E-12 - 0.001235875 - 1.10E-05 - 9.75E-07 - 2.13E-18 - 2.54E-16 - 1.20E-19 - 9.76E-14 - 0.359181534 - 0.03661672 - 0.400459987 - 0.387436466 - 0.342075061 - 0.904129283 - 0.031152635 - -Running the program will give the following output:: - - pi0: 0.140311054 - - FDR level: 0.05 - - p-value q-value significant - 0.1406275 0.02889212 1 - 0.4322499 0.06514199 0 - 0.1221209 0.02760624 1 - 0.1420102 0.02889212 1 - 0.01290986 0.00437754 1 - 0.000142807 6.46E-05 1 - 0.03984194 0.01013235 1 - 0.0351733 0.009932946 1 - 0.01134006 0.004194811 1 - 1.01E-05 5.59E-06 1 - 0.2127383 0.03934711 1 - 0.09125628 0.02184257 1 - 0.5473754 0.07954578 0 - 0.1895898 0.03673547 1 - 6.18E-12 5.03E-12 1 - 0.001235875 0.00050288 1 - 1.10E-05 5.59E-06 1 - 9.75E-07 6.61E-07 1 - 2.13E-18 4.33E-18 1 - 2.54E-16 3.45E-16 1 - 1.20E-19 4.88E-19 1 - 9.76E-14 9.93E-14 1 - 0.3591815 0.06089654 0 - 0.03661672 0.009932946 1 - 0.40046 0.0626723 0 - 0.3874365 0.0626723 0 - 0.3420751 0.06051785 0 - 0.9041293 0.1268593 0 - 0.03115264 0.009750824 1 - - -.. image:: ${static_path}/operation_icons/p_hist.png - - -.. image:: ${static_path}/operation_icons/q_hist.png - - -.. image:: ${static_path}/operation_icons/Q_plots.png - - - - - diff --git a/tools/regVariation/featureCounter.py b/tools/regVariation/featureCounter.py deleted file mode 100644 index 5cbfd428596..00000000000 --- a/tools/regVariation/featureCounter.py +++ /dev/null @@ -1,149 +0,0 @@ -#!/usr/bin/env python -#Guruprasad Ananda -""" -Calculate count and coverage of one query on another, and append the Coverage and counts to -the last four columns as bases covered, percent coverage, number of completely present features, number of partially present/overlapping features. - -usage: %prog bed_file_1 bed_file_2 out_file - -1, --cols1=N,N,N,N: Columns for chr, start, end, strand in first file - -2, --cols2=N,N,N,N: Columns for chr, start, end, strand in second file -""" -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -import sys, fileinput -from bx.intervals.io import * -from bx.cookbook import doc_optparse -from bx.intervals.operations import quicksect -from galaxy.tools.util.galaxyops import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - -def counter(node, start, end): - global full, partial - if node.start <= start and node.maxend > start: - if node.end >= end or (node.start == start and end > node.end > start): - full += 1 - elif end > node.end > start: - partial += 1 - if node.left and node.left.maxend > start: - counter(node.left, start, end) - if node.right: - counter(node.right, start, end) - elif start < node.start < end: - if node.end <= end: - full += 1 - else: - partial += 1 - if node.left and node.left.maxend > start: - counter(node.left, start, end) - if node.right: - counter(node.right, start, end) - else: - if node.left: - counter(node.left, start, end) - -def count_coverage( readers, comments=True ): - primary = readers[0] - secondary = readers[1] - secondary_copy = readers[2] - - rightTree = quicksect.IntervalTree() - for item in secondary: - if type( item ) is GenomicInterval: - rightTree.insert( item, secondary.linenum, item.fields ) - - bitsets = secondary_copy.binned_bitsets() - - global full, partial - - for interval in primary: - if type( interval ) is Header: - yield interval - if type( interval ) is Comment and comments: - yield interval - elif type( interval ) == GenomicInterval: - chrom = interval.chrom - start = int(interval.start) - end = int(interval.end) - full = 0 - partial = 0 - if chrom not in bitsets: - bases_covered = 0 - percent = 0.0 - full = 0 - partial = 0 - else: - bases_covered = bitsets[ chrom ].count_range( start, end-start ) - if (end - start) == 0: - percent = 0 - else: - percent = float(bases_covered) / float(end - start) - if bases_covered: - root = rightTree.chroms[chrom] #root node for the chrom tree - counter(root, start, end) - interval.fields.append(str(bases_covered)) - interval.fields.append(str(percent)) - interval.fields.append(str(full)) - interval.fields.append(str(partial)) - yield interval - - -def main(): - options, args = doc_optparse.parse( __doc__ ) - - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols1 ) - chr_col_2, start_col_2, end_col_2, strand_col_2 = parse_cols_arg( options.cols2 ) - in1_fname, in2_fname, out_fname = args - except: - stop_err( "Data issue: click the pencil icon in the history item to correct the metadata attributes." ) - - g1 = NiceReaderWrapper( fileinput.FileInput( in1_fname ), - chrom_col=chr_col_1, - start_col=start_col_1, - end_col=end_col_1, - strand_col=strand_col_1, - fix_strand=True ) - g2 = NiceReaderWrapper( fileinput.FileInput( in2_fname ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - g2_copy = NiceReaderWrapper( fileinput.FileInput( in2_fname ), - chrom_col=chr_col_2, - start_col=start_col_2, - end_col=end_col_2, - strand_col=strand_col_2, - fix_strand=True ) - - - out_file = open( out_fname, "w" ) - - try: - for line in count_coverage([g1, g2, g2_copy]): - if type( line ) is GenomicInterval: - out_file.write( "%s\n" % "\t".join( line.fields ) ) - else: - out_file.write( "%s\n" % line ) - except ParseError, exc: - out_file.close() - fail( str( exc ) ) - - out_file.close() - - if g1.skipped > 0: - print skipped( g1, filedesc=" of 1st dataset" ) - if g2.skipped > 0: - print skipped( g2, filedesc=" of 2nd dataset" ) - elif g2_copy.skipped > 0: - print skipped( g2_copy, filedesc=" of 2nd dataset" ) - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/featureCounter.xml b/tools/regVariation/featureCounter.xml deleted file mode 100644 index b85c107815b..00000000000 --- a/tools/regVariation/featureCounter.xml +++ /dev/null @@ -1,75 +0,0 @@ - - - featureCounter.py $input1 $input2 $output -1 ${input1.metadata.chromCol},${input1.metadata.startCol},${input1.metadata.endCol},${input1.metadata.strandCol} -2 ${input2.metadata.chromCol},${input2.metadata.startCol},${input2.metadata.endCol},${input2.metadata.strandCol} - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool finds the coverage of intervals in the first dataset on intervals in the second dataset. The coverage and count are appended as 4 new columns in the resulting dataset. - ------ - -**Example** - -- If **First dataset** consists of the following windows:: - - chrX 1 10001 seg 0 - - chrX 10001 20001 seg 0 - - chrX 20001 30001 seg 0 - - chrX 30001 40001 seg 0 - - -- and **Second dataset** consists of the following exons:: - - chrX 5000 6000 seg2 0 - - chrX 5500 7000 seg2 0 - - chrX 9000 22000 seg2 0 - - chrX 24000 34000 seg2 0 - - chrX 36000 38000 seg2 0 - - -- the **Result** is the coverage of exons of the second dataset in each of the windows contained in first dataset:: - - chrX 1 10001 seg 0 - 3001 0.3001 2 1 - chrX 10001 20001 seg 0 - 10000 1.0 1 0 - chrX 20001 30001 seg 0 - 8000 0.8 0 2 - chrX 30001 40001 seg 0 - 5999 0.5999 1 1 - -- To clarify, the following line of output ( added columns are indexed by a, b and c ):: - - a b c d - chrX 1 10001 seg 0 - 3001 0.3001 2 1 - - implies that 2 exons (c) fall fully in this window (chrX:1-10001), 1 exon (d) partially overlaps this window, and these 3 exons cover 30.01% (c) of the window size, spanning 3001 nucleotides (a). - - * a: number of nucleotides in this window covered by the features in (c) and (d) - features overlapping with each other will be merged to calculate (a) - * b: fraction of window size covered by features in (c) and (d) - features overlapping with each other will be merged to calculate (b) - * c: number of features in the 2nd dataset that fall **completely** within this window - * d: number of features in the 2nd dataset that **partially** overlap this window - - - diff --git a/tools/regVariation/getIndelRates_3way.py b/tools/regVariation/getIndelRates_3way.py deleted file mode 100755 index c209e2e1858..00000000000 --- a/tools/regVariation/getIndelRates_3way.py +++ /dev/null @@ -1,248 +0,0 @@ -#!/usr/bin/env python -#Guruprasad Ananda - -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) - -import sys, os, tempfile -import fileinput -from warnings import warn - -from galaxy.tools.util.galaxyops import * -from bx.intervals.io import * - -from bx.intervals.operations import quicksect - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -def counter(node, start, end, sort_col): - global full, blk_len, blk_list - if node.start < start: - if node.right: - counter(node.right, start, end, sort_col) - elif start <= node.start <= end and start <= node.end <= end: - full += 1 - if node.other[0] not in blk_list: - blk_list.append(node.other[0]) - blk_len += int(node.other[sort_col+2]) - if node.left and node.left.maxend > start: - counter(node.left, start, end, sort_col) - if node.right: - counter(node.right, start, end, sort_col) - elif node.start > end: - if node.left: - counter(node.left, start, end, sort_col) - - -infile = sys.argv[1] -fout = open(sys.argv[2],'w') -int_file = sys.argv[3] -if int_file != "None": #User has specified an interval file - try: - fint = open(int_file, 'r') - dbkey_i = sys.argv[4] - chr_col_i, start_col_i, end_col_i, strand_col_i = parse_cols_arg( sys.argv[5] ) - except: - stop_err("Unable to open input Interval file") - - -def main(): - for i, line in enumerate( file ( infile )): - line = line.rstrip('\r\n') - if len( line )>0 and not line.startswith( '#' ): - elems = line.split( '\t' ) - break - if i == 30: - break # Hopefully we'll never get here... - - if len( elems ) != 18: - stop_err( "This tool only works on tabular data output by 'Fetch Indels from 3-way alignments' tool. The data in your input dataset is either missing or not formatted properly." ) - - for i, line in enumerate( file ( infile )): - line = line.rstrip('\r\n') - elems = line.split('\t') - try: - assert int(elems[0]) - assert len(elems) == 18 - if int_file != "None": - if dbkey_i not in elems[3] and dbkey_i not in elems[8] and dbkey_i not in elems[13]: - stop_err("The species build corresponding to your interval file is not present in the Indel file.") - if dbkey_i in elems[3]: - sort_col = 4 - elif dbkey_i in elems[8]: - sort_col = 9 - elif dbkey_i in elems[13]: - sort_col = 14 - else: - species = [] - species.append( elems[3].split('.')[0] ) - species.append( elems[8].split('.')[0] ) - species.append( elems[13].split('.')[0] ) - sort_col = 0 #Based on block numbers - break - except: - continue - - fin = open(infile, 'r') - skipped = 0 - - if int_file == "None": - sorted_infile = tempfile.NamedTemporaryFile() - cmdline = "sort -n -k"+str(1)+" -o "+sorted_infile.name+" "+infile - try: - os.system(cmdline) - except: - stop_err("Encountered error while sorting the input file.") - print >> fout, "#Block\t%s_InsRate\t%s_InsRate\t%s_InsRate\t%s_DelRate\t%s_DelRate\t%s_DelRate" % ( species[0], species[1], species[2], species[0], species[1], species[2] ) - prev_bnum = -1 - sorted_infile.seek(0) - for line in sorted_infile.readlines(): - line = line.rstrip('\r\n') - elems = line.split('\t') - try: - assert int(elems[0]) - assert len(elems) == 18 - new_bnum = int(elems[0]) - if new_bnum != prev_bnum: - if prev_bnum != -1: - irate = [] - drate = [] - for i, elem in enumerate(inserts): - try: - irate.append(str("%.2e" % (inserts[i]/blen[i]))) - except: - irate.append('0') - try: - drate.append(str("%.2e" % (deletes[i]/blen[i]))) - except: - drate.append('0') - print >> fout, "%s\t%s\t%s" % ( prev_bnum, '\t'.join(irate) , '\t'.join(drate) ) - inserts = [0.0, 0.0, 0.0] - deletes = [0.0, 0.0, 0.0] - blen = [] - blen.append( int(elems[6]) ) - blen.append( int(elems[11]) ) - blen.append( int(elems[16]) ) - line_sp = elems[1].split('.')[0] - sp_ind = species.index(line_sp) - if elems[1].endswith('insert'): - inserts[sp_ind] += 1 - elif elems[1].endswith('delete'): - deletes[sp_ind] += 1 - prev_bnum = new_bnum - except Exception, ei: - #print >>sys.stderr, ei - continue - irate = [] - drate = [] - for i, elem in enumerate(inserts): - try: - irate.append(str("%.2e" % (inserts[i]/blen[i]))) - except: - irate.append('0') - try: - drate.append(str("%.2e" % (deletes[i]/blen[i]))) - except: - drate.append('0') - print >> fout, "%s\t%s\t%s" % ( prev_bnum, '\t'.join(irate) , '\t'.join(drate) ) - sys.exit() - - inf = open(infile, 'r') - start_met = False - end_met = False - sp_file = tempfile.NamedTemporaryFile() - for n, line in enumerate(inf): - line = line.rstrip('\r\n') - elems = line.split('\t') - try: - assert int(elems[0]) - assert len(elems) == 18 - if dbkey_i not in elems[1]: - if not(start_met): - continue - else: - sp_end = n - break - else: - print >> sp_file, line - if not(start_met): - start_met = True - sp_start = n - except: - continue - - try: - assert sp_end - except: - sp_end = n+1 - - sp_file.seek(0) - win = NiceReaderWrapper( fileinput.FileInput( int_file ), - chrom_col=chr_col_i, - start_col=start_col_i, - end_col=end_col_i, - strand_col=strand_col_i, - fix_strand=True) - - indel = NiceReaderWrapper( fileinput.FileInput( sp_file.name ), - chrom_col=1, - start_col=sort_col, - end_col=sort_col+1, - strand_col=-1, - fix_strand=True) - - indelTree = quicksect.IntervalTree() - for item in indel: - if type( item ) is GenomicInterval: - indelTree.insert( item, indel.linenum, item.fields ) - result = [] - - global full, blk_len, blk_list - for interval in win: - if type( interval ) is Header: - pass - if type( interval ) is Comment: - pass - elif type( interval ) == GenomicInterval: - chrom = interval.chrom - start = int(interval.start) - end = int(interval.end) - if start > end: - warn( "Interval start after end!" ) - ins_chr = "%s.%s_insert" % ( dbkey_i, chrom ) - del_chr = "%s.%s_delete" % ( dbkey_i, chrom ) - irate = 0 - drate = 0 - if ins_chr not in indelTree.chroms and del_chr not in indelTree.chroms: - pass - else: - if ins_chr in indelTree.chroms: - full = 0.0 - blk_len = 0 - blk_list = [] - root = indelTree.chroms[ins_chr] #root node for the chrom insertion tree - counter(root, start, end, sort_col) - if blk_len: - irate = full/blk_len - - if del_chr in indelTree.chroms: - full = 0.0 - blk_len = 0 - blk_list = [] - root = indelTree.chroms[del_chr] #root node for the chrom insertion tree - counter(root, start, end, sort_col) - if blk_len: - drate = full/blk_len - - interval.fields.append(str("%.2e" %irate)) - interval.fields.append(str("%.2e" %drate)) - print >> fout, "\t".join(interval.fields) - fout.flush() - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/getIndelRates_3way.xml b/tools/regVariation/getIndelRates_3way.xml deleted file mode 100644 index 8f4fe3126ec..00000000000 --- a/tools/regVariation/getIndelRates_3way.xml +++ /dev/null @@ -1,61 +0,0 @@ - - for 3-way alignments - - getIndelRates_3way.py $input1 $out_file1 - #if $region.type == "align" - "None" - #else - $region.input2 $input2.dbkey $input2.metadata.chromCol,$input2.metadata.startCol,$input2.metadata.endCol,$input2.metadata.strandCol - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool estimates the insertion and deletion rates for alignments in a window of specified size. Rates are computed over the total adjusted lengths (adjusted by disregarding masked bases) of all the alignments blocks from the indel file that fall within that window. - ------ - -.. class:: warningmark - -**Note** - -This tool only works on the output of the 'Estimate Indel Rates for 3-way alignments' tool. - - - - - diff --git a/tools/regVariation/getIndels.py b/tools/regVariation/getIndels.py deleted file mode 100644 index e9ea3f28230..00000000000 --- a/tools/regVariation/getIndels.py +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env python - -""" -Estimate INDELs for pair-wise alignments. - -usage: %prog maf_input out_file1 out_file2 -""" - -from __future__ import division -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -try: - pkg_resources.require("numpy") -except: - pass -import sys -from bx.cookbook import doc_optparse -from galaxy.tools.exception_handling import * -import bx.align.maf - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - # Parsing Command Line here - options, args = doc_optparse.parse( __doc__ ) - - try: - inp_file, out_file1 = args - except: - print >> sys.stderr, "Tool initialization error." - sys.exit() - - try: - open(inp_file, 'r') - except: - print >> sys.stderr, "Unable to open input file" - sys.exit() - try: - fout1 = open(out_file1, 'w') - #fout2 = open(out_file2, 'w') - except: - print >> sys.stderr, "Unable to open output file" - sys.exit() - - try: - maf_reader = bx.align.maf.Reader( open(inp_file, 'r') ) - except: - print >> sys.stderr, "Your MAF file appears to be malformed." - sys.exit() - - print >> fout1, "#Block\tSource\tSeq1_Start\tSeq1_End\tSeq2_Start\tSeq2_End\tIndel_length" - for block_ind, block in enumerate(maf_reader): - if len(block.components) < 2: - continue - seq1 = block.components[0].text - src1 = block.components[0].src - start1 = block.components[0].start - if len(block.components) == 2: - seq2 = block.components[1].text - src2 = block.components[1].src - start2 = block.components[1].start - #for pos in range(len(seq1)): - nt_pos1 = start1-1 #position of the nucleotide (without counting gaps) - nt_pos2 = start2-1 - pos = 0 #character column position - gaplen1 = 0 - gaplen2 = 0 - prev_pos_gap1 = 0 - prev_pos_gap2 = 0 - while pos < len(seq1): - if prev_pos_gap1 == 0: - gaplen1 = 0 - if prev_pos_gap2 == 0: - gaplen2 = 0 - - if seq1[pos] == '-': - if seq2[pos] != '-': - nt_pos2 += 1 - gaplen1 += 1 - prev_pos_gap1 = 1 - #write 2 - if prev_pos_gap2 == 1: - prev_pos_gap2 = 0 - print >> fout1, "%d\t%s\t%s\t%s\t%s\t%s\t%s" % ( block_ind+1, src2, nt_pos1, nt_pos1+1, nt_pos2-1, nt_pos2-1+gaplen2, gaplen2 ) - if pos == len(seq1)-1: - print >> fout1, "%d\t%s\t%s\t%s\t%s\t%s\t%s" % ( block_ind+1, src1, nt_pos1, nt_pos1+1, nt_pos2+1-gaplen1, nt_pos2+1, gaplen1 ) - else: - prev_pos_gap1 = 0 - prev_pos_gap2 = 0 - """ - if prev_pos_gap1 == 1: - prev_pos_gap1 = 0 - print >> fout1, "%d\t%s\t%s\t%s\t%s" % ( block_ind+1, src1, nt_pos1-1, nt_pos1, gaplen1 ) - elif prev_pos_gap2 == 1: - prev_pos_gap2 = 0 - print >> fout1, "%d\t%s\t%s\t%s\t%s" % ( block_ind+1, src2, nt_pos2-1, nt_pos2, gaplen2 ) - """ - else: - nt_pos1 += 1 - if seq2[pos] != '-': - nt_pos2 += 1 - #write both - if prev_pos_gap1 == 1: - prev_pos_gap1 = 0 - print >> fout1, "%d\t%s\t%s\t%s\t%s\t%s\t%s" % ( block_ind+1, src1, nt_pos1-1, nt_pos1, nt_pos2-gaplen1, nt_pos2, gaplen1 ) - elif prev_pos_gap2 == 1: - prev_pos_gap2 = 0 - print >> fout1, "%d\t%s\t%s\t%s\t%s\t%s\t%s" % ( block_ind+1, src2, nt_pos1-gaplen2, nt_pos1, nt_pos2-1, nt_pos2, gaplen2 ) - else: - gaplen2 += 1 - prev_pos_gap2 = 1 - #write 1 - if prev_pos_gap1 == 1: - prev_pos_gap1 = 0 - print >> fout1, "%d\t%s\t%s\t%s\t%s\t%s\t%s" % ( block_ind+1, src1, nt_pos1-1, nt_pos1, nt_pos2, nt_pos2+gaplen1, gaplen1 ) - if pos == len(seq1)-1: - print >> fout1, "%d\t%s\t%s\t%s\t%s\t%s\t%s" % ( block_ind+1, src2, nt_pos1+1-gaplen2, nt_pos1+1, nt_pos2, nt_pos2+1, gaplen2 ) - pos += 1 - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/getIndels_2way.xml b/tools/regVariation/getIndels_2way.xml deleted file mode 100644 index 1d4780ce49a..00000000000 --- a/tools/regVariation/getIndels_2way.xml +++ /dev/null @@ -1,59 +0,0 @@ - - from pairwise alignments - - getIndels.py $input1 $out_file1 - - - - - - - - - - - numpy - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool estimates the number of indels for every alignment block of the MAF file. - ------ - -.. class:: warningmark - -**Note** - -Any block/s not containing exactly 2 species will be omitted. - ------ - -**Example** - -- For the following alignment block:: - - a score=7233.0 - s hg18.chr1 100 35 + 247249719 AT--GACTGAGGACTTAGTTTAAGATGTTCCTACT - s rheMac2.chr11 200 31 + 134511895 ATAAG-CGGACGACTTAGTTTAAGATGTTCC---- - -- running this tool will return:: - - #Block Source Seq1_Start Seq1_End Seq2_Start Seq2_End Indel_length - 1 hg18.chr1 101 102 202 204 2 - 1 rheMac2.chr11 103 104 204 205 1 - 1 rheMac2.chr11 129 133 229 230 4 - - - - - diff --git a/tools/regVariation/linear_regression.py b/tools/regVariation/linear_regression.py deleted file mode 100644 index cf7afdc5ced..00000000000 --- a/tools/regVariation/linear_regression.py +++ /dev/null @@ -1,147 +0,0 @@ -#!/usr/bin/env python - -from galaxy import eggs -import sys -from rpy import * -import numpy - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - -infile = sys.argv[1] -y_col = int(sys.argv[2])-1 -x_cols = sys.argv[3].split(',') -outfile = sys.argv[4] -outfile2 = sys.argv[5] - -print "Predictor columns: %s; Response column: %d" % ( x_cols, y_col+1 ) -fout = open(outfile,'w') -elems = [] -for i, line in enumerate( file ( infile )): - line = line.rstrip('\r\n') - if len( line )>0 and not line.startswith( '#' ): - elems = line.split( '\t' ) - break - if i == 30: - break # Hopefully we'll never get here... - -if len( elems )<1: - stop_err( "The data in your input dataset is either missing or not formatted properly." ) - -y_vals = [] -x_vals = [] - -for k, col in enumerate(x_cols): - x_cols[k] = int(col)-1 - x_vals.append([]) - -NA = 'NA' -for ind, line in enumerate( file( infile )): - if line and not line.startswith( '#' ): - try: - fields = line.split("\t") - try: - yval = float(fields[y_col]) - except: - yval = r('NA') - y_vals.append(yval) - for k, col in enumerate(x_cols): - try: - xval = float(fields[col]) - except: - xval = r('NA') - x_vals[k].append(xval) - except: - pass - -x_vals1 = numpy.asarray(x_vals).transpose() - -dat = r.list(x=array(x_vals1), y=y_vals) - -set_default_mode(NO_CONVERSION) -try: - linear_model = r.lm(r("y ~ x"), data = r.na_exclude(dat)) -except RException, rex: - stop_err("Error performing linear regression on the input data.\nEither the response column or one of the predictor columns contain only non-numeric or invalid values.") -set_default_mode(BASIC_CONVERSION) - -coeffs = linear_model.as_py()['coefficients'] -yintercept = coeffs['(Intercept)'] -summary = r.summary(linear_model) - -co = summary.get('coefficients', 'NA') -""" -if len(co) != len(x_vals)+1: - stop_err("Stopped performing linear regression on the input data, since one of the predictor columns contains only non-numeric or invalid values.") -""" - -try: - yintercept = r.round(float(yintercept), digits=10) - pvaly = r.round(float(co[0][3]), digits=10) -except: - pass - -print >> fout, "Y-intercept\t%s" % (yintercept) -print >> fout, "p-value (Y-intercept)\t%s" % (pvaly) - -if len(x_vals) == 1: #Simple linear regression case with 1 predictor variable - try: - slope = r.round(float(coeffs['x']), digits=10) - except: - slope = 'NA' - try: - pval = r.round(float(co[1][3]), digits=10) - except: - pval = 'NA' - print >> fout, "Slope (c%d)\t%s" % ( x_cols[0]+1, slope ) - print >> fout, "p-value (c%d)\t%s" % ( x_cols[0]+1, pval ) -else: #Multiple regression case with >1 predictors - ind = 1 - while ind < len(coeffs.keys()): - try: - slope = r.round(float(coeffs['x'+str(ind)]), digits=10) - except: - slope = 'NA' - print >> fout, "Slope (c%d)\t%s" % ( x_cols[ind-1]+1, slope ) - try: - pval = r.round(float(co[ind][3]), digits=10) - except: - pval = 'NA' - print >> fout, "p-value (c%d)\t%s" % ( x_cols[ind-1]+1, pval ) - ind += 1 - -rsq = summary.get('r.squared','NA') -adjrsq = summary.get('adj.r.squared','NA') -fstat = summary.get('fstatistic','NA') -sigma = summary.get('sigma','NA') - -try: - rsq = r.round(float(rsq), digits=5) - adjrsq = r.round(float(adjrsq), digits=5) - fval = r.round(fstat['value'], digits=5) - fstat['value'] = str(fval) - sigma = r.round(float(sigma), digits=10) -except: - pass - -print >> fout, "R-squared\t%s" % (rsq) -print >> fout, "Adjusted R-squared\t%s" % (adjrsq) -print >> fout, "F-statistic\t%s" % (fstat) -print >> fout, "Sigma\t%s" % (sigma) - -r.pdf( outfile2, 8, 8 ) -if len(x_vals) == 1: #Simple linear regression case with 1 predictor variable - sub_title = "Slope = %s; Y-int = %s" % ( slope, yintercept ) - try: - r.plot(x=x_vals[0], y=y_vals, xlab="X", ylab="Y", sub=sub_title, main="Scatterplot with regression") - r.abline(a=yintercept, b=slope, col="red") - except: - pass -else: - r.pairs(dat, main="Scatterplot Matrix", col="blue") -try: - r.plot(linear_model) -except: - pass -r.dev_off() diff --git a/tools/regVariation/linear_regression.xml b/tools/regVariation/linear_regression.xml deleted file mode 100644 index d84639e0971..00000000000 --- a/tools/regVariation/linear_regression.xml +++ /dev/null @@ -1,71 +0,0 @@ - - - - linear_regression.py - $input1 - $response_col - $predictor_cols - $out_file1 - $out_file2 - 1>/dev/null - - - - - - - - - - - - - - rpy - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your data is not TAB delimited, use *Edit Datasets->Convert characters* - ------ - -.. class:: infomark - -**What it does** - -This tool uses the 'lm' function from R statistical package to perform linear regression on the input data. It outputs two files, one containing the summary statistics of the performed regression, and the other containing diagnostic plots to check whether model assumptions are satisfied. - -*R Development Core Team (2009). R: A language and environment for statistical computing. R Foundation for Statistical Computing, Vienna, Austria. ISBN 3-900051-07-0, URL http://www.R-project.org.* - ------ - -.. class:: warningmark - -**Note** - -- This tool currently treats all predictor and response variables as continuous numeric variables. Running the tool on categorical variables might result in incorrect results. - -- Rows containing non-numeric (or missing) data in any of the chosen columns will be skipped from the analysis. - -- The summary statistics in the output are described below: - - - sigma: the square root of the estimated variance of the random error (standard error of the residiuals) - - R-squared: the fraction of variance explained by the model - - Adjusted R-squared: the above R-squared statistic adjusted, penalizing for the number of the predictors (p) - - p-value: p-value for the t-test of the null hypothesis that the corresponding slope is equal to zero against the two-sided alternative. - - - - diff --git a/tools/regVariation/logistic_regression_vif.py b/tools/regVariation/logistic_regression_vif.py deleted file mode 100755 index 68524893aca..00000000000 --- a/tools/regVariation/logistic_regression_vif.py +++ /dev/null @@ -1,168 +0,0 @@ -#!/usr/bin/env python - -from galaxy import eggs -import sys -from rpy import * -import numpy - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - -infile = sys.argv[1] -y_col = int(sys.argv[2])-1 -x_cols = sys.argv[3].split(',') -outfile = sys.argv[4] - - -print "Predictor columns: %s; Response column: %d" % ( x_cols, y_col+1 ) -fout = open(outfile,'w') -elems = [] -for i, line in enumerate( file( infile ) ): - line = line.rstrip('\r\n') - if len( line )>0 and not line.startswith( '#' ): - elems = line.split( '\t' ) - break - if i == 30: - break # Hopefully we'll never get here... - -if len( elems )<1: - stop_err( "The data in your input dataset is either missing or not formatted properly." ) - -y_vals = [] -x_vals = [] - -for k, col in enumerate(x_cols): - x_cols[k] = int(col)-1 - x_vals.append([]) - -NA = 'NA' -for ind, line in enumerate( file( infile )): - if line and not line.startswith( '#' ): - try: - fields = line.split("\t") - try: - yval = float(fields[y_col]) - except: - yval = r('NA') - y_vals.append(yval) - for k, col in enumerate(x_cols): - try: - xval = float(fields[col]) - except: - xval = r('NA') - x_vals[k].append(xval) - except: - pass - -x_vals1 = numpy.asarray(x_vals).transpose() - -check1 = 0 -check0 = 0 -for i in y_vals: - if i == 1: - check1 = 1 - if i == 0: - check0 = 1 -if check1 == 0 or check0 == 0: - sys.exit("Warning: logistic regression must have at least two classes") - -for i in y_vals: - if i not in [1, 0, r('NA')]: - print >> fout, str(i) - sys.exit("Warning: the current version of this tool can run only with two classes and need to be labeled as 0 and 1.") - -dat = r.list(x=array(x_vals1), y=y_vals) -novif = 0 -set_default_mode(NO_CONVERSION) -try: - linear_model = r.glm(r("y ~ x"), data=r.na_exclude(dat), family="binomial") -except RException, rex: - stop_err("Error performing logistic regression on the input data.\nEither the response column or one of the predictor columns contain only non-numeric or invalid values.") -if len(x_cols)>1: - try: - r('suppressPackageStartupMessages(library(car))') - r.assign('dat', dat) - r.assign('ncols', len(x_cols)) - vif = r.vif(r('glm(dat$y ~ ., data = na.exclude(data.frame(as.matrix(dat$x,ncol=ncols))->datx), family="binomial")')) - except RException, rex: - print rex -else: - novif = 1 - -set_default_mode(BASIC_CONVERSION) - -coeffs = linear_model.as_py()['coefficients'] -null_deviance = linear_model.as_py()['null.deviance'] -residual_deviance = linear_model.as_py()['deviance'] -yintercept = coeffs['(Intercept)'] -summary = r.summary(linear_model) -co = summary.get('coefficients', 'NA') -""" -if len(co) != len(x_vals)+1: - stop_err("Stopped performing logistic regression on the input data, since one of the predictor columns contains only non-numeric or invalid values.") -""" - -try: - yintercept = r.round(float(yintercept), digits=10) - pvaly = r.round(float(co[0][3]), digits=10) -except: - pass -print >> fout, "response column\tc%d" % (y_col+1) -tempP = [] -for i in x_cols: - tempP.append('c'+str(i+1)) -tempP = ','.join(tempP) -print >> fout, "predictor column(s)\t%s" % (tempP) -print >> fout, "Y-intercept\t%s" % (yintercept) -print >> fout, "p-value (Y-intercept)\t%s" % (pvaly) - -if len(x_vals) == 1: #Simple linear regression case with 1 predictor variable - try: - slope = r.round(float(coeffs['x']), digits=10) - except: - slope = 'NA' - try: - pval = r.round(float(co[1][3]), digits=10) - except: - pval = 'NA' - print >> fout, "Slope (c%d)\t%s" % ( x_cols[0]+1, slope ) - print >> fout, "p-value (c%d)\t%s" % ( x_cols[0]+1, pval ) -else: #Multiple regression case with >1 predictors - ind = 1 - while ind < len(coeffs.keys()): - try: - slope = r.round(float(coeffs['x'+str(ind)]), digits=10) - except: - slope = 'NA' - print >> fout, "Slope (c%d)\t%s" % ( x_cols[ind-1]+1, slope ) - try: - pval = r.round(float(co[ind][3]), digits=10) - except: - pval = 'NA' - print >> fout, "p-value (c%d)\t%s" % ( x_cols[ind-1]+1, pval ) - ind += 1 - -rsq = summary.get('r.squared','NA') - -try: - rsq = r.round(float((null_deviance-residual_deviance)/null_deviance), digits=5) - null_deviance = r.round(float(null_deviance), digits=5) - residual_deviance = r.round(float(residual_deviance), digits=5) -except: - pass - -print >> fout, "Null deviance\t%s" % (null_deviance) -print >> fout, "Residual deviance\t%s" % (residual_deviance) -print >> fout, "pseudo R-squared\t%s" % (rsq) -print >> fout, "\n" -print >> fout, 'vif' - -if novif == 0: - py_vif = vif.as_py() - count = 0 - for i in sorted(py_vif.keys()): - print >> fout, 'c'+str(x_cols[count]+1), str(py_vif[i]) - count += 1 -elif novif == 1: - print >> fout, "vif can calculate only when model have more than 1 predictor" diff --git a/tools/regVariation/logistic_regression_vif.xml b/tools/regVariation/logistic_regression_vif.xml deleted file mode 100755 index 2b491bd1aac..00000000000 --- a/tools/regVariation/logistic_regression_vif.xml +++ /dev/null @@ -1,74 +0,0 @@ - - - - logistic_regression_vif.py - $input1 - $response_col - $predictor_cols - $out_file1 - 1>/dev/null - - - - - - - - - - - - - - rpy - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your data is not TAB delimited, use *Edit Datasets->Convert characters* - ------ - -.. class:: infomark - -**What it does** - -This tool uses the **'glm'** function from R statistical package to perform logistic regression on the input data. It outputs one file containing the summary statistics of the performed regression. Also, it calculates VIF(Variance Inflation Factor) with **'vif'** function from library (car) in R. - - -*R Development Core Team (2010). R: A language and environment for statistical computing. R Foundation for Statistical Computing, Vienna, Austria. ISBN 3-900051-07-0, URL http://www.R-project.org.* - ------ - -.. class:: warningmark - -**Note** - -- This tool currently treats all predictor variables as continuous numeric variables and response variable as categorical variable. Currently, the response variable can have only two classes, namely 0 and 1. The program will take 0 as base class. - -- Rows containing non-numeric (or missing) data in any of the chosen columns will be skipped from the analysis. - -- The summary statistics in the output are described below: - -- Pseudo R-squared: the proportion of model improvement from null model -- p-value: p-value for the z-test of the null hypothesis that the corresponding slope is equal to zero against the two-sided alternative. -- Coefficient indicates log ratio of (probability to be class 1 / probability to be class 0) - -- This tool also provides **Variance Inflation Factor or VIF** which quantifies the level of multicollinearity. The tool will automatic generate VIF if the model has more than one predictor. The higher the VIF, the higher is the multicollinearity. Multicollinearity will inflate standard error and reduce level of significance of the predictor. In the worst case, it can reverse direction of slope for highly correlated predictors if one of them is significant. A general thumb-rule is to use those predictors having VIF lower than 10 or 5. -- **vif** is calculated by - - First, regressing each predictor over all other predictors, and recording R-squared for each regression. - - Second, computing vif as 1/(1- R_squared) - - - diff --git a/tools/regVariation/maf_cpg_filter.py b/tools/regVariation/maf_cpg_filter.py deleted file mode 100644 index 0e3e4cb0b5d..00000000000 --- a/tools/regVariation/maf_cpg_filter.py +++ /dev/null @@ -1,60 +0,0 @@ -#!/usr/bin/env python -#Guruprasad Ananda -#Adapted from bx/scripts/maf_mask_cpg.py -""" -Mask out potential CpG sites from a maf. Restricted or inclusive definition -of CpG sites can be used. The total fraction masked is printed to stderr. - -usage: %prog < input > output restricted - -m, --mask=N: Character to use as mask ('?' is default) -""" - -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -try: - pkg_resources.require( "numpy" ) -except: - pass -import bx.align -import bx.align.maf -from bx.cookbook import doc_optparse -import sys -import bx.align.sitemask.cpg - -assert sys.version_info[:2] >= ( 2, 4 ) - -def main(): - options, args = doc_optparse.parse( __doc__ ) - try: - inp_file, out_file, sitetype, definition = args - if options.mask: - mask = int(options.mask) - else: - mask = 0 - except: - print >> sys.stderr, "Tool initialization error." - sys.exit() - - reader = bx.align.maf.Reader( open(inp_file, 'r') ) - writer = bx.align.maf.Writer( open(out_file,'w') ) - - mask_chr_dict = {0:'#', 1:'$', 2:'^', 3:'*', 4:'?', 5:'N'} - mask = mask_chr_dict[mask] - - if sitetype == "CpG": - if int(definition) == 1: - cpgfilter = bx.align.sitemask.cpg.Restricted( mask=mask ) - defn = "CpG-Restricted" - else: - cpgfilter = bx.align.sitemask.cpg.Inclusive( mask=mask ) - defn = "CpG-Inclusive" - else: - cpgfilter = bx.align.sitemask.cpg.nonCpG( mask=mask ) - defn = "non-CpG" - cpgfilter.run( reader, writer.write ) - - print "%2.2f percent bases masked; Mask character = %s, Definition = %s" % ( float(cpgfilter.masked)/float(cpgfilter.total) * 100, mask, defn ) - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/maf_cpg_filter.xml b/tools/regVariation/maf_cpg_filter.xml deleted file mode 100644 index 7d0d51ceda5..00000000000 --- a/tools/regVariation/maf_cpg_filter.xml +++ /dev/null @@ -1,87 +0,0 @@ - - from MAF file - - maf_cpg_filter.py - $input - $out_file1 - $masksite.type - #if $masksite.type == "CpG": - $masksite.definition - #else: - "NA" - #end if - -m $mask_char - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - numpy - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool takes a MAF file as input and masks CpG sites in every alignment block of the MAF file. - ------ - -.. class:: warningmark - -**Note** - -*Inclusive definition* defines CpG sites as those sites that are CG in at least one of the species. - -*Restricted definition* considers sites to be CpG if they are CG in at least one of the species, however, sites that are part of overlapping CpGs are excluded. - -For more information on CpG site definitions, please refer this article_. - -.. _article: http://mbe.oxfordjournals.org/cgi/content/full/23/3/565 - - - - - diff --git a/tools/regVariation/microsats_alignment_level.py b/tools/regVariation/microsats_alignment_level.py deleted file mode 100644 index 617261e7809..00000000000 --- a/tools/regVariation/microsats_alignment_level.py +++ /dev/null @@ -1,318 +0,0 @@ - #!/usr/bin/env python -#Guruprasad Ananda -""" -Uses SPUTNIK to fetch microsatellites and extracts orthologous repeats from the sputnik output. -""" -from galaxy import eggs -import os -import re -import string -import sys -import tempfile - -def reverse_complement(text): - DNA_COMP = string.maketrans( "ACGTacgt", "TGCAtgca" ) - comp = [ch for ch in text.translate(DNA_COMP)] - comp.reverse() - return "".join(comp) - - -def main(): - if len(sys.argv) != 8: - print >> sys.stderr, "Insufficient number of arguments." - sys.exit() - - infile = open(sys.argv[1],'r') - separation = int(sys.argv[2]) - outfile = sys.argv[3] - mono_threshold = int(sys.argv[5]) - non_mono_threshold = int(sys.argv[6]) - allow_different_units = int(sys.argv[7]) - - print "Min distance = %d bp; Min threshold for mono repeats = %d; Min threshold for non-mono repeats = %d; Allow different motifs = %s" % ( separation, mono_threshold, non_mono_threshold, allow_different_units==1 ) - try: - fout = open(outfile, "w") - print >> fout, "#Block\tSeq1_Name\tSeq1_Start\tSeq1_End\tSeq1_Type\tSeq1_Length\tSeq1_RepeatNumber\tSeq1_Unit\tSeq2_Name\tSeq2_Start\tSeq2_End\tSeq2_Type\tSeq2_Length\tSeq2_RepeatNumber\tSeq2_Unit" - #sputnik_cmd = os.path.join(os.path.split(sys.argv[0])[0], "sputnik") - sputnik_cmd = "sputnik" - input = infile.read() - block_num = 0 - input = input.replace('\r','\n') - for block in input.split('\n\n'): - block_num += 1 - tmpin = tempfile.NamedTemporaryFile() - tmpout = tempfile.NamedTemporaryFile() - tmpin.write(block.strip()) - cmdline = sputnik_cmd + " " + tmpin.name + " > /dev/null 2>&1 >> " + tmpout.name - try: - os.system(cmdline) - except Exception: - continue - sputnik_out = tmpout.read() - tmpin.close() - tmpout.close() - if sputnik_out != "": - if len(block.split('>')[1:]) != 2: #len(sputnik_out.split('>')): - continue - align_block = block.strip().split('>') - - lendict = {'mononucleotide':1, 'dinucleotide':2, 'trinucleotide':3, 'tetranucleotide':4, 'pentanucleotide':5, 'hexanucleotide':6} - blockdict = {} - r = 0 - namelist = [] - for k, sput_block in enumerate(sputnik_out.split('>')[1:]): - whole_seq = ''.join(align_block[k+1].split('\n')[1:]).replace('\n','').strip() - p = re.compile('\n(\S*nucleotide)') - repeats = p.split(sput_block.strip()) - repeats_count = len(repeats) - j = 1 - name = repeats[0].strip() - try: - coords = re.search('\d+[-_:]\d+', name).group() - coords = coords.replace('_', '-').replace(':', '-') - except Exception: - coords = '0-0' - r += 1 - blockdict[r] = {} - try: - sp_name = name[:name.index('.')] - chr_name = name[name.index('.'):name.index('(')] - namelist.append(sp_name + chr_name) - except: - namelist.append(name[:20]) - while j < repeats_count: - try: - if repeats[j].strip() not in lendict: - j += 2 - continue - - if blockdict[r].has_key('types'): - blockdict[r]['types'].append(repeats[j].strip()) #type of microsat - else: - blockdict[r]['types'] = [repeats[j].strip()] #type of microsat - - start = int(repeats[j+1].split('--')[0].split(':')[0].strip()) - #check to see if there are gaps before the start of the repeat, and change the start accordingly - sgaps = 0 - ch_pos = start - 1 - while ch_pos >= 0: - if whole_seq[ch_pos] == '-': - sgaps += 1 - else: - break #break at the 1st non-gap character - ch_pos -= 1 - if blockdict[r].has_key('starts'): - blockdict[r]['starts'].append(start+sgaps) #start co-ords adjusted with alignment co-ords to include GAPS - else: - blockdict[r]['starts'] = [start+sgaps] - - end = int(repeats[j+1].split('--')[0].split(':')[1].strip()) - #check to see if there are gaps after the end of the repeat, and change the end accordingly - egaps = 0 - for ch in whole_seq[end:]: - if ch == '-': - egaps += 1 - else: - break #break at the 1st non-gap character - if blockdict[r].has_key('ends'): - blockdict[r]['ends'].append(end+egaps) #end co-ords adjusted with alignment co-ords to include GAPS - else: - blockdict[r]['ends'] = [end+egaps] - - repeat_seq = ''.join(repeats[j+1].replace('\r','\n').split('\n')[1:]).strip() #Repeat Sequence - repeat_len = repeats[j+1].split('--')[1].split()[1].strip() - gap_count = repeat_seq.count('-') - #print repeats[j+1].split('--')[1], len(repeat_seq), repeat_len, gap_count - repeat_len = str(int(repeat_len) - gap_count) - - rel_start = blockdict[r]['starts'][-1] - gaps_before_start = whole_seq[:rel_start].count('-') - - if blockdict[r].has_key('gaps_before_start'): - blockdict[r]['gaps_before_start'].append(gaps_before_start) #lengths - else: - blockdict[r]['gaps_before_start'] = [gaps_before_start] #lengths - - whole_seq_start = int(coords.split('-')[0]) - if blockdict[r].has_key('whole_seq_start'): - blockdict[r]['whole_seq_start'].append(whole_seq_start) #lengths - else: - blockdict[r]['whole_seq_start'] = [whole_seq_start] #lengths - - if blockdict[r].has_key('lengths'): - blockdict[r]['lengths'].append(repeat_len) #lengths - else: - blockdict[r]['lengths'] = [repeat_len] #lengths - - if blockdict[r].has_key('counts'): - blockdict[r]['counts'].append(str(int(repeat_len)/lendict[repeats[j].strip()])) #Repeat Unit - else: - blockdict[r]['counts'] = [str(int(repeat_len)/lendict[repeats[j].strip()])] #Repeat Unit - - if blockdict[r].has_key('units'): - blockdict[r]['units'].append(repeat_seq[:lendict[repeats[j].strip()]]) #Repeat Unit - else: - blockdict[r]['units'] = [repeat_seq[:lendict[repeats[j].strip()]]] #Repeat Unit - - except Exception: - pass - j += 2 - #check the co-ords of all repeats corresponding to a sequence and remove adjacent repeats separated by less than the user-specified 'separation'. - delete_index_list = [] - for ind, item in enumerate(blockdict[r]['ends']): - try: - if blockdict[r]['starts'][ind+1]-item < separation: - if ind not in delete_index_list: - delete_index_list.append(ind) - if ind+1 not in delete_index_list: - delete_index_list.append(ind+1) - except Exception: - pass - for index in delete_index_list: #mark them for deletion - try: - blockdict[r]['starts'][index] = 'marked' - blockdict[r]['ends'][index] = 'marked' - blockdict[r]['types'][index] = 'marked' - blockdict[r]['gaps_before_start'][index] = 'marked' - blockdict[r]['whole_seq_start'][index] = 'marked' - blockdict[r]['lengths'][index] = 'marked' - blockdict[r]['counts'][index] = 'marked' - blockdict[r]['units'][index] = 'marked' - except Exception: - pass - #remove 'marked' elements from all the lists - """ - for key in blockdict[r].keys(): - for elem in blockdict[r][key]: - if elem == 'marked': - blockdict[r][key].remove(elem) - """ - #print blockdict - - #make sure that the blockdict has keys for both the species - if (1 not in blockdict) or (2 not in blockdict): - continue - - visited_2 = [0 for x in range(len(blockdict[2]['starts']))] - for ind1, coord_s1 in enumerate(blockdict[1]['starts']): - if coord_s1 == 'marked': - continue - coord_e1 = blockdict[1]['ends'][ind1] - out = [] - for ind2, coord_s2 in enumerate(blockdict[2]['starts']): - if coord_s2 == 'marked': - visited_2[ind2] = 1 - continue - coord_e2 = blockdict[2]['ends'][ind2] - #skip if the 2 repeats are not of the same type or don't have the same repeating unit. - if allow_different_units == 0: - if (blockdict[1]['types'][ind1] != blockdict[2]['types'][ind2]): - continue - else: - if (blockdict[1]['units'][ind1] not in blockdict[2]['units'][ind2]*2) and (reverse_complement(blockdict[1]['units'][ind1]) not in blockdict[2]['units'][ind2]*2): - continue - #print >> sys.stderr, (reverse_complement(blockdict[1]['units'][ind1]) not in blockdict[2]['units'][ind2]*2) - #skip if the repeat number thresholds are not met - if blockdict[1]['types'][ind1] == 'mononucleotide': - if (int(blockdict[1]['counts'][ind1]) < mono_threshold): - continue - else: - if (int(blockdict[1]['counts'][ind1]) < non_mono_threshold): - continue - - if blockdict[2]['types'][ind2] == 'mononucleotide': - if (int(blockdict[2]['counts'][ind2]) < mono_threshold): - continue - else: - if (int(blockdict[2]['counts'][ind2]) < non_mono_threshold): - continue - #print "s1,e1=%s,%s; s2,e2=%s,%s" % ( coord_s1, coord_e1, coord_s2, coord_e2 ) - if (coord_s1 in range(coord_s2, coord_e2)) or (coord_e1 in range(coord_s2, coord_e2)): - out.append(str(block_num)) - out.append(namelist[0]) - rel_start = blockdict[1]['whole_seq_start'][ind1] + coord_s1 - blockdict[1]['gaps_before_start'][ind1] - rel_end = rel_start + int(blockdict[1]['lengths'][ind1]) - out.append(str(rel_start)) - out.append(str(rel_end)) - out.append(blockdict[1]['types'][ind1]) - out.append(blockdict[1]['lengths'][ind1]) - out.append(blockdict[1]['counts'][ind1]) - out.append(blockdict[1]['units'][ind1]) - out.append(namelist[1]) - rel_start = blockdict[2]['whole_seq_start'][ind2] + coord_s2 - blockdict[2]['gaps_before_start'][ind2] - rel_end = rel_start + int(blockdict[2]['lengths'][ind2]) - out.append(str(rel_start)) - out.append(str(rel_end)) - out.append(blockdict[2]['types'][ind2]) - out.append(blockdict[2]['lengths'][ind2]) - out.append(blockdict[2]['counts'][ind2]) - out.append(blockdict[2]['units'][ind2]) - print >> fout, '\t'.join(out) - visited_2[ind2] = 1 - out = [] - - if 0 in visited_2: #there are still some elements in 2nd set which haven't found orthologs yet. - for ind2, coord_s2 in enumerate(blockdict[2]['starts']): - if coord_s2 == 'marked': - continue - if visited_2[ind] != 0: - continue - coord_e2 = blockdict[2]['ends'][ind2] - out = [] - for ind1, coord_s1 in enumerate(blockdict[1]['starts']): - if coord_s1 == 'marked': - continue - coord_e1 = blockdict[1]['ends'][ind1] - #skip if the 2 repeats are not of the same type or don't have the same repeating unit. - if allow_different_units == 0: - if (blockdict[1]['types'][ind1] != blockdict[2]['types'][ind2]): - continue - else: - if (blockdict[1]['units'][ind1] not in blockdict[2]['units'][ind2]*2):# and reverse_complement(blockdict[1]['units'][ind1]) not in blockdict[2]['units'][ind2]*2: - continue - #skip if the repeat number thresholds are not met - if blockdict[1]['types'][ind1] == 'mononucleotide': - if (int(blockdict[1]['counts'][ind1]) < mono_threshold): - continue - else: - if (int(blockdict[1]['counts'][ind1]) < non_mono_threshold): - continue - - if blockdict[2]['types'][ind2] == 'mononucleotide': - if (int(blockdict[2]['counts'][ind2]) < mono_threshold): - continue - else: - if (int(blockdict[2]['counts'][ind2]) < non_mono_threshold): - continue - - if (coord_s2 in range(coord_s1, coord_e1)) or (coord_e2 in range(coord_s1, coord_e1)): - out.append(str(block_num)) - out.append(namelist[0]) - rel_start = blockdict[1]['whole_seq_start'][ind1] + coord_s1 - blockdict[1]['gaps_before_start'][ind1] - rel_end = rel_start + int(blockdict[1]['lengths'][ind1]) - out.append(str(rel_start)) - out.append(str(rel_end)) - out.append(blockdict[1]['types'][ind1]) - out.append(blockdict[1]['lengths'][ind1]) - out.append(blockdict[1]['counts'][ind1]) - out.append(blockdict[1]['units'][ind1]) - out.append(namelist[1]) - rel_start = blockdict[2]['whole_seq_start'][ind2] + coord_s2 - blockdict[2]['gaps_before_start'][ind2] - rel_end = rel_start + int(blockdict[2]['lengths'][ind2]) - out.append(str(rel_start)) - out.append(str(rel_end)) - out.append(blockdict[2]['types'][ind2]) - out.append(blockdict[2]['lengths'][ind2]) - out.append(blockdict[2]['counts'][ind2]) - out.append(blockdict[2]['units'][ind2]) - print >> fout, '\t'.join(out) - visited_2[ind2] = 1 - out = [] - - #print >> fout, blockdict - except Exception, exc: - print >> sys.stderr, "type(exc),args,exc: %s, %s, %s" % ( type(exc), exc.args, exc ) - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/microsats_alignment_level.xml b/tools/regVariation/microsats_alignment_level.xml deleted file mode 100644 index 949f79ca440..00000000000 --- a/tools/regVariation/microsats_alignment_level.xml +++ /dev/null @@ -1,61 +0,0 @@ - - from pair-wise alignments - - microsats_alignment_level.py $input1 $separation $out_file1 "2way" $mono_threshold $non_mono_threshold $allow_different_units - - - - - - - - - - - - - - - - - - sputnik - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool uses a modified version of SPUTNIK to fetch microsatellite repeats from the input fasta sequences and extracts orthologous repeats from the sputnik output. The modified version allows detection of mononucleotide microsatellites. More information on SPUTNIK can be found on this website_. The modified version is available here_. - ------ - -.. class:: warningmark - -**Note** - -- Any block/s not containing exactly 2 species will be omitted. - -- This tool will filter out microsatellites based on the user input values for minimum distance and repeat number thresholds. Further, this tool will also filter out microsatellites that have no orthologous microsatellites in one of the species. - -.. _website: http://espressosoftware.com/pages/sputnik.jsp -.. _here: http://www.bx.psu.edu/svn/universe/dependencies/sputnik/ - - - - diff --git a/tools/regVariation/microsats_mutability.py b/tools/regVariation/microsats_mutability.py deleted file mode 100644 index e99101c896c..00000000000 --- a/tools/regVariation/microsats_mutability.py +++ /dev/null @@ -1,495 +0,0 @@ -#!/usr/bin/env python -#Guruprasad Ananda -""" -This tool computes microsatellite mutability for the orthologous microsatellites fetched from 'Extract Orthologous Microsatellites from pair-wise alignments' tool. -""" -from galaxy import eggs -import fileinput -import string -import sys -import tempfile -from galaxy.tools.util.galaxyops import * -from bx.intervals.io import * -from bx.intervals.operations import quicksect - -fout = open(sys.argv[2],'w') -p_group = int(sys.argv[3]) #primary "group-by" feature -p_bin_size = int(sys.argv[4]) -s_group = int(sys.argv[5]) #sub-group by feature -s_bin_size = int(sys.argv[6]) -mono_threshold = 9 -non_mono_threshold = 4 -p_group_cols = [p_group, p_group+7] -s_group_cols = [s_group, s_group+7] -num_generations = int(sys.argv[7]) -region = sys.argv[8] -int_file = sys.argv[9] -if int_file != "None": #User has specified an interval file - try: - fint = open(int_file, 'r') - dbkey_i = sys.argv[10] - chr_col_i, start_col_i, end_col_i, strand_col_i = parse_cols_arg( sys.argv[11] ) - except: - stop_err("Unable to open input Interval file") - - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -def reverse_complement(text): - DNA_COMP = string.maketrans( "ACGTacgt", "TGCAtgca" ) - comp = [ch for ch in text.translate(DNA_COMP)] - comp.reverse() - return "".join(comp) - - -def get_unique_elems(elems): - seen = set() - return[x for x in elems if x not in seen and not seen.add(x)] - - -def get_binned_lists(uniqlist, binsize): - binnedlist = [] - uniqlist.sort() - start = int(uniqlist[0]) - bin_ind = 0 - l_ind = 0 - binnedlist.append([]) - while l_ind < len(uniqlist): - elem = int(uniqlist[l_ind]) - if elem in range(start, start+binsize): - binnedlist[bin_ind].append(elem) - else: - start += binsize - bin_ind += 1 - binnedlist.append([]) - binnedlist[bin_ind].append(elem) - l_ind += 1 - return binnedlist - - -def fetch_weight(H, C, t): - if (H-(C-H)) < t: - return 2.0 - else: - return 1.0 - - -def mutabilityEstimator(repeats1, repeats2, thresholds): - mut_num = 0.0 #Mutability Numerator - mut_den = 0.0 #Mutability denominator - for ind, H in enumerate(repeats1): - C = repeats2[ind] - t = thresholds[ind] - w = fetch_weight(H, C, t) - mut_num += ((H-C)*(H-C)*w) - mut_den += w - return [mut_num, mut_den] - - -def output_writer(blk, blk_lines): - global winspecies, speciesind - all_elems_1 = [] - all_elems_2 = [] - all_s_elems_1 = [] - all_s_elems_2 = [] - for bline in blk_lines: - if not(bline): - continue - items = bline.split('\t') - seq1 = items[1] - seq2 = items[8] - if p_group_cols[0] == 6: - items[p_group_cols[0]] = int(items[p_group_cols[0]]) - items[p_group_cols[1]] = int(items[p_group_cols[1]]) - if s_group_cols[0] == 6: - items[s_group_cols[0]] = int(items[s_group_cols[0]]) - items[s_group_cols[1]] = int(items[s_group_cols[1]]) - all_elems_1.append(items[p_group_cols[0]]) #primary col elements for species 1 - all_elems_2.append(items[p_group_cols[1]]) #primary col elements for species 2 - if s_group_cols[0] != -1: #sub-group is not None - all_s_elems_1.append(items[s_group_cols[0]]) #secondary col elements for species 1 - all_s_elems_2.append(items[s_group_cols[1]]) #secondary col elements for species 2 - uniq_elems_1 = get_unique_elems(all_elems_1) - uniq_elems_2 = get_unique_elems(all_elems_2) - if s_group_cols[0] != -1: - uniq_s_elems_1 = get_unique_elems(all_s_elems_1) - uniq_s_elems_2 = get_unique_elems(all_s_elems_2) - mut1 = {} - mut2 = {} - count1 = {} - count2 = {} - """ - if p_group_cols[0] == 7: #i.e. the option chosen is group-by unit(AG, GTC, etc) - uniq_elems_1 = get_unique_units(j.sort(lambda x, y: len(x)-len(y))) - """ - if p_group_cols[0] == 6: #i.e. the option chosen is group-by repeat number. - uniq_elems_1 = get_binned_lists( uniq_elems_1, p_bin_size ) - uniq_elems_2 = get_binned_lists( uniq_elems_2, p_bin_size ) - - if s_group_cols[0] == 6: #i.e. the option chosen is subgroup-by repeat number. - uniq_s_elems_1 = get_binned_lists( uniq_s_elems_1, s_bin_size ) - uniq_s_elems_2 = get_binned_lists( uniq_s_elems_2, s_bin_size ) - - for pitem1 in uniq_elems_1: - #repeats1 = [] - #repeats2 = [] - thresholds = [] - if s_group_cols[0] != -1: #Sub-group by feature is not None - for sitem1 in uniq_s_elems_1: - repeats1 = [] - repeats2 = [] - if type(sitem1) == type(''): - sitem1 = sitem1.strip() - for bline in blk_lines: - belems = bline.split('\t') - if type(pitem1) == list: - if p_group_cols[0] == 6: - belems[p_group_cols[0]] = int(belems[p_group_cols[0]]) - if belems[p_group_cols[0]] in pitem1: - if belems[s_group_cols[0]] == sitem1: - repeats1.append(int(belems[6])) - repeats2.append(int(belems[13])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut1[str(pitem1)+'\t'+str(sitem1)] = mutabilityEstimator( repeats1, repeats2, thresholds ) - if region == 'align': - count1[str(pitem1)+'\t'+str(sitem1)] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count1["%s\t%s" % ( pitem1, sitem1 )] = sum(repeats1) - elif winspecies == 2: - count1["%s\t%s" % ( pitem1, sitem1 )] = sum(repeats2) - else: - if type(sitem1) == list: - if s_group_cols[0] == 6: - belems[s_group_cols[0]] = int(belems[s_group_cols[0]]) - if belems[p_group_cols[0]] == pitem1 and belems[s_group_cols[0]] in sitem1: - repeats1.append(int(belems[6])) - repeats2.append(int(belems[13])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut1["%s\t%s" % ( pitem1, sitem1 )] = mutabilityEstimator( repeats1, repeats2, thresholds ) - if region == 'align': - count1[str(pitem1)+'\t'+str(sitem1)] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count1[str(pitem1)+'\t'+str(sitem1)] = sum(repeats1) - elif winspecies == 2: - count1[str(pitem1)+'\t'+str(sitem1)] = sum(repeats2) - else: - if belems[p_group_cols[0]] == pitem1 and belems[s_group_cols[0]] == sitem1: - repeats1.append(int(belems[6])) - repeats2.append(int(belems[13])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut1["%s\t%s" % ( pitem1, sitem1 )] = mutabilityEstimator( repeats1, repeats2, thresholds ) - if region == 'align': - count1[str(pitem1)+'\t'+str(sitem1)] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count1["%s\t%s" % ( pitem1, sitem1 )] = sum(repeats1) - elif winspecies == 2: - count1["%s\t%s" % ( pitem1, sitem1 )] = sum(repeats2) - else: #Sub-group by feature is None - for bline in blk_lines: - belems = bline.split('\t') - if type(pitem1) == list: - #print >> sys.stderr, "item: " + str(item1) - if p_group_cols[0] == 6: - belems[p_group_cols[0]] = int(belems[p_group_cols[0]]) - if belems[p_group_cols[0]] in pitem1: - repeats1.append(int(belems[6])) - repeats2.append(int(belems[13])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - else: - if belems[p_group_cols[0]] == pitem1: - repeats1.append(int(belems[6])) - repeats2.append(int(belems[13])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut1["%s" % (pitem1)] = mutabilityEstimator( repeats1, repeats2, thresholds ) - if region == 'align': - count1["%s" % (pitem1)] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count1[str(pitem1)] = sum(repeats1) - elif winspecies == 2: - count1[str(pitem1)] = sum(repeats2) - - for pitem2 in uniq_elems_2: - #repeats1 = [] - #repeats2 = [] - thresholds = [] - if s_group_cols[0] != -1: #Sub-group by feature is not None - for sitem2 in uniq_s_elems_2: - repeats1 = [] - repeats2 = [] - if type(sitem2)==type(''): - sitem2 = sitem2.strip() - for bline in blk_lines: - belems = bline.split('\t') - if type(pitem2) == list: - if p_group_cols[0] == 6: - belems[p_group_cols[1]] = int(belems[p_group_cols[1]]) - if belems[p_group_cols[1]] in pitem2 and belems[s_group_cols[1]] == sitem2: - repeats2.append(int(belems[13])) - repeats1.append(int(belems[6])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut2["%s\t%s" % ( pitem2, sitem2 )] = mutabilityEstimator( repeats2, repeats1, thresholds ) - #count2[str(pitem2)+'\t'+str(sitem2)]=len(repeats2) - if region == 'align': - count2["%s\t%s" % ( pitem2, sitem2 )] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count2["%s\t%s" % ( pitem2, sitem2 )] = len(repeats2) - elif winspecies == 2: - count2["%s\t%s" % ( pitem2, sitem2 )] = len(repeats1) - else: - if type(sitem2) == list: - if s_group_cols[0] == 6: - belems[s_group_cols[1]] = int(belems[s_group_cols[1]]) - if belems[p_group_cols[1]] == pitem2 and belems[s_group_cols[1]] in sitem2: - repeats2.append(int(belems[13])) - repeats1.append(int(belems[6])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut2["%s\t%s" % ( pitem2, sitem2 )] = mutabilityEstimator( repeats2, repeats1, thresholds ) - if region == 'align': - count2["%s\t%s" % ( pitem2, sitem2 )] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count2["%s\t%s" % ( pitem2, sitem2 )] = len(repeats2) - elif winspecies == 2: - count2["%s\t%s" % ( pitem2, sitem2 )] = len(repeats1) - else: - if belems[p_group_cols[1]] == pitem2 and belems[s_group_cols[1]] == sitem2: - repeats1.append(int(belems[13])) - repeats2.append(int(belems[6])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut2["%s\t%s" % ( pitem2, sitem2 )] = mutabilityEstimator( repeats2, repeats1, thresholds ) - if region == 'align': - count2["%s\t%s" % ( pitem2, sitem2 )] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count2["%s\t%s" % ( pitem2, sitem2 )] = len(repeats2) - elif winspecies == 2: - count2["%s\t%s" % ( pitem2, sitem2 )] = len(repeats1) - else: #Sub-group by feature is None - for bline in blk_lines: - belems = bline.split('\t') - if type(pitem2) == list: - if p_group_cols[0] == 6: - belems[p_group_cols[1]] = int(belems[p_group_cols[1]]) - if belems[p_group_cols[1]] in pitem2: - repeats2.append(int(belems[13])) - repeats1.append(int(belems[6])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - else: - if belems[p_group_cols[1]] == pitem2: - repeats2.append(int(belems[13])) - repeats1.append(int(belems[6])) - if belems[4] == 'mononucleotide': - thresholds.append(mono_threshold) - else: - thresholds.append(non_mono_threshold) - mut2["%s" % (pitem2)] = mutabilityEstimator( repeats2, repeats1, thresholds ) - if region == 'align': - count2["%s" % (pitem2)] = min( sum(repeats1), sum(repeats2) ) - else: - if winspecies == 1: - count2["%s" % (pitem2)] = sum(repeats2) - elif winspecies == 2: - count2["%s" % (pitem2)] = sum(repeats1) - for key in mut1.keys(): - if key in mut2.keys(): - mut = (mut1[key][0]+mut2[key][0])/(mut1[key][1]+mut2[key][1]) - count = count1[key] - del mut2[key] - else: - unit_found = False - if p_group_cols[0] == 7 or s_group_cols[0] == 7: #if it is Repeat Unit (AG, GCT etc.) check for reverse-complements too - if p_group_cols[0] == 7: - this, other = 0, 1 - else: - this, other = 1, 0 - groups1 = key.split('\t') - mutn = mut1[key][0] - mutd = mut1[key][1] - count = 0 - for key2 in mut2.keys(): - groups2 = key2.split('\t') - if groups1[other] == groups2[other]: - if groups1[this] in groups2[this]*2 or reverse_complement(groups1[this]) in groups2[this]*2: - #mut = (mut1[key][0]+mut2[key2][0])/(mut1[key][1]+mut2[key2][1]) - mutn += mut2[key2][0] - mutd += mut2[key2][1] - count += int(count2[key2]) - unit_found = True - del mut2[key2] - #break - if unit_found: - mut = mutn/mutd - else: - mut = mut1[key][0]/mut1[key][1] - count = count1[key] - mut = "%.2e" % (mut/num_generations) - if region == 'align': - print >> fout, str(blk) + '\t'+seq1 + '\t' + seq2 + '\t' +key.strip()+ '\t'+str(mut) + '\t'+ str(count) - elif region == 'win': - fout.write("%s\t%s\t%s\t%s\n" % ( blk, key.strip(), mut, count )) - fout.flush() - - #catch any remaining repeats, for instance if the orthologous position contained different repeat units - for remaining_key in mut2.keys(): - mut = mut2[remaining_key][0]/mut2[remaining_key][1] - mut = "%.2e" % (mut/num_generations) - count = count2[remaining_key] - if region == 'align': - print >> fout, str(blk) + '\t'+seq1 + '\t'+seq2 + '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count) - elif region == 'win': - fout.write("%s\t%s\t%s\t%s\n" % ( blk, remaining_key.strip(), mut, count )) - fout.flush() - #print >> fout, blk + '\t'+remaining_key.strip()+ '\t'+str(mut)+ '\t'+ str(count) - - -def counter(node, start, end, report_func): - if start <= node.start < end and start < node.end <= end: - report_func(node) - if node.right: - counter(node.right, start, end, report_func) - if node.left: - counter(node.left, start, end, report_func) - elif node.start < start and node.right: - counter(node.right, start, end, report_func) - elif node.start >= end and node.left and node.left.maxend > start: - counter(node.left, start, end, report_func) - - -def main(): - infile = sys.argv[1] - - for i, line in enumerate( file ( infile )): - line = line.rstrip('\r\n') - if len( line )>0 and not line.startswith( '#' ): - elems = line.split( '\t' ) - break - if i == 30: - break # Hopefully we'll never get here... - - if len( elems ) != 15: - stop_err( "This tool only works on tabular data output by 'Extract Orthologous Microsatellites from pair-wise alignments' tool. The data in your input dataset is either missing or not formatted properly." ) - global winspecies, speciesind - if region == 'win': - if dbkey_i in elems[1]: - winspecies = 1 - speciesind = 1 - elif dbkey_i in elems[8]: - winspecies = 2 - speciesind = 8 - else: - stop_err("The species build corresponding to your interval file is not present in the Microsatellite file.") - - fin = open(infile, 'r') - skipped = 0 - linestr = "" - - if region == 'win': - msats = NiceReaderWrapper( fileinput.FileInput( infile ), - chrom_col = speciesind, - start_col = speciesind+1, - end_col = speciesind+2, - strand_col = -1, - fix_strand = True) - msatTree = quicksect.IntervalTree() - for item in msats: - if type( item ) is GenomicInterval: - msatTree.insert( item, msats.linenum, item.fields ) - - for iline in fint: - try: - iline = iline.rstrip('\r\n') - if not(iline) or iline == "": - continue - ielems = iline.strip("\r\n").split('\t') - ichr = ielems[chr_col_i] - istart = int(ielems[start_col_i]) - iend = int(ielems[end_col_i]) - isrc = "%s.%s" % ( dbkey_i, ichr ) - if isrc not in msatTree.chroms: - continue - result = [] - root = msatTree.chroms[isrc] #root node for the chrom - counter(root, istart, iend, lambda node: result.append( node )) - if not(result): - continue - tmpfile1 = tempfile.NamedTemporaryFile('wb+') - for node in result: - tmpfile1.write("%s\n" % "\t".join( node.other )) - - tmpfile1.seek(0) - output_writer(iline, tmpfile1.readlines()) - except: - skipped += 1 - if skipped: - print "Skipped %d intervals as invalid." % (skipped) - elif region == 'align': - if s_group_cols[0] != -1: - print >> fout, "#Window\tSpecies_1\tSpecies_2\tGroupby_Feature\tSubGroupby_Feature\tMutability\tCount" - else: - print >> fout, "#Window\tSpecies_1\tWindow_Start\tWindow_End\tSpecies_2\tGroupby_Feature\tMutability\tCount" - prev_bnum = -1 - try: - for line in fin: - line = line.strip("\r\n") - if not(line) or line == "": - continue - elems = line.split('\t') - try: - assert int(elems[0]) - assert len(elems) == 15 - except: - continue - new_bnum = int(elems[0]) - if new_bnum != prev_bnum: - if prev_bnum != -1: - output_writer(prev_bnum, linestr.strip().replace('\r','\n').split('\n')) - linestr = line + "\n" - else: - linestr += line - linestr += "\n" - prev_bnum = new_bnum - output_writer(prev_bnum, linestr.strip().replace('\r','\n').split('\n')) - except Exception, ea: - print >> sys.stderr, ea - skipped += 1 - if skipped: - print "Skipped %d lines as invalid." % (skipped) - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/microsats_mutability.xml b/tools/regVariation/microsats_mutability.xml deleted file mode 100644 index ef65084e794..00000000000 --- a/tools/regVariation/microsats_mutability.xml +++ /dev/null @@ -1,121 +0,0 @@ - - by specified attributes - - microsats_mutability.py - $input1 - $out_file1 - ${pri_condition.primary_group} - #if $pri_condition.primary_group == "6": - ${pri_condition.binsize} ${pri_condition.subgroup} -1 - #else: - 0 ${pri_condition.sub_condition.subgroup} - #if $pri_condition.sub_condition.subgroup == "6": - ${pri_condition.sub_condition.s_binsize} - #else: - -1 - #end if - #end if - $gens - ${region.type} - #if $region.type == "win": - ${region.input2} $input2.dbkey $input2.metadata.chromCol,$input2.metadata.startCol,$input2.metadata.endCol,$input2.metadata.strandCol - #else: - "None" - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool computes microsatellite mutability for the orthologous microsatellites fetched from 'Extract Orthologous Microsatellites from pair-wise alignments' tool. - -Mutability is computed according to the method described in the following paper: - -*Webster et al., Microsatellite evolution inferred from human-chimpanzee genomic sequence alignments, Proc Natl Acad Sci 2002 June 25; 99(13): 8748-8753* - ------ - -.. class:: warningmark - -**Note** - -The user selected group and subgroup by features, the computed mutability and the count of the number of repeats used to compute that mutability are added as columns to the output. - - diff --git a/tools/regVariation/partialR_square.py b/tools/regVariation/partialR_square.py deleted file mode 100755 index b4fbcd6dff1..00000000000 --- a/tools/regVariation/partialR_square.py +++ /dev/null @@ -1,147 +0,0 @@ -#!/usr/bin/env python - -from galaxy import eggs - -import sys -from rpy import * -import numpy - -#export PYTHONPATH=~/galaxy/lib/ -#running command python partialR_square.py reg_inp.tab 4 1,2,3 partialR_result.tabular - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -def sscombs(s): - if len(s) == 1: - return [s] - else: - ssc = sscombs(s[1:]) - return [s[0]] + [s[0]+comb for comb in ssc] + ssc - - -infile = sys.argv[1] -y_col = int(sys.argv[2])-1 -x_cols = sys.argv[3].split(',') -outfile = sys.argv[4] - -print "Predictor columns: %s; Response column: %d" % ( x_cols, y_col+1 ) -fout = open(outfile,'w') - -for i, line in enumerate( file ( infile )): - line = line.rstrip('\r\n') - if len( line )>0 and not line.startswith( '#' ): - elems = line.split( '\t' ) - break - if i == 30: - break # Hopefully we'll never get here... - -if len( elems )<1: - stop_err( "The data in your input dataset is either missing or not formatted properly." ) - -y_vals = [] -x_vals = [] - -for k, col in enumerate(x_cols): - x_cols[k] = int(col)-1 - x_vals.append([]) - """ - try: - float( elems[x_cols[k]] ) - except: - try: - msg = "This operation cannot be performed on non-numeric column %d containing value '%s'." % ( col, elems[x_cols[k]] ) - except: - msg = "This operation cannot be performed on non-numeric data." - stop_err( msg ) - """ -NA = 'NA' -for ind, line in enumerate( file( infile )): - if line and not line.startswith( '#' ): - try: - fields = line.split("\t") - try: - yval = float(fields[y_col]) - except Exception, ey: - yval = r('NA') - #print >> sys.stderr, "ey = %s" %ey - y_vals.append(yval) - for k, col in enumerate(x_cols): - try: - xval = float(fields[col]) - except Exception, ex: - xval = r('NA') - #print >> sys.stderr, "ex = %s" %ex - x_vals[k].append(xval) - except: - pass - -x_vals1 = numpy.asarray(x_vals).transpose() -dat = r.list(x=array(x_vals1), y=y_vals) - -set_default_mode(NO_CONVERSION) -try: - full = r.lm(r("y ~ x"), data= r.na_exclude(dat)) #full model includes all the predictor variables specified by the user -except RException, rex: - stop_err("Error performing linear regression on the input data.\nEither the response column or one of the predictor columns contain no numeric values.") -set_default_mode(BASIC_CONVERSION) - -summary = r.summary(full) -fullr2 = summary.get('r.squared','NA') - -if fullr2 == 'NA': - stop_err("Error in linear regression") - -if len(x_vals) < 10: - s = "" - for ch in range(len(x_vals)): - s += str(ch) -else: - stop_err("This tool only works with less than 10 predictors.") - -print >> fout, "#Model\tR-sq\tpartial_R_Terms\tpartial_R_Value" -all_combos = sorted(sscombs(s), key=len) -all_combos.reverse() -for j, cols in enumerate(all_combos): - #if len(cols) == len(s): #Same as the full model above - # continue - if len(cols) == 1: - x_vals1 = x_vals[int(cols)] - else: - x_v = [] - for col in cols: - x_v.append(x_vals[int(col)]) - x_vals1 = numpy.asarray(x_v).transpose() - dat = r.list(x=array(x_vals1), y=y_vals) - set_default_mode(NO_CONVERSION) - red = r.lm(r("y ~ x"), data= dat) #Reduced model - set_default_mode(BASIC_CONVERSION) - summary = r.summary(red) - redr2 = summary.get('r.squared','NA') - try: - partial_R = (float(fullr2)-float(redr2))/(1-float(redr2)) - except: - partial_R = 'NA' - col_str = "" - for col in cols: - col_str = col_str + str(int(x_cols[int(col)]) + 1) + " " - col_str.strip() - partial_R_col_str = "" - for col in s: - if col not in cols: - partial_R_col_str = partial_R_col_str + str(int(x_cols[int(col)]) + 1) + " " - partial_R_col_str.strip() - if len(cols) == len(s): #full model - partial_R_col_str = "-" - partial_R = "-" - try: - redr2 = "%.4f" % (float(redr2)) - except: - pass - try: - partial_R = "%.4f" % (float(partial_R)) - except: - pass - print >> fout, "%s\t%s\t%s\t%s" % ( col_str, redr2, partial_R_col_str, partial_R ) diff --git a/tools/regVariation/partialR_square.xml b/tools/regVariation/partialR_square.xml deleted file mode 100755 index 4068a07a374..00000000000 --- a/tools/regVariation/partialR_square.xml +++ /dev/null @@ -1,68 +0,0 @@ - - - - partialR_square.py - $input1 - $response_col - $predictor_cols - $out_file1 - 1>/dev/null - - - - - - - - - - - - - rpy - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your data is not TAB delimited, use *Edit Datasets->Convert characters* - ------ - -.. class:: infomark - -**What it does** - -This tool computes the Partial R squared for all possible variable subsets using the following formula: - -**Partial R squared = [SSE(without i: 1,2,...,p-1) - SSE (full: 1,2,..,i..,p-1) / SSE(without i: 1,2,...,p-1)]**, which denotes the case where the 'i'th predictor is dropped. - - - -In general, **Partial R squared = [SSE(without i: 1,2,...,p-1) - SSE (full: 1,2,..,i..,p-1) / SSE(without i: 1,2,...,p-1)]**, where, - -- SSE (full: 1,2,..,i..,p-1) = Sum of Squares left out by the full set of predictors SSE(X1, X2 … Xp) -- SSE (full: 1,2,..,i..,p-1) = Sum of Squares left out by the set of predictors excluding; for example, if we omit the first predictor, it will be SSE(X2 … Xp). - - -The 4 columns in the output are described below: - -- Column 1 (Model): denotes the variables present in the model -- Column 2 (R-sq): denotes the R-squared value corresponding to the model in Column 1 -- Column 3 (Partial R squared_Terms): denotes the variable/s for which Partial R squared is computed. These are the variables that are absent in the reduced model in Column 1. A '-' in this column indicates that the model in Column 1 is the Full model. -- Column 4 (Partial R squared): denotes the Partial R squared value corresponding to the variable/s in Column 3. A '-' in this column indicates that the model in Column 1 is the Full model. - -*R Development Core Team (2010). R: A language and environment for statistical computing. R Foundation for Statistical Computing, Vienna, Austria. ISBN 3-900051-07-0, URL http://www.R-project.org.* - - - diff --git a/tools/regVariation/quality_filter.py b/tools/regVariation/quality_filter.py deleted file mode 100644 index 12750060319..00000000000 --- a/tools/regVariation/quality_filter.py +++ /dev/null @@ -1,242 +0,0 @@ -#!/usr/bin/env python -#Guruprasad Ananda -""" -Filter based on nucleotide quality (PHRED score). - -usage: %prog input out_file primary_species mask_species score mask_char mask_region mask_region_length -""" - - -from __future__ import division -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -pkg_resources.require( "lrucache" ) -try: - pkg_resources.require("numpy") -except: - pass - -import sys -import os, os.path -from UserDict import DictMixin -from bx.binned_array import FileBinnedArray -from bx.bitset import * -from bx.bitset_builders import * -from bx.cookbook import doc_optparse -from galaxy.tools.exception_handling import * -import bx.align.maf - -class FileBinnedArrayDir( DictMixin ): - """ - Adapter that makes a directory of FileBinnedArray files look like - a regular dict of BinnedArray objects. - """ - def __init__( self, dir ): - self.dir = dir - self.cache = dict() - def __getitem__( self, key ): - value = None - if key in self.cache: - value = self.cache[key] - else: - fname = os.path.join( self.dir, "%s.qa.bqv" % key ) - if os.path.exists( fname ): - value = FileBinnedArray( open( fname ) ) - self.cache[key] = value - if value is None: - raise KeyError( "File does not exist: " + fname ) - return value - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - -def load_scores_ba_dir( dir ): - """ - Return a dict-like object (keyed by chromosome) that returns - FileBinnedArray objects created from "key.ba" files in `dir` - """ - return FileBinnedArrayDir( dir ) - -def bitwise_and ( string1, string2, maskch ): - result = [] - for i, ch in enumerate(string1): - try: - ch = int(ch) - except: - pass - if string2[i] == '-': - ch = 1 - if ch and string2[i]: - result.append(string2[i]) - else: - result.append(maskch) - return ''.join(result) - -def main(): - # Parsing Command Line here - options, args = doc_optparse.parse( __doc__ ) - - try: - #chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols ) - inp_file, out_file, pri_species, mask_species, qual_cutoff, mask_chr, mask_region, mask_length, loc_file = args - qual_cutoff = int(qual_cutoff) - mask_chr = int(mask_chr) - mask_region = int(mask_region) - if mask_region != 3: - mask_length = int(mask_length) - else: - mask_length_r = int(mask_length.split(',')[0]) - mask_length_l = int(mask_length.split(',')[1]) - except: - stop_err( "Data issue, click the pencil icon in the history item to correct the metadata attributes of the input dataset." ) - - if pri_species == 'None': - stop_err( "No primary species selected, try again by selecting at least one primary species." ) - if mask_species == 'None': - stop_err( "No mask species selected, try again by selecting at least one species to mask." ) - - mask_chr_count = 0 - mask_chr_dict = {0:'#', 1:'$', 2:'^', 3:'*', 4:'?', 5:'N'} - mask_reg_dict = {0:'Current pos', 1:'Current+Downstream', 2:'Current+Upstream', 3:'Current+Both sides'} - - #ensure dbkey is present in the twobit loc file - try: - pspecies_all = pri_species.split(',') - pspecies_all2 = pri_species.split(',') - pspecies = [] - filepaths = [] - for line in open(loc_file): - if pspecies_all2 == []: - break - if line[0:1] == "#": - continue - fields = line.split('\t') - try: - build = fields[0] - for i, dbkey in enumerate(pspecies_all2): - if dbkey == build: - pspecies.append(build) - filepaths.append(fields[1]) - del pspecies_all2[i] - else: - continue - except: - pass - except Exception, exc: - stop_err( 'Initialization errorL %s' % str( exc ) ) - - if len(pspecies) == 0: - stop_err( "Quality scores are not available for the following genome builds: %s" % ( pspecies_all2 ) ) - if len(pspecies) < len(pspecies_all): - print "Quality scores are not available for the following genome builds: %s" % (pspecies_all2) - - scores_by_chrom = [] - #Get scores for all the primary species - for file in filepaths: - scores_by_chrom.append(load_scores_ba_dir( file.strip() )) - - try: - maf_reader = bx.align.maf.Reader( open(inp_file, 'r') ) - maf_writer = bx.align.maf.Writer( open(out_file,'w') ) - except Exception, e: - stop_err( "Your MAF file appears to be malformed: %s" % str( e ) ) - - maf_count = 0 - for block in maf_reader: - status_strings = [] - for seq in range (len(block.components)): - src = block.components[seq].src - dbkey = src.split('.')[0] - chr = src.split('.')[1] - if not (dbkey in pspecies): - continue - else: #enter if the species is a primary species - index = pspecies.index(dbkey) - sequence = block.components[seq].text - s_start = block.components[seq].start - size = len(sequence) #this includes the gaps too - status_str = '1'*size - status_list = list(status_str) - if status_strings == []: - status_strings.append(status_str) - ind = 0 - s_end = block.components[seq].end - #Get scores for the entire sequence - try: - scores = scores_by_chrom[index][chr][s_start:s_end] - except: - continue - pos = 0 - while pos < (s_end-s_start): - if sequence[ind] == '-': #No score for GAPS - ind += 1 - continue - score = scores[pos] - if score < qual_cutoff: - score = 0 - - if not(score): - if mask_region == 0: #Mask Corresponding position only - status_list[ind] = '0' - ind += 1 - pos += 1 - elif mask_region == 1: #Mask Corresponding position + downstream neighbors - for n in range(mask_length+1): - try: - status_list[ind+n] = '0' - except: - pass - ind = ind + mask_length + 1 - pos = pos + mask_length + 1 - elif mask_region == 2: #Mask Corresponding position + upstream neighbors - for n in range(mask_length+1): - try: - status_list[ind-n] = '0' - except: - pass - ind += 1 - pos += 1 - elif mask_region == 3: #Mask Corresponding position + neighbors on both sides - for n in range(-mask_length_l, mask_length_r+1): - try: - status_list[ind+n] = '0' - except: - pass - ind = ind + mask_length_r + 1 - pos = pos + mask_length_r + 1 - else: - pos += 1 - ind += 1 - - status_strings.append(''.join(status_list)) - - if status_strings == []: #this block has no primary species - continue - output_status_str = status_strings[0] - for stat in status_strings[1:]: - try: - output_status_str = bitwise_and (status_strings[0], stat, '0') - except Exception, e: - break - - for seq in range (len(block.components)): - src = block.components[seq].src - dbkey = src.split('.')[0] - if dbkey not in mask_species.split(','): - continue - sequence = block.components[seq].text - sequence = bitwise_and (output_status_str, sequence, mask_chr_dict[mask_chr]) - block.components[seq].text = sequence - mask_chr_count += output_status_str.count('0') - maf_writer.write(block) - maf_count += 1 - - maf_reader.close() - maf_writer.close() - print "No. of blocks = %d; No. of masked nucleotides = %s; Mask character = %s; Mask region = %s; Cutoff used = %d" % (maf_count, mask_chr_count, mask_chr_dict[mask_chr], mask_reg_dict[mask_region], qual_cutoff) - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/quality_filter.xml b/tools/regVariation/quality_filter.xml deleted file mode 100644 index 0ebedc448b4..00000000000 --- a/tools/regVariation/quality_filter.xml +++ /dev/null @@ -1,115 +0,0 @@ - - based on quality scores - - quality_filter.py - $input - $out_file1 - $primary_species - $mask_species - $score - $mask_char - ${mask_region.region} - #if $mask_region.region == "3" - ${mask_region.lengthr},${mask_region.lengthl} - #elif $mask_region.region == "0" - 1 - #else - ${mask_region.length} - #end if - ${GALAXY_DATA_INDEX_DIR}/quality_scores.loc - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - numpy - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool takes a MAF file as input and filters nucleotides in every alignment block of the MAF file based on their quality/PHRED scores. - ------ - -.. class:: warningmark - -**Note** - -Any block/s not containing the primary species (species whose quality scores is to be used), will be omitted. -Also, any primary species whose quality scores are not available in Galaxy will be considered as a non-primary species. This info will appear as a message in the job history panel. - ------ - -**Example** - -- For the following alignment block:: - - a score=4050.0 - s hg18.chrX 3719221 48 - 154913754 tattttacatttaaaataaatatgtaaatatatattttatatttaaaa - s panTro2.chrX 3560945 48 - 155361357 tattttatatttaaaataaagatgtaaatatatattttatatttaaaa - -- running this tool with **Primary species as panTro2**, **Mask species as hg18, panTro2**, **Quality cutoff as 20**, **Mask character as #** and **Mask region as only the corresponding position** will return:: - - a score=4050.0 - s hg18.chrX 3719221 48 - 154913754 ###tttac#####a###a#atatgtaaat###tattt#####ttaaaa - s panTro2.chrX 3560945 48 - 155361357 ###tttat#####a###a#agatgtaaat###tattt#####ttaaaa - - where, the positions containing # represent panTro2 nucleotides having quality scores less than 20. - - diff --git a/tools/regVariation/qv_to_bqv.py b/tools/regVariation/qv_to_bqv.py deleted file mode 100755 index 172dce0e4cb..00000000000 --- a/tools/regVariation/qv_to_bqv.py +++ /dev/null @@ -1,89 +0,0 @@ -#!/usr/bin/env python - -""" -Adapted from bx/scripts/qv_to_bqv.py - -Convert a qual (qv) file to several BinnedArray files for fast seek. -This script takes approximately 4 seconds per 1 million base pairs. - -The input format is fasta style quality -- fasta headers followed by -whitespace separated integers. - -usage: %prog qual_file output_file -""" - -import pkg_resources -pkg_resources.require( "bx-python" ) -pkg_resources.require( "numpy" ) -import os -import sys -import tempfile -from bx.binned_array import BinnedArrayWriter -from bx.cookbook import * -import fileinput - -def load_scores_ba_dir( dir ): - """ - Return a dict-like object (keyed by chromosome) that returns - FileBinnedArray objects created from "key.ba" files in `dir` - """ - return FileBinnedArrayDir( dir ) - -def main(): - args = sys.argv[1:] - try: - qual_file_dir = args[0] - mydir = "/home/gua110/Desktop/rhesus_quality_scores/rheMac2.qual.qv" - qual_file_dir = mydir.replace(mydir.split("/")[-1], "") - output_file = args[1] - fo = open(output_file, "w") - except: - print "usage: qual_file output_file" - sys.exit() - - tmpfile = tempfile.NamedTemporaryFile() - cmdline = "ls " + qual_file_dir + "*.qa | cat >> " + tmpfile.name - os.system (cmdline) - for qual_file in tmpfile.readlines(): - qual = fileinput.FileInput( qual_file.strip() ) - outfile = None - outbin = None - base_count = 0 - mega_count = 0 - - for line in qual: - line = line.rstrip("\r\n") - if line.startswith(">"): - # close old - if outbin and outfile: - print "\nFinished region " + region + " at " + str(base_count) + " base pairs." - outbin.finish() - outfile.close() - # start new file - region = line.lstrip(">") - #outfname = output_file + "." + region + ".bqv" #CHANGED - outfname = qual_file.strip() + ".bqv" - print >> fo, "Writing region " + region + " to file " + outfname - outfile = open( outfname , "wb") - outbin = BinnedArrayWriter(outfile, typecode='b', default=0) - base_count = 0 - mega_count = 0 - else: - if outfile and outbin: - nums = line.split() - for val in nums: - outval = int(val) - assert outval <= 255 and outval >= 0 - outbin.write(outval) - base_count += 1 - if (mega_count * 1000000) <= base_count: - sys.stdout.write(str(mega_count)+" ") - sys.stdout.flush() - mega_count = base_count // 1000000 + 1 - if outbin and outfile: - print "\nFinished region " + region + " at " + str(base_count) + " base pairs." - outbin.finish() - outfile.close() - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/qv_to_bqv.xml b/tools/regVariation/qv_to_bqv.xml deleted file mode 100644 index d899a8d5fc2..00000000000 --- a/tools/regVariation/qv_to_bqv.xml +++ /dev/null @@ -1,17 +0,0 @@ - - - qv_to_bqv.py "$input1" $output - - - - - - - - - - - - - - \ No newline at end of file diff --git a/tools/regVariation/rcve.py b/tools/regVariation/rcve.py deleted file mode 100644 index 2e7113165a4..00000000000 --- a/tools/regVariation/rcve.py +++ /dev/null @@ -1,144 +0,0 @@ -#!/usr/bin/env python - -from galaxy import eggs - -import sys -from rpy import * -import numpy - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -def sscombs(s): - if len(s) == 1: - return [s] - else: - ssc = sscombs(s[1:]) - return [s[0]] + [s[0]+comb for comb in ssc] + ssc - - -infile = sys.argv[1] -y_col = int(sys.argv[2])-1 -x_cols = sys.argv[3].split(',') -outfile = sys.argv[4] - -print "Predictor columns: %s; Response column: %d" % ( x_cols, y_col+1 ) -fout = open(outfile,'w') - -for i, line in enumerate( file ( infile )): - line = line.rstrip('\r\n') - if len( line )>0 and not line.startswith( '#' ): - elems = line.split( '\t' ) - break - if i == 30: - break # Hopefully we'll never get here... - -if len( elems )<1: - stop_err( "The data in your input dataset is either missing or not formatted properly." ) - -y_vals = [] -x_vals = [] - -for k, col in enumerate(x_cols): - x_cols[k] = int(col)-1 - x_vals.append([]) - """ - try: - float( elems[x_cols[k]] ) - except: - try: - msg = "This operation cannot be performed on non-numeric column %d containing value '%s'." % ( col, elems[x_cols[k]] ) - except: - msg = "This operation cannot be performed on non-numeric data." - stop_err( msg ) - """ -NA = 'NA' -for ind, line in enumerate( file( infile )): - if line and not line.startswith( '#' ): - try: - fields = line.split("\t") - try: - yval = float(fields[y_col]) - except Exception, ey: - yval = r('NA') - #print >>sys.stderr, "ey = %s" %ey - y_vals.append(yval) - for k, col in enumerate(x_cols): - try: - xval = float(fields[col]) - except Exception, ex: - xval = r('NA') - #print >>sys.stderr, "ex = %s" %ex - x_vals[k].append(xval) - except: - pass - -x_vals1 = numpy.asarray(x_vals).transpose() -dat = r.list( x=array(x_vals1), y=y_vals ) - -set_default_mode(NO_CONVERSION) -try: - full = r.lm( r("y ~ x"), data=r.na_exclude(dat) ) #full model includes all the predictor variables specified by the user -except RException, rex: - stop_err("Error performing linear regression on the input data.\nEither the response column or one of the predictor columns contain no numeric values.") -set_default_mode(BASIC_CONVERSION) - -summary = r.summary(full) -fullr2 = summary.get('r.squared','NA') - -if fullr2 == 'NA': - stop_err("Error in linear regression") - -if len(x_vals) < 10: - s = "" - for ch in range(len(x_vals)): - s += str(ch) -else: - stop_err("This tool only works with less than 10 predictors.") - -print >> fout, "#Model\tR-sq\tRCVE_Terms\tRCVE_Value" -all_combos = sorted(sscombs(s), key=len) -all_combos.reverse() -for j, cols in enumerate(all_combos): - #if len(cols) == len(s): #Same as the full model above - # continue - if len(cols) == 1: - x_vals1 = x_vals[int(cols)] - else: - x_v = [] - for col in cols: - x_v.append(x_vals[int(col)]) - x_vals1 = numpy.asarray(x_v).transpose() - dat = r.list(x=array(x_vals1), y=y_vals) - set_default_mode(NO_CONVERSION) - red = r.lm(r("y ~ x"), data= dat) #Reduced model - set_default_mode(BASIC_CONVERSION) - summary = r.summary(red) - redr2 = summary.get('r.squared','NA') - try: - rcve = (float(fullr2)-float(redr2))/float(fullr2) - except: - rcve = 'NA' - col_str = "" - for col in cols: - col_str = col_str + str(int(x_cols[int(col)]) + 1) + " " - col_str.strip() - rcve_col_str = "" - for col in s: - if col not in cols: - rcve_col_str = rcve_col_str + str(int(x_cols[int(col)]) + 1) + " " - rcve_col_str.strip() - if len(cols) == len(s): #full model - rcve_col_str = "-" - rcve = "-" - try: - redr2 = "%.4f" % (float(redr2)) - except: - pass - try: - rcve = "%.4f" % (float(rcve)) - except: - pass - print >> fout, "%s\t%s\t%s\t%s" % ( col_str, redr2, rcve_col_str, rcve ) diff --git a/tools/regVariation/rcve.xml b/tools/regVariation/rcve.xml deleted file mode 100644 index 6807d9ac549..00000000000 --- a/tools/regVariation/rcve.xml +++ /dev/null @@ -1,70 +0,0 @@ - - - - rcve.py - $input1 - $response_col - $predictor_cols - $out_file1 - 1>/dev/null - - - - - - - - - - - - - rpy - - - - - - - - - - - - - -.. class:: infomark - -**TIP:** If your data is not TAB delimited, use *Edit Datasets->Convert characters* - ------ - -.. class:: infomark - -**What it does** - -This tool computes the RCVE (Relative Contribution to Variance) for all possible variable subsets using the following formula: - -**RCVE(i) = [R-sq (full: 1,2,..,i..,p-1) - R-sq(without i: 1,2,...,p-1)] / R-sq (full: 1,2,..,i..,p-1)**, -which denotes the case where the 'i'th predictor is dropped. - - -In general, -**RCVE(X+) = [R-sq (full: {X,X+}) - R-sq(reduced: {X})] / R-sq (full: {X,X+})**, -where, - -- {X,X+} denotes the set of all predictors, -- X+ is the set of predictors for which we compute RCVE (and therefore drop from the full model to obtain a reduced one), -- {X} is the set of the predictors that are left in the reduced model after excluding {X+} - - -The 4 columns in the output are described below: - -- Column 1 (Model): denotes the variables present in the model ({X}) -- Column 2 (R-sq): denotes the R-squared value corresponding to the model in Column 1 -- Column 3 (RCVE_Terms): denotes the variable/s for which RCVE is computed ({X+}). These are the variables that are absent in the reduced model in Column 1. A '-' in this column indicates that the model in Column 1 is the Full model. -- Column 4 (RCVE): denotes the RCVE value corresponding to the variable/s in Column 3. A '-' in this column indicates that the model in Column 1 is the Full model. - - - - diff --git a/tools/regVariation/substitution_rates.py b/tools/regVariation/substitution_rates.py deleted file mode 100644 index 2fe03d7210c..00000000000 --- a/tools/regVariation/substitution_rates.py +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env python -#guruprasad Ananda -""" -Estimates substitution rates from pairwise alignments using JC69 model. -""" - -from galaxy import eggs -from galaxy.tools.util.galaxyops import * -from galaxy.tools.util import maf_utilities -import bx.align.maf -import fileinput -import sys - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -if len(sys.argv) < 3: - stop_err("Incorrect number of arguments.") - -inp_file = sys.argv[1] -out_file = sys.argv[2] -fout = open(out_file, 'w') -int_file = sys.argv[3] -if int_file != "None": #The user has specified an interval file - dbkey_i = sys.argv[4] - chr_col_i, start_col_i, end_col_i, strand_col_i = parse_cols_arg( sys.argv[5] ) - - -def rateEstimator(block): - global alignlen, mismatches - - src1 = block.components[0].src - sequence1 = block.components[0].text - start1 = block.components[0].start - end1 = block.components[0].end - len1 = int(end1)-int(start1) - len1_withgap = len(sequence1) - mismatch = 0.0 - - for seq in range (1, len(block.components)): - src2 = block.components[seq].src - sequence2 = block.components[seq].text - start2 = block.components[seq].start - end2 = block.components[seq].end - len2 = int(end2)-int(start2) - for nt in range(len1_withgap): - if sequence1[nt] not in '-#$^*?' and sequence2[nt] not in '-#$^*?': # Not a gap or masked character - if sequence1[nt].upper() != sequence2[nt].upper(): - mismatch += 1 - - if int_file == "None": - p = mismatch/min(len1, len2) - print >> fout, "%s\t%s\t%s\t%s\t%s\t%s\t%d\t%d\t%.4f" % ( src1, start1, end1, src2, start2, end2, min(len1, len2), mismatch, p ) - else: - mismatches += mismatch - alignlen += min(len1, len2) - - -def main(): - skipped = 0 - not_pairwise = 0 - - if int_file == "None": - try: - maf_reader = bx.align.maf.Reader( open(inp_file, 'r') ) - except: - stop_err("Your MAF file appears to be malformed.") - print >> fout, "#Seq1\tStart1\tEnd1\tSeq2\tStart2\tEnd2\tL\tN\tp" - for block in maf_reader: - if len(block.components) != 2: - not_pairwise += 1 - continue - try: - rateEstimator(block) - except: - skipped += 1 - else: - index, index_filename = maf_utilities.build_maf_index( inp_file, species = [dbkey_i] ) - if index is None: - print >> sys.stderr, "Your MAF file appears to be malformed." - sys.exit() - win = NiceReaderWrapper( fileinput.FileInput( int_file ), - chrom_col=chr_col_i, - start_col=start_col_i, - end_col=end_col_i, - strand_col=strand_col_i, - fix_strand=True) - species = None - mincols = 0 - global alignlen, mismatches - - for interval in win: - alignlen = 0 - mismatches = 0.0 - src = "%s.%s" % ( dbkey_i, interval.chrom ) - for block in maf_utilities.get_chopped_blocks_for_region( index, src, interval, species, mincols ): - if len(block.components) != 2: - not_pairwise += 1 - continue - try: - rateEstimator(block) - except: - skipped += 1 - if alignlen: - p = mismatches/alignlen - else: - p = 'NA' - interval.fields.append(str(alignlen)) - interval.fields.append(str(mismatches)) - interval.fields.append(str(p)) - print >> fout, "\t".join(interval.fields) - #num_blocks += 1 - - if not_pairwise: - print "Skipped %d non-pairwise blocks" % (not_pairwise) - if skipped: - print "Skipped %d blocks as invalid" % (skipped) - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/substitution_rates.xml b/tools/regVariation/substitution_rates.xml deleted file mode 100644 index a4201a675d9..00000000000 --- a/tools/regVariation/substitution_rates.xml +++ /dev/null @@ -1,61 +0,0 @@ - - for non-coding regions - - substitution_rates.py - $input - $out_file1 - #if $region.type == "win": - ${region.input2} ${region.input2.dbkey} ${region.input2.metadata.chromCol},$region.input2.metadata.startCol,$region.input2.metadata.endCol,$region.input2.metadata.strandCol - #else: - "None" - #end if - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool takes a pairwise MAF file as input and estimates substitution rate according to Jukes-Cantor JC69 model. The 3 new columns appended to the output are explained below: - -- L: number of nucleotides compared -- N: number of different nucleotides -- p = N/L - ------ - -.. class:: warningmark - -**Note** - -Any block/s not containing exactly two sequences, will be omitted. - - - \ No newline at end of file diff --git a/tools/regVariation/substitutions.py b/tools/regVariation/substitutions.py deleted file mode 100644 index 7774ca5e274..00000000000 --- a/tools/regVariation/substitutions.py +++ /dev/null @@ -1,85 +0,0 @@ -#!/usr/bin/env python -#Guruprasad ANanda -""" -Fetches substitutions from pairwise alignments. -""" - -from galaxy import eggs - -from galaxy.tools.util import maf_utilities - -import bx.align.maf -import sys - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -if len(sys.argv) < 3: - stop_err("Incorrect number of arguments.") - -inp_file = sys.argv[1] -out_file = sys.argv[2] -fout = open(out_file, 'w') - -def fetchSubs(block): - src1 = block.components[0].src - sequence1 = block.components[0].text - start1 = block.components[0].start - end1 = block.components[0].end - len1_withgap = len(sequence1) - - for seq in range (1, len(block.components)): - src2 = block.components[seq].src - sequence2 = block.components[seq].text - start2 = block.components[seq].start - end2 = block.components[seq].end - sub_begin = None - sub_end = None - begin = False - - for nt in range(len1_withgap): - if sequence1[nt] not in '-#$^*?' and sequence2[nt] not in '-#$^*?': # Not a gap or masked character - if sequence1[nt].upper() != sequence2[nt].upper(): - if not(begin): - sub_begin = nt - begin = True - sub_end = nt - else: - if begin: - print >> fout, "%s\t%s\t%s" % ( src1, start1+sub_begin-sequence1[0:sub_begin].count('-'), start1+sub_end-sequence1[0:sub_end].count('-') ) - print >> fout, "%s\t%s\t%s" % ( src2, start2+sub_begin-sequence2[0:sub_begin].count('-'), start2+sub_end-sequence2[0:sub_end].count('-') ) - begin = False - else: - if begin: - print >> fout, "%s\t%s\t%s" % ( src1, start1+sub_begin-sequence1[0:sub_begin].count('-'), end1+sub_end-sequence1[0:sub_end].count('-') ) - print >> fout, "%s\t%s\t%s" % ( src2, start2+sub_begin-sequence2[0:sub_begin].count('-'), end2+sub_end-sequence2[0:sub_end].count('-') ) - begin = False - - -def main(): - skipped = 0 - not_pairwise = 0 - try: - maf_reader = bx.align.maf.Reader( open(inp_file, 'r') ) - except: - stop_err("Your MAF file appears to be malformed.") - print >> fout, "#Chr\tStart\tEnd" - for block in maf_reader: - if len(block.components) != 2: - not_pairwise += 1 - continue - try: - fetchSubs(block) - except: - skipped += 1 - - if not_pairwise: - print "Skipped %d non-pairwise blocks" % (not_pairwise) - if skipped: - print "Skipped %d blocks" % (skipped) - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/substitutions.xml b/tools/regVariation/substitutions.xml deleted file mode 100644 index d982db8725d..00000000000 --- a/tools/regVariation/substitutions.xml +++ /dev/null @@ -1,38 +0,0 @@ - - from pairwise alignments - - substitutions.py - $input - $out_file1 - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool takes a pairwise MAF file as input and fetches substitutions per alignment block. - ------ - -.. class:: warningmark - -**Note** - -Any block/s not containing exactly two sequences, will be omitted. - - - \ No newline at end of file diff --git a/tools/regVariation/windowSplitter.py b/tools/regVariation/windowSplitter.py deleted file mode 100644 index f1d9896d8ae..00000000000 --- a/tools/regVariation/windowSplitter.py +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env python - -""" -Split into windows. - -usage: %prog input size out_file - -l, --cols=N,N,N,N: Columns for chrom, start, end, strand in file -""" - -import sys - -from galaxy import eggs -import pkg_resources -pkg_resources.require( "bx-python" ) -from bx.cookbook import doc_optparse -from galaxy.tools.util.galaxyops import * - -def stop_err( msg ): - sys.stderr.write( msg ) - sys.exit() - - -def main(): - # Parsing Command Line here - options, args = doc_optparse.parse( __doc__ ) - - try: - chr_col_1, start_col_1, end_col_1, strand_col_1 = parse_cols_arg( options.cols ) - inp_file, winsize, out_file, makesliding, offset = args - winsize = int(winsize) - offset = int(offset) - makesliding = int(makesliding) - except: - stop_err( "Data issue, click the pencil icon in the history item to correct the metadata attributes of the input dataset." ) - - fo = open(out_file,'w') - - skipped_lines = 0 - first_invalid_line = 0 - invalid_line = None - if offset == 0: - makesliding = 0 - - for i, line in enumerate( file( inp_file ) ): - line = line.strip() - if line and line[0:1] != "#": - try: - elems = line.split('\t') - start = int(elems[start_col_1]) - end = int(elems[end_col_1]) - if makesliding == 0: - numwin = (end - start)/winsize - else: - numwin = (end - start)/offset - if numwin > 0: - for win in range(numwin): - elems_1 = elems - elems_1[start_col_1] = str(start) - elems_1[end_col_1] = str(start + winsize) - fo.write( "%s\n" % '\t'.join( elems_1 ) ) - if makesliding == 0: - start = start + winsize - else: - start = start + offset - if start+winsize > end: - break - except: - skipped_lines += 1 - if not invalid_line: - first_invalid_line = i + 1 - invalid_line = line - - fo.close() - - if makesliding == 1: - print 'Window size=%d, Sliding=Yes, Offset=%d' % ( winsize, offset ) - else: - print 'Window size=%d, Sliding=No' % (winsize) - if skipped_lines > 0: - print 'Skipped %d invalid lines starting with #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line ) - - -if __name__ == "__main__": - main() diff --git a/tools/regVariation/windowSplitter.xml b/tools/regVariation/windowSplitter.xml deleted file mode 100644 index 4caaf96edc6..00000000000 --- a/tools/regVariation/windowSplitter.xml +++ /dev/null @@ -1,104 +0,0 @@ - - - windowSplitter.py $input $size $out_file1 ${wintype.choice} ${wintype.offset} -l ${input.metadata.chromCol},${input.metadata.startCol},${input.metadata.endCol},${input.metadata.strandCol} - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -.. class:: infomark - -**What it does** - -This tool splits the intervals in the input file into smaller intervals based on the specified window-size and window type. - ------ - -.. class:: warningmark - -**Note** - -The positions at the end of the input interval which do not fit into the last window or a new window of required size, will be omitted from the output. - ------ - -.. class:: infomark - -**About formats** - -**BED format** Browser Extensible Data format was designed at UCSC for displaying data tracks in the Genome Browser. It has three required fields and several additional optional ones: - -The first three BED fields (required) are:: - - 1. chrom - The name of the chromosome (e.g. chr1, chrY_random). - 2. chromStart - The starting position in the chromosome. (The first base in a chromosome is numbered 0.) - 3. chromEnd - The ending position in the chromosome, plus 1 (i.e., a half-open interval). - -The additional BED fields (optional) are:: - - 4. name - The name of the BED line. - 5. score - A score between 0 and 1000. - 6. strand - Defines the strand - either '+' or '-'. - 7. thickStart - The starting position where the feature is drawn thickly at the Genome Browser. - 8. thickEnd - The ending position where the feature is drawn thickly at the Genome Browser. - 9. reserved - This should always be set to zero. - 10. blockCount - The number of blocks (exons) in the BED line. - 11. blockSizes - A comma-separated list of the block sizes. The number of items in this list should correspond to blockCount. - 12. blockStarts - A comma-separated list of block starts. All of the blockStart positions should be calculated relative to chromStart. The number of items in this list should correspond to blockCount. - 13. expCount - The number of experiments. - 14. expIds - A comma-separated list of experiment ids. The number of items in this list should correspond to expCount. - 15. expScores - A comma-separated list of experiment scores. All of the expScores should be relative to expIds. The number of items in this list should correspond to expCount. - ------ - -**Example** - -- For the following dataset:: - - chr22 1000 4700 NM_174568 0 + - -- running this tool with **Window size as 1000**, will return:: - - chr22 1000 2000 NM_174568 0 + - chr22 2000 3000 NM_174568 0 + - chr22 3000 4000 NM_174568 0 + - -- running this tool to make **Sliding windows** of **size 1000** and **offset 500**, will return:: - - chr22 1000 2000 NM_174568 0 + - chr22 1500 2500 NM_174568 0 + - chr22 2000 3000 NM_174568 0 + - chr22 2500 3500 NM_174568 0 + - chr22 3000 4000 NM_174568 0 + - chr22 3500 4500 NM_174568 0 + - - - - - \ No newline at end of file