More BLAST tests

This commit is contained in:
Kanwei Li
2011-04-06 12:38:00 -04:00
parent d969f29c01
commit af3f4a2ff9
8 changed files with 961 additions and 49 deletions
@@ -0,0 +1,722 @@
<?xml version="1.0"?>
<!DOCTYPE BlastOutput PUBLIC "-//NCBI//NCBI BlastOutput/EN" "NCBI_BlastOutput.dtd">
<BlastOutput>
<BlastOutput_program>tblastn</BlastOutput_program>
<BlastOutput_version>TBLASTN 2.2.25+</BlastOutput_version>
<BlastOutput_reference>Stephen F. Altschul, Thomas L. Madden, Alejandro A. Sch&amp;auml;ffer, Jinghui Zhang, Zheng Zhang, Webb Miller, and David J. Lipman (1997), &quot;Gapped BLAST and PSI-BLAST: a new generation of protein database search programs&quot;, Nucleic Acids Res. 25:3389-3402.</BlastOutput_reference>
<BlastOutput_db></BlastOutput_db>
<BlastOutput_query-ID>Query_1</BlastOutput_query-ID>
<BlastOutput_query-def>sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</BlastOutput_query-def>
<BlastOutput_query-len>406</BlastOutput_query-len>
<BlastOutput_param>
<Parameters>
<Parameters_matrix>BLOSUM80</Parameters_matrix>
<Parameters_expect>1e-10</Parameters_expect>
<Parameters_gap-open>10</Parameters_gap-open>
<Parameters_gap-extend>1</Parameters_gap-extend>
<Parameters_filter>F</Parameters_filter>
</Parameters>
</BlastOutput_param>
<BlastOutput_iterations>
<Iteration>
<Iteration_iter-num>1</Iteration_iter-num>
<Iteration_query-ID>Query_1</Iteration_query-ID>
<Iteration_query-def>sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>406</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>19</Statistics_hsp-len>
<Statistics_eff-space>127710</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>2</Iteration_iter-num>
<Iteration_query-ID>Query_1</Iteration_query-ID>
<Iteration_query-def>sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>406</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>19</Statistics_hsp-len>
<Statistics_eff-space>127710</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>3</Iteration_iter-num>
<Iteration_query-ID>Query_1</Iteration_query-ID>
<Iteration_query-def>sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>406</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>19</Statistics_hsp-len>
<Statistics_eff-space>127710</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>4</Iteration_iter-num>
<Iteration_query-ID>Query_1</Iteration_query-ID>
<Iteration_query-def>sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>406</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>19</Statistics_hsp-len>
<Statistics_eff-space>127710</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>5</Iteration_iter-num>
<Iteration_query-ID>Query_1</Iteration_query-ID>
<Iteration_query-def>sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>406</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>19</Statistics_hsp-len>
<Statistics_eff-space>127710</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>6</Iteration_iter-num>
<Iteration_query-ID>Query_1</Iteration_query-ID>
<Iteration_query-def>sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>406</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>19</Statistics_hsp-len>
<Statistics_eff-space>127710</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>7</Iteration_iter-num>
<Iteration_query-ID>Query_2</Iteration_query-ID>
<Iteration_query-def>sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2</Iteration_query-def>
<Iteration_query-len>1161</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>23</Statistics_hsp-len>
<Statistics_eff-space>370988</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>8</Iteration_iter-num>
<Iteration_query-ID>Query_2</Iteration_query-ID>
<Iteration_query-def>sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2</Iteration_query-def>
<Iteration_query-len>1161</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>23</Statistics_hsp-len>
<Statistics_eff-space>370988</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>9</Iteration_iter-num>
<Iteration_query-ID>Query_2</Iteration_query-ID>
<Iteration_query-def>sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2</Iteration_query-def>
<Iteration_query-len>1161</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>23</Statistics_hsp-len>
<Statistics_eff-space>370988</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>10</Iteration_iter-num>
<Iteration_query-ID>Query_2</Iteration_query-ID>
<Iteration_query-def>sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2</Iteration_query-def>
<Iteration_query-len>1161</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>23</Statistics_hsp-len>
<Statistics_eff-space>370988</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>11</Iteration_iter-num>
<Iteration_query-ID>Query_2</Iteration_query-ID>
<Iteration_query-def>sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2</Iteration_query-def>
<Iteration_query-len>1161</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>23</Statistics_hsp-len>
<Statistics_eff-space>370988</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>12</Iteration_iter-num>
<Iteration_query-ID>Query_2</Iteration_query-ID>
<Iteration_query-def>sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2</Iteration_query-def>
<Iteration_query-len>1161</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>23</Statistics_hsp-len>
<Statistics_eff-space>370988</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>13</Iteration_iter-num>
<Iteration_query-ID>Query_3</Iteration_query-ID>
<Iteration_query-def>sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4</Iteration_query-def>
<Iteration_query-len>1382</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>24</Statistics_hsp-len>
<Statistics_eff-space>441350</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>14</Iteration_iter-num>
<Iteration_query-ID>Query_3</Iteration_query-ID>
<Iteration_query-def>sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4</Iteration_query-def>
<Iteration_query-len>1382</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>24</Statistics_hsp-len>
<Statistics_eff-space>441350</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>15</Iteration_iter-num>
<Iteration_query-ID>Query_3</Iteration_query-ID>
<Iteration_query-def>sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4</Iteration_query-def>
<Iteration_query-len>1382</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>24</Statistics_hsp-len>
<Statistics_eff-space>441350</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>16</Iteration_iter-num>
<Iteration_query-ID>Query_3</Iteration_query-ID>
<Iteration_query-def>sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4</Iteration_query-def>
<Iteration_query-len>1382</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>24</Statistics_hsp-len>
<Statistics_eff-space>441350</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>17</Iteration_iter-num>
<Iteration_query-ID>Query_3</Iteration_query-ID>
<Iteration_query-def>sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4</Iteration_query-def>
<Iteration_query-len>1382</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>24</Statistics_hsp-len>
<Statistics_eff-space>441350</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>18</Iteration_iter-num>
<Iteration_query-ID>Query_3</Iteration_query-ID>
<Iteration_query-def>sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4</Iteration_query-def>
<Iteration_query-len>1382</Iteration_query-len>
<Iteration_hits></Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>24</Statistics_hsp-len>
<Statistics_eff-space>441350</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
<Iteration_message>No hits found</Iteration_message>
</Iteration>
<Iteration>
<Iteration_iter-num>19</Iteration_iter-num>
<Iteration_query-ID>Query_4</Iteration_query-ID>
<Iteration_query-def>sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>348</Iteration_query-len>
<Iteration_hits>
<Hit>
<Hit_num>1</Hit_num>
<Hit_id>Subject_1</Hit_id>
<Hit_def>gi|57163782|ref|NM_001009242.1| Felis catus rhodopsin (RHO), mRNA</Hit_def>
<Hit_accession>Subject_1</Hit_accession>
<Hit_len>1047</Hit_len>
<Hit_hsps>
<Hsp>
<Hsp_num>1</Hsp_num>
<Hsp_bit-score>732.392902459534</Hsp_bit-score>
<Hsp_score>1689</Hsp_score>
<Hsp_evalue>0</Hsp_evalue>
<Hsp_query-from>1</Hsp_query-from>
<Hsp_query-to>348</Hsp_query-to>
<Hsp_hit-from>1</Hsp_hit-from>
<Hsp_hit-to>1044</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>1</Hsp_hit-frame>
<Hsp_identity>336</Hsp_identity>
<Hsp_positive>343</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>348</Hsp_align-len>
<Hsp_qseq>MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEASATVSKTETSQVAPA</Hsp_qseq>
<Hsp_hseq>MNGTEGPNFYVPFSNKTGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVFGGFTTTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLVGWSRYIPEGMQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIVIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTLPAFFAKSSSIYNPVIYIMMNKQFRNCMLTTLCCGKNPLGDDEASTTGSKTETSQVAPA</Hsp_hseq>
<Hsp_midline>MNGTEGPNFYVPFSN TGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMV GGFT+TLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPL GWSRYIPEG+QCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMI+IFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMT+PAFFAKS++IYNPVIYIMMNKQFRNCMLTT+CCGKNPLGDDEAS T SKTETSQVAPA</Hsp_midline>
</Hsp>
</Hit_hsps>
</Hit>
</Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>18</Statistics_hsp-len>
<Statistics_eff-space>109230</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
</Iteration>
<Iteration>
<Iteration_iter-num>20</Iteration_iter-num>
<Iteration_query-ID>Query_4</Iteration_query-ID>
<Iteration_query-def>sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>348</Iteration_query-len>
<Iteration_hits>
<Hit>
<Hit_num>1</Hit_num>
<Hit_id>Subject_2</Hit_id>
<Hit_def>gi|2734705|gb|U59921.1|BBU59921 Bufo bufo rhodopsin mRNA, complete cds</Hit_def>
<Hit_accession>Subject_2</Hit_accession>
<Hit_len>1574</Hit_len>
<Hit_hsps>
<Hsp>
<Hsp_num>1</Hsp_num>
<Hsp_bit-score>646.119739014374</Hsp_bit-score>
<Hsp_score>1489</Hsp_score>
<Hsp_evalue>0</Hsp_evalue>
<Hsp_query-from>1</Hsp_query-from>
<Hsp_query-to>341</Hsp_query-to>
<Hsp_hit-from>42</Hsp_hit-from>
<Hsp_hit-to>1067</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>3</Hsp_hit-frame>
<Hsp_identity>290</Hsp_identity>
<Hsp_positive>320</Hsp_positive>
<Hsp_gaps>1</Hsp_gaps>
<Hsp_align-len>342</Hsp_align-len>
<Hsp_qseq>MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEA-SATVSKTE</Hsp_qseq>
<Hsp_hseq>MNGTEGPNFYIPMSNKTGVVRSPFEYPQYYLAEPWQYSILCAYMFLLILLGFPINFMTLYVTIQHKKLRTPLNYILLNLAFANHFMVLCGFTVTMYSSMNGYFILGATGCYVEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFSENHAVMGVAFTWIMALSCAVPPLLGWSRYIPEGMQCSCGVDYYTLKPEVNNESFVIYMFVVHFTIPLIIIFFCYGRLVCTVKEAAAQQQESATTQKAEKEVTRMVIIMVVFFLICWVPYASVAFFIFSNQGSEFGPIFMTVPAFFAKSSSIYNPVIYIMLNKQFRNCMITTLCCGKNPFGEDDASSAATSKTE</Hsp_hseq>
<Hsp_midline>MNGTEGPNFY+P SN TGVVRSPFEYPQYYLAEPWQ+S+L AYMFLLI+LGFPINF+TLYVT+QHKKLRTPLNYILLNLA A+ FMVL GFT T+Y+S+ GYF+ G TGC +EGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRF ENHA+MGVAFTW+MAL+CA PPL GWSRYIPEG+QCSCG+DYYTLKPEVNNESFVIYMFVVHFTIP+IIIFFCYG+LV TVKEAAAQQQESATTQKAEKEVTRMVIIMV+ FLICWVPYASVAF+IF+ QGS FGPIFMT+PAFFAKS++IYNPVIYIM+NKQFRNCM+TT+CCGKNP G+D+A SA SKTE</Hsp_midline>
</Hsp>
</Hit_hsps>
</Hit>
</Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>18</Statistics_hsp-len>
<Statistics_eff-space>109230</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
</Iteration>
<Iteration>
<Iteration_iter-num>21</Iteration_iter-num>
<Iteration_query-ID>Query_4</Iteration_query-ID>
<Iteration_query-def>sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>348</Iteration_query-len>
<Iteration_hits>
<Hit>
<Hit_num>1</Hit_num>
<Hit_id>Subject_3</Hit_id>
<Hit_def>gi|283855845|gb|GQ290303.1| Cynopterus brachyotis voucher 20020434 rhodopsin (RHO) gene, exons 1 through 5 and partial cds</Hit_def>
<Hit_accession>Subject_3</Hit_accession>
<Hit_len>4301</Hit_len>
<Hit_hsps>
<Hsp>
<Hsp_num>1</Hsp_num>
<Hsp_bit-score>151.343146656381</Hsp_bit-score>
<Hsp_score>342</Hsp_score>
<Hsp_evalue>1.39566684546685e-72</Hsp_evalue>
<Hsp_query-from>239</Hsp_query-from>
<Hsp_query-to>312</Hsp_query-to>
<Hsp_hit-from>3147</Hsp_hit-from>
<Hsp_hit-to>3368</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>3</Hsp_hit-frame>
<Hsp_identity>69</Hsp_identity>
<Hsp_positive>73</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>74</Hsp_align-len>
<Hsp_qseq>ESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQ</Hsp_qseq>
<Hsp_hseq>ESATTQKAEKEVTRMVIIMVIAFLICWLPYAGVAFYIFTHQGSNFGPIFMTLPAFFAKSSSIYNPVIYIMMNKQ</Hsp_hseq>
<Hsp_midline>ESATTQKAEKEVTRMVIIMVIAFLICW+PYA VAFYIFTHQGSNFGPIFMT+PAFFAKS++IYNPVIYIMMNKQ</Hsp_midline>
</Hsp>
<Hsp>
<Hsp_num>2</Hsp_num>
<Hsp_bit-score>126.323929257285</Hsp_bit-score>
<Hsp_score>284</Hsp_score>
<Hsp_evalue>1.39566684546685e-72</Hsp_evalue>
<Hsp_query-from>177</Hsp_query-from>
<Hsp_query-to>235</Hsp_query-to>
<Hsp_hit-from>2855</Hsp_hit-from>
<Hsp_hit-to>3031</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>2</Hsp_hit-frame>
<Hsp_identity>54</Hsp_identity>
<Hsp_positive>57</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>59</Hsp_align-len>
<Hsp_qseq>RYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAA</Hsp_qseq>
<Hsp_hseq>RYIPEGMQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIVIFFCYGQLVFTVKEVRS</Hsp_hseq>
<Hsp_midline>RYIPEG+QCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMI+IFFCYGQLVFTVKE +</Hsp_midline>
</Hsp>
<Hsp>
<Hsp_num>3</Hsp_num>
<Hsp_bit-score>229.420359574251</Hsp_bit-score>
<Hsp_score>523</Hsp_score>
<Hsp_evalue>9.84654801241353e-65</Hsp_evalue>
<Hsp_query-from>11</Hsp_query-from>
<Hsp_query-to>121</Hsp_query-to>
<Hsp_hit-from>1</Hsp_hit-from>
<Hsp_hit-to>333</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>1</Hsp_hit-frame>
<Hsp_identity>107</Hsp_identity>
<Hsp_positive>109</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>111</Hsp_align-len>
<Hsp_qseq>VPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGG</Hsp_qseq>
<Hsp_hseq>VPFSNKTGVVRSPFEHPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVFGGFTTTLYTSLHGYFVFGPTGCNLEGFFATLGG</Hsp_hseq>
<Hsp_midline>VPFSN TGVVRSPFE+PQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMV GGFT+TLYTSLHGYFVFGPTGCNLEGFFATLGG</Hsp_midline>
</Hsp>
<Hsp>
<Hsp_num>4</Hsp_num>
<Hsp_bit-score>122.873002719478</Hsp_bit-score>
<Hsp_score>276</Hsp_score>
<Hsp_evalue>1.40732096096596e-32</Hsp_evalue>
<Hsp_query-from>119</Hsp_query-from>
<Hsp_query-to>177</Hsp_query-to>
<Hsp_hit-from>1404</Hsp_hit-from>
<Hsp_hit-to>1580</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>3</Hsp_hit-frame>
<Hsp_identity>55</Hsp_identity>
<Hsp_positive>56</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>59</Hsp_align-len>
<Hsp_qseq>LGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSR</Hsp_qseq>
<Hsp_hseq>LAGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGLALTWVMALACAAPPLVGWSR</Hsp_hseq>
<Hsp_midline>L GEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMG+A TWVMALACAAPPL GWSR</Hsp_midline>
</Hsp>
<Hsp>
<Hsp_num>5</Hsp_num>
<Hsp_bit-score>57.7367643183824</Hsp_bit-score>
<Hsp_score>125</Hsp_score>
<Hsp_evalue>5.60065526485586e-13</Hsp_evalue>
<Hsp_query-from>312</Hsp_query-from>
<Hsp_query-to>337</Hsp_query-to>
<Hsp_hit-from>4222</Hsp_hit-from>
<Hsp_hit-to>4299</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>1</Hsp_hit-frame>
<Hsp_identity>23</Hsp_identity>
<Hsp_positive>24</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>26</Hsp_align-len>
<Hsp_qseq>QFRNCMLTTICCGKNPLGDDEASATV</Hsp_qseq>
<Hsp_hseq>QFRNCMLTTLCCGKNPLGDDEASTTA</Hsp_hseq>
<Hsp_midline>QFRNCMLTT+CCGKNPLGDDEAS T </Hsp_midline>
</Hsp>
</Hit_hsps>
</Hit>
</Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>18</Statistics_hsp-len>
<Statistics_eff-space>109230</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
</Iteration>
<Iteration>
<Iteration_iter-num>22</Iteration_iter-num>
<Iteration_query-ID>Query_4</Iteration_query-ID>
<Iteration_query-def>sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>348</Iteration_query-len>
<Iteration_hits>
<Hit>
<Hit_num>1</Hit_num>
<Hit_id>Subject_4</Hit_id>
<Hit_def>gi|283855822|gb|GQ290312.1| Myotis ricketti voucher GQX10 rhodopsin (RHO) mRNA, partial cds</Hit_def>
<Hit_accession>Subject_4</Hit_accession>
<Hit_len>983</Hit_len>
<Hit_hsps>
<Hsp>
<Hsp_num>1</Hsp_num>
<Hsp_bit-score>658.197981896696</Hsp_bit-score>
<Hsp_score>1517</Hsp_score>
<Hsp_evalue>0</Hsp_evalue>
<Hsp_query-from>11</Hsp_query-from>
<Hsp_query-to>336</Hsp_query-to>
<Hsp_hit-from>1</Hsp_hit-from>
<Hsp_hit-to>978</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>1</Hsp_hit-frame>
<Hsp_identity>310</Hsp_identity>
<Hsp_positive>322</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>326</Hsp_align-len>
<Hsp_qseq>VPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEASAT</Hsp_qseq>
<Hsp_hseq>VPFSNKTGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVANLFMVFGGFTTTLYTSMHGYFVFGATGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGLAFTWVMALACAAPPLAGWSRYIPEGMQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIVIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVVAFLICWLPYASVAFYIFTHQGSNFGPVFMTIPAFFAKSSSIYNPVIYIMMNKQFRNCMLTTLCCGKNPLGDDEASTT</Hsp_hseq>
<Hsp_midline>VPFSN TGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVA+LFMV GGFT+TLYTS+HGYFVFG TGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMG+AFTWVMALACAAPPLAGWSRYIPEG+QCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMI+IFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMV+AFLICW+PYASVAFYIFTHQGSNFGP+FMTIPAFFAKS++IYNPVIYIMMNKQFRNCMLTT+CCGKNPLGDDEAS T</Hsp_midline>
</Hsp>
</Hit_hsps>
</Hit>
</Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>18</Statistics_hsp-len>
<Statistics_eff-space>109230</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
</Iteration>
<Iteration>
<Iteration_iter-num>23</Iteration_iter-num>
<Iteration_query-ID>Query_4</Iteration_query-ID>
<Iteration_query-def>sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>348</Iteration_query-len>
<Iteration_hits>
<Hit>
<Hit_num>1</Hit_num>
<Hit_id>Subject_5</Hit_id>
<Hit_def>gi|18148870|dbj|AB062417.1| Synthetic construct Bos taurus gene for rhodopsin, complete cds</Hit_def>
<Hit_accession>Subject_5</Hit_accession>
<Hit_len>1047</Hit_len>
<Hit_hsps>
<Hsp>
<Hsp_num>1</Hsp_num>
<Hsp_bit-score>711.255977415469</Hsp_bit-score>
<Hsp_score>1640</Hsp_score>
<Hsp_evalue>0</Hsp_evalue>
<Hsp_query-from>1</Hsp_query-from>
<Hsp_query-to>348</Hsp_query-to>
<Hsp_hit-from>1</Hsp_hit-from>
<Hsp_hit-to>1044</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>1</Hsp_hit-frame>
<Hsp_identity>325</Hsp_identity>
<Hsp_positive>337</Hsp_positive>
<Hsp_gaps>0</Hsp_gaps>
<Hsp_align-len>348</Hsp_align-len>
<Hsp_qseq>MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEASATVSKTETSQVAPA</Hsp_qseq>
<Hsp_hseq>MNGTEGPNFYVPFSNKTGVVRSPFEAPQYYLAEPWQFSMLAAYMFLLIMLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVFGGFTTTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLVGWSRYIPEGMQCSCGIDYYTPHEETNNESFVIYMFVVHFIIPLIVIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWLPYAGVAFYIFTHQGSDFGPIFMTIPAFFAKTSAVYNPVIYIMMNKQFRNCMVTTLCCGKNPLGDDEASTTVSKTETSQVAPA</Hsp_hseq>
<Hsp_midline>MNGTEGPNFYVPFSN TGVVRSPFE PQYYLAEPWQFSMLAAYMFLLI+LGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMV GGFT+TLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPL GWSRYIPEG+QCSCGIDYYT E NNESFVIYMFVVHF IP+I+IFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICW+PYA VAFYIFTHQGS+FGPIFMTIPAFFAK++A+YNPVIYIMMNKQFRNCM+TT+CCGKNPLGDDEAS TVSKTETSQVAPA</Hsp_midline>
</Hsp>
</Hit_hsps>
</Hit>
</Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>18</Statistics_hsp-len>
<Statistics_eff-space>109230</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
</Iteration>
<Iteration>
<Iteration_iter-num>24</Iteration_iter-num>
<Iteration_query-ID>Query_4</Iteration_query-ID>
<Iteration_query-def>sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1</Iteration_query-def>
<Iteration_query-len>348</Iteration_query-len>
<Iteration_hits>
<Hit>
<Hit_num>1</Hit_num>
<Hit_id>Subject_6</Hit_id>
<Hit_def>gi|12583664|dbj|AB043817.1| Conger myriaster conf gene for fresh water form rod opsin, complete cds</Hit_def>
<Hit_accession>Subject_6</Hit_accession>
<Hit_len>1344</Hit_len>
<Hit_hsps>
<Hsp>
<Hsp_num>1</Hsp_num>
<Hsp_bit-score>626.708277239213</Hsp_bit-score>
<Hsp_score>1444</Hsp_score>
<Hsp_evalue>0</Hsp_evalue>
<Hsp_query-from>1</Hsp_query-from>
<Hsp_query-to>341</Hsp_query-to>
<Hsp_hit-from>23</Hsp_hit-from>
<Hsp_hit-to>1048</Hsp_hit-to>
<Hsp_query-frame>0</Hsp_query-frame>
<Hsp_hit-frame>2</Hsp_hit-frame>
<Hsp_identity>281</Hsp_identity>
<Hsp_positive>311</Hsp_positive>
<Hsp_gaps>1</Hsp_gaps>
<Hsp_align-len>342</Hsp_align-len>
<Hsp_qseq>MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPL-GDDEASATVSKTE</Hsp_qseq>
<Hsp_hseq>MNGTEGPNFYIPMSNATGVVRSPFEYPQYYLAEPWAFSALSAYMFFLIIAGFPINFLTLYVTIEHKKLRTPLNYILLNLAVADLFMVFGGFTTTMYTSMHGYFVFGPTGCNIEGFFATLGGEIALWCLVVLAIERWMVVCKPVTNFRFGESHAIMGVMVTWTMALACALPPLFGWSRYIPEGLQCSCGIDYYTRAPGINNESFVIYMFTCHFSIPLAVISFCYGRLVCTVKEAAAQQQESETTQRAEREVTRMVVIMVISFLVCWVPYASVAWYIFTHQGSTFGPIFMTIPSFFAKSSALYNPMIYICMNKQFRHCMITTLCCGKNPFEEEDGASATSSKTE</Hsp_hseq>
<Hsp_midline>MNGTEGPNFY+P SNATGVVRSPFEYPQYYLAEPW FS L+AYMF LI+ GFPINFLTLYVT++HKKLRTPLNYILLNLAVADLFMV GGFT+T+YTS+HGYFVFGPTGCN+EGFFATLGGEIALW LVVLAIER++VVCKP++NFRFGE HAIMGV TW MALACA PPL GWSRYIPEGLQCSCGIDYYT P +NNESFVIYMF HF+IP+ +I FCYG+LV TVKEAAAQQQES TTQ+AE+EVTRMV+IMVI+FL+CWVPYASVA YIFTHQGS FGPIFMTIP+FFAKS+A+YNP+IYI MNKQFR CM+TT+CCGKNP +D ASAT SKTE</Hsp_midline>
</Hsp>
</Hit_hsps>
</Hit>
</Iteration_hits>
<Iteration_stat>
<Statistics>
<Statistics_db-num>0</Statistics_db-num>
<Statistics_db-len>0</Statistics_db-len>
<Statistics_hsp-len>18</Statistics_hsp-len>
<Statistics_eff-space>109230</Statistics_eff-space>
<Statistics_kappa>0.071</Statistics_kappa>
<Statistics_lambda>0.299</Statistics_lambda>
<Statistics_entropy>0.27</Statistics_entropy>
</Statistics>
</Iteration_stat>
</Iteration>
</BlastOutput_iterations>
</BlastOutput>
+69 -13
View File
@@ -2,11 +2,12 @@
"""Convert a BLAST XML file to 12 column tabular output
Takes three command line options, input BLAST XML filename, output tabular
BLAST filename, output format (std for standard 12 columns, or x22 for the
extended 22 columns offered in the BLAST+ wrappers).
BLAST filename, output format (std for standard 12 columns, or ext for the
extended 24 columns offered in the BLAST+ wrappers).
The 12 colums output are 'qseqid sseqid pident length mismatch gapopen qstart
qend sstart send evalue bitscore' which mean:
qend sstart send evalue bitscore' or 'std' at the BLAST+ command line, which
mean:
====== ========= ============================================
Column NCBI name Description
@@ -25,7 +26,7 @@ Column NCBI name Description
12 bitscore Bit score
====== ========= ============================================
The additional columns are:
The additional columns offered in the Galaxy BLAST+ wrappers are:
====== ============= ===========================================
Column NCBI name Description
@@ -40,20 +41,26 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
23 qlen Query sequence length
24 slen Subject sequence length
====== ============= ===========================================
Most of these fields are given explicitly in the XML file, others some like
the percentage identity and the number of gap openings must be calculated.
Be aware that the sequence in the extended tabular output or XML direct from
BLAST+ may or may not use XXXX masking on regions of low complexity. This
can throw the off the calculation of percentage identity and gap openings.
[In fact, both BLAST 2.2.24+ and 2.2.25+ have a sutle bug in this regard,
with these numbers changing depending on whether or not the low complexity
filter is used.]
This script attempts to produce idential output to what BLAST+ would have done.
However, check this with "diff -b ..." since BLAST+ sometimes includes an extra
space character (probably a bug).
Beware that if using the extended output, the XML file contains the original
aligned sequences, but the tabular output direct from BLAST+ may use XXXX
masking on regions of low complexity (columns 21 and 22).
"""
import sys
import re
assert sys.version_info[:2] >= ( 2, 4 )
if sys.version_info[:2] >= ( 2, 5 ):
@@ -69,14 +76,16 @@ def stop_err( msg ):
try:
in_file, out_file, out_fmt = sys.argv[1:]
except:
stop_err("Expect 3 arguments: input BLAST XML file, output tabular file, out format (std or x22)")
stop_err("Expect 3 arguments: input BLAST XML file, output tabular file, out format (std or ext)")
if out_fmt == "std":
extended = False
elif out_fmt == "x22":
stop_err("Format argument x22 has been replaced with ext (extended 24 columns)")
elif out_fmt == "ext":
extended = True
else:
stop_err("Format argument should be std (12 column) or x22 (extended 22 column)")
stop_err("Format argument should be std (12 column) or ext (extended 24 columns)")
# get an iterable
@@ -92,6 +101,18 @@ try:
except:
stop_err( "Invalid data format." )
re_default_query_id = re.compile("^Query_\d+$")
assert re_default_query_id.match("Query_101")
assert not re_default_query_id.match("Query_101a")
assert not re_default_query_id.match("MyQuery_101")
re_default_subject_id = re.compile("^Subject_\d+$")
assert re_default_subject_id.match("Subject_1")
assert not re_default_subject_id.match("Subject_")
assert not re_default_subject_id.match("Subject_12a")
assert not re_default_subject_id.match("TheSubject_1")
outfile = open(out_file, 'w')
blast_program = None
for event, elem in context:
@@ -99,10 +120,42 @@ for event, elem in context:
blast_program = elem.text
# for every <Iteration> tag
if event == "end" and elem.tag == "Iteration":
qseqid = elem.findtext("Iteration_query-def").split(None,1)[0]
#Expecting either this, from BLAST 2.2.25+ using FASTA vs FASTA
# <Iteration_query-ID>sp|Q9BS26|ERP44_HUMAN</Iteration_query-ID>
# <Iteration_query-def>Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1</Iteration_query-def>
# <Iteration_query-len>406</Iteration_query-len>
# <Iteration_hits></Iteration_hits>
#
#Or, from BLAST 2.2.24+ run online
# <Iteration_query-ID>Query_1</Iteration_query-ID>
# <Iteration_query-def>Sample</Iteration_query-def>
# <Iteration_query-len>516</Iteration_query-len>
# <Iteration_hits>...
qseqid = elem.findtext("Iteration_query-ID")
if re_default_query_id.match(qseqid):
#Place holder ID, take the first word of the query definition
qseqid = elem.findtext("Iteration_query-def").split(None,1)[0]
qlen = int(elem.findtext("Iteration_query-len"))
# for every <Hit> within <Iteration>
for hit in elem.findall("Iteration_hits/Hit/"):
#Expecting either this,
# <Hit_id>gi|3024260|sp|P56514.1|OPSD_BUFBU</Hit_id>
# <Hit_def>RecName: Full=Rhodopsin</Hit_def>
# <Hit_accession>P56514</Hit_accession>
#or,
# <Hit_id>Subject_1</Hit_id>
# <Hit_def>gi|57163783|ref|NP_001009242.1| rhodopsin [Felis catus]</Hit_def>
# <Hit_accession>Subject_1</Hit_accession>
#
#apparently depending on the parse_deflines switch
sseqid = hit.findtext("Hit_id").split(None,1)[0]
hit_def = sseqid + " " + hit.findtext("Hit_def")
if re_default_subject_id.match(sseqid) \
and sseqid == hit.findtext("Hit_accession"):
#Place holder ID, take the first word of the subject definition
hit_def = hit.findtext("Hit_def")
sseqid = hit_def.split(None,1)[0]
# for every <Hsp> within <Hit>
for hsp in hit.findall("Hit_hsps/Hsp"):
nident = hsp.findtext("Hsp_identity")
@@ -164,7 +217,6 @@ for event, elem in context:
]
if extended:
hit_def = sseqid + " " + hit.findtext("Hit_def")
sallseqid = ";".join(name.split(None,1)[0] for name in hit_def.split(">"))
#print hit_def, "-->", sallseqid
positive = hsp.findtext("Hsp_positive")
@@ -175,6 +227,7 @@ for event, elem in context:
#Probably a bug in BLASTP that they use 0 or 1 depending on format
if qframe == "0": qframe = "1"
if sframe == "0": sframe = "1"
slen = int(hit.findtext("Hit_len"))
values.extend([sallseqid,
hsp.findtext("Hsp_score"), #score,
nident,
@@ -185,7 +238,10 @@ for event, elem in context:
sframe,
#NOTE - for blastp, XML shows original seq, tabular uses XXX masking
q_seq,
h_seq])
h_seq,
str(qlen),
str(slen),
])
#print "\t".join(values)
outfile.write("\t".join(values) + "\n")
# prevents ElementTree from growing large datastructure
+35 -8
View File
@@ -1,4 +1,4 @@
<tool id="blastxml_to_tabular" name="BLAST XML to tabular" version="0.0.6">
<tool id="blastxml_to_tabular" name="BLAST XML to tabular" version="0.0.8">
<description>Convert BLAST XML output to tabular</description>
<command interpreter="python">
blastxml_to_tabular.py $blastxml_file $tabular_file $out_format
@@ -7,7 +7,7 @@
<param name="blastxml_file" type="data" format="blastxml" label="BLAST results as XML"/>
<param name="out_format" type="select" label="Output format">
<option value="std" selected="True">Tabular (standard 12 columns)</option>
<option value="x22">Tabular (extended 22 columns)</option>
<option value="ext">Tabular (extended 24 columns)</option>
</param>
</inputs>
<outputs>
@@ -16,12 +16,36 @@
<requirements>
</requirements>
<tests>
<test>
<param name="blastxml_file" value="blastp_four_human_vs_rhodopsin.xml" ftype="blastxml" />
<param name="out_format" value="std" />
<!-- Note this has some white space differences from the actual blastp output blast_four_human_vs_rhodopsin.tabluar -->
<output name="tabular_file" file="blastp_four_human_vs_rhodopsin_converted.tabular" ftype="tabular" />
</test>
<test>
<param name="blastxml_file" value="blastp_four_human_vs_rhodopsin.xml" ftype="blastxml" />
<param name="out_format" value="ext" />
<!-- Note this has some white space differences from the actual blastp output blast_four_human_vs_rhodopsin_22c.tabluar -->
<output name="tabular_file" file="blastp_four_human_vs_rhodopsin_converted_ext.tabular" ftype="tabular" />
</test>
<test>
<param name="blastxml_file" value="blastp_sample.xml" ftype="blastxml" />
<param name="out_format" value="std" />
<!-- Note this has some white space differences from the actual blastx output -->
<!-- Note this has some white space differences from the actual blastp output -->
<output name="tabular_file" file="blastp_sample_converted.tabular" ftype="tabular" />
</test>
<test>
<param name="blastxml_file" value="blastx_rhodopsin_vs_four_human.xml" ftype="blastxml" />
<param name="out_format" value="std" />
<!-- Note this has some white space differences from the actual blastx output -->
<output name="tabular_file" file="blastx_rhodopsin_vs_four_human_converted.tabular" ftype="tabular" />
</test>
<test>
<param name="blastxml_file" value="blastx_rhodopsin_vs_four_human.xml" ftype="blastxml" />
<param name="out_format" value="ext" />
<!-- Note this has some white space and XXXX masking differences from the actual blastx output -->
<output name="tabular_file" file="blastx_rhodopsin_vs_four_human_converted_ext.tabular" ftype="tabular" />
</test>
<test>
<param name="blastxml_file" value="blastx_sample.xml" ftype="blastxml" />
<param name="out_format" value="std" />
@@ -36,9 +60,9 @@
</test>
<test>
<param name="blastxml_file" value="blastp_human_vs_pdb_seg_no.xml" ftype="blastxml" />
<param name="out_format" value="x22" />
<param name="out_format" value="ext" />
<!-- Note this has some white space differences from the actual blastp output -->
<output name="tabular_file" file="blastp_human_vs_pdb_seg_no_converted_x22.tabular" ftype="tabular" />
<output name="tabular_file" file="blastp_human_vs_pdb_seg_no_converted_ext.tabular" ftype="tabular" />
</test>
</tests>
<help>
@@ -90,11 +114,14 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
23 qlen Query sequence length
24 slen Subject sequence length
====== ============= ===========================================
Beware that the XML file (and thus the conversion) contains the original
aligned sequences, but the tabular output direct from BLAST+ may use XXXX
masking on regions of low complexity (columns 21 and 22).
Beware that the XML file (and thus the conversion) and the tabular output
direct from BLAST+ may differ in the presence of XXXX masking on regions
low complexity (columns 21 and 22), and thus also calculated figures like
the percentage idenity (column 3).
</help>
</tool>
+11 -6
View File
@@ -1,4 +1,4 @@
<tool id="ncbi_blastn_wrapper" name="NCBI BLAST+ blastn" version="0.0.8">
<tool id="ncbi_blastn_wrapper" name="NCBI BLAST+ blastn" version="0.0.9">
<description>Search nucleotide database with nucleotide query sequence(s)</description>
<command interpreter="python">hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -13,7 +13,12 @@ blastn
-task $blast_type
-evalue $evalue_cutoff
-out $output1
-outfmt "$out_format"
##Set the extended list here so if/when we add things, saved workflows are not affected
#if str($out_format)=="ext":
-outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
#else:
-outfmt "$out_format"
#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -66,7 +71,7 @@ $adv_opts.parse_deflines
<param name="evalue_cutoff" type="float" size="15" value="0.001" label="Set expectation value cutoff" />
<param name="out_format" type="select" label="Output format">
<option value="6" selected="True">Tabular (standard 12 columns)</option>
<option value="6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq">Tabular (extended 22 columns)</option>
<option value="ext">Tabular (extended 24 columns)</option>
<option value="5">BLAST XML</option>
<option value="0">Pairwise text</option>
<option value="0 -html">Pairwise HTML</option>
@@ -121,8 +126,6 @@ $adv_opts.parse_deflines
<requirements>
<requirement type="binary">blastn</requirement>
</requirements>
<tests>
</tests>
<help>
.. class:: warningmark
@@ -166,7 +169,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
workflow filtering steps that accept either the 12 or 22 column tabular
workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -182,6 +185,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
23 qlen Query sequence length
24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
+31 -8
View File
@@ -1,4 +1,4 @@
<tool id="ncbi_blastp_wrapper" name="NCBI BLAST+ blastp" version="0.0.8">
<tool id="ncbi_blastp_wrapper" name="NCBI BLAST+ blastp" version="0.0.9">
<description>Search protein database with protein query sequence(s)</description>
<command interpreter="python">hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -13,7 +13,12 @@ blastp
-task $blast_type
-evalue $evalue_cutoff
-out $output1
-outfmt "$out_format"
##Set the extended list here so if/when we add things, saved workflows are not affected
#if str($out_format)=="ext":
-outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
#else:
-outfmt "$out_format"
#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -61,7 +66,7 @@ $adv_opts.parse_deflines
<param name="evalue_cutoff" type="float" size="15" value="0.001" label="Set expectation value cutoff" />
<param name="out_format" type="select" label="Output format">
<option value="6" selected="True">Tabular (standard 12 columns)</option>
<option value="6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq">Tabular (extended 22 columns)</option>
<option value="ext">Tabular (extended 24 columns)</option>
<option value="5">BLAST XML</option>
<option value="0">Pairwise text</option>
<option value="0 -html">Pairwise HTML</option>
@@ -127,6 +132,22 @@ $adv_opts.parse_deflines
<requirement type="binary">blastp</requirement>
</requirements>
<tests>
<test>
<param name="query" value="four_human_proteins.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
<param name="subject" value="rhodopsin_proteins.fasta" ftype="fasta" />
<param name="database" value="" />
<param name="evalue_cutoff" value="1e-8" />
<param name="blast_type" value="blastp" />
<param name="out_format" value="5" />
<param name="adv_opts_selector" value="advanced" />
<param name="filter_query" value="False" />
<param name="matrix" value="BLOSUM62" />
<param name="max_hits" value="0" />
<param name="word_size" value="0" />
<param name="parse_deflines" value="True" />
<output name="output1" file="blastp_four_human_vs_rhodopsin.xml" ftype="blastxml" />
</test>
<test>
<param name="query" value="four_human_proteins.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
@@ -136,7 +157,7 @@ $adv_opts.parse_deflines
<param name="blast_type" value="blastp" />
<param name="out_format" value="6" />
<param name="adv_opts_selector" value="advanced" />
<param name="filter_query" value="True" />
<param name="filter_query" value="False" />
<param name="matrix" value="BLOSUM62" />
<param name="max_hits" value="0" />
<param name="word_size" value="0" />
@@ -150,14 +171,14 @@ $adv_opts.parse_deflines
<param name="database" value="" />
<param name="evalue_cutoff" value="1e-8" />
<param name="blast_type" value="blastp" />
<param name="out_format" value="6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq" />
<param name="out_format" value="ext" />
<param name="adv_opts_selector" value="advanced" />
<param name="filter_query" value="True" />
<param name="filter_query" value="False" />
<param name="matrix" value="BLOSUM62" />
<param name="max_hits" value="0" />
<param name="word_size" value="0" />
<param name="parse_deflines" value="True" />
<output name="output1" file="blastp_four_human_vs_rhodopsin_22c.tabular" ftype="tabular" />
<output name="output1" file="blastp_four_human_vs_rhodopsin_ext.tabular" ftype="tabular" />
</test>
<test>
<param name="query" value="rhodopsin_proteins.fasta" ftype="fasta" />
@@ -213,7 +234,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
workflow filtering steps that accept either the 12 or 22 column tabular
workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -229,6 +250,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
23 qlen Query sequence length
24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
+41 -4
View File
@@ -1,4 +1,4 @@
<tool id="ncbi_blastx_wrapper" name="NCBI BLAST+ blastx" version="0.0.8">
<tool id="ncbi_blastx_wrapper" name="NCBI BLAST+ blastx" version="0.0.9">
<description>Search protein database with translated nucleotide query sequence(s)</description>
<command interpreter="python">hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -12,7 +12,12 @@ blastx
#end if
-evalue $evalue_cutoff
-out $output1
-outfmt "$out_format"
##Set the extended list here so if/when we add things, saved workflows are not affected
#if str($out_format)=="ext":
-outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
#else:
-outfmt "$out_format"
#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -56,7 +61,7 @@ $adv_opts.parse_deflines
<param name="evalue_cutoff" type="float" size="15" value="0.001" label="Set expectation value cutoff" />
<param name="out_format" type="select" label="Output format">
<option value="6" selected="True">Tabular (standard 12 columns)</option>
<option value="6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq">Tabular (extended 22 columns)</option>
<option value="ext">Tabular (extended 24 columns)</option>
<option value="5">BLAST XML</option>
<option value="0">Pairwise text</option>
<option value="0 -html">Pairwise HTML</option>
@@ -122,6 +127,36 @@ $adv_opts.parse_deflines
<requirement type="binary">blastx</requirement>
</requirements>
<tests>
<test>
<param name="query" value="rhodopsin_nucs.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
<param name="subject" value="four_human_proteins.fasta" ftype="fasta" />
<param name="database" value="" />
<param name="evalue_cutoff" value="1e-10" />
<param name="out_format" value="5" />
<param name="adv_opts_selector" value="basic" />
<output name="output1" file="blastx_rhodopsin_vs_four_human.xml" ftype="blastxml" />
</test>
<test>
<param name="query" value="rhodopsin_nucs.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
<param name="subject" value="four_human_proteins.fasta" ftype="fasta" />
<param name="database" value="" />
<param name="evalue_cutoff" value="1e-10" />
<param name="out_format" value="6" />
<param name="adv_opts_selector" value="basic" />
<output name="output1" file="blastx_rhodopsin_vs_four_human.tabular" ftype="tabular" />
</test>
<test>
<param name="query" value="rhodopsin_nucs.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
<param name="subject" value="four_human_proteins.fasta" ftype="fasta" />
<param name="database" value="" />
<param name="evalue_cutoff" value="1e-10" />
<param name="out_format" value="ext" />
<param name="adv_opts_selector" value="basic" />
<output name="output1" file="blastx_rhodopsin_vs_four_human_ext.tabular" ftype="tabular" />
</test>
</tests>
<help>
@@ -165,7 +200,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
workflow filtering steps that accept either the 12 or 22 column tabular
workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -181,6 +216,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
23 qlen Query sequence length
24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
+41 -4
View File
@@ -1,4 +1,4 @@
<tool id="ncbi_tblastn_wrapper" name="NCBI BLAST+ tblastn" version="0.0.8">
<tool id="ncbi_tblastn_wrapper" name="NCBI BLAST+ tblastn" version="0.0.9">
<description>Search translated nucleotide database with protein query sequence(s)</description>
<command interpreter="python">hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -12,7 +12,12 @@ tblastn
#end if
-evalue $evalue_cutoff
-out $output1
-outfmt "$out_format"
##Set the extended list here so if/when we add things, saved workflows are not affected
#if str($out_format)=="ext":
-outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
#else:
-outfmt "$out_format"
#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -56,7 +61,7 @@ $adv_opts.parse_deflines
<param name="evalue_cutoff" type="float" size="15" value="0.001" label="Set expectation value cutoff" />
<param name="out_format" type="select" label="Output format">
<option value="6" selected="True">Tabular (standard 12 columns)</option>
<option value="6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq">Tabular (extended 22 columns)</option>
<option value="ext">Tabular (extended 24 columns)</option>
<option value="5">BLAST XML</option>
<option value="0">Pairwise text</option>
<option value="0 -html">Pairwise HTML</option>
@@ -122,6 +127,36 @@ $adv_opts.parse_deflines
<requirement type="binary">tblastn</requirement>
</requirements>
<tests>
<test>
<param name="query" value="four_human_proteins.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
<param name="subject" value="rhodopsin_nucs.fasta" ftype="fasta" />
<param name="database" value="" />
<param name="evalue_cutoff" value="1e-10" />
<param name="out_format" value="5" />
<param name="adv_opts_selector" value="advanced" />
<param name="filter_query" value="false" />
<param name="matrix" value="BLOSUM80" />
<param name="max_hits" value="0" />
<param name="word_size" value="0" />
<param name="parse_deflines" value="false" />
<output name="output1" file="tblastn_four_human_vs_rhodopsin.xml" ftype="blastxml" />
</test>
<test>
<param name="query" value="four_human_proteins.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
<param name="subject" value="rhodopsin_nucs.fasta" ftype="fasta" />
<param name="database" value="" />
<param name="evalue_cutoff" value="1e-10" />
<param name="out_format" value="ext" />
<param name="adv_opts_selector" value="advanced" />
<param name="filter_query" value="false" />
<param name="matrix" value="BLOSUM80" />
<param name="max_hits" value="0" />
<param name="word_size" value="0" />
<param name="parse_deflines" value="false" />
<output name="output1" file="tblastn_four_human_vs_rhodopsin_ext.tabular" ftype="tabular" />
</test>
<test>
<param name="query" value="four_human_proteins.fasta" ftype="fasta" />
<param name="db_opts_selector" value="file" />
@@ -196,7 +231,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
workflow filtering steps that accept either the 12 or 22 column tabular
workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -212,6 +247,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
23 qlen Query sequence length
24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
+11 -6
View File
@@ -1,4 +1,4 @@
<tool id="ncbi_tblastx_wrapper" name="NCBI BLAST+ tblastx" version="0.0.8">
<tool id="ncbi_tblastx_wrapper" name="NCBI BLAST+ tblastx" version="0.0.9">
<description>Search translated nucleotide database with translated nucleotide query sequence(s)</description>
<command interpreter="python">hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -12,7 +12,12 @@ tblastx
#end if
-evalue $evalue_cutoff
-out $output1
-outfmt "$out_format"
##Set the extended list here so if/when we add things, saved workflows are not affected
#if str($out_format)=="ext":
-outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
#else:
-outfmt "$out_format"
#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -55,7 +60,7 @@ $adv_opts.parse_deflines
<param name="evalue_cutoff" type="float" size="15" value="0.001" label="Set expectation value cutoff" />
<param name="out_format" type="select" label="Output format">
<option value="6" selected="True">Tabular (standard 12 columns)</option>
<option value="6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq">Tabular (extended 22 columns)</option>
<option value="ext">Tabular (extended 24 columns)</option>
<option value="5">BLAST XML</option>
<option value="0">Pairwise text</option>
<option value="0 -html">Pairwise HTML</option>
@@ -119,8 +124,6 @@ $adv_opts.parse_deflines
<requirements>
<requirement type="binary">tblastx</requirement>
</requirements>
<tests>
</tests>
<help>
.. class:: warningmark
@@ -163,7 +166,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
workflow filtering steps that accept either the 12 or 22 column tabular
workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -179,6 +182,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
23 qlen Query sequence length
24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by