diff --git a/test-data/tblastn_four_human_vs_rhodopsin.xml b/test-data/tblastn_four_human_vs_rhodopsin.xml
new file mode 100644
index 00000000000..9bccf112f2d
--- /dev/null
+++ b/test-data/tblastn_four_human_vs_rhodopsin.xml
@@ -0,0 +1,722 @@
+
+
+
+ tblastn
+ TBLASTN 2.2.25+
+ Stephen F. Altschul, Thomas L. Madden, Alejandro A. Schäffer, Jinghui Zhang, Zheng Zhang, Webb Miller, and David J. Lipman (1997), "Gapped BLAST and PSI-BLAST: a new generation of protein database search programs", Nucleic Acids Res. 25:3389-3402.
+
+ Query_1
+ sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ 406
+
+
+ BLOSUM80
+ 1e-10
+ 10
+ 1
+ F
+
+
+
+
+ 1
+ Query_1
+ sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ 406
+
+
+
+ 0
+ 0
+ 19
+ 127710
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 2
+ Query_1
+ sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ 406
+
+
+
+ 0
+ 0
+ 19
+ 127710
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 3
+ Query_1
+ sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ 406
+
+
+
+ 0
+ 0
+ 19
+ 127710
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 4
+ Query_1
+ sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ 406
+
+
+
+ 0
+ 0
+ 19
+ 127710
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 5
+ Query_1
+ sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ 406
+
+
+
+ 0
+ 0
+ 19
+ 127710
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 6
+ Query_1
+ sp|Q9BS26|ERP44_HUMAN Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ 406
+
+
+
+ 0
+ 0
+ 19
+ 127710
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 7
+ Query_2
+ sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2
+ 1161
+
+
+
+ 0
+ 0
+ 23
+ 370988
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 8
+ Query_2
+ sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2
+ 1161
+
+
+
+ 0
+ 0
+ 23
+ 370988
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 9
+ Query_2
+ sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2
+ 1161
+
+
+
+ 0
+ 0
+ 23
+ 370988
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 10
+ Query_2
+ sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2
+ 1161
+
+
+
+ 0
+ 0
+ 23
+ 370988
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 11
+ Query_2
+ sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2
+ 1161
+
+
+
+ 0
+ 0
+ 23
+ 370988
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 12
+ Query_2
+ sp|Q9NSY1|BMP2K_HUMAN BMP-2-inducible protein kinase OS=Homo sapiens GN=BMP2K PE=1 SV=2
+ 1161
+
+
+
+ 0
+ 0
+ 23
+ 370988
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 13
+ Query_3
+ sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4
+ 1382
+
+
+
+ 0
+ 0
+ 24
+ 441350
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 14
+ Query_3
+ sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4
+ 1382
+
+
+
+ 0
+ 0
+ 24
+ 441350
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 15
+ Query_3
+ sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4
+ 1382
+
+
+
+ 0
+ 0
+ 24
+ 441350
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 16
+ Query_3
+ sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4
+ 1382
+
+
+
+ 0
+ 0
+ 24
+ 441350
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 17
+ Query_3
+ sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4
+ 1382
+
+
+
+ 0
+ 0
+ 24
+ 441350
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 18
+ Query_3
+ sp|P06213|INSR_HUMAN Insulin receptor OS=Homo sapiens GN=INSR PE=1 SV=4
+ 1382
+
+
+
+ 0
+ 0
+ 24
+ 441350
+ 0.071
+ 0.299
+ 0.27
+
+
+ No hits found
+
+
+ 19
+ Query_4
+ sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1
+ 348
+
+
+ 1
+ Subject_1
+ gi|57163782|ref|NM_001009242.1| Felis catus rhodopsin (RHO), mRNA
+ Subject_1
+ 1047
+
+
+ 1
+ 732.392902459534
+ 1689
+ 0
+ 1
+ 348
+ 1
+ 1044
+ 0
+ 1
+ 336
+ 343
+ 0
+ 348
+ MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEASATVSKTETSQVAPA
+ MNGTEGPNFYVPFSNKTGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVFGGFTTTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLVGWSRYIPEGMQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIVIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTLPAFFAKSSSIYNPVIYIMMNKQFRNCMLTTLCCGKNPLGDDEASTTGSKTETSQVAPA
+ MNGTEGPNFYVPFSN TGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMV GGFT+TLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPL GWSRYIPEG+QCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMI+IFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMT+PAFFAKS++IYNPVIYIMMNKQFRNCMLTT+CCGKNPLGDDEAS T SKTETSQVAPA
+
+
+
+
+
+
+ 0
+ 0
+ 18
+ 109230
+ 0.071
+ 0.299
+ 0.27
+
+
+
+
+ 20
+ Query_4
+ sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1
+ 348
+
+
+ 1
+ Subject_2
+ gi|2734705|gb|U59921.1|BBU59921 Bufo bufo rhodopsin mRNA, complete cds
+ Subject_2
+ 1574
+
+
+ 1
+ 646.119739014374
+ 1489
+ 0
+ 1
+ 341
+ 42
+ 1067
+ 0
+ 3
+ 290
+ 320
+ 1
+ 342
+ MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEA-SATVSKTE
+ MNGTEGPNFYIPMSNKTGVVRSPFEYPQYYLAEPWQYSILCAYMFLLILLGFPINFMTLYVTIQHKKLRTPLNYILLNLAFANHFMVLCGFTVTMYSSMNGYFILGATGCYVEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFSENHAVMGVAFTWIMALSCAVPPLLGWSRYIPEGMQCSCGVDYYTLKPEVNNESFVIYMFVVHFTIPLIIIFFCYGRLVCTVKEAAAQQQESATTQKAEKEVTRMVIIMVVFFLICWVPYASVAFFIFSNQGSEFGPIFMTVPAFFAKSSSIYNPVIYIMLNKQFRNCMITTLCCGKNPFGEDDASSAATSKTE
+ MNGTEGPNFY+P SN TGVVRSPFEYPQYYLAEPWQ+S+L AYMFLLI+LGFPINF+TLYVT+QHKKLRTPLNYILLNLA A+ FMVL GFT T+Y+S+ GYF+ G TGC +EGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRF ENHA+MGVAFTW+MAL+CA PPL GWSRYIPEG+QCSCG+DYYTLKPEVNNESFVIYMFVVHFTIP+IIIFFCYG+LV TVKEAAAQQQESATTQKAEKEVTRMVIIMV+ FLICWVPYASVAF+IF+ QGS FGPIFMT+PAFFAKS++IYNPVIYIM+NKQFRNCM+TT+CCGKNP G+D+A SA SKTE
+
+
+
+
+
+
+ 0
+ 0
+ 18
+ 109230
+ 0.071
+ 0.299
+ 0.27
+
+
+
+
+ 21
+ Query_4
+ sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1
+ 348
+
+
+ 1
+ Subject_3
+ gi|283855845|gb|GQ290303.1| Cynopterus brachyotis voucher 20020434 rhodopsin (RHO) gene, exons 1 through 5 and partial cds
+ Subject_3
+ 4301
+
+
+ 1
+ 151.343146656381
+ 342
+ 1.39566684546685e-72
+ 239
+ 312
+ 3147
+ 3368
+ 0
+ 3
+ 69
+ 73
+ 0
+ 74
+ ESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQ
+ ESATTQKAEKEVTRMVIIMVIAFLICWLPYAGVAFYIFTHQGSNFGPIFMTLPAFFAKSSSIYNPVIYIMMNKQ
+ ESATTQKAEKEVTRMVIIMVIAFLICW+PYA VAFYIFTHQGSNFGPIFMT+PAFFAKS++IYNPVIYIMMNKQ
+
+
+ 2
+ 126.323929257285
+ 284
+ 1.39566684546685e-72
+ 177
+ 235
+ 2855
+ 3031
+ 0
+ 2
+ 54
+ 57
+ 0
+ 59
+ RYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAA
+ RYIPEGMQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIVIFFCYGQLVFTVKEVRS
+ RYIPEG+QCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMI+IFFCYGQLVFTVKE +
+
+
+ 3
+ 229.420359574251
+ 523
+ 9.84654801241353e-65
+ 11
+ 121
+ 1
+ 333
+ 0
+ 1
+ 107
+ 109
+ 0
+ 111
+ VPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGG
+ VPFSNKTGVVRSPFEHPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVFGGFTTTLYTSLHGYFVFGPTGCNLEGFFATLGG
+ VPFSN TGVVRSPFE+PQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMV GGFT+TLYTSLHGYFVFGPTGCNLEGFFATLGG
+
+
+ 4
+ 122.873002719478
+ 276
+ 1.40732096096596e-32
+ 119
+ 177
+ 1404
+ 1580
+ 0
+ 3
+ 55
+ 56
+ 0
+ 59
+ LGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSR
+ LAGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGLALTWVMALACAAPPLVGWSR
+ L GEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMG+A TWVMALACAAPPL GWSR
+
+
+ 5
+ 57.7367643183824
+ 125
+ 5.60065526485586e-13
+ 312
+ 337
+ 4222
+ 4299
+ 0
+ 1
+ 23
+ 24
+ 0
+ 26
+ QFRNCMLTTICCGKNPLGDDEASATV
+ QFRNCMLTTLCCGKNPLGDDEASTTA
+ QFRNCMLTT+CCGKNPLGDDEAS T
+
+
+
+
+
+
+ 0
+ 0
+ 18
+ 109230
+ 0.071
+ 0.299
+ 0.27
+
+
+
+
+ 22
+ Query_4
+ sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1
+ 348
+
+
+ 1
+ Subject_4
+ gi|283855822|gb|GQ290312.1| Myotis ricketti voucher GQX10 rhodopsin (RHO) mRNA, partial cds
+ Subject_4
+ 983
+
+
+ 1
+ 658.197981896696
+ 1517
+ 0
+ 11
+ 336
+ 1
+ 978
+ 0
+ 1
+ 310
+ 322
+ 0
+ 326
+ VPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEASAT
+ VPFSNKTGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVANLFMVFGGFTTTLYTSMHGYFVFGATGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGLAFTWVMALACAAPPLAGWSRYIPEGMQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIVIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVVAFLICWLPYASVAFYIFTHQGSNFGPVFMTIPAFFAKSSSIYNPVIYIMMNKQFRNCMLTTLCCGKNPLGDDEASTT
+ VPFSN TGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVA+LFMV GGFT+TLYTS+HGYFVFG TGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMG+AFTWVMALACAAPPLAGWSRYIPEG+QCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMI+IFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMV+AFLICW+PYASVAFYIFTHQGSNFGP+FMTIPAFFAKS++IYNPVIYIMMNKQFRNCMLTT+CCGKNPLGDDEAS T
+
+
+
+
+
+
+ 0
+ 0
+ 18
+ 109230
+ 0.071
+ 0.299
+ 0.27
+
+
+
+
+ 23
+ Query_4
+ sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1
+ 348
+
+
+ 1
+ Subject_5
+ gi|18148870|dbj|AB062417.1| Synthetic construct Bos taurus gene for rhodopsin, complete cds
+ Subject_5
+ 1047
+
+
+ 1
+ 711.255977415469
+ 1640
+ 0
+ 1
+ 348
+ 1
+ 1044
+ 0
+ 1
+ 325
+ 337
+ 0
+ 348
+ MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPLGDDEASATVSKTETSQVAPA
+ MNGTEGPNFYVPFSNKTGVVRSPFEAPQYYLAEPWQFSMLAAYMFLLIMLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVFGGFTTTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLVGWSRYIPEGMQCSCGIDYYTPHEETNNESFVIYMFVVHFIIPLIVIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWLPYAGVAFYIFTHQGSDFGPIFMTIPAFFAKTSAVYNPVIYIMMNKQFRNCMVTTLCCGKNPLGDDEASTTVSKTETSQVAPA
+ MNGTEGPNFYVPFSN TGVVRSPFE PQYYLAEPWQFSMLAAYMFLLI+LGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMV GGFT+TLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPL GWSRYIPEG+QCSCGIDYYT E NNESFVIYMFVVHF IP+I+IFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICW+PYA VAFYIFTHQGS+FGPIFMTIPAFFAK++A+YNPVIYIMMNKQFRNCM+TT+CCGKNPLGDDEAS TVSKTETSQVAPA
+
+
+
+
+
+
+ 0
+ 0
+ 18
+ 109230
+ 0.071
+ 0.299
+ 0.27
+
+
+
+
+ 24
+ Query_4
+ sp|P08100|OPSD_HUMAN Rhodopsin OS=Homo sapiens GN=RHO PE=1 SV=1
+ 348
+
+
+ 1
+ Subject_6
+ gi|12583664|dbj|AB043817.1| Conger myriaster conf gene for fresh water form rod opsin, complete cds
+ Subject_6
+ 1344
+
+
+ 1
+ 626.708277239213
+ 1444
+ 0
+ 1
+ 341
+ 23
+ 1048
+ 0
+ 2
+ 281
+ 311
+ 1
+ 342
+ MNGTEGPNFYVPFSNATGVVRSPFEYPQYYLAEPWQFSMLAAYMFLLIVLGFPINFLTLYVTVQHKKLRTPLNYILLNLAVADLFMVLGGFTSTLYTSLHGYFVFGPTGCNLEGFFATLGGEIALWSLVVLAIERYVVVCKPMSNFRFGENHAIMGVAFTWVMALACAAPPLAGWSRYIPEGLQCSCGIDYYTLKPEVNNESFVIYMFVVHFTIPMIIIFFCYGQLVFTVKEAAAQQQESATTQKAEKEVTRMVIIMVIAFLICWVPYASVAFYIFTHQGSNFGPIFMTIPAFFAKSAAIYNPVIYIMMNKQFRNCMLTTICCGKNPL-GDDEASATVSKTE
+ MNGTEGPNFYIPMSNATGVVRSPFEYPQYYLAEPWAFSALSAYMFFLIIAGFPINFLTLYVTIEHKKLRTPLNYILLNLAVADLFMVFGGFTTTMYTSMHGYFVFGPTGCNIEGFFATLGGEIALWCLVVLAIERWMVVCKPVTNFRFGESHAIMGVMVTWTMALACALPPLFGWSRYIPEGLQCSCGIDYYTRAPGINNESFVIYMFTCHFSIPLAVISFCYGRLVCTVKEAAAQQQESETTQRAEREVTRMVVIMVISFLVCWVPYASVAWYIFTHQGSTFGPIFMTIPSFFAKSSALYNPMIYICMNKQFRHCMITTLCCGKNPFEEEDGASATSSKTE
+ MNGTEGPNFY+P SNATGVVRSPFEYPQYYLAEPW FS L+AYMF LI+ GFPINFLTLYVT++HKKLRTPLNYILLNLAVADLFMV GGFT+T+YTS+HGYFVFGPTGCN+EGFFATLGGEIALW LVVLAIER++VVCKP++NFRFGE HAIMGV TW MALACA PPL GWSRYIPEGLQCSCGIDYYT P +NNESFVIYMF HF+IP+ +I FCYG+LV TVKEAAAQQQES TTQ+AE+EVTRMV+IMVI+FL+CWVPYASVA YIFTHQGS FGPIFMTIP+FFAKS+A+YNP+IYI MNKQFR CM+TT+CCGKNP +D ASAT SKTE
+
+
+
+
+
+
+ 0
+ 0
+ 18
+ 109230
+ 0.071
+ 0.299
+ 0.27
+
+
+
+
+
diff --git a/tools/ncbi_blast_plus/blastxml_to_tabular.py b/tools/ncbi_blast_plus/blastxml_to_tabular.py
index 880efe74467..41d8e582c0a 100644
--- a/tools/ncbi_blast_plus/blastxml_to_tabular.py
+++ b/tools/ncbi_blast_plus/blastxml_to_tabular.py
@@ -2,11 +2,12 @@
"""Convert a BLAST XML file to 12 column tabular output
Takes three command line options, input BLAST XML filename, output tabular
-BLAST filename, output format (std for standard 12 columns, or x22 for the
-extended 22 columns offered in the BLAST+ wrappers).
+BLAST filename, output format (std for standard 12 columns, or ext for the
+extended 24 columns offered in the BLAST+ wrappers).
The 12 colums output are 'qseqid sseqid pident length mismatch gapopen qstart
-qend sstart send evalue bitscore' which mean:
+qend sstart send evalue bitscore' or 'std' at the BLAST+ command line, which
+mean:
====== ========= ============================================
Column NCBI name Description
@@ -25,7 +26,7 @@ Column NCBI name Description
12 bitscore Bit score
====== ========= ============================================
-The additional columns are:
+The additional columns offered in the Galaxy BLAST+ wrappers are:
====== ============= ===========================================
Column NCBI name Description
@@ -40,20 +41,26 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
+ 23 qlen Query sequence length
+ 24 slen Subject sequence length
====== ============= ===========================================
Most of these fields are given explicitly in the XML file, others some like
the percentage identity and the number of gap openings must be calculated.
+Be aware that the sequence in the extended tabular output or XML direct from
+BLAST+ may or may not use XXXX masking on regions of low complexity. This
+can throw the off the calculation of percentage identity and gap openings.
+[In fact, both BLAST 2.2.24+ and 2.2.25+ have a sutle bug in this regard,
+with these numbers changing depending on whether or not the low complexity
+filter is used.]
+
This script attempts to produce idential output to what BLAST+ would have done.
However, check this with "diff -b ..." since BLAST+ sometimes includes an extra
space character (probably a bug).
-
-Beware that if using the extended output, the XML file contains the original
-aligned sequences, but the tabular output direct from BLAST+ may use XXXX
-masking on regions of low complexity (columns 21 and 22).
"""
import sys
+import re
assert sys.version_info[:2] >= ( 2, 4 )
if sys.version_info[:2] >= ( 2, 5 ):
@@ -69,14 +76,16 @@ def stop_err( msg ):
try:
in_file, out_file, out_fmt = sys.argv[1:]
except:
- stop_err("Expect 3 arguments: input BLAST XML file, output tabular file, out format (std or x22)")
+ stop_err("Expect 3 arguments: input BLAST XML file, output tabular file, out format (std or ext)")
if out_fmt == "std":
extended = False
elif out_fmt == "x22":
+ stop_err("Format argument x22 has been replaced with ext (extended 24 columns)")
+elif out_fmt == "ext":
extended = True
else:
- stop_err("Format argument should be std (12 column) or x22 (extended 22 column)")
+ stop_err("Format argument should be std (12 column) or ext (extended 24 columns)")
# get an iterable
@@ -92,6 +101,18 @@ try:
except:
stop_err( "Invalid data format." )
+
+re_default_query_id = re.compile("^Query_\d+$")
+assert re_default_query_id.match("Query_101")
+assert not re_default_query_id.match("Query_101a")
+assert not re_default_query_id.match("MyQuery_101")
+re_default_subject_id = re.compile("^Subject_\d+$")
+assert re_default_subject_id.match("Subject_1")
+assert not re_default_subject_id.match("Subject_")
+assert not re_default_subject_id.match("Subject_12a")
+assert not re_default_subject_id.match("TheSubject_1")
+
+
outfile = open(out_file, 'w')
blast_program = None
for event, elem in context:
@@ -99,10 +120,42 @@ for event, elem in context:
blast_program = elem.text
# for every tag
if event == "end" and elem.tag == "Iteration":
- qseqid = elem.findtext("Iteration_query-def").split(None,1)[0]
+ #Expecting either this, from BLAST 2.2.25+ using FASTA vs FASTA
+ # sp|Q9BS26|ERP44_HUMAN
+ # Endoplasmic reticulum resident protein 44 OS=Homo sapiens GN=ERP44 PE=1 SV=1
+ # 406
+ #
+ #
+ #Or, from BLAST 2.2.24+ run online
+ # Query_1
+ # Sample
+ # 516
+ # ...
+ qseqid = elem.findtext("Iteration_query-ID")
+ if re_default_query_id.match(qseqid):
+ #Place holder ID, take the first word of the query definition
+ qseqid = elem.findtext("Iteration_query-def").split(None,1)[0]
+ qlen = int(elem.findtext("Iteration_query-len"))
+
# for every within
for hit in elem.findall("Iteration_hits/Hit/"):
+ #Expecting either this,
+ # gi|3024260|sp|P56514.1|OPSD_BUFBU
+ # RecName: Full=Rhodopsin
+ # P56514
+ #or,
+ # Subject_1
+ # gi|57163783|ref|NP_001009242.1| rhodopsin [Felis catus]
+ # Subject_1
+ #
+ #apparently depending on the parse_deflines switch
sseqid = hit.findtext("Hit_id").split(None,1)[0]
+ hit_def = sseqid + " " + hit.findtext("Hit_def")
+ if re_default_subject_id.match(sseqid) \
+ and sseqid == hit.findtext("Hit_accession"):
+ #Place holder ID, take the first word of the subject definition
+ hit_def = hit.findtext("Hit_def")
+ sseqid = hit_def.split(None,1)[0]
# for every within
for hsp in hit.findall("Hit_hsps/Hsp"):
nident = hsp.findtext("Hsp_identity")
@@ -164,7 +217,6 @@ for event, elem in context:
]
if extended:
- hit_def = sseqid + " " + hit.findtext("Hit_def")
sallseqid = ";".join(name.split(None,1)[0] for name in hit_def.split(">"))
#print hit_def, "-->", sallseqid
positive = hsp.findtext("Hsp_positive")
@@ -175,6 +227,7 @@ for event, elem in context:
#Probably a bug in BLASTP that they use 0 or 1 depending on format
if qframe == "0": qframe = "1"
if sframe == "0": sframe = "1"
+ slen = int(hit.findtext("Hit_len"))
values.extend([sallseqid,
hsp.findtext("Hsp_score"), #score,
nident,
@@ -185,7 +238,10 @@ for event, elem in context:
sframe,
#NOTE - for blastp, XML shows original seq, tabular uses XXX masking
q_seq,
- h_seq])
+ h_seq,
+ str(qlen),
+ str(slen),
+ ])
#print "\t".join(values)
outfile.write("\t".join(values) + "\n")
# prevents ElementTree from growing large datastructure
diff --git a/tools/ncbi_blast_plus/blastxml_to_tabular.xml b/tools/ncbi_blast_plus/blastxml_to_tabular.xml
index 41ac37f6cdb..66bba2c06fc 100644
--- a/tools/ncbi_blast_plus/blastxml_to_tabular.xml
+++ b/tools/ncbi_blast_plus/blastxml_to_tabular.xml
@@ -1,4 +1,4 @@
-
+
Convert BLAST XML output to tabular
blastxml_to_tabular.py $blastxml_file $tabular_file $out_format
@@ -7,7 +7,7 @@
-
+
@@ -16,12 +16,36 @@
+
+
+
+
+
+
+
+
+
+
+
+
-
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -36,9 +60,9 @@
-
+
-
+
@@ -90,11 +114,14 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
+ 23 qlen Query sequence length
+ 24 slen Subject sequence length
====== ============= ===========================================
-Beware that the XML file (and thus the conversion) contains the original
-aligned sequences, but the tabular output direct from BLAST+ may use XXXX
-masking on regions of low complexity (columns 21 and 22).
+Beware that the XML file (and thus the conversion) and the tabular output
+direct from BLAST+ may differ in the presence of XXXX masking on regions
+low complexity (columns 21 and 22), and thus also calculated figures like
+the percentage idenity (column 3).
diff --git a/tools/ncbi_blast_plus/ncbi_blastn_wrapper.xml b/tools/ncbi_blast_plus/ncbi_blastn_wrapper.xml
index 2ab7fe55121..7d6880991be 100644
--- a/tools/ncbi_blast_plus/ncbi_blastn_wrapper.xml
+++ b/tools/ncbi_blast_plus/ncbi_blastn_wrapper.xml
@@ -1,4 +1,4 @@
-
+
Search nucleotide database with nucleotide query sequence(s)
hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -13,7 +13,12 @@ blastn
-task $blast_type
-evalue $evalue_cutoff
-out $output1
--outfmt "$out_format"
+##Set the extended list here so if/when we add things, saved workflows are not affected
+#if str($out_format)=="ext":
+ -outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
+#else:
+ -outfmt "$out_format"
+#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -66,7 +71,7 @@ $adv_opts.parse_deflines
-
+
@@ -121,8 +126,6 @@ $adv_opts.parse_deflines
blastn
-
-
.. class:: warningmark
@@ -166,7 +169,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
-workflow filtering steps that accept either the 12 or 22 column tabular
+workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -182,6 +185,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
+ 23 qlen Query sequence length
+ 24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
diff --git a/tools/ncbi_blast_plus/ncbi_blastp_wrapper.xml b/tools/ncbi_blast_plus/ncbi_blastp_wrapper.xml
index ed6606bdf00..70266c0522f 100644
--- a/tools/ncbi_blast_plus/ncbi_blastp_wrapper.xml
+++ b/tools/ncbi_blast_plus/ncbi_blastp_wrapper.xml
@@ -1,4 +1,4 @@
-
+
Search protein database with protein query sequence(s)
hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -13,7 +13,12 @@ blastp
-task $blast_type
-evalue $evalue_cutoff
-out $output1
--outfmt "$out_format"
+##Set the extended list here so if/when we add things, saved workflows are not affected
+#if str($out_format)=="ext":
+ -outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
+#else:
+ -outfmt "$out_format"
+#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -61,7 +66,7 @@ $adv_opts.parse_deflines
-
+
@@ -127,6 +132,22 @@ $adv_opts.parse_deflines
blastp
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -136,7 +157,7 @@ $adv_opts.parse_deflines
-
+
@@ -150,14 +171,14 @@ $adv_opts.parse_deflines
-
+
-
+
-
+
@@ -213,7 +234,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
-workflow filtering steps that accept either the 12 or 22 column tabular
+workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -229,6 +250,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
+ 23 qlen Query sequence length
+ 24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
diff --git a/tools/ncbi_blast_plus/ncbi_blastx_wrapper.xml b/tools/ncbi_blast_plus/ncbi_blastx_wrapper.xml
index 1240efba08f..e664b7f0801 100644
--- a/tools/ncbi_blast_plus/ncbi_blastx_wrapper.xml
+++ b/tools/ncbi_blast_plus/ncbi_blastx_wrapper.xml
@@ -1,4 +1,4 @@
-
+
Search protein database with translated nucleotide query sequence(s)
hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -12,7 +12,12 @@ blastx
#end if
-evalue $evalue_cutoff
-out $output1
--outfmt "$out_format"
+##Set the extended list here so if/when we add things, saved workflows are not affected
+#if str($out_format)=="ext":
+ -outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
+#else:
+ -outfmt "$out_format"
+#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -56,7 +61,7 @@ $adv_opts.parse_deflines
-
+
@@ -122,6 +127,36 @@ $adv_opts.parse_deflines
blastx
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -165,7 +200,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
-workflow filtering steps that accept either the 12 or 22 column tabular
+workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -181,6 +216,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
+ 23 qlen Query sequence length
+ 24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
diff --git a/tools/ncbi_blast_plus/ncbi_tblastn_wrapper.xml b/tools/ncbi_blast_plus/ncbi_tblastn_wrapper.xml
index 2b66816851b..cb0c39bc4f0 100644
--- a/tools/ncbi_blast_plus/ncbi_tblastn_wrapper.xml
+++ b/tools/ncbi_blast_plus/ncbi_tblastn_wrapper.xml
@@ -1,4 +1,4 @@
-
+
Search translated nucleotide database with protein query sequence(s)
hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -12,7 +12,12 @@ tblastn
#end if
-evalue $evalue_cutoff
-out $output1
--outfmt "$out_format"
+##Set the extended list here so if/when we add things, saved workflows are not affected
+#if str($out_format)=="ext":
+ -outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
+#else:
+ -outfmt "$out_format"
+#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -56,7 +61,7 @@ $adv_opts.parse_deflines
-
+
@@ -122,6 +127,36 @@ $adv_opts.parse_deflines
tblastn
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -196,7 +231,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
-workflow filtering steps that accept either the 12 or 22 column tabular
+workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -212,6 +247,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
+ 23 qlen Query sequence length
+ 24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by
diff --git a/tools/ncbi_blast_plus/ncbi_tblastx_wrapper.xml b/tools/ncbi_blast_plus/ncbi_tblastx_wrapper.xml
index 495c8b161a0..7ce09664027 100644
--- a/tools/ncbi_blast_plus/ncbi_tblastx_wrapper.xml
+++ b/tools/ncbi_blast_plus/ncbi_tblastx_wrapper.xml
@@ -1,4 +1,4 @@
-
+
Search translated nucleotide database with translated nucleotide query sequence(s)
hide_stderr.py
## The command is a Cheetah template which allows some Python based syntax.
@@ -12,7 +12,12 @@ tblastx
#end if
-evalue $evalue_cutoff
-out $output1
--outfmt "$out_format"
+##Set the extended list here so if/when we add things, saved workflows are not affected
+#if str($out_format)=="ext":
+ -outfmt "6 std sallseqid score nident positive gaps ppos qframe sframe qseq sseq qlen slen"
+#else:
+ -outfmt "$out_format"
+#end if
-num_threads 8
#if $adv_opts.adv_opts_selector=="advanced":
$adv_opts.filter_query
@@ -55,7 +60,7 @@ $adv_opts.parse_deflines
-
+
@@ -119,8 +124,6 @@ $adv_opts.parse_deflines
tblastx
-
-
.. class:: warningmark
@@ -163,7 +166,7 @@ The BLAST+ tools can optionally output additional columns of information,
but this takes longer to calculate. Most (but not all) of these columns are
included by selecting the extended tabular output. The extra columns are
included *after* the standard 12 columns. This is so that you can write
-workflow filtering steps that accept either the 12 or 22 column tabular
+workflow filtering steps that accept either the 12 or 24 column tabular
BLAST output.
====== ============= ===========================================
@@ -179,6 +182,8 @@ Column NCBI name Description
20 sframe Subject frame
21 qseq Aligned part of query sequence
22 sseq Aligned part of subject sequence
+ 23 qlen Query sequence length
+ 24 slen Subject sequence length
====== ============= ===========================================
The third option is BLAST XML output, which is designed to be parsed by