From 286c8f2af607e8f9f35e93fb4cb0a6c09ae47602 Mon Sep 17 00:00:00 2001 From: Kanwei Li Date: Tue, 12 Jul 2011 18:14:56 -0400 Subject: [PATCH] Improve BLAST parser tool error handling, and improve elementtree import as well. --- tools/metag_tools/megablast_xml_parser.py | 14 +++++++------- tools/metag_tools/megablast_xml_parser.xml | 17 +++++++---------- tools/ncbi_blast_plus/blastxml_to_tabular.py | 15 ++++++++------- 3 files changed, 22 insertions(+), 24 deletions(-) diff --git a/tools/metag_tools/megablast_xml_parser.py b/tools/metag_tools/megablast_xml_parser.py index 89e8fea44d1..c900115d3aa 100644 --- a/tools/metag_tools/megablast_xml_parser.py +++ b/tools/metag_tools/megablast_xml_parser.py @@ -2,12 +2,12 @@ import sys, os, re -assert sys.version_info[:2] >= ( 2, 4 ) - if sys.version_info[:2] >= ( 2, 5 ): - import xml.etree.cElementTree as cElementTree + import xml.etree.cElementTree as ElementTree else: - import cElementTree + from galaxy import eggs + import pkg_resources; pkg_resources.require( "elementtree" ) + from elementtree import ElementTree def stop_err( msg ): sys.stderr.write( "%s\n" % msg ) @@ -34,7 +34,7 @@ def __main__(): # get an iterable try: - context = cElementTree.iterparse( source, events=( "start", "end" ) ) + context = ElementTree.iterparse( source, events=( "start", "end" ) ) except: stop_err( "Invalid data format." ) # turn it into an iterator @@ -46,7 +46,7 @@ def __main__(): stop_err( "Invalid data format." ) outfile = open( sys.argv[2], 'w' ) - try: + try: for event, elem in context: # for every tag if event == "end" and elem.tag == "Iteration": @@ -71,7 +71,7 @@ def __main__(): elem.clear() except: outfile.close() - stop_err( "The input data contains tags that are not recognizable by the tool." ) + stop_err( "The input data is malformed, or there is more than one dataset in the input file. Error: %s" % sys.exc_info()[1] ) outfile.close() diff --git a/tools/metag_tools/megablast_xml_parser.xml b/tools/metag_tools/megablast_xml_parser.xml index 2729af91379..51218c2ec8e 100644 --- a/tools/metag_tools/megablast_xml_parser.xml +++ b/tools/metag_tools/megablast_xml_parser.xml @@ -2,26 +2,23 @@ megablast_xml_parser.py $input1 $output1 - + - + - - cElementTree - - - - - + + + + **What it does** This tool processes the XML output of any NCBI blast tool (if you run your own blast jobs, the XML output can be generated with **-m 7** option). - + ----- **Output fields** diff --git a/tools/ncbi_blast_plus/blastxml_to_tabular.py b/tools/ncbi_blast_plus/blastxml_to_tabular.py index 0f258cf6853..7098fec2d2b 100644 --- a/tools/ncbi_blast_plus/blastxml_to_tabular.py +++ b/tools/ncbi_blast_plus/blastxml_to_tabular.py @@ -5,7 +5,7 @@ Takes three command line options, input BLAST XML filename, output tabular BLAST filename, output format (std for standard 12 columns, or ext for the extended 24 columns offered in the BLAST+ wrappers). -The 12 colums output are 'qseqid sseqid pident length mismatch gapopen qstart +The 12 columns output are 'qseqid sseqid pident length mismatch gapopen qstart qend sstart send evalue bitscore' or 'std' at the BLAST+ command line, which mean: @@ -51,22 +51,23 @@ the percentage identity and the number of gap openings must be calculated. Be aware that the sequence in the extended tabular output or XML direct from BLAST+ may or may not use XXXX masking on regions of low complexity. This can throw the off the calculation of percentage identity and gap openings. -[In fact, both BLAST 2.2.24+ and 2.2.25+ have a sutle bug in this regard, +[In fact, both BLAST 2.2.24+ and 2.2.25+ have a subtle bug in this regard, with these numbers changing depending on whether or not the low complexity filter is used.] -This script attempts to produce idential output to what BLAST+ would have done. +This script attempts to produce identical output to what BLAST+ would have done. However, check this with "diff -b ..." since BLAST+ sometimes includes an extra space character (probably a bug). """ import sys import re -assert sys.version_info[:2] >= ( 2, 4 ) if sys.version_info[:2] >= ( 2, 5 ): - import xml.etree.cElementTree as cElementTree + import xml.etree.cElementTree as ElementTree else: - import cElementTree + from galaxy import eggs + import pkg_resources; pkg_resources.require( "elementtree" ) + from elementtree import ElementTree def stop_err( msg ): sys.stderr.write("%s\n" % msg) @@ -90,7 +91,7 @@ else: # get an iterable try: - context = cElementTree.iterparse(in_file, events=("start", "end")) + context = ElementTree.iterparse(in_file, events=("start", "end")) except: stop_err("Invalid data format.") # turn it into an iterator