From 9dca050a79d899cf6c2b6926e952ff5d7d8b0f21 Mon Sep 17 00:00:00 2001 From: Anton Nekrutenko Date: Mon, 3 Mar 2008 19:40:36 +0000 Subject: [PATCH] Fisrt pass on taxonomic rank fetcher --- test-data/taxonomyGI.dat | 5 +++ tools/taxonomy/gi2taxonomy.xml | 74 ++++++++++++++++++++++++++++++++++ tools/taxonomy/tax.py | 64 +++++++++++++++++++++++++++++ 3 files changed, 143 insertions(+) create mode 100644 test-data/taxonomyGI.dat create mode 100644 tools/taxonomy/gi2taxonomy.xml create mode 100644 tools/taxonomy/tax.py diff --git a/test-data/taxonomyGI.dat b/test-data/taxonomyGI.dat new file mode 100644 index 00000000000..d246b34882b --- /dev/null +++ b/test-data/taxonomyGI.dat @@ -0,0 +1,5 @@ +33001686 9443 root Eukaryota Metazoa n n Chordata Craniata Gnathostomata Mammalia n Euarchontoglires Primates n n n n n n n n n n +23236241 9604 root Eukaryota Metazoa n n Chordata Craniata Gnathostomata Mammalia n Euarchontoglires Primates Haplorrhini Hominoidea Hominidae n n n n n n n +12583 9606 root Eukaryota Metazoa n n Chordata Craniata Gnathostomata Mammalia n Euarchontoglires Primates Haplorrhini Hominoidea Hominidae n n n Homo n Homo sapiens n +410771 40674 root Eukaryota Metazoa n n Chordata Craniata Gnathostomata Mammalia n n n n n n n n n n n n n +2286205 63221 root Eukaryota Metazoa n n Chordata Craniata Gnathostomata Mammalia n Euarchontoglires Primates Haplorrhini Hominoidea Hominidae n n n Homo n Homo sapiens Homo sapiens neanderthalensis diff --git a/tools/taxonomy/gi2taxonomy.xml b/tools/taxonomy/gi2taxonomy.xml new file mode 100644 index 00000000000..049ad8db111 --- /dev/null +++ b/tools/taxonomy/gi2taxonomy.xml @@ -0,0 +1,74 @@ + + from a list of GIs + tax.py $input $out_file1 -c $giField + + + + + + + + + + + + + + + + + +.. class:: infomark + +Use *Filter and Sort->Filter* to restrict output of this tool to desired taxonomic ranks. You can also use *Text Manipulation->Cut* to remove unwanted columns from the output. + +------ + +**What it does** + +Fetches taxonomic information for a list of GI numbers (sequiences identifiers used by teh National Center for Biotechnology Information http://www.ncbi.nlm.nih.gov). + +------- + +**Example** + +Suppose you have BLAST output that looks like this:: + + +-----------------------+----------+----------+-----------------+------------+------+--------+ + | queryId | targetGI | identity | alignmentLength | mismatches | gaps | score | + +-----------------------+----------+----------+-----------------+------------+------+--------+ + | 1L_EYKX4VC01BXWX1_265 | 1430919 | 90.09 | 212 | 15 | 6 | 252.00 | + | 1L_EYKX4VC01BXWX1_265 | 516142 | 90.09 | 212 | 15 | 6 | 252.00 | + +-----------------------+----------+----------+-----------------+------------+------+--------+ + +and you want to obtain full taxonomic representation for GIs listed in *targetGI* column. The output of this tool will add 21 columns to the input data shown above. The 21 columns will correspond to:: + + 'root' :1, + 'superkingdom':2, + 'kingdom' :3, + 'subkingdom' :4, + 'superphylum' :5, + 'phylum' :6, + 'subphylum' :7, + 'superclass' :8, + 'class' :9, + 'subclass' :10, + 'superorder' :11, + 'order' :12, + 'suborder' :13, + 'superfamily' :14, + 'family' :15, + 'subfamily' :16, + 'tribe' :17, + 'subtribe' :18, + 'genus' :19, + 'subgenus' :20, + 'species' :21, + 'subspecies' :22 + + + + + + + diff --git a/tools/taxonomy/tax.py b/tools/taxonomy/tax.py new file mode 100644 index 00000000000..581a382bcfa --- /dev/null +++ b/tools/taxonomy/tax.py @@ -0,0 +1,64 @@ +#!/usr/bin/env python + +""" +Identify full taxonomic standing for sequences identified by gi number + +usage: %prog gi_list_file out_file + -c, --cols=N: Number of column containing gi within the gi_list file + + gi_list_file - user's input containing gi's in the column specified by option -c + + taxonomy_db - database containing collapsed NCBI taxonomy generated by prepareTaxonomy.sh script + distributed with Galaxy. See prepareTaxonomy.readme for information on how to generate + necessary files + +""" + +import pkg_resources +pkg_resources.require( 'bx-python' ) +pkg_resources.require( 'pysqlite' ) +import traceback +import fileinput +from pysqlite2 import dbapi2 as sqlite +from warnings import warn +from bx.cookbook import doc_optparse +import string, sys + +TAXONOMY = '/Users/anton/galaxy/static/taxonomy/taxonomy.db' + +def main(): + + options, args = doc_optparse.parse( __doc__ ) + + if len(args) < 2: + sys.stderr.write('Not enough arguments\n') + sys.exit(0) + + try: + gi_fname, out_fname = args + giCol = int( options.cols ) - 1 + except: + doc_optparse.exception + + con = sqlite.connect(TAXONOMY) + cur = con.cursor() + fg = open(gi_fname, 'r') + of = open( out_fname, "w" ) + + try: + for line in fg: + field = string.split(line.rstrip(), '\t') + sqlTemplate = string.Template('select gi2tax.gi, tax.* from gi2tax left join tax on gi2tax.taxId = tax.taxId where gi2tax.gi = $gi') + sql = sqlTemplate.substitute(gi = int(field[giCol])) + cur.execute(sql) + + for item in cur.fetchall(): + ranks = string.split(item[2], ",") + print >> of, str(item[0]) + "\t" + str(item[1]) + "\t" + "\t".join(ranks) + + finally: + fg.close() + of.close() + +if __name__ == "__main__": + main()