Finalizing taxonomy. Removed taxonomy processing from scripts. This does not require sqlite anymore. Script for fetching taxonomy now uses flat files directly and is much faster

This commit is contained in:
Anton Nekrutenko
2008-03-20 19:38:31 +00:00
parent 211673cf3b
commit a262ac1754
14 changed files with 2038 additions and 148 deletions
+4 -7
View File
@@ -1,5 +1,3 @@
PYTHONPATH="../../lib:../../eggs:../../eggs/`../check_python_ucs.py`"
export PYTHONPATH
echo "Getting files from NCBI..."
wget ftp://ftp.ncbi.nih.gov/pub/taxonomy/taxdump.tar.gz
wget ftp://ftp.ncbi.nih.gov/pub/taxonomy/gi_taxid_nucl.dmp.gz
@@ -9,9 +7,8 @@ gunzip -c taxdump.tar.gz | tar xvf -
gunzip gi_taxid_nucl.dmp.gz
gunzip gi_taxid_prot.dmp.gz
cat gi_taxid_nucl.dmp gi_taxid_prot.dmp > gi_taxid_all.dmp
rm gi_taxid_nucl.dmp gi_taxid_prot.dmp
echo "Parsing names.dmg"
cat names.dmp | cut -f 1,2,4 -d "|" | tr -s "\t" "|" | tr "|" "\t" | sed s/\"//g > names.txt
python process_NCBI_taxonomy.py gi_taxid_all.dmp names.txt taxonomy.db
echo "Done!.."
echo "Sorting gi2tax files..."
sort -n -k 1 gi_taxid_all.dmp > gi_taxid_sorted.txt
rm gi_taxid_nucl.dmp gi_taxid_prot.dmp gi_taxid_all.dmp
-50
View File
@@ -1,50 +0,0 @@
"""
process_NCBI_taxonomy.py <gi2tax.txt file> <name.txt> <database_name>
"""
import pkg_resources
pkg_resources.require( 'pysqlite' )
from pysqlite2 import dbapi2 as sqlite
import string, sys, tempfile
def stop_err(msg):
sys.stderr.write(msg)
sys.exit()
try:
gi2tax = open(sys.argv[1], 'r')
names = open(sys.argv[2], 'r')
db_name = sys.argv[3]
except:
stop_err('Check arguments: process_NCBI_taxonomy.py <gi2tax.txt file> <name.txt> <database_name>\n')
try:
con = sqlite.connect(db_name)
cur = con.cursor()
cur.execute('create table gi2tax(gi int unsigned not null, taxId int unsigned not null)')
cur.execute('create table t_names(taxId int unsigned not null, name text not null)')
cur.execute('create table names(taxId int unsigned not null, name text not null)')
for line in gi2tax:
fields = string.split(line.rstrip(), '\t')
cur.execute('insert into gi2tax values(%s, %s)' % ( fields[0], fields[1] ) )
gi2tax.close()
for line in names:
fields = string.split(line.rstrip(), '\t')
cur.execute('insert into t_names values(%s, "%s")' % ( fields[0], fields[1] ) )
names.close()
cur.execute('create index gi_i on gi2tax(gi)')
cur.execute('insert into names select * from t_names group by name')
cur.execute('drop table t_names')
cur.execute('create index name_i on names(name)')
cur.execute('vacuum')
con.commit()
con.close()
except Exception, e:
stop_err("%s\n" % e)