From 185b4353c21b5fa198728820fa6262371688eccf Mon Sep 17 00:00:00 2001 From: Greg Von Kuster Date: Thu, 1 May 2008 17:44:56 +0000 Subject: [PATCH] Deleted findcluster_mysql tool code. --- tool_conf.xml.sample | 5 - tools/patmat/findcluster_mysql.py | 301 -------------------- tools/patmat/findcluster_mysql.xml | 63 ----- tools/patmat/findcluster_mysql_subs.py | 368 ------------------------- 4 files changed, 737 deletions(-) delete mode 100755 tools/patmat/findcluster_mysql.py delete mode 100644 tools/patmat/findcluster_mysql.xml delete mode 100644 tools/patmat/findcluster_mysql_subs.py diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index 3db546f2bb6..9a19d986e3e 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -68,11 +68,6 @@ -
diff --git a/tools/patmat/findcluster_mysql.py b/tools/patmat/findcluster_mysql.py deleted file mode 100755 index 72f5c917fd4..00000000000 --- a/tools/patmat/findcluster_mysql.py +++ /dev/null @@ -1,301 +0,0 @@ -#!/usr/bin/env python -import urllib, sys, os, sets, findcluster_mysql_subs -from time import time, localtime, strftime -import pkg_resources -pkg_resources.require( "sqlalchemy>=0.2" ) -from sqlalchemy import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -if __name__ == '__main__': - nargv = len(sys.argv) - if nargv < 2 : - findcluster_mysql_subs.usage() - sys.exit() - - #---------------------------------------------------------------------------------------- - #--------------------------read params and files----------------------------------------- - genome_name = '' - chrom_name = '' - chroms = dict() - blocks = [] - range = 0 - blkstring = '' - chrs = [] - ptnstring = '' - ptns = [] - patterns = [] - wsize = 3500 - shsize = 32 - n = 1 - info_file = 'None' - while n < nargv : - #get the genome parameter - if sys.argv[n] == "-g" : genome_name = sys.argv[n+1] - - elif sys.argv[n] == "-x" : - flag = sys.argv[n+1].split(',') - flags=dict() - for ff in flag: - flags[ff] = 1 - #get the blocks from input box - elif sys.argv[n] == "-f" : -# if flags.get('f') == 1 : - if 1 : - while 1: - if sys.argv[n+1][0]!='-': blkstring = blkstring + sys.argv[n+1] + ' ' - else: break - n = n + 1 - nn = 0 - while nn < len(blkstring) : - if blkstring[nn]=='X' and blkstring[nn+1]=='X' : - blkstring=blkstring[:nn]+' '+blkstring[nn+2:] - nn = nn - 2 - if blkstring[nn]==' ' : - if blkstring[nn-1]!=' ': blkstring=blkstring[:nn]+' '+blkstring[nn+1:] - else : - blkstring=blkstring[:nn]+blkstring[nn+1:] - nn = nn - 1 - nn = nn + 1 - if len(blkstring) != 0 : - blks = blkstring.strip().split(' ') - for blk in blks : - block = [] - start = int(blk.split(':')[1].split('-')[0]) - end = int(blk.split(':')[1].split('-')[1]) - block.append(blk.split(':')[0].split('chr')[1]) - block.append(start-range) - block.append(end+range) - blocks.append(block) - chroms[blk.split(':')[0].split('chr')[1]] = 1 - #get the blocks from bed file - elif sys.argv[n] == "-b" : -# if flags.get('b')==1 and os.path.exists(sys.argv[n+1]) : - if os.path.exists(sys.argv[n+1]) : - bed_file = open(sys.argv[n+1], 'r') - for line in bed_file.readlines() : - block = [] - start = int(line.split('\t')[1]) - end = int(line.split('\t')[2]) - block.append(line.split('\t')[0].split('chr')[1]) - block.append(start-range) - block.append(end+range) - blocks.append(block) - chroms[line.split('\t')[0].split('chr')[1]] = 1 - #get the chroms from check box - elif sys.argv[n] == "-c" : - if flags.get('c')==1 and sys.argv[n+1][0]!='-' and sys.argv[n+1]!=None and sys.argv[n+1]!='None': - chromstmp = sys.argv[n+1].strip('\n').split(',') - for chromtmp in chromstmp: - block=[] - block.append(chromtmp) - block.append(0) - block.append(0) - blocks.append(block) - chroms[chromtmp] = 1 - #get the chroms from check box - elif sys.argv[n] == "-r" : range = int(sys.argv[n+1]) - - - #get the window size and header size. In our program, the header size is fixed to be 32. - elif sys.argv[n] == "-w" : wsize = int(sys.argv[n+1]) - elif sys.argv[n] == "-s" : shsize = int(sys.argv[n+1]) - - #output files - elif sys.argv[n] == "-o" : out_file = open(sys.argv[n+1], 'w') - elif sys.argv[n] == "-i" : info_file = open(sys.argv[n+1], 'w') - elif sys.argv[n] == "-l" : log_file = open(sys.argv[n+1], 'w') - - #get the patterns information - elif sys.argv[n] == "-p" : - while 1: - if sys.argv[n+1][0]!='-': ptnstring = ptnstring + sys.argv[n+1] + ' ' - else: break - n = n + 1 - nn = 0 - code = 'ACGTMRWSYKBDHVN0123456789' - while nn < len(ptnstring) : - if code.find(ptnstring[nn])==-1 : - if ptnstring[nn-1]!=' ': ptnstring=ptnstring[:nn]+' '+ptnstring[nn+1:] - else : - ptnstring=ptnstring[:nn]+ptnstring[nn+1:] - nn = nn - 1 - nn = nn + 1 - ptns = ptnstring.strip().split(' ') - n = n + 1 - - #check the genome, chroms, and ptns not to be empty - if genome_name=='' or len(chroms)==0 or ptns=='': - findcluster_mysql_subs.usage() - sys.exit() - - #---pattern data-------- - patterns = [] - patterns_c = ['A','B','C','D','E','F','G','H','I','J','K','L','M','N','O','P','Q','R','S','T','U','V','W','X','Y','Z'] - patterns_name = dict() - combines = dict() - n = 0 - nptns = len(ptns)/2 - while n < nptns: - patterns.append(ptns[2*n]) - combines[ptns[2*n]] = int(ptns[2*n+1]) - patterns_name[ptns[2*n]] = patterns_c[n] - n = n + 1 - - #restrict the pattern to be longer than 2bp(not include 2) - code = 'ACGTMRWSYKBDHV' - for pattern in patterns: - n = 0 - for c in pattern: - if code.find(c)!=-1 : n = n + 1 - if n < 3: sys.exit() - - - - #---------------------------------------------------------------------------------------- - #-------------------------- find clusters ----------------------------------------------- - #--------print bed file header--------- - out_file.write("#1. chrom") - out_file.write("\n#2. chromStart.Note:The first base in a chromosome is numbered 0.") - out_file.write("\n#3. chromEnd") - out_file.write("\n#4. Pattern order. E.g. BABCBD, each letter represents one pattern.") - out_file.write("\n#5. score. If the track line useScore attribute is set to 1 for this annotation data set, the score value will determine the level of gray.") - out_file.write("\n#6. strand") - out_file.write("\n#7. thickStart. The starting position at which the feature is drawn thickly.") - out_file.write("\n#8. thickEnd. The ending position at which the feature is drawn thickly.") - out_file.write("\n#9. itemRgb. An RGB value of the form R,G,B (e.g. 255,0,0). If the track line itemRgb is set to 'On', this RBG value will determine the display color. ") - out_file.write("\n#10. blockCount. The number of blocks (exons) in the BED line.") - out_file.write("\n#11. blockSizes. A comma-separated list of the block sizes. ") - out_file.write("\n#12. blockStarts. A comma-separated list of block starts.\n") - - result = dict() - for chrom_name in chroms.keys() : - if chrom_name != '' : - result[chrom_name] = findcluster_mysql_subs.scan_chromosome(genome_name, chrom_name, blocks, patterns, combines, patterns_name, wsize, shsize, out_file, log_file) - - #---------------------------------------------------------------------------------------- - #-------------------------- print out results ------------------------------------------- - if info_file == 'None' : - print "%-30s\t" % " ", - ii = 0 - while ii=block[1] and (pos<=block[2] or block[2]==0) : - npos = npos + 1 - print npos, "\t", - nclus = 0 - for key in result[block[0]]['NO'].keys() : - clus = result[block[0]]['NO'][key] - if clus['start']>=block[1] and ( clus['start']<=block[2] or block[2]==0) or clus['end']>=block[1] and ( clus['end']<=block[2] or block[2]==0) : - nclus = nclus + 1 - print result[chrom_name]['C'], "\t", nclus - print "Note:\n1)", - ii = 0 - while ii=block[1] and (pos<=block[2] or block[2]==0) : - npos = npos + 1 - info_file.write(str(npos)+"\t") - nclus = 0 - for key in result[block[0]]['NO'].keys() : - clus = result[block[0]]['NO'][key] - if clus['start']>=block[1] and ( clus['start']<=block[2] or block[2]==0) or clus['end']>=block[1] and ( clus['end']<=block[2] or block[2]==0) : - nclus = nclus + 1 - info_file.write(str(result[chrom_name]['C'])+"\t"+str(nclus)+"\n") - info_file.write("Note:\n1) ") - ii = 0 - while ii - search for clusters(specific combination of patterns in a window size) on a genome - findcluster_mysql.py -g $dbkey -x $flag -c $chroms -f $positions -b $input1 -r $range -p $patterns -w $wsize -o $out_file1 -i $out_file2 -l $out_file3 - - - - - - - - - - - - - - - - - - - - - - - - - -.. |INFO| image:: ../static/images/icon_info_sml.gif - ------ - -**Syntax** - -This tool uses a combined method (suffix-header approach and database support) to find clusters of patterns. - ------ - -**The steps** are: - -1. Select 'Genome' and click on 'Next step' button; -2. Select chroms on the genome; -3. Or input the blocks on the genome. E.g. chr2L:110020-0. Note: if the second number is 0, it means 'end'. -4. Or select a bed file that contains the blocks on the genome. -5. Input the block range. The search will extend the blocks in upstream and downstream. Default value is 0, which means no extension. -6. Input 'Patterns': patterns and occurrences in cluster. Here is a simple example: ARWYAKGCAART,1,YRTGRGAR,2,TGGYAATTW,1,GCCSSRGGV,2, -7. Input 'window size' for the cluster; -8. Click on 'Execute' button to run the tool. - ------ - -**Example** - -.. image:: ../static/patmat/findcluster.png - ------ - -**The outputs and display in ucsc genome browser** - -1. The results are 3 files: 1)a bed format file of the clusters, 2)statistic file, 3)running log. -2. To display the bed format file of result clusters in genome browser, click on 'display at UCSC main'. - -
diff --git a/tools/patmat/findcluster_mysql_subs.py b/tools/patmat/findcluster_mysql_subs.py deleted file mode 100644 index 7149ab87263..00000000000 --- a/tools/patmat/findcluster_mysql_subs.py +++ /dev/null @@ -1,368 +0,0 @@ -#!/usr/bin/env python -import sys, os, sets -from time import time, localtime, strftime -import pkg_resources -pkg_resources.require( "sqlalchemy>=0.2" ) -from sqlalchemy import * - -assert sys.version_info[:2] >= ( 2, 4 ) - -STDERR = sys.stderr - - -#--------------------------------------------------------------------------------------------------- -#--------------------sub-functions for findcluster.py---------------------------------------------------- - -#return genomes available in the datasets -def get_available_data_genomes( ): - sqlquery = "SELECT distinct genome, genome_desc from genome order by genome_desc" - db = create_engine('mysql://stree:12345@scofield.bx.psu.edu/stree') - conn = db.connect() - matchs = conn.execute(sqlquery) - - available_sets = [] - for match in matchs : - available_sets.append( (match[1], match[0], True) ) - available_sets.append( ("---", "---", True) ) - conn.close() - return available_sets - -#return chroms available of the genome in the datasets -def get_available_data_chroms( genome ): - sqlquery = "SELECT chrom from genome where genome= \'%s\' order by chrom" % genome - db = create_engine('mysql://stree:12345@scofield.bx.psu.edu/stree') - conn = db.connect() - available_sets = [] - - nn = 1 - while nn < 3 : - matchs = conn.execute(sqlquery) - for match in matchs : - if len(match[0]) == nn and match[0].isdigit(): - available_sets.append( (match[0], match[0], True) ) - nn = nn + 1 - matchs = conn.execute(sqlquery) - for match in matchs : - if not match[0].isdigit(): - available_sets.append( (match[0], match[0], True) ) - conn.close() - return available_sets - - -#--------------------------------------------------------------------------------------------------- -#--------------------sub-functions for findcluster.py---------------------------------------------------- - -def usage() : - print 'Usage: python findcluster_mysql.py -g dm2 -c 2L -p patterns -w 3500 -s 32 -o out_file -i info_file -l log_file' - print ' -g (required) genome name' - print ' -c (required) chromosome names, e.g."2L,2R," (seperated by comma) ' - print ' -p (required) string of patterns, per pattern per line, e.g."AYFGDFGA,2," (seperated by anything but pattern codes and -) ' - print ' -w (optional) cluster window size(default 3500)' - print ' -s (optional) suffix header size(default 32)' - print ' -o (required) output file' - print ' -o (optional) info file' - print ' -o (required) log file' - -def sn2num(sn) : - if sn == 'A' : return 0 - if sn == 'C' : return 1 - if sn == 'G' : return 2 - if sn == 'T' : return 3 - return 4 - -def dna2num(dna, mdna, bit) : - nn = 0 - num = [0,0] - while nn= 0 : - if matchs[n] == matchs[kk] : - del matchs[n] - kk = kk - 1 - else : - kk = n - n = n - 1 - return matchs - -def sortedDictValues(adict): - keys = adict.keys() - keys.sort() - return map(adict.get, keys) - -def pattern_match(conn, shsize, pat, table_name, log_file): - mpattern = len(pat) - patterns = expand_pattern(pat) - patterns_cmpl = expand_pattern_cmpl(pat) - - matchs = [] - log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime())) - log_file.write("\tmatch: %-10s..." % pat) - - num = dna2num(patterns[0], mpattern, shsize) - sqlquery = "SELECT chromStart from %s where " % table_name + "0 " - for pattern in patterns: - num=dna2num(pattern, mpattern, shsize) - sqlquery = sqlquery + "or num>=%s " % num[0] + "and num<%s " % num[1] - for pattern in patterns_cmpl: - num=dna2num(pattern, mpattern, shsize) - sqlquery = sqlquery + "or num>=%s " % num[0] + "and num<%s " % num[1] - sqlquery = sqlquery + " order by chromStart" - log_file.write(strftime("\t%H:%M:%S", localtime())) - log_file.write("...query database") - rr = conn.execute(sqlquery) - log_file.write(strftime("\t%H:%M:%S", localtime())) - log_file.write("...extract matchs") - for match in rr : - matchs.append(match[0]) - matchs_tmp = del_same(matchs) - log_file.write("-->(%d)" % len(matchs_tmp)) - - return matchs_tmp - -def scan_chromosome(genome_name, chrom, blocks, patterns, combines, patterns_name, wsize, shsize, out_file, log_file) : - chrom_name = chrom - log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime())) - log_file.write("\tStart on chrom %s " % chrom) - result = dict() - result['M'] = dict() - - - #---------------------------------------------------------------------------------------- - #----------------find matchs------------------------------------------------------------- - db = create_engine('mysql://stree:12345@scofield.bx.psu.edu/stree') - conn = db.connect() - total_matchs = 0 - table_name = genome_name + "_" + chrom_name - fp = open('hjb.txt', 'w') - for pattern in patterns: - result['M'][pattern] = pattern_match(conn, shsize, pattern, table_name, log_file) - nmatch = len (result['M'][pattern]) - while nmatch > 0 : - flag = 0 - for block in blocks : - if block[0] == chrom : - if int(block[2])==0 or int(block[1])(%d)" % len(result['M'][pattern])) - conn.close() - - #---------------------------------------------------------------------------------------- - #---------------find clusters in window size--------------------------------------------- - log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime())) - log_file.write("\tFind clusters in %d wsize ... " % wsize) - kk = 0 - clusters = dict() - nnclusters = 0 - - n = dict() - ismatch = dict() - for pattern in patterns : - n[pattern] = 0 - ismatch[pattern] = 0 - clusters_temp = dict() - clusters_temp["positions"] = [] - clusters_temp["patterns"] = [] - clusters_temp["strands"] = [] - - - nn = 0 - jj = 0 - while nn < total_matchs : - - # put matches of the patterns into clusters_temp: start from nnth match - if len(clusters_temp["positions"]) >0 : - ismatch[clusters_temp["patterns"][0]] = ismatch[clusters_temp["patterns"][0]] - 1 - del clusters_temp["positions"][0] - del clusters_temp["patterns"][0] - del clusters_temp["strands"][0] - while jj < total_matchs : - for pattern in patterns : - if len(result['M'][pattern])>0 : - current_pat = pattern - break - for pattern in patterns : - if len(result['M'][pattern])>0 : - if result['M'][pattern][n[pattern]] < result['M'][current_pat][n[current_pat]]: - current_pat = pattern - if len(clusters_temp["positions"]) > 0 : - if result['M'][current_pat][n[current_pat]] > clusters_temp["positions"][0] + wsize : - break - ismatch[current_pat] = ismatch[current_pat] + 1 - clusters_temp["positions"].append(result['M'][current_pat][n[current_pat]]) - clusters_temp["patterns"].append(current_pat) - clusters_temp["strands"].append('+') - n[current_pat] = n[current_pat] + 1 - if n[current_pat] == len(result['M'][current_pat]) : - for pattern in patterns : - while n[pattern] < len(result['M'][pattern]): - if result['M'][pattern][n[pattern]] < clusters_temp["positions"][0] + wsize : - clusters_temp["positions"].append(result['M'][pattern][n[pattern]]) - clusters_temp["patterns"].append(pattern) - clusters_temp["strands"].append('+') - n[pattern] = n[pattern] + 1 - jj = total_matchs - nn = total_matchs - jj = jj + 1 - - - #check if the cluster contains the minimum occurrences of patterns - allmatch = 0 - for pattern in patterns : - if combines[pattern] != 0 and ismatch[pattern] < combines[pattern]: - allmatch = 1 - - - #if yes, add the clusters_temp to clusters array - if allmatch == 0 : - nnclusters = nnclusters + 1 - - #kk==0, means it's the first cluster - if kk == 0 : - clusters[kk] = dict() - clusters[kk]["start"] = clusters_temp["positions"][0] - clusters[kk]["positions"] = [] - clusters[kk]["patterns"] = [] - clusters[kk]["strands"] = [] - ii = 0 - nclusters_temp = len(clusters_temp["positions"]) - while ii < nclusters_temp : - clusters[kk]["positions"].append(clusters_temp["positions"][ii]-clusters[kk]["start"]) - clusters[kk]["patterns"].append(patterns_name[clusters_temp["patterns"][ii]]) - ii = ii + 1 - clusters[kk]["strands"].extend(clusters_temp["strands"]) - clusters[kk]["end"] = clusters_temp["positions"][-1] + len(clusters_temp["patterns"][-1]) - kk = kk + 1 - else : - - #if the clusters_temp overlaps with the last cluster in the cluster array, then merge them - if clusters_temp["positions"][0] <= clusters[kk-1]["end"]: - ii = 0 - while ii < len(clusters_temp["positions"]) : - if clusters_temp["positions"][ii] > clusters[kk-1]["end"] - len(clusters_temp["patterns"][ii]) : - clusters[kk-1]["positions"].append(clusters_temp["positions"][ii]-clusters[kk-1]["start"]) - clusters[kk-1]["patterns"].append(patterns_name[clusters_temp["patterns"][ii]]) - clusters[kk-1]["strands"].append(clusters_temp["strands"][ii]) - ii = ii + 1 - clusters[kk-1]["end"] = clusters_temp["positions"][-1] + len(clusters_temp["patterns"][-1]) - - #otherwise, add as a new cluster to the array - else : - clusters[kk] = dict() - clusters[kk]["start"] = clusters_temp["positions"][0] - clusters[kk]["positions"] = [] - clusters[kk]["patterns"] = [] - clusters[kk]["strands"] = [] - ii = 0 - nclusters_temp = len(clusters_temp["positions"]) - while ii < nclusters_temp : - clusters[kk]["positions"].append(clusters_temp["positions"][ii]-clusters[kk]["start"]) - clusters[kk]["patterns"].append(patterns_name[clusters_temp["patterns"][ii]]) - ii = ii + 1 - clusters[kk]["strands"].extend(clusters_temp["strands"]) - clusters[kk]["end"] = clusters_temp["positions"][-1] + len(clusters_temp["patterns"][-1]) - kk = kk + 1 - nn = nn + 1 - log_file.write("\t%d" % nnclusters) - nclusters = len(clusters) - result['C'] = nnclusters - result['NO'] = clusters - - - #---------------------------------------------------------------------------------------- - #-----------------print clusters without overlap---------------------------------------------------------- - log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime())) - log_file.write("\tMerge overlap clusters...") - log_file.write("\t\t%d" % nclusters) - patterns_len = dict() - for pattern in patterns : - patterns_len[patterns_name[pattern]] = len(pattern) - - kk = 0 - while kk < nclusters: - if clusters.get(kk) != None : - if clusters[kk]["end"] - clusters[kk]["start"] != clusters[kk]["positions"][-1] + patterns_len[clusters[kk]["patterns"][-1]] : - log_file.write("\n"+chrom_name) - log_file.write("\t"+str(kk)+"/"+str(nclusters)) - log_file.write("\t"+str(clusters[kk]["end"] - clusters[kk]["start"] - clusters[kk]["positions"][-1]) ) - clusters[kk]["end"] = clusters[kk]["start"] + clusters[kk]["positions"][-1] + patterns_len[clusters[kk]["patterns"][-1]] - out_file.write(str("chr%s" % chrom_name)) - out_file.write("\t") - out_file.write(str(clusters[kk]["start"])) - out_file.write("\t") - out_file.write(str(clusters[kk]["end"])) - out_file.write("\t") - name = str(clusters[kk]["patterns"]).replace("'", "").replace(",","").replace(" ","").replace("[","").replace("]","") - if len(name) > 30 : - name = "c"+str(kk) - out_file.write(name) - out_file.write("\t0\t+\t") - out_file.write(str(clusters[kk]["start"])) - out_file.write("\t") - out_file.write(str(clusters[kk]["end"])) - out_file.write("\t0\t") - out_file.write(str(len(clusters[kk]["patterns"]))) - out_file.write("\t") - for pattern in clusters[kk]["patterns"] : - out_file.write("%d," % patterns_len[pattern]) - out_file.write("\t") - out_file.write(str(clusters[kk]["positions"]).replace("L", "").replace(" ","").replace("[","").replace("]","")) - out_file.write(",\n") - kk = kk + 1 - - log_file.write(strftime("\n%Y-%b-%d %H:%M:%S\t", localtime())) - log_file.write("OK!") - - return result - -