diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample
index 3db546f2bb6..9a19d986e3e 100644
--- a/tool_conf.xml.sample
+++ b/tool_conf.xml.sample
@@ -68,11 +68,6 @@
-
diff --git a/tools/patmat/findcluster_mysql.py b/tools/patmat/findcluster_mysql.py
deleted file mode 100755
index 72f5c917fd4..00000000000
--- a/tools/patmat/findcluster_mysql.py
+++ /dev/null
@@ -1,301 +0,0 @@
-#!/usr/bin/env python
-import urllib, sys, os, sets, findcluster_mysql_subs
-from time import time, localtime, strftime
-import pkg_resources
-pkg_resources.require( "sqlalchemy>=0.2" )
-from sqlalchemy import *
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-if __name__ == '__main__':
- nargv = len(sys.argv)
- if nargv < 2 :
- findcluster_mysql_subs.usage()
- sys.exit()
-
- #----------------------------------------------------------------------------------------
- #--------------------------read params and files-----------------------------------------
- genome_name = ''
- chrom_name = ''
- chroms = dict()
- blocks = []
- range = 0
- blkstring = ''
- chrs = []
- ptnstring = ''
- ptns = []
- patterns = []
- wsize = 3500
- shsize = 32
- n = 1
- info_file = 'None'
- while n < nargv :
- #get the genome parameter
- if sys.argv[n] == "-g" : genome_name = sys.argv[n+1]
-
- elif sys.argv[n] == "-x" :
- flag = sys.argv[n+1].split(',')
- flags=dict()
- for ff in flag:
- flags[ff] = 1
- #get the blocks from input box
- elif sys.argv[n] == "-f" :
-# if flags.get('f') == 1 :
- if 1 :
- while 1:
- if sys.argv[n+1][0]!='-': blkstring = blkstring + sys.argv[n+1] + ' '
- else: break
- n = n + 1
- nn = 0
- while nn < len(blkstring) :
- if blkstring[nn]=='X' and blkstring[nn+1]=='X' :
- blkstring=blkstring[:nn]+' '+blkstring[nn+2:]
- nn = nn - 2
- if blkstring[nn]==' ' :
- if blkstring[nn-1]!=' ': blkstring=blkstring[:nn]+' '+blkstring[nn+1:]
- else :
- blkstring=blkstring[:nn]+blkstring[nn+1:]
- nn = nn - 1
- nn = nn + 1
- if len(blkstring) != 0 :
- blks = blkstring.strip().split(' ')
- for blk in blks :
- block = []
- start = int(blk.split(':')[1].split('-')[0])
- end = int(blk.split(':')[1].split('-')[1])
- block.append(blk.split(':')[0].split('chr')[1])
- block.append(start-range)
- block.append(end+range)
- blocks.append(block)
- chroms[blk.split(':')[0].split('chr')[1]] = 1
- #get the blocks from bed file
- elif sys.argv[n] == "-b" :
-# if flags.get('b')==1 and os.path.exists(sys.argv[n+1]) :
- if os.path.exists(sys.argv[n+1]) :
- bed_file = open(sys.argv[n+1], 'r')
- for line in bed_file.readlines() :
- block = []
- start = int(line.split('\t')[1])
- end = int(line.split('\t')[2])
- block.append(line.split('\t')[0].split('chr')[1])
- block.append(start-range)
- block.append(end+range)
- blocks.append(block)
- chroms[line.split('\t')[0].split('chr')[1]] = 1
- #get the chroms from check box
- elif sys.argv[n] == "-c" :
- if flags.get('c')==1 and sys.argv[n+1][0]!='-' and sys.argv[n+1]!=None and sys.argv[n+1]!='None':
- chromstmp = sys.argv[n+1].strip('\n').split(',')
- for chromtmp in chromstmp:
- block=[]
- block.append(chromtmp)
- block.append(0)
- block.append(0)
- blocks.append(block)
- chroms[chromtmp] = 1
- #get the chroms from check box
- elif sys.argv[n] == "-r" : range = int(sys.argv[n+1])
-
-
- #get the window size and header size. In our program, the header size is fixed to be 32.
- elif sys.argv[n] == "-w" : wsize = int(sys.argv[n+1])
- elif sys.argv[n] == "-s" : shsize = int(sys.argv[n+1])
-
- #output files
- elif sys.argv[n] == "-o" : out_file = open(sys.argv[n+1], 'w')
- elif sys.argv[n] == "-i" : info_file = open(sys.argv[n+1], 'w')
- elif sys.argv[n] == "-l" : log_file = open(sys.argv[n+1], 'w')
-
- #get the patterns information
- elif sys.argv[n] == "-p" :
- while 1:
- if sys.argv[n+1][0]!='-': ptnstring = ptnstring + sys.argv[n+1] + ' '
- else: break
- n = n + 1
- nn = 0
- code = 'ACGTMRWSYKBDHVN0123456789'
- while nn < len(ptnstring) :
- if code.find(ptnstring[nn])==-1 :
- if ptnstring[nn-1]!=' ': ptnstring=ptnstring[:nn]+' '+ptnstring[nn+1:]
- else :
- ptnstring=ptnstring[:nn]+ptnstring[nn+1:]
- nn = nn - 1
- nn = nn + 1
- ptns = ptnstring.strip().split(' ')
- n = n + 1
-
- #check the genome, chroms, and ptns not to be empty
- if genome_name=='' or len(chroms)==0 or ptns=='':
- findcluster_mysql_subs.usage()
- sys.exit()
-
- #---pattern data--------
- patterns = []
- patterns_c = ['A','B','C','D','E','F','G','H','I','J','K','L','M','N','O','P','Q','R','S','T','U','V','W','X','Y','Z']
- patterns_name = dict()
- combines = dict()
- n = 0
- nptns = len(ptns)/2
- while n < nptns:
- patterns.append(ptns[2*n])
- combines[ptns[2*n]] = int(ptns[2*n+1])
- patterns_name[ptns[2*n]] = patterns_c[n]
- n = n + 1
-
- #restrict the pattern to be longer than 2bp(not include 2)
- code = 'ACGTMRWSYKBDHV'
- for pattern in patterns:
- n = 0
- for c in pattern:
- if code.find(c)!=-1 : n = n + 1
- if n < 3: sys.exit()
-
-
-
- #----------------------------------------------------------------------------------------
- #-------------------------- find clusters -----------------------------------------------
- #--------print bed file header---------
- out_file.write("#1. chrom")
- out_file.write("\n#2. chromStart.Note:The first base in a chromosome is numbered 0.")
- out_file.write("\n#3. chromEnd")
- out_file.write("\n#4. Pattern order. E.g. BABCBD, each letter represents one pattern.")
- out_file.write("\n#5. score. If the track line useScore attribute is set to 1 for this annotation data set, the score value will determine the level of gray.")
- out_file.write("\n#6. strand")
- out_file.write("\n#7. thickStart. The starting position at which the feature is drawn thickly.")
- out_file.write("\n#8. thickEnd. The ending position at which the feature is drawn thickly.")
- out_file.write("\n#9. itemRgb. An RGB value of the form R,G,B (e.g. 255,0,0). If the track line itemRgb is set to 'On', this RBG value will determine the display color. ")
- out_file.write("\n#10. blockCount. The number of blocks (exons) in the BED line.")
- out_file.write("\n#11. blockSizes. A comma-separated list of the block sizes. ")
- out_file.write("\n#12. blockStarts. A comma-separated list of block starts.\n")
-
- result = dict()
- for chrom_name in chroms.keys() :
- if chrom_name != '' :
- result[chrom_name] = findcluster_mysql_subs.scan_chromosome(genome_name, chrom_name, blocks, patterns, combines, patterns_name, wsize, shsize, out_file, log_file)
-
- #----------------------------------------------------------------------------------------
- #-------------------------- print out results -------------------------------------------
- if info_file == 'None' :
- print "%-30s\t" % " ",
- ii = 0
- while ii=block[1] and (pos<=block[2] or block[2]==0) :
- npos = npos + 1
- print npos, "\t",
- nclus = 0
- for key in result[block[0]]['NO'].keys() :
- clus = result[block[0]]['NO'][key]
- if clus['start']>=block[1] and ( clus['start']<=block[2] or block[2]==0) or clus['end']>=block[1] and ( clus['end']<=block[2] or block[2]==0) :
- nclus = nclus + 1
- print result[chrom_name]['C'], "\t", nclus
- print "Note:\n1)",
- ii = 0
- while ii=block[1] and (pos<=block[2] or block[2]==0) :
- npos = npos + 1
- info_file.write(str(npos)+"\t")
- nclus = 0
- for key in result[block[0]]['NO'].keys() :
- clus = result[block[0]]['NO'][key]
- if clus['start']>=block[1] and ( clus['start']<=block[2] or block[2]==0) or clus['end']>=block[1] and ( clus['end']<=block[2] or block[2]==0) :
- nclus = nclus + 1
- info_file.write(str(result[chrom_name]['C'])+"\t"+str(nclus)+"\n")
- info_file.write("Note:\n1) ")
- ii = 0
- while ii
- search for clusters(specific combination of patterns in a window size) on a genome
- findcluster_mysql.py -g $dbkey -x $flag -c $chroms -f $positions -b $input1 -r $range -p $patterns -w $wsize -o $out_file1 -i $out_file2 -l $out_file3
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-.. |INFO| image:: ../static/images/icon_info_sml.gif
-
------
-
-**Syntax**
-
-This tool uses a combined method (suffix-header approach and database support) to find clusters of patterns.
-
------
-
-**The steps** are:
-
-1. Select 'Genome' and click on 'Next step' button;
-2. Select chroms on the genome;
-3. Or input the blocks on the genome. E.g. chr2L:110020-0. Note: if the second number is 0, it means 'end'.
-4. Or select a bed file that contains the blocks on the genome.
-5. Input the block range. The search will extend the blocks in upstream and downstream. Default value is 0, which means no extension.
-6. Input 'Patterns': patterns and occurrences in cluster. Here is a simple example: ARWYAKGCAART,1,YRTGRGAR,2,TGGYAATTW,1,GCCSSRGGV,2,
-7. Input 'window size' for the cluster;
-8. Click on 'Execute' button to run the tool.
-
------
-
-**Example**
-
-.. image:: ../static/patmat/findcluster.png
-
------
-
-**The outputs and display in ucsc genome browser**
-
-1. The results are 3 files: 1)a bed format file of the clusters, 2)statistic file, 3)running log.
-2. To display the bed format file of result clusters in genome browser, click on 'display at UCSC main'.
-
-
diff --git a/tools/patmat/findcluster_mysql_subs.py b/tools/patmat/findcluster_mysql_subs.py
deleted file mode 100644
index 7149ab87263..00000000000
--- a/tools/patmat/findcluster_mysql_subs.py
+++ /dev/null
@@ -1,368 +0,0 @@
-#!/usr/bin/env python
-import sys, os, sets
-from time import time, localtime, strftime
-import pkg_resources
-pkg_resources.require( "sqlalchemy>=0.2" )
-from sqlalchemy import *
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-STDERR = sys.stderr
-
-
-#---------------------------------------------------------------------------------------------------
-#--------------------sub-functions for findcluster.py----------------------------------------------------
-
-#return genomes available in the datasets
-def get_available_data_genomes( ):
- sqlquery = "SELECT distinct genome, genome_desc from genome order by genome_desc"
- db = create_engine('mysql://stree:12345@scofield.bx.psu.edu/stree')
- conn = db.connect()
- matchs = conn.execute(sqlquery)
-
- available_sets = []
- for match in matchs :
- available_sets.append( (match[1], match[0], True) )
- available_sets.append( ("---", "---", True) )
- conn.close()
- return available_sets
-
-#return chroms available of the genome in the datasets
-def get_available_data_chroms( genome ):
- sqlquery = "SELECT chrom from genome where genome= \'%s\' order by chrom" % genome
- db = create_engine('mysql://stree:12345@scofield.bx.psu.edu/stree')
- conn = db.connect()
- available_sets = []
-
- nn = 1
- while nn < 3 :
- matchs = conn.execute(sqlquery)
- for match in matchs :
- if len(match[0]) == nn and match[0].isdigit():
- available_sets.append( (match[0], match[0], True) )
- nn = nn + 1
- matchs = conn.execute(sqlquery)
- for match in matchs :
- if not match[0].isdigit():
- available_sets.append( (match[0], match[0], True) )
- conn.close()
- return available_sets
-
-
-#---------------------------------------------------------------------------------------------------
-#--------------------sub-functions for findcluster.py----------------------------------------------------
-
-def usage() :
- print 'Usage: python findcluster_mysql.py -g dm2 -c 2L -p patterns -w 3500 -s 32 -o out_file -i info_file -l log_file'
- print ' -g (required) genome name'
- print ' -c (required) chromosome names, e.g."2L,2R," (seperated by comma) '
- print ' -p (required) string of patterns, per pattern per line, e.g."AYFGDFGA,2," (seperated by anything but pattern codes and -) '
- print ' -w (optional) cluster window size(default 3500)'
- print ' -s (optional) suffix header size(default 32)'
- print ' -o (required) output file'
- print ' -o (optional) info file'
- print ' -o (required) log file'
-
-def sn2num(sn) :
- if sn == 'A' : return 0
- if sn == 'C' : return 1
- if sn == 'G' : return 2
- if sn == 'T' : return 3
- return 4
-
-def dna2num(dna, mdna, bit) :
- nn = 0
- num = [0,0]
- while nn= 0 :
- if matchs[n] == matchs[kk] :
- del matchs[n]
- kk = kk - 1
- else :
- kk = n
- n = n - 1
- return matchs
-
-def sortedDictValues(adict):
- keys = adict.keys()
- keys.sort()
- return map(adict.get, keys)
-
-def pattern_match(conn, shsize, pat, table_name, log_file):
- mpattern = len(pat)
- patterns = expand_pattern(pat)
- patterns_cmpl = expand_pattern_cmpl(pat)
-
- matchs = []
- log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime()))
- log_file.write("\tmatch: %-10s..." % pat)
-
- num = dna2num(patterns[0], mpattern, shsize)
- sqlquery = "SELECT chromStart from %s where " % table_name + "0 "
- for pattern in patterns:
- num=dna2num(pattern, mpattern, shsize)
- sqlquery = sqlquery + "or num>=%s " % num[0] + "and num<%s " % num[1]
- for pattern in patterns_cmpl:
- num=dna2num(pattern, mpattern, shsize)
- sqlquery = sqlquery + "or num>=%s " % num[0] + "and num<%s " % num[1]
- sqlquery = sqlquery + " order by chromStart"
- log_file.write(strftime("\t%H:%M:%S", localtime()))
- log_file.write("...query database")
- rr = conn.execute(sqlquery)
- log_file.write(strftime("\t%H:%M:%S", localtime()))
- log_file.write("...extract matchs")
- for match in rr :
- matchs.append(match[0])
- matchs_tmp = del_same(matchs)
- log_file.write("-->(%d)" % len(matchs_tmp))
-
- return matchs_tmp
-
-def scan_chromosome(genome_name, chrom, blocks, patterns, combines, patterns_name, wsize, shsize, out_file, log_file) :
- chrom_name = chrom
- log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime()))
- log_file.write("\tStart on chrom %s " % chrom)
- result = dict()
- result['M'] = dict()
-
-
- #----------------------------------------------------------------------------------------
- #----------------find matchs-------------------------------------------------------------
- db = create_engine('mysql://stree:12345@scofield.bx.psu.edu/stree')
- conn = db.connect()
- total_matchs = 0
- table_name = genome_name + "_" + chrom_name
- fp = open('hjb.txt', 'w')
- for pattern in patterns:
- result['M'][pattern] = pattern_match(conn, shsize, pattern, table_name, log_file)
- nmatch = len (result['M'][pattern])
- while nmatch > 0 :
- flag = 0
- for block in blocks :
- if block[0] == chrom :
- if int(block[2])==0 or int(block[1])(%d)" % len(result['M'][pattern]))
- conn.close()
-
- #----------------------------------------------------------------------------------------
- #---------------find clusters in window size---------------------------------------------
- log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime()))
- log_file.write("\tFind clusters in %d wsize ... " % wsize)
- kk = 0
- clusters = dict()
- nnclusters = 0
-
- n = dict()
- ismatch = dict()
- for pattern in patterns :
- n[pattern] = 0
- ismatch[pattern] = 0
- clusters_temp = dict()
- clusters_temp["positions"] = []
- clusters_temp["patterns"] = []
- clusters_temp["strands"] = []
-
-
- nn = 0
- jj = 0
- while nn < total_matchs :
-
- # put matches of the patterns into clusters_temp: start from nnth match
- if len(clusters_temp["positions"]) >0 :
- ismatch[clusters_temp["patterns"][0]] = ismatch[clusters_temp["patterns"][0]] - 1
- del clusters_temp["positions"][0]
- del clusters_temp["patterns"][0]
- del clusters_temp["strands"][0]
- while jj < total_matchs :
- for pattern in patterns :
- if len(result['M'][pattern])>0 :
- current_pat = pattern
- break
- for pattern in patterns :
- if len(result['M'][pattern])>0 :
- if result['M'][pattern][n[pattern]] < result['M'][current_pat][n[current_pat]]:
- current_pat = pattern
- if len(clusters_temp["positions"]) > 0 :
- if result['M'][current_pat][n[current_pat]] > clusters_temp["positions"][0] + wsize :
- break
- ismatch[current_pat] = ismatch[current_pat] + 1
- clusters_temp["positions"].append(result['M'][current_pat][n[current_pat]])
- clusters_temp["patterns"].append(current_pat)
- clusters_temp["strands"].append('+')
- n[current_pat] = n[current_pat] + 1
- if n[current_pat] == len(result['M'][current_pat]) :
- for pattern in patterns :
- while n[pattern] < len(result['M'][pattern]):
- if result['M'][pattern][n[pattern]] < clusters_temp["positions"][0] + wsize :
- clusters_temp["positions"].append(result['M'][pattern][n[pattern]])
- clusters_temp["patterns"].append(pattern)
- clusters_temp["strands"].append('+')
- n[pattern] = n[pattern] + 1
- jj = total_matchs
- nn = total_matchs
- jj = jj + 1
-
-
- #check if the cluster contains the minimum occurrences of patterns
- allmatch = 0
- for pattern in patterns :
- if combines[pattern] != 0 and ismatch[pattern] < combines[pattern]:
- allmatch = 1
-
-
- #if yes, add the clusters_temp to clusters array
- if allmatch == 0 :
- nnclusters = nnclusters + 1
-
- #kk==0, means it's the first cluster
- if kk == 0 :
- clusters[kk] = dict()
- clusters[kk]["start"] = clusters_temp["positions"][0]
- clusters[kk]["positions"] = []
- clusters[kk]["patterns"] = []
- clusters[kk]["strands"] = []
- ii = 0
- nclusters_temp = len(clusters_temp["positions"])
- while ii < nclusters_temp :
- clusters[kk]["positions"].append(clusters_temp["positions"][ii]-clusters[kk]["start"])
- clusters[kk]["patterns"].append(patterns_name[clusters_temp["patterns"][ii]])
- ii = ii + 1
- clusters[kk]["strands"].extend(clusters_temp["strands"])
- clusters[kk]["end"] = clusters_temp["positions"][-1] + len(clusters_temp["patterns"][-1])
- kk = kk + 1
- else :
-
- #if the clusters_temp overlaps with the last cluster in the cluster array, then merge them
- if clusters_temp["positions"][0] <= clusters[kk-1]["end"]:
- ii = 0
- while ii < len(clusters_temp["positions"]) :
- if clusters_temp["positions"][ii] > clusters[kk-1]["end"] - len(clusters_temp["patterns"][ii]) :
- clusters[kk-1]["positions"].append(clusters_temp["positions"][ii]-clusters[kk-1]["start"])
- clusters[kk-1]["patterns"].append(patterns_name[clusters_temp["patterns"][ii]])
- clusters[kk-1]["strands"].append(clusters_temp["strands"][ii])
- ii = ii + 1
- clusters[kk-1]["end"] = clusters_temp["positions"][-1] + len(clusters_temp["patterns"][-1])
-
- #otherwise, add as a new cluster to the array
- else :
- clusters[kk] = dict()
- clusters[kk]["start"] = clusters_temp["positions"][0]
- clusters[kk]["positions"] = []
- clusters[kk]["patterns"] = []
- clusters[kk]["strands"] = []
- ii = 0
- nclusters_temp = len(clusters_temp["positions"])
- while ii < nclusters_temp :
- clusters[kk]["positions"].append(clusters_temp["positions"][ii]-clusters[kk]["start"])
- clusters[kk]["patterns"].append(patterns_name[clusters_temp["patterns"][ii]])
- ii = ii + 1
- clusters[kk]["strands"].extend(clusters_temp["strands"])
- clusters[kk]["end"] = clusters_temp["positions"][-1] + len(clusters_temp["patterns"][-1])
- kk = kk + 1
- nn = nn + 1
- log_file.write("\t%d" % nnclusters)
- nclusters = len(clusters)
- result['C'] = nnclusters
- result['NO'] = clusters
-
-
- #----------------------------------------------------------------------------------------
- #-----------------print clusters without overlap----------------------------------------------------------
- log_file.write(strftime("\n%Y-%b-%d %H:%M:%S", localtime()))
- log_file.write("\tMerge overlap clusters...")
- log_file.write("\t\t%d" % nclusters)
- patterns_len = dict()
- for pattern in patterns :
- patterns_len[patterns_name[pattern]] = len(pattern)
-
- kk = 0
- while kk < nclusters:
- if clusters.get(kk) != None :
- if clusters[kk]["end"] - clusters[kk]["start"] != clusters[kk]["positions"][-1] + patterns_len[clusters[kk]["patterns"][-1]] :
- log_file.write("\n"+chrom_name)
- log_file.write("\t"+str(kk)+"/"+str(nclusters))
- log_file.write("\t"+str(clusters[kk]["end"] - clusters[kk]["start"] - clusters[kk]["positions"][-1]) )
- clusters[kk]["end"] = clusters[kk]["start"] + clusters[kk]["positions"][-1] + patterns_len[clusters[kk]["patterns"][-1]]
- out_file.write(str("chr%s" % chrom_name))
- out_file.write("\t")
- out_file.write(str(clusters[kk]["start"]))
- out_file.write("\t")
- out_file.write(str(clusters[kk]["end"]))
- out_file.write("\t")
- name = str(clusters[kk]["patterns"]).replace("'", "").replace(",","").replace(" ","").replace("[","").replace("]","")
- if len(name) > 30 :
- name = "c"+str(kk)
- out_file.write(name)
- out_file.write("\t0\t+\t")
- out_file.write(str(clusters[kk]["start"]))
- out_file.write("\t")
- out_file.write(str(clusters[kk]["end"]))
- out_file.write("\t0\t")
- out_file.write(str(len(clusters[kk]["patterns"])))
- out_file.write("\t")
- for pattern in clusters[kk]["patterns"] :
- out_file.write("%d," % patterns_len[pattern])
- out_file.write("\t")
- out_file.write(str(clusters[kk]["positions"]).replace("L", "").replace(" ","").replace("[","").replace("]",""))
- out_file.write(",\n")
- kk = kk + 1
-
- log_file.write(strftime("\n%Y-%b-%d %H:%M:%S\t", localtime()))
- log_file.write("OK!")
-
- return result
-
-