From e75d08841366b1eb36d92347ced348beb1be8241 Mon Sep 17 00:00:00 2001 From: Guruprasad Anada Date: Thu, 5 Jul 2007 16:51:53 +0000 Subject: [PATCH] Forgot to add tool_conf.xml.sample. Also adding a small change to the tool code file. --- tool_conf.xml.sample | 1 + tools/stats/grouping.py | 101 +++++++++++++++++++++++----------------- 2 files changed, 58 insertions(+), 44 deletions(-) diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index fc8ff6038a1..0abda398e91 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -45,6 +45,7 @@
+ diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py index 302b1bfff5b..5b6c28bc6af 100644 --- a/tools/stats/grouping.py +++ b/tools/stats/grouping.py @@ -15,22 +15,31 @@ for var in sys.argv[4:]: ops.append(var.split()[0]) cols.append(var.split()[1]) +if os.path.exists( inputfile ): + for line in open( inputfile ): + line = line.strip() + if line and not line.startswith( '#' ): + elems = line.split( '\t' ) + break +else: + print 'The data file you selected for filtering does not exist.' + sys.exit() + groupcol = string.atoi(sys.argv[3]) -if groupcol > len( open(inputfile).readline().split('\t') ): +if groupcol > len( elems ): print >> sys.stderr, "Column %d does not exist." %(groupcol) sys.exit() groupcol = groupcol-1 for k,col in enumerate(cols): col = int(col) - flds = open(inputfile).readline().split('\t') - if col > len( flds ): + if col > len( elems ): print >> sys.stderr, "Column %d does not exist." %(col) sys.exit() else: if ops[k] != "c": try: - assert float(flds[col-1]) + assert float(elems[col-1]) except: print >> sys.stderr, "Operation '%s' cannot be performed on non-numeric column %d." %(ops[k],col) sys.exit() @@ -50,46 +59,50 @@ first_invalid_line = 0 invalid_line = None for line in open(tmpfile.name): - fields = line.split("\t") - item = fields[groupcol] - if previtem != "": - if item == previtem: #Keep iterating and storing values till a new item is encountered. - previtem = item - for i,col in enumerate(cols): - col = string.atoi(col) - col = col-1 - prevvals[i].append(fields[col].strip()) - else: #When a new item is encountered, write the previous item and the corresponding aggregate values into the output file. - outstr = previtem - try: - for i,op in enumerate(ops): - rfunc = "r." + op - if op != 'c': - for j,elem in enumerate(prevvals[i]): - prevvals[i][j] = float(elem) - rout = "%.2f" %(eval(rfunc)(prevvals[i])) - else: - rout = eval(rfunc)(prevvals[i]) - outstr += "\t" + str(rout) - print >>fout, outstr - except: - skipped_lines += 1 - previtem = item - prevvals = [] - for col in cols: - col = string.atoi(col) - col = col-1 - vallist = [] - vallist.append(fields[col].strip()) - prevvals.append(vallist) - else: #visited only once right at the start of the iteration. - previtem = item - for col in cols: - col = string.atoi(col) - col = col-1 - vallist = [] - vallist.append(fields[col].strip()) - prevvals.append(vallist) + if line and (not line.startswith( '#' )) and line != '': + try: + fields = line.split("\t") + item = fields[groupcol] + if previtem != "": + if item == previtem: #Keep iterating and storing values till a new item is encountered. + previtem = item + for i,col in enumerate(cols): + col = string.atoi(col) + col = col-1 + prevvals[i].append(fields[col].strip()) + else: #When a new item is encountered, write the previous item and the corresponding aggregate values into the output file. + outstr = previtem + try: + for i,op in enumerate(ops): + rfunc = "r." + op + if op != 'c': + for j,elem in enumerate(prevvals[i]): + prevvals[i][j] = float(elem) + rout = "%.2f" %(eval(rfunc)(prevvals[i])) + else: + rout = eval(rfunc)(prevvals[i]) + outstr += "\t" + str(rout) + print >>fout, outstr + except: + skipped_lines += 1 + previtem = item + prevvals = [] + for col in cols: + col = string.atoi(col) + col = col-1 + vallist = [] + vallist.append(fields[col].strip()) + prevvals.append(vallist) + else: #visited only once right at the start of the iteration. + previtem = item + for col in cols: + col = string.atoi(col) + col = col-1 + vallist = [] + vallist.append(fields[col].strip()) + prevvals.append(vallist) + except: + pass outstr = previtem for i,op in enumerate(ops):