diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample
index fc8ff6038a1..0abda398e91 100644
--- a/tool_conf.xml.sample
+++ b/tool_conf.xml.sample
@@ -45,6 +45,7 @@
+
diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py
index 302b1bfff5b..5b6c28bc6af 100644
--- a/tools/stats/grouping.py
+++ b/tools/stats/grouping.py
@@ -15,22 +15,31 @@ for var in sys.argv[4:]:
ops.append(var.split()[0])
cols.append(var.split()[1])
+if os.path.exists( inputfile ):
+ for line in open( inputfile ):
+ line = line.strip()
+ if line and not line.startswith( '#' ):
+ elems = line.split( '\t' )
+ break
+else:
+ print 'The data file you selected for filtering does not exist.'
+ sys.exit()
+
groupcol = string.atoi(sys.argv[3])
-if groupcol > len( open(inputfile).readline().split('\t') ):
+if groupcol > len( elems ):
print >> sys.stderr, "Column %d does not exist." %(groupcol)
sys.exit()
groupcol = groupcol-1
for k,col in enumerate(cols):
col = int(col)
- flds = open(inputfile).readline().split('\t')
- if col > len( flds ):
+ if col > len( elems ):
print >> sys.stderr, "Column %d does not exist." %(col)
sys.exit()
else:
if ops[k] != "c":
try:
- assert float(flds[col-1])
+ assert float(elems[col-1])
except:
print >> sys.stderr, "Operation '%s' cannot be performed on non-numeric column %d." %(ops[k],col)
sys.exit()
@@ -50,46 +59,50 @@ first_invalid_line = 0
invalid_line = None
for line in open(tmpfile.name):
- fields = line.split("\t")
- item = fields[groupcol]
- if previtem != "":
- if item == previtem: #Keep iterating and storing values till a new item is encountered.
- previtem = item
- for i,col in enumerate(cols):
- col = string.atoi(col)
- col = col-1
- prevvals[i].append(fields[col].strip())
- else: #When a new item is encountered, write the previous item and the corresponding aggregate values into the output file.
- outstr = previtem
- try:
- for i,op in enumerate(ops):
- rfunc = "r." + op
- if op != 'c':
- for j,elem in enumerate(prevvals[i]):
- prevvals[i][j] = float(elem)
- rout = "%.2f" %(eval(rfunc)(prevvals[i]))
- else:
- rout = eval(rfunc)(prevvals[i])
- outstr += "\t" + str(rout)
- print >>fout, outstr
- except:
- skipped_lines += 1
- previtem = item
- prevvals = []
- for col in cols:
- col = string.atoi(col)
- col = col-1
- vallist = []
- vallist.append(fields[col].strip())
- prevvals.append(vallist)
- else: #visited only once right at the start of the iteration.
- previtem = item
- for col in cols:
- col = string.atoi(col)
- col = col-1
- vallist = []
- vallist.append(fields[col].strip())
- prevvals.append(vallist)
+ if line and (not line.startswith( '#' )) and line != '':
+ try:
+ fields = line.split("\t")
+ item = fields[groupcol]
+ if previtem != "":
+ if item == previtem: #Keep iterating and storing values till a new item is encountered.
+ previtem = item
+ for i,col in enumerate(cols):
+ col = string.atoi(col)
+ col = col-1
+ prevvals[i].append(fields[col].strip())
+ else: #When a new item is encountered, write the previous item and the corresponding aggregate values into the output file.
+ outstr = previtem
+ try:
+ for i,op in enumerate(ops):
+ rfunc = "r." + op
+ if op != 'c':
+ for j,elem in enumerate(prevvals[i]):
+ prevvals[i][j] = float(elem)
+ rout = "%.2f" %(eval(rfunc)(prevvals[i]))
+ else:
+ rout = eval(rfunc)(prevvals[i])
+ outstr += "\t" + str(rout)
+ print >>fout, outstr
+ except:
+ skipped_lines += 1
+ previtem = item
+ prevvals = []
+ for col in cols:
+ col = string.atoi(col)
+ col = col-1
+ vallist = []
+ vallist.append(fields[col].strip())
+ prevvals.append(vallist)
+ else: #visited only once right at the start of the iteration.
+ previtem = item
+ for col in cols:
+ col = string.atoi(col)
+ col = col-1
+ vallist = []
+ vallist.append(fields[col].strip())
+ prevvals.append(vallist)
+ except:
+ pass
outstr = previtem
for i,op in enumerate(ops):