diff --git a/test-data/groupby_out1.dat b/test-data/groupby_out1.dat
deleted file mode 100644
index f822b20ca1a..00000000000
--- a/test-data/groupby_out1.dat
+++ /dev/null
@@ -1,20 +0,0 @@
-chr1 1.48053e+08
-chr10 5.52516e+07
-chr11 8.75888e+07
-chr12 3.84401e+07
-chr13 1.12382e+08
-chr14 9.87102e+07
-chr15 4.16664e+07
-chr16 206638
-chr18 5.05624e+07
-chr19 5.92262e+07
-chr2 1.69276e+08
-chr20 3.35042e+07
-chr21 3.31607e+07
-chr22 3.04712e+07
-chr5 1.31612e+08
-chr6 1.08564e+08
-chr7 1.15958e+08
-chr8 1.18881e+08
-chr9 1.28843e+08
-chrx 1.45195e+08
diff --git a/test-data/groupby_out2.dat b/test-data/groupby_out2.dat
deleted file mode 100644
index f89d723ab44..00000000000
--- a/test-data/groupby_out2.dat
+++ /dev/null
@@ -1,3 +0,0 @@
-
-chr10 chr10
-chr22 chr22
diff --git a/test-data/groupby_out3.tabular b/test-data/groupby_out3.tabular
deleted file mode 100644
index cc5214529db..00000000000
--- a/test-data/groupby_out3.tabular
+++ /dev/null
@@ -1,20 +0,0 @@
-chr1 1.48053e+08 1.48031e+08 -,+ 1.48185e+08 1.47962e+08 5.9221e+08 4 1 chr1,chr1,chr1,chr1 +,- 148185136 88085.9
-chr10 5.52516e+07 5.52516e+07 - 5.52516e+07 5.52516e+07 5.52516e+07 1 1 chr10 - 55251623 0
-chr11 8.75888e+07 1.16165e+08 -,+ 1.16212e+08 1.81238e+06 3.50355e+08 4 1 chr11,chr11,chr11,chr11 +,- 1812377 4.9523e+07
-chr12 3.84401e+07 3.84401e+07 - 3.84401e+07 3.84401e+07 3.84401e+07 1 1 chr12 - 38440094 0
-chr13 1.12382e+08 1.12382e+08 + 1.12382e+08 1.12382e+08 1.12382e+08 1 1 chr13 + 112381694 0
-chr14 9.87102e+07 9.87102e+07 - 9.87102e+07 9.87102e+07 9.87102e+07 1 1 chr14 - 98710240 0
-chr15 4.16664e+07 4.16764e+07 -,+ 4.1826e+07 4.14869e+07 1.66666e+08 4 1 chr15,chr15,chr15,chr15 +,- 41673708 120341
-chr16 206638 212188 +,- 259268 142908 826552 4 1 chr16,chr16,chr16,chr16 +,- 244413 47354.9
-chr18 5.05624e+07 5.94314e+07 -,+ 5.96006e+07 2.37861e+07 2.0225e+08 4 1 chr18,chr18,chr18,chr18 +,- 23786114 1.54594e+07
-chr19 5.92262e+07 5.9267e+07 +,- 5.93022e+07 5.90686e+07 2.36905e+08 4 1 chr19,chr19,chr19,chr19 +,- 59236026 94686.3
-chr2 1.69276e+08 1.69292e+08 +,- 2.2023e+08 1.18289e+08 6.77103e+08 4 1 chr2,chr2,chr2,chr2 +,- 220190202 5.09343e+07
-chr20 3.35042e+07 3.35466e+07 -,+ 3.35933e+07 3.33304e+07 1.34017e+08 4 1 chr20,chr20,chr20,chr20 +,- 33579500 104752
-chr21 3.31607e+07 3.30953e+07 +,- 3.3745e+07 3.2707e+07 1.32643e+08 4 1 chr21,chr21,chr21,chr21 +,- 33321040 405475
-chr22 3.04712e+07 3.04128e+07 +,- 3.09391e+07 3.01202e+07 1.21885e+08 4 1 chr22,chr22,chr22,chr22 +,- 30120223 345080
-chr5 1.31612e+08 1.31589e+08 +,- 1.31848e+08 1.31424e+08 5.2645e+08 4 1 chr5,chr5,chr5,chr5 +,- 131556601 153188
-chr6 1.08564e+08 1.08617e+08 -,+ 1.08723e+08 1.083e+08 4.34257e+08 4 1 chr6,chr6,chr6,chr6 +,- 108640045 159611
-chr7 1.15958e+08 1.16613e+08 +,- 1.16946e+08 1.13661e+08 4.63832e+08 4 1 chr7,chr7,chr7,chr7 +,- 113660517 1.33533e+06
-chr8 1.18881e+08 1.18881e+08 - 1.18881e+08 1.18881e+08 1.18881e+08 1 1 chr8 - 118881131 0
-chr9 1.28843e+08 1.28835e+08 +,- 1.28937e+08 1.28764e+08 5.15371e+08 4 1 chr9,chr9,chr9,chr9 +,- 128787519 70228.2
-chrX 1.45195e+08 1.5267e+08 +,- 1.52694e+08 1.22745e+08 5.80779e+08 4 1 chrX,chrX,chrX,chrX +,- 152648964 1.29614e+07
diff --git a/test-data/groupby_out4.tabular b/test-data/groupby_out4.tabular
deleted file mode 100644
index 3a9115ca22c..00000000000
--- a/test-data/groupby_out4.tabular
+++ /dev/null
@@ -1,3 +0,0 @@
- 1600
-chr10 1133.33
-chr22 1500
diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py
index 6dad86e9a2c..fc78a6c6b88 100644
--- a/tools/stats/grouping.py
+++ b/tools/stats/grouping.py
@@ -67,17 +67,12 @@ def main():
if sys.argv[5] != "None":
asciitodelete = sys.argv[5]
if asciitodelete:
- oldfile = open(inputfile, 'r')
newinputfile = "input_cleaned.tsv"
- newfile = open(newinputfile, 'w')
- asciitodelete = asciitodelete.split(',')
- for i in range(len(asciitodelete)):
- asciitodelete[i] = chr(int(asciitodelete[i]))
- for line in oldfile:
- if line[0] not in asciitodelete:
- newfile.write(line)
- oldfile.close()
- newfile.close()
+ with open(inputfile, 'r') as oldfile, open(newinputfile, 'w') as newfile:
+ asciitodelete = {chr(int(_)) for _ in asciitodelete.split(',')}
+ for line in oldfile:
+ if line[0] not in asciitodelete:
+ newfile.write(line)
inputfile = newinputfile
# get operations and options in separate arrays
@@ -115,17 +110,18 @@ def main():
case = ''
if ignorecase == 1:
case = '-f'
- command_line = "sort -t $'\\t' %s -k%s,%s -o %s %s" % (case, group_col + 1, group_col + 1, tmpfile.name, inputfile)
+ group_col_str = str(group_col + 1)
+ command_line = ["sort", "-t", "\t", "-k%s,%s" % (group_col_str, group_col_str), "-o", tmpfile.name, inputfile]
+ if case:
+ command_line.append(case)
except Exception as exc:
stop_err('Initialization error -> %s' % str(exc))
try:
- subprocess.check_output(command_line, stderr=subprocess.STDOUT, shell=True)
+ subprocess.check_output(command_line, stderr=subprocess.STDOUT)
except subprocess.CalledProcessError as e:
stop_err("Sorting input dataset resulted in error: %s: %s" % (e.returncode, e.output))
- fout = open(sys.argv[1], "w")
-
def is_new_item(line):
try:
item = line.split("\t")[group_col]
@@ -136,52 +132,56 @@ def main():
return item.lower()
return item
- for key, line_list in groupby(tmpfile, key=is_new_item):
- op_vals = [[] for _ in ops]
- out_str = key
+ with open(sys.argv[1], "w") as fout:
- for line in line_list:
- fields = line.split("\t")
- for i, col in enumerate(cols):
- col = int(col) - 1 # cXX from galaxy is 1-based
- try:
- val = fields[col].strip()
- op_vals[i].append(val)
- except IndexError:
- sys.stderr.write('Could not access the value for column %s on line: "%s". Make sure file is tab-delimited.\n' % (col + 1, line))
- sys.exit(1)
+ for key, line_list in groupby(tmpfile, key=is_new_item):
+ op_vals = [[] for _ in ops]
+ out_str = key
- # Generate string for each op for this group
- for i, op in enumerate(ops):
- data = op_vals[i]
- rval = ""
- if op == "mode":
- rval = mode(data)
- elif op == "length":
- rval = len(data)
- elif op == "random":
- rval = random.choice(data)
- elif op in ['cat', 'cat_uniq']:
- if op == 'cat_uniq':
- data = numpy.unique(data)
- rval = ','.join(data)
- elif op == "unique":
- rval = len(numpy.unique(data))
- else:
- # some kind of numpy fn
- try:
- data = float_wdefault(data, default_val[i], col + 1)
- except ValueError:
- sys.stderr.write("Operation %s expected number values but got %s instead.\n" % (op, data))
- sys.exit(1)
- rval = getattr(numpy, op)(data)
- if round_val[i] == 'yes':
- rval = int(round(rval))
+ for line in line_list:
+ fields = line.split("\t")
+ for i, col in enumerate(cols):
+ col = int(col) - 1 # cXX from galaxy is 1-based
+ try:
+ val = fields[col].strip()
+ op_vals[i].append(val)
+ except IndexError:
+ sys.stderr.write('Could not access the value for column %s on line: "%s". Make sure file is tab-delimited.\n' % (col + 1, line))
+ sys.exit(1)
+
+ # Generate string for each op for this group
+ for i, op in enumerate(ops):
+ data = op_vals[i]
+ rval = ""
+ if op == "mode":
+ rval = mode(data)
+ elif op == "length":
+ rval = len(data)
+ elif op == "random":
+ rval = random.choice(data)
+ elif op in ['cat', 'cat_uniq']:
+ if op == 'cat_uniq':
+ data = numpy.unique(data)
+ rval = ','.join(data)
+ elif op == "unique":
+ rval = len(numpy.unique(data))
else:
- rval = '%g' % rval
- out_str += "\t%s" % rval
+ # some kind of numpy fn
+ try:
+ data = float_wdefault(data, default_val[i], col + 1)
+ except ValueError:
+ sys.stderr.write("Operation %s expected number values but got %s instead.\n" % (op, data))
+ sys.exit(1)
+ rval = getattr(numpy, op)(data)
+ if round_val[i] == 'yes':
+ rval = int(round(rval))
+ else:
+ rval = '%g' % rval
+ out_str += "\t%s" % rval
- fout.write(out_str + "\n")
+ fout.write(out_str + "\n")
+
+ tmpfile.close()
# Generate a useful info message.
msg = "--Group by c%d: " % (group_col + 1)
@@ -200,8 +200,6 @@ def main():
msg += op + "[c" + cols[i] + "] "
print(msg)
- fout.close()
- tmpfile.close()
if __name__ == "__main__":
diff --git a/tools/stats/grouping.xml b/tools/stats/grouping.xml
index 9546ce139ae..ce7b6dd826c 100644
--- a/tools/stats/grouping.xml
+++ b/tools/stats/grouping.xml
@@ -2,6 +2,7 @@
data by a column and perform aggregate operation on other columns.
numpy
+ coreutils
python '$__tool_directory__/grouping.py'
@@ -79,7 +80,7 @@
-
+
@@ -170,7 +171,7 @@
-
+