diff --git a/test-data/groupby_out1.dat b/test-data/groupby_out1.dat deleted file mode 100644 index f822b20ca1a..00000000000 --- a/test-data/groupby_out1.dat +++ /dev/null @@ -1,20 +0,0 @@ -chr1 1.48053e+08 -chr10 5.52516e+07 -chr11 8.75888e+07 -chr12 3.84401e+07 -chr13 1.12382e+08 -chr14 9.87102e+07 -chr15 4.16664e+07 -chr16 206638 -chr18 5.05624e+07 -chr19 5.92262e+07 -chr2 1.69276e+08 -chr20 3.35042e+07 -chr21 3.31607e+07 -chr22 3.04712e+07 -chr5 1.31612e+08 -chr6 1.08564e+08 -chr7 1.15958e+08 -chr8 1.18881e+08 -chr9 1.28843e+08 -chrx 1.45195e+08 diff --git a/test-data/groupby_out2.dat b/test-data/groupby_out2.dat deleted file mode 100644 index f89d723ab44..00000000000 --- a/test-data/groupby_out2.dat +++ /dev/null @@ -1,3 +0,0 @@ - -chr10 chr10 -chr22 chr22 diff --git a/test-data/groupby_out3.tabular b/test-data/groupby_out3.tabular deleted file mode 100644 index cc5214529db..00000000000 --- a/test-data/groupby_out3.tabular +++ /dev/null @@ -1,20 +0,0 @@ -chr1 1.48053e+08 1.48031e+08 -,+ 1.48185e+08 1.47962e+08 5.9221e+08 4 1 chr1,chr1,chr1,chr1 +,- 148185136 88085.9 -chr10 5.52516e+07 5.52516e+07 - 5.52516e+07 5.52516e+07 5.52516e+07 1 1 chr10 - 55251623 0 -chr11 8.75888e+07 1.16165e+08 -,+ 1.16212e+08 1.81238e+06 3.50355e+08 4 1 chr11,chr11,chr11,chr11 +,- 1812377 4.9523e+07 -chr12 3.84401e+07 3.84401e+07 - 3.84401e+07 3.84401e+07 3.84401e+07 1 1 chr12 - 38440094 0 -chr13 1.12382e+08 1.12382e+08 + 1.12382e+08 1.12382e+08 1.12382e+08 1 1 chr13 + 112381694 0 -chr14 9.87102e+07 9.87102e+07 - 9.87102e+07 9.87102e+07 9.87102e+07 1 1 chr14 - 98710240 0 -chr15 4.16664e+07 4.16764e+07 -,+ 4.1826e+07 4.14869e+07 1.66666e+08 4 1 chr15,chr15,chr15,chr15 +,- 41673708 120341 -chr16 206638 212188 +,- 259268 142908 826552 4 1 chr16,chr16,chr16,chr16 +,- 244413 47354.9 -chr18 5.05624e+07 5.94314e+07 -,+ 5.96006e+07 2.37861e+07 2.0225e+08 4 1 chr18,chr18,chr18,chr18 +,- 23786114 1.54594e+07 -chr19 5.92262e+07 5.9267e+07 +,- 5.93022e+07 5.90686e+07 2.36905e+08 4 1 chr19,chr19,chr19,chr19 +,- 59236026 94686.3 -chr2 1.69276e+08 1.69292e+08 +,- 2.2023e+08 1.18289e+08 6.77103e+08 4 1 chr2,chr2,chr2,chr2 +,- 220190202 5.09343e+07 -chr20 3.35042e+07 3.35466e+07 -,+ 3.35933e+07 3.33304e+07 1.34017e+08 4 1 chr20,chr20,chr20,chr20 +,- 33579500 104752 -chr21 3.31607e+07 3.30953e+07 +,- 3.3745e+07 3.2707e+07 1.32643e+08 4 1 chr21,chr21,chr21,chr21 +,- 33321040 405475 -chr22 3.04712e+07 3.04128e+07 +,- 3.09391e+07 3.01202e+07 1.21885e+08 4 1 chr22,chr22,chr22,chr22 +,- 30120223 345080 -chr5 1.31612e+08 1.31589e+08 +,- 1.31848e+08 1.31424e+08 5.2645e+08 4 1 chr5,chr5,chr5,chr5 +,- 131556601 153188 -chr6 1.08564e+08 1.08617e+08 -,+ 1.08723e+08 1.083e+08 4.34257e+08 4 1 chr6,chr6,chr6,chr6 +,- 108640045 159611 -chr7 1.15958e+08 1.16613e+08 +,- 1.16946e+08 1.13661e+08 4.63832e+08 4 1 chr7,chr7,chr7,chr7 +,- 113660517 1.33533e+06 -chr8 1.18881e+08 1.18881e+08 - 1.18881e+08 1.18881e+08 1.18881e+08 1 1 chr8 - 118881131 0 -chr9 1.28843e+08 1.28835e+08 +,- 1.28937e+08 1.28764e+08 5.15371e+08 4 1 chr9,chr9,chr9,chr9 +,- 128787519 70228.2 -chrX 1.45195e+08 1.5267e+08 +,- 1.52694e+08 1.22745e+08 5.80779e+08 4 1 chrX,chrX,chrX,chrX +,- 152648964 1.29614e+07 diff --git a/test-data/groupby_out4.tabular b/test-data/groupby_out4.tabular deleted file mode 100644 index 3a9115ca22c..00000000000 --- a/test-data/groupby_out4.tabular +++ /dev/null @@ -1,3 +0,0 @@ - 1600 -chr10 1133.33 -chr22 1500 diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py index 6dad86e9a2c..fc78a6c6b88 100644 --- a/tools/stats/grouping.py +++ b/tools/stats/grouping.py @@ -67,17 +67,12 @@ def main(): if sys.argv[5] != "None": asciitodelete = sys.argv[5] if asciitodelete: - oldfile = open(inputfile, 'r') newinputfile = "input_cleaned.tsv" - newfile = open(newinputfile, 'w') - asciitodelete = asciitodelete.split(',') - for i in range(len(asciitodelete)): - asciitodelete[i] = chr(int(asciitodelete[i])) - for line in oldfile: - if line[0] not in asciitodelete: - newfile.write(line) - oldfile.close() - newfile.close() + with open(inputfile, 'r') as oldfile, open(newinputfile, 'w') as newfile: + asciitodelete = {chr(int(_)) for _ in asciitodelete.split(',')} + for line in oldfile: + if line[0] not in asciitodelete: + newfile.write(line) inputfile = newinputfile # get operations and options in separate arrays @@ -115,17 +110,18 @@ def main(): case = '' if ignorecase == 1: case = '-f' - command_line = "sort -t $'\\t' %s -k%s,%s -o %s %s" % (case, group_col + 1, group_col + 1, tmpfile.name, inputfile) + group_col_str = str(group_col + 1) + command_line = ["sort", "-t", "\t", "-k%s,%s" % (group_col_str, group_col_str), "-o", tmpfile.name, inputfile] + if case: + command_line.append(case) except Exception as exc: stop_err('Initialization error -> %s' % str(exc)) try: - subprocess.check_output(command_line, stderr=subprocess.STDOUT, shell=True) + subprocess.check_output(command_line, stderr=subprocess.STDOUT) except subprocess.CalledProcessError as e: stop_err("Sorting input dataset resulted in error: %s: %s" % (e.returncode, e.output)) - fout = open(sys.argv[1], "w") - def is_new_item(line): try: item = line.split("\t")[group_col] @@ -136,52 +132,56 @@ def main(): return item.lower() return item - for key, line_list in groupby(tmpfile, key=is_new_item): - op_vals = [[] for _ in ops] - out_str = key + with open(sys.argv[1], "w") as fout: - for line in line_list: - fields = line.split("\t") - for i, col in enumerate(cols): - col = int(col) - 1 # cXX from galaxy is 1-based - try: - val = fields[col].strip() - op_vals[i].append(val) - except IndexError: - sys.stderr.write('Could not access the value for column %s on line: "%s". Make sure file is tab-delimited.\n' % (col + 1, line)) - sys.exit(1) + for key, line_list in groupby(tmpfile, key=is_new_item): + op_vals = [[] for _ in ops] + out_str = key - # Generate string for each op for this group - for i, op in enumerate(ops): - data = op_vals[i] - rval = "" - if op == "mode": - rval = mode(data) - elif op == "length": - rval = len(data) - elif op == "random": - rval = random.choice(data) - elif op in ['cat', 'cat_uniq']: - if op == 'cat_uniq': - data = numpy.unique(data) - rval = ','.join(data) - elif op == "unique": - rval = len(numpy.unique(data)) - else: - # some kind of numpy fn - try: - data = float_wdefault(data, default_val[i], col + 1) - except ValueError: - sys.stderr.write("Operation %s expected number values but got %s instead.\n" % (op, data)) - sys.exit(1) - rval = getattr(numpy, op)(data) - if round_val[i] == 'yes': - rval = int(round(rval)) + for line in line_list: + fields = line.split("\t") + for i, col in enumerate(cols): + col = int(col) - 1 # cXX from galaxy is 1-based + try: + val = fields[col].strip() + op_vals[i].append(val) + except IndexError: + sys.stderr.write('Could not access the value for column %s on line: "%s". Make sure file is tab-delimited.\n' % (col + 1, line)) + sys.exit(1) + + # Generate string for each op for this group + for i, op in enumerate(ops): + data = op_vals[i] + rval = "" + if op == "mode": + rval = mode(data) + elif op == "length": + rval = len(data) + elif op == "random": + rval = random.choice(data) + elif op in ['cat', 'cat_uniq']: + if op == 'cat_uniq': + data = numpy.unique(data) + rval = ','.join(data) + elif op == "unique": + rval = len(numpy.unique(data)) else: - rval = '%g' % rval - out_str += "\t%s" % rval + # some kind of numpy fn + try: + data = float_wdefault(data, default_val[i], col + 1) + except ValueError: + sys.stderr.write("Operation %s expected number values but got %s instead.\n" % (op, data)) + sys.exit(1) + rval = getattr(numpy, op)(data) + if round_val[i] == 'yes': + rval = int(round(rval)) + else: + rval = '%g' % rval + out_str += "\t%s" % rval - fout.write(out_str + "\n") + fout.write(out_str + "\n") + + tmpfile.close() # Generate a useful info message. msg = "--Group by c%d: " % (group_col + 1) @@ -200,8 +200,6 @@ def main(): msg += op + "[c" + cols[i] + "] " print(msg) - fout.close() - tmpfile.close() if __name__ == "__main__": diff --git a/tools/stats/grouping.xml b/tools/stats/grouping.xml index 9546ce139ae..ce7b6dd826c 100644 --- a/tools/stats/grouping.xml +++ b/tools/stats/grouping.xml @@ -2,6 +2,7 @@ data by a column and perform aggregate operation on other columns. numpy + coreutils python '$__tool_directory__/grouping.py' @@ -79,7 +80,7 @@ - + @@ -170,7 +171,7 @@ - +