diff --git a/tools/stats/grouping.py b/tools/stats/grouping.py index 7c0e35d83e4..ad5849838ef 100644 --- a/tools/stats/grouping.py +++ b/tools/stats/grouping.py @@ -12,13 +12,13 @@ def stop_err(msg): def main(): inputfile = sys.argv[2] - + ignorecase = int(sys.argv[4]) ops = [] cols = [] rounds = [] elems = [] - for var in sys.argv[4:]: + for var in sys.argv[5:]: ops.append(var.split()[0]) cols.append(var.split()[1]) rounds.append(var.split()[2]) @@ -71,7 +71,10 @@ def main(): we need to add 1 to group_col. if POS2 is not specified, the newer versions of sort will consider the entire line for sorting. To prevent this, we set POS2=POS1. """ - command_line = "sort -f -k " + str(group_col+1) +"," + str(group_col+1) + " -o " + tmpfile.name + " " + inputfile + case = '' + if ignorecase == 1: + case = '-f' + command_line = "sort -t $'\t' " + case + " -k" + str(group_col+1) +"," + str(group_col+1) + " -o " + tmpfile.name + " " + inputfile except Exception, exc: stop_err( 'Initialization error -> %s' %str(exc) ) @@ -95,6 +98,8 @@ def main(): try: fields = line.split("\t") item = fields[group_col] + if ignorecase == 1: + item = item.lower() if prev_item != "": # At this level, we're grouping on values (item and prev_item) in group_col if item == prev_item: diff --git a/tools/stats/grouping.xml b/tools/stats/grouping.xml index 4d9a69336db..9a0bb749cee 100644 --- a/tools/stats/grouping.xml +++ b/tools/stats/grouping.xml @@ -1,10 +1,11 @@ - + data by a column and perform aggregate operation on other columns. grouping.py $out_file1 $input1 $groupcol + $ignorecase #for $op in $operations '${op.optype} ${op.opcol} @@ -14,6 +15,9 @@ + + + @@ -44,6 +48,7 @@ + @@ -54,6 +59,7 @@ + @@ -80,15 +86,22 @@ This tool allows you to group the input dataset by a particular column and perfo - For the following input:: - chr22 1000 NM_17 - chr22 2000 NM_18 - chr10 2200 NM_10 - chr10 1200 NM_11 - chr22 1600 NM_19 + chr22 1000 1003 TTT + chr22 2000 2003 aaa + chr10 2200 2203 TTT + chr10 1200 1203 ttt + chr22 1600 1603 AAA -- running this tool with **Group by column 1**, Operations **Mean on column 2** and **Concatenate on column 3** will return:: +- **Grouping on column 4** while ignoring case, and performing operation **Count on column 1** will return:: - chr10 1700.00 NM_11,NM_10 - chr22 1533.33 NM_17,NM_19,NM_18 + AAA 2 + TTT 3 + +- **Grouping on column 4** while not ignoring case, and performing operation **Count on column 1** will return:: + + aaa 1 + AAA 1 + ttt 1 + TTT 2