From fbcf3e2bbc3c70fd154f713dcfef186406c301e0 Mon Sep 17 00:00:00 2001 From: Nicola Soranzo Date: Tue, 28 Oct 2014 16:57:49 +0100 Subject: [PATCH] Make stripping and condensing optional. --- tools/filters/convert_characters.py | 60 +++++++++++++--------------- tools/filters/convert_characters.xml | 26 ++++++++++-- 2 files changed, 49 insertions(+), 37 deletions(-) diff --git a/tools/filters/convert_characters.py b/tools/filters/convert_characters.py index 76574adc430..e93b2f223be 100644 --- a/tools/filters/convert_characters.py +++ b/tools/filters/convert_characters.py @@ -1,30 +1,18 @@ #!/usr/bin/env python #By, Guruprasad Ananda. +import optparse import re -import sys - -def stop_err(msg): - sys.stderr.write(msg) - sys.exit() - - -def main(): - if len(sys.argv) != 4: - stop_err("usage: convert_characters infile from_char outfile") - - try: - fin = open(sys.argv[1], 'r') - except: - stop_err("Input file cannot be opened for reading.") - - from_char = sys.argv[2] - - try: - fout = open(sys.argv[3], 'w') - except: - stop_err("Output file cannot be opened for writing.") +def __main__(): + parser = optparse.OptionParser() + parser.add_option('--strip', action='store_true', + help='strip leading and trailing whitespaces') + parser.add_option('--condense', action='store_true', + help='condense consecutive delimiters') + (options, args) = parser.parse_args() + if len(args) != 3: + parser.error("usage: convert_characters.py infile from_char outfile") char_dict = { 'T': '\t', @@ -38,20 +26,26 @@ def main(): 'Sc': ';' } # regexp to match 1 or more occurences. - from_ch = char_dict[from_char] + '+' + from_char = args[1] + from_ch = char_dict[from_char] + if options.condense: + from_ch += '+' + skipped = 0 + with open(args[0], 'rU') as fin: + with open(args[2], 'w') as fout: + for line in fin: + if options.strip: + line = line.strip() + else: + line = line.rstrip('\n') + try: + fout.write("%s\n" % (re.sub(from_ch, '\t', line))) + except: + skipped += 1 - for line in fin: - line = line.strip() - try: - fout.write("%s\n" % (re.sub(from_ch, '\t', line))) - except: - skipped += 1 - - fin.close() - fout.close() if skipped: print "Skipped %d lines as invalid." % skipped if __name__ == "__main__": - main() + __main__() diff --git a/tools/filters/convert_characters.xml b/tools/filters/convert_characters.xml index 9507f792d8d..2de7602b9c4 100644 --- a/tools/filters/convert_characters.xml +++ b/tools/filters/convert_characters.xml @@ -1,6 +1,15 @@ delimiters to TAB - convert_characters.py $input $convert_from $out_file1 + +convert_characters.py +#if $strip + --strip +#end if +#if $condense + --condense +#end if +$input $convert_from $out_file1 + @@ -15,19 +24,28 @@ + + + + + + + + + @@ -35,7 +53,7 @@ **What it does** -Converts all delimiters of a specified type into TABs. Consecutive characters are condensed. For example, if columns are separated by 5 spaces they will converted into 1 tab. +Converts all delimiters of a specified type into TABs. Consecutive delimiters can be condensed in a single TAB. ----- @@ -48,12 +66,12 @@ Converts all delimiters of a specified type into TABs. Consecutive characters a chrX|151559494|151559583|NM_018558_exon_1_0_chrX_151559495_f|0|+ chrX|151564643|151564711|NM_018558_exon_2_0_chrX_151564644_f||||0|+ -- Converting all pipe delimiters of the above file to TABs will get:: +- Converting all pipe delimiters of the above file to TABs and condensing delimiters will get:: chrX 151283558 151283724 NM_000808_exon_8_0_chrX_151283559_r 0 - chrX 151370273 151370486 NM_000808_exon_9_0_chrX_151370274_r 0 - chrX 151559494 151559583 NM_018558_exon_1_0_chrX_151559495_f 0 + chrX 151564643 151564711 NM_018558_exon_2_0_chrX_151564644_f 0 + - +