diff --git a/datatype_converters_conf.xml.sample b/datatype_converters_conf.xml.sample index dc2aff53547..d69672c7743 100644 --- a/datatype_converters_conf.xml.sample +++ b/datatype_converters_conf.xml.sample @@ -1,8 +1,9 @@ - - + + + diff --git a/lib/galaxy/datatypes/converters/fasta_to_tabular_converter.py b/lib/galaxy/datatypes/converters/fasta_to_tabular_converter.py new file mode 100644 index 00000000000..7cd6eeb137e --- /dev/null +++ b/lib/galaxy/datatypes/converters/fasta_to_tabular_converter.py @@ -0,0 +1,43 @@ +#! /usr/bin/python +# This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools +""" +Input: fasta, minimal length, maximal length +Output: fasta +Return sequences whose lengths are within the range. +Wen-Yu Chung +""" + +import sys, os + +seq_hash = {} + +def __main__(): + infile = sys.argv[1] + outfile = sys.argv[2] + title = '' + sequence = '' + sequence_count = 0 + for i, line in enumerate( open( infile ) ): + line = line.rstrip( '\r\n' ) + if line.startswith( '>' ): + if sequence: + sequence_count += 1 + seq_hash[( sequence_count, title )] = sequence + title = line + sequence = '' + else: + sequence += line + if line.split()[0].isdigit(): + sequence += ' ' + if sequence: + seq_hash[( sequence_count, title )] = sequence + # return only those lengths are in the range + out = open( outfile, 'w' ) + title_keys = seq_hash.keys() + title_keys.sort() + for i, fasta_title in title_keys: + sequence = seq_hash[( i, fasta_title )] + print >> out, "%s\t%s" %( fasta_title, sequence ) + out.close() + +if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/lib/galaxy/datatypes/converters/fasta_to_tabular_converter.xml b/lib/galaxy/datatypes/converters/fasta_to_tabular_converter.xml new file mode 100644 index 00000000000..ca3519edc31 --- /dev/null +++ b/lib/galaxy/datatypes/converters/fasta_to_tabular_converter.xml @@ -0,0 +1,13 @@ + + + + fasta_to_tabular_converter.py $input $output + + + + + + + + + \ No newline at end of file diff --git a/tool_conf.xml.main b/tool_conf.xml.main index 83bb9678d3a..7016b8f15a8 100644 --- a/tool_conf.xml.main +++ b/tool_conf.xml.main @@ -69,8 +69,11 @@ -->
+ +
diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index 8a929b0354f..cd6365d6241 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -55,14 +55,16 @@
- - - - - + + + + + + +
@@ -303,8 +305,6 @@
- -
diff --git a/tools/fasta_tools/fasta_to_tabular.py b/tools/fasta_tools/fasta_to_tabular.py new file mode 100644 index 00000000000..7cd6eeb137e --- /dev/null +++ b/tools/fasta_tools/fasta_to_tabular.py @@ -0,0 +1,43 @@ +#! /usr/bin/python +# This code exists in 2 places: ~/datatypes/converters and ~/tools/fasta_tools +""" +Input: fasta, minimal length, maximal length +Output: fasta +Return sequences whose lengths are within the range. +Wen-Yu Chung +""" + +import sys, os + +seq_hash = {} + +def __main__(): + infile = sys.argv[1] + outfile = sys.argv[2] + title = '' + sequence = '' + sequence_count = 0 + for i, line in enumerate( open( infile ) ): + line = line.rstrip( '\r\n' ) + if line.startswith( '>' ): + if sequence: + sequence_count += 1 + seq_hash[( sequence_count, title )] = sequence + title = line + sequence = '' + else: + sequence += line + if line.split()[0].isdigit(): + sequence += ' ' + if sequence: + seq_hash[( sequence_count, title )] = sequence + # return only those lengths are in the range + out = open( outfile, 'w' ) + title_keys = seq_hash.keys() + title_keys.sort() + for i, fasta_title in title_keys: + sequence = seq_hash[( i, fasta_title )] + print >> out, "%s\t%s" %( fasta_title, sequence ) + out.close() + +if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_tool_fasta2tab.xml b/tools/fasta_tools/fasta_to_tabular.xml similarity index 74% rename from tools/fasta_tools/fasta_tool_fasta2tab.xml rename to tools/fasta_tools/fasta_to_tabular.xml index 13907391f03..06754d17ebd 100644 --- a/tools/fasta_tools/fasta_tool_fasta2tab.xml +++ b/tools/fasta_tools/fasta_to_tabular.xml @@ -1,6 +1,6 @@ - - Converts a FASTA formated file to TAB format - fasta_tool_fasta2tab.py $input $output + + Converts a FASTA file to Tabular format + fasta_to_tabular.py $input $output @@ -10,11 +10,11 @@ - + - - + + @@ -30,7 +30,7 @@ A fasta file with two sequences:: GGTGACATCGCCCACCACGGTACTCACTGGCTGGCTCTGGTTCCCGGCGGCATCGGAGGC CACCACGTTGAGGGTATTCCCCTCGGTTTGTGGCTCGGTGAGAACCACGTTGTAGTCGCC ATTGGTC - + Returns:: >EYKX4VC01B65GS length=54 xy=0784_1754 region=1 run=R_2007_11_07_16_15_57_ CCGGTATCCGGGTGCCGTGATGAGCGCCACCGGAACGAATTCGACTATGCCGAA diff --git a/tools/fasta_tools/fasta_tool_fasta2tab.py b/tools/fasta_tools/fasta_tool_fasta2tab.py deleted file mode 100644 index 04dc62da5d7..00000000000 --- a/tools/fasta_tools/fasta_tool_fasta2tab.py +++ /dev/null @@ -1,58 +0,0 @@ -#! /usr/bin/python -""" -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. -Wen-Yu Chung -""" - -import sys, os - -seq_hash = {} - -def parse_fasta_format(file_handle): - # detect whether it's score or seq files - # return a hash: key = title and value = seq - - # use seq_hash #tmp_hash = {} - tmp_title = '' - tmp_seq = '' - tmp_seq_count = 0 - for i, each_line in enumerate(file_handle): - each_line = each_line.strip('\r\n') - if (each_line[0] == '>'): - if (len(tmp_seq) > 0): - tmp_seq_count += 1 - seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq - tmp_title = each_line - tmp_seq = '' - else: - tmp_seq = tmp_seq + each_line - if (each_line.split()[0].isdigit()): - tmp_seq = tmp_seq + ' ' - if (len(tmp_seq) > 0): - seq_hash[(tmp_seq_count, tmp_title)] = tmp_seq - - return 0 - -def __main__(): - - input_filename = sys.argv[1] - output_filename = sys.argv[2] - - # parse fasta format into a hast table - input_handle = open(input_filename, 'r') - parse_fasta_format(input_handle) - input_handle.close() - - # return only those lengths are in the range - output_handle = open(output_filename, 'w') - title_keys = seq_hash.keys() - title_keys.sort() - for (i, fasta_title) in title_keys: - tmp_seq = seq_hash[(i, fasta_title)] - print >> output_handle, "%s\t%s" %(fasta_title, tmp_seq) - output_handle.close() - return 0 - -if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_tool_tab2fasta.py b/tools/fasta_tools/fasta_tool_tab2fasta.py deleted file mode 100644 index 5079ddbf1f3..00000000000 --- a/tools/fasta_tools/fasta_tool_tab2fasta.py +++ /dev/null @@ -1,59 +0,0 @@ -#! /usr/bin/python -""" -Input: fasta, minimal length, maximal length -Output: fasta -Return sequences whose lengths are within the range. -Wen-Yu Chung -""" - -import sys, os - -seq_hash = {} - -def __main__(): - - input_filename = sys.argv[1] - if (sys.argv[2] != ""): input_titleCol = sys.argv[2] - else: print >> sys.stderr, "Please indicate columns for the titles and sequences." - input_seqCol = int(sys.argv[3]) - 1 - output_filename = sys.argv[4] - - if (input_seqCol < 0): print >> sys.stderr, "Column index starts from 1." - - titleCols = input_titleCol.split(',') - - # print input_titleCol - # print input_seqCol - - input_handle = open(input_filename, 'r') - output_handle = open(output_filename, 'w') - - # need to check which column is the title, which is the sequence - for i, line in enumerate(input_handle): - line = line.strip('\r\n') - fields = line.split() - total_columns = len(fields) - - if (input_seqCol > total_columns): print >> sys.stderr, "Sequence column does not exist." - - fasta_title = [] - for j in titleCols: - if (j.isdigit()): - correct_index = int(j) - 1 - if (correct_index < total_columns) and (correct_index >= 0): - fasta_title.append(fields[correct_index]) - else: - print >> sys.stderr, "Title column does not exist. Column index starts from 1, not 0." - else: - print >> sys.stderr, "Title column is not an index." - - fasta_seq = fields[input_seqCol] - if (fasta_title[0].startswith(">")): fasta_title[0]=fasta_title[0][1:] - print >> output_handle, ">%s\n%s" %("_".join(fasta_title), fasta_seq) - - output_handle.close() - input_handle.close() - - return 0 - -if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/tabular_to_fasta.py b/tools/fasta_tools/tabular_to_fasta.py new file mode 100644 index 00000000000..0e849037eab --- /dev/null +++ b/tools/fasta_tools/tabular_to_fasta.py @@ -0,0 +1,46 @@ +#! /usr/bin/python +""" +Input: fasta, minimal length, maximal length +Output: fasta +Return sequences whose lengths are within the range. +Wen-Yu Chung +""" + +import sys, os + +seq_hash = {} + +def stop_err( msg ): + sys.stderr.write( msg ) + sys.exit() + +def __main__(): + infile = sys.argv[1] + title_col = sys.argv[2] + seq_col = sys.argv[3] + outfile = sys.argv[4] + + if title_col == None or title_col == 'None' or seq_col == None or seq_col == 'None': + stop_err( "Columns not specified." ) + try: + seq_col = int( seq_col ) - 1 + except: + stop_err( "Invalid Sequence Column: %s." %seq_col ) + + title_col_list = title_col.split( ',' ) + out = open( outfile, 'w' ) + + for i, line in enumerate( open( infile ) ): + line = line.strip( '\r\n' ) + fields = line.split() + fasta_title = [] + for j in title_col_list: + j = int( j ) - 1 + fasta_title.append( fields[j] ) + fasta_seq = fields[seq_col] + if ( fasta_title[0].startswith(">") ): + fasta_title[0] = fasta_title[0][1:] + print >> out, ">%s\n%s" % ( "_".join( fasta_title ), fasta_seq ) + out.close() + +if __name__ == "__main__" : __main__() \ No newline at end of file diff --git a/tools/fasta_tools/fasta_tool_tab2fasta.xml b/tools/fasta_tools/tabular_to_fasta.xml similarity index 53% rename from tools/fasta_tools/fasta_tool_tab2fasta.xml rename to tools/fasta_tools/tabular_to_fasta.xml index a2f6a0f1c1f..c7811fc8a4d 100644 --- a/tools/fasta_tools/fasta_tool_tab2fasta.xml +++ b/tools/fasta_tools/tabular_to_fasta.xml @@ -1,10 +1,10 @@ - - Converts a TAB formated file to FASTA format - fasta_tool_tab2fasta.py $input $input_titleCol $input_seqCol $output + + Converts a tabular file to FASTA format + tabular_to_fasta.py $input $title_col $seq_col $output - - + + @@ -12,9 +12,9 @@ - - - + + + @@ -22,15 +22,14 @@ **Example** Solexa data:: - 5 300 902 419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC 5 300 880 431 GTGATATGTATGTTGACGGCCATAAGGCTGCTTCTT 5 300 896 461 GTTGTCGATAGAACTTCATGTGCCTGTAAAACAAGT 5 300 890 751 ACCAACCAGAACGTGAAAAAGCGTCCTGCGTGTAGC 5 300 897 443 GTTTATGTTGGTTTCATGGTTTTGTCTAACTTTATC 5 300 906 879 GCTTTACCGTCTTTCCAGAAATTGTTCCAAGTATCG - -Choose **columns 3 and 4** (3,4) to be the fasta title, the tool returns:: + +Selecting **c3 and c4** as the Title Columns and **c5** as the Sequence Column will result in:: >902_419 GACTCATGATTTCTTACCTATTAGTGGTTGAACATC