diff --git a/tools/extract/liftOver_wrapper.py b/tools/extract/liftOver_wrapper.py index ea8450d5ae8..9ec918ce347 100644 --- a/tools/extract/liftOver_wrapper.py +++ b/tools/extract/liftOver_wrapper.py @@ -34,15 +34,27 @@ def safe_bed_file(infile): out_handle.close() return fname -if len( sys.argv ) != 7: - stop_err( "USAGE: prog input out_file1 out_file2 input_dbkey output_dbkey minMatch" ) +if len( sys.argv ) < 9: + stop_err( "USAGE: prog input out_file1 out_file2 input_dbkey output_dbkey infile_type minMatch multiple " ) infile = sys.argv[1] outfile1 = sys.argv[2] outfile2 = sys.argv[3] in_dbkey = sys.argv[4] mapfilepath = sys.argv[5] -minMatch = sys.argv[6] +infile_type = sys.argv[6] +gff_option = "" +if infile_type == "gff": + gff_option = "-gff " +minMatch = sys.argv[7] +multiple = int(sys.argv[8]) +multiple_option = "" +if multiple: + minChainT = sys.argv[9] + minChainQ = sys.argv[10] + minSizeQ = sys.argv[11] + multiple_option = " -multiple -minChainT=%s -minChainQ=%s -minSizeQ=%s " %(minChainT,minChainQ,minSizeQ) + try: assert float(minMatch) except: @@ -55,7 +67,8 @@ if not os.path.isfile( mapfilepath ): stop_err( "%s mapping is not currently available." % ( mapfilepath.split('/')[-1].split('.')[0] ) ) safe_infile = safe_bed_file(infile) -cmd_line = "liftOver -minMatch=" + str(minMatch) + " " + safe_infile + " " + mapfilepath + " " + outfile1 + " " + outfile2 + " > /dev/null" +cmd_line = "liftOver " + gff_option + "-minMatch=" + str(minMatch) + multiple_option + " " + safe_infile + " " + mapfilepath + " " + outfile1 + " " + outfile2 + " > /dev/null" + try: # have to nest try-except in try-finally to handle 2.4 try: diff --git a/tools/extract/liftOver_wrapper.xml b/tools/extract/liftOver_wrapper.xml index 2615f6aa615..726e5120afc 100644 --- a/tools/extract/liftOver_wrapper.xml +++ b/tools/extract/liftOver_wrapper.xml @@ -1,8 +1,21 @@ - + between assemblies and genomes - liftOver_wrapper.py $input "$out_file1" "$out_file2" $dbkey $to_dbkey $minMatch + + liftOver_wrapper.py + $input + "$out_file1" + "$out_file2" + $dbkey + $to_dbkey + #if isinstance( $input.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gff').__class__) or isinstance( $input.datatype, $__app__.datatypes_registry.get_datatype_by_extension('gtf').__class__): + "gff" + #else: + "interval" + #end if + $minMatch ${multiple.choice} ${multiple.minChainT} ${multiple.minChainQ} ${multiple.minSizeQ} + - + @@ -14,7 +27,23 @@ - + + + + + + + + + + + + + + + + + @@ -37,9 +66,40 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + .. class:: warningmark @@ -48,7 +108,7 @@ Make sure that the genome build of the input dataset is specified (click the pen .. class:: warningmark -This tool will only work on interval datasets with chromosome in column 1, +This tool can work with interval, GFF, and GTF datasets. It requires the interval datasets to have chromosome in column 1, start co-ordinate in column 2 and end co-ordinate in column 3. BED comments and track and browser lines will be ignored, but if other non-interval lines are present the tool will return empty output datasets. @@ -59,7 +119,11 @@ are present the tool will return empty output datasets. **What it does** -This tool converts coordinates and annotations between assemblies and genomes. It produces 2 files, one containing all the mapped coordinates and the other containing the unmapped coordinates, if any. +This tool is based on the LiftOver utility and Chain track from `the UC Santa Cruz Genome Browser`__. + +It converts coordinates and annotations between assemblies and genomes. It produces 2 files, one containing all the mapped coordinates and the other containing the unmapped coordinates, if any. + + .. __: http://genome.ucsc.edu/ -----