Bug fixes, miscellaneous cleanup for flanking_features, windowSplitter, quality_filter, featureCounter, get_flanks.

This commit is contained in:
Greg Von Kuster
2008-05-27 14:28:49 +00:00
parent fab2f2dcca
commit 8cf64e6a52
5 changed files with 50 additions and 86 deletions
@@ -84,8 +84,6 @@ def proximal_region_finder(readers, region, comments=True):
start = int(interval.start)
end = int(interval.end)
strand = interval.strand
if start > end:
warn( "Interval start after end!" )
if chrom not in rightTree.chroms:
continue
else:
+22 -32
View File
@@ -40,21 +40,15 @@ def main():
size = int(size)
except:
stop_err( "Invalid offset or length entered. Try again by entering valid integer values." )
try:
fi = open(inp_file,'r')
except:
stop_err( "Unable to open input file" )
try:
fo = open(out_file,'w')
except:
stop_err( "Unable to open output file" )
fo = open(out_file,'w')
skipped_lines = 0
first_invalid_line = 0
invalid_line = None
elems = []
j=0
for i, line in enumerate( fi ):
for i, line in enumerate( file( inp_file ) ):
line = line.strip()
if line and (not line.startswith( '#' )) and line != '':
j+=1
@@ -83,7 +77,7 @@ def main():
elems[start_col_1] = str(int(elems[end_col_1]) - offset)
elems[end_col_1] = str(int(elems[start_col_1]) + size)
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elif direction == 'Downstream':
if strand == '-':
@@ -101,7 +95,7 @@ def main():
elems[start_col_1] = str(int(elems[end_col_1]) + offset)
elems[end_col_1] = str(int(elems[start_col_1]) + size)
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elif direction == 'Both':
if strand == '-':
@@ -112,11 +106,11 @@ def main():
elems[start_col_1]=start
elems[end_col_1]=end1
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elems[start_col_1]=end2
elems[end_col_1]=start
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elif region == 'end':
start = str(int(elems[start_col_1]) - offset)
end1 = str(int(start) + size)
@@ -124,11 +118,11 @@ def main():
elems[start_col_1]=start
elems[end_col_1]=end1
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elems[start_col_1]=end2
elems[end_col_1]=start
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
else:
start1 = str(int(elems[end_col_1]) - offset)
end1 = str(int(start1) + size)
@@ -137,11 +131,11 @@ def main():
elems[start_col_1]=start1
elems[end_col_1]=end1
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elems[start_col_1]=end2
elems[end_col_1]=start2
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elif strand == '+':
if region == 'start':
start = str(int(elems[start_col_1]) + offset)
@@ -150,11 +144,11 @@ def main():
elems[start_col_1]=end1
elems[end_col_1]=start
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elems[start_col_1]=start
elems[end_col_1]=end2
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elif region == 'end':
start = str(int(elems[end_col_1]) + offset)
end1 = str(int(start) - size)
@@ -162,11 +156,11 @@ def main():
elems[start_col_1]=end1
elems[end_col_1]=start
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elems[start_col_1]=start
elems[end_col_1]=end2
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
else:
start1 = str(int(elems[start_col_1]) + offset)
end1 = str(int(start1) - size)
@@ -175,27 +169,23 @@ def main():
elems[start_col_1]=end1
elems[end_col_1]=start1
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
elems[start_col_1]=start2
elems[end_col_1]=end2
assert int(elems[start_col_1]) > 0 and int(elems[end_col_1]) > 0
print >>fo, '\t'.join(elems)
fo.write( "%s\n" % '\t'.join( elems ) )
except:
skipped_lines += 1
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
fo.close()
fi.close()
#If number of skipped lines = num of lines in the file, inform the user to check metadata attributes of the input file.
if skipped_lines == j:
print 'Data issue: Skipped all lines in your input. Check the metadata attributes of the chosen input by clicking on the pencil icon next to it.'
sys.exit()
elif skipped_lines > 0:
print '(Data issue: skipped %d invalid lines starting at line #%d which is "%s")' % ( skipped_lines, first_invalid_line, invalid_line )
print 'Location : %s, Region : %s, Flank-length : %d, Offset : %d ' %(direction, region, size, offset)
stop_err( "Data issue: click the pencil icon in the history item to correct the metadata attributes." )
if skipped_lines > 0:
print 'Skipped %d invalid lines starting with #%dL "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
print 'Location: %s, Region: %s, Flank-length: %d, Offset: %d ' %( direction, region, size, offset )
if __name__ == "__main__":
main()
+3 -4
View File
@@ -48,7 +48,7 @@ def counter(node, start, end):
if node.left:
counter(node.left, start, end)
def count_coverage(readers):
def count_coverage( readers, comments=True ):
primary = readers[0]
secondary = readers[1]
secondary_copy = readers[2]
@@ -71,7 +71,6 @@ def count_coverage(readers):
chrom = interval.chrom
start = int(interval.start)
end = int(interval.end)
if start > end: warn( "Interval start after end!" )
full = 0
partial = 0
if chrom not in bitsets:
@@ -129,9 +128,9 @@ def main():
try:
for line in count_coverage([g1,g2,g2_copy]):
if type( line ) is GenomicInterval:
print >> out_file, "\t".join( line.fields )
out_file.write( "%s\n" % "\t".join( line.fields ) )
else:
print >> out_file, line
out_file.write( "%s\n" % line )
except ParseError, exc:
out_file.close()
fail( str( exc ) )
+10 -27
View File
@@ -93,32 +93,17 @@ def main():
mask_length_r = int(mask_length.split(',')[0])
mask_length_l = int(mask_length.split(',')[1])
except:
print >> sys.stderr, "Data issue: Please check the metadata attributes of the chosen input by clicking on the pencil icon next to it."
sys.exit()
try:
fi = open(inp_file,'r')
except:
print >> sys.stderr, "Unable to open input file"
sys.exit()
try:
fo = open(out_file,'w')
except:
print >> sys.stderr, "Unable to open output file"
sys.exit()
stop_err( "Data issue, click the pencil icon in the history item to correct the metadata attributes of the input dataset." )
if pri_species == 'None':
print >>sys.stderr, "No primary species selected. Try again by selecting at least one primary species."
sys.exit()
stop_err( "No primary species selected, try again by selecting at least one primary species." )
if mask_species == 'None':
print >>sys.stderr, "No mask species selected. Try again by selecting at least one species to mask."
sys.exit()
stop_err( "No mask species selected, try again by selecting at least one species to mask." )
mask_chr_count = 0
mask_chr_dict = {0:'#', 1:'$', 2:'^', 3:'*', 4:'?'}
mask_reg_dict = {0:'Current pos', 1:'Current+Downstream', 2:'Current+Upstream', 3:'Current+Both sides'}
#ensure dbkey is present in the twobit loc file
filepath = None
try:
@@ -144,13 +129,12 @@ def main():
except:
pass
except Exception, exc:
print >>sys.stdout, 'quality_filter.py initialization error -> %s' % exc
stop_err( 'Initialization errorL %s' % str( exc ) )
if len(pspecies) == 0:
print >>sys.stderr, "Quality scores are not available for the following genome builds: %s" %(pspecies_all2)
sys.exit()
stop_err( "Quality scores are not available for the following genome builds: %s" % ( pspecies_all2 ) )
if len(pspecies) < len(pspecies_all):
print >>sys.stdout, "Quality scores are not available for the following genome builds: %s" %(pspecies_all2)
print "Quality scores are not available for the following genome builds: %s" %(pspecies_all2)
scores_by_chrom = []
#Get scores for all the primary species
@@ -160,9 +144,8 @@ def main():
try:
maf_reader = bx.align.maf.Reader( open(inp_file, 'r') )
maf_writer = bx.align.maf.Writer( open(out_file,'w') )
except:
print >>sys.stderr, "Your MAF file appears to be malformed."
sys.exit()
except Exception, e:
stop_err( "Your MAF file appears to be malformed: %s" % str( e ) )
maf_count = 0
for block in maf_reader:
+15 -21
View File
@@ -14,6 +14,10 @@ import pkg_resources; pkg_resources.require( "bx-python" )
from bx.cookbook import doc_optparse
from galaxy.tools.util.galaxyops import *
def stop_err( msg ):
sys.stderr.write( msg )
sys.exit()
def main():
# Parsing Command Line here
options, args = doc_optparse.parse( __doc__ )
@@ -27,27 +31,17 @@ def main():
if strand_col_1 <= 0:
strand = "+" #if strand is not defined, default it to +
except:
print >> sys.stderr, "Data issue: Please check the metadata attributes of the chosen input by clicking on the pencil icon next to it."
sys.exit()
stop_err( "Data issue, click the pencil icon in the history item to correct the metadata attributes of the input dataset." )
try:
fi = open(inp_file,'r')
except:
print >> sys.stderr, "Unable to open input file"
sys.exit()
try:
fo = open(out_file,'w')
except:
print >> sys.stderr, "Unable to open output file"
sys.exit()
fo = open(out_file,'w')
skipped_lines = 0
first_invalid_line = 0
invalid_line = None
if offset == 0:
makesliding = 0
for i, line in enumerate( fi ):
for i, line in enumerate( file( inp_file ) ):
line = line.strip()
if line and line[0:1] != "#":
try:
@@ -65,27 +59,27 @@ def main():
elems_1 = elems
elems_1[start_col_1] = str(start)
elems_1[end_col_1] = str(start + winsize)
print >>fo, '\t'.join(elems_1)
fo.write( "%s\n" % '\t'.join( elems_1 ) )
if makesliding == 0:
start = start + winsize
else:
start = start + offset
if start+winsize > end:
break
except Exception, exc:
print exc
except:
skipped_lines += 1
if not invalid_line:
first_invalid_line = i + 1
invalid_line = line
fo.close()
if makesliding == 1:
print 'Window size = %d, Sliding = Yes, Offset = %d' %(winsize, offset)
print 'Window size=%d, Sliding=Yes, Offset=%d' %(winsize, offset)
else:
print 'Window size = %d, Sliding = No' %(winsize)
print 'Window size=%d, Sliding=No' %(winsize)
if skipped_lines > 0:
print '(Data issue: skipped %d invalid lines starting at line #%d which is "%s")' % ( skipped_lines, first_invalid_line, invalid_line )
print 'Skipped %d invalid lines starting with #%d: "%s"' % ( skipped_lines, first_invalid_line, invalid_line )
if __name__ == "__main__":
main()