%s' % _ for _ in variants ]
print 'for files that are named %s' % ', '.join( poss_names[:-1] ),
if len( poss_names ) > 1:
print 'or %s' % poss_names[-1],
@@ -226,7 +227,7 @@ def __main__():
if cols:
col_values = [ col.strip() for col in cols.split( ',' ) ]
if not col_values or not loc_path:
- stop_err( 'No columns can be found for this data table (%s) in %s' % ( options.data_table, options.data_table_xml ) )
+ raise Exception( 'No columns can be found for this data table (%s) in %s' % ( options.data_table, options.data_table_xml ) )
# get all fasta paths under genome directory
fasta_locs = {}
@@ -234,7 +235,7 @@ def __main__():
genome_subdirs = [ dr for dr in os.listdir( options.genome_dir ) if dr not in exemptions ]
for genome_subdir in genome_subdirs:
possible_names = [ genome_subdir ]
- possible_names.extend( [ '%s%s' % ( genome_subdir, v ) for v in variants ] )
+ possible_names.extend( [ '%s%s' % ( genome_subdir, _ ) for _ in variants ] )
# get paths to all fasta files
for path_to_look_in in paths_to_look_in:
for dirpath, dirnames, filenames in os.walk( path_to_look_in % genome_subdir ):
@@ -257,10 +258,10 @@ def __main__():
if variant_exclusions.keys():
for k in variant_exclusions.keys():
leave_in = '%s%s' % ( genome_subdir, k )
- if fasta_locs.has_key( leave_in ):
+ if leave_in in fasta_locs:
to_remove = [ '%s%s' % ( genome_subdir, k ) for k in variant_exclusions[ k ] ]
for tr in to_remove:
- if fasta_locs.has_key( tr ):
+ if tr in fasta_locs:
del fasta_locs[ tr ]
# output results
@@ -291,10 +292,11 @@ def __main__():
try:
out_line.append( fasta_locs[ fb ][ col ] )
except KeyError:
- stop_err( 'Unexpected column (%s) encountered' % col )
+ raise Exception( 'Unexpected column (%s) encountered' % col )
if out_line:
all_fasta_loc.write( '%s\n' % '\t'.join( out_line ) )
# close up output loc file
all_fasta_loc.close()
-if __name__=='__main__': __main__()
+if __name__ == '__main__':
+ __main__()
diff --git a/scripts/microbes/BeautifulSoup.py b/scripts/microbes/BeautifulSoup.py
index 7efb82929a6..e588a0bd340 100644
--- a/scripts/microbes/BeautifulSoup.py
+++ b/scripts/microbes/BeautifulSoup.py
@@ -24,7 +24,7 @@ if you also install these three packages:
http://cjkpython.i18n.org/
Beautiful Soup defines classes for two main parsing strategies:
-
+
* BeautifulStoneSoup, for parsing XML, SGML, or your domain-specific
language that kind of looks like XML.
@@ -43,6 +43,15 @@ http://www.crummy.com/software/BeautifulSoup/documentation.html
"""
from __future__ import generators
+import codecs
+import re
+import string
+import sys
+import types
+import sgmllib
+from htmlentitydefs import name2codepoint
+from sgmllib import SGMLParser, SGMLParseError
+
__author__ = "Leonard Richardson (crummy.com)"
__contributors__ = ["Sam Ruby (intertwingly.net)",
"the unwitting Mark Pilgrim (diveintomark.org)",
@@ -51,13 +60,6 @@ __version__ = "3.0.3"
__copyright__ = "Copyright (c) 2004-2006 Leonard Richardson"
__license__ = "PSF"
-from sgmllib import SGMLParser, SGMLParseError
-import codecs
-import types
-import re
-import sgmllib
-from htmlentitydefs import name2codepoint
-
# This RE makes Beautiful Soup able to parse XML with namespaces.
sgmllib.tagfind = re.compile('[a-zA-Z][-_.:a-zA-Z0-9]*')
@@ -67,15 +69,15 @@ sgmllib.charref = re.compile('(\d+|x[0-9a-fA-F]+);')
DEFAULT_OUTPUT_ENCODING = "utf-8"
-# First, the classes that represent markup elements.
+# First, the classes that represent markup elements.
class PageElement:
"""Contains the navigational information for some part of the page
(either a tag or a piece of text)"""
def setup(self, parent=None, previous=None):
"""Sets up the initial relations between this element and
- other elements."""
+ other elements."""
self.parent = parent
self.previous = previous
self.next = None
@@ -85,7 +87,7 @@ class PageElement:
self.previousSibling = self.parent.contents[-1]
self.previousSibling.nextSibling = self
- def replaceWith(self, replaceWith):
+ def replaceWith(self, replaceWith):
oldParent = self.parent
myIndex = self.parent.contents.index(self)
if hasattr(replaceWith, 'parent') and replaceWith.parent == self.parent:
@@ -96,20 +98,20 @@ class PageElement:
# means that when we extract it, the index of this
# element will change.
myIndex = myIndex - 1
- self.extract()
+ self.extract()
oldParent.insert(myIndex, replaceWith)
-
+
def extract(self):
- """Destructively rips this element out of the tree."""
+ """Destructively rips this element out of the tree."""
if self.parent:
try:
self.parent.contents.remove(self)
except ValueError:
pass
- #Find the two elements that would be next to each other if
- #this element (and any children) hadn't been parsed. Connect
- #the two.
+ # Find the two elements that would be next to each other if
+ # this element (and any children) hadn't been parsed. Connect
+ # the two.
lastChild = self._lastRecursiveChild()
nextElement = lastChild.next
@@ -120,12 +122,12 @@ class PageElement:
self.previous = None
lastChild.next = None
- self.parent = None
+ self.parent = None
if self.previousSibling:
self.previousSibling.nextSibling = self.nextSibling
if self.nextSibling:
self.nextSibling.previousSibling = self.previousSibling
- self.previousSibling = self.nextSibling = None
+ self.previousSibling = self.nextSibling = None
def _lastRecursiveChild(self):
"Finds the last element beneath this object to be parsed."
@@ -135,15 +137,15 @@ class PageElement:
return lastChild
def insert(self, position, newChild):
- if (isinstance(newChild, basestring)
- or isinstance(newChild, unicode)) \
- and not isinstance(newChild, NavigableString):
- newChild = NavigableString(newChild)
+ if (isinstance(newChild, basestring) or
+ isinstance(newChild, unicode)) and \
+ not isinstance(newChild, NavigableString):
+ newChild = NavigableString(newChild)
- position = min(position, len(self.contents))
- if hasattr(newChild, 'parent') and newChild.parent != None:
+ position = min(position, len(self.contents))
+ if hasattr(newChild, 'parent') and newChild.parent is not None:
# We're 'inserting' an element that's already one
- # of this object's children.
+ # of this object's children.
if newChild.parent == self:
index = self.find(newChild)
if index and index < position:
@@ -153,39 +155,39 @@ class PageElement:
# will jump down one.
position = position - 1
newChild.extract()
-
+
newChild.parent = self
previousChild = None
if position == 0:
newChild.previousSibling = None
newChild.previous = self
else:
- previousChild = self.contents[position-1]
+ previousChild = self.contents[position - 1]
newChild.previousSibling = previousChild
newChild.previousSibling.nextSibling = newChild
newChild.previous = previousChild._lastRecursiveChild()
if newChild.previous:
- newChild.previous.next = newChild
+ newChild.previous.next = newChild
newChildsLastElement = newChild._lastRecursiveChild()
if position >= len(self.contents):
newChild.nextSibling = None
-
+
parent = self
parentsNextSibling = None
while not parentsNextSibling:
parentsNextSibling = parent.nextSibling
parent = parent.parent
- if not parent: # This is the last element in the document.
+ if not parent: # This is the last element in the document.
break
if parentsNextSibling:
newChildsLastElement.next = parentsNextSibling
else:
newChildsLastElement.next = None
else:
- nextChild = self.contents[position]
- newChild.nextSibling = nextChild
+ nextChild = self.contents[position]
+ newChild.nextSibling = nextChild
if newChild.nextSibling:
newChild.nextSibling.previousSibling = newChild
newChildsLastElement.next = nextChild
@@ -217,7 +219,7 @@ class PageElement:
criteria and appear after this Tag in the document."""
return self._findAll(name, attrs, text, limit,
self.nextSiblingGenerator, **kwargs)
- fetchNextSiblings = findNextSiblings # Compatibility with pre-3.x
+ fetchNextSiblings = findNextSiblings # Compatibility with pre-3.x
def findPrevious(self, name=None, attrs={}, text=None, **kwargs):
"""Returns the first item that matches the given criteria and
@@ -230,7 +232,7 @@ class PageElement:
before this Tag in the document."""
return self._findAll(name, attrs, text, limit, self.previousGenerator,
**kwargs)
- fetchPrevious = findAllPrevious # Compatibility with pre-3.x
+ fetchPrevious = findAllPrevious # Compatibility with pre-3.x
def findPreviousSibling(self, name=None, attrs={}, text=None, **kwargs):
"""Returns the closest sibling to this Tag that matches the
@@ -244,7 +246,7 @@ class PageElement:
criteria and appear before this Tag in the document."""
return self._findAll(name, attrs, text, limit,
self.previousSiblingGenerator, **kwargs)
- fetchPreviousSiblings = findPreviousSiblings # Compatibility with pre-3.x
+ fetchPreviousSiblings = findPreviousSiblings # Compatibility with pre-3.x
def findParent(self, name=None, attrs={}, **kwargs):
"""Returns the closest parent of this Tag that matches the given
@@ -263,9 +265,9 @@ class PageElement:
return self._findAll(name, attrs, None, limit, self.parentGenerator,
**kwargs)
- fetchParents = findParents # Compatibility with pre-3.x
+ fetchParents = findParents # Compatibility with pre-3.x
- #These methods do the real heavy lifting.
+ # These methods do the real heavy lifting.
def _findOne(self, method, name, attrs, text, **kwargs):
r = None
@@ -273,7 +275,7 @@ class PageElement:
if l:
r = l[0]
return r
-
+
def _findAll(self, name, attrs, text, limit, generator, **kwargs):
"Iterates over a generator looking for things that match."
@@ -297,8 +299,8 @@ class PageElement:
break
return results
- #These Generators can be used to navigate starting from both
- #NavigableStrings and Tags.
+ # These Generators can be used to navigate starting from both
+ # NavigableStrings and Tags.
def nextGenerator(self):
i = self
while i:
@@ -332,7 +334,7 @@ class PageElement:
# Utility methods
def substituteEncoding(self, str, encoding=None):
encoding = encoding or "utf-8"
- return str.replace("%SOUP-ENCODING%", encoding)
+ return str.replace("%SOUP-ENCODING%", encoding)
def toEncoding(self, s, encoding=None):
"""Encodes an object to a string in some encoding, or to Unicode.
@@ -347,11 +349,12 @@ class PageElement:
s = unicode(s)
else:
if encoding:
- s = self.toEncoding(str(s), encoding)
+ s = self.toEncoding(str(s), encoding)
else:
s = unicode(s)
return s
+
class NavigableString(unicode, PageElement):
def __getattr__(self, attr):
@@ -361,22 +364,24 @@ class NavigableString(unicode, PageElement):
if attr == 'string':
return self
else:
- raise AttributeError, "'%s' object has no attribute '%s'" % (self.__class__.__name__, attr)
+ raise AttributeError("'%s' object has no attribute '%s'" % (self.__class__.__name__, attr))
def __unicode__(self):
- return __str__(self, None)
+ return self.__str__()
def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING):
if encoding:
return self.encode(encoding)
else:
return self
-
+
+
class CData(NavigableString):
def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING):
return "" % NavigableString.__str__(self, encoding)
+
class ProcessingInstruction(NavigableString):
def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING):
output = self
@@ -384,13 +389,16 @@ class ProcessingInstruction(NavigableString):
output = self.substituteEncoding(output, encoding)
return "%s?>" % self.toEncoding(output, encoding)
+
class Comment(NavigableString):
def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING):
- return "" % NavigableString.__str__(self, encoding)
+ return "" % NavigableString.__str__(self, encoding)
+
class Declaration(NavigableString):
def __str__(self, encoding=DEFAULT_OUTPUT_ENCODING):
- return "" % NavigableString.__str__(self, encoding)
+ return "" % NavigableString.__str__(self, encoding)
+
class Tag(PageElement):
"""Represents a found HTML tag with its attributes and contents."""
@@ -415,7 +423,7 @@ class Tag(PageElement):
self.isSelfClosing = parser.isSelfClosingTag(name)
self.convertHTMLEntities = parser.convertHTMLEntities
self.name = name
- if attrs == None:
+ if attrs is None:
attrs = []
self.attrs = attrs
self.contents = []
@@ -427,10 +435,10 @@ class Tag(PageElement):
"""Returns the value of the 'key' attribute for the tag, or
the value given for 'default' if it doesn't have that
attribute."""
- return self._getAttrMap().get(key, default)
+ return self._getAttrMap().get(key, default)
def has_key(self, key):
- return self._getAttrMap().has_key(key)
+ return key in self._getAttrMap()
def __getitem__(self, key):
"""tag[key] returns the value of the 'key' attribute for the tag,
@@ -452,7 +460,7 @@ class Tag(PageElement):
"A tag is non-None even if it has no contents."
return True
- def __setitem__(self, key, value):
+ def __setitem__(self, key, value):
"""Setting tag[key] sets the value of the 'key' attribute for the
tag."""
self._getAttrMap()
@@ -471,10 +479,10 @@ class Tag(PageElement):
for item in self.attrs:
if item[0] == key:
self.attrs.remove(item)
- #We don't break because bad HTML can define the same
- #attribute multiple times.
+ # We don't break because bad HTML can define the same
+ # attribute multiple times.
self._getAttrMap()
- if self.attrMap.has_key(key):
+ if key in self.attrMap:
del self.attrMap[key]
def __call__(self, *args, **kwargs):
@@ -484,8 +492,7 @@ class Tag(PageElement):
return apply(self.findAll, args, kwargs)
def __getattr__(self, tag):
- #print "Getattr %s.%s" % (self.__class__, tag)
- if len(tag) > 3 and tag.rfind('Tag') == len(tag)-3:
+ if len(tag) > 3 and tag.rfind('Tag') == len(tag) - 3:
return self.find(tag[:-3])
elif tag.find('__') != 0:
return self.find(tag)
@@ -518,7 +525,7 @@ class Tag(PageElement):
def _convertEntities(self, match):
x = match.group(1)
if x in name2codepoint:
- return unichr(name2codepoint[x])
+ return unichr(name2codepoint[x])
elif "&" + x + ";" in self.XML_ENTITIES_TO_CHARS:
return '&%s;' % x
else:
@@ -539,7 +546,7 @@ class Tag(PageElement):
if self.attrs:
for key, val in self.attrs:
fmt = '%s="%s"'
- if isString(val):
+ if isString(val):
if self.containsSubstitutions and '%SOUP-ENCODING%' in val:
val = self.substituteEncoding(val, encoding)
@@ -577,7 +584,6 @@ class Tag(PageElement):
val = val.replace("<", "<").replace(">", ">")
val = self.BARE_AMPERSAND.sub("&", val)
-
attrs.append(fmt % (self.toEncoding(key, encoding),
self.toEncoding(val, encoding)))
close = ''
@@ -590,7 +596,7 @@ class Tag(PageElement):
indentTag, indentContents = 0, 0
if prettyPrint:
indentTag = indentLevel
- space = (' ' * (indentTag-1))
+ space = (' ' * (indentTag - 1))
indentContents = indentTag + 1
contents = self.renderContents(encoding, prettyPrint, indentContents)
if self.hidden:
@@ -599,7 +605,7 @@ class Tag(PageElement):
s = []
attributeString = ''
if attrs:
- attributeString = ' ' + ' '.join(attrs)
+ attributeString = ' ' + ' '.join(attrs)
if prettyPrint:
s.append(space)
s.append('<%s%s%s>' % (encodedName, attributeString, close))
@@ -623,7 +629,7 @@ class Tag(PageElement):
prettyPrint=False, indentLevel=0):
"""Renders the contents of this tag as a string in the given
encoding. If encoding is None, returns a Unicode string.."""
- s=[]
+ s = []
for c in self:
text = None
if isinstance(c, NavigableString):
@@ -631,16 +637,16 @@ class Tag(PageElement):
elif isinstance(c, Tag):
s.append(c.__str__(encoding, prettyPrint, indentLevel))
if text and prettyPrint:
- text = text.strip()
+ text = text.strip()
if text:
if prettyPrint:
- s.append(" " * (indentLevel-1))
+ s.append(" " * (indentLevel - 1))
s.append(text)
if prettyPrint:
s.append("\n")
- return ''.join(s)
+ return ''.join(s)
- #Soup methods
+ # Soup methods
def find(self, name=None, attrs={}, recursive=True, text=None,
**kwargs):
@@ -673,20 +679,20 @@ class Tag(PageElement):
# Pre-3.x compatibility methods
first = find
fetch = findAll
-
+
def fetchText(self, text=None, recursive=True, limit=None):
return self.findAll(text=text, recursive=recursive, limit=limit)
def firstText(self, text=None, recursive=True):
return self.find(text=text, recursive=recursive)
-
- #Utility methods
+
+ # Utility methods
def append(self, tag):
"""Appends the given tag to the contents of this tag."""
self.contents.append(tag)
- #Private methods
+ # Private methods
def _getAttrMap(self):
"""Initializes a map representation of this tag's attributes,
@@ -694,30 +700,31 @@ class Tag(PageElement):
if not getattr(self, 'attrMap'):
self.attrMap = {}
for (key, value) in self.attrs:
- self.attrMap[key] = value
+ self.attrMap[key] = value
return self.attrMap
- #Generator methods
+ # Generator methods
def childGenerator(self):
for i in range(0, len(self.contents)):
yield self.contents[i]
raise StopIteration
-
+
def recursiveChildGenerator(self):
stack = [(self, 0)]
while stack:
tag, start = stack.pop()
- if isinstance(tag, Tag):
+ if isinstance(tag, Tag):
for i in range(start, len(tag.contents)):
a = tag.contents[i]
yield a
if isinstance(a, Tag) and tag.contents:
if i < len(tag.contents) - 1:
- stack.append((tag, i+1))
+ stack.append((tag, i + 1))
stack.append((a, 0))
break
raise StopIteration
+
# Next, a couple classes to represent queries and their results.
class SoupStrainer:
"""Encapsulates a number of ways of matching a markup element (tag or
@@ -742,32 +749,31 @@ class SoupStrainer:
return self.text
else:
return "%s|%s" % (self.name, self.attrs)
-
+
def searchTag(self, markupName=None, markupAttrs={}):
found = None
markup = None
if isinstance(markupName, Tag):
markup = markupName
markupAttrs = markup
- callFunctionWithTagData = callable(self.name) \
- and not isinstance(markupName, Tag)
+ callFunctionWithTagData = callable(self.name) and \
+ not isinstance(markupName, Tag)
- if (not self.name) \
- or callFunctionWithTagData \
- or (markup and self._matches(markup, self.name)) \
- or (not markup and self._matches(markupName, self.name)):
+ if not self.name or callFunctionWithTagData or \
+ (markup and self._matches(markup, self.name)) or \
+ (not markup and self._matches(markupName, self.name)):
if callFunctionWithTagData:
match = self.name(markupName, markupAttrs)
else:
- match = True
+ match = True
markupAttrMap = None
for attr, matchAgainst in self.attrs.items():
if not markupAttrMap:
- if hasattr(markupAttrs, 'get'):
+ if hasattr(markupAttrs, 'get'):
markupAttrMap = markupAttrs
- else:
+ else:
markupAttrMap = {}
- for k,v in markupAttrs:
+ for k, v in markupAttrs:
markupAttrMap[k] = v
attrValue = markupAttrMap.get(attr)
if not self._matches(attrValue, matchAgainst):
@@ -781,14 +787,13 @@ class SoupStrainer:
return found
def search(self, markup):
- #print 'looking for %s in %s' % (self, markup)
found = None
# If given a list of items, scan it for a text element that
- # matches.
+ # matches.
if isList(markup) and not isinstance(markup, Tag):
for element in markup:
- if isinstance(element, NavigableString) \
- and self.search(element):
+ if isinstance(element, NavigableString) and \
+ self.search(element):
found = element
break
# If it's a Tag, make sure its name or attributes match.
@@ -797,37 +802,35 @@ class SoupStrainer:
if not self.text:
found = self.searchTag(markup)
# If it's text, make sure the text matches.
- elif isinstance(markup, NavigableString) or \
- isString(markup):
+ elif isinstance(markup, NavigableString) or isString(markup):
if self._matches(markup, self.text):
found = markup
else:
- raise Exception, "I don't know how to match against a %s" \
- % markup.__class__
+ raise Exception("I don't know how to match against a %s" %
+ markup.__class__)
return found
-
- def _matches(self, markup, matchAgainst):
- #print "Matching %s against %s" % (markup, matchAgainst)
+
+ def _matches(self, markup, matchAgainst):
result = False
if matchAgainst is True:
- result = markup != None
+ result = markup is not None
elif callable(matchAgainst):
result = matchAgainst(markup)
else:
- #Custom match methods take the tag as an argument, but all
- #other ways of matching match the tag name as a string.
+ # Custom match methods take the tag as an argument, but all
+ # other ways of matching match the tag name as a string.
if isinstance(markup, Tag):
markup = markup.name
if markup and not isString(markup):
markup = unicode(markup)
- #Now we know that chunk is either a string, or None.
+ # Now we know that chunk is either a string, or None.
if hasattr(matchAgainst, 'match'):
# It's a regexp object.
result = markup and matchAgainst.search(markup)
elif isList(matchAgainst):
result = markup in matchAgainst
elif hasattr(matchAgainst, 'items'):
- result = markup.has_key(matchAgainst)
+ result = matchAgainst in markup
elif matchAgainst and isString(markup):
if isinstance(markup, unicode):
matchAgainst = unicode(matchAgainst)
@@ -838,29 +841,34 @@ class SoupStrainer:
result = matchAgainst == markup
return result
+
class ResultSet(list):
"""A ResultSet is just a list that keeps track of the SoupStrainer
that created it."""
+
def __init__(self, source):
list.__init__([])
self.source = source
# Now, some helper functions.
+
def isList(l):
"""Convenience method that works with all 2.x versions of Python
to determine whether or not something is listlike."""
- return hasattr(l, '__iter__') \
- or (type(l) in (types.ListType, types.TupleType))
+ return hasattr(l, '__iter__') or \
+ type(l) in (types.ListType, types.TupleType)
+
def isString(s):
"""Convenience method that works with all 2.x versions of Python
to determine whether or not something is stringlike."""
try:
- return isinstance(s, unicode) or isintance(s, basestring)
+ return isinstance(s, unicode) or isintance(s, basestring)
except NameError:
return isinstance(s, str)
+
def buildTagMap(default, *args):
"""Turns a list of maps, lists, or scalars into a single map.
Used to build the SELF_CLOSING_TAGS, NESTABLE_TAGS, and
@@ -868,26 +876,27 @@ def buildTagMap(default, *args):
built = {}
for portion in args:
if hasattr(portion, 'items'):
- #It's a map. Merge it.
- for k,v in portion.items():
+ # It's a map. Merge it.
+ for k, v in portion.items():
built[k] = v
elif isList(portion):
- #It's a list. Map each item to the default.
+ # It's a list. Map each item to the default.
for k in portion:
built[k] = default
else:
- #It's a scalar. Map it to the default.
+ # It's a scalar. Map it to the default.
built[portion] = default
return built
# Now, the parser classes.
+
class BeautifulStoneSoup(Tag, SGMLParser):
"""This class contains the basic parser and search code. It defines
a parser that knows nothing about tag behavior except for the
following:
-
+
You can't close a tag without closing all the tags it encloses.
That is, "" actually means
"".
@@ -922,7 +931,7 @@ class BeautifulStoneSoup(Tag, SGMLParser):
convertEntities=None, selfClosingTags=None):
"""The Soup object is initialized as the 'root tag', and the
provided markup (which can be a string or a file-like object)
- is fed into the underlying parser.
+ is fed into the underlying parser.
sgmllib will process most bad HTML, and the BeautifulSoup
class has some tricks for dealing with some HTML that kills
@@ -964,7 +973,7 @@ class BeautifulStoneSoup(Tag, SGMLParser):
self.instanceSelfClosingTags = buildTagMap(None, selfClosingTags)
SGMLParser.__init__(self)
-
+
if hasattr(markup, 'read'): # It's a file-type object.
markup = markup.read()
self.markup = markup
@@ -982,15 +991,15 @@ class BeautifulStoneSoup(Tag, SGMLParser):
if not hasattr(self, 'originalEncoding'):
self.originalEncoding = None
else:
- dammit = UnicodeDammit\
- (markup, [self.fromEncoding, inDocumentEncoding],
- smartQuotesTo=self.smartQuotesTo)
+ dammit = UnicodeDammit(markup,
+ [self.fromEncoding, inDocumentEncoding],
+ smartQuotesTo=self.smartQuotesTo)
markup = dammit.unicode
self.originalEncoding = dammit.originalEncoding
if markup:
if self.markupMassage:
if not isList(self.markupMassage):
- self.markupMassage = self.MARKUP_MASSAGE
+ self.markupMassage = self.MARKUP_MASSAGE
for fix, m in self.markupMassage:
markup = fix.sub(m, markup)
self.reset()
@@ -1005,10 +1014,8 @@ class BeautifulStoneSoup(Tag, SGMLParser):
def __getattr__(self, methodName):
"""This method routes method call requests to either the SGMLParser
superclass or the Tag superclass, depending on the method name."""
- #print "__getattr__ called on %s.%s" % (self.__class__, methodName)
-
- if methodName.find('start_') == 0 or methodName.find('end_') == 0 \
- or methodName.find('do_') == 0:
+ if methodName.find('start_') == 0 or methodName.find('end_') == 0 or \
+ methodName.find('do_') == 0:
return SGMLParser.__getattr__(self, methodName)
elif methodName.find('__') != 0:
return Tag.__getattr__(self, methodName)
@@ -1018,9 +1025,9 @@ class BeautifulStoneSoup(Tag, SGMLParser):
def isSelfClosingTag(self, name):
"""Returns true iff the given string is the name of a
self-closing tag according to this parser."""
- return self.SELF_CLOSING_TAGS.has_key(name) \
- or self.instanceSelfClosingTags.has_key(name)
-
+ return name in self.SELF_CLOSING_TAGS or \
+ name in self.instanceSelfClosingTags
+
def reset(self):
Tag.__init__(self, self, self.ROOT_TAG_NAME)
self.hidden = 1
@@ -1030,23 +1037,21 @@ class BeautifulStoneSoup(Tag, SGMLParser):
self.tagStack = []
self.quoteStack = []
self.pushTag(self)
-
+
def popTag(self):
- tag = self.tagStack.pop()
+ self.tagStack.pop()
# Tags with just one string-owning child get the child as a
# 'string' property, so that soup.tag.string is shorthand for
# soup.tag.contents[0]
if len(self.currentTag.contents) == 1 and \
- isinstance(self.currentTag.contents[0], NavigableString):
+ isinstance(self.currentTag.contents[0], NavigableString):
self.currentTag.string = self.currentTag.contents[0]
- #print "Pop", tag.name
if self.tagStack:
self.currentTag = self.tagStack[-1]
return self.currentTag
def pushTag(self, tag):
- #print "Push", tag.name
if self.currentTag:
self.currentTag.append(tag)
self.tagStack.append(tag)
@@ -1064,8 +1069,8 @@ class BeautifulStoneSoup(Tag, SGMLParser):
currentData = ' '
self.currentData = []
if self.parseOnlyThese and len(self.tagStack) <= 1 and \
- (not self.parseOnlyThese.text or \
- not self.parseOnlyThese.search(currentData)):
+ (not self.parseOnlyThese.text or
+ not self.parseOnlyThese.search(currentData)):
return
o = containerClass(currentData)
o.setup(self.currentTag, self.previous)
@@ -1074,31 +1079,28 @@ class BeautifulStoneSoup(Tag, SGMLParser):
self.previous = o
self.currentTag.contents.append(o)
-
def _popToTag(self, name, inclusivePop=True):
"""Pops the tag stack up to and including the most recent
instance of the given tag. If inclusivePop is false, pops the tag
stack up to but *not* including the most recent instqance of
the given tag."""
- #print "Popping to %s" % name
if name == self.ROOT_TAG_NAME:
- return
+ return
numPops = 0
mostRecentTag = None
- for i in range(len(self.tagStack)-1, 0, -1):
+ for i in range(len(self.tagStack) - 1, 0, -1):
if name == self.tagStack[i].name:
- numPops = len(self.tagStack)-i
+ numPops = len(self.tagStack) - i
break
if not inclusivePop:
numPops = numPops - 1
for i in range(0, numPops):
mostRecentTag = self.popTag()
- return mostRecentTag
+ return mostRecentTag
def _smartPop(self, name):
-
"""We need to pop up to the previous tag of this type, unless
one of this tag's nesting reset triggers comes between this
tag and the previous tag of this type, OR unless this tag is a
@@ -1117,26 +1119,25 @@ class BeautifulStoneSoup(Tag, SGMLParser):
"""
nestingResetTriggers = self.NESTABLE_TAGS.get(name)
- isNestable = nestingResetTriggers != None
- isResetNesting = self.RESET_NESTING_TAGS.has_key(name)
+ isNestable = nestingResetTriggers is not None
+ isResetNesting = name in self.RESET_NESTING_TAGS
popTo = None
inclusive = True
- for i in range(len(self.tagStack)-1, 0, -1):
+ for i in range(len(self.tagStack) - 1, 0, -1):
p = self.tagStack[i]
if (not p or p.name == name) and not isNestable:
- #Non-nestable tags get popped to the top or to their
- #last occurance.
+ # Non-nestable tags get popped to the top or to their
+ # last occurance.
popTo = name
break
- if (nestingResetTriggers != None
- and p.name in nestingResetTriggers) \
- or (nestingResetTriggers == None and isResetNesting
- and self.RESET_NESTING_TAGS.has_key(p.name)):
-
- #If we encounter one of the nesting reset triggers
- #peculiar to this tag, or we encounter another tag
- #that causes nesting to reset, pop up to but not
- #including that tag.
+ if (nestingResetTriggers is not None and
+ p.name in nestingResetTriggers) or \
+ (nestingResetTriggers is None and isResetNesting and
+ p.name in self.RESET_NESTING_TAGS):
+ # If we encounter one of the nesting reset triggers
+ # peculiar to this tag, or we encounter another tag
+ # that causes nesting to reset, pop up to but not
+ # including that tag.
popTo = p.name
inclusive = False
break
@@ -1145,20 +1146,18 @@ class BeautifulStoneSoup(Tag, SGMLParser):
self._popToTag(popTo, inclusive)
def unknown_starttag(self, name, attrs, selfClosing=0):
- #print "Start tag %s: %s" % (name, attrs)
if self.quoteStack:
- #This is not a real tag.
- #print "<%s> is not real!" % name
+ # This is not a real tag.
attrs = ''.join(map(lambda(x, y): ' %s="%s"' % (x, y), attrs))
self.currentData.append('<%s%s>' % (name, attrs))
- return
+ return
self.endData()
if not self.isSelfClosingTag(name) and not selfClosing:
self._smartPop(name)
- if self.parseOnlyThese and len(self.tagStack) <= 1 \
- and (self.parseOnlyThese.text or not self.parseOnlyThese.searchTag(name, attrs)):
+ if self.parseOnlyThese and len(self.tagStack) <= 1 and \
+ (self.parseOnlyThese.text or not self.parseOnlyThese.searchTag(name, attrs)):
return
tag = Tag(self, name, attrs, self.currentTag, self.previous)
@@ -1167,18 +1166,15 @@ class BeautifulStoneSoup(Tag, SGMLParser):
self.previous = tag
self.pushTag(tag)
if selfClosing or self.isSelfClosingTag(name):
- self.popTag()
+ self.popTag()
if name in self.QUOTE_TAGS:
- #print "Beginning quote (%s)" % name
self.quoteStack.append(name)
self.literal = 1
return tag
def unknown_endtag(self, name):
- #print "End tag %s" % name
if self.quoteStack and self.quoteStack[-1] != name:
- #This is not a real end tag.
- #print "%s> is not real!" % name
+ # This is not a real end tag.
self.currentData.append('%s>' % name)
return
self.endData()
@@ -1190,11 +1186,11 @@ class BeautifulStoneSoup(Tag, SGMLParser):
def handle_data(self, data):
if self.convertHTMLEntities:
if data[0] == '&':
- data = self.BARE_AMPERSAND.sub("&",data)
+ data = self.BARE_AMPERSAND.sub("&", data)
else:
- data = data.replace('&','&') \
- .replace('<','<') \
- .replace('>','>')
+ data = data.replace('&', '&') \
+ .replace('<', '<') \
+ .replace('>', '>')
self.currentData.append(data)
def _toStringSubclass(self, text, subclass):
@@ -1219,10 +1215,10 @@ class BeautifulStoneSoup(Tag, SGMLParser):
def handle_charref(self, ref):
"Handle character references as data."
if ref[0] == 'x':
- data = unichr(int(ref[1:],16))
+ data = unichr(int(ref[1:], 16))
else:
data = unichr(int(ref))
-
+
if u'\x80' <= data <= u'\x9F':
data = UnicodeDammit.subMSChar(chr(ord(data)), self.smartQuotesTo)
elif not self.convertHTMLEntities and not self.convertXMLEntities:
@@ -1235,7 +1231,7 @@ class BeautifulStoneSoup(Tag, SGMLParser):
HTML entity references to the corresponding Unicode
characters."""
replaceWithXMLEntity = self.convertXMLEntities and \
- self.XML_ENTITIES_TO_CHARS.has_key(ref)
+ ref in self.XML_ENTITIES_TO_CHARS
if self.convertHTMLEntities or replaceWithXMLEntity:
try:
data = unichr(name2codepoint[ref])
@@ -1243,11 +1239,11 @@ class BeautifulStoneSoup(Tag, SGMLParser):
if replaceWithXMLEntity:
data = self.XML_ENTITIES_TO_CHARS.get(ref)
else:
- data="&%s" % ref
+ data = "&%s" % ref
else:
data = '&%s;' % ref
self.handle_data(data)
-
+
def handle_decl(self, data):
"Handle DOCTYPEs and the like as Declaration objects."
self._toStringSubclass(data, Declaration)
@@ -1256,13 +1252,13 @@ class BeautifulStoneSoup(Tag, SGMLParser):
"""Treat a bogus SGML declaration as raw data. Treat a CDATA
declaration as a CData object."""
j = None
- if self.rawdata[i:i+9] == '', i)
- if k == -1:
- k = len(self.rawdata)
- data = self.rawdata[i+9:k]
- j = k+3
- self._toStringSubclass(data, CData)
+ if self.rawdata[i:i + 9] == '', i)
+ if k == -1:
+ k = len(self.rawdata)
+ data = self.rawdata[i + 9:k]
+ j = k + 3
+ self._toStringSubclass(data, CData)
else:
try:
j = SGMLParser.parse_declaration(self, i)
@@ -1272,6 +1268,7 @@ class BeautifulStoneSoup(Tag, SGMLParser):
j = i + len(toHandle)
return j
+
class BeautifulSoup(BeautifulStoneSoup):
"""This parser knows the following facts about HTML:
@@ -1321,7 +1318,7 @@ class BeautifulSoup(BeautifulStoneSoup):
BeautifulStoneSoup before writing your own subclass."""
def __init__(self, *args, **kwargs):
- if not kwargs.has_key('smartQuotesTo'):
+ if 'smartQuotesTo' not in kwargs:
kwargs['smartQuotesTo'] = self.HTML_ENTITIES
BeautifulStoneSoup.__init__(self, *args, **kwargs)
@@ -1330,19 +1327,19 @@ class BeautifulSoup(BeautifulStoneSoup):
'spacer', 'link', 'frame', 'base'])
QUOTE_TAGS = {'script': None}
-
- #According to the HTML standard, each of these inline tags can
- #contain another tag of the same type. Furthermore, it's common
- #to actually use these tags this way.
+
+ # According to the HTML standard, each of these inline tags can
+ # contain another tag of the same type. Furthermore, it's common
+ # to actually use these tags this way.
NESTABLE_INLINE_TAGS = ['span', 'font', 'q', 'object', 'bdo', 'sub', 'sup',
'center']
- #According to the HTML standard, these block tags can contain
- #another tag of the same type. Furthermore, it's common
- #to actually use these tags this way.
+ # According to the HTML standard, these block tags can contain
+ # another tag of the same type. Furthermore, it's common
+ # to actually use these tags this way.
NESTABLE_BLOCK_TAGS = ['blockquote', 'div', 'fieldset', 'ins', 'del']
- #Lists can contain other lists, but there are restrictions.
+ # Lists can contain other lists, but there are restrictions.
NESTABLE_LIST_TAGS = { 'ol' : [],
'ul' : [],
'li' : ['ul', 'ol'],
@@ -1350,8 +1347,8 @@ class BeautifulSoup(BeautifulStoneSoup):
'dd' : ['dl'],
'dt' : ['dl'] }
- #Tables can contain other tables, but there are restrictions.
- NESTABLE_TABLE_TAGS = {'table' : [],
+ # Tables can contain other tables, but there are restrictions.
+ NESTABLE_TABLE_TAGS = {'table' : [],
'tr' : ['table', 'tbody', 'tfoot', 'thead'],
'td' : ['tr'],
'th' : ['tr'],
@@ -1362,8 +1359,8 @@ class BeautifulSoup(BeautifulStoneSoup):
NON_NESTABLE_BLOCK_TAGS = ['address', 'form', 'p', 'pre']
- #If one of these tags is encountered, all tags up to the next tag of
- #this type are popped.
+ # If one of these tags is encountered, all tags up to the next tag of
+ # this type are popped.
RESET_NESTING_TAGS = buildTagMap(None, NESTABLE_BLOCK_TAGS, 'noscript',
NON_NESTABLE_BLOCK_TAGS,
NESTABLE_LIST_TAGS,
@@ -1393,17 +1390,17 @@ class BeautifulSoup(BeautifulStoneSoup):
contentType = value
contentTypeIndex = i
- if httpEquiv and contentType: # It's an interesting meta tag.
+ if httpEquiv and contentType: # It's an interesting meta tag.
match = self.CHARSET_RE.search(contentType)
if match:
if getattr(self, 'declaredHTMLEncoding') or \
- (self.originalEncoding == self.fromEncoding):
+ self.originalEncoding == self.fromEncoding:
# This is our second pass through the document, or
# else an encoding was specified explicitly and it
# worked. Rewrite the meta tag.
- newAttr = self.CHARSET_RE.sub\
- (lambda(match):match.group(1) +
- "%SOUP-ENCODING%", value)
+ newAttr = self.CHARSET_RE.sub(
+ lambda(match): match.group(1) + "%SOUP-ENCODING%",
+ value)
attrs[contentTypeIndex] = (attrs[contentTypeIndex][0],
newAttr)
tagNeedsEncodingSubstitution = True
@@ -1419,9 +1416,11 @@ class BeautifulSoup(BeautifulStoneSoup):
if tag and tagNeedsEncodingSubstitution:
tag.containsSubstitutions = True
+
class StopParsing(Exception):
pass
-
+
+
class ICantBelieveItsBeautifulSoup(BeautifulSoup):
"""The BeautifulSoup class is oriented towards skipping over
@@ -1447,10 +1446,9 @@ class ICantBelieveItsBeautifulSoup(BeautifulSoup):
it's valid HTML and BeautifulSoup screwed up by assuming it
wouldn't be."""
- I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS = \
- ['em', 'big', 'i', 'small', 'tt', 'abbr', 'acronym', 'strong',
- 'cite', 'code', 'dfn', 'kbd', 'samp', 'strong', 'var', 'b',
- 'big']
+ I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS = ['em', 'big', 'i', 'small',
+ 'tt', 'abbr', 'acronym', 'strong', 'cite', 'code', 'dfn', 'kbd', 'samp',
+ 'strong', 'var', 'b', 'big']
I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS = ['noscript']
@@ -1458,6 +1456,7 @@ class ICantBelieveItsBeautifulSoup(BeautifulSoup):
I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS,
I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS)
+
class MinimalSoup(BeautifulSoup):
"""The MinimalSoup class is for parsing HTML that contains
pathologically bad markup. It makes no assumptions about tag
@@ -1467,10 +1466,11 @@ class MinimalSoup(BeautifulSoup):
This also makes it better for subclassing than BeautifulStoneSoup
or BeautifulSoup."""
-
+
RESET_NESTING_TAGS = buildTagMap('noscript')
NESTABLE_TAGS = {}
+
class BeautifulSOAP(BeautifulStoneSoup):
"""This class will push a tag with only a single string child into
the tag's parent as an attribute. The attribute's name is the tag
@@ -1496,28 +1496,38 @@ class BeautifulSOAP(BeautifulStoneSoup):
tag = self.tagStack[-1]
parent = self.tagStack[-2]
parent._getAttrMap()
- if (isinstance(tag, Tag) and len(tag.contents) == 1 and
- isinstance(tag.contents[0], NavigableString) and
- not parent.attrMap.has_key(tag.name)):
+ if isinstance(tag, Tag) and len(tag.contents) == 1 and \
+ isinstance(tag.contents[0], NavigableString) and \
+ tag.name not in parent.attrMap:
parent[tag.name] = tag.contents[0]
BeautifulStoneSoup.popTag(self)
-#Enterprise class names! It has come to our attention that some people
-#think the names of the Beautiful Soup parser classes are too silly
-#and "unprofessional" for use in enterprise screen-scraping. We feel
-#your pain! For such-minded folk, the Beautiful Soup Consortium And
-#All-Night Kosher Bakery recommends renaming this file to
-#"RobustParser.py" (or, in cases of extreme enterprisitude,
-#"RobustParserBeanInterface.class") and using the following
-#enterprise-friendly class aliases:
+# Enterprise class names! It has come to our attention that some people
+# think the names of the Beautiful Soup parser classes are too silly
+# and "unprofessional" for use in enterprise screen-scraping. We feel
+# your pain! For such-minded folk, the Beautiful Soup Consortium And
+# All-Night Kosher Bakery recommends renaming this file to
+# "RobustParser.py" (or, in cases of extreme enterprisitude,
+# "RobustParserBeanInterface.class") and using the following
+# enterprise-friendly class aliases:
+
+
class RobustXMLParser(BeautifulStoneSoup):
pass
+
+
class RobustHTMLParser(BeautifulSoup):
pass
+
+
class RobustWackAssHTMLParser(ICantBelieveItsBeautifulSoup):
pass
+
+
class RobustInsanelyWackAssHTMLParser(MinimalSoup):
pass
+
+
class SimplifyingSOAPParser(BeautifulSOAP):
pass
@@ -1541,17 +1551,6 @@ except:
chardet = None
chardet = None
-# cjkcodecs and iconv_codec make Python know about more character encodings.
-# Both are available from http://cjkpython.i18n.org/
-# They're built in if you use Python 2.4.
-try:
- import cjkcodecs.aliases
-except:
- pass
-try:
- import iconv_codec
-except:
- pass
class UnicodeDammit:
"""A class for detecting the encoding of a *ML document and
@@ -1565,11 +1564,11 @@ class UnicodeDammit:
# by the heuristics in find_codec.
CHARSET_ALIASES = { "macintosh" : "mac-roman",
"x-sjis" : "shift-jis" }
-
+
def __init__(self, markup, overrideEncodings=[],
smartQuotesTo='xml'):
self.markup, documentEncoding, sniffedEncoding = \
- self._detectEncoding(markup)
+ self._detectEncoding(markup)
self.smartQuotesTo = smartQuotesTo
self.triedEncodings = []
if isinstance(markup, unicode):
@@ -1578,12 +1577,14 @@ class UnicodeDammit:
u = None
for proposedEncoding in overrideEncodings:
u = self._convertFrom(proposedEncoding)
- if u: break
+ if u:
+ break
if not u:
for proposedEncoding in (documentEncoding, sniffedEncoding):
u = self._convertFrom(proposedEncoding)
- if u: break
-
+ if u:
+ break
+
# If no luck and we have auto-detection library, try that:
if not u and chardet and not isinstance(self.markup, unicode):
u = self._convertFrom(chardet.detect(self.markup)['encoding'])
@@ -1592,25 +1593,27 @@ class UnicodeDammit:
if not u:
for proposed_encoding in ("utf-8", "windows-1252"):
u = self._convertFrom(proposed_encoding)
- if u: break
+ if u:
+ break
self.unicode = u
- if not u: self.originalEncoding = None
+ if not u:
+ self.originalEncoding = None
def subMSChar(orig, smartQuotesTo):
"""Changes a MS smart quote character to an XML or HTML
entity."""
sub = UnicodeDammit.MS_CHARS.get(orig)
- if type(sub) == types.TupleType:
+ if isinstance(sub, types.TupleType):
if smartQuotesTo == 'xml':
sub = '%s;' % sub[1]
elif smartQuotesTo == 'html':
sub = '&%s;' % sub[0]
else:
- sub = unichr(int(sub[1],16))
- return sub
+ sub = unichr(int(sub[1], 16))
+ return sub
subMSChar = staticmethod(subMSChar)
- def _convertFrom(self, proposed):
+ def _convertFrom(self, proposed):
proposed = self.find_codec(proposed)
if not proposed or proposed in self.triedEncodings:
return None
@@ -1619,23 +1622,18 @@ class UnicodeDammit:
# Convert smart quotes to HTML if coming from an encoding
# that might have them.
- if self.smartQuotesTo and proposed in("windows-1252",
- "ISO-8859-1",
- "ISO-8859-2"):
- markup = re.compile("([\x80-\x9f])").sub \
- (lambda(x): self.subMSChar(x.group(1),self.smartQuotesTo),
- markup)
+ if self.smartQuotesTo and proposed in ("windows-1252", "ISO-8859-1",
+ "ISO-8859-2"):
+ markup = re.compile("([\x80-\x9f])").sub(
+ lambda(x): self.subMSChar(x.group(1), self.smartQuotesTo),
+ markup)
try:
- # print "Trying to convert document to %s" % proposed
u = self._toUnicode(markup, proposed)
- self.markup = u
+ self.markup = u
self.originalEncoding = proposed
- except Exception, e:
- # print "That didn't work!"
- # print e
- return None
- #print "Correct encoding: %s" % proposed
+ except Exception:
+ return None
return self.markup
def _toUnicode(self, data, encoding):
@@ -1643,12 +1641,12 @@ class UnicodeDammit:
%encoding is a string recognized by encodings.aliases'''
# strip Byte Order Mark (if present)
- if (len(data) >= 4) and (data[:2] == '\xfe\xff') \
- and (data[2:4] != '\x00\x00'):
+ if len(data) >= 4 and data[:2] == '\xfe\xff' and \
+ data[2:4] != '\x00\x00':
encoding = 'utf-16be'
data = data[2:]
- elif (len(data) >= 4) and (data[:2] == '\xff\xfe') \
- and (data[2:4] != '\x00\x00'):
+ elif len(data) >= 4 and data[:2] == '\xff\xfe' and \
+ data[2:4] != '\x00\x00':
encoding = 'utf-16le'
data = data[2:]
elif data[:3] == '\xef\xbb\xbf':
@@ -1662,7 +1660,7 @@ class UnicodeDammit:
data = data[4:]
newdata = unicode(data, encoding)
return newdata
-
+
def _detectEncoding(self, xml_data):
"""Given a document, tries to detect its XML encoding."""
xml_encoding = sniffed_xml_encoding = None
@@ -1674,8 +1672,8 @@ class UnicodeDammit:
# UTF-16BE
sniffed_xml_encoding = 'utf-16be'
xml_data = unicode(xml_data, 'utf-16be').encode('utf-8')
- elif (len(xml_data) >= 4) and (xml_data[:2] == '\xfe\xff') \
- and (xml_data[2:4] != '\x00\x00'):
+ elif len(xml_data) >= 4 and xml_data[:2] == '\xfe\xff' and \
+ xml_data[2:4] != '\x00\x00':
# UTF-16BE with BOM
sniffed_xml_encoding = 'utf-16be'
xml_data = unicode(xml_data[2:], 'utf-16be').encode('utf-8')
@@ -1683,8 +1681,8 @@ class UnicodeDammit:
# UTF-16LE
sniffed_xml_encoding = 'utf-16le'
xml_data = unicode(xml_data, 'utf-16le').encode('utf-8')
- elif (len(xml_data) >= 4) and (xml_data[:2] == '\xff\xfe') and \
- (xml_data[2:4] != '\x00\x00'):
+ elif len(xml_data) >= 4 and xml_data[:2] == '\xff\xfe' and \
+ xml_data[2:4] != '\x00\x00':
# UTF-16LE with BOM
sniffed_xml_encoding = 'utf-16le'
xml_data = unicode(xml_data[2:], 'utf-16le').encode('utf-8')
@@ -1711,9 +1709,8 @@ class UnicodeDammit:
else:
sniffed_xml_encoding = 'ascii'
pass
- xml_encoding_match = re.compile \
- ('^<\?.*encoding=[\'"](.*?)[\'"].*\?>')\
- .match(xml_data)
+ xml_encoding_match = re.compile('^<\?.*encoding=[\'"](.*?)[\'"].*\?>')\
+ .match(xml_data)
except:
xml_encoding_match = None
if xml_encoding_match:
@@ -1726,15 +1723,14 @@ class UnicodeDammit:
xml_encoding = sniffed_xml_encoding
return xml_data, xml_encoding, sniffed_xml_encoding
-
def find_codec(self, charset):
- return self._codec(self.CHARSET_ALIASES.get(charset, charset)) \
- or (charset and self._codec(charset.replace("-", ""))) \
- or (charset and self._codec(charset.replace("-", "_"))) \
- or charset
+ return self._codec(self.CHARSET_ALIASES.get(charset, charset)) or \
+ (charset and self._codec(charset.replace("-", ""))) or \
+ (charset and self._codec(charset.replace("-", "_"))) or charset
def _codec(self, charset):
- if not charset: return charset
+ if not charset:
+ return charset
codec = None
try:
codecs.lookup(charset)
@@ -1744,29 +1740,29 @@ class UnicodeDammit:
return codec
EBCDIC_TO_ASCII_MAP = None
+
def _ebcdic_to_ascii(self, s):
c = self.__class__
if not c.EBCDIC_TO_ASCII_MAP:
- emap = (0,1,2,3,156,9,134,127,151,141,142,11,12,13,14,15,
- 16,17,18,19,157,133,8,135,24,25,146,143,28,29,30,31,
- 128,129,130,131,132,10,23,27,136,137,138,139,140,5,6,7,
- 144,145,22,147,148,149,150,4,152,153,154,155,20,21,158,26,
- 32,160,161,162,163,164,165,166,167,168,91,46,60,40,43,33,
- 38,169,170,171,172,173,174,175,176,177,93,36,42,41,59,94,
- 45,47,178,179,180,181,182,183,184,185,124,44,37,95,62,63,
- 186,187,188,189,190,191,192,193,194,96,58,35,64,39,61,34,
- 195,97,98,99,100,101,102,103,104,105,196,197,198,199,200,
- 201,202,106,107,108,109,110,111,112,113,114,203,204,205,
- 206,207,208,209,126,115,116,117,118,119,120,121,122,210,
- 211,212,213,214,215,216,217,218,219,220,221,222,223,224,
- 225,226,227,228,229,230,231,123,65,66,67,68,69,70,71,72,
- 73,232,233,234,235,236,237,125,74,75,76,77,78,79,80,81,
- 82,238,239,240,241,242,243,92,159,83,84,85,86,87,88,89,
- 90,244,245,246,247,248,249,48,49,50,51,52,53,54,55,56,57,
- 250,251,252,253,254,255)
- import string
- c.EBCDIC_TO_ASCII_MAP = string.maketrans( \
- ''.join(map(chr, range(256))), ''.join(map(chr, emap)))
+ emap = (0, 1, 2, 3, 156, 9, 134, 127, 151, 141, 142, 11, 12, 13, 14, 15,
+ 16, 17, 18, 19, 157, 133, 8, 135, 24, 25, 146, 143, 28, 29, 30, 31,
+ 128, 129, 130, 131, 132, 10, 23, 27, 136, 137, 138, 139, 140, 5, 6, 7,
+ 144, 145, 22, 147, 148, 149, 150, 4, 152, 153, 154, 155, 20, 21, 158, 26,
+ 32, 160, 161, 162, 163, 164, 165, 166, 167, 168, 91, 46, 60, 40, 43, 33,
+ 38, 169, 170, 171, 172, 173, 174, 175, 176, 177, 93, 36, 42, 41, 59, 94,
+ 45, 47, 178, 179, 180, 181, 182, 183, 184, 185, 124, 44, 37, 95, 62, 63,
+ 186, 187, 188, 189, 190, 191, 192, 193, 194, 96, 58, 35, 64, 39, 61, 34,
+ 195, 97, 98, 99, 100, 101, 102, 103, 104, 105, 196, 197, 198, 199, 200,
+ 201, 202, 106, 107, 108, 109, 110, 111, 112, 113, 114, 203, 204, 205,
+ 206, 207, 208, 209, 126, 115, 116, 117, 118, 119, 120, 121, 122, 210,
+ 211, 212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224,
+ 225, 226, 227, 228, 229, 230, 231, 123, 65, 66, 67, 68, 69, 70, 71, 72,
+ 73, 232, 233, 234, 235, 236, 237, 125, 74, 75, 76, 77, 78, 79, 80, 81,
+ 82, 238, 239, 240, 241, 242, 243, 92, 159, 83, 84, 85, 86, 87, 88, 89,
+ 90, 244, 245, 246, 247, 248, 249, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57,
+ 250, 251, 252, 253, 254, 255)
+ c.EBCDIC_TO_ASCII_MAP = string.maketrans(
+ ''.join(map(chr, range(256))), ''.join(map(chr, emap)))
return s.translate(c.EBCDIC_TO_ASCII_MAP)
MS_CHARS = { '\x80' : ('euro', '20AC'),
@@ -1800,13 +1796,12 @@ class UnicodeDammit:
'\x9c' : ('oelig', '153'),
'\x9d' : '?',
'\x9e' : ('#x17E', '17E'),
- '\x9f' : ('Yuml', '178'),}
+ '\x9f' : ('Yuml', '178') }
#######################################################################
-#By default, act as an HTML pretty-printer.
+# By default, act as an HTML pretty-printer.
if __name__ == '__main__':
- import sys
soup = BeautifulSoup(sys.stdin.read())
print soup.prettify()
diff --git a/scripts/microbes/create_bacteria_loc_file.py b/scripts/microbes/create_bacteria_loc_file.py
index 9ba28b5f5d0..c8c5e24a1b9 100644
--- a/scripts/microbes/create_bacteria_loc_file.py
+++ b/scripts/microbes/create_bacteria_loc_file.py
@@ -1,69 +1,72 @@
-#!/usr/bin/env python
-#Dan Blankenberg
-
-import sys, os
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-def __main__():
- base_dir = os.path.join( os.getcwd(), "bacteria" )
- try:
- base_dir = sys.argv[1]
- except:
- pass
- #print "using default base_dir:", base_dir
-
- organisms = {}
- for result in os.walk(base_dir):
- this_base_dir,sub_dirs,files = result
- for file in files:
- if file[-5:] == ".info":
- dict = {}
- info_file = open(os.path.join(this_base_dir,file),'r')
- info = info_file.readlines()
- info_file.close()
- for line in info:
- fields = line.replace("\n","").split("=")
- dict[fields[0]]="=".join(fields[1:])
- if 'genome project id' in dict.keys():
- name = dict['genome project id']
- if 'build' in dict.keys():
- name = dict['build']
- if name not in organisms.keys():
- organisms[name] = {'chrs':{},'base_dir':this_base_dir}
- for key in dict.keys():
- organisms[name][key]=dict[key]
- else:
- if dict['organism'] not in organisms.keys():
- organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir}
- organisms[dict['organism']]['chrs'][dict['chromosome']]=dict
- for org in organisms:
- org = organisms[org]
- #if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation
- try:
- build = org['genome project id']
- except: continue
- if 'build' in org:
- build = org['build']
- print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tUCSC" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] )
- else:
- print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tNone" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] )
-
- for chr in org['chrs']:
- chr = org['chrs'][chr]
- print "CHR\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], chr['name'], chr['length'], chr['gi'], chr['gb'], "http://www.ncbi.nlm.nih.gov/entrez/viewer.fcgi?db=nucleotide&val="+chr['refseq'] )
- for feature in ['CDS','tRNA','rRNA']:
- print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], feature, build, chr['chromosome'], feature, "bed", os.path.join( org['base_dir'], "%s.%s.bed" % ( chr['chromosome'], feature ) ) )
- #FASTA
- print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "seq", build, chr['chromosome'], "sequence", "fasta", os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] ) )
- #GeneMark
- if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) ):
- print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMark", build, chr['chromosome'], "GeneMark", "bed", os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) )
- #GenMarkHMM
- if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) ):
- print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMarkHMM", build, chr['chromosome'], "GeneMarkHMM", "bed", os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) )
- #Glimmer3
- if os.path.exists( os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) ):
- print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "Glimmer3", build, chr['chromosome'], "Glimmer3", "bed", os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) )
-
-if __name__ == "__main__": __main__()
+#!/usr/bin/env python
+# Dan Blankenberg
+import os
+import sys
+
+assert sys.version_info[:2] >= ( 2, 4 )
+
+
+def __main__():
+ base_dir = os.path.join( os.getcwd(), "bacteria" )
+ try:
+ base_dir = sys.argv[1]
+ except:
+ pass
+ # print "using default base_dir:", base_dir
+
+ organisms = {}
+ for result in os.walk(base_dir):
+ this_base_dir, sub_dirs, files = result
+ for file in files:
+ if file[-5:] == ".info":
+ dict = {}
+ info_file = open(os.path.join(this_base_dir, file), 'r')
+ info = info_file.readlines()
+ info_file.close()
+ for line in info:
+ fields = line.replace("\n", "").split("=")
+ dict[fields[0]] = "=".join(fields[1:])
+ if 'genome project id' in dict.keys():
+ name = dict['genome project id']
+ if 'build' in dict.keys():
+ name = dict['build']
+ if name not in organisms.keys():
+ organisms[name] = {'chrs': {}, 'base_dir': this_base_dir}
+ for key in dict.keys():
+ organisms[name][key] = dict[key]
+ else:
+ if dict['organism'] not in organisms.keys():
+ organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir}
+ organisms[dict['organism']]['chrs'][dict['chromosome']] = dict
+ for org in organisms:
+ org = organisms[org]
+ # if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation
+ try:
+ build = org['genome project id']
+ except:
+ continue
+ if 'build' in org:
+ build = org['build']
+ print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tUCSC" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] )
+ else:
+ print "ORG\t%s\t%s\t%s\t%s\t%s\t%s\tNone" % ( build, org['name'], org['kingdom'], org['group'], org['chromosomes'], org['info url'] )
+
+ for chr in org['chrs']:
+ chr = org['chrs'][chr]
+ print "CHR\t%s\t%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], chr['name'], chr['length'], chr['gi'], chr['gb'], "http://www.ncbi.nlm.nih.gov/entrez/viewer.fcgi?db=nucleotide&val=" + chr['refseq'] )
+ for feature in ['CDS', 'tRNA', 'rRNA']:
+ print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], feature, build, chr['chromosome'], feature, "bed", os.path.join( org['base_dir'], "%s.%s.bed" % ( chr['chromosome'], feature ) ) )
+ # FASTA
+ print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "seq", build, chr['chromosome'], "sequence", "fasta", os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] ) )
+ # GeneMark
+ if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) ):
+ print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMark", build, chr['chromosome'], "GeneMark", "bed", os.path.join( org['base_dir'], "%s.GeneMark.bed" % chr['chromosome'] ) )
+ # GenMarkHMM
+ if os.path.exists( os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) ):
+ print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "GeneMarkHMM", build, chr['chromosome'], "GeneMarkHMM", "bed", os.path.join( org['base_dir'], "%s.GeneMarkHMM.bed" % chr['chromosome'] ) )
+ # Glimmer3
+ if os.path.exists( os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) ):
+ print "DATA\t%s_%s_%s\t%s\t%s\t%s\t%s\t%s" % ( build, chr['chromosome'], "Glimmer3", build, chr['chromosome'], "Glimmer3", "bed", os.path.join( org['base_dir'], "%s.Glimmer3.bed" % chr['chromosome'] ) )
+
+if __name__ == "__main__":
+ __main__()
diff --git a/scripts/microbes/create_bacteria_table.py b/scripts/microbes/create_bacteria_table.py
index 52e6c537754..42c8435d197 100644
--- a/scripts/microbes/create_bacteria_table.py
+++ b/scripts/microbes/create_bacteria_table.py
@@ -1,79 +1,78 @@
-#!/usr/bin/env python
-#Dan Blankenberg
-
-import sys, os
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-def __main__():
- base_dir = os.path.join( os.getcwd(), "bacteria" )
- try:
- base_dir = sys.argv[1]
- except:
- pass
- #print "using default base_dir:", base_dir
-
- organisms = {}
- for result in os.walk(base_dir):
- this_base_dir,sub_dirs,files = result
- for file in files:
- if file[-5:] == ".info":
- dict = {}
- info_file = open(os.path.join(this_base_dir,file),'r')
- info = info_file.readlines()
- info_file.close()
- for line in info:
- fields = line.replace("\n","").split("=")
- dict[fields[0]]="=".join(fields[1:])
- if 'genome project id' in dict.keys():
- name = dict['genome project id']
- if 'build' in dict.keys():
- name = dict['build']
- if name not in organisms.keys():
- organisms[name] = {'chrs':{},'base_dir':this_base_dir}
- for key in dict.keys():
- organisms[name][key]=dict[key]
- else:
- if dict['organism'] not in organisms.keys():
- organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir}
- organisms[dict['organism']]['chrs'][dict['chromosome']]=dict
-
- orgs = organisms.keys()
- for org in orgs:
- if 'name' not in organisms[org]:del organisms[org]
-
- orgs = organisms.keys()
- #need to sort by name
- swap_test = False
- for i in range(0, len(orgs) - 1):
- for j in range(0, len(orgs) - i - 1):
- if organisms[orgs[j]]['name'] > organisms[orgs[j + 1]]['name']:
- orgs[j], orgs[j + 1] = orgs[j + 1], orgs[j]
- swap_test = True
- if swap_test is False:
- break
-
-
-
-
- print "||'''Organism'''||'''Kingdom'''||'''Group'''||'''Links to UCSC Archaea Browser'''||"
-
-
- for org in orgs:
- org = organisms[org]
- at_ucsc = False
- #if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation
- try:
- build = org['genome project id']
- except: continue
- if 'build' in org:
- build = org['build']
- at_ucsc = True
-
- out_str = "||"+org['name']+"||"+org['kingdom']+"||"+org['group']+"||"
- if at_ucsc:
- out_str = out_str + "Yes"
- out_str = out_str + "||"
- print out_str
-
-if __name__ == "__main__": __main__()
+#!/usr/bin/env python
+# Dan Blankenberg
+import os
+import sys
+
+assert sys.version_info[:2] >= ( 2, 4 )
+
+
+def __main__():
+ base_dir = os.path.join( os.getcwd(), "bacteria" )
+ try:
+ base_dir = sys.argv[1]
+ except:
+ pass
+ # print "using default base_dir:", base_dir
+
+ organisms = {}
+ for result in os.walk(base_dir):
+ this_base_dir, sub_dirs, files = result
+ for file in files:
+ if file[-5:] == ".info":
+ dict = {}
+ info_file = open(os.path.join(this_base_dir, file), 'r')
+ info = info_file.readlines()
+ info_file.close()
+ for line in info:
+ fields = line.replace("\n", "").split("=")
+ dict[fields[0]] = "=".join(fields[1:])
+ if 'genome project id' in dict.keys():
+ name = dict['genome project id']
+ if 'build' in dict.keys():
+ name = dict['build']
+ if name not in organisms.keys():
+ organisms[name] = {'chrs': {}, 'base_dir': this_base_dir}
+ for key in dict.keys():
+ organisms[name][key] = dict[key]
+ else:
+ if dict['organism'] not in organisms.keys():
+ organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir}
+ organisms[dict['organism']]['chrs'][dict['chromosome']] = dict
+
+ orgs = organisms.keys()
+ for org in orgs:
+ if 'name' not in organisms[org]:
+ del organisms[org]
+
+ orgs = organisms.keys()
+ # need to sort by name
+ swap_test = False
+ for i in range(0, len(orgs) - 1):
+ for j in range(0, len(orgs) - i - 1):
+ if organisms[orgs[j]]['name'] > organisms[orgs[j + 1]]['name']:
+ orgs[j], orgs[j + 1] = orgs[j + 1], orgs[j]
+ swap_test = True
+ if swap_test is False:
+ break
+
+ print "||'''Organism'''||'''Kingdom'''||'''Group'''||'''Links to UCSC Archaea Browser'''||"
+
+ for org in orgs:
+ org = organisms[org]
+ at_ucsc = False
+ # if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation
+ try:
+ org['genome project id']
+ except:
+ continue
+ if 'build' in org:
+ at_ucsc = True
+
+ out_str = "||" + org['name'] + "||" + org['kingdom'] + "||" + org['group'] + "||"
+ if at_ucsc:
+ out_str = out_str + "Yes"
+ out_str = out_str + "||"
+ print out_str
+
+if __name__ == "__main__":
+ __main__()
diff --git a/scripts/microbes/create_nib_seq_loc_file.py b/scripts/microbes/create_nib_seq_loc_file.py
index 06408cb028a..5b4e9241391 100644
--- a/scripts/microbes/create_nib_seq_loc_file.py
+++ b/scripts/microbes/create_nib_seq_loc_file.py
@@ -1,88 +1,87 @@
-#!/usr/bin/env python
-#Dan Blankenberg
-
-import sys, os
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-def __main__():
- base_dir = os.path.join( os.getcwd(), "bacteria" )
- try:
- base_dir = sys.argv[1]
- except:
- print "using default base_dir:", base_dir
-
- loc_out = os.path.join( base_dir, "seq.loc" )
- try:
- loc_out = os.path.join( base_dir, sys.argv[2] )
- except:
- print "using default seq.loc:", loc_out
-
-
-
- organisms = {}
-
- loc_out = open( loc_out, 'wb' )
-
- for result in os.walk( base_dir ):
- this_base_dir, sub_dirs, files = result
- for file in files:
- if file[-5:] == ".info":
- dict = {}
- info_file = open( os.path.join( this_base_dir, file ), 'r' )
- info = info_file.readlines()
- info_file.close()
- for line in info:
- fields = line.replace( "\n", "" ).split( "=" )
- dict[fields[0]]="=".join(fields[1:])
- if 'genome project id' in dict.keys():
- name = dict['genome project id']
- if 'build' in dict.keys():
- name = dict['build']
- if name not in organisms.keys():
- organisms[name] = {'chrs':{}, 'base_dir':this_base_dir}
- for key in dict.keys():
- organisms[name][key] = dict[key]
- else:
- if dict['organism'] not in organisms.keys():
- organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir}
- organisms[dict['organism']]['chrs'][dict['chromosome']] = dict
-
-
- for org in organisms:
- org = organisms[org]
- try:
- build = org['genome project id']
- except: continue
- if 'build' in org:
- build = org['build']
-
- seq_path = os.path.join( org['base_dir'], "seq" )
-
- #create seq dir, if exists go to next org
- ##TODO: add better checking, i.e. for updating
- try:
- os.mkdir( seq_path )
- except:
- print "Skipping", build
- #continue
-
- loc_out.write( "seq %s %s\n" % ( build, seq_path ) )
-
- #print org info
-
- for chr in org['chrs']:
- chr = org['chrs'][chr]
-
- fasta_file = os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] )
- nib_out_file = os.path.join( seq_path, "%s.nib "% chr['chromosome'] )
- #create nibs using faToNib binary
- #TODO: when bx supports writing nib, use it here instead
- command = "faToNib %s %s" % ( fasta_file, nib_out_file )
- os.system( command )
-
- loc_out.close()
-
-
-
-if __name__ == "__main__": __main__()
+#!/usr/bin/env python
+# Dan Blankenberg
+import os
+import sys
+
+assert sys.version_info[:2] >= ( 2, 4 )
+
+
+def __main__():
+ base_dir = os.path.join( os.getcwd(), "bacteria" )
+ try:
+ base_dir = sys.argv[1]
+ except:
+ print "using default base_dir:", base_dir
+
+ loc_out = os.path.join( base_dir, "seq.loc" )
+ try:
+ loc_out = os.path.join( base_dir, sys.argv[2] )
+ except:
+ print "using default seq.loc:", loc_out
+
+ organisms = {}
+
+ loc_out = open( loc_out, 'wb' )
+
+ for result in os.walk( base_dir ):
+ this_base_dir, sub_dirs, files = result
+ for file in files:
+ if file[-5:] == ".info":
+ dict = {}
+ info_file = open( os.path.join( this_base_dir, file ), 'r' )
+ info = info_file.readlines()
+ info_file.close()
+ for line in info:
+ fields = line.replace( "\n", "" ).split( "=" )
+ dict[fields[0]] = "=".join(fields[1:])
+ if 'genome project id' in dict.keys():
+ name = dict['genome project id']
+ if 'build' in dict.keys():
+ name = dict['build']
+ if name not in organisms.keys():
+ organisms[name] = {'chrs': {}, 'base_dir': this_base_dir}
+ for key in dict.keys():
+ organisms[name][key] = dict[key]
+ else:
+ if dict['organism'] not in organisms.keys():
+ organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir}
+ organisms[dict['organism']]['chrs'][dict['chromosome']] = dict
+
+ for org in organisms:
+ org = organisms[org]
+ try:
+ build = org['genome project id']
+ except:
+ continue
+ if 'build' in org:
+ build = org['build']
+
+ seq_path = os.path.join( org['base_dir'], "seq" )
+
+ # create seq dir, if exists go to next org
+ # TODO: add better checking, i.e. for updating
+ try:
+ os.mkdir( seq_path )
+ except:
+ print "Skipping", build
+ # continue
+
+ loc_out.write( "seq %s %s\n" % ( build, seq_path ) )
+
+ # print org info
+
+ for chr in org['chrs']:
+ chr = org['chrs'][chr]
+
+ fasta_file = os.path.join( org['base_dir'], "%s.fna" % chr['chromosome'] )
+ nib_out_file = os.path.join( seq_path, "%s.nib " % chr['chromosome'] )
+ # create nibs using faToNib binary
+ # TODO: when bx supports writing nib, use it here instead
+ command = "faToNib %s %s" % ( fasta_file, nib_out_file )
+ os.system( command )
+
+ loc_out.close()
+
+
+if __name__ == "__main__":
+ __main__()
diff --git a/scripts/microbes/get_builds_lengths.py b/scripts/microbes/get_builds_lengths.py
index 3c473cb7393..398abcebaa8 100644
--- a/scripts/microbes/get_builds_lengths.py
+++ b/scripts/microbes/get_builds_lengths.py
@@ -1,56 +1,59 @@
-#!/usr/bin/env python
-#Dan Blankenberg
-
-import sys, os
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-def __main__():
- base_dir = os.path.join( os.getcwd(), "bacteria" )
- try:
- base_dir = sys.argv[1]
- except:
- pass
- #print "using default base_dir:", base_dir
-
- organisms = {}
- for result in os.walk(base_dir):
- this_base_dir,sub_dirs,files = result
- for file in files:
- if file[-5:] == ".info":
- dict = {}
- info_file = open(os.path.join(this_base_dir,file),'r')
- info = info_file.readlines()
- info_file.close()
- for line in info:
- fields = line.replace("\n","").split("=")
- dict[fields[0]]="=".join(fields[1:])
- if 'genome project id' in dict.keys():
- name = dict['genome project id']
- if 'build' in dict.keys():
- name = dict['build']
- if name not in organisms.keys():
- organisms[name] = {'chrs':{},'base_dir':this_base_dir}
- for key in dict.keys():
- organisms[name][key]=dict[key]
- else:
- if dict['organism'] not in organisms.keys():
- organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir}
- organisms[dict['organism']]['chrs'][dict['chromosome']]=dict
- for org in organisms:
- org = organisms[org]
- #if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation
- try:
- build = org['genome project id']
- except: continue
-
- if 'build' in org:
- build = org['build']
-
- chrs=[]
- for chrom in org['chrs']:
- chrom = org['chrs'][chrom]
- chrs.append( "%s=%s" % ( chrom['chromosome'], chrom['length'] ) )
- print "%s\t%s\t%s" % ( build, org['name'], ",".join( chrs ) )
-
-if __name__ == "__main__": __main__()
+#!/usr/bin/env python
+# Dan Blankenberg
+import os
+import sys
+
+assert sys.version_info[:2] >= ( 2, 4 )
+
+
+def __main__():
+ base_dir = os.path.join( os.getcwd(), "bacteria" )
+ try:
+ base_dir = sys.argv[1]
+ except:
+ pass
+ # print "using default base_dir:", base_dir
+
+ organisms = {}
+ for result in os.walk(base_dir):
+ this_base_dir, sub_dirs, files = result
+ for file in files:
+ if file[-5:] == ".info":
+ dict = {}
+ info_file = open(os.path.join(this_base_dir, file), 'r')
+ info = info_file.readlines()
+ info_file.close()
+ for line in info:
+ fields = line.replace("\n", "").split("=")
+ dict[fields[0]] = "=".join(fields[1:])
+ if 'genome project id' in dict.keys():
+ name = dict['genome project id']
+ if 'build' in dict.keys():
+ name = dict['build']
+ if name not in organisms.keys():
+ organisms[name] = {'chrs': {}, 'base_dir': this_base_dir}
+ for key in dict.keys():
+ organisms[name][key] = dict[key]
+ else:
+ if dict['organism'] not in organisms.keys():
+ organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir}
+ organisms[dict['organism']]['chrs'][dict['chromosome']] = dict
+ for org in organisms:
+ org = organisms[org]
+ # if no gpi, then must be a ncbi chr which corresponds to a UCSC org, w/o matching UCSC designation
+ try:
+ build = org['genome project id']
+ except:
+ continue
+
+ if 'build' in org:
+ build = org['build']
+
+ chrs = []
+ for chrom in org['chrs']:
+ chrom = org['chrs'][chrom]
+ chrs.append( "%s=%s" % ( chrom['chromosome'], chrom['length'] ) )
+ print "%s\t%s\t%s" % ( build, org['name'], ",".join( chrs ) )
+
+if __name__ == "__main__":
+ __main__()
diff --git a/scripts/microbes/harvest_bacteria.py b/scripts/microbes/harvest_bacteria.py
index 4e420ad305c..09fd6f6a17c 100644
--- a/scripts/microbes/harvest_bacteria.py
+++ b/scripts/microbes/harvest_bacteria.py
@@ -1,246 +1,254 @@
-#!/usr/bin/env python
-#Dan Blankenberg
-
-#Harvest Bacteria
-#Connects to NCBI's Microbial Genome Projects website and scrapes it for information.
-#Downloads and converts annotations for each Genome
-
-import sys, os, time
-from urllib2 import urlopen
-from urllib import urlretrieve
-from ftplib import FTP
-from BeautifulSoup import BeautifulSoup
-from util import get_bed_from_genbank, get_bed_from_glimmer3, get_bed_from_GeneMarkHMM, get_bed_from_GeneMark
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-#this defines the types of ftp files we are interested in, and how to process/convert them to a form for our use
-desired_ftp_files = {'GeneMark':{'ext':'GeneMark-2.5f','parser':'process_GeneMark'},
- 'GeneMarkHMM':{'ext':'GeneMarkHMM-2.6m','parser':'process_GeneMarkHMM'},
- 'Glimmer3':{'ext':'Glimmer3','parser':'process_Glimmer3'},
- 'fna':{'ext':'fna','parser':'process_FASTA'},
- 'gbk':{'ext':'gbk','parser':'process_Genbank'} }
-
-
-
-#number, name, chroms, kingdom, group, genbank, refseq, info_url, ftp_url
-def iter_genome_projects( url = "http://www.ncbi.nlm.nih.gov/genomes/lproks.cgi?view=1", info_url_base = "http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ):
- for row in BeautifulSoup( urlopen( url ) ).findAll( name = 'tr', bgcolor = ["#EEFFDD", "#E8E8DD"] ):
- row = str( row ).replace( "\n", "" ).replace( "\r", "" )
-
- fields = row.split( "" )
-
- org_num = fields[0].split( "list_uids=" )[-1].split( "\"" )[0]
-
- name = fields[1].split( "\">" )[-1].split( "<" )[0]
-
- kingdom = "archaea"
- if "| B" in fields[2]:
- kingdom = "bacteria"
-
- group = fields[3].split( ">" )[-1]
-
- info_url = "%s%s" % ( info_url_base, org_num )
-
- org_genbank = fields[7].split( "\">" )[-1].split( "<" )[0].split( "." )[0]
- org_refseq = fields[8].split( "\">" )[-1].split( "<" )[0].split( "." )[0]
-
- #seems some things donot have an ftp url, try and except it here:
- try:
- ftp_url = fields[22].split( "href=\"" )[1].split( "\"" )[0]
- except:
- print "FAILED TO AQUIRE FTP ADDRESS:", org_num, info_url
- ftp_url = None
-
- chroms = get_chroms_by_project_id( org_num )
-
- yield org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url
-
-def get_chroms_by_project_id( org_num, base_url = "http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ):
- html_count = 0
- html = None
- while html_count < 500 and html == None:
- html_count += 1
- url = "%s%s" % ( base_url, org_num )
- try:
- html = urlopen( url )
- except:
- print "GENOME PROJECT FAILED:", html_count, "org:", org_num, url
- html = None
- time.sleep( 1 ) #Throttle Connection
- if html is None:
- "GENOME PROJECT COMPLETELY FAILED TO LOAD", "org:", org_num,"http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids="+org_num
- return None
-
- chroms = []
- for chr_row in BeautifulSoup( html ).findAll( "tr", { "class" : "vvv" } ):
- chr_row = str( chr_row ).replace( "\n","" ).replace( "\r", "" )
- fields2 = chr_row.split( " | " )
- refseq = fields2[1].split( "" )[0].split( ">" )[-1]
- #genbank = fields2[2].split( "" )[0].split( ">" )[-1]
- chroms.append( refseq )
-
- return chroms
-
-def get_ftp_contents( ftp_url ):
- ftp_count = 0
- ftp_contents = None
- while ftp_count < 500 and ftp_contents == None:
- ftp_count += 1
- try:
- ftp = FTP( ftp_url.split("/")[2] )
- ftp.login()
- ftp.cwd( ftp_url.split( ftp_url.split( "/" )[2] )[-1] )
- ftp_contents = ftp.nlst()
- ftp.close()
- except:
- ftp_contents = None
- time.sleep( 1 ) #Throttle Connection
- return ftp_contents
-
-def scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url ):
- for file_type, items in desired_ftp_files.items():
- ext = items['ext']
- ftp_filename = "%s.%s" % ( refseq, ext )
- target_filename = os.path.join( org_dir, "%s.%s" % ( refseq, ext ) )
- if ftp_filename in ftp_contents:
- url_count = 0
- url = "%s/%s" % ( ftp_url, ftp_filename )
- results = None
- while url_count < 500 and results is None:
- url_count += 1
- try:
- results = urlretrieve( url, target_filename )
- except:
- results = None
- time.sleep(1) #Throttle Connection
- if results is None:
- "URL COMPLETELY FAILED TO LOAD:", url
- return
-
- #do special processing for each file type:
- if items['parser'] is not None:
- parser_results = globals()[items['parser']]( target_filename, org_num, refseq )
- else:
- print "FTP filetype:", file_type, "not found for", org_num, refseq
- #FTP Files have been Loaded
-
-
-def process_FASTA( filename, org_num, refseq ):
- fasta = []
- fasta = [line.strip() for line in open( filename, 'rb' ).readlines()]
- fasta_header = fasta.pop( 0 )[1:]
- fasta_header_split = fasta_header.split( "|" )
- chr_name = fasta_header_split.pop( -1 ).strip()
- accesions = {fasta_header_split[0]:fasta_header_split[1], fasta_header_split[2]:fasta_header_split[3]}
- fasta = "".join( fasta )
-
- #Create Chrom Info File:
- chrom_info_file = open( os.path.join( os.path.split( filename )[0], "%s.info" % refseq ), 'wb+' )
- chrom_info_file.write( "chromosome=%s\nname=%s\nlength=%s\norganism=%s\n" % ( refseq, chr_name, len( fasta ), org_num ) )
- try:
- chrom_info_file.write( "gi=%s\n" % accesions['gi'] )
- except:
- chrom_info_file.write( "gi=None\n" )
- try:
- chrom_info_file.write( "gb=%s\n" % accesions['gb'] )
- except:
- chrom_info_file.write( "gb=None\n" )
- try:
- chrom_info_file.write( "refseq=%s\n" % refseq )
- except:
- chrom_info_file.write( "refseq=None\n" )
- chrom_info_file.close()
-
-def process_Genbank( filename, org_num, refseq ):
- #extracts 'CDS', 'tRNA', 'rRNA' features from genbank file
- features = get_bed_from_genbank( filename, refseq, ['CDS', 'tRNA', 'rRNA'] )
- for feature in features.keys():
- feature_file = open( os.path.join( os.path.split( filename )[0], "%s.%s.bed" % ( refseq, feature ) ), 'wb+' )
- feature_file.write( '\n'.join( features[feature] ) )
- feature_file.close()
- print "Genbank extraction finished for chrom:", refseq, "file:", filename
-
-def process_Glimmer3( filename, org_num, refseq ):
- try:
- glimmer3_bed = get_bed_from_glimmer3( filename, refseq )
- except Exception, e:
- print "Converting Glimmer3 to bed FAILED! For chrom:", refseq, "file:", filename, e
- glimmer3_bed = []
- glimmer3_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.Glimmer3.bed" % refseq ), 'wb+' )
- glimmer3_bed_file.write( '\n'.join( glimmer3_bed ) )
- glimmer3_bed_file.close()
-
-def process_GeneMarkHMM( filename, org_num, refseq ):
- try:
- geneMarkHMM_bed = get_bed_from_GeneMarkHMM( filename, refseq )
- except Exception, e:
- print "Converting GeneMarkHMM to bed FAILED! For chrom:", refseq, "file:", filename, e
- geneMarkHMM_bed = []
- geneMarkHMM_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMarkHMM.bed" % refseq ), 'wb+' )
- geneMarkHMM_bed_bed_file.write( '\n'.join( geneMarkHMM_bed ) )
- geneMarkHMM_bed_bed_file.close()
-
-def process_GeneMark( filename, org_num, refseq ):
- try:
- geneMark_bed = get_bed_from_GeneMark( filename, refseq )
- except Exception, e:
- print "Converting GeneMark to bed FAILED! For chrom:", refseq, "file:", filename, e
- geneMark_bed = []
- geneMark_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMark.bed" % refseq ), 'wb+' )
- geneMark_bed_bed_file.write( '\n'.join( geneMark_bed ) )
- geneMark_bed_bed_file.close()
-
-
-
-def __main__():
- start_time = time.time()
- base_dir = os.path.join( os.getcwd(), "bacteria" )
- try:
- base_dir = sys.argv[1]
- except:
- print "using default base_dir:", base_dir
-
- try:
- os.mkdir( base_dir )
- print "path '%s' has been created" % base_dir
- except:
- print "path '%s' seems to already exist" % base_dir
-
- for org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url in iter_genome_projects():
- if chroms is None:
- continue #No chrom information, we can't really do anything with this organism
- #Create org directory, if exists, assume it is done and complete --> skip it
- try:
- org_dir = os.path.join( base_dir, org_num )
- os.mkdir( org_dir )
- except:
- print "Organism %s already exists on disk, skipping" % org_num
- continue
-
- #get ftp contents
- ftp_contents = get_ftp_contents( ftp_url )
- if ftp_contents is None:
- "FTP COMPLETELY FAILED TO LOAD", "org:", org_num, "ftp:", ftp_url
- else:
- for refseq in chroms:
- ftp_result = scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url )
- #FTP Files have been Loaded
- print "Org:", org_num, "chrom:", refseq, "[", time.time() - start_time, "seconds elapsed. ]"
-
- #Create org info file
- info_file = open( os.path.join( org_dir, "%s.info" % org_num ), 'wb+' )
- info_file.write("genome project id=%s\n" % org_num )
- info_file.write("name=%s\n" % name )
- info_file.write("kingdom=%s\n" % kingdom )
- info_file.write("group=%s\n" % group )
- info_file.write("chromosomes=%s\n" % ",".join( chroms ) )
- info_file.write("info url=%s\n" % info_url )
- info_file.write("ftp url=%s\n" % ftp_url )
- info_file.close()
-
- print "Finished Harvesting", "[", time.time() - start_time, "seconds elapsed. ]"
- print "[", ( time.time() - start_time )/60, "minutes. ]"
- print "[", ( time.time() - start_time )/60/60, "hours. ]"
-
-if __name__ == "__main__": __main__()
+#!/usr/bin/env python
+# Dan Blankenberg
+
+# Harvest Bacteria
+# Connects to NCBI's Microbial Genome Projects website and scrapes it for information.
+# Downloads and converts annotations for each Genome
+import os
+import sys
+import time
+from ftplib import FTP
+from urllib2 import urlopen
+from urllib import urlretrieve
+
+from BeautifulSoup import BeautifulSoup
+from util import get_bed_from_genbank, get_bed_from_glimmer3, get_bed_from_GeneMarkHMM, get_bed_from_GeneMark
+
+assert sys.version_info[:2] >= ( 2, 4 )
+
+# this defines the types of ftp files we are interested in, and how to process/convert them to a form for our use
+desired_ftp_files = {'GeneMark': {'ext': 'GeneMark-2.5f', 'parser': 'process_GeneMark'},
+ 'GeneMarkHMM': {'ext': 'GeneMarkHMM-2.6m', 'parser': 'process_GeneMarkHMM'},
+ 'Glimmer3': {'ext': 'Glimmer3', 'parser': 'process_Glimmer3'},
+ 'fna': {'ext': 'fna', 'parser': 'process_FASTA'},
+ 'gbk': {'ext': 'gbk', 'parser': 'process_Genbank'} }
+
+
+# number, name, chroms, kingdom, group, genbank, refseq, info_url, ftp_url
+def iter_genome_projects( url="http://www.ncbi.nlm.nih.gov/genomes/lproks.cgi?view=1", info_url_base="http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ):
+ for row in BeautifulSoup( urlopen( url ) ).findAll( name='tr', bgcolor=["#EEFFDD", "#E8E8DD"] ):
+ row = str( row ).replace( "\n", "" ).replace( "\r", "" )
+
+ fields = row.split( "" )
+
+ org_num = fields[0].split( "list_uids=" )[-1].split( "\"" )[0]
+
+ name = fields[1].split( "\">" )[-1].split( "<" )[0]
+
+ kingdom = "archaea"
+ if "B" in fields[2]:
+ kingdom = "bacteria"
+
+ group = fields[3].split( ">" )[-1]
+
+ info_url = "%s%s" % ( info_url_base, org_num )
+
+ org_genbank = fields[7].split( "\">" )[-1].split( "<" )[0].split( "." )[0]
+ org_refseq = fields[8].split( "\">" )[-1].split( "<" )[0].split( "." )[0]
+
+ # seems some things donot have an ftp url, try and except it here:
+ try:
+ ftp_url = fields[22].split( "href=\"" )[1].split( "\"" )[0]
+ except:
+ print "FAILED TO AQUIRE FTP ADDRESS:", org_num, info_url
+ ftp_url = None
+
+ chroms = get_chroms_by_project_id( org_num )
+
+ yield org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url
+
+
+def get_chroms_by_project_id( org_num, base_url="http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" ):
+ html_count = 0
+ html = None
+ while html_count < 500 and html is None:
+ html_count += 1
+ url = "%s%s" % ( base_url, org_num )
+ try:
+ html = urlopen( url )
+ except:
+ print "GENOME PROJECT FAILED:", html_count, "org:", org_num, url
+ html = None
+ time.sleep( 1 ) # Throttle Connection
+ if html is None:
+ "GENOME PROJECT COMPLETELY FAILED TO LOAD", "org:", org_num, "http://www.ncbi.nlm.nih.gov/entrez/query.fcgi?db=genomeprj&cmd=Retrieve&dopt=Overview&list_uids=" + org_num
+ return None
+
+ chroms = []
+ for chr_row in BeautifulSoup( html ).findAll( "tr", { "class" : "vvv" } ):
+ chr_row = str( chr_row ).replace( "\n", "" ).replace( "\r", "" )
+ fields2 = chr_row.split( " | " )
+ refseq = fields2[1].split( "" )[0].split( ">" )[-1]
+ # genbank = fields2[2].split( "" )[0].split( ">" )[-1]
+ chroms.append( refseq )
+
+ return chroms
+
+
+def get_ftp_contents( ftp_url ):
+ ftp_count = 0
+ ftp_contents = None
+ while ftp_count < 500 and ftp_contents is None:
+ ftp_count += 1
+ try:
+ ftp = FTP( ftp_url.split("/")[2] )
+ ftp.login()
+ ftp.cwd( ftp_url.split( ftp_url.split( "/" )[2] )[-1] )
+ ftp_contents = ftp.nlst()
+ ftp.close()
+ except:
+ ftp_contents = None
+ time.sleep( 1 ) # Throttle Connection
+ return ftp_contents
+
+
+def scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url ):
+ for file_type, items in desired_ftp_files.items():
+ ext = items['ext']
+ ftp_filename = "%s.%s" % ( refseq, ext )
+ target_filename = os.path.join( org_dir, "%s.%s" % ( refseq, ext ) )
+ if ftp_filename in ftp_contents:
+ url_count = 0
+ url = "%s/%s" % ( ftp_url, ftp_filename )
+ results = None
+ while url_count < 500 and results is None:
+ url_count += 1
+ try:
+ results = urlretrieve( url, target_filename )
+ except:
+ results = None
+ time.sleep(1) # Throttle Connection
+ if results is None:
+ "URL COMPLETELY FAILED TO LOAD:", url
+ return
+
+ # do special processing for each file type:
+ if items['parser'] is not None:
+ globals()[items['parser']]( target_filename, org_num, refseq )
+ else:
+ print "FTP filetype:", file_type, "not found for", org_num, refseq
+ # FTP Files have been Loaded
+
+
+def process_FASTA( filename, org_num, refseq ):
+ fasta = []
+ fasta = [line.strip() for line in open( filename, 'rb' ).readlines()]
+ fasta_header = fasta.pop( 0 )[1:]
+ fasta_header_split = fasta_header.split( "|" )
+ chr_name = fasta_header_split.pop( -1 ).strip()
+ accesions = {fasta_header_split[0]: fasta_header_split[1], fasta_header_split[2]: fasta_header_split[3]}
+ fasta = "".join( fasta )
+
+ # Create Chrom Info File:
+ chrom_info_file = open( os.path.join( os.path.split( filename )[0], "%s.info" % refseq ), 'wb+' )
+ chrom_info_file.write( "chromosome=%s\nname=%s\nlength=%s\norganism=%s\n" % ( refseq, chr_name, len( fasta ), org_num ) )
+ try:
+ chrom_info_file.write( "gi=%s\n" % accesions['gi'] )
+ except:
+ chrom_info_file.write( "gi=None\n" )
+ try:
+ chrom_info_file.write( "gb=%s\n" % accesions['gb'] )
+ except:
+ chrom_info_file.write( "gb=None\n" )
+ try:
+ chrom_info_file.write( "refseq=%s\n" % refseq )
+ except:
+ chrom_info_file.write( "refseq=None\n" )
+ chrom_info_file.close()
+
+
+def process_Genbank( filename, org_num, refseq ):
+ # extracts 'CDS', 'tRNA', 'rRNA' features from genbank file
+ features = get_bed_from_genbank( filename, refseq, ['CDS', 'tRNA', 'rRNA'] )
+ for feature in features.keys():
+ feature_file = open( os.path.join( os.path.split( filename )[0], "%s.%s.bed" % ( refseq, feature ) ), 'wb+' )
+ feature_file.write( '\n'.join( features[feature] ) )
+ feature_file.close()
+ print "Genbank extraction finished for chrom:", refseq, "file:", filename
+
+
+def process_Glimmer3( filename, org_num, refseq ):
+ try:
+ glimmer3_bed = get_bed_from_glimmer3( filename, refseq )
+ except Exception, e:
+ print "Converting Glimmer3 to bed FAILED! For chrom:", refseq, "file:", filename, e
+ glimmer3_bed = []
+ glimmer3_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.Glimmer3.bed" % refseq ), 'wb+' )
+ glimmer3_bed_file.write( '\n'.join( glimmer3_bed ) )
+ glimmer3_bed_file.close()
+
+
+def process_GeneMarkHMM( filename, org_num, refseq ):
+ try:
+ geneMarkHMM_bed = get_bed_from_GeneMarkHMM( filename, refseq )
+ except Exception, e:
+ print "Converting GeneMarkHMM to bed FAILED! For chrom:", refseq, "file:", filename, e
+ geneMarkHMM_bed = []
+ geneMarkHMM_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMarkHMM.bed" % refseq ), 'wb+' )
+ geneMarkHMM_bed_bed_file.write( '\n'.join( geneMarkHMM_bed ) )
+ geneMarkHMM_bed_bed_file.close()
+
+
+def process_GeneMark( filename, org_num, refseq ):
+ try:
+ geneMark_bed = get_bed_from_GeneMark( filename, refseq )
+ except Exception, e:
+ print "Converting GeneMark to bed FAILED! For chrom:", refseq, "file:", filename, e
+ geneMark_bed = []
+ geneMark_bed_bed_file = open( os.path.join( os.path.split( filename )[0], "%s.GeneMark.bed" % refseq ), 'wb+' )
+ geneMark_bed_bed_file.write( '\n'.join( geneMark_bed ) )
+ geneMark_bed_bed_file.close()
+
+
+def __main__():
+ start_time = time.time()
+ base_dir = os.path.join( os.getcwd(), "bacteria" )
+ try:
+ base_dir = sys.argv[1]
+ except:
+ print "using default base_dir:", base_dir
+
+ try:
+ os.mkdir( base_dir )
+ print "path '%s' has been created" % base_dir
+ except:
+ print "path '%s' seems to already exist" % base_dir
+
+ for org_num, name, chroms, kingdom, group, org_genbank, org_refseq, info_url, ftp_url in iter_genome_projects():
+ if chroms is None:
+ continue # No chrom information, we can't really do anything with this organism
+ # Create org directory, if exists, assume it is done and complete --> skip it
+ try:
+ org_dir = os.path.join( base_dir, org_num )
+ os.mkdir( org_dir )
+ except:
+ print "Organism %s already exists on disk, skipping" % org_num
+ continue
+
+ # get ftp contents
+ ftp_contents = get_ftp_contents( ftp_url )
+ if ftp_contents is None:
+ "FTP COMPLETELY FAILED TO LOAD", "org:", org_num, "ftp:", ftp_url
+ else:
+ for refseq in chroms:
+ scrape_ftp( ftp_contents, org_dir, org_num, refseq, ftp_url )
+ # FTP Files have been Loaded
+ print "Org:", org_num, "chrom:", refseq, "[", time.time() - start_time, "seconds elapsed. ]"
+
+ # Create org info file
+ info_file = open( os.path.join( org_dir, "%s.info" % org_num ), 'wb+' )
+ info_file.write("genome project id=%s\n" % org_num )
+ info_file.write("name=%s\n" % name )
+ info_file.write("kingdom=%s\n" % kingdom )
+ info_file.write("group=%s\n" % group )
+ info_file.write("chromosomes=%s\n" % ",".join( chroms ) )
+ info_file.write("info url=%s\n" % info_url )
+ info_file.write("ftp url=%s\n" % ftp_url )
+ info_file.close()
+
+ print "Finished Harvesting", "[", time.time() - start_time, "seconds elapsed. ]"
+ print "[", ( time.time() - start_time ) / 60, "minutes. ]"
+ print "[", ( time.time() - start_time ) / 60 / 60, "hours. ]"
+
+if __name__ == "__main__":
+ __main__()
diff --git a/scripts/microbes/ncbi_to_ucsc.py b/scripts/microbes/ncbi_to_ucsc.py
index 17916f6ca8a..6f37bb1eb95 100644
--- a/scripts/microbes/ncbi_to_ucsc.py
+++ b/scripts/microbes/ncbi_to_ucsc.py
@@ -1,15 +1,14 @@
#!/usr/bin/env python
-
"""
Walk downloaded Genome Projects and Convert, in place, IDs to match the UCSC Archaea browser, where applicable.
Uses UCSC Archaea DSN.
"""
-
-import sys, os
+import os
+import sys
import urllib
-from xml.etree import ElementTree
-from BeautifulSoup import BeautifulSoup
from shutil import move
+from xml.etree import ElementTree
+
def __main__():
base_dir = os.path.join( os.getcwd(), "bacteria" )
@@ -20,31 +19,30 @@ def __main__():
organisms = {}
for result in os.walk(base_dir):
- this_base_dir,sub_dirs,files = result
+ this_base_dir, sub_dirs, files = result
for file in files:
if file[-5:] == ".info":
dict = {}
- info_file = open(os.path.join(this_base_dir,file),'r')
+ info_file = open(os.path.join(this_base_dir, file), 'r')
info = info_file.readlines()
info_file.close()
for line in info:
- fields = line.replace("\n","").split("=")
- dict[fields[0]]="=".join(fields[1:])
+ fields = line.replace("\n", "").split("=")
+ dict[fields[0]] = "=".join(fields[1:])
if 'genome project id' in dict.keys():
if dict['genome project id'] not in organisms.keys():
- organisms[dict['genome project id']] = {'chrs':{},'base_dir':this_base_dir}
+ organisms[dict['genome project id']] = {'chrs': {}, 'base_dir': this_base_dir}
for key in dict.keys():
- organisms[dict['genome project id']][key]=dict[key]
+ organisms[dict['genome project id']][key] = dict[key]
else:
if dict['organism'] not in organisms.keys():
- organisms[dict['organism']] = {'chrs':{},'base_dir':this_base_dir}
- organisms[dict['organism']]['chrs'][dict['chromosome']]=dict
+ organisms[dict['organism']] = {'chrs': {}, 'base_dir': this_base_dir}
+ organisms[dict['organism']]['chrs'][dict['chromosome']] = dict
- ##get UCSC data
+ # get UCSC data
URL = "http://archaea.ucsc.edu/cgi-bin/das/dsn"
-
try:
page = urllib.urlopen(URL)
except:
@@ -60,21 +58,19 @@ def __main__():
print "?\tunspecified (?)"
sys.exit(1)
-
builds = {}
-
- #print "#Harvested from http://archaea.ucsc.edu/cgi-bin/das/dsn"
- #print "?\tunspecified (?)"
+ # print "#Harvested from http://archaea.ucsc.edu/cgi-bin/das/dsn"
+ # print "?\tunspecified (?)"
for dsn in tree:
build = dsn.find("SOURCE").attrib['id']
try:
- org_page = urllib.urlopen("http://archaea.ucsc.edu/cgi-bin/hgGateway?db="+build).read().replace("\n","").split("")[0].split("")
+ org_page = urllib.urlopen("http://archaea.ucsc.edu/cgi-bin/hgGateway?db=" + build).read().replace("\n", "").split("")[0].split("")
except:
- print "NO CHROMS FOR",build
+ print "NO CHROMS FOR", build
continue
org_page.pop(0)
- if org_page[-1]=="":
+ if org_page[-1] == "":
org_page.pop(-1)
for row in org_page:
@@ -82,65 +78,64 @@ def __main__():
refseq = row.split("")[-2].split(">")[-1]
for org in organisms:
for org_chr in organisms[org]['chrs']:
- if organisms[org]['chrs'][org_chr]['chromosome']==refseq:
+ if organisms[org]['chrs'][org_chr]['chromosome'] == refseq:
if org not in builds:
- builds[org]={'chrs':{},'build':build}
- builds[org]['chrs'][refseq]=chr
- #print build,org,chr,refseq
+ builds[org] = {'chrs': {}, 'build': build}
+ builds[org]['chrs'][refseq] = chr
+ # print build,org,chr,refseq
print
ext_to_edit = ['bed', 'info', ]
for org in builds:
- print org,"changed to",builds[org]['build']
+ print org, "changed to", builds[org]['build']
- #org info file
- info_file_old = os.path.join(base_dir+org,org+".info")
- info_file_new = os.path.join(base_dir+org,builds[org]['build']+".info")
+ # org info file
+ info_file_old = os.path.join(base_dir + org, org + ".info")
+ info_file_new = os.path.join(base_dir + org, builds[org]['build'] + ".info")
+ old_dir = base_dir + org
+ new_dir = base_dir + builds[org]['build']
- old_dir = base_dir+org
- new_dir = base_dir+builds[org]['build']
-
- #open and edit org info file
+ # open and edit org info file
info_file_contents = open(info_file_old).read()
- info_file_contents = info_file_contents+"build="+builds[org]['build']+"\n"
+ info_file_contents = info_file_contents + "build=" + builds[org]['build'] + "\n"
for chrom in builds[org]['chrs']:
- info_file_contents = info_file_contents.replace(chrom,builds[org]['chrs'][chrom])
- for result in os.walk(base_dir+org):
- this_base_dir,sub_dirs,files = result
+ info_file_contents = info_file_contents.replace(chrom, builds[org]['chrs'][chrom])
+ for result in os.walk(base_dir + org):
+ this_base_dir, sub_dirs, files = result
for file in files:
- if file[0:len(chrom)]==chrom:
- #rename file
- old_name = os.path.join(this_base_dir,file)
- new_name = os.path.join(this_base_dir,builds[org]['chrs'][chrom]+file[len(chrom):])
- move(old_name,new_name)
+ if file[0:len(chrom)] == chrom:
+ # rename file
+ old_name = os.path.join(this_base_dir, file)
+ new_name = os.path.join(this_base_dir, builds[org]['chrs'][chrom] + file[len(chrom):])
+ move(old_name, new_name)
- #edit contents of file, skiping those in list
- if file.split(".")[-1] not in ext_to_edit: continue
+ # edit contents of file, skiping those in list
+ if file.split(".")[-1] not in ext_to_edit:
+ continue
file_contents = open(new_name).read()
- file_contents = file_contents.replace(chrom,builds[org]['chrs'][chrom])
+ file_contents = file_contents.replace(chrom, builds[org]['chrs'][chrom])
- #special case fixes...
+ # special case fixes...
if file[-5:] == ".info":
- file_contents = file_contents.replace("organism="+org,"organism="+builds[org]['build'])
- file_contents = file_contents.replace("refseq="+builds[org]['chrs'][chrom],"refseq="+chrom)
+ file_contents = file_contents.replace("organism=" + org, "organism=" + builds[org]['build'])
+ file_contents = file_contents.replace("refseq=" + builds[org]['chrs'][chrom], "refseq=" + chrom)
- #write out new file
- file_out = open(new_name,'w')
+ # write out new file
+ file_out = open(new_name, 'w')
file_out.write(file_contents)
file_out.close()
-
-
- #write out org info file and remove old file
- org_info_out = open(info_file_new,'w')
+ # write out org info file and remove old file
+ org_info_out = open(info_file_new, 'w')
org_info_out.write(info_file_contents)
org_info_out.close()
os.unlink(info_file_old)
- #change org directory name
- move(old_dir,new_dir)
-
-
-if __name__ == "__main__": __main__()
+ # change org directory name
+ move(old_dir, new_dir)
+
+
+if __name__ == "__main__":
+ __main__()
diff --git a/scripts/microbes/util.py b/scripts/microbes/util.py
index a16a5795d25..b4a05a93e5b 100644
--- a/scripts/microbes/util.py
+++ b/scripts/microbes/util.py
@@ -1,224 +1,235 @@
-#!/usr/bin/env python
-#Dan Blankenberg
-
-import sys
-
-assert sys.version_info[:2] >= ( 2, 4 )
-
-#genbank_to_bed
-class Region:
- def __init__( self ):
- self.qualifiers = {}
- self.start = None
- self.end = None
- self.strand = '+'
- def set_coordinates_by_location( self, location ):
- location = location.strip().lower().replace( '..', ',' )
- if "complement(" in location: #if part of the sequence is on the negative strand, it all is?
- self.strand = '-' #default of + strand
- for remove_text in ["join(", "order(", "complement(", ")"]:
- location = location.replace( remove_text, "" )
- for number in location.split( ',' ):
- number = number.strip('\n\r\t <>,()')
- if number:
- if "^" in number:
- #a single point
- #check that this is correct for points, ie: 413/NC_005027.gbk: misc_feature 6636286^6636287 ===> 6636285,6636286
- end = int( number.split( '^' )[0] )
- start = end - 1
- else:
- end = int( number )
- start = end - 1 #match BED coordinates
- if self.start is None or start < self.start:
- self.start = start
- if self.end is None or end > self.end:
- self.end = end
-
-class GenBankFeatureParser:
- """Parses Features from Single Locus GenBank file"""
- def __init__( self, fh, features_list = [] ):
- self.fh = fh
- self.features = {}
- fh.seek(0)
- in_features = False
- last_feature_name = None
- base_indent = 0
- last_indent = 0
- last_attr_name = None
- for line in fh:
- if not in_features and line.startswith('FEATURES'):
- in_features = True
- continue
- if in_features:
- lstrip = line.lstrip()
- if line and lstrip == line:
- break #end of feature block
- cur_indent = len( line ) - len( lstrip )
- if last_feature_name is None:
- base_indent = cur_indent
- if cur_indent == base_indent:
- #a new feature
- last_attr_name = None
- fields = lstrip.split( None, 1 )
- last_feature_name = fields[0].strip()
- if not features_list or ( features_list and last_feature_name in features_list ):
- if last_feature_name not in self.features:
- self.features[last_feature_name] = []
- region = Region()
- region.set_coordinates_by_location( fields[1] )
- self.features[last_feature_name].append( region )
- else:
- #add info to last known feature
- line = line.strip()
- if line.startswith( '/' ):
- fields = line[1:].split( '=', 1 )
- if len( fields ) == 2:
- last_attr_name, content = fields
- else:
- #No data
- last_attr_name = line[1:]
- content = ""
- content = content.strip( '"' )
- if last_attr_name not in self.features[last_feature_name][-1].qualifiers:
- self.features[last_feature_name][-1].qualifiers[last_attr_name] = []
- self.features[last_feature_name][-1].qualifiers[last_attr_name].append( content )
- elif last_attr_name is None and last_feature_name:
- # must still be working on location
- self.features[last_feature_name][-1].set_coordinates_by_location( line )
- else:
- #continuation of multi-line qualifier content
- if last_feature_name.lower() in ['translation']:
- self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s%s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) )
- else:
- self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s %s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) )
-
-
- def get_features_by_type( self, feature_type ):
- if feature_type not in self.features:
- return []
- else:
- return self.features[feature_type]
-
-# Parse A GenBank file and return arrays of BED regions for the corresponding features
-def get_bed_from_genbank(gb_file, chrom, feature_list):
- genbank_parser = GenBankFeatureParser( open( gb_file ) )
- features = {}
- for feature_type in feature_list:
- features[feature_type]=[]
- for feature in genbank_parser.get_features_by_type( feature_type ):
- name = ""
- for name_tag in ['gene', 'locus_tag', 'db_xref']:
- if name_tag in feature.qualifiers:
- if name: name = name + ";"
- name = name + feature.qualifiers[name_tag][0].replace(" ","_")
- if not name:
- name = "unknown"
-
- features[feature_type].append( "%s\t%s\t%s\t%s\t%s\t%s" % ( chrom, feature.start, feature.end, name, 0, feature.strand ) )#append new bed field here
- return features
-
-#geneMark to bed
-import sys
-
-#converts GeneMarkHMM to bed
-#returns an array of bed regions
-def get_bed_from_GeneMark(geneMark_filename, chr):
- orfs = open(geneMark_filename).readlines()
- while True:
- line = orfs.pop(0).strip()
- if line.startswith("--------"):
- orfs.pop(0)
- break
- orfs = "".join(orfs)
- ctr = 0
- regions = []
- for block in orfs.split("\n\n"):
- if block.startswith("List of Regions of interest"): break
- best_block = {'start':0,'end':0,'strand':'+','avg_prob':-sys.maxint,'start_prob':-sys.maxint,'name':'DNE'}
- ctr+=1
- ctr2=0
- for line in block.split("\n"):
- ctr2+=1
- fields = line.split()
- start = int(fields.pop(0))-1
- end = int(fields.pop(0))
- strand = fields.pop(0)
- if strand == 'complement':
- strand = "-"
- else:
- strand = "+"
- frame = fields.pop(0)
- frame = frame + " " + fields.pop(0)
- avg_prob = float(fields.pop(0))
- try:
- start_prob = float(fields.pop(0))
- except:
- start_prob = 0
- name = "orf_"+str(ctr)+"_"+str(ctr2)
- if avg_prob >= best_block['avg_prob']:
- if start_prob > best_block['start_prob']:
- best_block = {'start':start,'end':end,'strand':strand,'avg_prob':avg_prob,'start_prob':start_prob,'name':name}
- regions.append(chr+"\t"+str(best_block['start'])+"\t"+str(best_block['end'])+"\t"+best_block['name']+"\t"+str(int(best_block['avg_prob']*1000))+"\t"+best_block['strand'])
- return regions
-
-#geneMarkHMM to bed
-#converts GeneMarkHMM to bed
-#returns an array of bed regions
-def get_bed_from_GeneMarkHMM(geneMarkHMM_filename, chr):
- orfs = open(geneMarkHMM_filename).readlines()
- while True:
- line = orfs.pop(0).strip()
- if line == "Predicted genes":
- orfs.pop(0)
- orfs.pop(0)
- break
- regions = []
- for line in orfs:
- fields = line.split()
- name = "gene_number_"+fields.pop(0)
- strand = fields.pop(0)
- start = fields.pop(0)
- if start.startswith("<"): start = 1
- start = int(start)-1
- end = fields.pop(0)
- if end.startswith(">"): end = end[1:]
- end = int(end)
- score = 0 # no scores provided
- regions.append(chr+"\t"+str(start)+"\t"+str(end)+"\t"+name+"\t"+str(score)+"\t"+strand)
- return regions
-
-#glimmer3 to bed
-#converts glimmer3 to bed, doing some linear scaling (probably not correct?) on scores
-#returns an array of bed regions
-def get_bed_from_glimmer3(glimmer3_filename, chr):
- max_score = -sys.maxint
- min_score = sys.maxint
- orfs = []
- for line in open(glimmer3_filename).readlines():
- if line.startswith(">"): continue
- fields = line.split()
- name = fields.pop(0)
- start = int(fields.pop(0))
- end = int(fields.pop(0))
- if int(fields.pop(0))<0:
- strand = "-"
- temp = start
- start = end
- end = temp
- else:
- strand = "+"
- start = start - 1
- score = (float(fields.pop(0)))
- if score > max_score: max_score = score
- if score < min_score: min_score = score
- orfs.append((chr,start,end,name,score,strand))
-
- delta = 0
- if min_score < 0: delta = min_score * -1
- regions = []
- for (chr,start,end,name,score,strand) in orfs:
- #need to cast to str because was having the case where 1000.0 was rounded to 999 by int, some sort of precision bug?
- my_score = int(float(str( ( (score+delta) * (1000-0-(min_score+delta)) ) / ( (max_score + delta) + 0 ))))
-
- regions.append(chr+"\t"+str(start)+"\t"+str(end)+"\t"+name+"\t"+str(my_score)+"\t"+strand)
- return regions
+#!/usr/bin/env python
+# Dan Blankenberg
+
+import sys
+
+assert sys.version_info[:2] >= ( 2, 4 )
+
+
+# genbank_to_bed
+class Region:
+ def __init__( self ):
+ self.qualifiers = {}
+ self.start = None
+ self.end = None
+ self.strand = '+'
+
+ def set_coordinates_by_location( self, location ):
+ location = location.strip().lower().replace( '..', ',' )
+ if "complement(" in location: # if part of the sequence is on the negative strand, it all is?
+ self.strand = '-' # default of + strand
+ for remove_text in ["join(", "order(", "complement(", ")"]:
+ location = location.replace( remove_text, "" )
+ for number in location.split( ',' ):
+ number = number.strip('\n\r\t <>,()')
+ if number:
+ if "^" in number:
+ # a single point
+ # check that this is correct for points, ie: 413/NC_005027.gbk: misc_feature 6636286^6636287 ===> 6636285,6636286
+ end = int( number.split( '^' )[0] )
+ start = end - 1
+ else:
+ end = int( number )
+ start = end - 1 # match BED coordinates
+ if self.start is None or start < self.start:
+ self.start = start
+ if self.end is None or end > self.end:
+ self.end = end
+
+
+class GenBankFeatureParser:
+ """Parses Features from Single Locus GenBank file"""
+ def __init__( self, fh, features_list=[] ):
+ self.fh = fh
+ self.features = {}
+ fh.seek(0)
+ in_features = False
+ last_feature_name = None
+ base_indent = 0
+ last_attr_name = None
+ for line in fh:
+ if not in_features and line.startswith('FEATURES'):
+ in_features = True
+ continue
+ if in_features:
+ lstrip = line.lstrip()
+ if line and lstrip == line:
+ break # end of feature block
+ cur_indent = len( line ) - len( lstrip )
+ if last_feature_name is None:
+ base_indent = cur_indent
+ if cur_indent == base_indent:
+ # a new feature
+ last_attr_name = None
+ fields = lstrip.split( None, 1 )
+ last_feature_name = fields[0].strip()
+ if not features_list or ( features_list and last_feature_name in features_list ):
+ if last_feature_name not in self.features:
+ self.features[last_feature_name] = []
+ region = Region()
+ region.set_coordinates_by_location( fields[1] )
+ self.features[last_feature_name].append( region )
+ else:
+ # add info to last known feature
+ line = line.strip()
+ if line.startswith( '/' ):
+ fields = line[1:].split( '=', 1 )
+ if len( fields ) == 2:
+ last_attr_name, content = fields
+ else:
+ # No data
+ last_attr_name = line[1:]
+ content = ""
+ content = content.strip( '"' )
+ if last_attr_name not in self.features[last_feature_name][-1].qualifiers:
+ self.features[last_feature_name][-1].qualifiers[last_attr_name] = []
+ self.features[last_feature_name][-1].qualifiers[last_attr_name].append( content )
+ elif last_attr_name is None and last_feature_name:
+ # must still be working on location
+ self.features[last_feature_name][-1].set_coordinates_by_location( line )
+ else:
+ # continuation of multi-line qualifier content
+ if last_feature_name.lower() in ['translation']:
+ self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s%s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) )
+ else:
+ self.features[last_feature_name][-1].qualifiers[last_attr_name][-1] = "%s %s" % ( self.features[last_feature_name][-1].qualifiers[last_attr_name][-1], line.rstrip( '"' ) )
+
+ def get_features_by_type( self, feature_type ):
+ if feature_type not in self.features:
+ return []
+ else:
+ return self.features[feature_type]
+
+
+# Parse A GenBank file and return arrays of BED regions for the corresponding features
+def get_bed_from_genbank(gb_file, chrom, feature_list):
+ genbank_parser = GenBankFeatureParser( open( gb_file ) )
+ features = {}
+ for feature_type in feature_list:
+ features[feature_type] = []
+ for feature in genbank_parser.get_features_by_type( feature_type ):
+ name = ""
+ for name_tag in ['gene', 'locus_tag', 'db_xref']:
+ if name_tag in feature.qualifiers:
+ if name:
+ name = name + ";"
+ name = name + feature.qualifiers[name_tag][0].replace(" ", "_")
+ if not name:
+ name = "unknown"
+
+ features[feature_type].append( "%s\t%s\t%s\t%s\t%s\t%s" % ( chrom, feature.start, feature.end, name, 0, feature.strand ) ) # append new bed field here
+ return features
+
+
+# geneMark to bed
+# converts GeneMarkHMM to bed
+# returns an array of bed regions
+def get_bed_from_GeneMark(geneMark_filename, chr):
+ orfs = open(geneMark_filename).readlines()
+ while True:
+ line = orfs.pop(0).strip()
+ if line.startswith("--------"):
+ orfs.pop(0)
+ break
+ orfs = "".join(orfs)
+ ctr = 0
+ regions = []
+ for block in orfs.split("\n\n"):
+ if block.startswith("List of Regions of interest"):
+ break
+ best_block = {'start': 0, 'end': 0, 'strand': '+', 'avg_prob': -sys.maxint, 'start_prob': -sys.maxint, 'name': 'DNE'}
+ ctr += 1
+ ctr2 = 0
+ for line in block.split("\n"):
+ ctr2 += 1
+ fields = line.split()
+ start = int(fields.pop(0)) - 1
+ end = int(fields.pop(0))
+ strand = fields.pop(0)
+ if strand == 'complement':
+ strand = "-"
+ else:
+ strand = "+"
+ frame = fields.pop(0)
+ frame = frame + " " + fields.pop(0)
+ avg_prob = float(fields.pop(0))
+ try:
+ start_prob = float(fields.pop(0))
+ except:
+ start_prob = 0
+ name = "orf_" + str(ctr) + "_" + str(ctr2)
+ if avg_prob >= best_block['avg_prob']:
+ if start_prob > best_block['start_prob']:
+ best_block = {'start': start, 'end': end, 'strand': strand, 'avg_prob': avg_prob, 'start_prob': start_prob, 'name': name}
+ regions.append(chr + "\t" + str(best_block['start']) + "\t" + str(best_block['end']) + "\t" + best_block['name'] + "\t" + str(int(best_block['avg_prob'] * 1000)) + "\t" + best_block['strand'])
+ return regions
+
+
+# geneMarkHMM to bed
+# converts GeneMarkHMM to bed
+# returns an array of bed regions
+def get_bed_from_GeneMarkHMM(geneMarkHMM_filename, chr):
+ orfs = open(geneMarkHMM_filename).readlines()
+ while True:
+ line = orfs.pop(0).strip()
+ if line == "Predicted genes":
+ orfs.pop(0)
+ orfs.pop(0)
+ break
+ regions = []
+ for line in orfs:
+ fields = line.split()
+ name = "gene_number_" + fields.pop(0)
+ strand = fields.pop(0)
+ start = fields.pop(0)
+ if start.startswith("<"):
+ start = 1
+ start = int(start) - 1
+ end = fields.pop(0)
+ if end.startswith(">"):
+ end = end[1:]
+ end = int(end)
+ score = 0 # no scores provided
+ regions.append(chr + "\t" + str(start) + "\t" + str(end) + "\t" + name + "\t" + str(score) + "\t" + strand)
+ return regions
+
+
+# glimmer3 to bed
+# converts glimmer3 to bed, doing some linear scaling (probably not correct?) on scores
+# returns an array of bed regions
+def get_bed_from_glimmer3(glimmer3_filename, chr):
+ max_score = -sys.maxint
+ min_score = sys.maxint
+ orfs = []
+ for line in open(glimmer3_filename).readlines():
+ if line.startswith(">"):
+ continue
+ fields = line.split()
+ name = fields.pop(0)
+ start = int(fields.pop(0))
+ end = int(fields.pop(0))
+ if int(fields.pop(0)) < 0:
+ strand = "-"
+ temp = start
+ start = end
+ end = temp
+ else:
+ strand = "+"
+ start = start - 1
+ score = (float(fields.pop(0)))
+ if score > max_score:
+ max_score = score
+ if score < min_score:
+ min_score = score
+ orfs.append((chr, start, end, name, score, strand))
+
+ delta = 0
+ if min_score < 0:
+ delta = min_score * -1
+ regions = []
+ for (chr, start, end, name, score, strand) in orfs:
+ # need to cast to str because was having the case where 1000.0 was rounded to 999 by int, some sort of precision bug?
+ my_score = int(float(str( ( (score + delta) * (1000 - 0 - (min_score + delta)) ) / ( (max_score + delta) + 0 ))))
+
+ regions.append(chr + "\t" + str(start) + "\t" + str(end) + "\t" + name + "\t" + str(my_score) + "\t" + strand)
+ return regions
diff --git a/scripts/others/incorrect_gops_jobs.py b/scripts/others/incorrect_gops_jobs.py
index da0782d268e..a96d136ec28 100755
--- a/scripts/others/incorrect_gops_jobs.py
+++ b/scripts/others/incorrect_gops_jobs.py
@@ -1,17 +1,21 @@
#!/usr/bin/env python
"""
-Fetch jobs using gops_intersect, gops_merge, gops_subtract, gops_complement, gops_coverage
+Fetch jobs using gops_intersect, gops_merge, gops_subtract, gops_complement, gops_coverage
wherein the second dataset doesn't have chr, start and end in standard columns 1, 2 and 3.
"""
-
-import sys, os, ConfigParser, tempfile
-import galaxy.app
-import galaxy.model.mapping
+import ConfigParser
+import os
+import sys
+import tempfile
import sqlalchemy as sa
+import galaxy.app
+import galaxy.model.mapping
+
assert sys.version_info[:2] >= ( 2, 4 )
+
class TestApplication( object ):
"""Encapsulates the state of a Universe application"""
def __init__( self, database_connection=None, file_path=None ):
@@ -25,9 +29,10 @@ class TestApplication( object ):
# Setup the database engine and ORM
self.model = galaxy.model.mapping.init( self.file_path, self.database_connection, engine_options={}, create_tables=False )
+
def main():
ini_file = sys.argv[1]
- conf_parser = ConfigParser.ConfigParser( {'here':os.getcwd()} )
+ conf_parser = ConfigParser.ConfigParser( {'here': os.getcwd()} )
conf_parser.read( ini_file )
configuration = {}
for key, value in conf_parser.items( "app:main" ):
@@ -39,21 +44,13 @@ def main():
try:
for job in app.model.Job.filter( sa.and_( app.model.Job.table.c.create_time.between( '2008-05-23', '2008-11-29' ),
app.model.Job.table.c.state == 'ok',
- sa.or_(
- sa.and_( sa.or_( app.model.Job.table.c.tool_id == 'gops_intersect_1',
+ sa.or_( sa.and_( sa.or_( app.model.Job.table.c.tool_id == 'gops_intersect_1',
app.model.Job.table.c.tool_id == 'gops_subtract_1',
- app.model.Job.table.c.tool_id == 'gops_coverage_1',
- ),
- sa.not_( app.model.Job.table.c.command_line.like( '%-2 1,2,3%' ) )
- ),
+ app.model.Job.table.c.tool_id == 'gops_coverage_1' ),
+ sa.not_( app.model.Job.table.c.command_line.like( '%-2 1,2,3%' ) ) ),
sa.and_( sa.or_( app.model.Job.table.c.tool_id == 'gops_complement_1',
- app.model.Job.table.c.tool_id == 'gops_merge_1',
- ),
- sa.not_( app.model.Job.table.c.command_line.like( '%-1 1,2,3%' ) )
- )
- )
- )
- ).all():
+ app.model.Job.table.c.tool_id == 'gops_merge_1' ),
+ sa.not_( app.model.Job.table.c.command_line.like( '%-1 1,2,3%' ) ) ) ) ) ).all():
print "# processing job id %s" % str( job.id )
for jtoda in job.output_datasets:
print "# --> processing JobToOutputDatasetAssociation id %s" % str( jtoda.id )
@@ -67,17 +64,17 @@ def main():
if history.user_id:
cmd_line = str( job.command_line )
new_output = tempfile.NamedTemporaryFile('w')
- if job.tool_id in ['gops_intersect_1','gops_subtract_1','gops_coverage_1']:
- new_cmd_line = " ".join(map(str,cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str,cmd_line.split()[5:]))
+ if job.tool_id in ['gops_intersect_1', 'gops_subtract_1', 'gops_coverage_1']:
+ new_cmd_line = " ".join(map(str, cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str, cmd_line.split()[5:]))
job_output = cmd_line.split()[4]
else:
- new_cmd_line = " ".join(map(str,cmd_line.split()[:3])) + " " + new_output.name + " " + " ".join(map(str,cmd_line.split()[4:]))
+ new_cmd_line = " ".join(map(str, cmd_line.split()[:3])) + " " + new_output.name + " " + " ".join(map(str, cmd_line.split()[4:]))
job_output = cmd_line.split()[3]
try:
os.system(new_cmd_line)
except:
pass
- diff_status = os.system('diff %s %s >> /dev/null' %(new_output.name, job_output))
+ diff_status = os.system('diff %s %s >> /dev/null' % (new_output.name, job_output))
if diff_status == 0:
continue
print "# --------> Outputs differ"
@@ -86,26 +83,25 @@ def main():
jobs[ job.id ][ 'hda_id' ] = hda.id
jobs[ job.id ][ 'hda_name' ] = hda.name
jobs[ job.id ][ 'hda_info' ] = hda.info
- jobs[ job.id ][ 'history_id' ] = history.id
- jobs[ job.id ][ 'history_name' ] = history.name
+ jobs[ job.id ][ 'history_id' ] = history.id
+ jobs[ job.id ][ 'history_name' ] = history.name
jobs[ job.id ][ 'history_update_time' ] = history.update_time
jobs[ job.id ][ 'user_email' ] = user.email
except Exception, e:
print "# caught exception: %s" % str( e )
-
+
print "\n\n# Number of incorrect Jobs: %d\n\n" % ( len( jobs ) )
print "#job_id\thda_id\thda_name\thda_info\thistory_id\thistory_name\thistory_update_time\tuser_email"
for jid in jobs:
print '%s\t%s\t"%s"\t"%s"\t%s\t"%s"\t"%s"\t%s' % \
- ( str( jid ),
- str( jobs[ jid ][ 'hda_id' ] ),
- jobs[ jid ][ 'hda_name' ],
+ ( str( jid ),
+ str( jobs[ jid ][ 'hda_id' ] ),
+ jobs[ jid ][ 'hda_name' ],
jobs[ jid ][ 'hda_info' ],
str( jobs[ jid ][ 'history_id' ] ),
jobs[ jid ][ 'history_name' ],
jobs[ jid ][ 'history_update_time' ],
- jobs[ jid ][ 'user_email' ]
- )
+ jobs[ jid ][ 'user_email' ] )
sys.exit(0)
if __name__ == "__main__":
diff --git a/scripts/others/incorrect_gops_join_jobs.py b/scripts/others/incorrect_gops_join_jobs.py
index 8290cf9894e..bcefdb6f4c7 100644
--- a/scripts/others/incorrect_gops_join_jobs.py
+++ b/scripts/others/incorrect_gops_join_jobs.py
@@ -2,15 +2,19 @@
"""
Fetch gops_join wherein the use specified minimum coverage is not 1.
"""
-
-import sys, os, ConfigParser, tempfile
-import galaxy.app
-import galaxy.model.mapping
+import ConfigParser
+import os
+import sys
+import tempfile
import sqlalchemy as sa
+import galaxy.app
+import galaxy.model.mapping
+
assert sys.version_info[:2] >= ( 2, 4 )
+
class TestApplication( object ):
"""Encapsulates the state of a Universe application"""
def __init__( self, database_connection=None, file_path=None ):
@@ -24,9 +28,10 @@ class TestApplication( object ):
# Setup the database engine and ORM
self.model = galaxy.model.mapping.init( self.file_path, self.database_connection, engine_options={}, create_tables=False )
+
def main():
ini_file = sys.argv[1]
- conf_parser = ConfigParser.ConfigParser( {'here':os.getcwd()} )
+ conf_parser = ConfigParser.ConfigParser( {'here': os.getcwd()} )
conf_parser.read( ini_file )
configuration = {}
for key, value in conf_parser.items( "app:main" ):
@@ -39,9 +44,7 @@ def main():
for job in app.model.Job.filter( sa.and_( app.model.Job.table.c.create_time < '2008-12-16',
app.model.Job.table.c.state == 'ok',
app.model.Job.table.c.tool_id == 'gops_join_1',
- sa.not_( app.model.Job.table.c.command_line.like( '%-m 1 %' ) )
- )
- ).all():
+ sa.not_( app.model.Job.table.c.command_line.like( '%-m 1 %' ) ) ) ).all():
print "# processing job id %s" % str( job.id )
for jtoda in job.output_datasets:
print "# --> processing JobToOutputDatasetAssociation id %s" % str( jtoda.id )
@@ -55,13 +58,13 @@ def main():
if history.user_id:
cmd_line = str( job.command_line )
new_output = tempfile.NamedTemporaryFile('w')
- new_cmd_line = " ".join(map(str,cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str,cmd_line.split()[5:]))
+ new_cmd_line = " ".join(map(str, cmd_line.split()[:4])) + " " + new_output.name + " " + " ".join(map(str, cmd_line.split()[5:]))
job_output = cmd_line.split()[4]
try:
os.system(new_cmd_line)
except:
pass
- diff_status = os.system('diff %s %s >> /dev/null' %(new_output.name, job_output))
+ diff_status = os.system('diff %s %s >> /dev/null' % (new_output.name, job_output))
if diff_status == 0:
continue
print "# --------> Outputs differ"
@@ -70,26 +73,25 @@ def main():
jobs[ job.id ][ 'hda_id' ] = hda.id
jobs[ job.id ][ 'hda_name' ] = hda.name
jobs[ job.id ][ 'hda_info' ] = hda.info
- jobs[ job.id ][ 'history_id' ] = history.id
- jobs[ job.id ][ 'history_name' ] = history.name
+ jobs[ job.id ][ 'history_id' ] = history.id
+ jobs[ job.id ][ 'history_name' ] = history.name
jobs[ job.id ][ 'history_update_time' ] = history.update_time
jobs[ job.id ][ 'user_email' ] = user.email
except Exception, e:
print "# caught exception: %s" % str( e )
-
+
print "\n\n# Number of incorrect Jobs: %d\n\n" % ( len( jobs ) )
print "#job_id\thda_id\thda_name\thda_info\thistory_id\thistory_name\thistory_update_time\tuser_email"
for jid in jobs:
print '%s\t%s\t"%s"\t"%s"\t%s\t"%s"\t"%s"\t%s' % \
- ( str( jid ),
- str( jobs[ jid ][ 'hda_id' ] ),
- jobs[ jid ][ 'hda_name' ],
+ ( str( jid ),
+ str( jobs[ jid ][ 'hda_id' ] ),
+ jobs[ jid ][ 'hda_name' ],
jobs[ jid ][ 'hda_info' ],
str( jobs[ jid ][ 'history_id' ] ),
jobs[ jid ][ 'history_name' ],
jobs[ jid ][ 'history_update_time' ],
- jobs[ jid ][ 'user_email' ]
- )
+ jobs[ jid ][ 'user_email' ] )
sys.exit(0)
if __name__ == "__main__":
diff --git a/scripts/tool_shed/check_download_urls.py b/scripts/tool_shed/check_download_urls.py
index b5c6a4ca422..4698eb404c3 100644
--- a/scripts/tool_shed/check_download_urls.py
+++ b/scripts/tool_shed/check_download_urls.py
@@ -1,23 +1,22 @@
#!/usr/bin/env python
-##Dan Blankenberg
-##Script that checks toolshed tags to see if URLs are accessible.
-##Does not currently handle 'download_binary'
-
+# Dan Blankenberg
+# Script that checks toolshed tags to see if URLs are accessible.
+# Does not currently handle 'download_binary'
import os
-import sys
-from optparse import OptionParser
-import xml.etree.ElementTree as ET
import urllib2
+import xml.etree.ElementTree as ET
+from optparse import OptionParser
FILENAMES = [ 'tool_dependencies.xml' ]
ACTION_TYPES = [ 'download_by_url', 'download_file' ]
+
def main():
parser = OptionParser()
parser.add_option( '-d', '--directory', dest='directory', action='store', type="string", default='.', help='Root directory' )
-
+
( options, args ) = parser.parse_args()
-
+
for (dirpath, dirnames, filenames) in os.walk( options.directory ):
for filename in filenames:
if filename in FILENAMES:
diff --git a/scripts/tool_shed/migrate_tools_to_repositories.py b/scripts/tool_shed/migrate_tools_to_repositories.py
index d10e359d391..529808d2e39 100644
--- a/scripts/tool_shed/migrate_tools_to_repositories.py
+++ b/scripts/tool_shed/migrate_tools_to_repositories.py
@@ -15,23 +15,29 @@ associated with them, and migrates old tool shed stuff to new tool shed stuff.
# Enable next-gen tool shed features
enable_next_gen_tool_shed = True
-2. This script requires the Galaxy instance to use Postgres for database storage.
+2. This script requires the Galaxy instance to use Postgres for database storage.
To run this script, use "sh migrate_tools_to_repositories.sh" from this directory
'''
-import sys, os, subprocess, ConfigParser, shutil, tarfile, tempfile
+import ConfigParser
+import os
+import shutil
+import sys
+import tarfile
+import tempfile
+from time import strftime
+
+from mercurial import hg, ui
-assert sys.version_info[:2] >= ( 2, 4 )
new_path = [ os.path.join( os.getcwd(), "lib" ) ]
-new_path.extend( sys.path[1:] ) # remove scripts/ from the path
+new_path.extend( sys.path[1:] ) # remove scripts/ from the path
sys.path = new_path
-import psycopg2
-
import galaxy.webapps.tool_shed.app
-from mercurial import hg, ui, httprepo, commands
-from time import strftime
+
+assert sys.version_info[:2] >= ( 2, 4 )
+
def directory_hash_id( id ):
s = str( id )
@@ -44,13 +50,14 @@ def directory_hash_id( id ):
# Drop the last three digits -- 1000 files per directory
padded = padded[:-3]
# Break into chunks of three
- return [ padded[i*3:(i+1)*3] for i in range( len( padded ) // 3 ) ]
+ return [ padded[i * 3:(i + 1) * 3] for i in range( len( padded ) // 3 ) ]
+
def get_versions( app, item ):
"""Get all versions of item whose state is a valid state"""
- valid_states = [ app.model.Tool.states.NEW,
- app.model.Tool.states.WAITING,
- app.model.Tool.states.APPROVED,
+ valid_states = [ app.model.Tool.states.NEW,
+ app.model.Tool.states.WAITING,
+ app.model.Tool.states.APPROVED,
app.model.Tool.states.ARCHIVED ]
versions = [ item ]
this_item = item
@@ -65,6 +72,7 @@ def get_versions( app, item ):
item = item.older_version[ 0 ]
return versions
+
def get_approved_tools( app, sa_session ):
"""Get only the latest version of each tool from the database whose state is approved"""
tools = []
@@ -74,6 +82,7 @@ def get_approved_tools( app, sa_session ):
tools.append( tool )
return tools
+
def create_repository_from_tool( app, sa_session, tool ):
# Make the repository name a form of the tool's tool_id by
# lower-casing everything and replacing any blank spaces with underscores.
@@ -81,7 +90,7 @@ def create_repository_from_tool( app, sa_session, tool ):
print "Creating repository '%s' in database" % ( repo_name )
repository = app.model.Repository( name=repo_name,
description=tool.description,
- user_id = tool.user_id )
+ user_id=tool.user_id )
# Flush to get the id
sa_session.add( repository )
sa_session.flush()
@@ -97,7 +106,7 @@ def create_repository_from_tool( app, sa_session, tool ):
os.makedirs( repository_path )
# Create the local hg repository
print "Creating repository '%s' on disk" % ( os.path.abspath( repository_path ) )
- repo = hg.repository( ui.ui(), os.path.abspath( repository_path ), create=True )
+ hg.repository( ui.ui(), os.path.abspath( repository_path ), create=True )
# Add an entry in the hgweb.config file for the new repository - this enables calls to repository.repo_path
add_hgweb_config_entry( repository, repository_path )
# Migrate tool categories
@@ -113,14 +122,15 @@ def create_repository_from_tool( app, sa_session, tool ):
rra = app.model.RepositoryRatingAssociation( user=tra.user,
rating=tra.rating,
comment=tra.comment )
- rra.repository=repository
+ rra.repository = repository
sa_session.add( rra )
sa_session.flush()
+
def add_hgweb_config_entry( repository, repository_path ):
# Add an entry in the hgweb.config file for a new repository. This enables calls to repository.repo_path.
# An entry looks something like: repos/test/mira_assembler = database/community_files/000/repo_123
- hgweb_config = "%s/hgweb.config" % os.getcwd()
+ hgweb_config = "%s/hgweb.config" % os.getcwd()
entry = "repos/%s/%s = %s" % ( repository.user.username, repository.name, repository_path.lstrip( './' ) )
if os.path.exists( hgweb_config ):
output = open( hgweb_config, 'a' )
@@ -130,6 +140,7 @@ def add_hgweb_config_entry( repository, repository_path ):
output.write( "%s\n" % entry )
output.close()
+
def create_hgrc_file( repository ):
# At this point, an entry for the repository is required to be in the hgweb.config
# file so we can call repository.repo_path.
@@ -150,6 +161,7 @@ def create_hgrc_file( repository ):
output.flush()
output.close()
+
def add_tool_files_to_repository( app, sa_session, tool ):
current_working_dir = os.getcwd()
# Get the repository to which the tool will be migrated
@@ -169,7 +181,7 @@ def add_tool_files_to_repository( app, sa_session, tool ):
cmd = "hg clone %s" % repo_path
os.chdir( tmp_archive_dir )
os.system( cmd )
- os.chdir( current_working_dir )
+ os.chdir( current_working_dir )
cloned_repo_dir = os.path.join( tmp_archive_dir, 'repo_%d' % repository.id )
# We want these change sets to be associated with the owner of the repository, so we'll
# set the HGUSER environment variable accordingly. We do this because in the mercurial
@@ -194,8 +206,8 @@ def add_tool_files_to_repository( app, sa_session, tool ):
# Don't visit .hg directories
dirs.remove( '.hg' )
if 'hgrc' in files:
- # Don't include hgrc files in commit - should be impossible
- # since we don't visit .hg dirs, but just in case...
+ # Don't include hgrc files in commit - should be impossible
+ # since we don't visit .hg dirs, but just in case...
files.remove( 'hgrc' )
for dir in dirs:
os.system( "hg add %s" % dir )
@@ -221,13 +233,16 @@ def add_tool_files_to_repository( app, sa_session, tool ):
# Remove tmp directory
shutil.rmtree( tmp_dir )
+
def get_repository_by_name( app, sa_session, repo_name ):
"""Get a repository from the database"""
return sa_session.query( app.model.Repository ).filter_by( name=repo_name ).one()
+
def contains( containing_str, contained_str ):
return containing_str.lower().find( contained_str.lower() ) >= 0
+
def tool_archive_extension( file_name ):
extension = None
if extension is None:
@@ -248,9 +263,11 @@ def tool_archive_extension( file_name ):
extension = 'tar'
return extension
+
def tool_archive_file_name( tool, file_name ):
return '%s_%s.%s' % ( tool.tool_id, tool.version, tool_archive_extension( file_name ) )
-
+
+
def main():
if len( sys.argv ) < 2:
print "Usage: python %s " % sys.argv[0]
@@ -261,24 +278,13 @@ def main():
print "%s - Migrating current tool archives to new tool repositories" % now
# tool_shed_wsgi.ini file
ini_file = sys.argv[1]
- conf_parser = ConfigParser.ConfigParser( {'here':os.getcwd()} )
+ conf_parser = ConfigParser.ConfigParser( {'here': os.getcwd()} )
conf_parser.read( ini_file )
try:
db_conn_str = conf_parser.get( "app:main", "database_connection" )
- except ConfigParser.NoOptionError, e:
+ except ConfigParser.NoOptionError:
db_conn_str = conf_parser.get( "app:main", "database_file" )
print 'DB Connection: ', db_conn_str
- # Determine db connection - only postgres is supported
- if contains( db_conn_str, '///' ) and contains( db_conn_str, '?' ) and contains( db_conn_str, '&' ):
- # postgres:///galaxy_test?user=postgres&password=postgres
- db_str = db_conn_str.split( '///' )[1]
- db_name = db_str.split( '?' )[0]
- db_user = db_str.split( '?' )[1].split( '&' )[0].split( '=' )[1]
- db_password = db_str.split( '?' )[1].split( '&' )[1].split( '=' )[1]
- elif contains( db_conn_str, '//' ) and contains( db_conn_str, ':' ):
- # dialect://user:password@host/db_name
- db_name = db_conn_str.split('/')[-1]
- db_user = db_conn_str.split('//')[1].split(':')[0]
# Instantiate app
configuration = {}
for key, value in conf_parser.items( "app:main" ):
@@ -286,7 +292,7 @@ def main():
app = galaxy.webapps.tool_shed.app.UniverseApplication( global_conf=dict( __file__=ini_file ), **configuration )
sa_session = app.model.context
# Remove the hgweb.config file if it exists
- hgweb_config = "%s/hgweb.config" % os.getcwd()
+ hgweb_config = "%s/hgweb.config" % os.getcwd()
if os.path.exists( hgweb_config ):
print "Removing old file: ", hgweb_config
os.remove( hgweb_config )
@@ -300,7 +306,7 @@ def main():
if os.path.exists( dir ):
print "Removing old repository file directory: ", dir
shutil.rmtree( dir )
- # Delete all records from db tables:
+ # Delete all records from db tables:
# repository_category_association, repository_rating_association, repository
print "Deleting db records for repository: ", repo.name
for rca in repo.categories:
@@ -315,7 +321,7 @@ def main():
print "Deleted %d rows from the repository table" % repo_records
print "Deleted %d rows from the repository_category_association table" % rca_records
print "Deleted %d rows from the repository_rating_association table" % rra_records
- # Migrate database tool, tool category and tool rating records to new
+ # Migrate database tool, tool category and tool rating records to new
# database repository, repository category and repository rating records
# and create the hg repository on disk for each.
for tool in get_approved_tools( app, sa_session ):
diff --git a/scripts/tools/maf/check_loc_file.py b/scripts/tools/maf/check_loc_file.py
index ee6207c1522..efd95320ce6 100644
--- a/scripts/tools/maf/check_loc_file.py
+++ b/scripts/tools/maf/check_loc_file.py
@@ -7,6 +7,7 @@ import sys
assert sys.version_info[:2] >= ( 2, 4 )
+
def __main__():
index_location_file = sys.argv[ 1 ]
for i, line in enumerate( open( index_location_file ) ):
@@ -20,12 +21,12 @@ def __main__():
species_indexed_in_maf = []
species_found_in_maf = []
for maf_file in maf_files:
- indexed_maf = bx.align.maf.MAFIndexedAccess( maf_file, keep_open = True, parse_e_rows = False )
+ indexed_maf = bx.align.maf.MAFIndexedAccess( maf_file, keep_open=True, parse_e_rows=False )
for key in indexed_maf.indexes.indexes.keys():
spec = maf_utilities.src_split( key )[0]
if spec not in species_indexed_in_maf:
species_indexed_in_maf.append( spec )
- while True: #reading entire maf set will take some time
+ while True: # reading entire maf set will take some time
block = indexed_maf.read_at_current_offset( indexed_maf.f )
if block is None:
break
@@ -33,14 +34,14 @@ def __main__():
spec = maf_utilities.src_split( comp.src )[0]
if spec not in species_found_in_maf:
species_found_in_maf.append( spec )
- #indexed species
+ # indexed species
for spec in indexed_for_species:
if spec not in species_indexed_in_maf:
print "Line %i, %s claims to be indexed for %s, but indexes do not exist." % ( i, uid, spec )
for spec in species_indexed_in_maf:
if spec not in indexed_for_species:
print "Line %i, %s is indexed for %s, but is not listed in loc file." % ( i, uid, spec )
- #existing species
+ # existing species
for spec in species_exist:
if spec not in species_found_in_maf:
print "Line %i, %s claims to have blocks for %s, but was not found in MAF files." % ( i, uid, spec )
@@ -50,4 +51,5 @@ def __main__():
except Exception, e:
print "Line %i is invalid: %s" % ( i, e )
-if __name__ == "__main__": __main__()
+if __name__ == "__main__":
+ __main__()
diff --git a/scripts/tools/re_escape_output.py b/scripts/tools/re_escape_output.py
index 3bfebb2418c..9fd74d68d11 100644
--- a/scripts/tools/re_escape_output.py
+++ b/scripts/tools/re_escape_output.py
@@ -1,20 +1,19 @@
#!/usr/bin/env python
-
"""
Escapes a file into a form suitable for use with tool tests using re_match or re_match_multiline (when -m/--multiline option is used)
usage: re_escape_output.py [options] input_file [output_file]
-m: Use Multiline Matching
"""
+import optparse
+import re
-import optparse, re
def __main__():
- #Parse Command Line
parser = optparse.OptionParser()
parser.add_option( "-m", "--multiline", action="store_true", dest="multiline", default=False, help="Use Multiline Matching")
( options, args ) = parser.parse_args()
- input = open( args[0] ,'rb' )
+ input = open( args[0], 'rb' )
if len( args ) > 1:
output = open( args[1], 'wb' )
else:
diff --git a/scripts/transfer.py b/scripts/transfer.py
index e194d2bd7fa..5c2110720ba 100644
--- a/scripts/transfer.py
+++ b/scripts/transfer.py
@@ -4,7 +4,17 @@ Downloads files to temp locations. This script is invoked by the Transfer
Manager (galaxy.jobs.transfer_manager) and should not normally be invoked by
hand.
"""
-import os, sys, optparse, ConfigParser, socket, SocketServer, threading, logging, random, urllib2, tempfile, time
+import ConfigParser
+import logging
+import optparse
+import os
+import random
+import SocketServer
+import sys
+import tempfile
+import threading
+import time
+import urllib2
galaxy_root = os.path.abspath( os.path.join( os.path.dirname( __file__ ), os.pardir ) )
sys.path.insert( 0, os.path.abspath( os.path.join( galaxy_root, 'lib' ) ) )
@@ -14,9 +24,9 @@ try:
except ImportError:
pexpect = None
-from sqlalchemy import *
-from sqlalchemy.orm import *
from daemon import DaemonContext
+from sqlalchemy import create_engine, MetaData, Table
+from sqlalchemy.orm import scoped_session, sessionmaker
import galaxy.model
from galaxy.util import json, bunch
@@ -33,6 +43,7 @@ log.addHandler( handler )
debug = False
slow = False
+
class ArgHandler( object ):
"""
Collect command line flags.
@@ -40,10 +51,11 @@ class ArgHandler( object ):
def __init__( self ):
self.parser = optparse.OptionParser()
self.parser.add_option( '-c', '--config', dest='config', help='Path to Galaxy config file (config/galaxy.ini)',
- default=os.path.abspath( os.path.join( galaxy_root, 'config/galaxy.ini' ) ) )
+ default=os.path.abspath( os.path.join( galaxy_root, 'config/galaxy.ini' ) ) )
self.parser.add_option( '-d', '--debug', action='store_true', dest='debug', help="Debug (don't detach)" )
self.parser.add_option( '-s', '--slow', action='store_true', dest='slow', help="Transfer slowly (for debugging)" )
self.opts = None
+
def parse( self ):
self.opts, args = self.parser.parse_args()
if len( args ) != 1:
@@ -62,19 +74,21 @@ class ArgHandler( object ):
global slow
slow = True
+
class GalaxyApp( object ):
"""
A shell Galaxy App to provide access to the Galaxy configuration and
model/database.
"""
def __init__( self, config_file ):
- self.config = ConfigParser.ConfigParser( dict( database_file = 'database/universe.sqlite',
- file_path = 'database/files',
- transfer_worker_port_range = '12275-12675',
- transfer_worker_log = None ) )
+ self.config = ConfigParser.ConfigParser( dict( database_file='database/universe.sqlite',
+ file_path='database/files',
+ transfer_worker_port_range='12275-12675',
+ transfer_worker_log=None ) )
self.config.read( config_file )
self.model = bunch.Bunch()
self.connect_database()
+
def connect_database( self ):
# Avoid loading the entire model since doing so is exceptionally slow
default_dburl = 'sqlite:///%s?isolation_level=IMMEDIATE' % self.config.get( 'app:main', 'database_file' )
@@ -87,9 +101,11 @@ class GalaxyApp( object ):
self.sa_session = scoped_session( sessionmaker( bind=engine, autoflush=False, autocommit=True ) )
self.model.TransferJob = galaxy.model.TransferJob
self.model.TransferJob.table = Table( "transfer_job", metadata, autoload=True )
+
def get_transfer_job( self, id ):
return self.sa_session.query( self.model.TransferJob ).get( int( id ) )
+
class ListenerServer( SocketServer.ThreadingTCPServer ):
"""
The listener will accept state requests and new transfers for as long as
@@ -110,6 +126,7 @@ class ListenerServer( SocketServer.ThreadingTCPServer ):
app.sa_session.add( transfer_job )
app.sa_session.flush()
+
class ListenerRequestHandler( SocketServer.BaseRequestHandler ):
"""
Handle state or transfer requests received on the socket.
@@ -127,6 +144,7 @@ class ListenerRequestHandler( SocketServer.BaseRequestHandler ):
log.error( error_msg )
log.debug( 'Original request was: %s' % request )
+
class StateResult( object ):
"""
A mutable container for the 'result' portion of JSON-RPC responses to state requests.
@@ -134,6 +152,7 @@ class StateResult( object ):
def __init__( self, result=None ):
self.result = result
+
def transfer( app, transfer_job_id ):
transfer_job = app.get_transfer_job( transfer_job_id )
if transfer_job is None:
@@ -149,14 +168,14 @@ def transfer( app, transfer_job_id ):
if protocol not in ( 'http', 'https', 'scp' ):
log.error( 'Unsupported protocol: %s' % protocol )
return False
- state_result = StateResult( result = dict( state = transfer_job.states.RUNNING, info='Transfer process starting up.' ) )
+ state_result = StateResult( result=dict( state=transfer_job.states.RUNNING, info='Transfer process starting up.' ) )
listener_server = ListenerServer( range( port_range[0], port_range[1] + 1 ), ListenerRequestHandler, app, transfer_job, state_result )
# daemonize here (if desired)
if not debug:
daemon_context = DaemonContext( files_preserve=[ listener_server.fileno() ], working_directory=os.getcwd() )
daemon_context.open()
# If this fails, it'll never be detected. Hopefully it won't fail since it succeeded once.
- app.connect_database() # daemon closed the database fd
+ app.connect_database() # daemon closed the database fd
transfer_job = app.get_transfer_job( transfer_job_id )
listener_thread = threading.Thread( target=listener_server.serve_forever )
listener_thread.setDaemon( True )
@@ -191,13 +210,14 @@ def transfer( app, transfer_job_id ):
app.sa_session.flush()
return True
+
def http_transfer( transfer_job ):
"""Plugin" for handling http(s) transfers."""
url = transfer_job.params['url']
try:
f = urllib2.urlopen( url )
except urllib2.URLError, e:
- yield dict( state = transfer_job.states.ERROR, info = 'Unable to open URL: %s' % str( e ) )
+ yield dict( state=transfer_job.states.ERROR, info='Unable to open URL: %s' % str( e ) )
return
size = f.info().getheader( 'Content-Length' )
if size is not None:
@@ -210,7 +230,7 @@ def http_transfer( transfer_job ):
try:
fh, fn = tempfile.mkstemp()
except Exception, e:
- yield dict( state = transfer_job.states.ERROR, info = 'Unable to create temporary file for transfer: %s' % str( e ) )
+ yield dict( state=transfer_job.states.ERROR, info='Unable to create temporary file for transfer: %s' % str( e ) )
return
log.debug( 'Writing %s to %s, size is %s' % ( url, fn, size or 'unknown' ) )
try:
@@ -223,19 +243,20 @@ def http_transfer( transfer_job ):
if size is not None and read < size:
percent = int( float( read ) / size * 100 )
if percent != last:
- yield dict( state = transfer_job.states.PROGRESS, read = read, percent = '%s' % percent )
+ yield dict( state=transfer_job.states.PROGRESS, read=read, percent='%s' % percent )
last = percent
elif size is None:
- yield dict( state = transfer_job.states.PROGRESS, read = read )
+ yield dict( state=transfer_job.states.PROGRESS, read=read )
if slow:
time.sleep( 1 )
os.close( fh )
- yield dict( state = transfer_job.states.DONE, path = fn )
+ yield dict( state=transfer_job.states.DONE, path=fn )
except Exception, e:
- yield dict( state = transfer_job.states.ERROR, info = 'Error during file transfer: %s' % str( e ) )
+ yield dict( state=transfer_job.states.ERROR, info='Error during file transfer: %s' % str( e ) )
return
return
+
def scp_transfer( transfer_job ):
"""Plugin" for handling scp transfers using pexpect"""
def print_ticks( d ):
@@ -245,24 +266,23 @@ def scp_transfer( transfer_job ):
password = transfer_job.params[ 'password' ]
file_path = transfer_job.params[ 'file_path' ]
if pexpect is None:
- return dict( state = transfer_job.states.ERROR, info = PEXPECT_IMPORT_MESSAGE )
+ return dict( state=transfer_job.states.ERROR, info=PEXPECT_IMPORT_MESSAGE )
try:
fh, fn = tempfile.mkstemp()
except Exception, e:
- return dict( state = transfer_job.states.ERROR, info = 'Unable to create temporary file for transfer: %s' % str( e ) )
+ return dict( state=transfer_job.states.ERROR, info='Unable to create temporary file for transfer: %s' % str( e ) )
try:
# TODO: add the ability to determine progress of the copy here like we do in the http_transfer above.
cmd = "scp %s@%s:'%s' '%s'" % ( user_name,
host,
file_path.replace( ' ', '\ ' ),
fn )
- output = pexpect.run( cmd,
- events={ '.ssword:*': password + '\r\n',
- pexpect.TIMEOUT: print_ticks },
- timeout=10 )
- return dict( state = transfer_job.states.DONE, path = fn )
+ pexpect.run( cmd, events={ '.ssword:*': password + '\r\n',
+ pexpect.TIMEOUT: print_ticks },
+ timeout=10 )
+ return dict( state=transfer_job.states.DONE, path=fn )
except Exception, e:
- return dict( state = transfer_job.states.ERROR, info = 'Error during file transfer: %s' % str( e ) )
+ return dict( state=transfer_job.states.ERROR, info='Error during file transfer: %s' % str( e ) )
if __name__ == '__main__':
arg_handler = ArgHandler()