Merge pull request #7889 from mvdbeek/unicode_null_character_issue

Strip unicode control characters from TEXT before entering the database
This commit is contained in:
Marius van den Beek
2019-06-17 10:06:45 +02:00
committed by GitHub
5 changed files with 46 additions and 13 deletions
+2 -1
View File
@@ -1454,7 +1454,8 @@ class JobWrapper(HasResourceParameters):
if not os.path.exists(version_filename):
version_filename = self.get_version_string_path_legacy()
if os.path.exists(version_filename):
self.version_string = open(version_filename).read()
with open(version_filename, 'rb') as fh:
self.version_string = galaxy.util.shrink_and_unicodify(fh.read())
os.unlink(version_filename)
outputs_to_working_directory = util.asbool(self.get_destination_configuration("outputs_to_working_directory", False))
+15 -5
View File
@@ -289,11 +289,13 @@ class JobLike(object):
def set_streams(self, tool_stdout, tool_stderr, job_stdout=None, job_stderr=None, job_messages=None):
def shrink_and_unicodify(what, stream):
stream = galaxy.util.unicodify(stream) or u''
if (len(stream) > galaxy.util.DATABASE_MAX_STRING_SIZE):
stream = galaxy.util.shrink_string_by_size(tool_stdout, galaxy.util.DATABASE_MAX_STRING_SIZE, join_by="\n..\n", left_larger=True, beginning_on_size_error=True)
log.info("%s for %s %d is greater than %s, only a portion will be logged to database", what, type(self), self.id, galaxy.util.DATABASE_MAX_STRING_SIZE_PRETTY)
return stream
if len(stream) > galaxy.util.DATABASE_MAX_STRING_SIZE:
log.info("%s for %s %d is greater than %s, only a portion will be logged to database",
what,
type(self),
self.id,
galaxy.util.DATABASE_MAX_STRING_SIZE_PRETTY)
return galaxy.util.shrink_and_unicodify(stream)
self.tool_stdout = shrink_and_unicodify('tool_stdout', tool_stdout)
self.tool_stderr = shrink_and_unicodify('tool_stderr', tool_stderr)
@@ -2397,6 +2399,14 @@ class DatasetInstance(object):
self.parent_id = parent_id
self.validation_errors = validation_errors
@property
def peek(self):
return self._peek
@peek.setter
def peek(self, peek):
self._peek = unicodify(peek, strip_null=True)
def update(self):
self.update_time = galaxy.model.orm.now.now()
+2 -2
View File
@@ -222,7 +222,7 @@ model.HistoryDatasetAssociation.table = Table(
Column("name", TrimmedString(255)),
Column("info", TrimmedString(255)),
Column("blurb", TrimmedString(255)),
Column("peek", TEXT),
Column("peek", TEXT, key="_peek"),
Column("tool_version", TEXT),
Column("extension", TrimmedString(64)),
Column("metadata", MetadataType(), key="_metadata"),
@@ -506,7 +506,7 @@ model.LibraryDatasetDatasetAssociation.table = Table(
Column("name", TrimmedString(255), index=True),
Column("info", TrimmedString(255)),
Column("blurb", TrimmedString(255)),
Column("peek", TEXT),
Column("peek", TEXT, key="_peek"),
Column("tool_version", TEXT),
Column("extension", TrimmedString(64)),
Column("metadata", MetadataType(), key="_metadata"),
+15 -2
View File
@@ -402,6 +402,17 @@ def shrink_stream_by_size(value, size, join_by=b"..", left_larger=True, beginnin
return unicodify(rval)
def shrink_and_unicodify(stream):
stream = unicodify(stream, strip_null=True) or u''
if (len(stream) > DATABASE_MAX_STRING_SIZE):
stream = shrink_string_by_size(stream,
DATABASE_MAX_STRING_SIZE,
join_by="\n..\n",
left_larger=True,
beginning_on_size_error=True)
return stream
def shrink_string_by_size(value, size, join_by="..", left_larger=True, beginning_on_size_error=False, end_on_size_error=False):
if len(value) > size:
len_join_by = len(join_by)
@@ -993,7 +1004,7 @@ def roundify(amount, sfs=2):
return amount[0:sfs] + '0' * (len(amount) - sfs)
def unicodify(value, encoding=DEFAULT_ENCODING, error='replace'):
def unicodify(value, encoding=DEFAULT_ENCODING, error='replace', strip_null=False):
u"""
Returns a Unicode string or None.
@@ -1008,7 +1019,7 @@ def unicodify(value, encoding=DEFAULT_ENCODING, error='replace'):
>>> s = u'lâtín strìñg'; assert unicodify(s.encode('latin-1')) == u'l\ufffdt\ufffdn str\ufffd\ufffdg'
>>> s = u'lâtín strìñg'; assert unicodify(s.encode('latin-1'), error='ignore') == u'ltn strg'
"""
if value is None or isinstance(value, text_type):
if value is None:
return value
try:
if isinstance(value, bytearray):
@@ -1025,6 +1036,8 @@ def unicodify(value, encoding=DEFAULT_ENCODING, error='replace'):
msg = "Value '%s' could not be coerced to Unicode" % value
log.exception(msg)
raise Exception(msg)
if strip_null:
return value.replace('\0', '')
return value
+12 -3
View File
@@ -1,21 +1,26 @@
<tool id="unicode_stream" name="unicode_stream" version="0.1.0">
<description>
</description>
<version_command>echo "\x00"</version_command>
<command detect_errors="exit_code"><![CDATA[
echo '$input1' > '$out_file1';
#if $include_null:
echo "\x00" > $out_file1;
#end if
echo '$input1' >> '$out_file1';
cat '$cf';
echo "\x00";
>&2 cat '$cf';
sh -c "exit $exit"
]]></command>
<configfiles>
<configfile name="cf">ვეპხის ტყაოსანი შოთა რუსთაველი
</configfile>
<configfile name="cf">ვეპხის ტყაოსანი შოთა რუსთაველი</configfile>
</configfiles>
<inputs>
<param name="input1" type="text" label="Input">
<sanitizer sanitize="False" />
</param>
<param name="exit" type="integer" value="0" label="Exit Code" />
<param name="include_null" type="boolean" label="Include unicode null in output?"/>
</inputs>
<outputs>
<data name="out_file1" format="txt" />
@@ -34,6 +39,10 @@ sh -c "exit $exit"
<param name="input1" value="ვვვვვ"/>
<param name="exit" value="0" />
</test>
<test expect_exit_code="0" expect_failure="false">
<param name="include_null" value="true"/>
<param name="exit" value="0" />
</test>
</tests>
<help>
</help>