mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Merge pull request #1352 from jmchilton/multibyte_refactor
Refactor is_multi_byte out of top-level galaxy.util.
This commit is contained in:
@@ -14,6 +14,7 @@ import zipfile
|
||||
from encodings import search_function as encodings_search_function
|
||||
|
||||
from galaxy import util
|
||||
from galaxy.util import multi_byte
|
||||
from galaxy.datatypes.checkers import check_binary, check_html, is_gzip
|
||||
from galaxy.datatypes.binary import Binary
|
||||
|
||||
@@ -56,7 +57,7 @@ def stream_to_open_named_file( stream, fd, filename, source_encoding=None, sourc
|
||||
if not is_compressed:
|
||||
# See if we have a multi-byte character file
|
||||
chars = chunk[:100]
|
||||
is_multi_byte = util.is_multi_byte( chars )
|
||||
is_multi_byte = multi_byte.is_multi_byte( chars )
|
||||
if not is_multi_byte:
|
||||
is_binary = util.is_binary( chunk )
|
||||
data_checked = True
|
||||
|
||||
@@ -36,7 +36,8 @@ from galaxy.datatypes.metadata import MetadataCollection
|
||||
from galaxy.model.item_attrs import UsesAnnotations
|
||||
from galaxy.util.dictifiable import Dictifiable
|
||||
from galaxy.security import get_permitted_actions
|
||||
from galaxy.util import is_multi_byte, Params, restore_text, send_mail
|
||||
from galaxy.util import Params, restore_text, send_mail
|
||||
from galaxy.util.multi_byte import is_multi_byte
|
||||
from galaxy.util import ready_name_for_url, unique_id
|
||||
from galaxy.util.bunch import Bunch
|
||||
from galaxy.util.hash_util import new_secure_hash
|
||||
|
||||
@@ -38,8 +38,6 @@ import docutils.writers.html4css1
|
||||
|
||||
from xml.etree import ElementTree, ElementInclude
|
||||
|
||||
import wchartype
|
||||
|
||||
from .inflection import Inflector, English
|
||||
inflector = Inflector(English)
|
||||
|
||||
@@ -58,22 +56,6 @@ NULL_CHAR = '\000'
|
||||
BINARY_CHARS = [ NULL_CHAR ]
|
||||
|
||||
|
||||
def is_multi_byte( chars ):
|
||||
for char in chars:
|
||||
try:
|
||||
char = unicode( char )
|
||||
except UnicodeDecodeError:
|
||||
# Probably binary
|
||||
return False
|
||||
if ( wchartype.is_asian( char ) or wchartype.is_full_width( char ) or
|
||||
wchartype.is_kanji( char ) or wchartype.is_hiragana( char ) or
|
||||
wchartype.is_katakana( char ) or wchartype.is_half_katakana( char ) or
|
||||
wchartype.is_hangul( char ) or wchartype.is_full_digit( char ) or
|
||||
wchartype.is_full_letter( char )):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_binary( value, binary_chars=None ):
|
||||
"""
|
||||
File is binary if it contains a null-byte by default (e.g. behavior of grep, etc.).
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
import wchartype
|
||||
|
||||
|
||||
def is_multi_byte( chars ):
|
||||
for char in chars:
|
||||
try:
|
||||
char = unicode( char )
|
||||
except UnicodeDecodeError:
|
||||
# Probably binary
|
||||
return False
|
||||
if ( wchartype.is_asian( char ) or wchartype.is_full_width( char ) or
|
||||
wchartype.is_kanji( char ) or wchartype.is_hiragana( char ) or
|
||||
wchartype.is_katakana( char ) or wchartype.is_half_katakana( char ) or
|
||||
wchartype.is_hangul( char ) or wchartype.is_full_digit( char ) or
|
||||
wchartype.is_full_letter( char )):
|
||||
return True
|
||||
return False
|
||||
@@ -22,6 +22,7 @@ from galaxy.datatypes.checkers import check_binary, check_bz2, check_gzip, check
|
||||
from galaxy.datatypes.registry import Registry
|
||||
from galaxy.datatypes.util.image_util import get_image_ext
|
||||
from galaxy.util.json import dumps, loads
|
||||
from galaxy.util import multi_byte
|
||||
|
||||
try:
|
||||
import Image as PIL
|
||||
@@ -111,7 +112,7 @@ def add_file( dataset, registry, json_file, output_path ):
|
||||
if not dataset.type == 'url':
|
||||
# Already set is_multi_byte above if type == 'url'
|
||||
try:
|
||||
dataset.is_multi_byte = util.is_multi_byte( codecs.open( dataset.path, 'r', 'utf-8' ).read( 100 ) )
|
||||
dataset.is_multi_byte = multi_byte.is_multi_byte( codecs.open( dataset.path, 'r', 'utf-8' ).read( 100 ) )
|
||||
except UnicodeDecodeError, e:
|
||||
dataset.is_multi_byte = False
|
||||
# Is dataset an image?
|
||||
|
||||
Reference in New Issue
Block a user