Tornado update
This commit is contained in:
+41
-54
@@ -20,49 +20,34 @@ Also includes a few other miscellaneous string manipulation functions that
|
||||
have crept in over time.
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import, division, with_statement
|
||||
from __future__ import absolute_import, division, print_function, with_statement
|
||||
|
||||
import htmlentitydefs
|
||||
import re
|
||||
import sys
|
||||
import urllib
|
||||
|
||||
# Python3 compatibility: On python2.5, introduce the bytes alias from 2.6
|
||||
try:
|
||||
bytes
|
||||
except Exception:
|
||||
bytes = str
|
||||
from tornado.util import bytes_type, unicode_type, basestring_type, u
|
||||
|
||||
try:
|
||||
from urlparse import parse_qs # Python 2.6+
|
||||
from urllib.parse import parse_qs # py3
|
||||
except ImportError:
|
||||
from cgi import parse_qs
|
||||
from urlparse import parse_qs # Python 2.6+
|
||||
|
||||
# json module is in the standard library as of python 2.6; fall back to
|
||||
# simplejson if present for older versions.
|
||||
try:
|
||||
import json
|
||||
assert hasattr(json, "loads") and hasattr(json, "dumps")
|
||||
_json_decode = json.loads
|
||||
_json_encode = json.dumps
|
||||
except Exception:
|
||||
try:
|
||||
import simplejson
|
||||
_json_decode = lambda s: simplejson.loads(_unicode(s))
|
||||
_json_encode = lambda v: simplejson.dumps(v)
|
||||
except ImportError:
|
||||
try:
|
||||
# For Google AppEngine
|
||||
from django.utils import simplejson
|
||||
_json_decode = lambda s: simplejson.loads(_unicode(s))
|
||||
_json_encode = lambda v: simplejson.dumps(v)
|
||||
except ImportError:
|
||||
def _json_decode(s):
|
||||
raise NotImplementedError(
|
||||
"A JSON parser is required, e.g., simplejson at "
|
||||
"http://pypi.python.org/pypi/simplejson/")
|
||||
_json_encode = _json_decode
|
||||
import htmlentitydefs # py2
|
||||
except ImportError:
|
||||
import html.entities as htmlentitydefs # py3
|
||||
|
||||
try:
|
||||
import urllib.parse as urllib_parse # py3
|
||||
except ImportError:
|
||||
import urllib as urllib_parse # py2
|
||||
|
||||
import json
|
||||
|
||||
try:
|
||||
unichr
|
||||
except NameError:
|
||||
unichr = chr
|
||||
|
||||
_XHTML_ESCAPE_RE = re.compile('[&<>"]')
|
||||
_XHTML_ESCAPE_DICT = {'&': '&', '<': '<', '>': '>', '"': '"'}
|
||||
@@ -87,12 +72,12 @@ def json_encode(value):
|
||||
# the javscript. Some json libraries do this escaping by default,
|
||||
# although python's standard library does not, so we do it here.
|
||||
# http://stackoverflow.com/questions/1580647/json-why-are-forward-slashes-escaped
|
||||
return _json_encode(recursive_unicode(value)).replace("</", "<\\/")
|
||||
return json.dumps(recursive_unicode(value)).replace("</", "<\\/")
|
||||
|
||||
|
||||
def json_decode(value):
|
||||
"""Returns Python objects for the given JSON string."""
|
||||
return _json_decode(to_basestring(value))
|
||||
return json.loads(to_basestring(value))
|
||||
|
||||
|
||||
def squeeze(value):
|
||||
@@ -102,7 +87,7 @@ def squeeze(value):
|
||||
|
||||
def url_escape(value):
|
||||
"""Returns a valid URL-encoded version of the given value."""
|
||||
return urllib.quote_plus(utf8(value))
|
||||
return urllib_parse.quote_plus(utf8(value))
|
||||
|
||||
# python 3 changed things around enough that we need two separate
|
||||
# implementations of url_unescape. We also need our own implementation
|
||||
@@ -117,9 +102,9 @@ if sys.version_info[0] < 3:
|
||||
the result is a unicode string in the specified encoding.
|
||||
"""
|
||||
if encoding is None:
|
||||
return urllib.unquote_plus(utf8(value))
|
||||
return urllib_parse.unquote_plus(utf8(value))
|
||||
else:
|
||||
return unicode(urllib.unquote_plus(utf8(value)), encoding)
|
||||
return unicode_type(urllib_parse.unquote_plus(utf8(value)), encoding)
|
||||
|
||||
parse_qs_bytes = parse_qs
|
||||
else:
|
||||
@@ -132,9 +117,9 @@ else:
|
||||
the result is a unicode string in the specified encoding.
|
||||
"""
|
||||
if encoding is None:
|
||||
return urllib.parse.unquote_to_bytes(value)
|
||||
return urllib_parse.unquote_to_bytes(value)
|
||||
else:
|
||||
return urllib.unquote_plus(to_basestring(value), encoding=encoding)
|
||||
return urllib_parse.unquote_plus(to_basestring(value), encoding=encoding)
|
||||
|
||||
def parse_qs_bytes(qs, keep_blank_values=False, strict_parsing=False):
|
||||
"""Parses a query string like urlparse.parse_qs, but returns the
|
||||
@@ -149,12 +134,12 @@ else:
|
||||
result = parse_qs(qs, keep_blank_values, strict_parsing,
|
||||
encoding='latin1', errors='strict')
|
||||
encoded = {}
|
||||
for k, v in result.iteritems():
|
||||
for k, v in result.items():
|
||||
encoded[k] = [i.encode('latin1') for i in v]
|
||||
return encoded
|
||||
|
||||
|
||||
_UTF8_TYPES = (bytes, type(None))
|
||||
_UTF8_TYPES = (bytes_type, type(None))
|
||||
|
||||
|
||||
def utf8(value):
|
||||
@@ -165,10 +150,10 @@ def utf8(value):
|
||||
"""
|
||||
if isinstance(value, _UTF8_TYPES):
|
||||
return value
|
||||
assert isinstance(value, unicode)
|
||||
assert isinstance(value, unicode_type)
|
||||
return value.encode("utf-8")
|
||||
|
||||
_TO_UNICODE_TYPES = (unicode, type(None))
|
||||
_TO_UNICODE_TYPES = (unicode_type, type(None))
|
||||
|
||||
|
||||
def to_unicode(value):
|
||||
@@ -179,7 +164,7 @@ def to_unicode(value):
|
||||
"""
|
||||
if isinstance(value, _TO_UNICODE_TYPES):
|
||||
return value
|
||||
assert isinstance(value, bytes)
|
||||
assert isinstance(value, bytes_type)
|
||||
return value.decode("utf-8")
|
||||
|
||||
# to_unicode was previously named _unicode not because it was private,
|
||||
@@ -188,12 +173,12 @@ _unicode = to_unicode
|
||||
|
||||
# When dealing with the standard library across python 2 and 3 it is
|
||||
# sometimes useful to have a direct conversion to the native string type
|
||||
if str is unicode:
|
||||
if str is unicode_type:
|
||||
native_str = to_unicode
|
||||
else:
|
||||
native_str = utf8
|
||||
|
||||
_BASESTRING_TYPES = (basestring, type(None))
|
||||
_BASESTRING_TYPES = (basestring_type, type(None))
|
||||
|
||||
|
||||
def to_basestring(value):
|
||||
@@ -207,7 +192,7 @@ def to_basestring(value):
|
||||
"""
|
||||
if isinstance(value, _BASESTRING_TYPES):
|
||||
return value
|
||||
assert isinstance(value, bytes)
|
||||
assert isinstance(value, bytes_type)
|
||||
return value.decode("utf-8")
|
||||
|
||||
|
||||
@@ -217,12 +202,12 @@ def recursive_unicode(obj):
|
||||
Supports lists, tuples, and dictionaries.
|
||||
"""
|
||||
if isinstance(obj, dict):
|
||||
return dict((recursive_unicode(k), recursive_unicode(v)) for (k, v) in obj.iteritems())
|
||||
return dict((recursive_unicode(k), recursive_unicode(v)) for (k, v) in obj.items())
|
||||
elif isinstance(obj, list):
|
||||
return list(recursive_unicode(i) for i in obj)
|
||||
elif isinstance(obj, tuple):
|
||||
return tuple(recursive_unicode(i) for i in obj)
|
||||
elif isinstance(obj, bytes):
|
||||
elif isinstance(obj, bytes_type):
|
||||
return to_unicode(obj)
|
||||
else:
|
||||
return obj
|
||||
@@ -232,7 +217,9 @@ def recursive_unicode(obj):
|
||||
# but it gets all exponential on certain patterns (such as too many trailing
|
||||
# dots), causing the regex matcher to never return.
|
||||
# This regex should avoid those problems.
|
||||
_URL_RE = re.compile(ur"""\b((?:([\w-]+):(/{1,3})|www[.])(?:(?:(?:[^\s&()]|&|")*(?:[^!"#$%&'()*+,.:;<=>?@\[\]^`{|}~\s]))|(?:\((?:[^\s&()]|&|")*\)))+)""")
|
||||
# Use to_unicode instead of tornado.util.u - we don't want backslashes getting
|
||||
# processed as escapes.
|
||||
_URL_RE = re.compile(to_unicode(r"""\b((?:([\w-]+):(/{1,3})|www[.])(?:(?:(?:[^\s&()]|&|")*(?:[^!"#$%&'()*+,.:;<=>?@\[\]^`{|}~\s]))|(?:\((?:[^\s&()]|&|")*\)))+)"""))
|
||||
|
||||
|
||||
def linkify(text, shorten=False, extra_params="",
|
||||
@@ -302,7 +289,7 @@ def linkify(text, shorten=False, extra_params="",
|
||||
# (no more slug, etc), so it really just provides a little
|
||||
# extra indication of shortening.
|
||||
url = url[:proto_len] + parts[0] + "/" + \
|
||||
parts[1][:8].split('?')[0].split('.')[0]
|
||||
parts[1][:8].split('?')[0].split('.')[0]
|
||||
|
||||
if len(url) > max_len * 1.5: # still too long
|
||||
url = url[:max_len]
|
||||
@@ -321,7 +308,7 @@ def linkify(text, shorten=False, extra_params="",
|
||||
# have a status bar, such as Safari by default)
|
||||
params += ' title="%s"' % href
|
||||
|
||||
return u'<a href="%s"%s>%s</a>' % (href, params, url)
|
||||
return u('<a href="%s"%s>%s</a>') % (href, params, url)
|
||||
|
||||
# First HTML-escape so that our strings are all safe.
|
||||
# The regex is modified to avoid character entites other than & so
|
||||
@@ -344,7 +331,7 @@ def _convert_entity(m):
|
||||
|
||||
def _build_unicode_map():
|
||||
unicode_map = {}
|
||||
for name, value in htmlentitydefs.name2codepoint.iteritems():
|
||||
for name, value in htmlentitydefs.name2codepoint.items():
|
||||
unicode_map[name] = unichr(value)
|
||||
return unicode_map
|
||||
|
||||
|
||||
Reference in New Issue
Block a user