faster re.compile in validators
This commit is contained in:
@@ -1 +1 @@
|
|||||||
Version 2.0.6 (2012-09-02 12:12:55) stable
|
Version 2.0.6 (2012-09-02 12:26:12) stable
|
||||||
|
|||||||
+23
-20
@@ -1376,6 +1376,7 @@ class IS_GENERIC_URL(Validator):
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
error_message='enter a valid URL',
|
error_message='enter a valid URL',
|
||||||
@@ -1402,6 +1403,9 @@ class IS_GENERIC_URL(Validator):
|
|||||||
"prepend_scheme='%s' is not in allowed_schemes=%s" \
|
"prepend_scheme='%s' is not in allowed_schemes=%s" \
|
||||||
% (self.prepend_scheme, self.allowed_schemes)
|
% (self.prepend_scheme, self.allowed_schemes)
|
||||||
|
|
||||||
|
GENERIC_URL = re.compile(r"%[^0-9A-Fa-f]{2}|%[^0-9A-Fa-f][0-9A-Fa-f]|%[0-9A-Fa-f][^0-9A-Fa-f]|%$|%[0-9A-Fa-f]$|%[^0-9A-Fa-f]$")
|
||||||
|
GENERIC_URL_VALID = re.compile(r"[A-Za-z0-9;/?:@&=+$,\-_\.!~*'\(\)%#]+$")
|
||||||
|
|
||||||
def __call__(self, value):
|
def __call__(self, value):
|
||||||
"""
|
"""
|
||||||
:param value: a string, the URL to validate
|
:param value: a string, the URL to validate
|
||||||
@@ -1411,12 +1415,9 @@ class IS_GENERIC_URL(Validator):
|
|||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
# if the URL does not misuse the '%' character
|
# if the URL does not misuse the '%' character
|
||||||
if not re.compile(
|
if not self.GENERIC_URL.search(value):
|
||||||
r"%[^0-9A-Fa-f]{2}|%[^0-9A-Fa-f][0-9A-Fa-f]|%[0-9A-Fa-f][^0-9A-Fa-f]|%$|%[0-9A-Fa-f]$|%[^0-9A-Fa-f]$"
|
|
||||||
).search(value):
|
|
||||||
# if the URL is only composed of valid characters
|
# if the URL is only composed of valid characters
|
||||||
if re.compile(
|
if self.GENERIC_URL_VALID.match(value):
|
||||||
r"[A-Za-z0-9;/?:@&=+$,\-_\.!~*'\(\)%#]+$").match(value):
|
|
||||||
# Then split up the URL into its components and check on
|
# Then split up the URL into its components and check on
|
||||||
# the scheme
|
# the scheme
|
||||||
scheme = url_split_regex.match(value).group(2)
|
scheme = url_split_regex.match(value).group(2)
|
||||||
@@ -1432,11 +1433,10 @@ class IS_GENERIC_URL(Validator):
|
|||||||
# ports, check to see if adding a valid scheme fixes
|
# ports, check to see if adding a valid scheme fixes
|
||||||
# the problem (but only do this if it doesn't have
|
# the problem (but only do this if it doesn't have
|
||||||
# one already!)
|
# one already!)
|
||||||
if not re.compile('://').search(value) and None\
|
if value.find('://')<0 and None in self.allowed_schemes:
|
||||||
in self.allowed_schemes:
|
|
||||||
schemeToUse = self.prepend_scheme or 'http'
|
schemeToUse = self.prepend_scheme or 'http'
|
||||||
prependTest = self.__call__(schemeToUse
|
prependTest = self.__call__(
|
||||||
+ '://' + value)
|
schemeToUse + '://' + value)
|
||||||
# if the prepend test succeeded
|
# if the prepend test succeeded
|
||||||
if prependTest[1] is None:
|
if prependTest[1] is None:
|
||||||
# if prepending in the output is enabled
|
# if prepending in the output is enabled
|
||||||
@@ -1791,6 +1791,9 @@ class IS_HTTP_URL(Validator):
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
GENERIC_VALID_IP = re.compile("([\w.!~*'|;:&=+$,-]+@)?\d+\.\d+\.\d+\.\d+(:\d*)*$")
|
||||||
|
GENERIC_VALID_DOMAIN = re.compile("([\w.!~*'|;:&=+$,-]+@)?(([A-Za-z0-9]+[A-Za-z0-9\-]*[A-Za-z0-9]+\.)*([A-Za-z0-9]+\.)*)*([A-Za-z]+[A-Za-z0-9\-]*[A-Za-z0-9]+)\.?(:\d*)*$")
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
error_message='enter a valid URL',
|
error_message='enter a valid URL',
|
||||||
@@ -1843,16 +1846,12 @@ class IS_HTTP_URL(Validator):
|
|||||||
# if there is an authority component
|
# if there is an authority component
|
||||||
if authority:
|
if authority:
|
||||||
# if authority is a valid IP address
|
# if authority is a valid IP address
|
||||||
if re.compile(
|
if self.GENERIC_VALID_IP.match(authority):
|
||||||
"([\w.!~*'|;:&=+$,-]+@)?\d+\.\d+\.\d+\.\d+(:\d*)*$").match(authority):
|
|
||||||
# Then this HTTP URL is valid
|
# Then this HTTP URL is valid
|
||||||
return (value, None)
|
return (value, None)
|
||||||
else:
|
else:
|
||||||
# else if authority is a valid domain name
|
# else if authority is a valid domain name
|
||||||
domainMatch = \
|
domainMatch = self.GENERIC_VALID_DOMAIN.match(authority)
|
||||||
re.compile(
|
|
||||||
"([\w.!~*'|;:&=+$,-]+@)?(([A-Za-z0-9]+[A-Za-z0-9\-]*[A-Za-z0-9]+\.)*([A-Za-z0-9]+\.)*)*([A-Za-z]+[A-Za-z0-9\-]*[A-Za-z0-9]+)\.?(:\d*)*$"
|
|
||||||
).match(authority)
|
|
||||||
if domainMatch:
|
if domainMatch:
|
||||||
# if the top-level domain really exists
|
# if the top-level domain really exists
|
||||||
if domainMatch.group(5).lower()\
|
if domainMatch.group(5).lower()\
|
||||||
@@ -1865,13 +1864,13 @@ class IS_HTTP_URL(Validator):
|
|||||||
path = componentsMatch.group(5)
|
path = componentsMatch.group(5)
|
||||||
# relative case: if this is a valid path (if it starts with
|
# relative case: if this is a valid path (if it starts with
|
||||||
# a slash)
|
# a slash)
|
||||||
if re.compile('/').match(path):
|
if path.startswith('/'):
|
||||||
# Then this HTTP URL is valid
|
# Then this HTTP URL is valid
|
||||||
return (value, None)
|
return (value, None)
|
||||||
else:
|
else:
|
||||||
# abbreviated case: if we haven't already, prepend a
|
# abbreviated case: if we haven't already, prepend a
|
||||||
# scheme and see if it fixes the problem
|
# scheme and see if it fixes the problem
|
||||||
if not re.compile('://').search(value):
|
if value.find('://')<0:
|
||||||
schemeToUse = self.prepend_scheme or 'http'
|
schemeToUse = self.prepend_scheme or 'http'
|
||||||
prependTest = self.__call__(schemeToUse
|
prependTest = self.__call__(schemeToUse
|
||||||
+ '://' + value)
|
+ '://' + value)
|
||||||
@@ -2521,9 +2520,11 @@ class CLEANUP(Validator):
|
|||||||
|
|
||||||
removes special characters on validation
|
removes special characters on validation
|
||||||
"""
|
"""
|
||||||
|
REGEX_CLEANUP = re.compile('[^\x09\x0a\x0d\x20-\x7e]')
|
||||||
|
|
||||||
def __init__(self, regex='[^\x09\x0a\x0d\x20-\x7e]'):
|
def __init__(self, regex=None):
|
||||||
self.regex = re.compile(regex)
|
self.regex = self.REGEX_CLEANUP if regex is None \
|
||||||
|
else re.compile(regex)
|
||||||
|
|
||||||
def __call__(self, value):
|
def __call__(self, value):
|
||||||
v = self.regex.sub('',str(value).strip())
|
v = self.regex.sub('',str(value).strip())
|
||||||
@@ -2790,11 +2791,13 @@ class IS_STRONG(object):
|
|||||||
|
|
||||||
class IS_IN_SUBSET(IS_IN_SET):
|
class IS_IN_SUBSET(IS_IN_SET):
|
||||||
|
|
||||||
|
REGEX_W = re.compile('\w+')
|
||||||
|
|
||||||
def __init__(self, *a, **b):
|
def __init__(self, *a, **b):
|
||||||
IS_IN_SET.__init__(self, *a, **b)
|
IS_IN_SET.__init__(self, *a, **b)
|
||||||
|
|
||||||
def __call__(self, value):
|
def __call__(self, value):
|
||||||
values = re.compile("\w+").findall(str(value))
|
values = self.REGEX_W.findall(str(value))
|
||||||
failures = [x for x in values if IS_IN_SET.__call__(self, x)[1]]
|
failures = [x for x in values if IS_IN_SET.__call__(self, x)[1]]
|
||||||
if failures:
|
if failures:
|
||||||
return (value, translate(self.error_message))
|
return (value, translate(self.error_message))
|
||||||
|
|||||||
Reference in New Issue
Block a user