Library update

This commit is contained in:
Ruud
2012-06-11 09:54:15 +02:00
parent bd5170bc8e
commit 02855c7b9c
115 changed files with 22575 additions and 2880 deletions
Regular → Executable
View File
Regular → Executable
View File
Regular → Executable
+1 -1
View File
@@ -18,7 +18,7 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
__version__ = '0.4'
__version__ = '0.5-dev'
__all__ = ['Guess', 'Language',
'guess_file_info', 'guess_video_info',
'guess_movie_info', 'guess_episode_info']
Regular → Executable
+3 -1
View File
@@ -20,6 +20,7 @@
from __future__ import unicode_literals
from guessit import fileutils
from guessit.textutils import to_unicode
import logging
log = logging.getLogger(__name__)
@@ -66,7 +67,8 @@ class Country(object):
"""
def __init__(self, country, strict=False):
self.alpha3 = country_to_alpha3.get(country.lower())
country = to_unicode(country.strip().lower())
self.alpha3 = country_to_alpha3.get(country)
if self.alpha3 is None and strict:
msg = 'The given string "%s" could not be identified as a country'
Regular → Executable
View File
Regular → Executable
+1 -6
View File
@@ -29,6 +29,7 @@ def split_path(path):
If the given path was an absolute path, the first element will always be:
- the '/' root folder on Unix systems
- the drive letter on Windows systems (eg: r'C:\')
- the mount point '\\' on Windows systems (eg: r'\\host\share')
>>> split_path('/usr/bin/smewt')
['/', 'usr', 'bin', 'smewt']
@@ -36,12 +37,6 @@ def split_path(path):
>>> split_path('relative_path/to/my_folder/')
['relative_path', 'to', 'my_folder']
>>> split_path(r'C:\Program Files\Smewt\smewt.exe')
['C:\\', 'Program Files', 'Smewt', 'smewt.exe']
>>> split_path(r'Documents and Settings\User\config\\')
['Documents and Settings', 'User', 'config']
"""
result = []
while True:
Regular → Executable
View File
Regular → Executable
View File
Regular → Executable
View File
Regular → Executable
+66 -15
View File
@@ -21,11 +21,13 @@
from __future__ import unicode_literals
from guessit import fileutils
from guessit.country import Country
from guessit.textutils import to_unicode
import re
import logging
__all__ = [ 'is_iso_language', 'is_language', 'lang_set', 'Language',
'ALL_LANGUAGES', 'ALL_LANGUAGES_NAMES', 'search_language' ]
'ALL_LANGUAGES', 'ALL_LANGUAGES_NAMES', 'UNDETERMINED',
'search_language' ]
log = logging.getLogger(__name__)
@@ -46,14 +48,21 @@ _iso639_contents = _iso639_contents[1:]
language_matrix = [ l.strip().split('|')
for l in _iso639_contents.strip().split('\n') ]
language_matrix += [ [ 'unk', '', 'un', 'Unknown', 'inconnu' ] ]
# update information in the language matrix
language_matrix += [['mol', '', 'mo', 'Moldavian', 'moldave'],
['ass', '', '', 'Assyrian', 'assyrien']]
# remove unused languages that shadow other common ones with a non-official form
for lang in language_matrix:
# remove unused languages that shadow other common ones with a non-official form
if (lang[2] == 'se' or # Northern Sami shadows Swedish
lang[2] == 'br'): # Breton shadows Brazilian
language_matrix.remove(lang)
lang[2] = ''
# add missing information
if lang[0] == 'und':
lang[2] = 'un'
if lang[0] == 'srp':
lang[1] = 'scc' # from OpenSubtitles
lng3 = frozenset(l[0] for l in language_matrix if l[0])
@@ -87,12 +96,17 @@ lng_fr_name_to_lng3 = dict((fr_name.lower(), l[0])
# contains a list of exceptions: strings that should be parsed as a language
# but which are not in an ISO form
lng_exceptions = { 'gr': ('gre', None),
lng_exceptions = { 'unknown': ('und', None),
'inconnu': ('und', None),
'unk': ('und', None),
'un': ('und', None),
'gr': ('gre', None),
'greek': ('gre', None),
'esp': ('spa', None),
'español': ('spa', None),
'se': ('swe', None),
'po': ('pt', 'br'),
'pb': ('pt', 'br'),
'pob': ('pt', 'br'),
'br': ('pt', 'br'),
'brazilian': ('pt', 'br'),
@@ -101,7 +115,8 @@ lng_exceptions = { 'gr': ('gre', None),
'ua': ('ukr', None),
'cn': ('chi', None),
'chs': ('chi', None),
'jp': ('jpn', None)
'jp': ('jpn', None),
'scr': ('hrv', None)
}
@@ -130,6 +145,11 @@ class Language(object):
You can also distinguish languages for specific countries, such as
Portuguese and Brazilian Portuguese.
There are various properties on the language object that give you the
representation of the language for a specific usage, such as .alpha3
to get the ISO 3-letter code, or .opensubtitles to get the OpenSubtitles
language code.
>>> Language('fr')
Language(French)
@@ -146,16 +166,19 @@ class Language(object):
True
>>> Language('zz', strict=False).english_name
u'Unknown'
u'Undetermined'
>>> Language('pt(br)').opensubtitles
u'pob'
"""
_with_country_regexp = re.compile('(.*)\((.*)\)')
_with_country_regexp2 = re.compile('(.*)-(.*)')
def __init__(self, language, country=None, strict=False):
language = language.strip().lower()
if isinstance(language, str):
language = language.decode('utf-8')
with_country = Language._with_country_regexp.match(language)
def __init__(self, language, country=None, strict=False, scheme=None):
language = to_unicode(language.strip().lower())
with_country = (Language._with_country_regexp.match(language) or
Language._with_country_regexp2.match(language))
if with_country:
self.lang = Language(with_country.group(1)).lang
self.country = Country(with_country.group(2))
@@ -164,6 +187,18 @@ class Language(object):
self.lang = None
self.country = Country(country) if country else None
# first look for scheme specific languages
if scheme == 'opensubtitles':
if language == 'br':
self.lang = 'bre'
return
elif language == 'se':
self.lang = 'sme'
return
elif scheme is not None:
log.warning('Unrecognized scheme: "%s" - Proceeding with standard one' % scheme)
# look for ISO language codes
if len(language) == 2:
self.lang = lng2_to_lng3.get(language)
elif len(language) == 3:
@@ -174,6 +209,7 @@ class Language(object):
self.lang = (lng_en_name_to_lng3.get(language) or
lng_fr_name_to_lng3.get(language))
# general language exceptions
if self.lang is None and language in lng_exceptions:
lang, country = lng_exceptions[language]
self.lang = Language(lang).alpha3
@@ -186,7 +222,7 @@ class Language(object):
if self.lang is None:
log.debug(msg)
self.lang = 'unk'
self.lang = 'und'
@property
def alpha2(self):
@@ -208,6 +244,20 @@ class Language(object):
def french_name(self):
return lng3_to_lng_fr_name[self.lang]
@property
def opensubtitles(self):
if self.lang == 'por' and self.country and self.country.alpha2 == 'br':
return 'pob'
elif self.lang in ['gre', 'srp']:
return self.alpha3term
return self.alpha3
@property
def tmdb(self):
if self.country:
return '%s-%s' % (self.alpha2, self.country.alpha2.upper())
return self.alpha2
def __hash__(self):
return hash(self.lang)
@@ -227,7 +277,7 @@ class Language(object):
return not self == other
def __nonzero__(self):
return self.lang != 'unk'
return self.lang != 'und'
def __unicode__(self):
if self.country:
@@ -245,7 +295,8 @@ class Language(object):
return 'Language(%s)' % self.english_name
ALL_LANGUAGES = frozenset(Language(lng) for lng in lng_all_names) - frozenset([Language('unk')])
UNDETERMINED = Language('und')
ALL_LANGUAGES = frozenset(Language(lng) for lng in lng_all_names) - frozenset([UNDETERMINED])
ALL_LANGUAGES_NAMES = lng_all_names
def search_language(string, lang_filter=None):
Regular → Executable
View File
Regular → Executable
View File
+3 -3
View File
@@ -40,10 +40,10 @@ episode_rexps = [ # ... Season 2 ...
(r'saison (?P<season>[0-9]+)', 1.0, (0, 0)),
# ... s02e13 ...
(r'[Ss](?P<season>[0-9]{1,2}).{,3}[EeXx](?P<episodeNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
(r'[Ss](?P<season>[0-9]{1,2}).{,3}(?P<episodeNumber>(?:[EeXx][0-9]{1,2})+)[^0-9]', 1.0, (0, -1)),
# ... 2x13 ...
(r'[^0-9](?P<season>[0-9]{1,2})x(?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
(r'[^0-9](?P<season>[0-9]{1,2})(?P<episodeNumber>(?:[xX][0-9]{1,2})+)[^0-9]', 0.8, (1, -1)),
# ... s02 ...
#(sep + r's(?P<season>[0-9]{1,2})' + sep, 0.6, (1, -1)),
@@ -61,7 +61,7 @@ weak_episode_rexps = [ # ... 213 or 0106 ...
(sep + r'(?P<episodeNumber>[0-9]{1,4})' + sep, (1, -1)),
# ... 2x13 ...
(sep + r'[^0-9](?P<season>[0-9]{1,2})\.(?P<episodeNumber>[0-9]{2})[^0-9]' + sep, (1, -1)),
(sep + r'[^0-9](?P<season>[0-9]{1,2})\.(?P<episodeNumber>[0-9]{1,2})[^0-9]' + sep, (1, -1)),
# ... e13 ... for a mini-series without a season number
(r'e(?P<episodeNumber>[0-9]{1,4})[^0-9]', (0, -1)),
Regular → Executable
View File
Regular → Executable
+22
View File
@@ -19,6 +19,7 @@
#
from guessit.patterns import sep
import unicodedata
import copy
# string-related functions
@@ -70,6 +71,7 @@ def to_utf8(o):
return [ to_utf8(i) for i in o ]
elif isinstance(o, dict):
# need to do it like that to handle Guess instances correctly
# FIXME: why is that necessary?
result = copy.deepcopy(o)
for key, value in o.items():
result[to_utf8(key)] = to_utf8(value)
@@ -78,6 +80,26 @@ def to_utf8(o):
else:
return o
def to_unicode(o):
"""Convert all strings found in the given object to normalized
unicode strings, using the UTF-8 codec if needed."""
if isinstance(o, unicode):
return unicodedata.normalize('NFC', o)
if isinstance(o, str):
return unicodedata.normalize('NFC', o.decode('utf-8'))
elif isinstance(o, list):
return [ to_unicode(i) for i in o ]
elif isinstance(o, dict):
# need to do it like that to handle Guess instances correctly
#result = copy.deepcopy(o)
for key, value in o.items():
result[to_unicode(key)] = to_unicode(value)
return result
else:
return o
def levenshtein(a, b):
if not a:
Regular → Executable
View File
View File
View File
View File
+20 -3
View File
@@ -26,14 +26,31 @@ import logging
log = logging.getLogger(__name__)
def number_list(s):
return re.sub('[^0-9]+', ' ', s).split()
def guess_episodes_rexps(string):
for rexp, confidence, span_adjust in episode_rexps:
match = re.search(rexp, string, re.IGNORECASE)
if match:
return (Guess(match.groupdict(), confidence=confidence),
(match.start() + span_adjust[0],
match.end() + span_adjust[1]))
result = (Guess(match.groupdict(), confidence=confidence),
(match.start() + span_adjust[0],
match.end() + span_adjust[1]))
# episodes which have a season > 25 are most likely errors
# (Simpsons is at 23!)
if int(result[0].get('season', 0)) > 25:
continue
# decide whether we have only a single episode number or an
# episode list
if result[0].get('episodeNumber'):
eplist = number_list(result[0]['episodeNumber'])
result[0].set('episodeNumber', int(eplist[0]), confidence=confidence)
if len(eplist) > 1:
result[0].set('episodeList', map(int, eplist), confidence=confidence)
return result
return None, None
+3
View File
@@ -87,6 +87,9 @@ def guess_filetype(filename, filetype):
upgrade_episode()
break
if 'tvu.org.ru' in filename:
upgrade_episode()
# if no episode info found, assume it's a movie
upgrade_movie()
View File
View File
View File
View File
View File
View File
View File
View File
View File
View File
View File
View File