Library update
This commit is contained in:
Regular → Executable
Regular → Executable
Regular → Executable
+1
-1
@@ -18,7 +18,7 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__version__ = '0.4'
|
||||
__version__ = '0.5-dev'
|
||||
__all__ = ['Guess', 'Language',
|
||||
'guess_file_info', 'guess_video_info',
|
||||
'guess_movie_info', 'guess_episode_info']
|
||||
|
||||
Regular → Executable
+3
-1
@@ -20,6 +20,7 @@
|
||||
|
||||
from __future__ import unicode_literals
|
||||
from guessit import fileutils
|
||||
from guessit.textutils import to_unicode
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -66,7 +67,8 @@ class Country(object):
|
||||
"""
|
||||
|
||||
def __init__(self, country, strict=False):
|
||||
self.alpha3 = country_to_alpha3.get(country.lower())
|
||||
country = to_unicode(country.strip().lower())
|
||||
self.alpha3 = country_to_alpha3.get(country)
|
||||
|
||||
if self.alpha3 is None and strict:
|
||||
msg = 'The given string "%s" could not be identified as a country'
|
||||
|
||||
Regular → Executable
Regular → Executable
+1
-6
@@ -29,6 +29,7 @@ def split_path(path):
|
||||
If the given path was an absolute path, the first element will always be:
|
||||
- the '/' root folder on Unix systems
|
||||
- the drive letter on Windows systems (eg: r'C:\')
|
||||
- the mount point '\\' on Windows systems (eg: r'\\host\share')
|
||||
|
||||
>>> split_path('/usr/bin/smewt')
|
||||
['/', 'usr', 'bin', 'smewt']
|
||||
@@ -36,12 +37,6 @@ def split_path(path):
|
||||
>>> split_path('relative_path/to/my_folder/')
|
||||
['relative_path', 'to', 'my_folder']
|
||||
|
||||
>>> split_path(r'C:\Program Files\Smewt\smewt.exe')
|
||||
['C:\\', 'Program Files', 'Smewt', 'smewt.exe']
|
||||
|
||||
>>> split_path(r'Documents and Settings\User\config\\')
|
||||
['Documents and Settings', 'User', 'config']
|
||||
|
||||
"""
|
||||
result = []
|
||||
while True:
|
||||
|
||||
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
+66
-15
@@ -21,11 +21,13 @@
|
||||
from __future__ import unicode_literals
|
||||
from guessit import fileutils
|
||||
from guessit.country import Country
|
||||
from guessit.textutils import to_unicode
|
||||
import re
|
||||
import logging
|
||||
|
||||
__all__ = [ 'is_iso_language', 'is_language', 'lang_set', 'Language',
|
||||
'ALL_LANGUAGES', 'ALL_LANGUAGES_NAMES', 'search_language' ]
|
||||
'ALL_LANGUAGES', 'ALL_LANGUAGES_NAMES', 'UNDETERMINED',
|
||||
'search_language' ]
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -46,14 +48,21 @@ _iso639_contents = _iso639_contents[1:]
|
||||
language_matrix = [ l.strip().split('|')
|
||||
for l in _iso639_contents.strip().split('\n') ]
|
||||
|
||||
language_matrix += [ [ 'unk', '', 'un', 'Unknown', 'inconnu' ] ]
|
||||
|
||||
# update information in the language matrix
|
||||
language_matrix += [['mol', '', 'mo', 'Moldavian', 'moldave'],
|
||||
['ass', '', '', 'Assyrian', 'assyrien']]
|
||||
|
||||
# remove unused languages that shadow other common ones with a non-official form
|
||||
for lang in language_matrix:
|
||||
# remove unused languages that shadow other common ones with a non-official form
|
||||
if (lang[2] == 'se' or # Northern Sami shadows Swedish
|
||||
lang[2] == 'br'): # Breton shadows Brazilian
|
||||
language_matrix.remove(lang)
|
||||
lang[2] = ''
|
||||
# add missing information
|
||||
if lang[0] == 'und':
|
||||
lang[2] = 'un'
|
||||
if lang[0] == 'srp':
|
||||
lang[1] = 'scc' # from OpenSubtitles
|
||||
|
||||
|
||||
lng3 = frozenset(l[0] for l in language_matrix if l[0])
|
||||
@@ -87,12 +96,17 @@ lng_fr_name_to_lng3 = dict((fr_name.lower(), l[0])
|
||||
|
||||
# contains a list of exceptions: strings that should be parsed as a language
|
||||
# but which are not in an ISO form
|
||||
lng_exceptions = { 'gr': ('gre', None),
|
||||
lng_exceptions = { 'unknown': ('und', None),
|
||||
'inconnu': ('und', None),
|
||||
'unk': ('und', None),
|
||||
'un': ('und', None),
|
||||
'gr': ('gre', None),
|
||||
'greek': ('gre', None),
|
||||
'esp': ('spa', None),
|
||||
'español': ('spa', None),
|
||||
'se': ('swe', None),
|
||||
'po': ('pt', 'br'),
|
||||
'pb': ('pt', 'br'),
|
||||
'pob': ('pt', 'br'),
|
||||
'br': ('pt', 'br'),
|
||||
'brazilian': ('pt', 'br'),
|
||||
@@ -101,7 +115,8 @@ lng_exceptions = { 'gr': ('gre', None),
|
||||
'ua': ('ukr', None),
|
||||
'cn': ('chi', None),
|
||||
'chs': ('chi', None),
|
||||
'jp': ('jpn', None)
|
||||
'jp': ('jpn', None),
|
||||
'scr': ('hrv', None)
|
||||
}
|
||||
|
||||
|
||||
@@ -130,6 +145,11 @@ class Language(object):
|
||||
You can also distinguish languages for specific countries, such as
|
||||
Portuguese and Brazilian Portuguese.
|
||||
|
||||
There are various properties on the language object that give you the
|
||||
representation of the language for a specific usage, such as .alpha3
|
||||
to get the ISO 3-letter code, or .opensubtitles to get the OpenSubtitles
|
||||
language code.
|
||||
|
||||
>>> Language('fr')
|
||||
Language(French)
|
||||
|
||||
@@ -146,16 +166,19 @@ class Language(object):
|
||||
True
|
||||
|
||||
>>> Language('zz', strict=False).english_name
|
||||
u'Unknown'
|
||||
u'Undetermined'
|
||||
|
||||
>>> Language('pt(br)').opensubtitles
|
||||
u'pob'
|
||||
"""
|
||||
|
||||
_with_country_regexp = re.compile('(.*)\((.*)\)')
|
||||
_with_country_regexp2 = re.compile('(.*)-(.*)')
|
||||
|
||||
def __init__(self, language, country=None, strict=False):
|
||||
language = language.strip().lower()
|
||||
if isinstance(language, str):
|
||||
language = language.decode('utf-8')
|
||||
with_country = Language._with_country_regexp.match(language)
|
||||
def __init__(self, language, country=None, strict=False, scheme=None):
|
||||
language = to_unicode(language.strip().lower())
|
||||
with_country = (Language._with_country_regexp.match(language) or
|
||||
Language._with_country_regexp2.match(language))
|
||||
if with_country:
|
||||
self.lang = Language(with_country.group(1)).lang
|
||||
self.country = Country(with_country.group(2))
|
||||
@@ -164,6 +187,18 @@ class Language(object):
|
||||
self.lang = None
|
||||
self.country = Country(country) if country else None
|
||||
|
||||
# first look for scheme specific languages
|
||||
if scheme == 'opensubtitles':
|
||||
if language == 'br':
|
||||
self.lang = 'bre'
|
||||
return
|
||||
elif language == 'se':
|
||||
self.lang = 'sme'
|
||||
return
|
||||
elif scheme is not None:
|
||||
log.warning('Unrecognized scheme: "%s" - Proceeding with standard one' % scheme)
|
||||
|
||||
# look for ISO language codes
|
||||
if len(language) == 2:
|
||||
self.lang = lng2_to_lng3.get(language)
|
||||
elif len(language) == 3:
|
||||
@@ -174,6 +209,7 @@ class Language(object):
|
||||
self.lang = (lng_en_name_to_lng3.get(language) or
|
||||
lng_fr_name_to_lng3.get(language))
|
||||
|
||||
# general language exceptions
|
||||
if self.lang is None and language in lng_exceptions:
|
||||
lang, country = lng_exceptions[language]
|
||||
self.lang = Language(lang).alpha3
|
||||
@@ -186,7 +222,7 @@ class Language(object):
|
||||
|
||||
if self.lang is None:
|
||||
log.debug(msg)
|
||||
self.lang = 'unk'
|
||||
self.lang = 'und'
|
||||
|
||||
@property
|
||||
def alpha2(self):
|
||||
@@ -208,6 +244,20 @@ class Language(object):
|
||||
def french_name(self):
|
||||
return lng3_to_lng_fr_name[self.lang]
|
||||
|
||||
@property
|
||||
def opensubtitles(self):
|
||||
if self.lang == 'por' and self.country and self.country.alpha2 == 'br':
|
||||
return 'pob'
|
||||
elif self.lang in ['gre', 'srp']:
|
||||
return self.alpha3term
|
||||
return self.alpha3
|
||||
|
||||
@property
|
||||
def tmdb(self):
|
||||
if self.country:
|
||||
return '%s-%s' % (self.alpha2, self.country.alpha2.upper())
|
||||
return self.alpha2
|
||||
|
||||
def __hash__(self):
|
||||
return hash(self.lang)
|
||||
|
||||
@@ -227,7 +277,7 @@ class Language(object):
|
||||
return not self == other
|
||||
|
||||
def __nonzero__(self):
|
||||
return self.lang != 'unk'
|
||||
return self.lang != 'und'
|
||||
|
||||
def __unicode__(self):
|
||||
if self.country:
|
||||
@@ -245,7 +295,8 @@ class Language(object):
|
||||
return 'Language(%s)' % self.english_name
|
||||
|
||||
|
||||
ALL_LANGUAGES = frozenset(Language(lng) for lng in lng_all_names) - frozenset([Language('unk')])
|
||||
UNDETERMINED = Language('und')
|
||||
ALL_LANGUAGES = frozenset(Language(lng) for lng in lng_all_names) - frozenset([UNDETERMINED])
|
||||
ALL_LANGUAGES_NAMES = lng_all_names
|
||||
|
||||
def search_language(string, lang_filter=None):
|
||||
|
||||
Regular → Executable
Regular → Executable
@@ -40,10 +40,10 @@ episode_rexps = [ # ... Season 2 ...
|
||||
(r'saison (?P<season>[0-9]+)', 1.0, (0, 0)),
|
||||
|
||||
# ... s02e13 ...
|
||||
(r'[Ss](?P<season>[0-9]{1,2}).{,3}[EeXx](?P<episodeNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
|
||||
(r'[Ss](?P<season>[0-9]{1,2}).{,3}(?P<episodeNumber>(?:[EeXx][0-9]{1,2})+)[^0-9]', 1.0, (0, -1)),
|
||||
|
||||
# ... 2x13 ...
|
||||
(r'[^0-9](?P<season>[0-9]{1,2})x(?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
|
||||
(r'[^0-9](?P<season>[0-9]{1,2})(?P<episodeNumber>(?:[xX][0-9]{1,2})+)[^0-9]', 0.8, (1, -1)),
|
||||
|
||||
# ... s02 ...
|
||||
#(sep + r's(?P<season>[0-9]{1,2})' + sep, 0.6, (1, -1)),
|
||||
@@ -61,7 +61,7 @@ weak_episode_rexps = [ # ... 213 or 0106 ...
|
||||
(sep + r'(?P<episodeNumber>[0-9]{1,4})' + sep, (1, -1)),
|
||||
|
||||
# ... 2x13 ...
|
||||
(sep + r'[^0-9](?P<season>[0-9]{1,2})\.(?P<episodeNumber>[0-9]{2})[^0-9]' + sep, (1, -1)),
|
||||
(sep + r'[^0-9](?P<season>[0-9]{1,2})\.(?P<episodeNumber>[0-9]{1,2})[^0-9]' + sep, (1, -1)),
|
||||
|
||||
# ... e13 ... for a mini-series without a season number
|
||||
(r'e(?P<episodeNumber>[0-9]{1,4})[^0-9]', (0, -1)),
|
||||
|
||||
Regular → Executable
Regular → Executable
+22
@@ -19,6 +19,7 @@
|
||||
#
|
||||
|
||||
from guessit.patterns import sep
|
||||
import unicodedata
|
||||
import copy
|
||||
|
||||
# string-related functions
|
||||
@@ -70,6 +71,7 @@ def to_utf8(o):
|
||||
return [ to_utf8(i) for i in o ]
|
||||
elif isinstance(o, dict):
|
||||
# need to do it like that to handle Guess instances correctly
|
||||
# FIXME: why is that necessary?
|
||||
result = copy.deepcopy(o)
|
||||
for key, value in o.items():
|
||||
result[to_utf8(key)] = to_utf8(value)
|
||||
@@ -78,6 +80,26 @@ def to_utf8(o):
|
||||
else:
|
||||
return o
|
||||
|
||||
def to_unicode(o):
|
||||
"""Convert all strings found in the given object to normalized
|
||||
unicode strings, using the UTF-8 codec if needed."""
|
||||
|
||||
if isinstance(o, unicode):
|
||||
return unicodedata.normalize('NFC', o)
|
||||
if isinstance(o, str):
|
||||
return unicodedata.normalize('NFC', o.decode('utf-8'))
|
||||
elif isinstance(o, list):
|
||||
return [ to_unicode(i) for i in o ]
|
||||
elif isinstance(o, dict):
|
||||
# need to do it like that to handle Guess instances correctly
|
||||
#result = copy.deepcopy(o)
|
||||
for key, value in o.items():
|
||||
result[to_unicode(key)] = to_unicode(value)
|
||||
return result
|
||||
|
||||
else:
|
||||
return o
|
||||
|
||||
|
||||
def levenshtein(a, b):
|
||||
if not a:
|
||||
|
||||
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
+20
-3
@@ -26,14 +26,31 @@ import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
def number_list(s):
|
||||
return re.sub('[^0-9]+', ' ', s).split()
|
||||
|
||||
def guess_episodes_rexps(string):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, string, re.IGNORECASE)
|
||||
if match:
|
||||
return (Guess(match.groupdict(), confidence=confidence),
|
||||
(match.start() + span_adjust[0],
|
||||
match.end() + span_adjust[1]))
|
||||
result = (Guess(match.groupdict(), confidence=confidence),
|
||||
(match.start() + span_adjust[0],
|
||||
match.end() + span_adjust[1]))
|
||||
# episodes which have a season > 25 are most likely errors
|
||||
# (Simpsons is at 23!)
|
||||
if int(result[0].get('season', 0)) > 25:
|
||||
continue
|
||||
|
||||
# decide whether we have only a single episode number or an
|
||||
# episode list
|
||||
if result[0].get('episodeNumber'):
|
||||
eplist = number_list(result[0]['episodeNumber'])
|
||||
result[0].set('episodeNumber', int(eplist[0]), confidence=confidence)
|
||||
|
||||
if len(eplist) > 1:
|
||||
result[0].set('episodeList', map(int, eplist), confidence=confidence)
|
||||
|
||||
return result
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
Regular → Executable
+3
@@ -87,6 +87,9 @@ def guess_filetype(filename, filetype):
|
||||
upgrade_episode()
|
||||
break
|
||||
|
||||
if 'tvu.org.ru' in filename:
|
||||
upgrade_episode()
|
||||
|
||||
# if no episode info found, assume it's a movie
|
||||
upgrade_movie()
|
||||
|
||||
|
||||
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Reference in New Issue
Block a user