Library update
This commit is contained in:
+10
-7
@@ -18,7 +18,8 @@
|
||||
from .core import (SERVICES, LANGUAGE_INDEX, SERVICE_INDEX, SERVICE_CONFIDENCE,
|
||||
MATCHING_CONFIDENCE, create_list_tasks, consume_task, create_download_tasks,
|
||||
group_by_video, key_subtitles)
|
||||
from .languages import list_languages
|
||||
import guessit
|
||||
from guessit.language import ALL_LANGUAGES
|
||||
import logging
|
||||
|
||||
|
||||
@@ -26,7 +27,7 @@ __all__ = ['list_subtitles', 'download_subtitles']
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def list_subtitles(paths, languages=None, services=None, force=True, multi=False, cache_dir=None, max_depth=3):
|
||||
def list_subtitles(paths, languages=None, services=None, force=True, multi=False, cache_dir=None, max_depth=3, scan_filter=None):
|
||||
"""List subtitles in given paths according to the criteria
|
||||
|
||||
:param paths: path(s) to video file or folder
|
||||
@@ -37,19 +38,20 @@ def list_subtitles(paths, languages=None, services=None, force=True, multi=False
|
||||
:param bool multi: search multiple languages for the same video
|
||||
:param string cache_dir: path to the cache directory to use
|
||||
:param int max_depth: maximum depth for scanning entries
|
||||
:param function scan_filter: filter function that takes a path as argument and returns a boolean indicating whether it has to be filtered out (``True``) or not (``False``)
|
||||
:return: found subtitles
|
||||
:rtype: dict of :class:`~subliminal.videos.Video` => [:class:`~subliminal.subtitles.ResultSubtitle`]
|
||||
|
||||
"""
|
||||
services = services or SERVICES
|
||||
languages = set(languages or list_languages(1))
|
||||
languages = set(map(guessit.Language, languages or []) or ALL_LANGUAGES)
|
||||
if isinstance(paths, basestring):
|
||||
paths = [paths]
|
||||
if any([not isinstance(p, unicode) for p in paths]):
|
||||
logger.warning(u'Not all entries are unicode')
|
||||
results = []
|
||||
service_instances = {}
|
||||
tasks = create_list_tasks(paths, languages, services, force, multi, cache_dir, max_depth)
|
||||
tasks = create_list_tasks(paths, languages, services, force, multi, cache_dir, max_depth, scan_filter)
|
||||
for task in tasks:
|
||||
try:
|
||||
result = consume_task(task, service_instances)
|
||||
@@ -61,7 +63,7 @@ def list_subtitles(paths, languages=None, services=None, force=True, multi=False
|
||||
return group_by_video(results)
|
||||
|
||||
|
||||
def download_subtitles(paths, languages=None, services=None, force=True, multi=False, cache_dir=None, max_depth=3, order=None):
|
||||
def download_subtitles(paths, languages=None, services=None, force=True, multi=False, cache_dir=None, max_depth=3, scan_filter=None, order=None):
|
||||
"""Download subtitles in given paths according to the criteria
|
||||
|
||||
:param paths: path(s) to video file or folder
|
||||
@@ -72,6 +74,7 @@ def download_subtitles(paths, languages=None, services=None, force=True, multi=F
|
||||
:param bool multi: search multiple languages for the same video
|
||||
:param string cache_dir: path to the cache directory to use
|
||||
:param int max_depth: maximum depth for scanning entries
|
||||
:param function scan_filter: filter function that takes a path as argument and returns a boolean indicating whether it has to be filtered out (``True``) or not (``False``)
|
||||
:param order: preferred order for subtitles sorting
|
||||
:type list: list of :data:`~subliminal.core.LANGUAGE_INDEX`, :data:`~subliminal.core.SERVICE_INDEX`, :data:`~subliminal.core.SERVICE_CONFIDENCE`, :data:`~subliminal.core.MATCHING_CONFIDENCE`
|
||||
:return: found subtitles
|
||||
@@ -79,11 +82,11 @@ def download_subtitles(paths, languages=None, services=None, force=True, multi=F
|
||||
|
||||
"""
|
||||
services = services or SERVICES
|
||||
languages = languages or list_languages(1)
|
||||
languages = map(guessit.Language, languages or []) or list(ALL_LANGUAGES)
|
||||
if isinstance(paths, basestring):
|
||||
paths = [paths]
|
||||
order = order or [LANGUAGE_INDEX, SERVICE_INDEX, SERVICE_CONFIDENCE, MATCHING_CONFIDENCE]
|
||||
subtitles_by_video = list_subtitles(paths, set(languages), services, force, multi, cache_dir, max_depth)
|
||||
subtitles_by_video = list_subtitles(paths, set(languages), services, force, multi, cache_dir, max_depth, scan_filter)
|
||||
for video, subtitles in subtitles_by_video.iteritems():
|
||||
subtitles.sort(key=lambda s: key_subtitles(s, video, languages, services, order), reverse=True)
|
||||
results = []
|
||||
|
||||
@@ -18,7 +18,7 @@
|
||||
from .core import (consume_task, LANGUAGE_INDEX, SERVICE_INDEX,
|
||||
SERVICE_CONFIDENCE, MATCHING_CONFIDENCE, SERVICES, create_list_tasks,
|
||||
create_download_tasks, group_by_video, key_subtitles)
|
||||
from .languages import list_languages
|
||||
from guessit.language import ALL_LANGUAGES
|
||||
from .tasks import StopTask
|
||||
import Queue
|
||||
import logging
|
||||
@@ -108,29 +108,29 @@ class Pool(object):
|
||||
break
|
||||
return results
|
||||
|
||||
def list_subtitles(self, paths, languages=None, services=None, force=True, multi=False, cache_dir=None, max_depth=3):
|
||||
def list_subtitles(self, paths, languages=None, services=None, force=True, multi=False, cache_dir=None, max_depth=3, scan_filter=None):
|
||||
"""See :meth:`subliminal.list_subtitles`"""
|
||||
services = services or SERVICES
|
||||
languages = set(languages or list_languages(1))
|
||||
languages = set(languages or ALL_LANGUAGES)
|
||||
if isinstance(paths, basestring):
|
||||
paths = [paths]
|
||||
if any([not isinstance(p, unicode) for p in paths]):
|
||||
logger.warning(u'Not all entries are unicode')
|
||||
tasks = create_list_tasks(paths, languages, services, force, multi, cache_dir, max_depth)
|
||||
tasks = create_list_tasks(paths, languages, services, force, multi, cache_dir, max_depth, scan_filter)
|
||||
for task in tasks:
|
||||
self.tasks.put(task)
|
||||
self.join()
|
||||
results = self.collect()
|
||||
return group_by_video(results)
|
||||
|
||||
def download_subtitles(self, paths, languages=None, services=None, cache_dir=None, max_depth=3, force=True, multi=False, order=None):
|
||||
def download_subtitles(self, paths, languages=None, services=None, force=True, multi=False, cache_dir=None, max_depth=3, scan_filter=None, order=None):
|
||||
"""See :meth:`subliminal.download_subtitles`"""
|
||||
services = services or SERVICES
|
||||
languages = languages or list_languages(1)
|
||||
languages = languages or list(ALL_LANGUAGES)
|
||||
if isinstance(paths, basestring):
|
||||
paths = [paths]
|
||||
order = order or [LANGUAGE_INDEX, SERVICE_INDEX, SERVICE_CONFIDENCE, MATCHING_CONFIDENCE]
|
||||
subtitles_by_video = self.list_subtitles(paths, set(languages), services, force, multi, cache_dir, max_depth)
|
||||
subtitles_by_video = self.list_subtitles(paths, set(languages), services, force, multi, cache_dir, max_depth, scan_filter)
|
||||
for video, subtitles in subtitles_by_video.iteritems():
|
||||
subtitles.sort(key=lambda s: key_subtitles(s, video, languages, services, order), reverse=True)
|
||||
tasks = create_download_tasks(subtitles_by_video, multi)
|
||||
|
||||
Executable
+132
@@ -0,0 +1,132 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# This file is part of subliminal.
|
||||
#
|
||||
# subliminal is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU Lesser General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# subliminal is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU Lesser General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
import os.path
|
||||
from collections import defaultdict
|
||||
import threading
|
||||
from functools import wraps
|
||||
import logging
|
||||
try:
|
||||
import cPickle as pickle
|
||||
except ImportError:
|
||||
import pickle
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Cache(object):
|
||||
"""A Cache object contains cached values for methods. It can have
|
||||
separate internal caches, one for each service.
|
||||
"""
|
||||
|
||||
def __init__(self, cache_dir):
|
||||
self.cache_dir = cache_dir
|
||||
self.cache = defaultdict(dict)
|
||||
self.lock = threading.RLock()
|
||||
|
||||
def __del__(self):
|
||||
for service_name in self.cache:
|
||||
self.save(service_name)
|
||||
|
||||
def cache_location(self, service_name):
|
||||
return os.path.join(self.cache_dir, 'subliminal_%s.cache' % service_name)
|
||||
|
||||
def load(self, service_name):
|
||||
with self.lock:
|
||||
if service_name in self.cache:
|
||||
# already loaded
|
||||
return
|
||||
|
||||
self.cache[service_name] = defaultdict(dict)
|
||||
filename = self.cache_location(service_name)
|
||||
logger.debug(u'Cache: loading cache from %s' % filename)
|
||||
try:
|
||||
self.cache[service_name] = pickle.load(open(filename, 'rb'))
|
||||
except IOError:
|
||||
logger.info('Cache: Cache file "%s" doesn\'t exist, creating it' % filename)
|
||||
except EOFError:
|
||||
logger.error('Cache: cache file "%s" is corrupted... Removing it.' % filename)
|
||||
os.remove(filename)
|
||||
|
||||
def save(self, service_name):
|
||||
filename = self.cache_location(service_name)
|
||||
logger.debug(u'Cache: saving cache to %s' % filename)
|
||||
with self.lock:
|
||||
pickle.dump(self.cache[service_name], open(filename, 'wb'))
|
||||
|
||||
def clear(self, service_name):
|
||||
try:
|
||||
os.remove(self.cache_location(service_name))
|
||||
except OSError:
|
||||
pass
|
||||
self.cache[service_name] = defaultdict(dict)
|
||||
|
||||
def cached_func_key(self, func, cls=None):
|
||||
try:
|
||||
cls = func.im_class
|
||||
except:
|
||||
pass
|
||||
return ('%s.%s' % (cls.__module__, cls.__name__), func.__name__)
|
||||
|
||||
def function_cache(self, service_name, func):
|
||||
func_key = self.cached_func_key(func)
|
||||
return self.cache[service_name][func_key]
|
||||
|
||||
def cache_for(self, service_name, func, args, result):
|
||||
# no need to lock here, dict ops are atomic
|
||||
self.function_cache(service_name, func)[args] = result
|
||||
|
||||
def cached_value(self, service_name, func, args):
|
||||
"""Raises KeyError if not found"""
|
||||
# no need to lock here, dict ops are atomic
|
||||
return self.function_cache(service_name, func)[args]
|
||||
|
||||
|
||||
def cachedmethod(function):
|
||||
"""Decorator to make a method use the cache.
|
||||
|
||||
WARNING: this can NOT be used with static functions, it has to be used on
|
||||
methods of some class."""
|
||||
|
||||
@wraps(function)
|
||||
def cached(*args):
|
||||
c = args[0].config.cache
|
||||
service_name = args[0].__class__.__name__
|
||||
func_key = c.cached_func_key(function, cls=args[0].__class__)
|
||||
func_cache = c.cache[service_name][func_key]
|
||||
|
||||
# we need to remove the first element of args for the key, as it is the
|
||||
# instance pointer and we don't want the cache to know which instance
|
||||
# called it, it is shared among all instances of the same class
|
||||
key = args[1:]
|
||||
|
||||
if key in func_cache:
|
||||
result = func_cache[key]
|
||||
logger.debug(u'Using cached value for %s(%s), returns: %s' % (func_key, key, result))
|
||||
return result
|
||||
|
||||
result = function(*args)
|
||||
|
||||
# note: another thread could have already cached a value in the
|
||||
# meantime, but that's ok as we prefer to keep the latest value in
|
||||
# the cache
|
||||
func_cache[key] = result
|
||||
|
||||
return result
|
||||
|
||||
return cached
|
||||
+53
-22
@@ -20,6 +20,8 @@ from .services import ServiceConfig
|
||||
from .tasks import DownloadTask, ListTask
|
||||
from .utils import get_keywords
|
||||
from .videos import Episode, Movie, scan
|
||||
from guessit.language import lang_set
|
||||
import bs4
|
||||
from collections import defaultdict
|
||||
from itertools import groupby
|
||||
import guessit
|
||||
@@ -30,11 +32,11 @@ __all__ = ['SERVICES', 'LANGUAGE_INDEX', 'SERVICE_INDEX', 'SERVICE_CONFIDENCE',
|
||||
'create_list_tasks', 'create_download_tasks', 'consume_task', 'matching_confidence',
|
||||
'key_subtitles', 'group_by_video']
|
||||
logger = logging.getLogger(__name__)
|
||||
SERVICES = ['opensubtitles', 'bierdopje', 'subswiki', 'subtitulos', 'thesubdb']
|
||||
SERVICES = ['opensubtitles', 'bierdopje', 'subswiki', 'subtitulos', 'thesubdb', 'addic7ed', 'tvsubtitles']
|
||||
LANGUAGE_INDEX, SERVICE_INDEX, SERVICE_CONFIDENCE, MATCHING_CONFIDENCE = range(4)
|
||||
|
||||
|
||||
def create_list_tasks(paths, languages, services, force, multi, cache_dir, max_depth):
|
||||
def create_list_tasks(paths, languages, services, force, multi, cache_dir, max_depth, scan_filter):
|
||||
"""Create a list of :class:`~subliminal.tasks.ListTask` from one or more paths using the given criteria
|
||||
|
||||
:param paths: path(s) to video file or folder
|
||||
@@ -45,18 +47,20 @@ def create_list_tasks(paths, languages, services, force, multi, cache_dir, max_d
|
||||
:param bool multi: search multiple languages for the same video
|
||||
:param string cache_dir: path to the cache directory to use
|
||||
:param int max_depth: maximum depth for scanning entries
|
||||
:param function scan_filter: filter function that takes a path as argument and returns a boolean indicating whether it has to be filtered out (``True``) or not (``False``)
|
||||
:return: the created tasks
|
||||
:rtype: list of :class:`~subliminal.tasks.ListTask`
|
||||
|
||||
"""
|
||||
scan_result = []
|
||||
for p in paths:
|
||||
scan_result.extend(scan(p, max_depth))
|
||||
scan_result.extend(scan(p, max_depth, scan_filter))
|
||||
logger.debug(u'Found %d videos in %r with maximum depth %d' % (len(scan_result), paths, max_depth))
|
||||
tasks = []
|
||||
config = ServiceConfig(multi, cache_dir)
|
||||
services = filter_services(services)
|
||||
for video, detected_subtitles in scan_result:
|
||||
detected_languages = set([s.language for s in detected_subtitles])
|
||||
detected_languages = set(s.language for s in detected_subtitles)
|
||||
wanted_languages = languages.copy()
|
||||
if not force and multi:
|
||||
wanted_languages -= detected_languages
|
||||
@@ -70,14 +74,9 @@ def create_list_tasks(paths, languages, services, force, multi, cache_dir, max_d
|
||||
for service_name in services:
|
||||
mod = __import__('services.' + service_name, globals=globals(), locals=locals(), fromlist=['Service'], level=-1)
|
||||
service = mod.Service
|
||||
service_languages = wanted_languages & service.available_languages()
|
||||
if not service_languages:
|
||||
logger.debug(u'Skipping %r: none of wanted languages %r available for service %s' % (video, wanted_languages, service_name))
|
||||
if not service.check_validity(video, wanted_languages):
|
||||
continue
|
||||
if not service.is_valid_video(video):
|
||||
logger.debug(u'Skipping %r: not part of supported videos %r for service %s' % (video, service.videos, service_name))
|
||||
continue
|
||||
task = ListTask(video, service_languages, service_name, config)
|
||||
task = ListTask(video, wanted_languages & service.languages, service_name, config)
|
||||
logger.debug(u'Created task %r' % task)
|
||||
tasks.append(task)
|
||||
return tasks
|
||||
@@ -128,25 +127,19 @@ def consume_task(task, services=None):
|
||||
logger.info(u'Consuming %r' % task)
|
||||
result = None
|
||||
if isinstance(task, ListTask):
|
||||
if task.service not in services:
|
||||
mod = __import__('services.' + task.service, globals=globals(), locals=locals(), fromlist=['Service'], level=-1)
|
||||
services[task.service] = mod.Service(task.config)
|
||||
services[task.service].init()
|
||||
subtitles = services[task.service].list(task.video, task.languages)
|
||||
result = subtitles
|
||||
service = get_service(services, task.service, config=task.config)
|
||||
result = service.list(task.video, task.languages)
|
||||
elif isinstance(task, DownloadTask):
|
||||
for subtitle in task.subtitles:
|
||||
if subtitle.service not in services:
|
||||
mod = __import__('services.' + subtitle.service, globals=globals(), locals=locals(), fromlist=['Service'], level=-1)
|
||||
services[subtitle.service] = mod.Service()
|
||||
services[subtitle.service].init()
|
||||
service = get_service(services, subtitle.service)
|
||||
try:
|
||||
services[subtitle.service].download(subtitle)
|
||||
service.download(subtitle)
|
||||
result = subtitle
|
||||
break
|
||||
except DownloadFailedError:
|
||||
logger.warning(u'Could not download subtitle %r, trying next' % subtitle)
|
||||
continue
|
||||
|
||||
if result is None:
|
||||
logger.error(u'No subtitles could be downloaded for video %r' % task.video)
|
||||
return result
|
||||
@@ -193,6 +186,26 @@ def matching_confidence(video, subtitle):
|
||||
return confidence
|
||||
|
||||
|
||||
def get_service(services, service_name, config=None):
|
||||
"""Get a service from its name in the service dict with the specified config.
|
||||
If the service does not exist in the service dict, it is created and added to the dict.
|
||||
|
||||
:param dict services: dict where to get existing services or put created ones
|
||||
:param string service_name: name of the service to get
|
||||
:param config: config to use for the service
|
||||
:type config: :class:`~subliminal.services.ServiceConfig` or None
|
||||
:return: the corresponding service
|
||||
:rtype: :class:`~subliminal.services.ServiceBase`
|
||||
|
||||
"""
|
||||
if service_name not in services:
|
||||
mod = __import__('services.' + service_name, globals=globals(), locals=locals(), fromlist=['Service'], level=-1)
|
||||
services[service_name] = mod.Service()
|
||||
services[service_name].init()
|
||||
services[service_name].config = config
|
||||
return services[service_name]
|
||||
|
||||
|
||||
def key_subtitles(subtitle, video, languages, services, order):
|
||||
"""Create a key to sort subtitle using the given order
|
||||
|
||||
@@ -238,3 +251,21 @@ def group_by_video(list_results):
|
||||
for video, subtitles in list_results:
|
||||
result[video] += subtitles
|
||||
return result
|
||||
|
||||
|
||||
def filter_services(services):
|
||||
"""Filter out services that are not available because of a missing feature
|
||||
|
||||
:param list services: service names to filter
|
||||
:return: a copy of the initial list of service names without unavailable ones
|
||||
:rtype: list
|
||||
|
||||
"""
|
||||
filtered_services = services[:]
|
||||
for service_name in services:
|
||||
mod = __import__('services.' + service_name, globals=globals(), locals=locals(), fromlist=['Service'], level=-1)
|
||||
service = mod.Service
|
||||
if service.required_features is not None and bs4.builder_registry.lookup(*service.required_features) is None:
|
||||
logger.warning(u'Service %s not available: none of available features could be used. One of %r required' % (service_name, service.required_features))
|
||||
filtered_services.remove(service_name)
|
||||
return filtered_services
|
||||
|
||||
@@ -15,4 +15,4 @@
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
__version__ = '0.5.1'
|
||||
__version__ = '0.6.0'
|
||||
|
||||
@@ -1,547 +0,0 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright 2011-2012 Antoine Bertin <diaoulael@gmail.com>
|
||||
#
|
||||
# This file is part of subliminal.
|
||||
#
|
||||
# subliminal is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU Lesser General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# subliminal is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU Lesser General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
__all__ = ['convert_language', 'list_languages', 'LANGUAGES']
|
||||
|
||||
|
||||
def convert_language(language, to_iso, from_iso=None):
|
||||
"""Convert a language into another format
|
||||
|
||||
:param string language: language
|
||||
:param int to_iso: convert language to ISO-639-x
|
||||
:param int from_iso: convert language from ISO-639-x
|
||||
:return: converted language
|
||||
:rtype: string
|
||||
|
||||
"""
|
||||
if from_iso == None: # if no from_iso is given, try to guess it
|
||||
if language.startswith(language[:1].upper()):
|
||||
from_iso = 0
|
||||
elif len(language) == 2:
|
||||
from_iso = 1
|
||||
elif len(language) == 3:
|
||||
from_iso = 2
|
||||
else:
|
||||
raise ValueError('Invalid input language format')
|
||||
if isinstance(language, unicode):
|
||||
language = language.encode('utf-8')
|
||||
converted_language = None
|
||||
for language_tuple in LANGUAGES:
|
||||
if language_tuple[from_iso] == language and language_tuple[to_iso]:
|
||||
converted_language = language_tuple[to_iso]
|
||||
break
|
||||
return converted_language
|
||||
|
||||
|
||||
def list_languages(iso):
|
||||
"""List languages in the given ISO-639-x format
|
||||
|
||||
:param int iso: ISO-639-x format to list
|
||||
:return: languages in the requested format
|
||||
:rtype: list
|
||||
|
||||
"""
|
||||
return [l[iso] for l in LANGUAGES if l[iso]]
|
||||
|
||||
#: ISO-639-2 languages list from http://www.loc.gov/standards/iso639-2/ISO-639-2_utf-8.txt
|
||||
#: + ('Brazilian', 'po', 'pob')
|
||||
LANGUAGES = [('Afar', 'aa', 'aar'),
|
||||
('Abkhazian', 'ab', 'abk'),
|
||||
('Achinese', '', 'ace'),
|
||||
('Acoli', '', 'ach'),
|
||||
('Adangme', '', 'ada'),
|
||||
('Adyghe; Adygei', '', 'ady'),
|
||||
('Afro-Asiatic languages', '', 'afa'),
|
||||
('Afrihili', '', 'afh'),
|
||||
('Afrikaans', 'af', 'afr'),
|
||||
('Ainu', '', 'ain'),
|
||||
('Akan', 'ak', 'aka'),
|
||||
('Akkadian', '', 'akk'),
|
||||
('Albanian', 'sq', 'alb'),
|
||||
('Aleut', '', 'ale'),
|
||||
('Algonquian languages', '', 'alg'),
|
||||
('Southern Altai', '', 'alt'),
|
||||
('Amharic', 'am', 'amh'),
|
||||
('English, Old (ca.450-1100)', '', 'ang'),
|
||||
('Angika', '', 'anp'),
|
||||
('Apache languages', '', 'apa'),
|
||||
('Arabic', 'ar', 'ara'),
|
||||
('Official Aramaic (700-300 BCE); Imperial Aramaic (700-300 BCE)', '', 'arc'),
|
||||
('Aragonese', 'an', 'arg'),
|
||||
('Armenian', 'hy', 'arm'),
|
||||
('Mapudungun; Mapuche', '', 'arn'),
|
||||
('Arapaho', '', 'arp'),
|
||||
('Artificial languages', '', 'art'),
|
||||
('Arawak', '', 'arw'),
|
||||
('Assamese', 'as', 'asm'),
|
||||
('Asturian; Bable; Leonese; Asturleonese', '', 'ast'),
|
||||
('Athapascan languages', '', 'ath'),
|
||||
('Australian languages', '', 'aus'),
|
||||
('Avaric', 'av', 'ava'),
|
||||
('Avestan', 'ae', 'ave'),
|
||||
('Awadhi', '', 'awa'),
|
||||
('Aymara', 'ay', 'aym'),
|
||||
('Azerbaijani', 'az', 'aze'),
|
||||
('Banda languages', '', 'bad'),
|
||||
('Bamileke languages', '', 'bai'),
|
||||
('Bashkir', 'ba', 'bak'),
|
||||
('Baluchi', '', 'bal'),
|
||||
('Bambara', 'bm', 'bam'),
|
||||
('Balinese', '', 'ban'),
|
||||
('Basque', 'eu', 'baq'),
|
||||
('Basa', '', 'bas'),
|
||||
('Baltic languages', '', 'bat'),
|
||||
('Beja; Bedawiyet', '', 'bej'),
|
||||
('Belarusian', 'be', 'bel'),
|
||||
('Bemba', '', 'bem'),
|
||||
('Bengali', 'bn', 'ben'),
|
||||
('Berber languages', '', 'ber'),
|
||||
('Bhojpuri', '', 'bho'),
|
||||
('Bihari languages', 'bh', 'bih'),
|
||||
('Bikol', '', 'bik'),
|
||||
('Bini; Edo', '', 'bin'),
|
||||
('Bislama', 'bi', 'bis'),
|
||||
('Siksika', '', 'bla'),
|
||||
('Bantu (Other)', '', 'bnt'),
|
||||
('Bosnian', 'bs', 'bos'),
|
||||
('Braj', '', 'bra'),
|
||||
('Breton', 'br', 'bre'),
|
||||
('Batak languages', '', 'btk'),
|
||||
('Buriat', '', 'bua'),
|
||||
('Buginese', '', 'bug'),
|
||||
('Bulgarian', 'bg', 'bul'),
|
||||
('Burmese', 'my', 'bur'),
|
||||
('Blin; Bilin', '', 'byn'),
|
||||
('Caddo', '', 'cad'),
|
||||
('Central American Indian languages', '', 'cai'),
|
||||
('Galibi Carib', '', 'car'),
|
||||
('Catalan; Valencian', 'ca', 'cat'),
|
||||
('Caucasian languages', '', 'cau'),
|
||||
('Cebuano', '', 'ceb'),
|
||||
('Celtic languages', '', 'cel'),
|
||||
('Chamorro', 'ch', 'cha'),
|
||||
('Chibcha', '', 'chb'),
|
||||
('Chechen', 'ce', 'che'),
|
||||
('Chagatai', '', 'chg'),
|
||||
('Chinese', 'zh', 'chi'),
|
||||
('Chuukese', '', 'chk'),
|
||||
('Mari', '', 'chm'),
|
||||
('Chinook jargon', '', 'chn'),
|
||||
('Choctaw', '', 'cho'),
|
||||
('Chipewyan; Dene Suline', '', 'chp'),
|
||||
('Cherokee', '', 'chr'),
|
||||
('Church Slavic; Old Slavonic; Church Slavonic; Old Bulgarian; Old Church Slavonic', 'cu', 'chu'),
|
||||
('Chuvash', 'cv', 'chv'),
|
||||
('Cheyenne', '', 'chy'),
|
||||
('Chamic languages', '', 'cmc'),
|
||||
('Coptic', '', 'cop'),
|
||||
('Cornish', 'kw', 'cor'),
|
||||
('Corsican', 'co', 'cos'),
|
||||
('Creoles and pidgins, English based', '', 'cpe'),
|
||||
('Creoles and pidgins, French-based ', '', 'cpf'),
|
||||
('Creoles and pidgins, Portuguese-based ', '', 'cpp'),
|
||||
('Cree', 'cr', 'cre'),
|
||||
('Crimean Tatar; Crimean Turkish', '', 'crh'),
|
||||
('Creoles and pidgins ', '', 'crp'),
|
||||
('Kashubian', '', 'csb'),
|
||||
('Cushitic languages', '', 'cus'),
|
||||
('Czech', 'cs', 'cze'),
|
||||
('Dakota', '', 'dak'),
|
||||
('Danish', 'da', 'dan'),
|
||||
('Dargwa', '', 'dar'),
|
||||
('Land Dayak languages', '', 'day'),
|
||||
('Delaware', '', 'del'),
|
||||
('Slave (Athapascan)', '', 'den'),
|
||||
('Dogrib', '', 'dgr'),
|
||||
('Dinka', '', 'din'),
|
||||
('Divehi; Dhivehi; Maldivian', 'dv', 'div'),
|
||||
('Dogri', '', 'doi'),
|
||||
('Dravidian languages', '', 'dra'),
|
||||
('Lower Sorbian', '', 'dsb'),
|
||||
('Duala', '', 'dua'),
|
||||
('Dutch, Middle (ca.1050-1350)', '', 'dum'),
|
||||
('Dutch; Flemish', 'nl', 'dut'),
|
||||
('Dyula', '', 'dyu'),
|
||||
('Dzongkha', 'dz', 'dzo'),
|
||||
('Efik', '', 'efi'),
|
||||
('Egyptian (Ancient)', '', 'egy'),
|
||||
('Ekajuk', '', 'eka'),
|
||||
('Elamite', '', 'elx'),
|
||||
('English', 'en', 'eng'),
|
||||
('English, Middle (1100-1500)', '', 'enm'),
|
||||
('Esperanto', 'eo', 'epo'),
|
||||
('Estonian', 'et', 'est'),
|
||||
('Ewe', 'ee', 'ewe'),
|
||||
('Ewondo', '', 'ewo'),
|
||||
('Fang', '', 'fan'),
|
||||
('Faroese', 'fo', 'fao'),
|
||||
('Fanti', '', 'fat'),
|
||||
('Fijian', 'fj', 'fij'),
|
||||
('Filipino; Pilipino', '', 'fil'),
|
||||
('Finnish', 'fi', 'fin'),
|
||||
('Finno-Ugrian languages', '', 'fiu'),
|
||||
('Fon', '', 'fon'),
|
||||
('French', 'fr', 'fre'),
|
||||
('French, Middle (ca.1400-1600)', '', 'frm'),
|
||||
('French, Old (842-ca.1400)', '', 'fro'),
|
||||
('Northern Frisian', '', 'frr'),
|
||||
('Eastern Frisian', '', 'frs'),
|
||||
('Western Frisian', 'fy', 'fry'),
|
||||
('Fulah', 'ff', 'ful'),
|
||||
('Friulian', '', 'fur'),
|
||||
('Ga', '', 'gaa'),
|
||||
('Gayo', '', 'gay'),
|
||||
('Gbaya', '', 'gba'),
|
||||
('Germanic languages', '', 'gem'),
|
||||
('Georgian', 'ka', 'geo'),
|
||||
('German', 'de', 'ger'),
|
||||
('Geez', '', 'gez'),
|
||||
('Gilbertese', '', 'gil'),
|
||||
('Gaelic; Scottish Gaelic', 'gd', 'gla'),
|
||||
('Irish', 'ga', 'gle'),
|
||||
('Galician', 'gl', 'glg'),
|
||||
('Manx', 'gv', 'glv'),
|
||||
('German, Middle High (ca.1050-1500)', '', 'gmh'),
|
||||
('German, Old High (ca.750-1050)', '', 'goh'),
|
||||
('Gondi', '', 'gon'),
|
||||
('Gorontalo', '', 'gor'),
|
||||
('Gothic', '', 'got'),
|
||||
('Grebo', '', 'grb'),
|
||||
('Greek, Ancient (to 1453)', '', 'grc'),
|
||||
('Greek, Modern (1453-)', 'el', 'gre'),
|
||||
('Guarani', 'gn', 'grn'),
|
||||
('Swiss German; Alemannic; Alsatian', '', 'gsw'),
|
||||
('Gujarati', 'gu', 'guj'),
|
||||
('Gwich\'in', '', 'gwi'),
|
||||
('Haida', '', 'hai'),
|
||||
('Haitian; Haitian Creole', 'ht', 'hat'),
|
||||
('Hausa', 'ha', 'hau'),
|
||||
('Hawaiian', '', 'haw'),
|
||||
('Hebrew', 'he', 'heb'),
|
||||
('Herero', 'hz', 'her'),
|
||||
('Hiligaynon', '', 'hil'),
|
||||
('Himachali languages; Western Pahari languages', '', 'him'),
|
||||
('Hindi', 'hi', 'hin'),
|
||||
('Hittite', '', 'hit'),
|
||||
('Hmong; Mong', '', 'hmn'),
|
||||
('Hiri Motu', 'ho', 'hmo'),
|
||||
('Croatian', 'hr', 'hrv'),
|
||||
('Upper Sorbian', '', 'hsb'),
|
||||
('Hungarian', 'hu', 'hun'),
|
||||
('Hupa', '', 'hup'),
|
||||
('Iban', '', 'iba'),
|
||||
('Igbo', 'ig', 'ibo'),
|
||||
('Icelandic', 'is', 'ice'),
|
||||
('Ido', 'io', 'ido'),
|
||||
('Sichuan Yi; Nuosu', 'ii', 'iii'),
|
||||
('Ijo languages', '', 'ijo'),
|
||||
('Inuktitut', 'iu', 'iku'),
|
||||
('Interlingue; Occidental', 'ie', 'ile'),
|
||||
('Iloko', '', 'ilo'),
|
||||
('Interlingua (International Auxiliary Language Association)', 'ia', 'ina'),
|
||||
('Indic languages', '', 'inc'),
|
||||
('Indonesian', 'id', 'ind'),
|
||||
('Indo-European languages', '', 'ine'),
|
||||
('Ingush', '', 'inh'),
|
||||
('Inupiaq', 'ik', 'ipk'),
|
||||
('Iranian languages', '', 'ira'),
|
||||
('Iroquoian languages', '', 'iro'),
|
||||
('Italian', 'it', 'ita'),
|
||||
('Javanese', 'jv', 'jav'),
|
||||
('Lojban', '', 'jbo'),
|
||||
('Japanese', 'ja', 'jpn'),
|
||||
('Judeo-Persian', '', 'jpr'),
|
||||
('Judeo-Arabic', '', 'jrb'),
|
||||
('Kara-Kalpak', '', 'kaa'),
|
||||
('Kabyle', '', 'kab'),
|
||||
('Kachin; Jingpho', '', 'kac'),
|
||||
('Kalaallisut; Greenlandic', 'kl', 'kal'),
|
||||
('Kamba', '', 'kam'),
|
||||
('Kannada', 'kn', 'kan'),
|
||||
('Karen languages', '', 'kar'),
|
||||
('Kashmiri', 'ks', 'kas'),
|
||||
('Kanuri', 'kr', 'kau'),
|
||||
('Kawi', '', 'kaw'),
|
||||
('Kazakh', 'kk', 'kaz'),
|
||||
('Kabardian', '', 'kbd'),
|
||||
('Khasi', '', 'kha'),
|
||||
('Khoisan languages', '', 'khi'),
|
||||
('Central Khmer', 'km', 'khm'),
|
||||
('Khotanese; Sakan', '', 'kho'),
|
||||
('Kikuyu; Gikuyu', 'ki', 'kik'),
|
||||
('Kinyarwanda', 'rw', 'kin'),
|
||||
('Kirghiz; Kyrgyz', 'ky', 'kir'),
|
||||
('Kimbundu', '', 'kmb'),
|
||||
('Konkani', '', 'kok'),
|
||||
('Komi', 'kv', 'kom'),
|
||||
('Kongo', 'kg', 'kon'),
|
||||
('Korean', 'ko', 'kor'),
|
||||
('Kosraean', '', 'kos'),
|
||||
('Kpelle', '', 'kpe'),
|
||||
('Karachay-Balkar', '', 'krc'),
|
||||
('Karelian', '', 'krl'),
|
||||
('Kru languages', '', 'kro'),
|
||||
('Kurukh', '', 'kru'),
|
||||
('Kuanyama; Kwanyama', 'kj', 'kua'),
|
||||
('Kumyk', '', 'kum'),
|
||||
('Kurdish', 'ku', 'kur'),
|
||||
('Kutenai', '', 'kut'),
|
||||
('Ladino', '', 'lad'),
|
||||
('Lahnda', '', 'lah'),
|
||||
('Lamba', '', 'lam'),
|
||||
('Lao', 'lo', 'lao'),
|
||||
('Latin', 'la', 'lat'),
|
||||
('Latvian', 'lv', 'lav'),
|
||||
('Lezghian', '', 'lez'),
|
||||
('Limburgan; Limburger; Limburgish', 'li', 'lim'),
|
||||
('Lingala', 'ln', 'lin'),
|
||||
('Lithuanian', 'lt', 'lit'),
|
||||
('Mongo', '', 'lol'),
|
||||
('Lozi', '', 'loz'),
|
||||
('Luxembourgish; Letzeburgesch', 'lb', 'ltz'),
|
||||
('Luba-Lulua', '', 'lua'),
|
||||
('Luba-Katanga', 'lu', 'lub'),
|
||||
('Ganda', 'lg', 'lug'),
|
||||
('Luiseno', '', 'lui'),
|
||||
('Lunda', '', 'lun'),
|
||||
('Luo (Kenya and Tanzania)', '', 'luo'),
|
||||
('Lushai', '', 'lus'),
|
||||
('Macedonian', 'mk', 'mac'),
|
||||
('Madurese', '', 'mad'),
|
||||
('Magahi', '', 'mag'),
|
||||
('Marshallese', 'mh', 'mah'),
|
||||
('Maithili', '', 'mai'),
|
||||
('Makasar', '', 'mak'),
|
||||
('Malayalam', 'ml', 'mal'),
|
||||
('Mandingo', '', 'man'),
|
||||
('Maori', 'mi', 'mao'),
|
||||
('Austronesian languages', '', 'map'),
|
||||
('Marathi', 'mr', 'mar'),
|
||||
('Masai', '', 'mas'),
|
||||
('Malay', 'ms', 'may'),
|
||||
('Moksha', '', 'mdf'),
|
||||
('Mandar', '', 'mdr'),
|
||||
('Mende', '', 'men'),
|
||||
('Irish, Middle (900-1200)', '', 'mga'),
|
||||
('Mi\'kmaq; Micmac', '', 'mic'),
|
||||
('Minangkabau', '', 'min'),
|
||||
('Uncoded languages', '', 'mis'),
|
||||
('Mon-Khmer languages', '', 'mkh'),
|
||||
('Malagasy', 'mg', 'mlg'),
|
||||
('Maltese', 'mt', 'mlt'),
|
||||
('Manchu', '', 'mnc'),
|
||||
('Manipuri', '', 'mni'),
|
||||
('Manobo languages', '', 'mno'),
|
||||
('Mohawk', '', 'moh'),
|
||||
('Mongolian', 'mn', 'mon'),
|
||||
('Mossi', '', 'mos'),
|
||||
('Multiple languages', '', 'mul'),
|
||||
('Munda languages', '', 'mun'),
|
||||
('Creek', '', 'mus'),
|
||||
('Mirandese', '', 'mwl'),
|
||||
('Marwari', '', 'mwr'),
|
||||
('Mayan languages', '', 'myn'),
|
||||
('Erzya', '', 'myv'),
|
||||
('Nahuatl languages', '', 'nah'),
|
||||
('North American Indian languages', '', 'nai'),
|
||||
('Neapolitan', '', 'nap'),
|
||||
('Nauru', 'na', 'nau'),
|
||||
('Navajo; Navaho', 'nv', 'nav'),
|
||||
('Ndebele, South; South Ndebele', 'nr', 'nbl'),
|
||||
('Ndebele, North; North Ndebele', 'nd', 'nde'),
|
||||
('Ndonga', 'ng', 'ndo'),
|
||||
('Low German; Low Saxon; German, Low; Saxon, Low', '', 'nds'),
|
||||
('Nepali', 'ne', 'nep'),
|
||||
('Nepal Bhasa; Newari', '', 'new'),
|
||||
('Nias', '', 'nia'),
|
||||
('Niger-Kordofanian languages', '', 'nic'),
|
||||
('Niuean', '', 'niu'),
|
||||
('Norwegian Nynorsk; Nynorsk, Norwegian', 'nn', 'nno'),
|
||||
('Bokmål, Norwegian; Norwegian Bokmål', 'nb', 'nob'),
|
||||
('Nogai', '', 'nog'),
|
||||
('Norse, Old', '', 'non'),
|
||||
('Norwegian', 'no', 'nor'),
|
||||
('N\'Ko', '', 'nqo'),
|
||||
('Pedi; Sepedi; Northern Sotho', '', 'nso'),
|
||||
('Nubian languages', '', 'nub'),
|
||||
('Classical Newari; Old Newari; Classical Nepal Bhasa', '', 'nwc'),
|
||||
('Chichewa; Chewa; Nyanja', 'ny', 'nya'),
|
||||
('Nyamwezi', '', 'nym'),
|
||||
('Nyankole', '', 'nyn'),
|
||||
('Nyoro', '', 'nyo'),
|
||||
('Nzima', '', 'nzi'),
|
||||
('Occitan (post 1500); Provençal', 'oc', 'oci'),
|
||||
('Ojibwa', 'oj', 'oji'),
|
||||
('Oriya', 'or', 'ori'),
|
||||
('Oromo', 'om', 'orm'),
|
||||
('Osage', '', 'osa'),
|
||||
('Ossetian; Ossetic', 'os', 'oss'),
|
||||
('Turkish, Ottoman (1500-1928)', '', 'ota'),
|
||||
('Otomian languages', '', 'oto'),
|
||||
('Papuan languages', '', 'paa'),
|
||||
('Pangasinan', '', 'pag'),
|
||||
('Pahlavi', '', 'pal'),
|
||||
('Pampanga; Kapampangan', '', 'pam'),
|
||||
('Panjabi; Punjabi', 'pa', 'pan'),
|
||||
('Papiamento', '', 'pap'),
|
||||
('Palauan', '', 'pau'),
|
||||
('Persian, Old (ca.600-400 B.C.)', '', 'peo'),
|
||||
('Persian', 'fa', 'per'),
|
||||
('Philippine languages', '', 'phi'),
|
||||
('Phoenician', '', 'phn'),
|
||||
('Pali', 'pi', 'pli'),
|
||||
('Polish', 'pl', 'pol'),
|
||||
('Pohnpeian', '', 'pon'),
|
||||
('Portuguese', 'pt', 'por'),
|
||||
('Prakrit languages', '', 'pra'),
|
||||
('Provençal, Old (to 1500)', '', 'pro'),
|
||||
('Pushto; Pashto', 'ps', 'pus'),
|
||||
('Reserved for local use', '', 'qaa-qtz'),
|
||||
('Quechua', 'qu', 'que'),
|
||||
('Rajasthani', '', 'raj'),
|
||||
('Rapanui', '', 'rap'),
|
||||
('Rarotongan; Cook Islands Maori', '', 'rar'),
|
||||
('Romance languages', '', 'roa'),
|
||||
('Romansh', 'rm', 'roh'),
|
||||
('Romany', '', 'rom'),
|
||||
('Romanian; Moldavian; Moldovan', 'ro', 'rum'),
|
||||
('Rundi', 'rn', 'run'),
|
||||
('Aromanian; Arumanian; Macedo-Romanian', '', 'rup'),
|
||||
('Russian', 'ru', 'rus'),
|
||||
('Sandawe', '', 'sad'),
|
||||
('Sango', 'sg', 'sag'),
|
||||
('Yakut', '', 'sah'),
|
||||
('South American Indian (Other)', '', 'sai'),
|
||||
('Salishan languages', '', 'sal'),
|
||||
('Samaritan Aramaic', '', 'sam'),
|
||||
('Sanskrit', 'sa', 'san'),
|
||||
('Sasak', '', 'sas'),
|
||||
('Santali', '', 'sat'),
|
||||
('Sicilian', '', 'scn'),
|
||||
('Scots', '', 'sco'),
|
||||
('Selkup', '', 'sel'),
|
||||
('Semitic languages', '', 'sem'),
|
||||
('Irish, Old (to 900)', '', 'sga'),
|
||||
('Sign Languages', '', 'sgn'),
|
||||
('Shan', '', 'shn'),
|
||||
('Sidamo', '', 'sid'),
|
||||
('Sinhala; Sinhalese', 'si', 'sin'),
|
||||
('Siouan languages', '', 'sio'),
|
||||
('Sino-Tibetan languages', '', 'sit'),
|
||||
('Slavic languages', '', 'sla'),
|
||||
('Slovak', 'sk', 'slo'),
|
||||
('Slovenian', 'sl', 'slv'),
|
||||
('Southern Sami', '', 'sma'),
|
||||
('Northern Sami', 'se', 'sme'),
|
||||
('Sami languages', '', 'smi'),
|
||||
('Lule Sami', '', 'smj'),
|
||||
('Inari Sami', '', 'smn'),
|
||||
('Samoan', 'sm', 'smo'),
|
||||
('Skolt Sami', '', 'sms'),
|
||||
('Shona', 'sn', 'sna'),
|
||||
('Sindhi', 'sd', 'snd'),
|
||||
('Soninke', '', 'snk'),
|
||||
('Sogdian', '', 'sog'),
|
||||
('Somali', 'so', 'som'),
|
||||
('Songhai languages', '', 'son'),
|
||||
('Sotho, Southern', 'st', 'sot'),
|
||||
('Spanish; Castilian', 'es', 'spa'),
|
||||
('Sardinian', 'sc', 'srd'),
|
||||
('Sranan Tongo', '', 'srn'),
|
||||
('Serbian', 'sr', 'srp'),
|
||||
('Serer', '', 'srr'),
|
||||
('Nilo-Saharan languages', '', 'ssa'),
|
||||
('Swati', 'ss', 'ssw'),
|
||||
('Sukuma', '', 'suk'),
|
||||
('Sundanese', 'su', 'sun'),
|
||||
('Susu', '', 'sus'),
|
||||
('Sumerian', '', 'sux'),
|
||||
('Swahili', 'sw', 'swa'),
|
||||
('Swedish', 'sv', 'swe'),
|
||||
('Classical Syriac', '', 'syc'),
|
||||
('Syriac', '', 'syr'),
|
||||
('Tahitian', 'ty', 'tah'),
|
||||
('Tai languages', '', 'tai'),
|
||||
('Tamil', 'ta', 'tam'),
|
||||
('Tatar', 'tt', 'tat'),
|
||||
('Telugu', 'te', 'tel'),
|
||||
('Timne', '', 'tem'),
|
||||
('Tereno', '', 'ter'),
|
||||
('Tetum', '', 'tet'),
|
||||
('Tajik', 'tg', 'tgk'),
|
||||
('Tagalog', 'tl', 'tgl'),
|
||||
('Thai', 'th', 'tha'),
|
||||
('Tibetan', 'bo', 'tib'),
|
||||
('Tigre', '', 'tig'),
|
||||
('Tigrinya', 'ti', 'tir'),
|
||||
('Tiv', '', 'tiv'),
|
||||
('Tokelau', '', 'tkl'),
|
||||
('Klingon; tlhIngan-Hol', '', 'tlh'),
|
||||
('Tlingit', '', 'tli'),
|
||||
('Tamashek', '', 'tmh'),
|
||||
('Tonga (Nyasa)', '', 'tog'),
|
||||
('Tonga (Tonga Islands)', 'to', 'ton'),
|
||||
('Tok Pisin', '', 'tpi'),
|
||||
('Tsimshian', '', 'tsi'),
|
||||
('Tswana', 'tn', 'tsn'),
|
||||
('Tsonga', 'ts', 'tso'),
|
||||
('Turkmen', 'tk', 'tuk'),
|
||||
('Tumbuka', '', 'tum'),
|
||||
('Tupi languages', '', 'tup'),
|
||||
('Turkish', 'tr', 'tur'),
|
||||
('Altaic languages', '', 'tut'),
|
||||
('Tuvalu', '', 'tvl'),
|
||||
('Twi', 'tw', 'twi'),
|
||||
('Tuvinian', '', 'tyv'),
|
||||
('Udmurt', '', 'udm'),
|
||||
('Ugaritic', '', 'uga'),
|
||||
('Uighur; Uyghur', 'ug', 'uig'),
|
||||
('Ukrainian', 'uk', 'ukr'),
|
||||
('Umbundu', '', 'umb'),
|
||||
('Undetermined', '', 'und'),
|
||||
('Urdu', 'ur', 'urd'),
|
||||
('Uzbek', 'uz', 'uzb'),
|
||||
('Vai', '', 'vai'),
|
||||
('Venda', 've', 'ven'),
|
||||
('Vietnamese', 'vi', 'vie'),
|
||||
('Volapük', 'vo', 'vol'),
|
||||
('Votic', '', 'vot'),
|
||||
('Wakashan languages', '', 'wak'),
|
||||
('Walamo', '', 'wal'),
|
||||
('Waray', '', 'war'),
|
||||
('Washo', '', 'was'),
|
||||
('Welsh', 'cy', 'wel'),
|
||||
('Sorbian languages', '', 'wen'),
|
||||
('Walloon', 'wa', 'wln'),
|
||||
('Wolof', 'wo', 'wol'),
|
||||
('Kalmyk; Oirat', '', 'xal'),
|
||||
('Xhosa', 'xh', 'xho'),
|
||||
('Yao', '', 'yao'),
|
||||
('Yapese', '', 'yap'),
|
||||
('Yiddish', 'yi', 'yid'),
|
||||
('Yoruba', 'yo', 'yor'),
|
||||
('Yupik languages', '', 'ypk'),
|
||||
('Zapotec', '', 'zap'),
|
||||
('Blissymbols; Blissymbolics; Bliss', '', 'zbl'),
|
||||
('Zenaga', '', 'zen'),
|
||||
('Zhuang; Chuang', 'za', 'zha'),
|
||||
('Zande languages', '', 'znd'),
|
||||
('Zulu', 'zu', 'zul'),
|
||||
('Zuni', '', 'zun'),
|
||||
('No linguistic content; Not applicable', '', 'zxx'),
|
||||
('Zaza; Dimili; Dimli; Kirdki; Kirmanjki; Zazaki', '', 'zza'),
|
||||
('Brazilian', 'po', 'pob')]
|
||||
@@ -15,11 +15,15 @@
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
from ..exceptions import MissingLanguageError, DownloadFailedError
|
||||
from .. import cache
|
||||
from ..exceptions import MissingLanguageError, DownloadFailedError, ServiceError
|
||||
from ..subtitles import EXTENSIONS
|
||||
from guessit.language import lang_set, UNDETERMINED
|
||||
import logging
|
||||
import os
|
||||
import requests
|
||||
import threading
|
||||
import zipfile
|
||||
|
||||
|
||||
__all__ = ['ServiceBase', 'ServiceConfig']
|
||||
@@ -37,7 +41,7 @@ class ServiceBase(object):
|
||||
server_url = ''
|
||||
|
||||
#: User Agent for any HTTP-based requests
|
||||
user_agent = 'subliminal v0.5'
|
||||
user_agent = 'subliminal v0.6'
|
||||
|
||||
#: Whether based on an API or not
|
||||
api_based = False
|
||||
@@ -45,21 +49,18 @@ class ServiceBase(object):
|
||||
#: Timeout for web requests
|
||||
timeout = 5
|
||||
|
||||
#: Lock for cache interactions
|
||||
lock = threading.Lock()
|
||||
|
||||
#: Mapping to Service's language codes and subliminal's
|
||||
languages = {}
|
||||
|
||||
#: Whether the mapping is reverted or not
|
||||
reverted_languages = False
|
||||
|
||||
#: Accepted video classes (:class:`~subliminal.videos.Episode`, :class:`~subliminal.videos.Movie`, :class:`~subliminal.videos.UnknownVideo`)
|
||||
videos = []
|
||||
|
||||
#: Whether the video has to exist or not
|
||||
require_video = False
|
||||
|
||||
#: List of required features for BeautifulSoup
|
||||
required_features = None
|
||||
|
||||
def __init__(self, config=None):
|
||||
self.config = config or ServiceConfig()
|
||||
|
||||
@@ -75,6 +76,30 @@ class ServiceBase(object):
|
||||
logger.debug(u'Initializing %s' % self.__class__.__name__)
|
||||
self.session = requests.session(timeout=10, headers={'User-Agent': self.user_agent})
|
||||
|
||||
def init_cache(self):
|
||||
"""Initialize cache, make sure it is loaded from disk"""
|
||||
if not self.config or not self.config.cache:
|
||||
raise ServiceError('Cache directory is required')
|
||||
|
||||
service_name = self.__class__.__name__
|
||||
self.config.cache.load(service_name)
|
||||
|
||||
def save_cache(self):
|
||||
service_name = self.__class__.__name__
|
||||
self.config.cache.save(service_name)
|
||||
|
||||
def clear_cache(self):
|
||||
service_name = self.__class__.__name__
|
||||
self.config.cache.clear(service_name)
|
||||
|
||||
def cache_for(self, func, args, result):
|
||||
service_name = self.__class__.__name__
|
||||
return self.config.cache.cache_for(service_name, func, args, result)
|
||||
|
||||
def cached_value(self, func, args):
|
||||
service_name = self.__class__.__name__
|
||||
return self.config.cache.cached_value(service_name, func, args)
|
||||
|
||||
def terminate(self):
|
||||
"""Terminate connection"""
|
||||
logger.debug(u'Terminating %s' % self.__class__.__name__)
|
||||
@@ -84,25 +109,20 @@ class ServiceBase(object):
|
||||
pass
|
||||
|
||||
def list(self, video, languages):
|
||||
"""List subtitles"""
|
||||
pass
|
||||
"""List subtitles
|
||||
|
||||
As a service writer, you can either override this method or implement
|
||||
:meth:`list_checked` instead to have the languages pre-filtered for you
|
||||
|
||||
"""
|
||||
if not self.check_validity(video, languages):
|
||||
return []
|
||||
return self.list_checked(video, languages)
|
||||
|
||||
def download(self, subtitle):
|
||||
"""Download a subtitle"""
|
||||
self.download_file(subtitle.link, subtitle.path)
|
||||
|
||||
@classmethod
|
||||
def available_languages(cls):
|
||||
"""Available languages in the Service
|
||||
|
||||
:return: available languages
|
||||
:rtype: set
|
||||
|
||||
"""
|
||||
if not cls.reverted_languages:
|
||||
return set(cls.languages.keys())
|
||||
return set(cls.languages.values())
|
||||
|
||||
@classmethod
|
||||
def check_validity(cls, video, languages):
|
||||
"""Check for video and languages validity in the Service
|
||||
@@ -113,72 +133,15 @@ class ServiceBase(object):
|
||||
:rtype: bool
|
||||
|
||||
"""
|
||||
languages &= cls.available_languages()
|
||||
languages = (lang_set(languages) & cls.languages) - set([UNDETERMINED])
|
||||
if not languages:
|
||||
logger.debug(u'No language available for service %s' % cls.__class__.__name__.lower())
|
||||
logger.debug(u'No language available for service %s' % cls.__name__.lower())
|
||||
return False
|
||||
if not cls.is_valid_video(video):
|
||||
logger.debug(u'%r is not valid for service %s' % (video, cls.__class__.__name__.lower()))
|
||||
if cls.require_video and not video.exists or not isinstance(video, tuple(cls.videos)):
|
||||
logger.debug(u'%r is not valid for service %s' % (video, cls.__name__.lower()))
|
||||
return False
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def is_valid_video(cls, video):
|
||||
"""Check if video is valid in the Service
|
||||
|
||||
:param video: the video to check
|
||||
:type video: :class:`~subliminal.videos.Video`
|
||||
:rtype: bool
|
||||
|
||||
"""
|
||||
if cls.require_video and not video.exists:
|
||||
return False
|
||||
if not isinstance(video, tuple(cls.videos)):
|
||||
return False
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def is_valid_language(cls, language):
|
||||
"""Check if language is valid in the Service
|
||||
|
||||
:param string language: the language to check
|
||||
:rtype: bool
|
||||
|
||||
"""
|
||||
if language in cls.available_languages():
|
||||
return True
|
||||
return False
|
||||
|
||||
@classmethod
|
||||
def get_revert_language(cls, language):
|
||||
"""ISO-639-1 language code from service language code
|
||||
|
||||
:param string language: service language code
|
||||
:return: ISO-639-1 language code
|
||||
:rtype: string
|
||||
|
||||
"""
|
||||
if not cls.reverted_languages and language in cls.languages.values():
|
||||
return [k for k, v in cls.languages.iteritems() if v == language][0]
|
||||
if cls.reverted_languages and language in cls.languages.keys():
|
||||
return cls.languages[language]
|
||||
raise MissingLanguageError(language)
|
||||
|
||||
@classmethod
|
||||
def get_language(cls, language):
|
||||
"""Service language code from ISO-639-1 language code
|
||||
|
||||
:param string language: ISO-639-1 language code
|
||||
:return: service language code
|
||||
:rtype: string
|
||||
|
||||
"""
|
||||
if not cls.reverted_languages and language in cls.languages.keys():
|
||||
return cls.languages[language]
|
||||
if cls.reverted_languages and language in cls.languages.values():
|
||||
return [k for k, v in cls.languages.iteritems() if v == language][0]
|
||||
raise MissingLanguageError(language)
|
||||
|
||||
def download_file(self, url, filepath):
|
||||
"""Attempt to download a file and remove it in case of failure
|
||||
|
||||
@@ -198,6 +161,43 @@ class ServiceBase(object):
|
||||
raise DownloadFailedError(str(e))
|
||||
logger.debug(u'Download finished for file %s. Size: %s' % (filepath, os.path.getsize(filepath)))
|
||||
|
||||
def download_zip_file(self, url, filepath):
|
||||
"""Attempt to download a zip file and extract any subtitle file from it, if any.
|
||||
This cleans up after itself if anything fails.
|
||||
|
||||
:param string url: URL of the zip file to download
|
||||
:param string filepath: destination path for the subtitle
|
||||
|
||||
"""
|
||||
logger.info(u'Downloading %s' % url)
|
||||
try:
|
||||
zippath = filepath + '.zip'
|
||||
r = self.session.get(url, headers={'Referer': url, 'User-Agent': self.user_agent})
|
||||
with open(zippath, 'wb') as f:
|
||||
f.write(r.content)
|
||||
if not zipfile.is_zipfile(zippath):
|
||||
# TODO: could check if maybe we already have a text file and
|
||||
# download it directly
|
||||
raise DownloadFailedError('Downloaded file is not a zip file')
|
||||
zipsub = zipfile.ZipFile(zippath)
|
||||
for subfile in zipsub.namelist():
|
||||
if os.path.splitext(subfile)[1] in EXTENSIONS:
|
||||
open(filepath, 'w').write(zipsub.open(subfile).read())
|
||||
break
|
||||
else:
|
||||
logger.debug(u'No subtitles found in zip file')
|
||||
raise DownloadFailedError('No subtitles found in zip file')
|
||||
os.remove(zippath)
|
||||
logger.debug(u'Download finished for file %s. Size: %s' % (filepath, os.path.getsize(filepath)))
|
||||
return
|
||||
except Exception as e:
|
||||
logger.error(u'Download %s failed: %s' % (url, e))
|
||||
if os.path.exists(zippath):
|
||||
os.remove(zippath)
|
||||
if os.path.exists(filepath):
|
||||
os.remove(filepath)
|
||||
raise DownloadFailedError(str(e))
|
||||
|
||||
|
||||
class ServiceConfig(object):
|
||||
"""Configuration for any :class:`Service`
|
||||
@@ -209,6 +209,9 @@ class ServiceConfig(object):
|
||||
def __init__(self, multi=False, cache_dir=None):
|
||||
self.multi = multi
|
||||
self.cache_dir = cache_dir
|
||||
self.cache = None
|
||||
if cache_dir is not None:
|
||||
self.cache = cache.Cache(cache_dir)
|
||||
|
||||
def __repr__(self):
|
||||
return 'ServiceConfig(%r, %s)' % (self.multi, self.cache_dir)
|
||||
return 'ServiceConfig(%r, %s)' % (self.multi, self.cache.cache_dir)
|
||||
|
||||
Executable
+161
@@ -0,0 +1,161 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright 2012 Olivier Leveau <olifozzy@gmail.com>
|
||||
#
|
||||
# This file is part of subliminal.
|
||||
#
|
||||
# subliminal is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU Lesser General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# subliminal is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU Lesser General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
from . import ServiceBase
|
||||
from ..cache import cachedmethod
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..videos import Episode
|
||||
from bs4 import BeautifulSoup
|
||||
from guessit.language import lang_set
|
||||
from subliminal.utils import get_keywords
|
||||
import guessit
|
||||
import logging
|
||||
import re
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def match(pattern, string):
|
||||
try:
|
||||
return re.search(pattern, string).group(1)
|
||||
except AttributeError:
|
||||
logger.debug(u'Could not match %r on %r' % (pattern, string))
|
||||
return None
|
||||
|
||||
|
||||
def matches(pattern, string):
|
||||
try:
|
||||
return re.search(pattern, string).group(1, 2)
|
||||
except AttributeError:
|
||||
logger.debug(u'Could not match %r on %r' % (pattern, string))
|
||||
return None
|
||||
|
||||
|
||||
class Addic7ed(ServiceBase):
|
||||
server_url = 'http://www.addic7ed.com'
|
||||
api_based = False
|
||||
languages = lang_set([u'English', u'Italian', u'Portuguese',
|
||||
u'Portuguese (Brazilian)', u'Romanian',
|
||||
u'Spanish', u'French', u'Greek', u'Arabic',
|
||||
u'German', u'Croatian', u'Indonesian', u'Hebrew',
|
||||
u'Russian', u'Turkish', u'Swedish', u'Czech',
|
||||
u'Dutch', u'Hungarian', u'Norwegian', u'Polish',
|
||||
u'Persian'], strict=True)
|
||||
videos = [Episode]
|
||||
require_video = False
|
||||
required_features = ['permissive']
|
||||
|
||||
@cachedmethod
|
||||
def get_likely_series_id(self, name):
|
||||
r = self.session.get('%s/shows.php' % self.server_url)
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
for elem in soup.find_all('h3'):
|
||||
show_name = elem.a.text.lower()
|
||||
show_id = int(match('show/([0-9]+)', elem.a['href']))
|
||||
# we could just return the id of the queried show, but as we
|
||||
# already downloaded the whole page we might as well fill in the
|
||||
# information for all the shows
|
||||
self.cache_for(self.get_likely_series_id, args=(show_name,), result=show_id)
|
||||
return self.cached_value(self.get_likely_series_id, args=(name,))
|
||||
|
||||
@cachedmethod
|
||||
def get_episode_url(self, series_id, season, number):
|
||||
"""Get the Addic7ed id for the given episode. Raises KeyError if none
|
||||
could be found
|
||||
|
||||
"""
|
||||
# download the page of the show, contains ids for all episodes all seasons
|
||||
r = self.session.get('%s/show/%d' % (self.server_url, series_id))
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
form = soup.find('form', attrs={'name': 'multidl'})
|
||||
for table in form.find_all('table'):
|
||||
for row in table.find_all('tr'):
|
||||
cell = row.find('td', 'MultiDldS')
|
||||
if not cell:
|
||||
continue
|
||||
m = matches('/serie/.+/([0-9]+)/([0-9]+)/', cell.a['href'])
|
||||
if not m:
|
||||
continue
|
||||
episode_url = cell.a['href']
|
||||
season_number = int(m[0])
|
||||
episode_number = int(m[1])
|
||||
# we could just return the url of the queried episode, but as we
|
||||
# already downloaded the whole page we might as well fill in the
|
||||
# information for all the episodes of the show
|
||||
self.cache_for(self.get_episode_url, args=(series_id, season_number, episode_number), result=episode_url)
|
||||
# raises KeyError if not found
|
||||
return self.cached_value(self.get_episode_url, args=(series_id, season, number))
|
||||
|
||||
# Do not cache this method in order to always check for the most recent
|
||||
# subtitles
|
||||
def get_sub_urls(self, episode_url):
|
||||
suburls = []
|
||||
r = self.session.get('%s/%s' % (self.server_url, episode_url))
|
||||
epsoup = BeautifulSoup(r.content, self.required_features)
|
||||
for releaseTable in epsoup.find_all('table', 'tabel95'):
|
||||
releaseRow = releaseTable.find('td', 'NewsTitle')
|
||||
if not releaseRow:
|
||||
continue
|
||||
release = releaseRow.text.strip()
|
||||
for row in releaseTable.find_all('tr'):
|
||||
link = row.find('a', 'buttonDownload')
|
||||
if not link:
|
||||
continue
|
||||
if 'href' not in link.attrs or not (link['href'].startswith('/original') or link['href'].startswith('/updated')):
|
||||
continue
|
||||
suburl = link['href']
|
||||
lang = guessit.Language(row.find('td', 'language').text.strip())
|
||||
result = {'suburl': suburl, 'language': lang, 'release': release}
|
||||
suburls.append(result)
|
||||
return suburls
|
||||
|
||||
def list_checked(self, video, languages):
|
||||
return self.query(video.path or video.release, languages, get_keywords(video.guess), video.series, video.season, video.episode)
|
||||
|
||||
def query(self, filepath, languages, keywords, series, season, episode):
|
||||
logger.debug(u'Getting subtitles for %s season %d episode %d with languages %r' % (series, season, episode, languages))
|
||||
self.init_cache()
|
||||
try:
|
||||
sid = self.get_likely_series_id(series.lower())
|
||||
except KeyError:
|
||||
logger.debug(u'Could not find series id for %s' % series)
|
||||
return []
|
||||
|
||||
try:
|
||||
ep_url = self.get_episode_url(sid, season, episode)
|
||||
except KeyError:
|
||||
logger.debug(u'Could not find episode id for %s season %d episode %d' % (series, season, episode))
|
||||
return []
|
||||
suburls = self.get_sub_urls(ep_url)
|
||||
|
||||
# filter the subtitles with our queried languages
|
||||
subtitles = []
|
||||
for suburl in suburls:
|
||||
language = suburl['language']
|
||||
if language not in languages:
|
||||
continue
|
||||
|
||||
path = get_subtitle_path(filepath, language, self.config.multi)
|
||||
subtitle = ResultSubtitle(path, language, self.__class__.__name__.lower(),
|
||||
'%s/%s' % (self.server_url, suburl['suburl']),
|
||||
keywords=[suburl['release']])
|
||||
subtitles.append(subtitle)
|
||||
return subtitles
|
||||
|
||||
|
||||
Service = Addic7ed
|
||||
@@ -16,13 +16,14 @@
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
from . import ServiceBase
|
||||
from ..cache import cachedmethod
|
||||
from ..exceptions import ServiceError
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..videos import Episode
|
||||
from ..utils import to_unicode
|
||||
import BeautifulSoup
|
||||
from ..videos import Episode
|
||||
from bs4 import BeautifulSoup
|
||||
from guessit.language import lang_set
|
||||
import logging
|
||||
import os.path
|
||||
import urllib
|
||||
try:
|
||||
import cPickle as pickle
|
||||
@@ -36,30 +37,23 @@ logger = logging.getLogger(__name__)
|
||||
class BierDopje(ServiceBase):
|
||||
server_url = 'http://api.bierdopje.com/A2B638AC5D804C2E/'
|
||||
api_based = True
|
||||
languages = {'en': 'en', 'nl': 'nl'}
|
||||
reverted_languages = False
|
||||
languages = lang_set(['en', 'nl'])
|
||||
videos = [Episode]
|
||||
require_video = False
|
||||
required_features = ['xml']
|
||||
|
||||
def __init__(self, config=None):
|
||||
super(BierDopje, self).__init__(config)
|
||||
self.showids = {}
|
||||
if self.config and self.config.cache_dir:
|
||||
self.init_cache()
|
||||
@cachedmethod
|
||||
def get_show_id(self, series):
|
||||
r = self.session.get('%sGetShowByName/%s' % (self.server_url, urllib.quote(series.lower())))
|
||||
if r.status_code != 200:
|
||||
logger.error(u'Request %s returned status code %d' % (r.url, r.status_code))
|
||||
return None
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
if soup.status.contents[0] == 'false':
|
||||
logger.debug(u'Could not find show %s' % series)
|
||||
return None
|
||||
|
||||
def init_cache(self):
|
||||
logger.debug(u'Initializing cache...')
|
||||
if not self.config or not self.config.cache_dir:
|
||||
raise ServiceError('Cache directory is required')
|
||||
self.showids_cache = os.path.join(self.config.cache_dir, 'bierdopje_showids.cache')
|
||||
if not os.path.exists(self.showids_cache):
|
||||
self.save_cache()
|
||||
|
||||
def save_cache(self):
|
||||
logger.debug(u'Saving showids to cache...')
|
||||
with self.lock:
|
||||
with open(self.showids_cache, 'w') as f:
|
||||
pickle.dump(self.showids, f)
|
||||
return int(soup.showid.contents[0])
|
||||
|
||||
def load_cache(self):
|
||||
logger.debug(u'Loading showids from cache...')
|
||||
@@ -67,25 +61,12 @@ class BierDopje(ServiceBase):
|
||||
with open(self.showids_cache, 'r') as f:
|
||||
self.showids = pickle.load(f)
|
||||
|
||||
def query(self, season, episode, languages, filepath, tvdbid=None, series=None):
|
||||
self.load_cache()
|
||||
def query(self, filepath, season, episode, languages, tvdbid=None, series=None):
|
||||
self.init_cache()
|
||||
if series:
|
||||
if series.lower() in self.showids: # from cache
|
||||
request_id = self.showids[series.lower()]
|
||||
logger.debug(u'Retreived showid %d for %s from cache' % (request_id, series))
|
||||
else: # query to get showid
|
||||
logger.debug(u'Getting showid from show name %s...' % series)
|
||||
r = self.session.get('%sGetShowByName/%s' % (self.server_url, urllib.quote(series.lower())))
|
||||
if r.status_code != 200:
|
||||
logger.error(u'Request %s returned status code %d' % (r.url, r.status_code))
|
||||
return []
|
||||
soup = BeautifulSoup.BeautifulStoneSoup(r.content)
|
||||
if soup.status.contents[0] == 'false':
|
||||
logger.debug(u'Could not find show %s' % series)
|
||||
return []
|
||||
request_id = int(soup.showid.contents[0])
|
||||
self.showids[series.lower()] = request_id
|
||||
self.save_cache()
|
||||
request_id = self.get_show_id(series.lower())
|
||||
if request_id is None:
|
||||
return []
|
||||
request_source = 'showid'
|
||||
request_is_tvdbid = 'false'
|
||||
elif tvdbid:
|
||||
@@ -96,14 +77,14 @@ class BierDopje(ServiceBase):
|
||||
raise ServiceError('One or more parameter missing')
|
||||
subtitles = []
|
||||
for language in languages:
|
||||
logger.debug(u'Getting subtitles for %s %d season %d episode %d with language %s' % (request_source, request_id, season, episode, language))
|
||||
r = self.session.get('%sGetAllSubsFor/%s/%s/%s/%s/%s' % (self.server_url, request_id, season, episode, language, request_is_tvdbid))
|
||||
logger.debug(u'Getting subtitles for %s %d season %d episode %d with language %s' % (request_source, request_id, season, episode, language.alpha2))
|
||||
r = self.session.get('%sGetAllSubsFor/%s/%s/%s/%s/%s' % (self.server_url, request_id, season, episode, language.alpha2, request_is_tvdbid))
|
||||
if r.status_code != 200:
|
||||
logger.error(u'Request %s returned status code %d' % (r.url, r.status_code))
|
||||
return []
|
||||
soup = BeautifulSoup.BeautifulStoneSoup(r.content)
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
if soup.status.contents[0] == 'false':
|
||||
logger.debug(u'Could not find subtitles for %s %d season %d episode %d with language %s' % (request_source, request_id, season, episode, language))
|
||||
logger.debug(u'Could not find subtitles for %s %d season %d episode %d with language %s' % (request_source, request_id, season, episode, language.alpha2))
|
||||
continue
|
||||
path = get_subtitle_path(filepath, language, self.config.multi)
|
||||
for result in soup.results('result'):
|
||||
@@ -112,11 +93,8 @@ class BierDopje(ServiceBase):
|
||||
subtitles.append(subtitle)
|
||||
return subtitles
|
||||
|
||||
def list(self, video, languages):
|
||||
if not self.check_validity(video, languages):
|
||||
return []
|
||||
results = self.query(video.season, video.episode, languages, video.path or video.release, video.tvdbid, video.series)
|
||||
return results
|
||||
def list_checked(self, video, languages):
|
||||
return self.query(video.path or video.release, video.season, video.episode, languages, video.tvdbid, video.series)
|
||||
|
||||
|
||||
Service = BierDopje
|
||||
|
||||
@@ -18,8 +18,10 @@
|
||||
from . import ServiceBase
|
||||
from ..exceptions import ServiceError, DownloadFailedError
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..videos import Episode, Movie
|
||||
from ..utils import to_unicode
|
||||
from ..videos import Episode, Movie
|
||||
from guessit.language import lang_set
|
||||
import guessit
|
||||
import gzip
|
||||
import logging
|
||||
import os.path
|
||||
@@ -32,34 +34,71 @@ logger = logging.getLogger(__name__)
|
||||
class OpenSubtitles(ServiceBase):
|
||||
server_url = 'http://api.opensubtitles.org/xml-rpc'
|
||||
api_based = True
|
||||
languages = {'aa': 'aar', 'ab': 'abk', 'af': 'afr', 'ak': 'aka', 'sq': 'alb', 'am': 'amh', 'ar': 'ara',
|
||||
'an': 'arg', 'hy': 'arm', 'as': 'asm', 'av': 'ava', 'ae': 'ave', 'ay': 'aym', 'az': 'aze',
|
||||
'ba': 'bak', 'bm': 'bam', 'eu': 'baq', 'be': 'bel', 'bn': 'ben', 'bh': 'bih', 'bi': 'bis',
|
||||
'bs': 'bos', 'br': 'bre', 'bg': 'bul', 'my': 'bur', 'ca': 'cat', 'ch': 'cha', 'ce': 'che',
|
||||
'zh': 'chi', 'cu': 'chu', 'cv': 'chv', 'kw': 'cor', 'co': 'cos', 'cr': 'cre', 'cs': 'cze',
|
||||
'da': 'dan', 'dv': 'div', 'nl': 'dut', 'dz': 'dzo', 'en': 'eng', 'eo': 'epo', 'et': 'est',
|
||||
'ee': 'ewe', 'fo': 'fao', 'fj': 'fij', 'fi': 'fin', 'fr': 'fre', 'fy': 'fry', 'ff': 'ful',
|
||||
'ka': 'geo', 'de': 'ger', 'gd': 'gla', 'ga': 'gle', 'gl': 'glg', 'gv': 'glv', 'el': 'ell',
|
||||
'gn': 'grn', 'gu': 'guj', 'ht': 'hat', 'ha': 'hau', 'he': 'heb', 'hz': 'her', 'hi': 'hin',
|
||||
'ho': 'hmo', 'hr': 'hrv', 'hu': 'hun', 'ig': 'ibo', 'is': 'ice', 'io': 'ido', 'ii': 'iii',
|
||||
'iu': 'iku', 'ie': 'ile', 'ia': 'ina', 'id': 'ind', 'ik': 'ipk', 'it': 'ita', 'jv': 'jav',
|
||||
'ja': 'jpn', 'kl': 'kal', 'kn': 'kan', 'ks': 'kas', 'kr': 'kau', 'kk': 'kaz', 'km': 'khm',
|
||||
'ki': 'kik', 'rw': 'kin', 'ky': 'kir', 'kv': 'kom', 'kg': 'kon', 'ko': 'kor', 'kj': 'kua',
|
||||
'ku': 'kur', 'lo': 'lao', 'la': 'lat', 'lv': 'lav', 'li': 'lim', 'ln': 'lin', 'lt': 'lit',
|
||||
'lb': 'ltz', 'lu': 'lub', 'lg': 'lug', 'mk': 'mac', 'mh': 'mah', 'ml': 'mal', 'mi': 'mao',
|
||||
'mr': 'mar', 'ms': 'may', 'mg': 'mlg', 'mt': 'mlt', 'mo': 'mol', 'mn': 'mon', 'na': 'nau',
|
||||
'nv': 'nav', 'nr': 'nbl', 'nd': 'nde', 'ng': 'ndo', 'ne': 'nep', 'nn': 'nno', 'nb': 'nob',
|
||||
'no': 'nor', 'ny': 'nya', 'oc': 'oci', 'oj': 'oji', 'or': 'ori', 'om': 'orm', 'os': 'oss',
|
||||
'pa': 'pan', 'fa': 'per', 'pi': 'pli', 'pl': 'pol', 'pt': 'por', 'ps': 'pus', 'qu': 'que',
|
||||
'rm': 'roh', 'rn': 'run', 'ru': 'rus', 'sg': 'sag', 'sa': 'san', 'sr': 'scc', 'si': 'sin',
|
||||
'sk': 'slo', 'sl': 'slv', 'se': 'sme', 'sm': 'smo', 'sn': 'sna', 'sd': 'snd', 'so': 'som',
|
||||
'st': 'sot', 'es': 'spa', 'sc': 'srd', 'ss': 'ssw', 'su': 'sun', 'sw': 'swa', 'sv': 'swe',
|
||||
'ty': 'tah', 'ta': 'tam', 'tt': 'tat', 'te': 'tel', 'tg': 'tgk', 'tl': 'tgl', 'th': 'tha',
|
||||
'bo': 'tib', 'ti': 'tir', 'to': 'ton', 'tn': 'tsn', 'ts': 'tso', 'tk': 'tuk', 'tr': 'tur',
|
||||
'tw': 'twi', 'ug': 'uig', 'uk': 'ukr', 'ur': 'urd', 'uz': 'uzb', 've': 'ven', 'vi': 'vie',
|
||||
'vo': 'vol', 'cy': 'wel', 'wa': 'wln', 'wo': 'wol', 'xh': 'xho', 'yi': 'yid', 'yo': 'yor',
|
||||
'za': 'zha', 'zu': 'zul', 'ro': 'rum', 'po': 'pob', 'un': 'unk', 'ay': 'ass'}
|
||||
reverted_languages = False
|
||||
# language list fetched from:
|
||||
# http://www.opensubtitles.org/addons/export_languages.php
|
||||
languages = lang_set(['aar', 'abk', 'ace', 'ach', 'ada', 'ady', 'afa', 'afh',
|
||||
'afr', 'ain', 'aka', 'akk', 'alb', 'ale', 'alg', 'alt',
|
||||
'amh', 'ang', 'apa', 'ara', 'arc', 'arg', 'arm', 'arn',
|
||||
'arp', 'art', 'arw', 'asm', 'ast', 'ath', 'aus', 'ava',
|
||||
'ave', 'awa', 'aym', 'aze', 'bad', 'bai', 'bak', 'bal',
|
||||
'bam', 'ban', 'baq', 'bas', 'bat', 'bej', 'bel', 'bem',
|
||||
'ben', 'ber', 'bho', 'bih', 'bik', 'bin', 'bis', 'bla',
|
||||
'bnt', 'bod', 'bos', 'bra', 'bre', 'btk', 'bua', 'bug',
|
||||
'bul', 'bur', 'byn', 'cad', 'cai', 'car', 'cat', 'cau',
|
||||
'ceb', 'cel', 'cha', 'chb', 'che', 'chg', 'chi', 'chk',
|
||||
'chm', 'chn', 'cho', 'chp', 'chr', 'chu', 'chv', 'chy',
|
||||
'cmc', 'cop', 'cor', 'cos', 'cpe', 'cpf', 'cpp', 'cre',
|
||||
'crh', 'crp', 'csb', 'cus', 'cym', 'cze', 'dak', 'dan',
|
||||
'dar', 'day', 'del', 'den', 'deu', 'dgr', 'din', 'div',
|
||||
'doi', 'dra', 'dua', 'dum', 'dut', 'dyu', 'dzo', 'efi',
|
||||
'egy', 'eka', 'elx', 'eng', 'enm', 'epo', 'est', 'eus',
|
||||
'ewe', 'ewo', 'fan', 'fao', 'fas', 'fat', 'fij', 'fil',
|
||||
'fin', 'fiu', 'fon', 'fra', 'fre', 'frm', 'fro', 'fry',
|
||||
'ful', 'fur', 'gaa', 'gay', 'gba', 'gem', 'geo', 'ger',
|
||||
'gez', 'gil', 'gla', 'gle', 'glg', 'glv', 'gmh', 'goh',
|
||||
'gon', 'gor', 'got', 'grb', 'grc', 'ell', 'grn', 'guj',
|
||||
'gwi', 'hai', 'hat', 'hau', 'haw', 'heb', 'her', 'hil',
|
||||
'him', 'hin', 'hit', 'hmn', 'hmo', 'hrv', 'hun', 'hup',
|
||||
'hye', 'iba', 'ibo', 'ice', 'ido', 'iii', 'ijo', 'iku',
|
||||
'ile', 'ilo', 'ina', 'inc', 'ind', 'ine', 'inh', 'ipk',
|
||||
'ira', 'iro', 'isl', 'ita', 'jav', 'jpn', 'jpr', 'jrb',
|
||||
'kaa', 'kab', 'kac', 'kal', 'kam', 'kan', 'kar', 'kas',
|
||||
'kat', 'kau', 'kaw', 'kaz', 'kbd', 'kha', 'khi', 'khm',
|
||||
'kho', 'kik', 'kin', 'kir', 'kmb', 'kok', 'kom', 'kon',
|
||||
'kor', 'kos', 'kpe', 'krc', 'kro', 'kru', 'kua', 'kum',
|
||||
'kur', 'kut', 'lad', 'lah', 'lam', 'lao', 'lat', 'lav',
|
||||
'lez', 'lim', 'lin', 'lit', 'lol', 'loz', 'ltz', 'lua',
|
||||
'lub', 'lug', 'lui', 'lun', 'luo', 'lus', 'mac', 'mad',
|
||||
'mag', 'mah', 'mai', 'mak', 'mal', 'man', 'mao', 'map',
|
||||
'mar', 'mas', 'may', 'mdf', 'mdr', 'men', 'mga', 'mic',
|
||||
'min', 'mis', 'mkd', 'mkh', 'mlg', 'mlt', 'mnc', 'mni',
|
||||
'mno', 'moh', 'mol', 'mon', 'mos', 'mri', 'msa', 'mwl',
|
||||
'mul', 'mun', 'mus', 'mwr', 'mya', 'myn', 'myv', 'nah',
|
||||
'nai', 'nap', 'nau', 'nav', 'nbl', 'nde', 'ndo', 'nds',
|
||||
'nep', 'new', 'nia', 'nic', 'niu', 'nld', 'nno', 'nob',
|
||||
'nog', 'non', 'nor', 'nso', 'nub', 'nwc', 'nya', 'nym',
|
||||
'nyn', 'nyo', 'nzi', 'oci', 'oji', 'ori', 'orm', 'osa',
|
||||
'oss', 'ota', 'oto', 'paa', 'pag', 'pal', 'pam', 'pan',
|
||||
'pap', 'pau', 'peo', 'per', 'phi', 'phn', 'pli', 'pol',
|
||||
'pon', 'por', 'pra', 'pro', 'pus', 'que', 'raj', 'rap',
|
||||
'rar', 'roa', 'roh', 'rom', 'ron', 'run', 'rup', 'rus',
|
||||
'sad', 'sag', 'sah', 'sai', 'sal', 'sam', 'san', 'sas',
|
||||
'sat', 'scc', 'scn', 'sco', 'scr', 'sel', 'sem', 'sga',
|
||||
'sgn', 'shn', 'sid', 'sin', 'sio', 'sit', 'sla', 'slk',
|
||||
'slo', 'slv', 'sma', 'sme', 'smi', 'smj', 'smn', 'smo',
|
||||
'sms', 'sna', 'snd', 'snk', 'sog', 'som', 'son', 'sot',
|
||||
'spa', 'sqi', 'srd', 'srp', 'srr', 'ssa', 'ssw', 'suk',
|
||||
'sun', 'sus', 'sux', 'swa', 'swe', 'syr', 'tah', 'tai',
|
||||
'tam', 'tat', 'tel', 'tem', 'ter', 'tet', 'tgk', 'tgl',
|
||||
'tha', 'tib', 'tig', 'tir', 'tiv', 'tkl', 'tlh', 'tli',
|
||||
'tmh', 'tog', 'ton', 'tpi', 'tsi', 'tsn', 'tso', 'tuk',
|
||||
'tum', 'tup', 'tur', 'tut', 'tvl', 'twi', 'tyv', 'udm',
|
||||
'uga', 'uig', 'ukr', 'umb', 'und', 'urd', 'uzb', 'vai',
|
||||
'ven', 'vie', 'vol', 'vot', 'wak', 'wal', 'war', 'was',
|
||||
'wel', 'wen', 'wln', 'wol', 'xal', 'xho', 'yao', 'yap',
|
||||
'yid', 'yor', 'ypk', 'zap', 'zen', 'zha', 'zho', 'znd',
|
||||
'zul', 'zun', 'rum', 'pob', 'unk', 'ass'])
|
||||
|
||||
videos = [Episode, Movie]
|
||||
require_video = False
|
||||
confidence_order = ['moviehash', 'imdbid', 'fulltext']
|
||||
@@ -92,7 +131,7 @@ class OpenSubtitles(ServiceBase):
|
||||
if not searches:
|
||||
raise ServiceError('One or more parameter missing')
|
||||
for search in searches:
|
||||
search['sublanguageid'] = ','.join([self.get_language(l) for l in languages])
|
||||
search['sublanguageid'] = ','.join(l.opensubtitles for l in languages)
|
||||
logger.debug(u'Getting subtitles %r with token %s' % (searches, self.token))
|
||||
results = self.server.SearchSubtitles(self.token, searches)
|
||||
if not results['data']:
|
||||
@@ -100,7 +139,7 @@ class OpenSubtitles(ServiceBase):
|
||||
return []
|
||||
subtitles = []
|
||||
for result in results['data']:
|
||||
language = self.get_revert_language(result['SubLanguageID'])
|
||||
language = guessit.Language(result['SubLanguageID'])
|
||||
path = get_subtitle_path(filepath, language, self.config.multi)
|
||||
confidence = 1 - float(self.confidence_order.index(result['MatchedBy'])) / float(len(self.confidence_order))
|
||||
subtitle = ResultSubtitle(path, language, service=self.__class__.__name__.lower(), link=result['SubDownloadLink'],
|
||||
@@ -108,9 +147,7 @@ class OpenSubtitles(ServiceBase):
|
||||
subtitles.append(subtitle)
|
||||
return subtitles
|
||||
|
||||
def list(self, video, languages):
|
||||
if not self.check_validity(video, languages):
|
||||
return []
|
||||
def list_checked(self, video, languages):
|
||||
results = []
|
||||
if video.exists:
|
||||
results = self.query(video.path or video.release, languages, moviehash=video.hashes['OpenSubtitles'], size=str(video.size))
|
||||
|
||||
Executable
+106
@@ -0,0 +1,106 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright 2011-2012 Antoine Bertin <diaoulael@gmail.com>
|
||||
#
|
||||
# This file is part of subliminal.
|
||||
#
|
||||
# subliminal is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU Lesser General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# subliminal is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU Lesser General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
from . import ServiceBase
|
||||
from ..exceptions import ServiceError, DownloadFailedError
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..utils import to_unicode
|
||||
from ..videos import Episode, Movie
|
||||
from guessit.language import lang_set
|
||||
from hashlib import md5, sha256
|
||||
import guessit
|
||||
import logging
|
||||
import xmlrpclib
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Podnapisi(ServiceBase):
|
||||
server_url = 'http://ssp.podnapisi.net:8000'
|
||||
api_based = True
|
||||
languages = lang_set(['sl', 'en', 'nn', 'ko', 'de', 'is', 'cs', 'fr', 'it', 'bs', 'jp', 'ar', 'ro',
|
||||
'hu', 'gr', 'zh', 'lt', 'et', 'lv', 'he', 'nl', 'da', 'sv', 'pl', 'ru', 'es',
|
||||
'sq', 'tr', 'fi', 'pt', 'bg', 'mk', 'sr', 'sk', 'hr', 'hi', 'th', 'ca', 'uk',
|
||||
'pb', 'ga', 'be', 'vi', 'fa', 'ca', 'id', 'ms'])
|
||||
#FIXME: ag and cyr not recognized by guessit
|
||||
videos = [Episode, Movie]
|
||||
require_video = True
|
||||
|
||||
def __init__(self, config=None):
|
||||
super(Podnapisi, self).__init__(config)
|
||||
self.server = xmlrpclib.ServerProxy(self.server_url)
|
||||
self.token = None
|
||||
|
||||
def init(self):
|
||||
super(Podnapisi, self).init()
|
||||
result = self.server.initiate(self.user_agent)
|
||||
if result['status'] != 200:
|
||||
raise ServiceError('Initiate failed')
|
||||
username = 'python_subliminal'
|
||||
password = sha256(md5('XWFXQ6gE5Oe12rv4qxXX').hexdigest() + result['nonce']).hexdigest()
|
||||
self.token = result['session']
|
||||
result = self.server.authenticate(self.token, username, password)
|
||||
if result['status'] != 200:
|
||||
raise ServiceError('Authenticate failed')
|
||||
|
||||
def terminate(self):
|
||||
super(Podnapisi, self).terminate()
|
||||
|
||||
def query(self, filepath, languages, moviehash):
|
||||
results = self.server.search(self.token, [moviehash])
|
||||
if results['status'] != 200:
|
||||
logger.error('Search failed with error code %d' % results['status'])
|
||||
return []
|
||||
if not results['results'] or not results['results'][moviehash]['subtitles']:
|
||||
logger.debug(u'Could not find subtitles for %r with token %s' % (moviehash, self.token))
|
||||
return []
|
||||
subtitles = []
|
||||
for result in results['results'][moviehash]['subtitles']:
|
||||
language = guessit.Language(result['lang'])
|
||||
if language == guessit.language.UNDETERMINED or language not in languages:
|
||||
continue
|
||||
path = get_subtitle_path(filepath, language, self.config.multi)
|
||||
subtitle = ResultSubtitle(path, language, service=self.__class__.__name__.lower(), link=result['id'],
|
||||
release=to_unicode(result['release']), confidence=result['weight'])
|
||||
subtitles.append(subtitle)
|
||||
if not subtitles:
|
||||
return []
|
||||
# Convert weight to confidence
|
||||
max_weight = float(max([s.confidence for s in subtitles]))
|
||||
min_weight = float(min([s.confidence for s in subtitles]))
|
||||
for subtitle in subtitles:
|
||||
if max_weight == 0 and min_weight == 0:
|
||||
subtitle.confidence = 1.0
|
||||
else:
|
||||
subtitle.confidence = (subtitle.confidence - min_weight) / (max_weight - min_weight)
|
||||
return subtitles
|
||||
|
||||
def list_checked(self, video, languages):
|
||||
results = self.query(video.path, languages, video.hashes['OpenSubtitles'])
|
||||
return results
|
||||
|
||||
def download(self, subtitle):
|
||||
results = self.server.download(self.token, [subtitle.link])
|
||||
if results['status'] != 200:
|
||||
raise DownloadFailedError()
|
||||
subtitle.link = 'http://www.podnapisi.net/static/podnapisi/' + results['names'][0]['filename']
|
||||
self.download_file(subtitle.link, subtitle.path)
|
||||
return subtitle
|
||||
|
||||
|
||||
Service = Podnapisi
|
||||
@@ -19,8 +19,10 @@ from . import ServiceBase
|
||||
from ..exceptions import ServiceError
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..videos import Episode, Movie
|
||||
from bs4 import BeautifulSoup
|
||||
from guessit.language import lang_set
|
||||
from subliminal.utils import get_keywords, split_keyword
|
||||
import BeautifulSoup
|
||||
import guessit
|
||||
import logging
|
||||
import re
|
||||
import urllib
|
||||
@@ -32,17 +34,15 @@ logger = logging.getLogger(__name__)
|
||||
class SubsWiki(ServiceBase):
|
||||
server_url = 'http://www.subswiki.com'
|
||||
api_based = False
|
||||
languages = {u'English (US)': 'en', u'English (UK)': 'en', u'English': 'en', u'French': 'fr', u'Brazilian': 'po',
|
||||
u'Portuguese': 'pt', u'Español (Latinoamérica)': 'es', u'Español (España)': 'es', u'Español': 'es',
|
||||
u'Italian': 'it', u'Català': 'ca'}
|
||||
reverted_languages = True
|
||||
languages = lang_set([u'English (US)', u'English (UK)', u'English', u'French', u'Brazilian',
|
||||
u'Portuguese', u'Español (Latinoamérica)', u'Español (España)',
|
||||
u'Español', u'Italian', u'Català'], strict=True)
|
||||
videos = [Episode, Movie]
|
||||
require_video = False
|
||||
release_pattern = re.compile('\nVersion (.+), ([0-9]+).([0-9])+ MBs')
|
||||
required_features = ['permissive']
|
||||
|
||||
def list(self, video, languages):
|
||||
if not self.check_validity(video, languages):
|
||||
return []
|
||||
def list_checked(self, video, languages):
|
||||
results = []
|
||||
if isinstance(video, Episode):
|
||||
results = self.query(video.path or video.release, languages, get_keywords(video.guess), series=video.series, season=video.season, episode=video.episode)
|
||||
@@ -74,7 +74,7 @@ class SubsWiki(ServiceBase):
|
||||
if r.status_code != 200:
|
||||
logger.error(u'Request %s returned status code %d' % (r.url, r.status_code))
|
||||
return []
|
||||
soup = BeautifulSoup.BeautifulSoup(r.content)
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
subtitles = []
|
||||
for sub in soup('td', {'class': 'NewsTitle'}):
|
||||
sub_keywords = split_keyword(self.release_pattern.search(sub.contents[1]).group(1).lower())
|
||||
@@ -82,8 +82,8 @@ class SubsWiki(ServiceBase):
|
||||
logger.debug(u'None of subtitle keywords %r in %r' % (sub_keywords, keywords))
|
||||
continue
|
||||
for html_language in sub.parent.parent.findAll('td', {'class': 'language'}):
|
||||
language = self.get_revert_language(html_language.string.strip())
|
||||
if not language in languages:
|
||||
language = guessit.Language(html_language.string.strip())
|
||||
if language not in languages:
|
||||
logger.debug(u'Language %r not in wanted languages %r' % (language, languages))
|
||||
continue
|
||||
html_status = html_language.findNextSibling('td')
|
||||
@@ -96,4 +96,5 @@ class SubsWiki(ServiceBase):
|
||||
subtitles.append(subtitle)
|
||||
return subtitles
|
||||
|
||||
|
||||
Service = SubsWiki
|
||||
|
||||
@@ -18,8 +18,10 @@
|
||||
from . import ServiceBase
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..videos import Episode
|
||||
from bs4 import BeautifulSoup
|
||||
from guessit.language import lang_set
|
||||
from subliminal.utils import get_keywords, split_keyword
|
||||
import BeautifulSoup
|
||||
import guessit
|
||||
import logging
|
||||
import re
|
||||
import unicodedata
|
||||
@@ -32,19 +34,19 @@ logger = logging.getLogger(__name__)
|
||||
class Subtitulos(ServiceBase):
|
||||
server_url = 'http://www.subtitulos.es'
|
||||
api_based = False
|
||||
languages = {u'English (US)': 'en', u'English (UK)': 'en', u'English': 'en', u'French': 'fr', u'Brazilian': 'po',
|
||||
u'Portuguese': 'pt', u'Español (Latinoamérica)': 'es', u'Español (España)': 'es', u'Español': 'es',
|
||||
u'Italian': 'it', u'Català': 'ca'}
|
||||
reverted_languages = True
|
||||
languages = lang_set([u'English (US)', u'English (UK)', u'English', u'French', u'Brazilian',
|
||||
u'Portuguese', u'Español (Latinoamérica)', u'Español (España)', u'Español',
|
||||
u'Italian', u'Català'], strict=True)
|
||||
videos = [Episode]
|
||||
require_video = False
|
||||
release_pattern = re.compile('Versión (.+) ([0-9]+).([0-9])+ megabytes')
|
||||
required_features = ['permissive']
|
||||
# the '.+' in the pattern for Version allows us to match both 'ó'
|
||||
# and the 'ó' char directly. This is because now BS4 converts the html
|
||||
# code chars into their equivalent unicode char
|
||||
release_pattern = re.compile('Versi.+n (.+) ([0-9]+).([0-9])+ megabytes')
|
||||
|
||||
def list(self, video, languages):
|
||||
if not self.check_validity(video, languages):
|
||||
return []
|
||||
results = self.query(video.path or video.release, languages, get_keywords(video.guess), video.series, video.season, video.episode)
|
||||
return results
|
||||
def list_checked(self, video, languages):
|
||||
return self.query(video.path or video.release, languages, get_keywords(video.guess), video.series, video.season, video.episode)
|
||||
|
||||
def query(self, filepath, languages, keywords, series, season, episode):
|
||||
request_series = series.lower().replace(' ', '_')
|
||||
@@ -58,7 +60,7 @@ class Subtitulos(ServiceBase):
|
||||
if r.status_code != 200:
|
||||
logger.error(u'Request %s returned status code %d' % (r.url, r.status_code))
|
||||
return []
|
||||
soup = BeautifulSoup.BeautifulSoup(r.content)
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
subtitles = []
|
||||
for sub in soup('div', {'id': 'version'}):
|
||||
sub_keywords = split_keyword(self.release_pattern.search(sub.find('p', {'class': 'title-sub'}).contents[1]).group(1).lower())
|
||||
@@ -66,8 +68,8 @@ class Subtitulos(ServiceBase):
|
||||
logger.debug(u'None of subtitle keywords %r in %r' % (sub_keywords, keywords))
|
||||
continue
|
||||
for html_language in sub.findAllNext('ul', {'class': 'sslist'}):
|
||||
language = self.get_revert_language(html_language.findNext('li', {'class': 'li-idioma'}).find('strong').contents[0].string.strip())
|
||||
if not language in languages:
|
||||
language = guessit.Language(html_language.findNext('li', {'class': 'li-idioma'}).find('strong').contents[0].string.strip())
|
||||
if language not in languages:
|
||||
logger.debug(u'Language %r not in wanted languages %r' % (language, languages))
|
||||
continue
|
||||
html_status = html_language.findNext('li', {'class': 'li-estado green'})
|
||||
@@ -80,4 +82,5 @@ class Subtitulos(ServiceBase):
|
||||
subtitles.append(subtitle)
|
||||
return subtitles
|
||||
|
||||
|
||||
Service = Subtitulos
|
||||
|
||||
@@ -18,6 +18,8 @@
|
||||
from . import ServiceBase
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..videos import Episode, Movie, UnknownVideo
|
||||
from guessit.language import lang_set
|
||||
import guessit
|
||||
import logging
|
||||
|
||||
|
||||
@@ -26,21 +28,17 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
class TheSubDB(ServiceBase):
|
||||
server_url = 'http://api.thesubdb.com/' # for testing purpose, use http://sandbox.thesubdb.com/ instead
|
||||
user_agent = 'SubDB/1.0 (subliminal/0.5; https://github.com/Diaoul/subliminal)' # defined by the API
|
||||
user_agent = 'SubDB/1.0 (subliminal/0.6; https://github.com/Diaoul/subliminal)' # defined by the API
|
||||
api_based = True
|
||||
languages = {'af': 'af', 'cs': 'cs', 'da': 'da', 'de': 'de', 'en': 'en', 'es': 'es', 'fi': 'fi',
|
||||
'fr': 'fr', 'hu': 'hu', 'id': 'id', 'it': 'it', 'la': 'la', 'nl': 'nl', 'no': 'no',
|
||||
'oc': 'oc', 'pl': 'pl', 'pt': 'pt', 'ro': 'ro', 'ru': 'ru', 'sl': 'sl', 'sr': 'sr',
|
||||
'sv': 'sv', 'tr': 'tr'} # list available with the API at http://sandbox.thesubdb.com/?action=languages
|
||||
reverted_languages = False
|
||||
languages = lang_set(['af', 'cs', 'da', 'de', 'en', 'es', 'fi',
|
||||
'fr', 'hu', 'id', 'it', 'la', 'nl', 'no',
|
||||
'oc', 'pl', 'pt', 'ro', 'ru', 'sl', 'sr',
|
||||
'sv', 'tr'], strict=True) # list available with the API at http://sandbox.thesubdb.com/?action=languages
|
||||
videos = [Movie, Episode, UnknownVideo]
|
||||
require_video = True
|
||||
|
||||
def list(self, video, languages):
|
||||
if not self.check_validity(video, languages):
|
||||
return []
|
||||
results = self.query(video.path, video.hashes['TheSubDB'], languages)
|
||||
return results
|
||||
def list_checked(self, video, languages):
|
||||
return self.query(video.path, video.hashes['TheSubDB'], languages)
|
||||
|
||||
def query(self, filepath, moviehash, languages):
|
||||
r = self.session.get(self.server_url, params={'action': 'search', 'hash': moviehash})
|
||||
@@ -50,7 +48,7 @@ class TheSubDB(ServiceBase):
|
||||
if r.status_code != 200:
|
||||
logger.error(u'Request %s returned status code %d' % (r.url, r.status_code))
|
||||
return []
|
||||
available_languages = set([self.get_revert_language(l) for l in r.content.split(',')])
|
||||
available_languages = set(guessit.Language(l) for l in r.content.split(','))
|
||||
languages &= available_languages
|
||||
if not languages:
|
||||
logger.debug(u'Could not find subtitles for hash %s with languages %r (only %r available)' % (moviehash, languages, available_languages))
|
||||
@@ -58,8 +56,9 @@ class TheSubDB(ServiceBase):
|
||||
subtitles = []
|
||||
for language in languages:
|
||||
path = get_subtitle_path(filepath, language, self.config.multi)
|
||||
subtitle = ResultSubtitle(path, language, service=self.__class__.__name__.lower(), link='%s?action=download&hash=%s&language=%s' % (self.server_url, moviehash, self.get_language(language)))
|
||||
subtitle = ResultSubtitle(path, language, self.__class__.__name__.lower(), '%s?action=download&hash=%s&language=%s' % (self.server_url, moviehash, language.alpha2))
|
||||
subtitles.append(subtitle)
|
||||
return subtitles
|
||||
|
||||
|
||||
Service = TheSubDB
|
||||
|
||||
Executable
+146
@@ -0,0 +1,146 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Copyright 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# This file is part of subliminal.
|
||||
#
|
||||
# subliminal is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the GNU Lesser General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# subliminal is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU Lesser General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
from . import ServiceBase
|
||||
from ..cache import cachedmethod
|
||||
from ..subtitles import get_subtitle_path, ResultSubtitle
|
||||
from ..videos import Episode
|
||||
from bs4 import BeautifulSoup
|
||||
from guessit.language import lang_set
|
||||
from subliminal.utils import get_keywords
|
||||
import guessit
|
||||
import logging
|
||||
import re
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def match(pattern, string):
|
||||
try:
|
||||
return re.search(pattern, string).group(1)
|
||||
except AttributeError:
|
||||
logger.debug(u'Could not match %r on %r' % (pattern, string))
|
||||
return None
|
||||
|
||||
|
||||
class TvSubtitles(ServiceBase):
|
||||
server_url = 'http://www.tvsubtitles.net'
|
||||
api_based = False
|
||||
languages = lang_set([u'English', u'Español', u'French', u'German',
|
||||
u'Brazilian', u'Russian', u'Ukrainian', u'Italian',
|
||||
u'Greek', u'Arabic', u'Hungarian', u'Polish',
|
||||
u'Turkish', u'Dutch', u'Portuguese', u'Swedish',
|
||||
u'Danish', u'Finnish', u'Korean', u'Chinese',
|
||||
u'Japanese', u'Bulgarian', u'Czech', u'Romanian'], strict=True)
|
||||
videos = [Episode]
|
||||
require_video = False
|
||||
required_features = ['permissive']
|
||||
|
||||
@cachedmethod
|
||||
def get_likely_series_id(self, name):
|
||||
r = self.session.post('%s/search.php' % self.server_url, data={'q': name})
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
maindiv = soup.find('div', 'left')
|
||||
results = []
|
||||
for elem in maindiv.find_all('li'):
|
||||
sid = int(match('tvshow-([0-9]+)\.html', elem.a['href']))
|
||||
show_name = match('(.*) \(', elem.a.text)
|
||||
results.append((show_name, sid))
|
||||
#TODO: pick up the best one in a smart way
|
||||
result = results[0]
|
||||
return result[1]
|
||||
|
||||
@cachedmethod
|
||||
def get_episode_id(self, series_id, season, number):
|
||||
"""Get the TvSubtitles id for the given episode. Raises KeyError if none
|
||||
could be found."""
|
||||
# download the page of the season, contains ids for all episodes
|
||||
episode_id = None
|
||||
r = self.session.get('%s/tvshow-%d-%d.html' % (self.server_url, series_id, season))
|
||||
soup = BeautifulSoup(r.content, self.required_features)
|
||||
table = soup.find('table', id='table5')
|
||||
for row in table.find_all('tr'):
|
||||
cells = row.find_all('td')
|
||||
if not cells:
|
||||
continue
|
||||
|
||||
episode_number = match('x([0-9]+)', cells[0].text)
|
||||
if not episode_number:
|
||||
continue
|
||||
|
||||
episode_number = int(episode_number)
|
||||
episode_id = int(match('episode-([0-9]+)', cells[1].a['href']))
|
||||
# we could just return the id of the queried episode, but as we
|
||||
# already downloaded the whole page we might as well fill in the
|
||||
# information for all the episodes of the season
|
||||
self.cache_for(self.get_episode_id, args=(series_id, season, episode_number), result=episode_id)
|
||||
# raises KeyError if not found
|
||||
return self.cached_value(self.get_episode_id, args=(series_id, season, number))
|
||||
|
||||
# Do not cache this method in order to always check for the most recent
|
||||
# subtitles
|
||||
def get_sub_ids(self, episode_id):
|
||||
subids = []
|
||||
r = self.session.get('%s/episode-%d.html' % (self.server_url, episode_id))
|
||||
epsoup = BeautifulSoup(r.content, self.required_features)
|
||||
for subdiv in epsoup.find_all('a'):
|
||||
if 'href' not in subdiv.attrs or not subdiv['href'].startswith('/subtitle'):
|
||||
continue
|
||||
subid = int(match('([0-9]+)', subdiv['href']))
|
||||
lang = guessit.Language(match('flags/(.*).gif', subdiv.img['src']))
|
||||
result = {'subid': subid, 'language': lang}
|
||||
for p in subdiv.find_all('p'):
|
||||
if 'alt' in p.attrs and p['alt'] == 'rip':
|
||||
result['rip'] = p.text.strip()
|
||||
if 'alt' in p.attrs and p['alt'] == 'release':
|
||||
result['release'] = p.text.strip()
|
||||
|
||||
subids.append(result)
|
||||
return subids
|
||||
|
||||
def list_checked(self, video, languages):
|
||||
return self.query(video.path or video.release, languages, get_keywords(video.guess), video.series, video.season, video.episode)
|
||||
|
||||
def query(self, filepath, languages, keywords, series, season, episode):
|
||||
logger.debug(u'Getting subtitles for %s season %d episode %d with languages %r' % (series, season, episode, languages))
|
||||
self.init_cache()
|
||||
sid = self.get_likely_series_id(series.lower())
|
||||
try:
|
||||
ep_id = self.get_episode_id(sid, season, episode)
|
||||
except KeyError:
|
||||
logger.debug(u'Could not find episode id for %s season %d episode %d' % (series, season, episode))
|
||||
return []
|
||||
subids = self.get_sub_ids(ep_id)
|
||||
# filter the subtitles with our queried languages
|
||||
subtitles = []
|
||||
for subid in subids:
|
||||
language = subid['language']
|
||||
if language not in languages:
|
||||
continue
|
||||
path = get_subtitle_path(filepath, language, self.config.multi)
|
||||
subtitle = ResultSubtitle(path, language, self.__class__.__name__.lower(),
|
||||
'%s/download-%d.html' % (self.server_url, subid['subid']),
|
||||
keywords=[subid['rip'], subid['release']])
|
||||
subtitles.append(subtitle)
|
||||
return subtitles
|
||||
|
||||
def download(self, subtitle):
|
||||
self.download_zip_file(subtitle.link, subtitle.path)
|
||||
|
||||
|
||||
Service = TvSubtitles
|
||||
@@ -15,13 +15,13 @@
|
||||
#
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
from .languages import list_languages, convert_language
|
||||
import os.path
|
||||
import guessit
|
||||
from guessit.language import is_language
|
||||
|
||||
|
||||
__all__ = ['Subtitle', 'EmbeddedSubtitle', 'ExternalSubtitle', 'ResultSubtitle', 'get_subtitle_path']
|
||||
|
||||
|
||||
#: Subtitles extensions
|
||||
EXTENSIONS = ['.srt', '.sub', '.txt']
|
||||
|
||||
@@ -30,7 +30,8 @@ class Subtitle(object):
|
||||
"""Base class for subtitles
|
||||
|
||||
:param string path: path to the subtitle
|
||||
:param string language: language of the subtitle (second element of :class:`~subliminal.languages.LANGUAGES`)
|
||||
:param language: language of the subtitle
|
||||
:type language: :class:`guessit.Language`
|
||||
|
||||
"""
|
||||
def __init__(self, path, language):
|
||||
@@ -49,7 +50,8 @@ class EmbeddedSubtitle(Subtitle):
|
||||
"""Subtitle embedded in a container
|
||||
|
||||
:param string path: path to the subtitle
|
||||
:param string language: language of the subtitle (second element of :class:`~subliminal.languages.LANGUAGES`)
|
||||
:param language: language of the subtitle
|
||||
:type language: :class:`guessit.Language`
|
||||
:param int track_id: id of the subtitle track in the container
|
||||
|
||||
"""
|
||||
@@ -59,7 +61,7 @@ class EmbeddedSubtitle(Subtitle):
|
||||
|
||||
@classmethod
|
||||
def from_enzyme(cls, path, subtitle):
|
||||
language = convert_language(subtitle.language, 1, 2)
|
||||
language = guessit.Language(subtitle.language) or None
|
||||
return cls(path, language, subtitle.trackno)
|
||||
|
||||
|
||||
@@ -76,8 +78,8 @@ class ExternalSubtitle(Subtitle):
|
||||
if not extension:
|
||||
raise ValueError('Not a supported subtitle extension')
|
||||
language = os.path.splitext(path[:len(path) - len(extension)])[1][1:]
|
||||
if not language in list_languages(1):
|
||||
language = None
|
||||
language = guessit.Language(language) or None
|
||||
|
||||
return cls(path, language)
|
||||
|
||||
|
||||
@@ -85,7 +87,8 @@ class ResultSubtitle(ExternalSubtitle):
|
||||
"""Subtitle found using :mod:`~subliminal.services`
|
||||
|
||||
:param string path: path to the subtitle
|
||||
:param string language: language of the subtitle (second element of :class:`~subliminal.languages.LANGUAGES`)
|
||||
:param language: language of the subtitle
|
||||
:type language: :class:`guessit.Language`
|
||||
:param string service: name of the service
|
||||
:param string link: download link for the subtitle
|
||||
:param string release: release name of the video
|
||||
@@ -111,20 +114,27 @@ class ResultSubtitle(ExternalSubtitle):
|
||||
"""
|
||||
extension = os.path.splitext(self.path)[0]
|
||||
language = os.path.splitext(self.path[:len(self.path) - len(extension)])[1][1:]
|
||||
if not language in list_languages(1):
|
||||
return True
|
||||
return False
|
||||
return not is_language(language)
|
||||
|
||||
def __repr__(self):
|
||||
return 'ResultSubtitle(%s, %s, %.2f, %s)' % (self.language, self.service, self.confidence, self.release)
|
||||
|
||||
|
||||
def get_subtitle_path(video_path, language, multi):
|
||||
"""Create the subtitle path from the given video path using language if multi"""
|
||||
"""Create the subtitle path from the given video path using language if multi
|
||||
|
||||
:param string video_path: path to the video
|
||||
:param language: language of the subtitle
|
||||
:type language: :class:`guessit.Language`
|
||||
:param bool multi: whether to use multi language naming or not
|
||||
:return: path of the subtitle
|
||||
:rtype: string
|
||||
|
||||
"""
|
||||
if not os.path.exists(video_path):
|
||||
path = os.path.splitext(os.path.basename(video_path))[0]
|
||||
else:
|
||||
path = os.path.splitext(video_path)[0]
|
||||
if multi and language:
|
||||
return path + '.%s%s' % (language, EXTENSIONS[0])
|
||||
return path + '.%s%s' % (language.alpha2, EXTENSIONS[0])
|
||||
return path + '%s' % EXTENSIONS[0]
|
||||
|
||||
+22
-11
@@ -16,7 +16,6 @@
|
||||
# You should have received a copy of the GNU Lesser General Public License
|
||||
# along with subliminal. If not, see <http://www.gnu.org/licenses/>.
|
||||
from . import subtitles
|
||||
from .languages import list_languages
|
||||
import enzyme
|
||||
import guessit
|
||||
import hashlib
|
||||
@@ -130,14 +129,23 @@ class Video(object):
|
||||
logger.debug(u'Failed parsing %s with enzyme' % self.path)
|
||||
if isinstance(video_infos, enzyme.core.AVContainer):
|
||||
results.extend([subtitles.EmbeddedSubtitle.from_enzyme(self.path, s) for s in video_infos.subtitles])
|
||||
for l in list_languages(1):
|
||||
for e in subtitles.EXTENSIONS:
|
||||
single_path = basepath + '%s' % e
|
||||
if os.path.exists(single_path):
|
||||
results.append(subtitles.ExternalSubtitle(single_path, None))
|
||||
multi_path = basepath + '.%s%s' % (l, e)
|
||||
if os.path.exists(multi_path):
|
||||
results.append(subtitles.ExternalSubtitle(multi_path, l))
|
||||
|
||||
# cannot use glob here because it chokes if there are any square
|
||||
# brackets inside the filename, so we have to use basic string
|
||||
# startswith/endswith comparisons
|
||||
folder, basename = os.path.split(basepath)
|
||||
existing = [f for f in os.listdir(folder) if f.startswith(basename)]
|
||||
for path in existing:
|
||||
for ext in subtitles.EXTENSIONS:
|
||||
if path.endswith(ext):
|
||||
possible_lang = path[len(basename) + 1:-len(ext)]
|
||||
if possible_lang == '':
|
||||
results.append(subtitles.ExternalSubtitle(path, None))
|
||||
else:
|
||||
lang = guessit.Language(possible_lang)
|
||||
if lang:
|
||||
results.append(subtitles.ExternalSubtitle(path, lang))
|
||||
|
||||
return results
|
||||
|
||||
def __repr__(self):
|
||||
@@ -189,11 +197,12 @@ class UnknownVideo(Video):
|
||||
pass
|
||||
|
||||
|
||||
def scan(entry, max_depth=3, depth=0):
|
||||
def scan(entry, max_depth=3, scan_filter=None, depth=0):
|
||||
"""Scan a path for videos and subtitles
|
||||
|
||||
:param string entry: path
|
||||
:param int max_depth: maximum folder depth
|
||||
:param function scan_filter: filter function that takes a path as argument and returns a boolean indicating whether it has to be filtered out (``True``) or not (``False``)
|
||||
:param int depth: starting depth
|
||||
:return: found videos and subtitles
|
||||
:rtype: list of (:class:`Video`, [:class:`~subliminal.subtitles.Subtitle`])
|
||||
@@ -207,13 +216,15 @@ def scan(entry, max_depth=3, depth=0):
|
||||
logger.debug(u'Scanning directory %s with depth %d/%d' % (entry, depth, max_depth))
|
||||
result = []
|
||||
for e in os.listdir(entry):
|
||||
result.extend(scan(os.path.join(entry, e), max_depth, depth + 1))
|
||||
result.extend(scan(os.path.join(entry, e), max_depth, scan_filter, depth + 1))
|
||||
return result
|
||||
if os.path.isfile(entry) or depth == 0:
|
||||
logger.debug(u'Scanning file %s with depth %d/%d' % (entry, depth, max_depth))
|
||||
if depth != 0: # trust the user: only check for valid format if recursing
|
||||
if mimetypes.guess_type(entry)[0] not in MIMETYPES and os.path.splitext(entry)[1] not in EXTENSIONS:
|
||||
return []
|
||||
if scan_filter is not None and scan_filter(entry):
|
||||
return []
|
||||
video = Video.from_path(entry)
|
||||
return [(video, video.scan())]
|
||||
logger.warning(u'Scanning entry %s failed with depth %d/%d' % (entry, depth, max_depth))
|
||||
|
||||
Reference in New Issue
Block a user