Subliminal update
This commit is contained in:
+24
-35
@@ -18,10 +18,10 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__version__ = '0.3-dev'
|
||||
__all__ = [ 'Guess', 'Language',
|
||||
'guess_file_info', 'guess_video_info',
|
||||
'guess_movie_info', 'guess_episode_info' ]
|
||||
__version__ = '0.3.1'
|
||||
__all__ = ['Guess', 'Language',
|
||||
'guess_file_info', 'guess_video_info',
|
||||
'guess_movie_info', 'guess_episode_info']
|
||||
|
||||
|
||||
from guessit.guess import Guess, merge_all
|
||||
@@ -31,6 +31,7 @@ import logging
|
||||
|
||||
log = logging.getLogger("guessit")
|
||||
|
||||
|
||||
class NullHandler(logging.Handler):
|
||||
def emit(self, record):
|
||||
pass
|
||||
@@ -40,9 +41,7 @@ h = NullHandler()
|
||||
log.addHandler(h)
|
||||
|
||||
|
||||
|
||||
|
||||
def guess_file_info(filename, filetype, info = [ 'filename' ]):
|
||||
def guess_file_info(filename, filetype, info=None):
|
||||
"""info can contain the names of the various plugins, such as 'filename' to
|
||||
detect filename info, or 'hash_md5' to get the md5 hash of the file.
|
||||
|
||||
@@ -52,27 +51,30 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
|
||||
result = []
|
||||
hashers = []
|
||||
|
||||
if info is None:
|
||||
info = ['filename']
|
||||
|
||||
if isinstance(info, basestring):
|
||||
info = [ info ]
|
||||
info = [info]
|
||||
|
||||
for infotype in info:
|
||||
if infotype == 'filename':
|
||||
m = IterativeMatcher(filename, filetype = filetype)
|
||||
m = IterativeMatcher(filename, filetype=filetype)
|
||||
result.append(m.matched())
|
||||
|
||||
elif infotype == 'hash_mpc':
|
||||
import hash_mpc
|
||||
from guessit.hash_mpc import hash_file
|
||||
try:
|
||||
result.append(Guess({ 'hash_mpc': hash_mpc.hash_file(filename) },
|
||||
confidence = 1.0))
|
||||
result.append(Guess({'hash_mpc': hash_file(filename)},
|
||||
confidence=1.0))
|
||||
except Exception, e:
|
||||
log.warning('Could not compute MPC-style hash because: %s' % e)
|
||||
|
||||
elif infotype == 'hash_ed2k':
|
||||
import hash_ed2k
|
||||
from guessit.hash_ed2k import hash_file
|
||||
try:
|
||||
result.append(Guess({ 'hash_ed2k': hash_ed2k.hash_file(filename) },
|
||||
confidence = 1.0))
|
||||
result.append(Guess({'hash_ed2k': hash_file(filename)},
|
||||
confidence=1.0))
|
||||
except Exception, e:
|
||||
log.warning('Could not compute ed2k hash because: %s' % e)
|
||||
|
||||
@@ -88,18 +90,6 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
|
||||
else:
|
||||
log.warning('Invalid infotype: %s' % infotype)
|
||||
|
||||
|
||||
"""For plugins which depend on some optional library, import them like that:
|
||||
|
||||
if infotype == 'plugin_name':
|
||||
try:
|
||||
import optional_lib
|
||||
except ImportError:
|
||||
raise Exception, 'The plugin module cannot be loaded because the optional_lib lib is missing'
|
||||
|
||||
# do some stuff
|
||||
"""
|
||||
|
||||
# do all the hashes now, but on a single pass
|
||||
if hashers:
|
||||
try:
|
||||
@@ -112,22 +102,21 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
|
||||
hasher.update(chunk)
|
||||
|
||||
for infotype, hasher in hashers:
|
||||
result.append(Guess({ infotype: hasher.hexdigest() },
|
||||
confidence = 1.0))
|
||||
result.append(Guess({infotype: hasher.hexdigest()},
|
||||
confidence=1.0))
|
||||
except Exception, e:
|
||||
log.warning('Could not compute hash because: %s' % e)
|
||||
|
||||
|
||||
return merge_all(result)
|
||||
|
||||
|
||||
def guess_video_info(filename, info = [ 'filename' ]):
|
||||
def guess_video_info(filename, info=None):
|
||||
return guess_file_info(filename, 'autodetect', info)
|
||||
|
||||
def guess_movie_info(filename, info = [ 'filename' ]):
|
||||
|
||||
def guess_movie_info(filename, info=None):
|
||||
return guess_file_info(filename, 'movie', info)
|
||||
|
||||
def guess_episode_info(filename, info = [ 'filename' ]):
|
||||
|
||||
def guess_episode_info(filename, info=None):
|
||||
return guess_file_info(filename, 'episode', info)
|
||||
|
||||
|
||||
|
||||
@@ -1,75 +0,0 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
#from guessit import movie, episode
|
||||
import os, os.path
|
||||
import logging
|
||||
|
||||
log = logging.getLogger('guessit.autodetect')
|
||||
|
||||
def within(x, nrange):
|
||||
"""Return whether a number is inside a given range, specified as a list or tuple
|
||||
of the lower and upper bounds."""
|
||||
low, high = nrange
|
||||
return low <= x <= high
|
||||
|
||||
def guess_filename_info(filename):
|
||||
log.debug('Trying to guess info for file: ' + filename)
|
||||
|
||||
# try to guess info as if it were an episode
|
||||
episode_info = episode.guess_episode_filename(filename)
|
||||
|
||||
# 1- if we found either season/episodeNumber, then we're pretty sure it must
|
||||
# be an episode
|
||||
if 'season' in episode_info or 'episodeNumber' in episode_info:
|
||||
log.debug('Likely an episode as it contains season and/or episodeNumber: ' + filename)
|
||||
episode_info.update({ 'type': 'episode' }, confidence = 0.9)
|
||||
return episode_info
|
||||
|
||||
# try to guess info as if it were a movie
|
||||
movie_info = movie.guess_movie_filename(filename)
|
||||
|
||||
# 2- if the file exists, try to guess its type using its size
|
||||
if os.path.exists(filename):
|
||||
size = os.stat(filename).st_size / (1024 * 1024)
|
||||
|
||||
# if size <= 1/2 of 1CD -> episode (very unlikely a movie so small)
|
||||
if size < 400:
|
||||
log.debug('Likely an episode due to its small size (%dMB): %s' % (size, filename))
|
||||
episode_info.update({ 'type': 'episode' }, confidence = 0.8)
|
||||
return episode_info
|
||||
|
||||
# if size > 2G -> movie (even fullHD eps aren't that big yet)
|
||||
if size > 2048:
|
||||
log.debug('Likely a movie due to its big size (%dMB): %s' % (size, filename))
|
||||
movie_info.update({ 'type': 'movie' }, confidence = 0.8)
|
||||
return movie_info
|
||||
|
||||
# if size == 1CD or 2CDs -> movie
|
||||
if within(size, [690, 710]) or within(size, [1380, 1420]):
|
||||
log.debug('Likely a movie due to its size close to a CD size (%dMB): %s' % (size, filename))
|
||||
movie_info.update({ 'type': 'movie' }, confidence = 0.8)
|
||||
return movie_info
|
||||
|
||||
|
||||
# 3- if all else fails, assume it's a movie
|
||||
log.debug('Couldn\'t make an informed guess... Assuming file is a movie: %s' % filename)
|
||||
movie_info.update({ 'type': 'movie' }, confidence = 0.5)
|
||||
return movie_info
|
||||
+31
-28
@@ -21,6 +21,7 @@
|
||||
import datetime
|
||||
import re
|
||||
|
||||
|
||||
def search_year(string):
|
||||
"""Looks for year patterns, and if found return the year and group span.
|
||||
Assumes there are sentinels at the beginning and end of the string that
|
||||
@@ -62,34 +63,35 @@ def search_date(string):
|
||||
|
||||
dsep = r'[-/ \.]'
|
||||
|
||||
date_rexps = [ # 20010823
|
||||
r'[^0-9]' +
|
||||
r'(?P<year>[0-9]{4})' +
|
||||
r'(?P<month>[0-9]{2})' +
|
||||
r'(?P<day>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
date_rexps = [
|
||||
# 20010823
|
||||
r'[^0-9]' +
|
||||
r'(?P<year>[0-9]{4})' +
|
||||
r'(?P<month>[0-9]{2})' +
|
||||
r'(?P<day>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
|
||||
# 2001-08-23
|
||||
r'[^0-9]' +
|
||||
r'(?P<year>[0-9]{4})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<day>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
# 2001-08-23
|
||||
r'[^0-9]' +
|
||||
r'(?P<year>[0-9]{4})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<day>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
|
||||
# 23-08-2001
|
||||
r'[^0-9]' +
|
||||
r'(?P<day>[0-9]{2})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<year>[0-9]{4})' +
|
||||
r'[^0-9]',
|
||||
# 23-08-2001
|
||||
r'[^0-9]' +
|
||||
r'(?P<day>[0-9]{2})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<year>[0-9]{4})' +
|
||||
r'[^0-9]',
|
||||
|
||||
# 23-08-01
|
||||
r'[^0-9]' +
|
||||
r'(?P<day>[0-9]{2})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<year>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
]
|
||||
# 23-08-01
|
||||
r'[^0-9]' +
|
||||
r'(?P<day>[0-9]{2})' + dsep +
|
||||
r'(?P<month>[0-9]{2})' + dsep +
|
||||
r'(?P<year>[0-9]{2})' +
|
||||
r'[^0-9]',
|
||||
]
|
||||
|
||||
for drexp in date_rexps:
|
||||
match = re.search(drexp, string)
|
||||
@@ -98,7 +100,7 @@ def search_date(string):
|
||||
year, month, day = int(d['year']), int(d['month']), int(d['day'])
|
||||
# years specified as 2 digits should be adjusted here
|
||||
if year < 100:
|
||||
if year > (datetime.date.today().year % 100)+ 5:
|
||||
if year > (datetime.date.today().year % 100) + 5:
|
||||
year = 1900 + year
|
||||
else:
|
||||
year = 2000 + year
|
||||
@@ -120,8 +122,9 @@ def search_date(string):
|
||||
continue
|
||||
|
||||
# looks like we have a valid date
|
||||
# note: span is [+1,-1] because we don't want to include the non-digit char
|
||||
# note: span is [+1,-1] because we don't want to include the
|
||||
# non-digit char
|
||||
start, end = match.span()
|
||||
return (date, (start+1, end-1))
|
||||
return (date, (start + 1, end - 1))
|
||||
|
||||
return None, None
|
||||
|
||||
@@ -1,76 +0,0 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import subtitle_exts, video_exts, episode_rexps, find_properties, canonical_form
|
||||
import os.path
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.filetype")
|
||||
|
||||
|
||||
def guess_filetype(filename, filetype = 'autodetect'):
|
||||
other = {}
|
||||
|
||||
# look at the extension first
|
||||
fileext = os.path.splitext(filename)[1][1:].lower()
|
||||
if fileext in subtitle_exts:
|
||||
if 'movie' in filetype:
|
||||
filetype = 'moviesubtitle'
|
||||
elif 'episode' in filetype:
|
||||
filetype = 'episodesubtitle'
|
||||
else:
|
||||
filetype = 'subtitle'
|
||||
other = { 'container': fileext }
|
||||
elif fileext in video_exts:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'video'
|
||||
other = { 'container': fileext }
|
||||
else:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'unknown'
|
||||
other = { 'extension': fileext }
|
||||
|
||||
# now look whether there are some specific hints for episode vs movie
|
||||
if filetype in ('video', 'subtitle'):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, filename, re.IGNORECASE)
|
||||
if match:
|
||||
if filetype == 'video':
|
||||
filetype = 'episode'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'episodesubtitle'
|
||||
break
|
||||
|
||||
for prop, value, start, end in find_properties(filename):
|
||||
if canonical_form(value) == 'DVB':
|
||||
if filetype == 'video':
|
||||
filetype = 'episode'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'episodesubtitle'
|
||||
break
|
||||
|
||||
# if no episode info found, assume it's a movie
|
||||
if filetype == 'video':
|
||||
filetype = 'movie'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'moviesubtitle'
|
||||
|
||||
return filetype, other
|
||||
@@ -50,11 +50,11 @@ def split_path(path):
|
||||
|
||||
# on Unix systems, the root folder is '/'
|
||||
if head == '/' and tail == '':
|
||||
return [ '/' ] + result
|
||||
return ['/'] + result
|
||||
|
||||
# on Windows, the root folder is a drive letter (eg: 'C:\')
|
||||
if len(head) == 3 and head[1:] == ':\\' and tail == '':
|
||||
return [ head ] + result
|
||||
return [head] + result
|
||||
|
||||
if head == '' and tail == '':
|
||||
return result
|
||||
@@ -64,17 +64,10 @@ def split_path(path):
|
||||
path = head
|
||||
continue
|
||||
|
||||
result = [ tail ] + result
|
||||
result = [tail] + result
|
||||
path = head
|
||||
|
||||
|
||||
def split_path_components(filename):
|
||||
"""Returns the filename split into [ dir*, basename, ext ]."""
|
||||
result = split_path(filename)
|
||||
basename = result.pop(-1)
|
||||
return result + list(os.path.splitext(basename))
|
||||
|
||||
|
||||
def file_in_same_dir(ref_file, desired_file):
|
||||
"""Return the path for a file in the same dir as a given reference file.
|
||||
|
||||
@@ -82,17 +75,17 @@ def file_in_same_dir(ref_file, desired_file):
|
||||
'~/smewt/smewt.settings'
|
||||
|
||||
"""
|
||||
return os.path.join(*(split_path(ref_file)[:-1] + [ desired_file ]))
|
||||
return os.path.join(*(split_path(ref_file)[:-1] + [desired_file]))
|
||||
|
||||
|
||||
def load_file_in_same_dir(ref_file, filename):
|
||||
"""Load a given file. Works even when the file is contained inside a zip."""
|
||||
path = split_path(ref_file)[:-1] + [ filename ]
|
||||
path = split_path(ref_file)[:-1] + [filename]
|
||||
|
||||
for i, p in enumerate(path):
|
||||
if p.endswith('.zip'):
|
||||
zfilename = os.path.join(*path[:i+1])
|
||||
zfilename = os.path.join(*path[:i + 1])
|
||||
zfile = zipfile.ZipFile(zfilename)
|
||||
return zfile.read('/'.join(path[i+1:]))
|
||||
return zfile.read('/'.join(path[i + 1:]))
|
||||
|
||||
return open(os.path.join(*path)).read()
|
||||
|
||||
+60
-45
@@ -26,9 +26,12 @@ log = logging.getLogger("guessit.guess")
|
||||
|
||||
|
||||
class Guess(dict):
|
||||
"""A Guess is a dictionary which has an associated confidence for each of its values.
|
||||
"""A Guess is a dictionary which has an associated confidence for each of
|
||||
its values.
|
||||
|
||||
As it is a subclass of dict, you can use it everywhere you expect a
|
||||
simple dict."""
|
||||
|
||||
As it is a subclass of dict, you can use it everywhere you expect a simple dict"""
|
||||
def __init__(self, *args, **kwargs):
|
||||
try:
|
||||
confidence = kwargs.pop('confidence')
|
||||
@@ -52,20 +55,20 @@ class Guess(dict):
|
||||
elif isinstance(value, unicode):
|
||||
data[prop] = value.encode('utf-8')
|
||||
elif isinstance(value, list):
|
||||
data[prop] = [ str(x) for x in value ]
|
||||
data[prop] = [str(x) for x in value]
|
||||
|
||||
return data
|
||||
|
||||
def nice_string(self):
|
||||
data = self.to_utf8_dict()
|
||||
|
||||
parts = json.dumps(data, indent = 4).split('\n')
|
||||
parts = json.dumps(data, indent=4).split('\n')
|
||||
for i, p in enumerate(parts):
|
||||
if p[:5] != ' "':
|
||||
continue
|
||||
|
||||
prop = p.split('"')[1]
|
||||
parts[i] = (' [%.2f] "' % (self._confidence.get(prop) or -1)) + p[5:]
|
||||
parts[i] = (' [%.2f] "' % self.confidence(prop)) + p[5:]
|
||||
|
||||
return '\n'.join(parts)
|
||||
|
||||
@@ -73,9 +76,9 @@ class Guess(dict):
|
||||
return str(self.to_utf8_dict())
|
||||
|
||||
def confidence(self, prop):
|
||||
return self._confidence[prop]
|
||||
return self._confidence.get(prop, -1)
|
||||
|
||||
def set(self, prop, value, confidence = None):
|
||||
def set(self, prop, value, confidence=None):
|
||||
self[prop] = value
|
||||
if confidence is not None:
|
||||
self._confidence[prop] = confidence
|
||||
@@ -83,7 +86,7 @@ class Guess(dict):
|
||||
def set_confidence(self, prop, value):
|
||||
self._confidence[prop] = value
|
||||
|
||||
def update(self, other, confidence = None):
|
||||
def update(self, other, confidence=None):
|
||||
dict.update(self, other)
|
||||
if isinstance(other, Guess):
|
||||
for prop in other:
|
||||
@@ -94,36 +97,36 @@ class Guess(dict):
|
||||
self._confidence[prop] = confidence
|
||||
|
||||
def update_highest_confidence(self, other):
|
||||
"""Update this guess with the values from the given one. In case there is
|
||||
property present in both, only the one with the highest one is kept."""
|
||||
"""Update this guess with the values from the given one. In case
|
||||
there is property present in both, only the one with the highest one
|
||||
is kept."""
|
||||
if not isinstance(other, Guess):
|
||||
raise ValueError, 'Can only call this function on Guess instances'
|
||||
raise ValueError('Can only call this function on Guess instances')
|
||||
|
||||
for prop in other:
|
||||
if prop in self and self._confidence[prop] >= other._confidence[prop]:
|
||||
if prop in self and self.confidence(prop) >= other.confidence(prop):
|
||||
continue
|
||||
self[prop] = other[prop]
|
||||
self._confidence[prop] = other._confidence[prop]
|
||||
|
||||
|
||||
self._confidence[prop] = other.confidence(prop)
|
||||
|
||||
|
||||
def choose_int(g1, g2):
|
||||
"""Function used by merge_similar_guesses to choose between 2 possible properties
|
||||
when they are integers."""
|
||||
"""Function used by merge_similar_guesses to choose between 2 possible
|
||||
properties when they are integers."""
|
||||
v1, c1 = g1 # value, confidence
|
||||
v2, c2 = g2
|
||||
if (v1 == v2):
|
||||
return (v1, 1 - (1-c1)*(1-c2))
|
||||
return (v1, 1 - (1 - c1) * (1 - c2))
|
||||
else:
|
||||
if c1 > c2:
|
||||
return (v1, c1 - c2)
|
||||
else:
|
||||
return (v2, c2 - c1)
|
||||
|
||||
|
||||
def choose_string(g1, g2):
|
||||
"""Function used by merge_similar_guesses to choose between 2 possible properties
|
||||
when they are strings.
|
||||
"""Function used by merge_similar_guesses to choose between 2 possible
|
||||
properties when they are strings.
|
||||
|
||||
If the 2 strings are similar, or one is contained in the other, the latter is returned
|
||||
with an increased confidence.
|
||||
@@ -142,7 +145,7 @@ def choose_string(g1, g2):
|
||||
('Hello', 0.75)
|
||||
|
||||
>>> choose_string(('Hello', 0.4), ('Hello World', 0.4))
|
||||
('Hello', 0.64000000000000001)
|
||||
('Hello', 0.64)
|
||||
|
||||
>>> choose_string(('simpsons', 0.5), ('The Simpsons', 0.5))
|
||||
('The Simpsons', 0.75)
|
||||
@@ -159,7 +162,7 @@ def choose_string(g1, g2):
|
||||
v1, v2 = v1.strip(), v2.strip()
|
||||
v1l, v2l = v1.lower(), v2.lower()
|
||||
|
||||
combined_prob = 1 - (1-c1)*(1-c2)
|
||||
combined_prob = 1 - (1 - c1) * (1 - c2)
|
||||
|
||||
if v1l == v2l:
|
||||
return (v1, combined_prob)
|
||||
@@ -191,26 +194,31 @@ def _merge_similar_guesses_nocheck(guesses, prop, choose):
|
||||
|
||||
This function assumes there are at least 2 valid guesses."""
|
||||
|
||||
similar = [ guess for guess in guesses if prop in guess ]
|
||||
similar = [guess for guess in guesses if prop in guess]
|
||||
|
||||
g1, g2 = similar[0], similar[1]
|
||||
|
||||
other_props = set(g1) & set(g2) - set([prop])
|
||||
if other_props:
|
||||
log.debug('guess 1: %s' % g1)
|
||||
log.debug('guess 2: %s' % g2)
|
||||
for prop in other_props:
|
||||
if g1[prop] != g2[prop]:
|
||||
log.warning('both guesses to be merged have more than one different property in common, bailing out...')
|
||||
log.warning('both guesses to be merged have more than one '
|
||||
'different property in common, bailing out...')
|
||||
return
|
||||
|
||||
# merge all props of s2 into s1, updating the confidence for the considered property
|
||||
# merge all props of s2 into s1, updating the confidence for the
|
||||
# considered property
|
||||
v1, v2 = g1[prop], g2[prop]
|
||||
c1, c2 = g1.confidence(prop), g2.confidence(prop)
|
||||
|
||||
new_value, new_confidence = choose((v1, c1), (v2, c2))
|
||||
if new_confidence >= c1:
|
||||
log.debug("Updating matching property '%s' with confidence %.2f" % (prop, new_confidence))
|
||||
msg = "Updating matching property '%s' with confidence %.2f"
|
||||
else:
|
||||
log.debug("Updating non-matching property '%s' with confidence %.2f" % (prop, new_confidence))
|
||||
msg = "Updating non-matching property '%s' with confidence %.2f"
|
||||
log.debug(msg % (prop, new_confidence))
|
||||
|
||||
g2[prop] = new_value
|
||||
g2.set_confidence(prop, new_confidence)
|
||||
@@ -218,12 +226,13 @@ def _merge_similar_guesses_nocheck(guesses, prop, choose):
|
||||
g1.update(g2)
|
||||
guesses.remove(g2)
|
||||
|
||||
|
||||
def merge_similar_guesses(guesses, prop, choose):
|
||||
"""Take a list of guesses and merge those which have the same properties,
|
||||
increasing or decreasing the confidence depending on whether their values
|
||||
are similar."""
|
||||
|
||||
similar = [ guess for guess in guesses if prop in guess ]
|
||||
similar = [guess for guess in guesses if prop in guess]
|
||||
if len(similar) < 2:
|
||||
# nothing to merge
|
||||
return
|
||||
@@ -233,9 +242,13 @@ def merge_similar_guesses(guesses, prop, choose):
|
||||
|
||||
if len(similar) > 2:
|
||||
log.debug('complex merge, trying our best...')
|
||||
before = len(guesses)
|
||||
_merge_similar_guesses_nocheck(guesses, prop, choose)
|
||||
merge_similar_guesses(guesses, prop, choose)
|
||||
return
|
||||
after = len(guesses)
|
||||
if after < before:
|
||||
# recurse only when the previous call actually did something,
|
||||
# otherwise we end up in an infinite loop
|
||||
merge_similar_guesses(guesses, prop, choose)
|
||||
|
||||
|
||||
def merge_append_guesses(guesses, prop):
|
||||
@@ -245,14 +258,12 @@ def merge_append_guesses(guesses, prop):
|
||||
DEPRECATED, remove with old guessers
|
||||
|
||||
"""
|
||||
|
||||
|
||||
similar = [ guess for guess in guesses if prop in guess ]
|
||||
similar = [guess for guess in guesses if prop in guess]
|
||||
if not similar:
|
||||
return
|
||||
|
||||
merged = similar[0]
|
||||
merged[prop] = [ merged[prop] ]
|
||||
merged[prop] = [merged[prop]]
|
||||
# TODO: what to do with global confidence? mean of them all?
|
||||
|
||||
for m in similar[1:]:
|
||||
@@ -261,17 +272,18 @@ def merge_append_guesses(guesses, prop):
|
||||
merged[prop].append(m[prop])
|
||||
else:
|
||||
if prop2 in m:
|
||||
log.warning('overwriting property "%s" with value ' % (prop2, m[prop2]))
|
||||
log.warning('overwriting property "%s" with value %s' % (prop2, m[prop2]))
|
||||
merged[prop2] = m[prop2]
|
||||
# TODO: confidence also
|
||||
|
||||
guesses.remove(m)
|
||||
|
||||
|
||||
def merge_all(guesses, append = []):
|
||||
"""Merges all the guesses in a single result, removes very unlikely values, and returns it.
|
||||
You can specify a list of properties that should be appended into a list instead of being
|
||||
merged.
|
||||
def merge_all(guesses, append=None):
|
||||
"""Merge all the guesses in a single result, remove very unlikely values,
|
||||
and return it.
|
||||
You can specify a list of properties that should be appended into a list
|
||||
instead of being merged.
|
||||
|
||||
>>> merge_all([ Guess({ 'season': 2 }, confidence = 0.6),
|
||||
... Guess({ 'episodeNumber': 13 }, confidence = 0.8) ])
|
||||
@@ -286,20 +298,24 @@ def merge_all(guesses, append = []):
|
||||
return Guess()
|
||||
|
||||
result = guesses[0]
|
||||
if append is None:
|
||||
append = []
|
||||
|
||||
for g in guesses[1:]:
|
||||
# first append our appendable properties
|
||||
for prop in append:
|
||||
if prop in g:
|
||||
result.set(prop, result.get(prop, []) + [ g[prop] ],
|
||||
# TODO: what to do with confidence here? maybe an arithmetic mean...
|
||||
confidence = g.confidence(prop))
|
||||
result.set(prop, result.get(prop, []) + [g[prop]],
|
||||
# TODO: what to do with confidence here? maybe an
|
||||
# arithmetic mean...
|
||||
confidence=g.confidence(prop))
|
||||
|
||||
del g[prop]
|
||||
|
||||
# then merge the remaining ones
|
||||
if set(result) & set(g):
|
||||
log.warning('duplicate properties %s in merged result...' % (set(result) & set(g)))
|
||||
dups = set(result) & set(g)
|
||||
if dups:
|
||||
log.warning('duplicate properties %s in merged result...' % dups)
|
||||
|
||||
result.update_highest_confidence(g)
|
||||
|
||||
@@ -314,4 +330,3 @@ def merge_all(guesses, append = []):
|
||||
result[prop] = list(set(result[prop]))
|
||||
|
||||
return result
|
||||
|
||||
|
||||
@@ -18,8 +18,9 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
import hashlib, os.path
|
||||
import hashlib
|
||||
import os.path
|
||||
|
||||
|
||||
def hash_file(filename):
|
||||
"""Returns the ed2k hash of a given file.
|
||||
@@ -31,6 +32,7 @@ def hash_file(filename):
|
||||
os.path.getsize(filename),
|
||||
hash_filehash(filename).upper())
|
||||
|
||||
|
||||
def hash_filehash(filename):
|
||||
"""Returns the ed2k hash of a given file.
|
||||
|
||||
@@ -42,8 +44,10 @@ def hash_filehash(filename):
|
||||
def gen(f):
|
||||
while True:
|
||||
x = f.read(9728000)
|
||||
if x: yield x
|
||||
else: return
|
||||
if x:
|
||||
yield x
|
||||
else:
|
||||
return
|
||||
|
||||
def md4_hash(data):
|
||||
m = md4()
|
||||
@@ -55,4 +59,5 @@ def hash_filehash(filename):
|
||||
hashes = [md4_hash(data).digest() for data in a]
|
||||
if len(hashes) == 1:
|
||||
return hashes[0].encode("hex")
|
||||
else: return md4_hash(reduce(lambda a,d: a + d, hashes, "")).hexd
|
||||
else:
|
||||
return md4_hash(reduce(lambda a, d: a + d, hashes, "")).hexd
|
||||
|
||||
+18
-18
@@ -18,8 +18,9 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
import struct, os
|
||||
import struct
|
||||
import os
|
||||
|
||||
|
||||
def hash_file(filename):
|
||||
"""This function is taken from:
|
||||
@@ -32,25 +33,24 @@ def hash_file(filename):
|
||||
f = open(filename, "rb")
|
||||
|
||||
filesize = os.path.getsize(filename)
|
||||
hash = filesize
|
||||
hash_value = filesize
|
||||
|
||||
if filesize < 65536 * 2:
|
||||
raise Exception, "SizeError: size is %d, should be > 132K..." % filesize
|
||||
raise Exception("SizeError: size is %d, should be > 132K..." % filesize)
|
||||
|
||||
for x in range(65536/bytesize):
|
||||
buffer = f.read(bytesize)
|
||||
(l_value,)= struct.unpack(longlongformat, buffer)
|
||||
hash += l_value
|
||||
hash = hash & 0xFFFFFFFFFFFFFFFF #to remain as 64bit number
|
||||
for x in range(65536 / bytesize):
|
||||
buf = f.read(bytesize)
|
||||
(l_value,) = struct.unpack(longlongformat, buf)
|
||||
hash_value += l_value
|
||||
hash_value = hash_value & 0xFFFFFFFFFFFFFFFF #to remain as 64bit number
|
||||
|
||||
|
||||
f.seek(max(0,filesize-65536),0)
|
||||
for x in range(65536/bytesize):
|
||||
buffer = f.read(bytesize)
|
||||
(l_value,)= struct.unpack(longlongformat, buffer)
|
||||
hash += l_value
|
||||
hash = hash & 0xFFFFFFFFFFFFFFFF
|
||||
f.seek(max(0, filesize - 65536), 0)
|
||||
for x in range(65536 / bytesize):
|
||||
buf = f.read(bytesize)
|
||||
(l_value,) = struct.unpack(longlongformat, buf)
|
||||
hash_value += l_value
|
||||
hash_value = hash_value & 0xFFFFFFFFFFFFFFFF
|
||||
|
||||
f.close()
|
||||
returnedhash = "%016x" % hash
|
||||
return returnedhash
|
||||
|
||||
return "%016x" % hash_value
|
||||
|
||||
+53
-44
@@ -19,28 +19,28 @@
|
||||
#
|
||||
|
||||
from guessit import fileutils
|
||||
import os.path
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger('guessit.language')
|
||||
|
||||
|
||||
|
||||
# downloaded from http://www.loc.gov/standards/iso639-2/ISO-639-2_utf-8.txt
|
||||
#
|
||||
# Description of the fields:
|
||||
# "An alpha-3 (bibliographic) code, an alpha-3 (terminologic) code (when given),
|
||||
# an alpha-2 code (when given), an English name, and a French name of a language
|
||||
# are all separated by pipe (|) characters."
|
||||
_iso639_contents = fileutils.load_file_in_same_dir(__file__,
|
||||
'ISO-639-2_utf-8.txt')
|
||||
language_matrix = [ l.strip().decode('utf-8').split('|')
|
||||
for l in fileutils.load_file_in_same_dir(__file__, 'ISO-639-2_utf-8.txt').split('\n') ]
|
||||
for l in _iso639_contents.split('\n') ]
|
||||
|
||||
lng3 = frozenset(filter(bool, (l[0] for l in language_matrix)))
|
||||
lng3term = frozenset(filter(bool, (l[1] for l in language_matrix)))
|
||||
lng2 = frozenset(filter(bool, (l[2] for l in language_matrix)))
|
||||
lng_en_name = frozenset(filter(bool, (lng for l in language_matrix for lng in l[3].lower().split('; '))))
|
||||
lng_fr_name = frozenset(filter(bool, (lng for l in language_matrix for lng in l[4].lower().split('; '))))
|
||||
lng3 = frozenset(l[0] for l in language_matrix if l[0])
|
||||
lng3term = frozenset(l[1] for l in language_matrix if l[1])
|
||||
lng2 = frozenset(l[2] for l in language_matrix if l[2])
|
||||
lng_en_name = frozenset(lng for l in language_matrix
|
||||
for lng in l[3].lower().split('; ') if lng)
|
||||
lng_fr_name = frozenset(lng for l in language_matrix
|
||||
for lng in l[4].lower().split('; ') if lng)
|
||||
lng_all_names = lng3 | lng3term | lng2 | lng_en_name | lng_fr_name
|
||||
|
||||
lng3_to_lng3term = dict((l[0], l[1]) for l in language_matrix if l[1])
|
||||
@@ -50,22 +50,29 @@ lng3_to_lng2 = dict((l[0], l[2]) for l in language_matrix if l[2])
|
||||
lng2_to_lng3 = dict((l[2], l[0]) for l in language_matrix if l[2])
|
||||
|
||||
# we only return the first given english name, hoping it is the most used one
|
||||
lng3_to_lng_en_name = dict((l[0], l[3].split('; ')[0]) for l in language_matrix if l[3])
|
||||
lng_en_name_to_lng3 = dict((en_name.lower(), l[0]) for l in language_matrix if l[3] for en_name in l[3].split('; '))
|
||||
lng3_to_lng_en_name = dict((l[0], l[3].split('; ')[0])
|
||||
for l in language_matrix if l[3])
|
||||
lng_en_name_to_lng3 = dict((en_name.lower(), l[0])
|
||||
for l in language_matrix if l[3]
|
||||
for en_name in l[3].split('; '))
|
||||
|
||||
# we only return the first given french name, hoping it is the most used one
|
||||
lng3_to_lng_fr_name = dict((l[0], l[4].split('; ')[0]) for l in language_matrix if l[4])
|
||||
lng_fr_name_to_lng3 = dict((fr_name.lower(), l[0]) for l in language_matrix if l[4] for fr_name in l[4].split('; '))
|
||||
lng3_to_lng_fr_name = dict((l[0], l[4].split('; ')[0])
|
||||
for l in language_matrix if l[4])
|
||||
lng_fr_name_to_lng3 = dict((fr_name.lower(), l[0])
|
||||
for l in language_matrix if l[4]
|
||||
for fr_name in l[4].split('; '))
|
||||
|
||||
|
||||
def is_language(language):
|
||||
return language.lower() in lng_all_names
|
||||
|
||||
|
||||
class Language(object):
|
||||
"""This class represents a human language.
|
||||
|
||||
You can initialize it with pretty much everything, as it knows conversion from
|
||||
ISO-639 2-letter and 3-letter codes, English and French names.
|
||||
You can initialize it with pretty much everything, as it knows conversion
|
||||
from ISO-639 2-letter and 3-letter codes, English and French names.
|
||||
|
||||
>>> Language('fr')
|
||||
Language(French)
|
||||
@@ -79,12 +86,16 @@ class Language(object):
|
||||
if len(language) == 2:
|
||||
lang = lng2_to_lng3.get(language)
|
||||
elif len(language) == 3:
|
||||
lang = language if language in lng3 else lng3term_to_lng3.get(language)
|
||||
lang = (language
|
||||
if language in lng3
|
||||
else lng3term_to_lng3.get(language))
|
||||
else:
|
||||
lang = lng_en_name_to_lng3.get(language) or lng_fr_name_to_lng3.get(language)
|
||||
lang = (lng_en_name_to_lng3.get(language) or
|
||||
lng_fr_name_to_lng3.get(language))
|
||||
|
||||
if lang is None:
|
||||
raise ValueError, 'The given string "%s" could not be identified as a language' % language
|
||||
msg = 'The given string "%s" could not be identified as a language'
|
||||
raise ValueError(msg % language)
|
||||
|
||||
self.lang = lang
|
||||
|
||||
@@ -103,7 +114,6 @@ class Language(object):
|
||||
def french_name(self):
|
||||
return lng3_to_lng_fr_name[self.lang]
|
||||
|
||||
|
||||
def __hash__(self):
|
||||
return hash(self.lang)
|
||||
|
||||
@@ -132,19 +142,15 @@ class Language(object):
|
||||
return 'Language(%s)' % self
|
||||
|
||||
|
||||
|
||||
def search_language(string, lang_filter = None):
|
||||
def search_language(string, lang_filter=None):
|
||||
"""Looks for language patterns, and if found return the language object,
|
||||
its group span and an associated confidence.
|
||||
|
||||
you can specify a list of allowed languages using the lang_filter argument,
|
||||
as in lang_filter = [ 'fr', 'eng', 'spanish' ]
|
||||
|
||||
Assumes there are sentinels at the beginning and end of the string that
|
||||
always allow matching a non-letter delimiting the language.
|
||||
|
||||
>>> search_language('movie [en].avi')
|
||||
(Language(English), (7, 9), 0.80000000000000004)
|
||||
(Language(English), (7, 9), 0.8)
|
||||
|
||||
>>> search_language('the zen fat cat and the gay mad men got a new fan', lang_filter = ['en', 'fr', 'es'])
|
||||
(None, None, None)
|
||||
@@ -153,25 +159,27 @@ def search_language(string, lang_filter = None):
|
||||
# list of common words which could be interpreted as languages, but which
|
||||
# are far too common to be able to say they represent a language in the
|
||||
# middle of a string (where they most likely carry their commmon meaning)
|
||||
lng_common_words = frozenset([ # english words
|
||||
'is', 'it', 'am', 'mad', 'men', 'man', 'run', 'sin', 'st', 'to',
|
||||
'no', 'non', 'war', 'min', 'new', 'car', 'day', 'bad', 'bat', 'fan',
|
||||
'fry', 'cop', 'zen', 'gay', 'fat', 'cherokee', 'got', 'an', 'as',
|
||||
'cat', 'her', 'be', 'hat', 'sun', 'may', 'my', 'mr',
|
||||
# french words
|
||||
'bas', 'de', 'le', 'son', 'vo', 'vf', 'ne', 'ca', 'ce', 'et', 'que',
|
||||
'mal', 'est', 'vol', 'or', 'mon', 'se',
|
||||
# spanish words
|
||||
'la', 'el', 'del', 'por', 'mar',
|
||||
# other
|
||||
'ind', 'arw', 'ts', 'ii', 'bin', 'chan', 'ss', 'san'
|
||||
])
|
||||
lng_common_words = frozenset([
|
||||
# english words
|
||||
'is', 'it', 'am', 'mad', 'men', 'man', 'run', 'sin', 'st', 'to',
|
||||
'no', 'non', 'war', 'min', 'new', 'car', 'day', 'bad', 'bat', 'fan',
|
||||
'fry', 'cop', 'zen', 'gay', 'fat', 'cherokee', 'got', 'an', 'as',
|
||||
'cat', 'her', 'be', 'hat', 'sun', 'may', 'my', 'mr',
|
||||
# french words
|
||||
'bas', 'de', 'le', 'son', 'vo', 'vf', 'ne', 'ca', 'ce', 'et', 'que',
|
||||
'mal', 'est', 'vol', 'or', 'mon', 'se',
|
||||
# spanish words
|
||||
'la', 'el', 'del', 'por', 'mar',
|
||||
# other
|
||||
'ind', 'arw', 'ts', 'ii', 'bin', 'chan', 'ss', 'san', 'oss', 'iii',
|
||||
'vi'
|
||||
])
|
||||
sep = r'[](){} \._-+'
|
||||
|
||||
if lang_filter:
|
||||
lang_filter = set(Language(l) for l in lang_filter)
|
||||
|
||||
slow = string.lower()
|
||||
slow = ' %s ' % string.lower()
|
||||
confidence = 1.0 # for all of them
|
||||
for lang in lng_all_names:
|
||||
|
||||
@@ -183,7 +191,7 @@ def search_language(string, lang_filter = None):
|
||||
if pos != -1:
|
||||
end = pos + len(lang)
|
||||
# make sure our word is always surrounded by separators
|
||||
if slow[pos-1] not in sep or slow[end] not in sep:
|
||||
if slow[pos - 1] not in sep or slow[end] not in sep:
|
||||
continue
|
||||
|
||||
language = Language(slow[pos:end])
|
||||
@@ -201,10 +209,11 @@ def search_language(string, lang_filter = None):
|
||||
elif len(lang) == 3:
|
||||
confidence = 0.9
|
||||
else:
|
||||
# Note: we could either be really confident that we found a language
|
||||
# or assume that full language names are too common words
|
||||
# Note: we could either be really confident that we found a
|
||||
# language or assume that full language names are too
|
||||
# common words
|
||||
confidence = 0.3 # going with the low-confidence route here
|
||||
|
||||
return language, (pos, end), confidence
|
||||
return language, (pos - 1, end - 1), confidence
|
||||
|
||||
return None, None, None
|
||||
|
||||
+63
-505
@@ -2,8 +2,7 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
# Copyright (c) 2011 Ricard Marxer <ricardmp@gmail.com>
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
@@ -19,282 +18,18 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import fileutils, textutils
|
||||
from guessit.guess import Guess, merge_similar_guesses, merge_all, choose_int, choose_string
|
||||
from guessit.date import search_date, search_year
|
||||
from guessit.language import search_language
|
||||
from guessit.filetype import guess_filetype
|
||||
from guessit.patterns import video_exts, subtitle_exts, sep, deleted, video_rexps, websites, episode_rexps, weak_episode_rexps, non_episode_title, find_properties, canonical_form, unlikely_series
|
||||
from guessit.matchtree import get_group, find_group, leftover_valid_groups, tree_to_string
|
||||
from guessit.textutils import find_first_level_groups, split_on_groups, blank_region, clean_string, to_utf8
|
||||
from guessit.fileutils import split_path_components
|
||||
import datetime
|
||||
import os.path
|
||||
import re
|
||||
from guessit.matchtree import MatchTree
|
||||
from guessit.textutils import to_utf8
|
||||
from guessit.guess import (merge_similar_guesses, merge_all,
|
||||
choose_int, choose_string)
|
||||
import copy
|
||||
import logging
|
||||
import mimetypes
|
||||
|
||||
log = logging.getLogger("guessit.matcher")
|
||||
|
||||
|
||||
|
||||
def split_explicit_groups(string):
|
||||
"""return the string split into explicit groups, that is, those either
|
||||
between parenthese, square brackets or curly braces, and those separated
|
||||
by a dash."""
|
||||
result = find_first_level_groups(string, '()')
|
||||
result = reduce(lambda l, x: l + find_first_level_groups(x, '[]'), result, [])
|
||||
result = reduce(lambda l, x: l + find_first_level_groups(x, '{}'), result, [])
|
||||
# do not do this at this moment, it is not strong enough and can break other
|
||||
# patterns, such as dates, etc...
|
||||
#result = reduce(lambda l, x: l + x.split('-'), result, [])
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def format_guess(guess):
|
||||
"""Format all the found values to their natural type.
|
||||
For instance, a year would be stored as an int value, etc...
|
||||
|
||||
Note that this modifies the dictionary given as input.
|
||||
"""
|
||||
for prop, value in guess.items():
|
||||
if prop in ('season', 'episodeNumber', 'year', 'cdNumber', 'cdNumberTotal'):
|
||||
guess[prop] = int(guess[prop])
|
||||
elif isinstance(value, basestring):
|
||||
if prop in ('edition',):
|
||||
value = clean_string(value)
|
||||
guess[prop] = canonical_form(value)
|
||||
|
||||
return guess
|
||||
|
||||
|
||||
def guess_groups(string, result, filetype):
|
||||
# add sentinels so we can match a separator char at either end of
|
||||
# our groups, even when they are at the beginning or end of the string
|
||||
# we will adjust the span accordingly later
|
||||
#
|
||||
# filetype can either be movie, moviesubtitle, episode, episodesubtitle
|
||||
current = ' ' + string + ' '
|
||||
|
||||
regions = [] # list of (start, end) of matched regions
|
||||
|
||||
def guessed(match_dict, confidence):
|
||||
guess = format_guess(Guess(match_dict, confidence = confidence))
|
||||
result.append(guess)
|
||||
log.debug('Found with confidence %.2f: %s' % (confidence, guess))
|
||||
return guess
|
||||
|
||||
def update_found(string, guess, span, span_adjust = (0,0)):
|
||||
span = (span[0] + span_adjust[0],
|
||||
span[1] + span_adjust[1])
|
||||
regions.append((span, guess))
|
||||
return blank_region(string, span)
|
||||
|
||||
# try to find dates first, as they are very specific
|
||||
date, span = search_date(current)
|
||||
if date:
|
||||
guess = guessed({ 'date': date }, confidence = 1.0)
|
||||
current = update_found(current, guess, span)
|
||||
|
||||
# for non episodes only, look for year information
|
||||
if filetype not in ('episode', 'episodesubtitle'):
|
||||
year, span = search_year(current)
|
||||
if year:
|
||||
guess = guessed({ 'year': year }, confidence = 1.0)
|
||||
current = update_found(current, guess, span)
|
||||
|
||||
# specific regexps (ie: cd number, season X episode, ...)
|
||||
for rexp, confidence, span_adjust in video_rexps:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
# is this the better place to put it? (maybe, as it is at least the soonest that we can catch it)
|
||||
if 'cdNumberTotal' in metadata and metadata['cdNumberTotal'] is None:
|
||||
del metadata['cdNumberTotal']
|
||||
|
||||
guess = guessed(metadata, confidence = confidence)
|
||||
current = update_found(current, guess, match.span(), span_adjust)
|
||||
|
||||
if filetype in ('episode', 'episodesubtitle'):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
guess = guessed(metadata, confidence = confidence)
|
||||
current = update_found(current, guess, match.span(), span_adjust)
|
||||
|
||||
|
||||
# Now websites, but as exact string instead of regexps
|
||||
clow = current.lower()
|
||||
for site in websites:
|
||||
pos = clow.find(site.lower())
|
||||
if pos != -1:
|
||||
guess = guessed({ 'website': site }, confidence = confidence)
|
||||
current = update_found(current, guess, (pos, pos+len(site)))
|
||||
clow = current.lower()
|
||||
|
||||
|
||||
# release groups have certain constraints, cannot be included in the previous general regexps
|
||||
group_names = [ r'\.(Xvid)-(?P<releaseGroup>.*?)[ \.]',
|
||||
r'\.(DivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
r'\.(DVDivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
]
|
||||
for rexp in group_names:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
metadata.update({ 'videoCodec': match.group(1) })
|
||||
guess = guessed(metadata, confidence = 0.8)
|
||||
current = update_found(current, guess, match.span(), span_adjust = (1, -1))
|
||||
|
||||
|
||||
# common well-defined words and regexps
|
||||
confidence = 1.0 # for all of them
|
||||
for prop, value, pos, end in find_properties(current):
|
||||
guess = guessed({ prop: value }, confidence = confidence)
|
||||
current = update_found(current, guess, (pos, end))
|
||||
|
||||
|
||||
# weak guesses for episode number, only run it if we don't have an estimate already
|
||||
if filetype in ('episode', 'episodesubtitle'):
|
||||
if not any('episodeNumber' in match for match in result):
|
||||
for rexp, _, span_adjust in weak_episode_rexps:
|
||||
match = re.search(rexp, current, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
epnum = int(metadata['episodeNumber'])
|
||||
if epnum > 100:
|
||||
guess = guessed({ 'season': epnum // 100,
|
||||
'episodeNumber': epnum % 100 }, confidence = 0.6)
|
||||
else:
|
||||
guess = guessed(metadata, confidence = 0.3)
|
||||
current = update_found(current, guess, match.span(), span_adjust)
|
||||
|
||||
# try to find languages now
|
||||
language, span, confidence = search_language(current)
|
||||
while language:
|
||||
# is it a subtitle language?
|
||||
if 'sub' in clean_string(current[:span[0]]).lower().split(' '):
|
||||
guess = guessed({ 'subtitleLanguage': language }, confidence = confidence)
|
||||
else:
|
||||
guess = guessed({ 'language': language }, confidence = confidence)
|
||||
current = update_found(current, guess, span)
|
||||
|
||||
language, span, confidence = search_language(current)
|
||||
|
||||
|
||||
# remove our sentinels now and ajust spans accordingly
|
||||
assert(current[0] == ' ' and current[-1] == ' ')
|
||||
current = current[1:-1]
|
||||
regions = [ ((start-1, end-1), guess) for (start, end), guess in regions ]
|
||||
|
||||
# split into '-' separated subgroups (with required separator chars
|
||||
# around the dash)
|
||||
didx = current.find('-')
|
||||
while didx > 0:
|
||||
regions.append(((didx, didx), None))
|
||||
didx = current.find('-', didx+1)
|
||||
|
||||
# cut our final groups, and rematch the guesses to the group that created
|
||||
# id, None if it is a leftover group
|
||||
region_spans = [ span for span, guess in regions ]
|
||||
string_groups = split_on_groups(string, region_spans)
|
||||
remaining_groups = split_on_groups(current, region_spans)
|
||||
guesses = []
|
||||
|
||||
pos = 0
|
||||
for group in string_groups:
|
||||
found = False
|
||||
for span, guess in regions:
|
||||
if span[0] == pos:
|
||||
guesses.append(guess)
|
||||
found = True
|
||||
if not found:
|
||||
guesses.append(None)
|
||||
|
||||
pos += len(group)
|
||||
|
||||
return zip(string_groups,
|
||||
remaining_groups,
|
||||
guesses)
|
||||
|
||||
|
||||
def match_from_epnum_position(match_tree, epnum_pos, guessed, update_found):
|
||||
"""guessed is a callback function to call with the guessed group
|
||||
update_found is a callback to update the match group and returns leftover groups."""
|
||||
pidx, eidx, gidx = epnum_pos
|
||||
|
||||
# a few helper functions to be able to filter using high-level semantics
|
||||
def same_pgroup_before(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and (eeidx, ggidx) < (eidx, gidx)
|
||||
|
||||
def same_pgroup_after(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and (eeidx, ggidx) > (eidx, gidx)
|
||||
|
||||
def same_egroup_before(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and eeidx == eidx and ggidx < gidx
|
||||
|
||||
def same_egroup_after(group):
|
||||
_, (ppidx, eeidx, ggidx) = group
|
||||
return ppidx == pidx and eeidx == eidx and ggidx > gidx
|
||||
|
||||
leftover = leftover_valid_groups(match_tree)
|
||||
|
||||
# if we have at least 1 valid group before the episodeNumber, then it's probably
|
||||
# the series name
|
||||
series_candidates = filter(same_pgroup_before, leftover)
|
||||
if len(series_candidates) >= 1:
|
||||
guess = guessed({ 'series': series_candidates[0][0] }, confidence = 0.7)
|
||||
leftover = update_found(leftover, series_candidates[0][1], guess)
|
||||
|
||||
# only 1 group after (in the same path group) and it's probably the episode title
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(same_pgroup_after, leftover))
|
||||
if len(title_candidates) == 1:
|
||||
guess = guessed({ 'title': title_candidates[0][0] }, confidence = 0.5)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
else:
|
||||
# try in the same explicit group, with lower confidence
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(same_egroup_after, leftover))
|
||||
if len(title_candidates) == 1:
|
||||
guess = guessed({ 'title': title_candidates[0][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
|
||||
# epnumber is the first group and there are only 2 after it in same path group
|
||||
# -> season title - episode title
|
||||
already_has_title = (find_group(match_tree, 'title') != [])
|
||||
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(same_pgroup_after, leftover))
|
||||
if (not already_has_title and # no title
|
||||
not filter(same_pgroup_before, leftover) and # no groups before
|
||||
len(title_candidates) == 2): # only 2 groups after
|
||||
|
||||
guess = guessed({ 'series': title_candidates[0][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
guess = guessed({ 'title': title_candidates[1][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[1][1], guess)
|
||||
|
||||
|
||||
# if we only have 1 remaining valid group in the pathpart before the filename,
|
||||
# then it's likely that it is the series name
|
||||
series_candidates = [ group for group in leftover if group[1][0] == pidx-1 ]
|
||||
if len(series_candidates) == 1:
|
||||
guess = guessed({ 'series': series_candidates[0][0] }, confidence = 0.5)
|
||||
leftover = update_found(leftover, series_candidates[0][1], guess)
|
||||
|
||||
return match_tree
|
||||
|
||||
|
||||
|
||||
class IterativeMatcher(object):
|
||||
def __init__(self, filename, filetype = 'autodetect'):
|
||||
def __init__(self, filename, filetype='autodetect'):
|
||||
"""An iterative matcher tries to match different patterns that appear
|
||||
in the filename.
|
||||
|
||||
@@ -325,7 +60,7 @@ class IterativeMatcher(object):
|
||||
The first 3 lines indicates the group index in which a char in the
|
||||
filename is located. So for instance, x264 is the group (0, 4, 1), and
|
||||
it corresponds to a video codec, denoted by the letter'v' in the 4th line.
|
||||
(for more info, see guess.matchtree.tree_to_string)
|
||||
(for more info, see guess.matchtree.to_string)
|
||||
|
||||
|
||||
Second, it tries to merge all this information into a single object
|
||||
@@ -333,266 +68,89 @@ class IterativeMatcher(object):
|
||||
resolution when they arise.
|
||||
"""
|
||||
|
||||
if filetype not in ('autodetect', 'subtitle', 'video',
|
||||
valid_filetypes = ('autodetect', 'subtitle', 'video',
|
||||
'movie', 'moviesubtitle',
|
||||
'episode', 'episodesubtitle'):
|
||||
raise ValueError, "filetype needs to be one of ('autodetect', 'subtitle', 'video', 'movie', 'moviesubtitle', 'episode', 'episodesubtitle')"
|
||||
'episode', 'episodesubtitle')
|
||||
if filetype not in valid_filetypes:
|
||||
raise ValueError("filetype needs to be one of %s" % valid_filetypes)
|
||||
if not isinstance(filename, unicode):
|
||||
log.debug('WARNING: given filename to matcher is not unicode...')
|
||||
|
||||
match_tree = []
|
||||
result = [] # list of found metadata
|
||||
|
||||
def guessed(match_dict, confidence):
|
||||
guess = format_guess(Guess(match_dict, confidence = confidence))
|
||||
result.append(guess)
|
||||
log.debug('Found with confidence %.2f: %s' % (confidence, guess))
|
||||
return guess
|
||||
|
||||
def update_found(leftover, group_pos, guess):
|
||||
pidx, eidx, gidx = group_pos
|
||||
group = match_tree[pidx][eidx][gidx]
|
||||
match_tree[pidx][eidx][gidx] = (group[0],
|
||||
deleted * len(group[0]),
|
||||
guess)
|
||||
return [ g for g in leftover if g[1] != group_pos ]
|
||||
self.match_tree = MatchTree(filename)
|
||||
mtree = self.match_tree
|
||||
mtree.guess.set('type', filetype, confidence=1.0)
|
||||
|
||||
def apply_transfo(transfo_name, *args, **kwargs):
|
||||
transfo = __import__('guessit.transfo.' + transfo_name,
|
||||
globals=globals(), locals=locals(),
|
||||
fromlist=['process'], level=-1)
|
||||
transfo.process(mtree, *args, **kwargs)
|
||||
|
||||
# 1- first split our path into dirs + basename + ext
|
||||
match_tree = split_path_components(filename)
|
||||
apply_transfo('split_path_components')
|
||||
|
||||
# try to detect the file type
|
||||
filetype, other = guess_filetype(filename, filetype)
|
||||
guessed({ 'type': filetype }, confidence = 1.0)
|
||||
extguess = guessed(other, confidence = 1.0)
|
||||
# 2- guess the file type now (will be useful later)
|
||||
apply_transfo('guess_filetype', filetype)
|
||||
if mtree.guess['type'] == 'unknown':
|
||||
return
|
||||
|
||||
# guess the mimetype of the filename
|
||||
# TODO: handle other mimetypes not found on the default type_maps
|
||||
# mimetypes.types_map['.srt']='text/subtitle'
|
||||
mime, _ = mimetypes.guess_type(filename, strict=False)
|
||||
if mime is not None:
|
||||
guessed({ 'mimetype': mime }, confidence = 1.0)
|
||||
# 3- split each of those into explicit groups (separated by parentheses
|
||||
# or square brackets)
|
||||
apply_transfo('split_explicit_groups')
|
||||
|
||||
# remove the extension from the match tree, as all indices relative
|
||||
# the the filename groups assume the basename is the last one
|
||||
fileext = match_tree.pop(-1)[1:].lower()
|
||||
# 4- try to match information for specific patterns
|
||||
if mtree.guess['type'] in ('episode', 'episodesubtitle'):
|
||||
strategy = ['guess_date', 'guess_video_rexps',
|
||||
'guess_episodes_rexps', 'guess_website',
|
||||
'guess_release_group', 'guess_properties',
|
||||
'guess_weak_episodes_rexps', 'guess_language']
|
||||
else:
|
||||
strategy = ['guess_date', 'guess_year', 'guess_video_rexps',
|
||||
'guess_website', 'guess_release_group',
|
||||
'guess_properties', 'guess_language']
|
||||
|
||||
for name in strategy:
|
||||
apply_transfo(name)
|
||||
|
||||
# 2- split each of those into explicit groups, if any
|
||||
# note: be careful, as this might split some regexps with more confidence such as
|
||||
# Alfleni-Team, or [XCT] or split a date such as (14-01-2008)
|
||||
match_tree = [ split_explicit_groups(part) for part in match_tree ]
|
||||
# more guessers for both movies and episodes
|
||||
for name in ['guess_bonus_features']:
|
||||
apply_transfo(name)
|
||||
|
||||
# split into '-' separated subgroups (with required separator chars
|
||||
# around the dash)
|
||||
apply_transfo('split_on_dash')
|
||||
|
||||
# 3- try to match information in decreasing order of confidence and
|
||||
# blank the matching group in the string if we found something
|
||||
for pathpart in match_tree:
|
||||
for gidx, explicit_group in enumerate(pathpart):
|
||||
pathpart[gidx] = guess_groups(explicit_group, result, filetype = filetype)
|
||||
# 5- try to identify the remaining unknown groups by looking at their
|
||||
# position relative to other known elements
|
||||
if mtree.guess['type'] in ('episode', 'episodesubtitle'):
|
||||
apply_transfo('guess_episode_info_from_position')
|
||||
else:
|
||||
apply_transfo('guess_movie_title_from_position')
|
||||
|
||||
# 4- try to identify the remaining unknown groups by looking at their position
|
||||
# relative to other known elements
|
||||
|
||||
if filetype in ('episode', 'episodesubtitle'):
|
||||
eps = find_group(match_tree, 'episodeNumber')
|
||||
if eps:
|
||||
match_tree = match_from_epnum_position(match_tree, eps[0], guessed, update_found)
|
||||
|
||||
leftover = leftover_valid_groups(match_tree)
|
||||
|
||||
if not eps:
|
||||
# if we don't have the episode number, but at least 2 groups in the
|
||||
# last path group, then it's probably series - eptitle
|
||||
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
|
||||
filter(lambda g: g[1][0] == len(match_tree)-1,
|
||||
leftover_valid_groups(match_tree)))
|
||||
if len(title_candidates) >= 2:
|
||||
guess = guessed({ 'series': title_candidates[0][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[0][1], guess)
|
||||
guess = guessed({ 'title': title_candidates[1][0] }, confidence = 0.4)
|
||||
leftover = update_found(leftover, title_candidates[1][1], guess)
|
||||
|
||||
|
||||
# if there's a path group that only contains the season info, then the previous one
|
||||
# is most likely the series title (ie: .../series/season X/...)
|
||||
eps = [ gpos for gpos in find_group(match_tree, 'season')
|
||||
if 'episodeNumber' not in get_group(match_tree, gpos)[2] ]
|
||||
|
||||
if eps:
|
||||
pidx, eidx, gidx = eps[0]
|
||||
previous = [ group for group in leftover if group[1][0] == pidx - 1 ]
|
||||
if len(previous) == 1:
|
||||
guess = guessed({ 'series': previous[0][0] }, confidence = 0.5)
|
||||
leftover = update_found(leftover, previous[0][1], guess)
|
||||
|
||||
# reduce the confidence of unlikely series
|
||||
for guess in result:
|
||||
if 'series' in guess:
|
||||
if guess['series'].lower() in unlikely_series:
|
||||
guess.set_confidence('series', guess.confidence('series') * 0.5)
|
||||
|
||||
|
||||
elif filetype in ('movie', 'moviesubtitle'):
|
||||
leftover_all = leftover_valid_groups(match_tree)
|
||||
|
||||
# specific cases:
|
||||
# - movies/tttttt (yyyy)/tttttt.ccc
|
||||
try:
|
||||
if match_tree[-3][0][0][0].lower() == 'movies':
|
||||
# Note:too generic, might solve all the unittests as they all contain 'movies'
|
||||
# in their path
|
||||
#
|
||||
#if len(match_tree[-2][0]) == 1:
|
||||
# title = match_tree[-2][0][0]
|
||||
# guess = guessed({ 'title': clean_string(title[0]) }, confidence = 0.7)
|
||||
# update_found(leftover_all, title, guess)
|
||||
|
||||
year_group = filter(lambda gpos: gpos[0] == len(match_tree)-2,
|
||||
find_group(match_tree, 'year'))[0]
|
||||
leftover = leftover_valid_groups(match_tree,
|
||||
valid = lambda g: ((g[0] and g[0][0] not in sep) and
|
||||
g[1][0] == len(match_tree) - 2))
|
||||
if len(match_tree[-2]) == 2 and year_group[1] == 1:
|
||||
title = leftover[0]
|
||||
guess = guessed({ 'title': clean_string(title[0]) },
|
||||
confidence = 0.8)
|
||||
update_found(leftover_all, title[1], guess)
|
||||
raise Exception # to exit the try catch now
|
||||
|
||||
leftover = [ g for g in leftover_all if (g[1][0] == year_group[0] and
|
||||
g[1][1] < year_group[1] and
|
||||
g[1][2] < year_group[2]) ]
|
||||
leftover = sorted(leftover, key = lambda x:x[1])
|
||||
title = leftover[0]
|
||||
guess = guessed({ 'title': title[0] }, confidence = 0.8)
|
||||
leftover = update_found(leftover, title[1], guess)
|
||||
except:
|
||||
pass
|
||||
|
||||
# if we have either format or videoCodec in the folder containing the file
|
||||
# or one of its parents, then we should probably look for the title in
|
||||
# there rather than in the basename
|
||||
props = filter(lambda g: g[0] <= len(match_tree) - 2,
|
||||
find_group(match_tree, 'videoCodec') +
|
||||
find_group(match_tree, 'format') +
|
||||
find_group(match_tree, 'language'))
|
||||
leftover = None
|
||||
if props and all(g[0] == props[0][0] for g in props):
|
||||
leftover = [ g for g in leftover_all if g[1][0] == props[0][0] ]
|
||||
|
||||
if props and leftover:
|
||||
guess = guessed({ 'title': leftover[0][0] }, confidence = 0.7)
|
||||
leftover = update_found(leftover, leftover[0][1], guess)
|
||||
|
||||
else:
|
||||
# first leftover group in the last path part sounds like a good candidate for title,
|
||||
# except if it's only one word and that the first group before has at least 3 words in it
|
||||
# (case where the filename contains an 8 chars short name and the movie title is
|
||||
# actually in the parent directory name)
|
||||
leftover = [ g for g in leftover_all if g[1][0] == len(match_tree)-1 ]
|
||||
if leftover:
|
||||
title, (pidx, eidx, gidx) = leftover[0]
|
||||
previous_pgroup_leftover = filter(lambda g: g[1][0] == pidx-1, leftover_all)
|
||||
|
||||
if (title.count(' ') == 0 and
|
||||
previous_pgroup_leftover and
|
||||
previous_pgroup_leftover[0][0].count(' ') >= 2):
|
||||
|
||||
guess = guessed({ 'title': previous_pgroup_leftover[0][0] }, confidence = 0.6)
|
||||
leftover = update_found(leftover, previous_pgroup_leftover[0][1], guess)
|
||||
|
||||
else:
|
||||
guess = guessed({ 'title': title }, confidence = 0.6)
|
||||
leftover = update_found(leftover, leftover[0][1], guess)
|
||||
else:
|
||||
# if there were no leftover groups in the last path part, look in the one before that
|
||||
previous_pgroup_leftover = filter(lambda g: g[1][0] == len(match_tree)-2, leftover_all)
|
||||
if previous_pgroup_leftover:
|
||||
guess = guessed({ 'title': previous_pgroup_leftover[0][0] }, confidence = 0.6)
|
||||
leftover = update_found(leftover, previous_pgroup_leftover[0][1], guess)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
# 5- perform some post-processing steps
|
||||
|
||||
# 5.1- try to promote language to subtitle language where it makes sense
|
||||
for pidx, eidx, gidx in find_group(match_tree, 'language'):
|
||||
string, remaining, guess = get_group(match_tree, (pidx, eidx, gidx))
|
||||
|
||||
def promote_subtitle():
|
||||
guess.set('subtitleLanguage', guess['language'], confidence = guess.confidence('language'))
|
||||
del guess['language']
|
||||
|
||||
# - if we matched a language in a file with a sub extension and that the group
|
||||
# is the last group of the filename, it is probably the language of the subtitle
|
||||
# (eg: 'xxx.english.srt')
|
||||
if (fileext in subtitle_exts and
|
||||
pidx == len(match_tree) - 1 and
|
||||
eidx == len(match_tree[pidx]) - 1):
|
||||
promote_subtitle()
|
||||
|
||||
# - if a language is in an explicit group just preceded by "st", it is a subtitle
|
||||
# language (eg: '...st[fr-eng]...')
|
||||
if eidx > 0:
|
||||
previous = get_group(match_tree, (pidx, eidx-1, -1))
|
||||
if previous[0][-2:].lower() == 'st':
|
||||
promote_subtitle()
|
||||
|
||||
|
||||
|
||||
# re-append the extension now
|
||||
match_tree.append([[(fileext, deleted*len(fileext), extguess)]])
|
||||
|
||||
self.parts = result
|
||||
self.match_tree = match_tree
|
||||
|
||||
if filename.startswith('/'):
|
||||
filename = ' ' + filename
|
||||
|
||||
log.debug('Found match tree:\n%s\n%s' % (to_utf8(tree_to_string(match_tree)),
|
||||
to_utf8(filename)))
|
||||
# 6- perform some post-processing steps
|
||||
apply_transfo('post_process')
|
||||
|
||||
log.debug('Found match tree:\n%s' % (to_utf8(unicode(mtree))))
|
||||
|
||||
def matched(self):
|
||||
# we need to make a copy here, as the merge functions work in place and
|
||||
# calling them on the match tree would modify it
|
||||
parts = copy.deepcopy(self.parts)
|
||||
|
||||
# 1- start by doing some common preprocessing tasks
|
||||
parts = [node.guess for node in self.match_tree.nodes() if node.guess]
|
||||
parts = copy.deepcopy(parts)
|
||||
|
||||
# 1.1- ", the" at the end of a series title should be prepended to it
|
||||
for part in parts:
|
||||
if 'series' not in part:
|
||||
continue
|
||||
|
||||
series = part['series']
|
||||
lseries = series.lower()
|
||||
|
||||
if lseries[-4:] == ',the':
|
||||
part['series'] = 'The ' + series[:-4]
|
||||
|
||||
if lseries[-5:] == ', the':
|
||||
part['series'] = 'The ' + series[:-5]
|
||||
|
||||
|
||||
# 2- try to merge similar information together and give it a higher confidence
|
||||
# 1- try to merge similar information together and give it a higher
|
||||
# confidence
|
||||
for int_part in ('year', 'season', 'episodeNumber'):
|
||||
merge_similar_guesses(parts, int_part, choose_int)
|
||||
|
||||
for string_part in ('title', 'series', 'container', 'format', 'releaseGroup', 'website',
|
||||
'audioCodec', 'videoCodec', 'screenSize', 'episodeFormat'):
|
||||
for string_part in ('title', 'series', 'container', 'format',
|
||||
'releaseGroup', 'website', 'audioCodec',
|
||||
'videoCodec', 'screenSize', 'episodeFormat'):
|
||||
merge_similar_guesses(parts, string_part, choose_string)
|
||||
|
||||
result = merge_all(parts, append = ['language', 'subtitleLanguage', 'other'])
|
||||
|
||||
# 3- some last minute post-processing
|
||||
if (result['type'] == 'episode' and
|
||||
'season' not in result and
|
||||
result.get('episodeFormat', '') == 'Minisode'):
|
||||
result['season'] = 0
|
||||
result = merge_all(parts,
|
||||
append=['language', 'subtitleLanguage', 'other'])
|
||||
|
||||
log.debug('Final result: ' + result.nice_string())
|
||||
return result
|
||||
|
||||
+206
-99
@@ -18,136 +18,243 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import deleted
|
||||
from guessit.textutils import clean_string
|
||||
from guessit import Guess
|
||||
from guessit.textutils import clean_string, str_fill, to_utf8
|
||||
from guessit.patterns import group_delimiters
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.matchtree")
|
||||
|
||||
|
||||
class BaseMatchTree(object):
|
||||
"""A MatchTree represents the hierarchical split of a string into its
|
||||
constituent semantic groups."""
|
||||
|
||||
def tree_to_string(tree):
|
||||
"""Return a string representation for the given tree.
|
||||
def __init__(self, string='', span=None, parent=None):
|
||||
self.string = string
|
||||
self.span = span or (0, len(string))
|
||||
self.parent = parent
|
||||
self.children = []
|
||||
self.guess = Guess()
|
||||
|
||||
The lines convey the following information:
|
||||
- line 1: path idx
|
||||
- line 2: explicit group idx
|
||||
- line 3: group index
|
||||
- line 4: remaining info
|
||||
- line 5: meaning conveyed
|
||||
@property
|
||||
def value(self):
|
||||
return self.string[self.span[0]:self.span[1]]
|
||||
|
||||
Meaning is a letter indicating what type of info was matched by this group,
|
||||
for instance 't' = title, 'f' = format, 'l' = language, etc...
|
||||
@property
|
||||
def clean_value(self):
|
||||
return clean_string(self.value)
|
||||
|
||||
An example is the following:
|
||||
@property
|
||||
def offset(self):
|
||||
return self.span[0]
|
||||
|
||||
0000000000000000000000000000000000000000000000000000000000000000000000000000000000 111
|
||||
0000011111111111112222222222222233333333444444444444444455555555666777777778888888 000
|
||||
0000000000000000000000000000000001111112011112222333333401123334000011233340000000 000
|
||||
__________________(The.Prestige).______.[____.HP.______.{__-___}.St{__-___}.Chaps].___
|
||||
xxxxxttttttttttttt ffffff vvvv xxxxxx ll lll xx xxx ccc
|
||||
[XCT].Le.Prestige.(The.Prestige).DVDRip.[x264.HP.He-Aac.{Fr-Eng}.St{Fr-Eng}.Chaps].mkv
|
||||
@property
|
||||
def info(self):
|
||||
result = dict(self.guess)
|
||||
|
||||
(note: the last line representing the filename is not pat of the tree representation)
|
||||
"""
|
||||
m_tree = [ '', # path level index
|
||||
'', # explicit group index
|
||||
'', # matched regexp and dash-separated
|
||||
'', # groups leftover that couldn't be matched
|
||||
'', # meaning conveyed: E = episodenumber, S = season, ...
|
||||
]
|
||||
for c in self.children:
|
||||
result.update(c.info)
|
||||
|
||||
return result
|
||||
|
||||
@property
|
||||
def root(self):
|
||||
if not self.parent:
|
||||
return self
|
||||
|
||||
return self.parent.root
|
||||
|
||||
@property
|
||||
def depth(self):
|
||||
if self.is_leaf():
|
||||
return 0
|
||||
|
||||
return 1 + max(c.depth for c in self.children)
|
||||
|
||||
def is_leaf(self):
|
||||
return self.children == []
|
||||
|
||||
def add_child(self, span):
|
||||
child = MatchTree(self.string, span=span, parent=self)
|
||||
self.children.append(child)
|
||||
|
||||
def partition(self, indices):
|
||||
indices = sorted(indices)
|
||||
if indices[0] != 0:
|
||||
indices.insert(0, 0)
|
||||
if indices[-1] != len(self.value):
|
||||
indices.append(len(self.value))
|
||||
|
||||
for start, end in zip(indices[:-1], indices[1:]):
|
||||
self.add_child(span=(self.offset + start,
|
||||
self.offset + end))
|
||||
|
||||
def split_on_components(self, components):
|
||||
offset = 0
|
||||
for c in components:
|
||||
start = self.value.find(c, offset)
|
||||
end = start + len(c)
|
||||
self.add_child(span=(self.offset + start,
|
||||
self.offset + end))
|
||||
offset = end
|
||||
|
||||
def nodes_at_depth(self, depth):
|
||||
if depth == 0:
|
||||
yield self
|
||||
|
||||
for child in self.children:
|
||||
for node in child.nodes_at_depth(depth - 1):
|
||||
yield node
|
||||
|
||||
@property
|
||||
def node_idx(self):
|
||||
if self.parent is None:
|
||||
return ()
|
||||
return self.parent.node_idx + (self.parent.children.index(self),)
|
||||
|
||||
def node_at(self, idx):
|
||||
if not idx:
|
||||
return self
|
||||
|
||||
try:
|
||||
return self.children[idx[0]].node_at(idx[1:])
|
||||
except:
|
||||
raise ValueError('Non-existent node index: %s' % (idx,))
|
||||
|
||||
def nodes(self):
|
||||
yield self
|
||||
for child in self.children:
|
||||
for node in child.nodes():
|
||||
yield node
|
||||
|
||||
def _leaves(self):
|
||||
if self.is_leaf():
|
||||
yield self
|
||||
else:
|
||||
for child in self.children:
|
||||
# pylint: disable=W0212
|
||||
for leaf in child._leaves():
|
||||
yield leaf
|
||||
|
||||
def leaves(self):
|
||||
return list(self._leaves())
|
||||
|
||||
def to_string(self):
|
||||
empty_line = ' ' * len(self.string)
|
||||
|
||||
def add_char(pidx, eidx, gidx, remaining, meaning = None):
|
||||
nr = len(remaining)
|
||||
def to_hex(x):
|
||||
if isinstance(x, int):
|
||||
return str(x) if x < 10 else chr(55+x)
|
||||
return str(x) if x < 10 else chr(55 + x)
|
||||
return x
|
||||
m_tree[0] = m_tree[0] + to_hex(pidx) * nr
|
||||
m_tree[1] = m_tree[1] + to_hex(eidx) * nr
|
||||
m_tree[2] = m_tree[2] + to_hex(gidx) * nr
|
||||
m_tree[3] = m_tree[3] + remaining
|
||||
m_tree[4] = m_tree[4] + str(meaning or ' ') * nr
|
||||
|
||||
def meaning(result):
|
||||
mmap = { 'episodeNumber': 'E',
|
||||
'season': 'S',
|
||||
'extension': 'e',
|
||||
'format': 'f',
|
||||
'language': 'l',
|
||||
'videoCodec': 'v',
|
||||
'audioCodec': 'a',
|
||||
'website': 'w',
|
||||
'container': 'c',
|
||||
'series': 'T',
|
||||
'title': 't',
|
||||
'date': 'd',
|
||||
'year': 'y',
|
||||
'releaseGroup': 'r',
|
||||
'screenSize': 's'
|
||||
}
|
||||
def meaning(result):
|
||||
mmap = { 'episodeNumber': 'E',
|
||||
'season': 'S',
|
||||
'extension': 'e',
|
||||
'format': 'f',
|
||||
'language': 'l',
|
||||
'videoCodec': 'v',
|
||||
'audioCodec': 'a',
|
||||
'website': 'w',
|
||||
'container': 'c',
|
||||
'series': 'T',
|
||||
'title': 't',
|
||||
'date': 'd',
|
||||
'year': 'y',
|
||||
'releaseGroup': 'r',
|
||||
'screenSize': 's'
|
||||
}
|
||||
|
||||
if result is None:
|
||||
return ' '
|
||||
if result is None:
|
||||
return ' '
|
||||
|
||||
for prop, l in mmap.items():
|
||||
if prop in result:
|
||||
return l
|
||||
for prop, l in mmap.items():
|
||||
if prop in result:
|
||||
return l
|
||||
|
||||
return 'x'
|
||||
return 'x'
|
||||
|
||||
for pidx, pathpart in enumerate(tree):
|
||||
for eidx, explicit_group in enumerate(pathpart):
|
||||
for gidx, (group, remaining, result) in enumerate(explicit_group):
|
||||
add_char(pidx, eidx, gidx, remaining, meaning(result))
|
||||
lines = [ empty_line ] * (self.depth + 2) # +2: remaining, meaning
|
||||
lines[-2] = self.string
|
||||
|
||||
# special conditions for the path separator
|
||||
if pidx < len(tree) - 2:
|
||||
add_char(' ', ' ', ' ', '/')
|
||||
elif pidx == len(tree) - 2:
|
||||
add_char(' ', ' ', ' ', '.')
|
||||
for node in self.nodes():
|
||||
if node == self:
|
||||
continue
|
||||
|
||||
return '\n'.join(m_tree)
|
||||
idx = node.node_idx
|
||||
depth = len(idx) - 1
|
||||
if idx:
|
||||
lines[depth] = str_fill(lines[depth], node.span,
|
||||
to_hex(idx[-1]))
|
||||
if node.guess:
|
||||
lines[-2] = str_fill(lines[-2], node.span, '_')
|
||||
lines[-1] = str_fill(lines[-1], node.span, meaning(node.guess))
|
||||
|
||||
lines.append(self.string)
|
||||
|
||||
return '\n'.join(lines)
|
||||
|
||||
def __unicode__(self):
|
||||
return self.to_string()
|
||||
|
||||
def __str__(self):
|
||||
return to_utf8(unicode(self))
|
||||
|
||||
|
||||
class MatchTree(BaseMatchTree):
|
||||
"""The MatchTree contains a few "utility" methods which are not necessary
|
||||
for the BaseMatchTree, but add a lot of convenience for writing
|
||||
higher-level rules."""
|
||||
|
||||
def iterate_groups(match_tree):
|
||||
"""Iterate over all the groups in a match_tree and return them as pairs
|
||||
of (group_pos, group) where:
|
||||
- group_pos = (pidx, eidx, gidx)
|
||||
- group = (string, remaining, guess)
|
||||
"""
|
||||
for pidx, pathpart in enumerate(match_tree):
|
||||
for eidx, explicit_group in enumerate(pathpart):
|
||||
for gidx, group in enumerate(explicit_group):
|
||||
yield (pidx, eidx, gidx), group
|
||||
def _unidentified_leaves(self,
|
||||
valid=lambda leaf: len(leaf.clean_value) >= 2):
|
||||
for leaf in self._leaves():
|
||||
if not leaf.guess and valid(leaf):
|
||||
yield leaf
|
||||
|
||||
def unidentified_leaves(self,
|
||||
valid=lambda leaf: len(leaf.clean_value) >= 2):
|
||||
return list(self._unidentified_leaves(valid))
|
||||
|
||||
def find_group(match_tree, prop):
|
||||
"""Find the list of groups that resulted in a guess that contains the
|
||||
asked property."""
|
||||
result = []
|
||||
for gpos, (string, remaining, guess) in iterate_groups(match_tree):
|
||||
if guess and prop in guess:
|
||||
result.append(gpos)
|
||||
return result
|
||||
def _leaves_containing(self, property_name):
|
||||
if isinstance(property_name, basestring):
|
||||
property_name = [ property_name ]
|
||||
|
||||
def get_group(match_tree, gpos):
|
||||
pidx, eidx, gidx = gpos
|
||||
return match_tree[pidx][eidx][gidx]
|
||||
for leaf in self._leaves():
|
||||
for prop in property_name:
|
||||
if prop in leaf.guess:
|
||||
yield leaf
|
||||
break
|
||||
|
||||
def leaves_containing(self, property_name):
|
||||
return list(self._leaves_containing(property_name))
|
||||
|
||||
def leftover_valid_groups(match_tree, valid = lambda s: len(s[0]) > 3):
|
||||
"""Return the list of valid string groups (eg: len(s) > 3) that could not be
|
||||
matched to anything as a list of pairs (cleaned_str, group_pos)."""
|
||||
leftover = []
|
||||
for gpos, (group, remaining, guess) in iterate_groups(match_tree):
|
||||
if not guess:
|
||||
clean_str = clean_string(remaining)
|
||||
if valid((clean_str, gpos)):
|
||||
leftover.append((clean_str, gpos))
|
||||
def first_leaf_containing(self, property_name):
|
||||
try:
|
||||
return next(self._leaves_containing(property_name))
|
||||
except StopIteration:
|
||||
return None
|
||||
|
||||
return leftover
|
||||
def _previous_unidentified_leaves(self, node):
|
||||
node_idx = node.node_idx
|
||||
for leaf in self._unidentified_leaves():
|
||||
if leaf.node_idx < node_idx:
|
||||
yield leaf
|
||||
|
||||
def previous_unidentified_leaves(self, node):
|
||||
return list(self._previous_unidentified_leaves(node))
|
||||
|
||||
def _previous_leaves_containing(self, node, property_name):
|
||||
node_idx = node.node_idx
|
||||
for leaf in self._leaves_containing(property_name):
|
||||
if leaf.node_idx < node_idx:
|
||||
yield leaf
|
||||
|
||||
def previous_leaves_containing(self, node, property_name):
|
||||
return list(self._previous_leaves_containing(node, property_name))
|
||||
|
||||
def is_explicit(self):
|
||||
"""Return whether the group was explicitly enclosed by
|
||||
parentheses/square brackets/etc."""
|
||||
return (self.value[0] + self.value[-1]) in group_delimiters
|
||||
|
||||
Regular → Executable
+45
-23
@@ -22,10 +22,13 @@
|
||||
|
||||
subtitle_exts = [ 'srt', 'idx', 'sub', 'ssa', 'txt' ]
|
||||
|
||||
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'm4v', 'mov', 'ogg', 'ogm', 'ogv', 'wmv', 'divx' ]
|
||||
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'm4v', 'mov', 'ogg', 'ogm', 'ogv',
|
||||
'wmv', 'divx' ]
|
||||
|
||||
group_delimiters = [ '()', '[]', '{}' ]
|
||||
|
||||
# separator character regexp
|
||||
sep = r'[][)(}{+ \._-]' # regexp art, hehe :D
|
||||
sep = r'[][)(}{+ /\._-]' # regexp art, hehe :D
|
||||
|
||||
# character used to represent a deleted char (when matching groups)
|
||||
deleted = '_'
|
||||
@@ -35,28 +38,33 @@ episode_rexps = [ # ... Season 2 ...
|
||||
(r'season (?P<season>[0-9]+)', 1.0, (0, 0)),
|
||||
(r'saison (?P<season>[0-9]+)', 1.0, (0, 0)),
|
||||
|
||||
# ... s02-x01 ...
|
||||
(r's(?P<season>[0-9]{1,2})-x(?P<bonusNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
|
||||
|
||||
# ... s02e13 ...
|
||||
(r'[Ss](?P<season>[0-9]{1,2}).{,3}[EeXx](?P<episodeNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
|
||||
|
||||
# ... 2x13 ...
|
||||
(r'[^0-9](?P<season>[0-9]{1,2})[x\.](?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
|
||||
(r'[^0-9](?P<season>[0-9]{1,2})x(?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
|
||||
|
||||
# ... s02 ...
|
||||
(sep + r's(?P<season>[0-9]{1,2})' + sep + '?', 0.6, (1, -1)),
|
||||
#(sep + r's(?P<season>[0-9]{1,2})' + sep, 0.6, (1, -1)),
|
||||
(r's(?P<season>[0-9]{1,2})[^0-9]', 0.6, (0, -1)),
|
||||
|
||||
# v2 or v3 for some mangas which have multiples rips
|
||||
(sep + r'(?P<episodeNumber>[0-9]{1,3})v[23]' + sep, 0.6, (0, 0)),
|
||||
(r'(?P<episodeNumber>[0-9]{1,3})v[23]' + sep, 0.6, (0, 0)),
|
||||
|
||||
# ... ep 23 ...
|
||||
('ep' + sep + r'(?P<episodeNumber>[0-9]{1,2})[^0-9]', 0.7, (0, -1))
|
||||
]
|
||||
|
||||
|
||||
weak_episode_rexps = [ # ... 213 or 0106 ...
|
||||
(sep + r'(?P<episodeNumber>[0-9]{1,4})' + sep, 0.3, (1, -1)),
|
||||
(sep + r'(?P<episodeNumber>[0-9]{1,4})' + sep, (1, -1)),
|
||||
|
||||
# ... 2x13 ...
|
||||
(sep + r'[^0-9](?P<season>[0-9]{1,2})\.(?P<episodeNumber>[0-9]{2})[^0-9]' + sep, (1, -1)),
|
||||
|
||||
]
|
||||
|
||||
non_episode_title = [ 'extras' ]
|
||||
non_episode_title = [ 'extras', 'rip' ]
|
||||
|
||||
|
||||
video_rexps = [ # cd number
|
||||
@@ -76,7 +84,13 @@ video_rexps = [ # cd number
|
||||
(r'(?P<width>[0-9]{3,4})x(?P<height>[0-9]{3,4})', 0.9, (0, 0)),
|
||||
|
||||
# website
|
||||
(r'(?P<website>www(\.[a-zA-Z0-9]+){2,3})', 0.8, (0, 0))
|
||||
(r'(?P<website>www(\.[a-zA-Z0-9]+){2,3})', 0.8, (0, 0)),
|
||||
|
||||
# bonusNumber: ... x01 ...
|
||||
(r'x(?P<bonusNumber>[0-9]{1,2})', 1.0, (0, 0)),
|
||||
|
||||
# filmNumber: ... f01 ...
|
||||
(r'f(?P<filmNumber>[0-9]{1,2})', 1.0, (0, 0))
|
||||
]
|
||||
|
||||
websites = [ 'tvu.org.ru', 'emule-island.com', 'UsaBit.com', 'www.divx-overnet.com', 'sharethefiles.com' ]
|
||||
@@ -87,7 +101,7 @@ properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'B
|
||||
'HDRip', 'DVD', 'DVDivX', 'HDTV', 'DVB', 'DVBRip', 'PDTV', 'WEBRip',
|
||||
'DVDSCR', 'Screener', 'VHS', 'VIDEO_TS' ],
|
||||
|
||||
'screenSize': [ '720p', '720' ],
|
||||
'screenSize': [ '720p', '720', '1080p', '1080' ],
|
||||
|
||||
'videoCodec': [ 'XviD', 'DivX', 'x264', 'h264', 'Rv10' ],
|
||||
|
||||
@@ -98,18 +112,18 @@ properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'B
|
||||
'releaseGroup': [ 'ESiR', 'WAF', 'SEPTiC', '[XCT]', 'iNT', 'PUKKA',
|
||||
'CHD', 'ViTE', 'TLF', 'DEiTY', 'FLAiTE',
|
||||
'MDX', 'GM4F', 'DVL', 'SVD', 'iLUMiNADOS', ' FiNaLe',
|
||||
'UnSeeN', 'aXXo', 'KLAXXON', 'NoTV', 'ZeaL', 'LOL' ],
|
||||
'UnSeeN', 'aXXo', 'KLAXXON', 'NoTV', 'ZeaL', 'LOL',
|
||||
'HDBRiSe' ],
|
||||
|
||||
'episodeFormat': [ 'Minisode', 'Minisodes' ],
|
||||
|
||||
'other': [ '5ch', 'PROPER', 'REPACK', 'LIMITED', 'DualAudio', 'iNTERNAL', 'Audiofixed', 'R5',
|
||||
'complete', 'classic', # not so sure about these ones, could appear in a title
|
||||
'ws', # widescreen
|
||||
#'SE', # special edition
|
||||
# TODO: director's cut
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def find_properties(filename):
|
||||
result = []
|
||||
clow = filename.lower()
|
||||
@@ -119,7 +133,7 @@ def find_properties(filename):
|
||||
if pos != -1:
|
||||
end = pos + len(value)
|
||||
# make sure our word is always surrounded by separators
|
||||
if ((pos > 0 and clow[pos-1] not in sep) or
|
||||
if ((pos > 0 and clow[pos - 1] not in sep) or
|
||||
(end < len(clow) and clow[end] not in sep)):
|
||||
# note: sep is a regexp, but in this case using it as
|
||||
# a sequence achieves the same goal
|
||||
@@ -137,6 +151,7 @@ property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
|
||||
'DivX': [ 'DVDivX' ],
|
||||
'h264': [ 'x264' ],
|
||||
'720p': [ '720' ],
|
||||
'1080p': [ '1080' ],
|
||||
'AAC': [ 'He-AAC', 'AAC-He' ],
|
||||
'Special Edition': [ 'Special' ],
|
||||
'Collector Edition': [ 'Collector' ],
|
||||
@@ -145,14 +160,21 @@ property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
|
||||
}
|
||||
|
||||
|
||||
reverse_synonyms = {}
|
||||
for prop, values in properties.items():
|
||||
for value in values:
|
||||
reverse_synonyms[value.lower()] = value
|
||||
def revert_synonyms():
|
||||
reverse = {}
|
||||
|
||||
for _, values in properties.items():
|
||||
for value in values:
|
||||
reverse[value.lower()] = value
|
||||
|
||||
for canonical, synonyms in property_synonyms.items():
|
||||
for synonym in synonyms:
|
||||
reverse[synonym.lower()] = canonical
|
||||
|
||||
return reverse
|
||||
|
||||
reverse_synonyms = revert_synonyms()
|
||||
|
||||
for canonical, synonyms in property_synonyms.items():
|
||||
for synonym in synonyms:
|
||||
reverse_synonyms[synonym.lower()] = canonical
|
||||
|
||||
def canonical_form(string):
|
||||
return reverse_synonyms.get(string.lower(), string)
|
||||
|
||||
@@ -28,8 +28,8 @@ RED_FONT = "\x1B[0;31m"
|
||||
RESET_FONT = "\x1B[0m"
|
||||
|
||||
|
||||
def setupLogging(colored = True):
|
||||
"""Sets up a nice colored logger as the main application logger (not only smewt itself)."""
|
||||
def setupLogging(colored=True):
|
||||
"""Set up a nice colored logger as the main application logger."""
|
||||
|
||||
class SimpleFormatter(logging.Formatter):
|
||||
def __init__(self):
|
||||
@@ -38,7 +38,9 @@ def setupLogging(colored = True):
|
||||
|
||||
class ColoredFormatter(logging.Formatter):
|
||||
def __init__(self):
|
||||
self.fmt = '%(levelname)-8s ' + BLUE_FONT + '%(module)s:%(funcName)s' + RESET_FONT + ' -- %(message)s'
|
||||
self.fmt = ('%(levelname)-8s ' +
|
||||
BLUE_FONT + '%(name)s:%(funcName)s' +
|
||||
RESET_FONT + ' -- %(message)s')
|
||||
logging.Formatter.__init__(self, self.fmt)
|
||||
|
||||
def format(self, record):
|
||||
@@ -50,11 +52,9 @@ def setupLogging(colored = True):
|
||||
else:
|
||||
return RED_FONT + result
|
||||
|
||||
|
||||
ch = logging.StreamHandler()
|
||||
if colored and sys.platform != 'win32':
|
||||
ch.setFormatter(ColoredFormatter())
|
||||
else:
|
||||
ch.setFormatter(SimpleFormatter())
|
||||
logging.getLogger().addHandler(ch)
|
||||
|
||||
|
||||
+33
-25
@@ -18,47 +18,59 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import sep, deleted
|
||||
from guessit.patterns import sep
|
||||
import copy
|
||||
|
||||
# string-related functions
|
||||
|
||||
|
||||
def strip_brackets(s):
|
||||
if not s:
|
||||
return s
|
||||
if s[0] == '[' and s[-1] == ']': return s[1:-1]
|
||||
if s[0] == '(' and s[-1] == ')': return s[1:-1]
|
||||
if s[0] == '{' and s[-1] == '}': return s[1:-1]
|
||||
|
||||
if ((s[0] == '[' and s[-1] == ']') or
|
||||
(s[0] == '(' and s[-1] == ')') or
|
||||
(s[0] == '{' and s[-1] == '}')):
|
||||
return s[1:-1]
|
||||
|
||||
return s
|
||||
|
||||
|
||||
def clean_string(s):
|
||||
for c in sep:
|
||||
for c in sep[:-2]: # do not remove dashes ('-')
|
||||
s = s.replace(c, ' ')
|
||||
parts = s.split()
|
||||
return ' '.join(p for p in parts if p != '')
|
||||
result = ' '.join(p for p in parts if p != '')
|
||||
|
||||
# now also remove dashes on the outer part of the string
|
||||
while result and result[0] in sep:
|
||||
result = result[1:]
|
||||
while result and result[-1] in sep:
|
||||
result = result[:-1]
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def str_replace(string, pos, c):
|
||||
return string[:pos] + c + string[pos+1:]
|
||||
|
||||
def blank_region(string, region, blank_sep = deleted):
|
||||
|
||||
def str_fill(string, region, c):
|
||||
start, end = region
|
||||
return string[:start] + blank_sep * (end - start) + string[end:]
|
||||
return string[:start] + c * (end - start) + string[end:]
|
||||
|
||||
|
||||
def between(s, left, right):
|
||||
return s.split(left)[1].split(right)[0]
|
||||
|
||||
def to_utf8(o):
|
||||
'''converts all unicode strings found in the given object to utf-8 strings'''
|
||||
"""Convert all unicode strings found in the given object to utf-8
|
||||
strings."""
|
||||
|
||||
if isinstance(o, unicode):
|
||||
return o.encode('utf-8')
|
||||
elif isinstance(o, list):
|
||||
return [ to_utf8(i) for i in o ]
|
||||
elif isinstance(o, dict):
|
||||
result = copy.deepcopy(o) # need to do it like that to handle Guess instances correctly
|
||||
# need to do it like that to handle Guess instances correctly
|
||||
result = copy.deepcopy(o)
|
||||
for key, value in o.items():
|
||||
result[to_utf8(key)] = to_utf8(value)
|
||||
return result
|
||||
@@ -68,8 +80,10 @@ def to_utf8(o):
|
||||
|
||||
|
||||
def levenshtein(a, b):
|
||||
if not a: return len(b)
|
||||
if not b: return len(a)
|
||||
if not a:
|
||||
return len(b)
|
||||
if not b:
|
||||
return len(a)
|
||||
|
||||
m = len(a)
|
||||
n = len(b)
|
||||
@@ -160,14 +174,13 @@ def split_on_groups(string, groups):
|
||||
if boundaries[-1] != len(string):
|
||||
boundaries.append(len(string))
|
||||
|
||||
groups = [ string[start:end] for start, end in zip(boundaries[:-1], boundaries[1:]) ]
|
||||
groups = [ string[start:end] for start, end in zip(boundaries[:-1],
|
||||
boundaries[1:]) ]
|
||||
|
||||
return filter(bool, groups) # return only non-empty groups
|
||||
return [ g for g in groups if g ] # return only non-empty groups
|
||||
|
||||
|
||||
|
||||
|
||||
def find_first_level_groups(string, enclosing, blank_sep = None):
|
||||
def find_first_level_groups(string, enclosing, blank_sep=None):
|
||||
"""Return a list of groups that could be split because of explicit grouping.
|
||||
The groups are delimited by the given enclosing characters.
|
||||
|
||||
@@ -203,8 +216,3 @@ def find_first_level_groups(string, enclosing, blank_sep = None):
|
||||
string = str_replace(string, end-1, blank_sep)
|
||||
|
||||
return split_on_groups(string, groups)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
from guessit.patterns import canonical_form
|
||||
from guessit.textutils import clean_string
|
||||
import logging
|
||||
|
||||
log = logging.getLogger('guessit.transfo')
|
||||
|
||||
|
||||
def found_property(node, name, confidence):
|
||||
node.guess = Guess({name: node.clean_value}, confidence=confidence)
|
||||
log.debug('Found with confidence %.2f: %s' % (confidence, node.guess))
|
||||
|
||||
|
||||
def format_guess(guess):
|
||||
"""Format all the found values to their natural type.
|
||||
For instance, a year would be stored as an int value, etc...
|
||||
|
||||
Note that this modifies the dictionary given as input.
|
||||
"""
|
||||
for prop, value in guess.items():
|
||||
if prop in ('season', 'episodeNumber', 'year', 'cdNumber',
|
||||
'cdNumberTotal', 'bonusNumber', 'filmNumber'):
|
||||
guess[prop] = int(guess[prop])
|
||||
elif isinstance(value, basestring):
|
||||
if prop in ('edition',):
|
||||
value = clean_string(value)
|
||||
guess[prop] = canonical_form(value)
|
||||
|
||||
return guess
|
||||
|
||||
|
||||
def find_and_split_node(node, strategy, logger):
|
||||
string = ' %s ' % node.value # add sentinels
|
||||
for matcher, confidence in strategy:
|
||||
if getattr(matcher, 'use_node', False):
|
||||
result, span = matcher(string, node)
|
||||
else:
|
||||
result, span = matcher(string)
|
||||
|
||||
if result:
|
||||
# readjust span to compensate for sentinels
|
||||
span = (span[0] - 1, span[1] - 1)
|
||||
|
||||
if isinstance(result, Guess):
|
||||
if confidence is None:
|
||||
confidence = result.confidence(result.keys()[0])
|
||||
else:
|
||||
if confidence is None:
|
||||
confidence = 1.0
|
||||
|
||||
guess = format_guess(Guess(result, confidence=confidence))
|
||||
msg = 'Found with confidence %.2f: %s' % (confidence, guess)
|
||||
(logger or log).debug(msg)
|
||||
|
||||
node.partition(span)
|
||||
absolute_span = (span[0] + node.offset, span[1] + node.offset)
|
||||
for child in node.children:
|
||||
if child.span == absolute_span:
|
||||
child.guess = guess
|
||||
else:
|
||||
find_and_split_node(child, strategy, logger)
|
||||
return
|
||||
|
||||
|
||||
class SingleNodeGuesser(object):
|
||||
def __init__(self, guess_func, confidence, logger=None):
|
||||
self.guess_func = guess_func
|
||||
self.confidence = confidence
|
||||
self.logger = logger
|
||||
|
||||
def process(self, mtree):
|
||||
# strategy is a list of pairs (guesser, confidence)
|
||||
# - if the guesser returns a guessit.Guess and confidence is specified,
|
||||
# it will override it, otherwise it will leave the guess confidence
|
||||
# - if the guesser returns a simple dict as a guess and confidence is
|
||||
# specified, it will use it, or 1.0 otherwise
|
||||
strategy = [ (self.guess_func, self.confidence) ]
|
||||
|
||||
for node in mtree.unidentified_leaves():
|
||||
find_and_split_node(node, strategy, self.logger)
|
||||
@@ -0,0 +1,60 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.transfo import found_property
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_bonus_features")
|
||||
|
||||
|
||||
def process(mtree):
|
||||
def previous_group(g):
|
||||
for leaf in mtree.unidentified_leaves()[::-1]:
|
||||
if leaf.node_idx < g.node_idx:
|
||||
return leaf
|
||||
|
||||
def next_group(g):
|
||||
for leaf in mtree.unidentified_leaves():
|
||||
if leaf.node_idx > g.node_idx:
|
||||
return leaf
|
||||
|
||||
def same_group(g1, g2):
|
||||
return g1.node_idx[:2] == g2.node_idx[:2]
|
||||
|
||||
bonus = [ node for node in mtree.leaves() if 'bonusNumber' in node.guess ]
|
||||
if bonus:
|
||||
bonusTitle = next_group(bonus[0])
|
||||
if same_group(bonusTitle, bonus[0]):
|
||||
found_property(bonusTitle, 'bonusTitle', 0.8)
|
||||
|
||||
filmNumber = [ node for node in mtree.leaves()
|
||||
if 'filmNumber' in node.guess ]
|
||||
if filmNumber:
|
||||
filmSeries = previous_group(filmNumber[0])
|
||||
found_property(filmSeries, 'filmSeries', 0.9)
|
||||
|
||||
title = next_group(filmNumber[0])
|
||||
found_property(title, 'title', 0.9)
|
||||
|
||||
season = [ node for node in mtree.leaves() if 'season' in node.guess ]
|
||||
if season and 'bonusNumber' in mtree.info:
|
||||
series = previous_group(season[0])
|
||||
if same_group(series, season[0]):
|
||||
found_property(series, 'series', 0.9)
|
||||
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.date import search_date
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_date")
|
||||
|
||||
|
||||
def guess_date(string):
|
||||
date, span = search_date(string)
|
||||
if date:
|
||||
return { 'date': date }, span
|
||||
else:
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_date, 1.0, log).process(mtree)
|
||||
@@ -0,0 +1,142 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.transfo import found_property
|
||||
from guessit.patterns import non_episode_title, unlikely_series
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_episode_info_from_position")
|
||||
|
||||
|
||||
def match_from_epnum_position(mtree, node):
|
||||
epnum_idx = node.node_idx
|
||||
|
||||
# a few helper functions to be able to filter using high-level semantics
|
||||
def before_epnum_in_same_pathgroup():
|
||||
return [ leaf for leaf in mtree.unidentified_leaves()
|
||||
if (leaf.node_idx[0] == epnum_idx[0] and
|
||||
leaf.node_idx[1:] < epnum_idx[1:]) ]
|
||||
|
||||
def after_epnum_in_same_pathgroup():
|
||||
return [ leaf for leaf in mtree.unidentified_leaves()
|
||||
if (leaf.node_idx[0] == epnum_idx[0] and
|
||||
leaf.node_idx[1:] > epnum_idx[1:]) ]
|
||||
|
||||
def after_epnum_in_same_explicitgroup():
|
||||
return [ leaf for leaf in mtree.unidentified_leaves()
|
||||
if (leaf.node_idx[:2] == epnum_idx[:2] and
|
||||
leaf.node_idx[2:] > epnum_idx[2:]) ]
|
||||
|
||||
# epnumber is the first group and there are only 2 after it in same
|
||||
# path group
|
||||
# -> series title - episode title
|
||||
title_candidates = [ n for n in after_epnum_in_same_pathgroup()
|
||||
if n.clean_value.lower() not in non_episode_title ]
|
||||
if ('title' not in mtree.info and # no title
|
||||
before_epnum_in_same_pathgroup() == [] and # no groups before
|
||||
len(title_candidates) == 2): # only 2 groups after
|
||||
|
||||
found_property(title_candidates[0], 'series', confidence=0.4)
|
||||
found_property(title_candidates[1], 'title', confidence=0.4)
|
||||
return
|
||||
|
||||
# if we have at least 1 valid group before the episodeNumber, then it's
|
||||
# probably the series name
|
||||
series_candidates = before_epnum_in_same_pathgroup()
|
||||
if len(series_candidates) >= 1:
|
||||
found_property(series_candidates[0], 'series', confidence=0.7)
|
||||
|
||||
# only 1 group after (in the same path group) and it's probably the
|
||||
# episode title
|
||||
title_candidates = [ n for n in after_epnum_in_same_pathgroup()
|
||||
if n.clean_value.lower() not in non_episode_title ]
|
||||
|
||||
if len(title_candidates) == 1:
|
||||
found_property(title_candidates[0], 'title', confidence=0.5)
|
||||
return
|
||||
else:
|
||||
# try in the same explicit group, with lower confidence
|
||||
title_candidates = [ n for n in after_epnum_in_same_explicitgroup()
|
||||
if n.clean_value.lower() not in non_episode_title
|
||||
]
|
||||
if len(title_candidates) == 1:
|
||||
found_property(title_candidates[0], 'title', confidence=0.4)
|
||||
return
|
||||
elif len(title_candidates) > 1:
|
||||
found_property(title_candidates[0], 'title', confidence=0.3)
|
||||
return
|
||||
|
||||
# get the one with the longest value
|
||||
title_candidates = [ n for n in after_epnum_in_same_pathgroup()
|
||||
if n.clean_value.lower() not in non_episode_title ]
|
||||
if title_candidates:
|
||||
maxidx = -1
|
||||
maxv = -1
|
||||
for i, c in enumerate(title_candidates):
|
||||
if len(c.clean_value) > maxv:
|
||||
maxidx = i
|
||||
maxv = len(c.clean_value)
|
||||
found_property(title_candidates[maxidx], 'title', confidence=0.3)
|
||||
|
||||
|
||||
def process(mtree):
|
||||
eps = [node for node in mtree.leaves() if 'episodeNumber' in node.guess]
|
||||
if eps:
|
||||
match_from_epnum_position(mtree, eps[0])
|
||||
|
||||
else:
|
||||
# if we don't have the episode number, but at least 2 groups in the
|
||||
# basename, then it's probably series - eptitle
|
||||
basename = mtree.node_at((-2,))
|
||||
title_candidates = [ n for n in basename.unidentified_leaves()
|
||||
if n.clean_value.lower() not in non_episode_title
|
||||
]
|
||||
|
||||
if len(title_candidates) >= 2:
|
||||
found_property(title_candidates[0], 'series', 0.4)
|
||||
found_property(title_candidates[1], 'title', 0.4)
|
||||
|
||||
# if we only have 1 remaining valid group in the folder containing the
|
||||
# file, then it's likely that it is the series name
|
||||
try:
|
||||
series_candidates = mtree.node_at((-3,)).unidentified_leaves()
|
||||
except ValueError:
|
||||
series_candidates = []
|
||||
|
||||
if len(series_candidates) == 1:
|
||||
found_property(series_candidates[0], 'series', 0.3)
|
||||
|
||||
# if there's a path group that only contains the season info, then the
|
||||
# previous one is most likely the series title (ie: ../series/season X/..)
|
||||
eps = [ node for node in mtree.nodes()
|
||||
if 'season' in node.guess and 'episodeNumber' not in node.guess ]
|
||||
|
||||
if eps:
|
||||
previous = [ node for node in mtree.unidentified_leaves()
|
||||
if node.node_idx[0] == eps[0].node_idx[0] - 1 ]
|
||||
if len(previous) == 1:
|
||||
found_property(previous[0], 'series', 0.5)
|
||||
|
||||
# reduce the confidence of unlikely series
|
||||
for node in mtree.nodes():
|
||||
if 'series' in node.guess:
|
||||
if node.guess['series'].lower() in unlikely_series:
|
||||
new_confidence = node.guess.confidence('series') * 0.5
|
||||
node.guess.set_confidence('series', new_confidence)
|
||||
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.patterns import episode_rexps
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_episodes_rexps")
|
||||
|
||||
|
||||
def guess_episodes_rexps(string):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, string, re.IGNORECASE)
|
||||
if match:
|
||||
return (Guess(match.groupdict(), confidence=confidence),
|
||||
(match.start() + span_adjust[0],
|
||||
match.end() + span_adjust[1]))
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_episodes_rexps, None, log).process(mtree)
|
||||
@@ -0,0 +1,113 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
from guessit.patterns import (subtitle_exts, video_exts, episode_rexps,
|
||||
find_properties, canonical_form)
|
||||
import os.path
|
||||
import re
|
||||
import mimetypes
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_filetype")
|
||||
|
||||
|
||||
def guess_filetype(filename, filetype):
|
||||
other = {}
|
||||
|
||||
# look at the extension first
|
||||
fileext = os.path.splitext(filename)[1][1:].lower()
|
||||
if fileext in subtitle_exts:
|
||||
if 'movie' in filetype:
|
||||
filetype = 'moviesubtitle'
|
||||
elif 'episode' in filetype:
|
||||
filetype = 'episodesubtitle'
|
||||
else:
|
||||
filetype = 'subtitle'
|
||||
other = { 'container': fileext }
|
||||
elif fileext in video_exts:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'video'
|
||||
other = { 'container': fileext }
|
||||
else:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'unknown'
|
||||
other = { 'extension': fileext }
|
||||
|
||||
# put the filetype inside a dummy container to be able to have the
|
||||
# following functions work correctly as closures
|
||||
# this is a workaround for python 2 which doesn't have the
|
||||
# 'nonlocal' keyword (python 3 does have it)
|
||||
filetype_container = [filetype]
|
||||
|
||||
def upgrade_episode():
|
||||
if filetype_container[0] == 'video':
|
||||
filetype_container[0] = 'episode'
|
||||
elif filetype_container[0] == 'subtitle':
|
||||
filetype_container[0] = 'episodesubtitle'
|
||||
|
||||
def upgrade_movie():
|
||||
if filetype_container[0] == 'video':
|
||||
filetype_container[0] = 'movie'
|
||||
elif filetype_container[0] == 'subtitle':
|
||||
filetype_container[0] = 'moviesubtitle'
|
||||
|
||||
# now look whether there are some specific hints for episode vs movie
|
||||
if filetype in ('video', 'subtitle'):
|
||||
for rexp, _, _ in episode_rexps:
|
||||
match = re.search(rexp, filename, re.IGNORECASE)
|
||||
if match:
|
||||
upgrade_episode()
|
||||
break
|
||||
|
||||
for prop, value, _, _ in find_properties(filename):
|
||||
log.debug('prop: %s = %s' % (prop, value))
|
||||
if prop == 'episodeFormat':
|
||||
upgrade_episode()
|
||||
break
|
||||
|
||||
elif canonical_form(value) == 'DVB':
|
||||
upgrade_episode()
|
||||
break
|
||||
|
||||
# if no episode info found, assume it's a movie
|
||||
upgrade_movie()
|
||||
|
||||
filetype = filetype_container[0]
|
||||
return filetype, other
|
||||
|
||||
|
||||
def process(mtree, filetype='autodetect'):
|
||||
filetype, other = guess_filetype(mtree.string, filetype)
|
||||
|
||||
mtree.guess.set('type', filetype, confidence=1.0)
|
||||
log.debug('Found with confidence %.2f: %s' % (1.0, mtree.guess))
|
||||
|
||||
filetype_info = Guess(other, confidence=1.0)
|
||||
# guess the mimetype of the filename
|
||||
# TODO: handle other mimetypes not found on the default type_maps
|
||||
# mimetypes.types_map['.srt']='text/subtitle'
|
||||
mime, _ = mimetypes.guess_type(mtree.string, strict=False)
|
||||
if mime is not None:
|
||||
filetype_info.update({'mimetype': mime}, confidence=1.0)
|
||||
|
||||
node_ext = mtree.node_at((-1,))
|
||||
node_ext.guess = filetype_info
|
||||
log.debug('Found with confidence %.2f: %s' % (1.0, node_ext.guess))
|
||||
@@ -0,0 +1,47 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.language import search_language
|
||||
from guessit.textutils import clean_string
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_language")
|
||||
|
||||
|
||||
def guess_language(string):
|
||||
language, span, confidence = search_language(string)
|
||||
if language:
|
||||
# is it a subtitle language?
|
||||
if 'sub' in clean_string(string[:span[0]]).lower().split(' '):
|
||||
return (Guess({'subtitleLanguage': language},
|
||||
confidence=confidence),
|
||||
span)
|
||||
else:
|
||||
return (Guess({'language': language},
|
||||
confidence=confidence),
|
||||
span)
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_language, None, log).process(mtree)
|
||||
@@ -0,0 +1,171 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_movie_title_from_position")
|
||||
|
||||
|
||||
def process(mtree):
|
||||
def found_property(node, name, value, confidence):
|
||||
node.guess = Guess({ name: value },
|
||||
confidence=confidence)
|
||||
log.debug('Found with confidence %.2f: %s' % (confidence, node.guess))
|
||||
|
||||
def found_title(node, confidence):
|
||||
found_property(node, 'title', node.clean_value, confidence)
|
||||
|
||||
basename = mtree.node_at((-2,))
|
||||
all_valid = lambda leaf: len(leaf.clean_value) > 0
|
||||
basename_leftover = basename.unidentified_leaves(valid=all_valid)
|
||||
|
||||
try:
|
||||
folder = mtree.node_at((-3,))
|
||||
folder_leftover = folder.unidentified_leaves()
|
||||
except ValueError:
|
||||
folder = None
|
||||
folder_leftover = []
|
||||
|
||||
log.debug('folder: %s' % folder_leftover)
|
||||
log.debug('basename: %s' % basename_leftover)
|
||||
|
||||
# specific cases:
|
||||
# if we find the same group both in the folder name and the filename,
|
||||
# it's a good candidate for title
|
||||
if (folder_leftover and basename_leftover and
|
||||
folder_leftover[0].clean_value == basename_leftover[0].clean_value):
|
||||
|
||||
found_title(folder_leftover[0], confidence=0.8)
|
||||
return
|
||||
|
||||
# specific cases:
|
||||
# if the basename contains a number first followed by an unidentified
|
||||
# group, and the folder only contains 1 unidentified one, then we have
|
||||
# a series
|
||||
# ex: Millenium Trilogy (2009)/(1)The Girl With The Dragon Tattoo(2009).mkv
|
||||
try:
|
||||
series = folder_leftover[0]
|
||||
filmNumber = basename_leftover[0]
|
||||
title = basename_leftover[1]
|
||||
|
||||
basename_leaves = basename.leaves()
|
||||
|
||||
num = int(filmNumber.clean_value)
|
||||
|
||||
log.debug('series: %s' % series.clean_value)
|
||||
log.debug('title: %s' % title.clean_value)
|
||||
if (series.clean_value != title.clean_value and
|
||||
series.clean_value != filmNumber.clean_value and
|
||||
basename_leaves.index(filmNumber) == 0 and
|
||||
basename_leaves.index(title) == 1):
|
||||
|
||||
found_title(title, confidence=0.6)
|
||||
found_property(series, 'filmSeries',
|
||||
series.clean_value, confidence=0.6)
|
||||
found_property(filmNumber, 'filmNumber',
|
||||
num, confidence=0.6)
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# specific cases:
|
||||
# - movies/tttttt (yyyy)/tttttt.ccc
|
||||
try:
|
||||
if mtree.node_at((-4, 0)).value.lower() == 'movies':
|
||||
folder = mtree.node_at((-3,))
|
||||
|
||||
# Note:too generic, might solve all the unittests as they all
|
||||
# contain 'movies' in their path
|
||||
#
|
||||
#if containing_folder.is_leaf() and not containing_folder.guess:
|
||||
# containing_folder.guess =
|
||||
# Guess({ 'title': clean_string(containing_folder.value) },
|
||||
# confidence=0.7)
|
||||
|
||||
year_group = folder.first_leaf_containing('year')
|
||||
groups_before = folder.previous_unidentified_leaves(year_group)
|
||||
|
||||
found_title(groups_before[0], confidence=0.8)
|
||||
return
|
||||
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# if we have either format or videoCodec in the folder containing the file
|
||||
# or one of its parents, then we should probably look for the title in
|
||||
# there rather than in the basename
|
||||
try:
|
||||
props = mtree.previous_leaves_containing(mtree.children[-2],
|
||||
[ 'videoCodec', 'format',
|
||||
'language' ])
|
||||
except IndexError:
|
||||
props = []
|
||||
|
||||
if props:
|
||||
group_idx = props[0].node_idx[0]
|
||||
if all(g.node_idx[0] == group_idx for g in props):
|
||||
# if they're all in the same group, take leftover info from there
|
||||
leftover = mtree.node_at((group_idx,)).unidentified_leaves()
|
||||
|
||||
if leftover:
|
||||
found_title(leftover[0], confidence=0.7)
|
||||
return
|
||||
|
||||
# look for title in basename if there are some remaining undidentified
|
||||
# groups there
|
||||
if basename_leftover:
|
||||
title_candidate = basename_leftover[0]
|
||||
|
||||
# if basename is only one word and the containing folder has at least
|
||||
# 3 words in it, we should take the title from the folder name
|
||||
# ex: Movies/Alice in Wonderland DVDRip.XviD-DiAMOND/dmd-aw.avi
|
||||
# ex: Movies/Somewhere.2010.DVDRip.XviD-iLG/i-smwhr.avi <-- TODO: gets caught here?
|
||||
if (title_candidate.clean_value.count(' ') == 0 and
|
||||
folder_leftover and
|
||||
folder_leftover[0].clean_value.count(' ') >= 2):
|
||||
|
||||
found_title(folder_leftover[0], confidence=0.7)
|
||||
return
|
||||
|
||||
# if there are only 2 unidentified groups, the first of which is inside
|
||||
# brackets or parentheses, we take the second one for the title:
|
||||
# ex: Movies/[阿维达].Avida.2006.FRENCH.DVDRiP.XViD-PROD.avi
|
||||
if len(basename_leftover) == 2 and basename_leftover[0].is_explicit():
|
||||
found_title(basename_leftover[1], confidence=0.8)
|
||||
return
|
||||
|
||||
# if all else fails, take the first remaining unidentified group in the
|
||||
# basename as title
|
||||
found_title(title_candidate, confidence=0.6)
|
||||
return
|
||||
|
||||
# if there are no leftover groups in the basename, look in the folder name
|
||||
if folder_leftover:
|
||||
found_title(folder_leftover[0], confidence=0.5)
|
||||
return
|
||||
|
||||
# if nothing worked, look if we have a very small group at the beginning
|
||||
# of the basename
|
||||
basename = mtree.node_at((-2,))
|
||||
basename_leftover = basename.unidentified_leaves(valid=lambda leaf: True)
|
||||
if basename_leftover:
|
||||
found_title(basename_leftover[0], confidence=0.4)
|
||||
return
|
||||
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.patterns import find_properties
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_properties")
|
||||
|
||||
|
||||
def guess_properties(string):
|
||||
try:
|
||||
prop, value, pos, end = find_properties(string)[0]
|
||||
return { prop: value }, (pos, end)
|
||||
except IndexError:
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_properties, 1.0, log).process(mtree)
|
||||
@@ -0,0 +1,44 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_release_group")
|
||||
|
||||
|
||||
def guess_release_group(string):
|
||||
group_names = [ r'\.(Xvid)-(?P<releaseGroup>.*?)[ \.]',
|
||||
r'\.(DivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
r'\.(DVDivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
]
|
||||
for rexp in group_names:
|
||||
match = re.search(rexp, string, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
metadata.update({ 'videoCodec': match.group(1) })
|
||||
return metadata, (match.start() + 1, match.end() - 1)
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_release_group, 0.8, log).process(mtree)
|
||||
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.patterns import video_rexps, sep
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_video_rexps")
|
||||
|
||||
|
||||
def guess_video_rexps(string):
|
||||
string = '-' + string + '-'
|
||||
for rexp, confidence, span_adjust in video_rexps:
|
||||
match = re.search(sep + rexp + sep, string, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
# is this the better place to put it? (maybe, as it is at least
|
||||
# the soonest that we can catch it)
|
||||
if metadata.get('cdNumberTotal', -1) is None:
|
||||
del metadata['cdNumberTotal']
|
||||
return (Guess(metadata, confidence=confidence),
|
||||
(match.start() + span_adjust[0],
|
||||
match.end() + span_adjust[1] - 2))
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_video_rexps, None, log).process(mtree)
|
||||
@@ -0,0 +1,56 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import Guess
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.patterns import weak_episode_rexps
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_weak_episodes_rexps")
|
||||
|
||||
|
||||
def guess_weak_episodes_rexps(string, node):
|
||||
if 'episodeNumber' in node.root.info:
|
||||
return None, None
|
||||
|
||||
for rexp, span_adjust in weak_episode_rexps:
|
||||
match = re.search(rexp, string, re.IGNORECASE)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
span = (match.start() + span_adjust[0],
|
||||
match.end() + span_adjust[1])
|
||||
|
||||
epnum = int(metadata['episodeNumber'])
|
||||
if epnum > 100:
|
||||
return Guess({ 'season': epnum // 100,
|
||||
'episodeNumber': epnum % 100 },
|
||||
confidence=0.6), span
|
||||
else:
|
||||
return Guess(metadata, confidence=0.3), span
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
guess_weak_episodes_rexps.use_node = True
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_weak_episodes_rexps, 0.6, log).process(mtree)
|
||||
@@ -0,0 +1,38 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.patterns import websites
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_website")
|
||||
|
||||
|
||||
def guess_website(string):
|
||||
low = string.lower()
|
||||
for site in websites:
|
||||
pos = low.find(site.lower())
|
||||
if pos != -1:
|
||||
return {'website': site}, (pos, pos + len(site))
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_website, 1.0, log).process(mtree)
|
||||
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.date import search_year
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.guess_year")
|
||||
|
||||
|
||||
def guess_year(string):
|
||||
year, span = search_year(string)
|
||||
if year:
|
||||
return { 'year': year }, span
|
||||
else:
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_year, 1.0, log).process(mtree)
|
||||
@@ -0,0 +1,69 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import subtitle_exts
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.post_process")
|
||||
|
||||
|
||||
def process(mtree):
|
||||
# 1- try to promote language to subtitle language where it makes sense
|
||||
for node in mtree.nodes():
|
||||
if 'language' not in node.guess:
|
||||
continue
|
||||
|
||||
def promote_subtitle():
|
||||
# pylint: disable=W0631
|
||||
node.guess.set('subtitleLanguage', node.guess['language'],
|
||||
confidence=node.guess.confidence('language'))
|
||||
del node.guess['language']
|
||||
|
||||
# - if we matched a language in a file with a sub extension and that
|
||||
# the group is the last group of the filename, it is probably the
|
||||
# language of the subtitle
|
||||
# (eg: 'xxx.english.srt')
|
||||
if (mtree.node_at((-1,)).value.lower() in subtitle_exts and
|
||||
node == mtree.leaves()[-2]):
|
||||
promote_subtitle()
|
||||
|
||||
# - if a language is in an explicit group just preceded by "st",
|
||||
# it is a subtitle language (eg: '...st[fr-eng]...')
|
||||
try:
|
||||
idx = node.node_idx
|
||||
previous = mtree.node_at((idx[0], idx[1] - 1)).leaves()[-1]
|
||||
if previous.value.lower()[-2:] == 'st':
|
||||
promote_subtitle()
|
||||
except IndexError:
|
||||
pass
|
||||
|
||||
# 2- ", the" at the end of a series title should be prepended to it
|
||||
for node in mtree.nodes():
|
||||
if 'series' not in node.guess:
|
||||
continue
|
||||
|
||||
series = node.guess['series']
|
||||
lseries = series.lower()
|
||||
|
||||
if lseries[-4:] == ',the':
|
||||
node.guess['series'] = 'The ' + series[:-4]
|
||||
|
||||
if lseries[-5:] == ', the':
|
||||
node.guess['series'] = 'The ' + series[:-5]
|
||||
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.textutils import find_first_level_groups
|
||||
from guessit.patterns import group_delimiters
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.split_explicit_groups")
|
||||
|
||||
|
||||
def process(mtree):
|
||||
"""return the string split into explicit groups, that is, those either
|
||||
between parenthese, square brackets or curly braces, and those separated
|
||||
by a dash."""
|
||||
for c in mtree.children:
|
||||
groups = find_first_level_groups(c.value, group_delimiters[0])
|
||||
for delimiters in group_delimiters:
|
||||
flatten = lambda l, x: l + find_first_level_groups(x, delimiters)
|
||||
groups = reduce(flatten, groups, [])
|
||||
|
||||
# do not do this at this moment, it is not strong enough and can break other
|
||||
# patterns, such as dates, etc...
|
||||
#groups = reduce(lambda l, x: l + x.split('-'), groups, [])
|
||||
|
||||
c.split_on_components(groups)
|
||||
@@ -0,0 +1,51 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import sep
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.split_on_dash")
|
||||
|
||||
|
||||
def process(mtree):
|
||||
for node in mtree.unidentified_leaves():
|
||||
indices = []
|
||||
|
||||
didx = 0
|
||||
pattern = re.compile(sep + '-' + sep)
|
||||
match = pattern.search(node.value)
|
||||
while match:
|
||||
span = match.span()
|
||||
indices.extend([ span[0], span[1] ])
|
||||
match = pattern.search(node.value, span[1])
|
||||
|
||||
didx = node.value.find('-')
|
||||
while didx > 0:
|
||||
if (didx > 10 and
|
||||
(didx - 1 not in indices and
|
||||
didx + 2 not in indices)):
|
||||
|
||||
indices.extend([ didx, didx + 1 ])
|
||||
|
||||
didx = node.value.find('-', didx + 1)
|
||||
|
||||
if indices:
|
||||
node.partition(indices)
|
||||
@@ -0,0 +1,35 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit import fileutils
|
||||
import os.path
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.transfo.split_path_components")
|
||||
|
||||
|
||||
def process(mtree):
|
||||
"""Returns the filename split into [ dir*, basename, ext ]."""
|
||||
components = fileutils.split_path(mtree.value)
|
||||
basename = components.pop(-1)
|
||||
components += list(os.path.splitext(basename))
|
||||
components[-1] = components[-1][1:] # remove the '.' from the extension
|
||||
|
||||
mtree.split_on_components(components)
|
||||
Reference in New Issue
Block a user