Subliminal update

This commit is contained in:
Ruud
2012-04-08 17:35:39 +02:00
parent 5321037f08
commit 148e1940ac
85 changed files with 4375 additions and 3420 deletions
+24 -35
View File
@@ -18,10 +18,10 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
__version__ = '0.3-dev'
__all__ = [ 'Guess', 'Language',
'guess_file_info', 'guess_video_info',
'guess_movie_info', 'guess_episode_info' ]
__version__ = '0.3.1'
__all__ = ['Guess', 'Language',
'guess_file_info', 'guess_video_info',
'guess_movie_info', 'guess_episode_info']
from guessit.guess import Guess, merge_all
@@ -31,6 +31,7 @@ import logging
log = logging.getLogger("guessit")
class NullHandler(logging.Handler):
def emit(self, record):
pass
@@ -40,9 +41,7 @@ h = NullHandler()
log.addHandler(h)
def guess_file_info(filename, filetype, info = [ 'filename' ]):
def guess_file_info(filename, filetype, info=None):
"""info can contain the names of the various plugins, such as 'filename' to
detect filename info, or 'hash_md5' to get the md5 hash of the file.
@@ -52,27 +51,30 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
result = []
hashers = []
if info is None:
info = ['filename']
if isinstance(info, basestring):
info = [ info ]
info = [info]
for infotype in info:
if infotype == 'filename':
m = IterativeMatcher(filename, filetype = filetype)
m = IterativeMatcher(filename, filetype=filetype)
result.append(m.matched())
elif infotype == 'hash_mpc':
import hash_mpc
from guessit.hash_mpc import hash_file
try:
result.append(Guess({ 'hash_mpc': hash_mpc.hash_file(filename) },
confidence = 1.0))
result.append(Guess({'hash_mpc': hash_file(filename)},
confidence=1.0))
except Exception, e:
log.warning('Could not compute MPC-style hash because: %s' % e)
elif infotype == 'hash_ed2k':
import hash_ed2k
from guessit.hash_ed2k import hash_file
try:
result.append(Guess({ 'hash_ed2k': hash_ed2k.hash_file(filename) },
confidence = 1.0))
result.append(Guess({'hash_ed2k': hash_file(filename)},
confidence=1.0))
except Exception, e:
log.warning('Could not compute ed2k hash because: %s' % e)
@@ -88,18 +90,6 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
else:
log.warning('Invalid infotype: %s' % infotype)
"""For plugins which depend on some optional library, import them like that:
if infotype == 'plugin_name':
try:
import optional_lib
except ImportError:
raise Exception, 'The plugin module cannot be loaded because the optional_lib lib is missing'
# do some stuff
"""
# do all the hashes now, but on a single pass
if hashers:
try:
@@ -112,22 +102,21 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
hasher.update(chunk)
for infotype, hasher in hashers:
result.append(Guess({ infotype: hasher.hexdigest() },
confidence = 1.0))
result.append(Guess({infotype: hasher.hexdigest()},
confidence=1.0))
except Exception, e:
log.warning('Could not compute hash because: %s' % e)
return merge_all(result)
def guess_video_info(filename, info = [ 'filename' ]):
def guess_video_info(filename, info=None):
return guess_file_info(filename, 'autodetect', info)
def guess_movie_info(filename, info = [ 'filename' ]):
def guess_movie_info(filename, info=None):
return guess_file_info(filename, 'movie', info)
def guess_episode_info(filename, info = [ 'filename' ]):
def guess_episode_info(filename, info=None):
return guess_file_info(filename, 'episode', info)
-75
View File
@@ -1,75 +0,0 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
#from guessit import movie, episode
import os, os.path
import logging
log = logging.getLogger('guessit.autodetect')
def within(x, nrange):
"""Return whether a number is inside a given range, specified as a list or tuple
of the lower and upper bounds."""
low, high = nrange
return low <= x <= high
def guess_filename_info(filename):
log.debug('Trying to guess info for file: ' + filename)
# try to guess info as if it were an episode
episode_info = episode.guess_episode_filename(filename)
# 1- if we found either season/episodeNumber, then we're pretty sure it must
# be an episode
if 'season' in episode_info or 'episodeNumber' in episode_info:
log.debug('Likely an episode as it contains season and/or episodeNumber: ' + filename)
episode_info.update({ 'type': 'episode' }, confidence = 0.9)
return episode_info
# try to guess info as if it were a movie
movie_info = movie.guess_movie_filename(filename)
# 2- if the file exists, try to guess its type using its size
if os.path.exists(filename):
size = os.stat(filename).st_size / (1024 * 1024)
# if size <= 1/2 of 1CD -> episode (very unlikely a movie so small)
if size < 400:
log.debug('Likely an episode due to its small size (%dMB): %s' % (size, filename))
episode_info.update({ 'type': 'episode' }, confidence = 0.8)
return episode_info
# if size > 2G -> movie (even fullHD eps aren't that big yet)
if size > 2048:
log.debug('Likely a movie due to its big size (%dMB): %s' % (size, filename))
movie_info.update({ 'type': 'movie' }, confidence = 0.8)
return movie_info
# if size == 1CD or 2CDs -> movie
if within(size, [690, 710]) or within(size, [1380, 1420]):
log.debug('Likely a movie due to its size close to a CD size (%dMB): %s' % (size, filename))
movie_info.update({ 'type': 'movie' }, confidence = 0.8)
return movie_info
# 3- if all else fails, assume it's a movie
log.debug('Couldn\'t make an informed guess... Assuming file is a movie: %s' % filename)
movie_info.update({ 'type': 'movie' }, confidence = 0.5)
return movie_info
+31 -28
View File
@@ -21,6 +21,7 @@
import datetime
import re
def search_year(string):
"""Looks for year patterns, and if found return the year and group span.
Assumes there are sentinels at the beginning and end of the string that
@@ -62,34 +63,35 @@ def search_date(string):
dsep = r'[-/ \.]'
date_rexps = [ # 20010823
r'[^0-9]' +
r'(?P<year>[0-9]{4})' +
r'(?P<month>[0-9]{2})' +
r'(?P<day>[0-9]{2})' +
r'[^0-9]',
date_rexps = [
# 20010823
r'[^0-9]' +
r'(?P<year>[0-9]{4})' +
r'(?P<month>[0-9]{2})' +
r'(?P<day>[0-9]{2})' +
r'[^0-9]',
# 2001-08-23
r'[^0-9]' +
r'(?P<year>[0-9]{4})' + dsep +
r'(?P<month>[0-9]{2})' + dsep +
r'(?P<day>[0-9]{2})' +
r'[^0-9]',
# 2001-08-23
r'[^0-9]' +
r'(?P<year>[0-9]{4})' + dsep +
r'(?P<month>[0-9]{2})' + dsep +
r'(?P<day>[0-9]{2})' +
r'[^0-9]',
# 23-08-2001
r'[^0-9]' +
r'(?P<day>[0-9]{2})' + dsep +
r'(?P<month>[0-9]{2})' + dsep +
r'(?P<year>[0-9]{4})' +
r'[^0-9]',
# 23-08-2001
r'[^0-9]' +
r'(?P<day>[0-9]{2})' + dsep +
r'(?P<month>[0-9]{2})' + dsep +
r'(?P<year>[0-9]{4})' +
r'[^0-9]',
# 23-08-01
r'[^0-9]' +
r'(?P<day>[0-9]{2})' + dsep +
r'(?P<month>[0-9]{2})' + dsep +
r'(?P<year>[0-9]{2})' +
r'[^0-9]',
]
# 23-08-01
r'[^0-9]' +
r'(?P<day>[0-9]{2})' + dsep +
r'(?P<month>[0-9]{2})' + dsep +
r'(?P<year>[0-9]{2})' +
r'[^0-9]',
]
for drexp in date_rexps:
match = re.search(drexp, string)
@@ -98,7 +100,7 @@ def search_date(string):
year, month, day = int(d['year']), int(d['month']), int(d['day'])
# years specified as 2 digits should be adjusted here
if year < 100:
if year > (datetime.date.today().year % 100)+ 5:
if year > (datetime.date.today().year % 100) + 5:
year = 1900 + year
else:
year = 2000 + year
@@ -120,8 +122,9 @@ def search_date(string):
continue
# looks like we have a valid date
# note: span is [+1,-1] because we don't want to include the non-digit char
# note: span is [+1,-1] because we don't want to include the
# non-digit char
start, end = match.span()
return (date, (start+1, end-1))
return (date, (start + 1, end - 1))
return None, None
-76
View File
@@ -1,76 +0,0 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.patterns import subtitle_exts, video_exts, episode_rexps, find_properties, canonical_form
import os.path
import re
import logging
log = logging.getLogger("guessit.filetype")
def guess_filetype(filename, filetype = 'autodetect'):
other = {}
# look at the extension first
fileext = os.path.splitext(filename)[1][1:].lower()
if fileext in subtitle_exts:
if 'movie' in filetype:
filetype = 'moviesubtitle'
elif 'episode' in filetype:
filetype = 'episodesubtitle'
else:
filetype = 'subtitle'
other = { 'container': fileext }
elif fileext in video_exts:
if filetype == 'autodetect':
filetype = 'video'
other = { 'container': fileext }
else:
if filetype == 'autodetect':
filetype = 'unknown'
other = { 'extension': fileext }
# now look whether there are some specific hints for episode vs movie
if filetype in ('video', 'subtitle'):
for rexp, confidence, span_adjust in episode_rexps:
match = re.search(rexp, filename, re.IGNORECASE)
if match:
if filetype == 'video':
filetype = 'episode'
elif filetype == 'subtitle':
filetype = 'episodesubtitle'
break
for prop, value, start, end in find_properties(filename):
if canonical_form(value) == 'DVB':
if filetype == 'video':
filetype = 'episode'
elif filetype == 'subtitle':
filetype = 'episodesubtitle'
break
# if no episode info found, assume it's a movie
if filetype == 'video':
filetype = 'movie'
elif filetype == 'subtitle':
filetype = 'moviesubtitle'
return filetype, other
+7 -14
View File
@@ -50,11 +50,11 @@ def split_path(path):
# on Unix systems, the root folder is '/'
if head == '/' and tail == '':
return [ '/' ] + result
return ['/'] + result
# on Windows, the root folder is a drive letter (eg: 'C:\')
if len(head) == 3 and head[1:] == ':\\' and tail == '':
return [ head ] + result
return [head] + result
if head == '' and tail == '':
return result
@@ -64,17 +64,10 @@ def split_path(path):
path = head
continue
result = [ tail ] + result
result = [tail] + result
path = head
def split_path_components(filename):
"""Returns the filename split into [ dir*, basename, ext ]."""
result = split_path(filename)
basename = result.pop(-1)
return result + list(os.path.splitext(basename))
def file_in_same_dir(ref_file, desired_file):
"""Return the path for a file in the same dir as a given reference file.
@@ -82,17 +75,17 @@ def file_in_same_dir(ref_file, desired_file):
'~/smewt/smewt.settings'
"""
return os.path.join(*(split_path(ref_file)[:-1] + [ desired_file ]))
return os.path.join(*(split_path(ref_file)[:-1] + [desired_file]))
def load_file_in_same_dir(ref_file, filename):
"""Load a given file. Works even when the file is contained inside a zip."""
path = split_path(ref_file)[:-1] + [ filename ]
path = split_path(ref_file)[:-1] + [filename]
for i, p in enumerate(path):
if p.endswith('.zip'):
zfilename = os.path.join(*path[:i+1])
zfilename = os.path.join(*path[:i + 1])
zfile = zipfile.ZipFile(zfilename)
return zfile.read('/'.join(path[i+1:]))
return zfile.read('/'.join(path[i + 1:]))
return open(os.path.join(*path)).read()
+60 -45
View File
@@ -26,9 +26,12 @@ log = logging.getLogger("guessit.guess")
class Guess(dict):
"""A Guess is a dictionary which has an associated confidence for each of its values.
"""A Guess is a dictionary which has an associated confidence for each of
its values.
As it is a subclass of dict, you can use it everywhere you expect a
simple dict."""
As it is a subclass of dict, you can use it everywhere you expect a simple dict"""
def __init__(self, *args, **kwargs):
try:
confidence = kwargs.pop('confidence')
@@ -52,20 +55,20 @@ class Guess(dict):
elif isinstance(value, unicode):
data[prop] = value.encode('utf-8')
elif isinstance(value, list):
data[prop] = [ str(x) for x in value ]
data[prop] = [str(x) for x in value]
return data
def nice_string(self):
data = self.to_utf8_dict()
parts = json.dumps(data, indent = 4).split('\n')
parts = json.dumps(data, indent=4).split('\n')
for i, p in enumerate(parts):
if p[:5] != ' "':
continue
prop = p.split('"')[1]
parts[i] = (' [%.2f] "' % (self._confidence.get(prop) or -1)) + p[5:]
parts[i] = (' [%.2f] "' % self.confidence(prop)) + p[5:]
return '\n'.join(parts)
@@ -73,9 +76,9 @@ class Guess(dict):
return str(self.to_utf8_dict())
def confidence(self, prop):
return self._confidence[prop]
return self._confidence.get(prop, -1)
def set(self, prop, value, confidence = None):
def set(self, prop, value, confidence=None):
self[prop] = value
if confidence is not None:
self._confidence[prop] = confidence
@@ -83,7 +86,7 @@ class Guess(dict):
def set_confidence(self, prop, value):
self._confidence[prop] = value
def update(self, other, confidence = None):
def update(self, other, confidence=None):
dict.update(self, other)
if isinstance(other, Guess):
for prop in other:
@@ -94,36 +97,36 @@ class Guess(dict):
self._confidence[prop] = confidence
def update_highest_confidence(self, other):
"""Update this guess with the values from the given one. In case there is
property present in both, only the one with the highest one is kept."""
"""Update this guess with the values from the given one. In case
there is property present in both, only the one with the highest one
is kept."""
if not isinstance(other, Guess):
raise ValueError, 'Can only call this function on Guess instances'
raise ValueError('Can only call this function on Guess instances')
for prop in other:
if prop in self and self._confidence[prop] >= other._confidence[prop]:
if prop in self and self.confidence(prop) >= other.confidence(prop):
continue
self[prop] = other[prop]
self._confidence[prop] = other._confidence[prop]
self._confidence[prop] = other.confidence(prop)
def choose_int(g1, g2):
"""Function used by merge_similar_guesses to choose between 2 possible properties
when they are integers."""
"""Function used by merge_similar_guesses to choose between 2 possible
properties when they are integers."""
v1, c1 = g1 # value, confidence
v2, c2 = g2
if (v1 == v2):
return (v1, 1 - (1-c1)*(1-c2))
return (v1, 1 - (1 - c1) * (1 - c2))
else:
if c1 > c2:
return (v1, c1 - c2)
else:
return (v2, c2 - c1)
def choose_string(g1, g2):
"""Function used by merge_similar_guesses to choose between 2 possible properties
when they are strings.
"""Function used by merge_similar_guesses to choose between 2 possible
properties when they are strings.
If the 2 strings are similar, or one is contained in the other, the latter is returned
with an increased confidence.
@@ -142,7 +145,7 @@ def choose_string(g1, g2):
('Hello', 0.75)
>>> choose_string(('Hello', 0.4), ('Hello World', 0.4))
('Hello', 0.64000000000000001)
('Hello', 0.64)
>>> choose_string(('simpsons', 0.5), ('The Simpsons', 0.5))
('The Simpsons', 0.75)
@@ -159,7 +162,7 @@ def choose_string(g1, g2):
v1, v2 = v1.strip(), v2.strip()
v1l, v2l = v1.lower(), v2.lower()
combined_prob = 1 - (1-c1)*(1-c2)
combined_prob = 1 - (1 - c1) * (1 - c2)
if v1l == v2l:
return (v1, combined_prob)
@@ -191,26 +194,31 @@ def _merge_similar_guesses_nocheck(guesses, prop, choose):
This function assumes there are at least 2 valid guesses."""
similar = [ guess for guess in guesses if prop in guess ]
similar = [guess for guess in guesses if prop in guess]
g1, g2 = similar[0], similar[1]
other_props = set(g1) & set(g2) - set([prop])
if other_props:
log.debug('guess 1: %s' % g1)
log.debug('guess 2: %s' % g2)
for prop in other_props:
if g1[prop] != g2[prop]:
log.warning('both guesses to be merged have more than one different property in common, bailing out...')
log.warning('both guesses to be merged have more than one '
'different property in common, bailing out...')
return
# merge all props of s2 into s1, updating the confidence for the considered property
# merge all props of s2 into s1, updating the confidence for the
# considered property
v1, v2 = g1[prop], g2[prop]
c1, c2 = g1.confidence(prop), g2.confidence(prop)
new_value, new_confidence = choose((v1, c1), (v2, c2))
if new_confidence >= c1:
log.debug("Updating matching property '%s' with confidence %.2f" % (prop, new_confidence))
msg = "Updating matching property '%s' with confidence %.2f"
else:
log.debug("Updating non-matching property '%s' with confidence %.2f" % (prop, new_confidence))
msg = "Updating non-matching property '%s' with confidence %.2f"
log.debug(msg % (prop, new_confidence))
g2[prop] = new_value
g2.set_confidence(prop, new_confidence)
@@ -218,12 +226,13 @@ def _merge_similar_guesses_nocheck(guesses, prop, choose):
g1.update(g2)
guesses.remove(g2)
def merge_similar_guesses(guesses, prop, choose):
"""Take a list of guesses and merge those which have the same properties,
increasing or decreasing the confidence depending on whether their values
are similar."""
similar = [ guess for guess in guesses if prop in guess ]
similar = [guess for guess in guesses if prop in guess]
if len(similar) < 2:
# nothing to merge
return
@@ -233,9 +242,13 @@ def merge_similar_guesses(guesses, prop, choose):
if len(similar) > 2:
log.debug('complex merge, trying our best...')
before = len(guesses)
_merge_similar_guesses_nocheck(guesses, prop, choose)
merge_similar_guesses(guesses, prop, choose)
return
after = len(guesses)
if after < before:
# recurse only when the previous call actually did something,
# otherwise we end up in an infinite loop
merge_similar_guesses(guesses, prop, choose)
def merge_append_guesses(guesses, prop):
@@ -245,14 +258,12 @@ def merge_append_guesses(guesses, prop):
DEPRECATED, remove with old guessers
"""
similar = [ guess for guess in guesses if prop in guess ]
similar = [guess for guess in guesses if prop in guess]
if not similar:
return
merged = similar[0]
merged[prop] = [ merged[prop] ]
merged[prop] = [merged[prop]]
# TODO: what to do with global confidence? mean of them all?
for m in similar[1:]:
@@ -261,17 +272,18 @@ def merge_append_guesses(guesses, prop):
merged[prop].append(m[prop])
else:
if prop2 in m:
log.warning('overwriting property "%s" with value ' % (prop2, m[prop2]))
log.warning('overwriting property "%s" with value %s' % (prop2, m[prop2]))
merged[prop2] = m[prop2]
# TODO: confidence also
guesses.remove(m)
def merge_all(guesses, append = []):
"""Merges all the guesses in a single result, removes very unlikely values, and returns it.
You can specify a list of properties that should be appended into a list instead of being
merged.
def merge_all(guesses, append=None):
"""Merge all the guesses in a single result, remove very unlikely values,
and return it.
You can specify a list of properties that should be appended into a list
instead of being merged.
>>> merge_all([ Guess({ 'season': 2 }, confidence = 0.6),
... Guess({ 'episodeNumber': 13 }, confidence = 0.8) ])
@@ -286,20 +298,24 @@ def merge_all(guesses, append = []):
return Guess()
result = guesses[0]
if append is None:
append = []
for g in guesses[1:]:
# first append our appendable properties
for prop in append:
if prop in g:
result.set(prop, result.get(prop, []) + [ g[prop] ],
# TODO: what to do with confidence here? maybe an arithmetic mean...
confidence = g.confidence(prop))
result.set(prop, result.get(prop, []) + [g[prop]],
# TODO: what to do with confidence here? maybe an
# arithmetic mean...
confidence=g.confidence(prop))
del g[prop]
# then merge the remaining ones
if set(result) & set(g):
log.warning('duplicate properties %s in merged result...' % (set(result) & set(g)))
dups = set(result) & set(g)
if dups:
log.warning('duplicate properties %s in merged result...' % dups)
result.update_highest_confidence(g)
@@ -314,4 +330,3 @@ def merge_all(guesses, append = []):
result[prop] = list(set(result[prop]))
return result
+10 -5
View File
@@ -18,8 +18,9 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
import hashlib, os.path
import hashlib
import os.path
def hash_file(filename):
"""Returns the ed2k hash of a given file.
@@ -31,6 +32,7 @@ def hash_file(filename):
os.path.getsize(filename),
hash_filehash(filename).upper())
def hash_filehash(filename):
"""Returns the ed2k hash of a given file.
@@ -42,8 +44,10 @@ def hash_filehash(filename):
def gen(f):
while True:
x = f.read(9728000)
if x: yield x
else: return
if x:
yield x
else:
return
def md4_hash(data):
m = md4()
@@ -55,4 +59,5 @@ def hash_filehash(filename):
hashes = [md4_hash(data).digest() for data in a]
if len(hashes) == 1:
return hashes[0].encode("hex")
else: return md4_hash(reduce(lambda a,d: a + d, hashes, "")).hexd
else:
return md4_hash(reduce(lambda a, d: a + d, hashes, "")).hexd
+18 -18
View File
@@ -18,8 +18,9 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
import struct, os
import struct
import os
def hash_file(filename):
"""This function is taken from:
@@ -32,25 +33,24 @@ def hash_file(filename):
f = open(filename, "rb")
filesize = os.path.getsize(filename)
hash = filesize
hash_value = filesize
if filesize < 65536 * 2:
raise Exception, "SizeError: size is %d, should be > 132K..." % filesize
raise Exception("SizeError: size is %d, should be > 132K..." % filesize)
for x in range(65536/bytesize):
buffer = f.read(bytesize)
(l_value,)= struct.unpack(longlongformat, buffer)
hash += l_value
hash = hash & 0xFFFFFFFFFFFFFFFF #to remain as 64bit number
for x in range(65536 / bytesize):
buf = f.read(bytesize)
(l_value,) = struct.unpack(longlongformat, buf)
hash_value += l_value
hash_value = hash_value & 0xFFFFFFFFFFFFFFFF #to remain as 64bit number
f.seek(max(0,filesize-65536),0)
for x in range(65536/bytesize):
buffer = f.read(bytesize)
(l_value,)= struct.unpack(longlongformat, buffer)
hash += l_value
hash = hash & 0xFFFFFFFFFFFFFFFF
f.seek(max(0, filesize - 65536), 0)
for x in range(65536 / bytesize):
buf = f.read(bytesize)
(l_value,) = struct.unpack(longlongformat, buf)
hash_value += l_value
hash_value = hash_value & 0xFFFFFFFFFFFFFFFF
f.close()
returnedhash = "%016x" % hash
return returnedhash
return "%016x" % hash_value
+53 -44
View File
@@ -19,28 +19,28 @@
#
from guessit import fileutils
import os.path
import re
import logging
log = logging.getLogger('guessit.language')
# downloaded from http://www.loc.gov/standards/iso639-2/ISO-639-2_utf-8.txt
#
# Description of the fields:
# "An alpha-3 (bibliographic) code, an alpha-3 (terminologic) code (when given),
# an alpha-2 code (when given), an English name, and a French name of a language
# are all separated by pipe (|) characters."
_iso639_contents = fileutils.load_file_in_same_dir(__file__,
'ISO-639-2_utf-8.txt')
language_matrix = [ l.strip().decode('utf-8').split('|')
for l in fileutils.load_file_in_same_dir(__file__, 'ISO-639-2_utf-8.txt').split('\n') ]
for l in _iso639_contents.split('\n') ]
lng3 = frozenset(filter(bool, (l[0] for l in language_matrix)))
lng3term = frozenset(filter(bool, (l[1] for l in language_matrix)))
lng2 = frozenset(filter(bool, (l[2] for l in language_matrix)))
lng_en_name = frozenset(filter(bool, (lng for l in language_matrix for lng in l[3].lower().split('; '))))
lng_fr_name = frozenset(filter(bool, (lng for l in language_matrix for lng in l[4].lower().split('; '))))
lng3 = frozenset(l[0] for l in language_matrix if l[0])
lng3term = frozenset(l[1] for l in language_matrix if l[1])
lng2 = frozenset(l[2] for l in language_matrix if l[2])
lng_en_name = frozenset(lng for l in language_matrix
for lng in l[3].lower().split('; ') if lng)
lng_fr_name = frozenset(lng for l in language_matrix
for lng in l[4].lower().split('; ') if lng)
lng_all_names = lng3 | lng3term | lng2 | lng_en_name | lng_fr_name
lng3_to_lng3term = dict((l[0], l[1]) for l in language_matrix if l[1])
@@ -50,22 +50,29 @@ lng3_to_lng2 = dict((l[0], l[2]) for l in language_matrix if l[2])
lng2_to_lng3 = dict((l[2], l[0]) for l in language_matrix if l[2])
# we only return the first given english name, hoping it is the most used one
lng3_to_lng_en_name = dict((l[0], l[3].split('; ')[0]) for l in language_matrix if l[3])
lng_en_name_to_lng3 = dict((en_name.lower(), l[0]) for l in language_matrix if l[3] for en_name in l[3].split('; '))
lng3_to_lng_en_name = dict((l[0], l[3].split('; ')[0])
for l in language_matrix if l[3])
lng_en_name_to_lng3 = dict((en_name.lower(), l[0])
for l in language_matrix if l[3]
for en_name in l[3].split('; '))
# we only return the first given french name, hoping it is the most used one
lng3_to_lng_fr_name = dict((l[0], l[4].split('; ')[0]) for l in language_matrix if l[4])
lng_fr_name_to_lng3 = dict((fr_name.lower(), l[0]) for l in language_matrix if l[4] for fr_name in l[4].split('; '))
lng3_to_lng_fr_name = dict((l[0], l[4].split('; ')[0])
for l in language_matrix if l[4])
lng_fr_name_to_lng3 = dict((fr_name.lower(), l[0])
for l in language_matrix if l[4]
for fr_name in l[4].split('; '))
def is_language(language):
return language.lower() in lng_all_names
class Language(object):
"""This class represents a human language.
You can initialize it with pretty much everything, as it knows conversion from
ISO-639 2-letter and 3-letter codes, English and French names.
You can initialize it with pretty much everything, as it knows conversion
from ISO-639 2-letter and 3-letter codes, English and French names.
>>> Language('fr')
Language(French)
@@ -79,12 +86,16 @@ class Language(object):
if len(language) == 2:
lang = lng2_to_lng3.get(language)
elif len(language) == 3:
lang = language if language in lng3 else lng3term_to_lng3.get(language)
lang = (language
if language in lng3
else lng3term_to_lng3.get(language))
else:
lang = lng_en_name_to_lng3.get(language) or lng_fr_name_to_lng3.get(language)
lang = (lng_en_name_to_lng3.get(language) or
lng_fr_name_to_lng3.get(language))
if lang is None:
raise ValueError, 'The given string "%s" could not be identified as a language' % language
msg = 'The given string "%s" could not be identified as a language'
raise ValueError(msg % language)
self.lang = lang
@@ -103,7 +114,6 @@ class Language(object):
def french_name(self):
return lng3_to_lng_fr_name[self.lang]
def __hash__(self):
return hash(self.lang)
@@ -132,19 +142,15 @@ class Language(object):
return 'Language(%s)' % self
def search_language(string, lang_filter = None):
def search_language(string, lang_filter=None):
"""Looks for language patterns, and if found return the language object,
its group span and an associated confidence.
you can specify a list of allowed languages using the lang_filter argument,
as in lang_filter = [ 'fr', 'eng', 'spanish' ]
Assumes there are sentinels at the beginning and end of the string that
always allow matching a non-letter delimiting the language.
>>> search_language('movie [en].avi')
(Language(English), (7, 9), 0.80000000000000004)
(Language(English), (7, 9), 0.8)
>>> search_language('the zen fat cat and the gay mad men got a new fan', lang_filter = ['en', 'fr', 'es'])
(None, None, None)
@@ -153,25 +159,27 @@ def search_language(string, lang_filter = None):
# list of common words which could be interpreted as languages, but which
# are far too common to be able to say they represent a language in the
# middle of a string (where they most likely carry their commmon meaning)
lng_common_words = frozenset([ # english words
'is', 'it', 'am', 'mad', 'men', 'man', 'run', 'sin', 'st', 'to',
'no', 'non', 'war', 'min', 'new', 'car', 'day', 'bad', 'bat', 'fan',
'fry', 'cop', 'zen', 'gay', 'fat', 'cherokee', 'got', 'an', 'as',
'cat', 'her', 'be', 'hat', 'sun', 'may', 'my', 'mr',
# french words
'bas', 'de', 'le', 'son', 'vo', 'vf', 'ne', 'ca', 'ce', 'et', 'que',
'mal', 'est', 'vol', 'or', 'mon', 'se',
# spanish words
'la', 'el', 'del', 'por', 'mar',
# other
'ind', 'arw', 'ts', 'ii', 'bin', 'chan', 'ss', 'san'
])
lng_common_words = frozenset([
# english words
'is', 'it', 'am', 'mad', 'men', 'man', 'run', 'sin', 'st', 'to',
'no', 'non', 'war', 'min', 'new', 'car', 'day', 'bad', 'bat', 'fan',
'fry', 'cop', 'zen', 'gay', 'fat', 'cherokee', 'got', 'an', 'as',
'cat', 'her', 'be', 'hat', 'sun', 'may', 'my', 'mr',
# french words
'bas', 'de', 'le', 'son', 'vo', 'vf', 'ne', 'ca', 'ce', 'et', 'que',
'mal', 'est', 'vol', 'or', 'mon', 'se',
# spanish words
'la', 'el', 'del', 'por', 'mar',
# other
'ind', 'arw', 'ts', 'ii', 'bin', 'chan', 'ss', 'san', 'oss', 'iii',
'vi'
])
sep = r'[](){} \._-+'
if lang_filter:
lang_filter = set(Language(l) for l in lang_filter)
slow = string.lower()
slow = ' %s ' % string.lower()
confidence = 1.0 # for all of them
for lang in lng_all_names:
@@ -183,7 +191,7 @@ def search_language(string, lang_filter = None):
if pos != -1:
end = pos + len(lang)
# make sure our word is always surrounded by separators
if slow[pos-1] not in sep or slow[end] not in sep:
if slow[pos - 1] not in sep or slow[end] not in sep:
continue
language = Language(slow[pos:end])
@@ -201,10 +209,11 @@ def search_language(string, lang_filter = None):
elif len(lang) == 3:
confidence = 0.9
else:
# Note: we could either be really confident that we found a language
# or assume that full language names are too common words
# Note: we could either be really confident that we found a
# language or assume that full language names are too
# common words
confidence = 0.3 # going with the low-confidence route here
return language, (pos, end), confidence
return language, (pos - 1, end - 1), confidence
return None, None, None
+63 -505
View File
@@ -2,8 +2,7 @@
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
# Copyright (c) 2011 Ricard Marxer <ricardmp@gmail.com>
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
@@ -19,282 +18,18 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import fileutils, textutils
from guessit.guess import Guess, merge_similar_guesses, merge_all, choose_int, choose_string
from guessit.date import search_date, search_year
from guessit.language import search_language
from guessit.filetype import guess_filetype
from guessit.patterns import video_exts, subtitle_exts, sep, deleted, video_rexps, websites, episode_rexps, weak_episode_rexps, non_episode_title, find_properties, canonical_form, unlikely_series
from guessit.matchtree import get_group, find_group, leftover_valid_groups, tree_to_string
from guessit.textutils import find_first_level_groups, split_on_groups, blank_region, clean_string, to_utf8
from guessit.fileutils import split_path_components
import datetime
import os.path
import re
from guessit.matchtree import MatchTree
from guessit.textutils import to_utf8
from guessit.guess import (merge_similar_guesses, merge_all,
choose_int, choose_string)
import copy
import logging
import mimetypes
log = logging.getLogger("guessit.matcher")
def split_explicit_groups(string):
"""return the string split into explicit groups, that is, those either
between parenthese, square brackets or curly braces, and those separated
by a dash."""
result = find_first_level_groups(string, '()')
result = reduce(lambda l, x: l + find_first_level_groups(x, '[]'), result, [])
result = reduce(lambda l, x: l + find_first_level_groups(x, '{}'), result, [])
# do not do this at this moment, it is not strong enough and can break other
# patterns, such as dates, etc...
#result = reduce(lambda l, x: l + x.split('-'), result, [])
return result
def format_guess(guess):
"""Format all the found values to their natural type.
For instance, a year would be stored as an int value, etc...
Note that this modifies the dictionary given as input.
"""
for prop, value in guess.items():
if prop in ('season', 'episodeNumber', 'year', 'cdNumber', 'cdNumberTotal'):
guess[prop] = int(guess[prop])
elif isinstance(value, basestring):
if prop in ('edition',):
value = clean_string(value)
guess[prop] = canonical_form(value)
return guess
def guess_groups(string, result, filetype):
# add sentinels so we can match a separator char at either end of
# our groups, even when they are at the beginning or end of the string
# we will adjust the span accordingly later
#
# filetype can either be movie, moviesubtitle, episode, episodesubtitle
current = ' ' + string + ' '
regions = [] # list of (start, end) of matched regions
def guessed(match_dict, confidence):
guess = format_guess(Guess(match_dict, confidence = confidence))
result.append(guess)
log.debug('Found with confidence %.2f: %s' % (confidence, guess))
return guess
def update_found(string, guess, span, span_adjust = (0,0)):
span = (span[0] + span_adjust[0],
span[1] + span_adjust[1])
regions.append((span, guess))
return blank_region(string, span)
# try to find dates first, as they are very specific
date, span = search_date(current)
if date:
guess = guessed({ 'date': date }, confidence = 1.0)
current = update_found(current, guess, span)
# for non episodes only, look for year information
if filetype not in ('episode', 'episodesubtitle'):
year, span = search_year(current)
if year:
guess = guessed({ 'year': year }, confidence = 1.0)
current = update_found(current, guess, span)
# specific regexps (ie: cd number, season X episode, ...)
for rexp, confidence, span_adjust in video_rexps:
match = re.search(rexp, current, re.IGNORECASE)
if match:
metadata = match.groupdict()
# is this the better place to put it? (maybe, as it is at least the soonest that we can catch it)
if 'cdNumberTotal' in metadata and metadata['cdNumberTotal'] is None:
del metadata['cdNumberTotal']
guess = guessed(metadata, confidence = confidence)
current = update_found(current, guess, match.span(), span_adjust)
if filetype in ('episode', 'episodesubtitle'):
for rexp, confidence, span_adjust in episode_rexps:
match = re.search(rexp, current, re.IGNORECASE)
if match:
metadata = match.groupdict()
guess = guessed(metadata, confidence = confidence)
current = update_found(current, guess, match.span(), span_adjust)
# Now websites, but as exact string instead of regexps
clow = current.lower()
for site in websites:
pos = clow.find(site.lower())
if pos != -1:
guess = guessed({ 'website': site }, confidence = confidence)
current = update_found(current, guess, (pos, pos+len(site)))
clow = current.lower()
# release groups have certain constraints, cannot be included in the previous general regexps
group_names = [ r'\.(Xvid)-(?P<releaseGroup>.*?)[ \.]',
r'\.(DivX)-(?P<releaseGroup>.*?)[\. ]',
r'\.(DVDivX)-(?P<releaseGroup>.*?)[\. ]',
]
for rexp in group_names:
match = re.search(rexp, current, re.IGNORECASE)
if match:
metadata = match.groupdict()
metadata.update({ 'videoCodec': match.group(1) })
guess = guessed(metadata, confidence = 0.8)
current = update_found(current, guess, match.span(), span_adjust = (1, -1))
# common well-defined words and regexps
confidence = 1.0 # for all of them
for prop, value, pos, end in find_properties(current):
guess = guessed({ prop: value }, confidence = confidence)
current = update_found(current, guess, (pos, end))
# weak guesses for episode number, only run it if we don't have an estimate already
if filetype in ('episode', 'episodesubtitle'):
if not any('episodeNumber' in match for match in result):
for rexp, _, span_adjust in weak_episode_rexps:
match = re.search(rexp, current, re.IGNORECASE)
if match:
metadata = match.groupdict()
epnum = int(metadata['episodeNumber'])
if epnum > 100:
guess = guessed({ 'season': epnum // 100,
'episodeNumber': epnum % 100 }, confidence = 0.6)
else:
guess = guessed(metadata, confidence = 0.3)
current = update_found(current, guess, match.span(), span_adjust)
# try to find languages now
language, span, confidence = search_language(current)
while language:
# is it a subtitle language?
if 'sub' in clean_string(current[:span[0]]).lower().split(' '):
guess = guessed({ 'subtitleLanguage': language }, confidence = confidence)
else:
guess = guessed({ 'language': language }, confidence = confidence)
current = update_found(current, guess, span)
language, span, confidence = search_language(current)
# remove our sentinels now and ajust spans accordingly
assert(current[0] == ' ' and current[-1] == ' ')
current = current[1:-1]
regions = [ ((start-1, end-1), guess) for (start, end), guess in regions ]
# split into '-' separated subgroups (with required separator chars
# around the dash)
didx = current.find('-')
while didx > 0:
regions.append(((didx, didx), None))
didx = current.find('-', didx+1)
# cut our final groups, and rematch the guesses to the group that created
# id, None if it is a leftover group
region_spans = [ span for span, guess in regions ]
string_groups = split_on_groups(string, region_spans)
remaining_groups = split_on_groups(current, region_spans)
guesses = []
pos = 0
for group in string_groups:
found = False
for span, guess in regions:
if span[0] == pos:
guesses.append(guess)
found = True
if not found:
guesses.append(None)
pos += len(group)
return zip(string_groups,
remaining_groups,
guesses)
def match_from_epnum_position(match_tree, epnum_pos, guessed, update_found):
"""guessed is a callback function to call with the guessed group
update_found is a callback to update the match group and returns leftover groups."""
pidx, eidx, gidx = epnum_pos
# a few helper functions to be able to filter using high-level semantics
def same_pgroup_before(group):
_, (ppidx, eeidx, ggidx) = group
return ppidx == pidx and (eeidx, ggidx) < (eidx, gidx)
def same_pgroup_after(group):
_, (ppidx, eeidx, ggidx) = group
return ppidx == pidx and (eeidx, ggidx) > (eidx, gidx)
def same_egroup_before(group):
_, (ppidx, eeidx, ggidx) = group
return ppidx == pidx and eeidx == eidx and ggidx < gidx
def same_egroup_after(group):
_, (ppidx, eeidx, ggidx) = group
return ppidx == pidx and eeidx == eidx and ggidx > gidx
leftover = leftover_valid_groups(match_tree)
# if we have at least 1 valid group before the episodeNumber, then it's probably
# the series name
series_candidates = filter(same_pgroup_before, leftover)
if len(series_candidates) >= 1:
guess = guessed({ 'series': series_candidates[0][0] }, confidence = 0.7)
leftover = update_found(leftover, series_candidates[0][1], guess)
# only 1 group after (in the same path group) and it's probably the episode title
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
filter(same_pgroup_after, leftover))
if len(title_candidates) == 1:
guess = guessed({ 'title': title_candidates[0][0] }, confidence = 0.5)
leftover = update_found(leftover, title_candidates[0][1], guess)
else:
# try in the same explicit group, with lower confidence
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
filter(same_egroup_after, leftover))
if len(title_candidates) == 1:
guess = guessed({ 'title': title_candidates[0][0] }, confidence = 0.4)
leftover = update_found(leftover, title_candidates[0][1], guess)
# epnumber is the first group and there are only 2 after it in same path group
# -> season title - episode title
already_has_title = (find_group(match_tree, 'title') != [])
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
filter(same_pgroup_after, leftover))
if (not already_has_title and # no title
not filter(same_pgroup_before, leftover) and # no groups before
len(title_candidates) == 2): # only 2 groups after
guess = guessed({ 'series': title_candidates[0][0] }, confidence = 0.4)
leftover = update_found(leftover, title_candidates[0][1], guess)
guess = guessed({ 'title': title_candidates[1][0] }, confidence = 0.4)
leftover = update_found(leftover, title_candidates[1][1], guess)
# if we only have 1 remaining valid group in the pathpart before the filename,
# then it's likely that it is the series name
series_candidates = [ group for group in leftover if group[1][0] == pidx-1 ]
if len(series_candidates) == 1:
guess = guessed({ 'series': series_candidates[0][0] }, confidence = 0.5)
leftover = update_found(leftover, series_candidates[0][1], guess)
return match_tree
class IterativeMatcher(object):
def __init__(self, filename, filetype = 'autodetect'):
def __init__(self, filename, filetype='autodetect'):
"""An iterative matcher tries to match different patterns that appear
in the filename.
@@ -325,7 +60,7 @@ class IterativeMatcher(object):
The first 3 lines indicates the group index in which a char in the
filename is located. So for instance, x264 is the group (0, 4, 1), and
it corresponds to a video codec, denoted by the letter'v' in the 4th line.
(for more info, see guess.matchtree.tree_to_string)
(for more info, see guess.matchtree.to_string)
Second, it tries to merge all this information into a single object
@@ -333,266 +68,89 @@ class IterativeMatcher(object):
resolution when they arise.
"""
if filetype not in ('autodetect', 'subtitle', 'video',
valid_filetypes = ('autodetect', 'subtitle', 'video',
'movie', 'moviesubtitle',
'episode', 'episodesubtitle'):
raise ValueError, "filetype needs to be one of ('autodetect', 'subtitle', 'video', 'movie', 'moviesubtitle', 'episode', 'episodesubtitle')"
'episode', 'episodesubtitle')
if filetype not in valid_filetypes:
raise ValueError("filetype needs to be one of %s" % valid_filetypes)
if not isinstance(filename, unicode):
log.debug('WARNING: given filename to matcher is not unicode...')
match_tree = []
result = [] # list of found metadata
def guessed(match_dict, confidence):
guess = format_guess(Guess(match_dict, confidence = confidence))
result.append(guess)
log.debug('Found with confidence %.2f: %s' % (confidence, guess))
return guess
def update_found(leftover, group_pos, guess):
pidx, eidx, gidx = group_pos
group = match_tree[pidx][eidx][gidx]
match_tree[pidx][eidx][gidx] = (group[0],
deleted * len(group[0]),
guess)
return [ g for g in leftover if g[1] != group_pos ]
self.match_tree = MatchTree(filename)
mtree = self.match_tree
mtree.guess.set('type', filetype, confidence=1.0)
def apply_transfo(transfo_name, *args, **kwargs):
transfo = __import__('guessit.transfo.' + transfo_name,
globals=globals(), locals=locals(),
fromlist=['process'], level=-1)
transfo.process(mtree, *args, **kwargs)
# 1- first split our path into dirs + basename + ext
match_tree = split_path_components(filename)
apply_transfo('split_path_components')
# try to detect the file type
filetype, other = guess_filetype(filename, filetype)
guessed({ 'type': filetype }, confidence = 1.0)
extguess = guessed(other, confidence = 1.0)
# 2- guess the file type now (will be useful later)
apply_transfo('guess_filetype', filetype)
if mtree.guess['type'] == 'unknown':
return
# guess the mimetype of the filename
# TODO: handle other mimetypes not found on the default type_maps
# mimetypes.types_map['.srt']='text/subtitle'
mime, _ = mimetypes.guess_type(filename, strict=False)
if mime is not None:
guessed({ 'mimetype': mime }, confidence = 1.0)
# 3- split each of those into explicit groups (separated by parentheses
# or square brackets)
apply_transfo('split_explicit_groups')
# remove the extension from the match tree, as all indices relative
# the the filename groups assume the basename is the last one
fileext = match_tree.pop(-1)[1:].lower()
# 4- try to match information for specific patterns
if mtree.guess['type'] in ('episode', 'episodesubtitle'):
strategy = ['guess_date', 'guess_video_rexps',
'guess_episodes_rexps', 'guess_website',
'guess_release_group', 'guess_properties',
'guess_weak_episodes_rexps', 'guess_language']
else:
strategy = ['guess_date', 'guess_year', 'guess_video_rexps',
'guess_website', 'guess_release_group',
'guess_properties', 'guess_language']
for name in strategy:
apply_transfo(name)
# 2- split each of those into explicit groups, if any
# note: be careful, as this might split some regexps with more confidence such as
# Alfleni-Team, or [XCT] or split a date such as (14-01-2008)
match_tree = [ split_explicit_groups(part) for part in match_tree ]
# more guessers for both movies and episodes
for name in ['guess_bonus_features']:
apply_transfo(name)
# split into '-' separated subgroups (with required separator chars
# around the dash)
apply_transfo('split_on_dash')
# 3- try to match information in decreasing order of confidence and
# blank the matching group in the string if we found something
for pathpart in match_tree:
for gidx, explicit_group in enumerate(pathpart):
pathpart[gidx] = guess_groups(explicit_group, result, filetype = filetype)
# 5- try to identify the remaining unknown groups by looking at their
# position relative to other known elements
if mtree.guess['type'] in ('episode', 'episodesubtitle'):
apply_transfo('guess_episode_info_from_position')
else:
apply_transfo('guess_movie_title_from_position')
# 4- try to identify the remaining unknown groups by looking at their position
# relative to other known elements
if filetype in ('episode', 'episodesubtitle'):
eps = find_group(match_tree, 'episodeNumber')
if eps:
match_tree = match_from_epnum_position(match_tree, eps[0], guessed, update_found)
leftover = leftover_valid_groups(match_tree)
if not eps:
# if we don't have the episode number, but at least 2 groups in the
# last path group, then it's probably series - eptitle
title_candidates = filter(lambda g:g[0].lower() not in non_episode_title,
filter(lambda g: g[1][0] == len(match_tree)-1,
leftover_valid_groups(match_tree)))
if len(title_candidates) >= 2:
guess = guessed({ 'series': title_candidates[0][0] }, confidence = 0.4)
leftover = update_found(leftover, title_candidates[0][1], guess)
guess = guessed({ 'title': title_candidates[1][0] }, confidence = 0.4)
leftover = update_found(leftover, title_candidates[1][1], guess)
# if there's a path group that only contains the season info, then the previous one
# is most likely the series title (ie: .../series/season X/...)
eps = [ gpos for gpos in find_group(match_tree, 'season')
if 'episodeNumber' not in get_group(match_tree, gpos)[2] ]
if eps:
pidx, eidx, gidx = eps[0]
previous = [ group for group in leftover if group[1][0] == pidx - 1 ]
if len(previous) == 1:
guess = guessed({ 'series': previous[0][0] }, confidence = 0.5)
leftover = update_found(leftover, previous[0][1], guess)
# reduce the confidence of unlikely series
for guess in result:
if 'series' in guess:
if guess['series'].lower() in unlikely_series:
guess.set_confidence('series', guess.confidence('series') * 0.5)
elif filetype in ('movie', 'moviesubtitle'):
leftover_all = leftover_valid_groups(match_tree)
# specific cases:
# - movies/tttttt (yyyy)/tttttt.ccc
try:
if match_tree[-3][0][0][0].lower() == 'movies':
# Note:too generic, might solve all the unittests as they all contain 'movies'
# in their path
#
#if len(match_tree[-2][0]) == 1:
# title = match_tree[-2][0][0]
# guess = guessed({ 'title': clean_string(title[0]) }, confidence = 0.7)
# update_found(leftover_all, title, guess)
year_group = filter(lambda gpos: gpos[0] == len(match_tree)-2,
find_group(match_tree, 'year'))[0]
leftover = leftover_valid_groups(match_tree,
valid = lambda g: ((g[0] and g[0][0] not in sep) and
g[1][0] == len(match_tree) - 2))
if len(match_tree[-2]) == 2 and year_group[1] == 1:
title = leftover[0]
guess = guessed({ 'title': clean_string(title[0]) },
confidence = 0.8)
update_found(leftover_all, title[1], guess)
raise Exception # to exit the try catch now
leftover = [ g for g in leftover_all if (g[1][0] == year_group[0] and
g[1][1] < year_group[1] and
g[1][2] < year_group[2]) ]
leftover = sorted(leftover, key = lambda x:x[1])
title = leftover[0]
guess = guessed({ 'title': title[0] }, confidence = 0.8)
leftover = update_found(leftover, title[1], guess)
except:
pass
# if we have either format or videoCodec in the folder containing the file
# or one of its parents, then we should probably look for the title in
# there rather than in the basename
props = filter(lambda g: g[0] <= len(match_tree) - 2,
find_group(match_tree, 'videoCodec') +
find_group(match_tree, 'format') +
find_group(match_tree, 'language'))
leftover = None
if props and all(g[0] == props[0][0] for g in props):
leftover = [ g for g in leftover_all if g[1][0] == props[0][0] ]
if props and leftover:
guess = guessed({ 'title': leftover[0][0] }, confidence = 0.7)
leftover = update_found(leftover, leftover[0][1], guess)
else:
# first leftover group in the last path part sounds like a good candidate for title,
# except if it's only one word and that the first group before has at least 3 words in it
# (case where the filename contains an 8 chars short name and the movie title is
# actually in the parent directory name)
leftover = [ g for g in leftover_all if g[1][0] == len(match_tree)-1 ]
if leftover:
title, (pidx, eidx, gidx) = leftover[0]
previous_pgroup_leftover = filter(lambda g: g[1][0] == pidx-1, leftover_all)
if (title.count(' ') == 0 and
previous_pgroup_leftover and
previous_pgroup_leftover[0][0].count(' ') >= 2):
guess = guessed({ 'title': previous_pgroup_leftover[0][0] }, confidence = 0.6)
leftover = update_found(leftover, previous_pgroup_leftover[0][1], guess)
else:
guess = guessed({ 'title': title }, confidence = 0.6)
leftover = update_found(leftover, leftover[0][1], guess)
else:
# if there were no leftover groups in the last path part, look in the one before that
previous_pgroup_leftover = filter(lambda g: g[1][0] == len(match_tree)-2, leftover_all)
if previous_pgroup_leftover:
guess = guessed({ 'title': previous_pgroup_leftover[0][0] }, confidence = 0.6)
leftover = update_found(leftover, previous_pgroup_leftover[0][1], guess)
# 5- perform some post-processing steps
# 5.1- try to promote language to subtitle language where it makes sense
for pidx, eidx, gidx in find_group(match_tree, 'language'):
string, remaining, guess = get_group(match_tree, (pidx, eidx, gidx))
def promote_subtitle():
guess.set('subtitleLanguage', guess['language'], confidence = guess.confidence('language'))
del guess['language']
# - if we matched a language in a file with a sub extension and that the group
# is the last group of the filename, it is probably the language of the subtitle
# (eg: 'xxx.english.srt')
if (fileext in subtitle_exts and
pidx == len(match_tree) - 1 and
eidx == len(match_tree[pidx]) - 1):
promote_subtitle()
# - if a language is in an explicit group just preceded by "st", it is a subtitle
# language (eg: '...st[fr-eng]...')
if eidx > 0:
previous = get_group(match_tree, (pidx, eidx-1, -1))
if previous[0][-2:].lower() == 'st':
promote_subtitle()
# re-append the extension now
match_tree.append([[(fileext, deleted*len(fileext), extguess)]])
self.parts = result
self.match_tree = match_tree
if filename.startswith('/'):
filename = ' ' + filename
log.debug('Found match tree:\n%s\n%s' % (to_utf8(tree_to_string(match_tree)),
to_utf8(filename)))
# 6- perform some post-processing steps
apply_transfo('post_process')
log.debug('Found match tree:\n%s' % (to_utf8(unicode(mtree))))
def matched(self):
# we need to make a copy here, as the merge functions work in place and
# calling them on the match tree would modify it
parts = copy.deepcopy(self.parts)
# 1- start by doing some common preprocessing tasks
parts = [node.guess for node in self.match_tree.nodes() if node.guess]
parts = copy.deepcopy(parts)
# 1.1- ", the" at the end of a series title should be prepended to it
for part in parts:
if 'series' not in part:
continue
series = part['series']
lseries = series.lower()
if lseries[-4:] == ',the':
part['series'] = 'The ' + series[:-4]
if lseries[-5:] == ', the':
part['series'] = 'The ' + series[:-5]
# 2- try to merge similar information together and give it a higher confidence
# 1- try to merge similar information together and give it a higher
# confidence
for int_part in ('year', 'season', 'episodeNumber'):
merge_similar_guesses(parts, int_part, choose_int)
for string_part in ('title', 'series', 'container', 'format', 'releaseGroup', 'website',
'audioCodec', 'videoCodec', 'screenSize', 'episodeFormat'):
for string_part in ('title', 'series', 'container', 'format',
'releaseGroup', 'website', 'audioCodec',
'videoCodec', 'screenSize', 'episodeFormat'):
merge_similar_guesses(parts, string_part, choose_string)
result = merge_all(parts, append = ['language', 'subtitleLanguage', 'other'])
# 3- some last minute post-processing
if (result['type'] == 'episode' and
'season' not in result and
result.get('episodeFormat', '') == 'Minisode'):
result['season'] = 0
result = merge_all(parts,
append=['language', 'subtitleLanguage', 'other'])
log.debug('Final result: ' + result.nice_string())
return result
+206 -99
View File
@@ -18,136 +18,243 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.patterns import deleted
from guessit.textutils import clean_string
from guessit import Guess
from guessit.textutils import clean_string, str_fill, to_utf8
from guessit.patterns import group_delimiters
import logging
log = logging.getLogger("guessit.matchtree")
class BaseMatchTree(object):
"""A MatchTree represents the hierarchical split of a string into its
constituent semantic groups."""
def tree_to_string(tree):
"""Return a string representation for the given tree.
def __init__(self, string='', span=None, parent=None):
self.string = string
self.span = span or (0, len(string))
self.parent = parent
self.children = []
self.guess = Guess()
The lines convey the following information:
- line 1: path idx
- line 2: explicit group idx
- line 3: group index
- line 4: remaining info
- line 5: meaning conveyed
@property
def value(self):
return self.string[self.span[0]:self.span[1]]
Meaning is a letter indicating what type of info was matched by this group,
for instance 't' = title, 'f' = format, 'l' = language, etc...
@property
def clean_value(self):
return clean_string(self.value)
An example is the following:
@property
def offset(self):
return self.span[0]
0000000000000000000000000000000000000000000000000000000000000000000000000000000000 111
0000011111111111112222222222222233333333444444444444444455555555666777777778888888 000
0000000000000000000000000000000001111112011112222333333401123334000011233340000000 000
__________________(The.Prestige).______.[____.HP.______.{__-___}.St{__-___}.Chaps].___
xxxxxttttttttttttt ffffff vvvv xxxxxx ll lll xx xxx ccc
[XCT].Le.Prestige.(The.Prestige).DVDRip.[x264.HP.He-Aac.{Fr-Eng}.St{Fr-Eng}.Chaps].mkv
@property
def info(self):
result = dict(self.guess)
(note: the last line representing the filename is not pat of the tree representation)
"""
m_tree = [ '', # path level index
'', # explicit group index
'', # matched regexp and dash-separated
'', # groups leftover that couldn't be matched
'', # meaning conveyed: E = episodenumber, S = season, ...
]
for c in self.children:
result.update(c.info)
return result
@property
def root(self):
if not self.parent:
return self
return self.parent.root
@property
def depth(self):
if self.is_leaf():
return 0
return 1 + max(c.depth for c in self.children)
def is_leaf(self):
return self.children == []
def add_child(self, span):
child = MatchTree(self.string, span=span, parent=self)
self.children.append(child)
def partition(self, indices):
indices = sorted(indices)
if indices[0] != 0:
indices.insert(0, 0)
if indices[-1] != len(self.value):
indices.append(len(self.value))
for start, end in zip(indices[:-1], indices[1:]):
self.add_child(span=(self.offset + start,
self.offset + end))
def split_on_components(self, components):
offset = 0
for c in components:
start = self.value.find(c, offset)
end = start + len(c)
self.add_child(span=(self.offset + start,
self.offset + end))
offset = end
def nodes_at_depth(self, depth):
if depth == 0:
yield self
for child in self.children:
for node in child.nodes_at_depth(depth - 1):
yield node
@property
def node_idx(self):
if self.parent is None:
return ()
return self.parent.node_idx + (self.parent.children.index(self),)
def node_at(self, idx):
if not idx:
return self
try:
return self.children[idx[0]].node_at(idx[1:])
except:
raise ValueError('Non-existent node index: %s' % (idx,))
def nodes(self):
yield self
for child in self.children:
for node in child.nodes():
yield node
def _leaves(self):
if self.is_leaf():
yield self
else:
for child in self.children:
# pylint: disable=W0212
for leaf in child._leaves():
yield leaf
def leaves(self):
return list(self._leaves())
def to_string(self):
empty_line = ' ' * len(self.string)
def add_char(pidx, eidx, gidx, remaining, meaning = None):
nr = len(remaining)
def to_hex(x):
if isinstance(x, int):
return str(x) if x < 10 else chr(55+x)
return str(x) if x < 10 else chr(55 + x)
return x
m_tree[0] = m_tree[0] + to_hex(pidx) * nr
m_tree[1] = m_tree[1] + to_hex(eidx) * nr
m_tree[2] = m_tree[2] + to_hex(gidx) * nr
m_tree[3] = m_tree[3] + remaining
m_tree[4] = m_tree[4] + str(meaning or ' ') * nr
def meaning(result):
mmap = { 'episodeNumber': 'E',
'season': 'S',
'extension': 'e',
'format': 'f',
'language': 'l',
'videoCodec': 'v',
'audioCodec': 'a',
'website': 'w',
'container': 'c',
'series': 'T',
'title': 't',
'date': 'd',
'year': 'y',
'releaseGroup': 'r',
'screenSize': 's'
}
def meaning(result):
mmap = { 'episodeNumber': 'E',
'season': 'S',
'extension': 'e',
'format': 'f',
'language': 'l',
'videoCodec': 'v',
'audioCodec': 'a',
'website': 'w',
'container': 'c',
'series': 'T',
'title': 't',
'date': 'd',
'year': 'y',
'releaseGroup': 'r',
'screenSize': 's'
}
if result is None:
return ' '
if result is None:
return ' '
for prop, l in mmap.items():
if prop in result:
return l
for prop, l in mmap.items():
if prop in result:
return l
return 'x'
return 'x'
for pidx, pathpart in enumerate(tree):
for eidx, explicit_group in enumerate(pathpart):
for gidx, (group, remaining, result) in enumerate(explicit_group):
add_char(pidx, eidx, gidx, remaining, meaning(result))
lines = [ empty_line ] * (self.depth + 2) # +2: remaining, meaning
lines[-2] = self.string
# special conditions for the path separator
if pidx < len(tree) - 2:
add_char(' ', ' ', ' ', '/')
elif pidx == len(tree) - 2:
add_char(' ', ' ', ' ', '.')
for node in self.nodes():
if node == self:
continue
return '\n'.join(m_tree)
idx = node.node_idx
depth = len(idx) - 1
if idx:
lines[depth] = str_fill(lines[depth], node.span,
to_hex(idx[-1]))
if node.guess:
lines[-2] = str_fill(lines[-2], node.span, '_')
lines[-1] = str_fill(lines[-1], node.span, meaning(node.guess))
lines.append(self.string)
return '\n'.join(lines)
def __unicode__(self):
return self.to_string()
def __str__(self):
return to_utf8(unicode(self))
class MatchTree(BaseMatchTree):
"""The MatchTree contains a few "utility" methods which are not necessary
for the BaseMatchTree, but add a lot of convenience for writing
higher-level rules."""
def iterate_groups(match_tree):
"""Iterate over all the groups in a match_tree and return them as pairs
of (group_pos, group) where:
- group_pos = (pidx, eidx, gidx)
- group = (string, remaining, guess)
"""
for pidx, pathpart in enumerate(match_tree):
for eidx, explicit_group in enumerate(pathpart):
for gidx, group in enumerate(explicit_group):
yield (pidx, eidx, gidx), group
def _unidentified_leaves(self,
valid=lambda leaf: len(leaf.clean_value) >= 2):
for leaf in self._leaves():
if not leaf.guess and valid(leaf):
yield leaf
def unidentified_leaves(self,
valid=lambda leaf: len(leaf.clean_value) >= 2):
return list(self._unidentified_leaves(valid))
def find_group(match_tree, prop):
"""Find the list of groups that resulted in a guess that contains the
asked property."""
result = []
for gpos, (string, remaining, guess) in iterate_groups(match_tree):
if guess and prop in guess:
result.append(gpos)
return result
def _leaves_containing(self, property_name):
if isinstance(property_name, basestring):
property_name = [ property_name ]
def get_group(match_tree, gpos):
pidx, eidx, gidx = gpos
return match_tree[pidx][eidx][gidx]
for leaf in self._leaves():
for prop in property_name:
if prop in leaf.guess:
yield leaf
break
def leaves_containing(self, property_name):
return list(self._leaves_containing(property_name))
def leftover_valid_groups(match_tree, valid = lambda s: len(s[0]) > 3):
"""Return the list of valid string groups (eg: len(s) > 3) that could not be
matched to anything as a list of pairs (cleaned_str, group_pos)."""
leftover = []
for gpos, (group, remaining, guess) in iterate_groups(match_tree):
if not guess:
clean_str = clean_string(remaining)
if valid((clean_str, gpos)):
leftover.append((clean_str, gpos))
def first_leaf_containing(self, property_name):
try:
return next(self._leaves_containing(property_name))
except StopIteration:
return None
return leftover
def _previous_unidentified_leaves(self, node):
node_idx = node.node_idx
for leaf in self._unidentified_leaves():
if leaf.node_idx < node_idx:
yield leaf
def previous_unidentified_leaves(self, node):
return list(self._previous_unidentified_leaves(node))
def _previous_leaves_containing(self, node, property_name):
node_idx = node.node_idx
for leaf in self._leaves_containing(property_name):
if leaf.node_idx < node_idx:
yield leaf
def previous_leaves_containing(self, node, property_name):
return list(self._previous_leaves_containing(node, property_name))
def is_explicit(self):
"""Return whether the group was explicitly enclosed by
parentheses/square brackets/etc."""
return (self.value[0] + self.value[-1]) in group_delimiters
Regular → Executable
+45 -23
View File
@@ -22,10 +22,13 @@
subtitle_exts = [ 'srt', 'idx', 'sub', 'ssa', 'txt' ]
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'm4v', 'mov', 'ogg', 'ogm', 'ogv', 'wmv', 'divx' ]
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'm4v', 'mov', 'ogg', 'ogm', 'ogv',
'wmv', 'divx' ]
group_delimiters = [ '()', '[]', '{}' ]
# separator character regexp
sep = r'[][)(}{+ \._-]' # regexp art, hehe :D
sep = r'[][)(}{+ /\._-]' # regexp art, hehe :D
# character used to represent a deleted char (when matching groups)
deleted = '_'
@@ -35,28 +38,33 @@ episode_rexps = [ # ... Season 2 ...
(r'season (?P<season>[0-9]+)', 1.0, (0, 0)),
(r'saison (?P<season>[0-9]+)', 1.0, (0, 0)),
# ... s02-x01 ...
(r's(?P<season>[0-9]{1,2})-x(?P<bonusNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
# ... s02e13 ...
(r'[Ss](?P<season>[0-9]{1,2}).{,3}[EeXx](?P<episodeNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
# ... 2x13 ...
(r'[^0-9](?P<season>[0-9]{1,2})[x\.](?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
(r'[^0-9](?P<season>[0-9]{1,2})x(?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
# ... s02 ...
(sep + r's(?P<season>[0-9]{1,2})' + sep + '?', 0.6, (1, -1)),
#(sep + r's(?P<season>[0-9]{1,2})' + sep, 0.6, (1, -1)),
(r's(?P<season>[0-9]{1,2})[^0-9]', 0.6, (0, -1)),
# v2 or v3 for some mangas which have multiples rips
(sep + r'(?P<episodeNumber>[0-9]{1,3})v[23]' + sep, 0.6, (0, 0)),
(r'(?P<episodeNumber>[0-9]{1,3})v[23]' + sep, 0.6, (0, 0)),
# ... ep 23 ...
('ep' + sep + r'(?P<episodeNumber>[0-9]{1,2})[^0-9]', 0.7, (0, -1))
]
weak_episode_rexps = [ # ... 213 or 0106 ...
(sep + r'(?P<episodeNumber>[0-9]{1,4})' + sep, 0.3, (1, -1)),
(sep + r'(?P<episodeNumber>[0-9]{1,4})' + sep, (1, -1)),
# ... 2x13 ...
(sep + r'[^0-9](?P<season>[0-9]{1,2})\.(?P<episodeNumber>[0-9]{2})[^0-9]' + sep, (1, -1)),
]
non_episode_title = [ 'extras' ]
non_episode_title = [ 'extras', 'rip' ]
video_rexps = [ # cd number
@@ -76,7 +84,13 @@ video_rexps = [ # cd number
(r'(?P<width>[0-9]{3,4})x(?P<height>[0-9]{3,4})', 0.9, (0, 0)),
# website
(r'(?P<website>www(\.[a-zA-Z0-9]+){2,3})', 0.8, (0, 0))
(r'(?P<website>www(\.[a-zA-Z0-9]+){2,3})', 0.8, (0, 0)),
# bonusNumber: ... x01 ...
(r'x(?P<bonusNumber>[0-9]{1,2})', 1.0, (0, 0)),
# filmNumber: ... f01 ...
(r'f(?P<filmNumber>[0-9]{1,2})', 1.0, (0, 0))
]
websites = [ 'tvu.org.ru', 'emule-island.com', 'UsaBit.com', 'www.divx-overnet.com', 'sharethefiles.com' ]
@@ -87,7 +101,7 @@ properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'B
'HDRip', 'DVD', 'DVDivX', 'HDTV', 'DVB', 'DVBRip', 'PDTV', 'WEBRip',
'DVDSCR', 'Screener', 'VHS', 'VIDEO_TS' ],
'screenSize': [ '720p', '720' ],
'screenSize': [ '720p', '720', '1080p', '1080' ],
'videoCodec': [ 'XviD', 'DivX', 'x264', 'h264', 'Rv10' ],
@@ -98,18 +112,18 @@ properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'B
'releaseGroup': [ 'ESiR', 'WAF', 'SEPTiC', '[XCT]', 'iNT', 'PUKKA',
'CHD', 'ViTE', 'TLF', 'DEiTY', 'FLAiTE',
'MDX', 'GM4F', 'DVL', 'SVD', 'iLUMiNADOS', ' FiNaLe',
'UnSeeN', 'aXXo', 'KLAXXON', 'NoTV', 'ZeaL', 'LOL' ],
'UnSeeN', 'aXXo', 'KLAXXON', 'NoTV', 'ZeaL', 'LOL',
'HDBRiSe' ],
'episodeFormat': [ 'Minisode', 'Minisodes' ],
'other': [ '5ch', 'PROPER', 'REPACK', 'LIMITED', 'DualAudio', 'iNTERNAL', 'Audiofixed', 'R5',
'complete', 'classic', # not so sure about these ones, could appear in a title
'ws', # widescreen
#'SE', # special edition
# TODO: director's cut
],
}
def find_properties(filename):
result = []
clow = filename.lower()
@@ -119,7 +133,7 @@ def find_properties(filename):
if pos != -1:
end = pos + len(value)
# make sure our word is always surrounded by separators
if ((pos > 0 and clow[pos-1] not in sep) or
if ((pos > 0 and clow[pos - 1] not in sep) or
(end < len(clow) and clow[end] not in sep)):
# note: sep is a regexp, but in this case using it as
# a sequence achieves the same goal
@@ -137,6 +151,7 @@ property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
'DivX': [ 'DVDivX' ],
'h264': [ 'x264' ],
'720p': [ '720' ],
'1080p': [ '1080' ],
'AAC': [ 'He-AAC', 'AAC-He' ],
'Special Edition': [ 'Special' ],
'Collector Edition': [ 'Collector' ],
@@ -145,14 +160,21 @@ property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
}
reverse_synonyms = {}
for prop, values in properties.items():
for value in values:
reverse_synonyms[value.lower()] = value
def revert_synonyms():
reverse = {}
for _, values in properties.items():
for value in values:
reverse[value.lower()] = value
for canonical, synonyms in property_synonyms.items():
for synonym in synonyms:
reverse[synonym.lower()] = canonical
return reverse
reverse_synonyms = revert_synonyms()
for canonical, synonyms in property_synonyms.items():
for synonym in synonyms:
reverse_synonyms[synonym.lower()] = canonical
def canonical_form(string):
return reverse_synonyms.get(string.lower(), string)
+5 -5
View File
@@ -28,8 +28,8 @@ RED_FONT = "\x1B[0;31m"
RESET_FONT = "\x1B[0m"
def setupLogging(colored = True):
"""Sets up a nice colored logger as the main application logger (not only smewt itself)."""
def setupLogging(colored=True):
"""Set up a nice colored logger as the main application logger."""
class SimpleFormatter(logging.Formatter):
def __init__(self):
@@ -38,7 +38,9 @@ def setupLogging(colored = True):
class ColoredFormatter(logging.Formatter):
def __init__(self):
self.fmt = '%(levelname)-8s ' + BLUE_FONT + '%(module)s:%(funcName)s' + RESET_FONT + ' -- %(message)s'
self.fmt = ('%(levelname)-8s ' +
BLUE_FONT + '%(name)s:%(funcName)s' +
RESET_FONT + ' -- %(message)s')
logging.Formatter.__init__(self, self.fmt)
def format(self, record):
@@ -50,11 +52,9 @@ def setupLogging(colored = True):
else:
return RED_FONT + result
ch = logging.StreamHandler()
if colored and sys.platform != 'win32':
ch.setFormatter(ColoredFormatter())
else:
ch.setFormatter(SimpleFormatter())
logging.getLogger().addHandler(ch)
+33 -25
View File
@@ -18,47 +18,59 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.patterns import sep, deleted
from guessit.patterns import sep
import copy
# string-related functions
def strip_brackets(s):
if not s:
return s
if s[0] == '[' and s[-1] == ']': return s[1:-1]
if s[0] == '(' and s[-1] == ')': return s[1:-1]
if s[0] == '{' and s[-1] == '}': return s[1:-1]
if ((s[0] == '[' and s[-1] == ']') or
(s[0] == '(' and s[-1] == ')') or
(s[0] == '{' and s[-1] == '}')):
return s[1:-1]
return s
def clean_string(s):
for c in sep:
for c in sep[:-2]: # do not remove dashes ('-')
s = s.replace(c, ' ')
parts = s.split()
return ' '.join(p for p in parts if p != '')
result = ' '.join(p for p in parts if p != '')
# now also remove dashes on the outer part of the string
while result and result[0] in sep:
result = result[1:]
while result and result[-1] in sep:
result = result[:-1]
return result
def str_replace(string, pos, c):
return string[:pos] + c + string[pos+1:]
def blank_region(string, region, blank_sep = deleted):
def str_fill(string, region, c):
start, end = region
return string[:start] + blank_sep * (end - start) + string[end:]
return string[:start] + c * (end - start) + string[end:]
def between(s, left, right):
return s.split(left)[1].split(right)[0]
def to_utf8(o):
'''converts all unicode strings found in the given object to utf-8 strings'''
"""Convert all unicode strings found in the given object to utf-8
strings."""
if isinstance(o, unicode):
return o.encode('utf-8')
elif isinstance(o, list):
return [ to_utf8(i) for i in o ]
elif isinstance(o, dict):
result = copy.deepcopy(o) # need to do it like that to handle Guess instances correctly
# need to do it like that to handle Guess instances correctly
result = copy.deepcopy(o)
for key, value in o.items():
result[to_utf8(key)] = to_utf8(value)
return result
@@ -68,8 +80,10 @@ def to_utf8(o):
def levenshtein(a, b):
if not a: return len(b)
if not b: return len(a)
if not a:
return len(b)
if not b:
return len(a)
m = len(a)
n = len(b)
@@ -160,14 +174,13 @@ def split_on_groups(string, groups):
if boundaries[-1] != len(string):
boundaries.append(len(string))
groups = [ string[start:end] for start, end in zip(boundaries[:-1], boundaries[1:]) ]
groups = [ string[start:end] for start, end in zip(boundaries[:-1],
boundaries[1:]) ]
return filter(bool, groups) # return only non-empty groups
return [ g for g in groups if g ] # return only non-empty groups
def find_first_level_groups(string, enclosing, blank_sep = None):
def find_first_level_groups(string, enclosing, blank_sep=None):
"""Return a list of groups that could be split because of explicit grouping.
The groups are delimited by the given enclosing characters.
@@ -203,8 +216,3 @@ def find_first_level_groups(string, enclosing, blank_sep = None):
string = str_replace(string, end-1, blank_sep)
return split_on_groups(string, groups)
+100
View File
@@ -0,0 +1,100 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
from guessit.patterns import canonical_form
from guessit.textutils import clean_string
import logging
log = logging.getLogger('guessit.transfo')
def found_property(node, name, confidence):
node.guess = Guess({name: node.clean_value}, confidence=confidence)
log.debug('Found with confidence %.2f: %s' % (confidence, node.guess))
def format_guess(guess):
"""Format all the found values to their natural type.
For instance, a year would be stored as an int value, etc...
Note that this modifies the dictionary given as input.
"""
for prop, value in guess.items():
if prop in ('season', 'episodeNumber', 'year', 'cdNumber',
'cdNumberTotal', 'bonusNumber', 'filmNumber'):
guess[prop] = int(guess[prop])
elif isinstance(value, basestring):
if prop in ('edition',):
value = clean_string(value)
guess[prop] = canonical_form(value)
return guess
def find_and_split_node(node, strategy, logger):
string = ' %s ' % node.value # add sentinels
for matcher, confidence in strategy:
if getattr(matcher, 'use_node', False):
result, span = matcher(string, node)
else:
result, span = matcher(string)
if result:
# readjust span to compensate for sentinels
span = (span[0] - 1, span[1] - 1)
if isinstance(result, Guess):
if confidence is None:
confidence = result.confidence(result.keys()[0])
else:
if confidence is None:
confidence = 1.0
guess = format_guess(Guess(result, confidence=confidence))
msg = 'Found with confidence %.2f: %s' % (confidence, guess)
(logger or log).debug(msg)
node.partition(span)
absolute_span = (span[0] + node.offset, span[1] + node.offset)
for child in node.children:
if child.span == absolute_span:
child.guess = guess
else:
find_and_split_node(child, strategy, logger)
return
class SingleNodeGuesser(object):
def __init__(self, guess_func, confidence, logger=None):
self.guess_func = guess_func
self.confidence = confidence
self.logger = logger
def process(self, mtree):
# strategy is a list of pairs (guesser, confidence)
# - if the guesser returns a guessit.Guess and confidence is specified,
# it will override it, otherwise it will leave the guess confidence
# - if the guesser returns a simple dict as a guess and confidence is
# specified, it will use it, or 1.0 otherwise
strategy = [ (self.guess_func, self.confidence) ]
for node in mtree.unidentified_leaves():
find_and_split_node(node, strategy, self.logger)
@@ -0,0 +1,60 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.transfo import found_property
import logging
log = logging.getLogger("guessit.transfo.guess_bonus_features")
def process(mtree):
def previous_group(g):
for leaf in mtree.unidentified_leaves()[::-1]:
if leaf.node_idx < g.node_idx:
return leaf
def next_group(g):
for leaf in mtree.unidentified_leaves():
if leaf.node_idx > g.node_idx:
return leaf
def same_group(g1, g2):
return g1.node_idx[:2] == g2.node_idx[:2]
bonus = [ node for node in mtree.leaves() if 'bonusNumber' in node.guess ]
if bonus:
bonusTitle = next_group(bonus[0])
if same_group(bonusTitle, bonus[0]):
found_property(bonusTitle, 'bonusTitle', 0.8)
filmNumber = [ node for node in mtree.leaves()
if 'filmNumber' in node.guess ]
if filmNumber:
filmSeries = previous_group(filmNumber[0])
found_property(filmSeries, 'filmSeries', 0.9)
title = next_group(filmNumber[0])
found_property(title, 'title', 0.9)
season = [ node for node in mtree.leaves() if 'season' in node.guess ]
if season and 'bonusNumber' in mtree.info:
series = previous_group(season[0])
if same_group(series, season[0]):
found_property(series, 'series', 0.9)
+37
View File
@@ -0,0 +1,37 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.transfo import SingleNodeGuesser
from guessit.date import search_date
import logging
log = logging.getLogger("guessit.transfo.guess_date")
def guess_date(string):
date, span = search_date(string)
if date:
return { 'date': date }, span
else:
return None, None
def process(mtree):
SingleNodeGuesser(guess_date, 1.0, log).process(mtree)
@@ -0,0 +1,142 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.transfo import found_property
from guessit.patterns import non_episode_title, unlikely_series
import logging
log = logging.getLogger("guessit.transfo.guess_episode_info_from_position")
def match_from_epnum_position(mtree, node):
epnum_idx = node.node_idx
# a few helper functions to be able to filter using high-level semantics
def before_epnum_in_same_pathgroup():
return [ leaf for leaf in mtree.unidentified_leaves()
if (leaf.node_idx[0] == epnum_idx[0] and
leaf.node_idx[1:] < epnum_idx[1:]) ]
def after_epnum_in_same_pathgroup():
return [ leaf for leaf in mtree.unidentified_leaves()
if (leaf.node_idx[0] == epnum_idx[0] and
leaf.node_idx[1:] > epnum_idx[1:]) ]
def after_epnum_in_same_explicitgroup():
return [ leaf for leaf in mtree.unidentified_leaves()
if (leaf.node_idx[:2] == epnum_idx[:2] and
leaf.node_idx[2:] > epnum_idx[2:]) ]
# epnumber is the first group and there are only 2 after it in same
# path group
# -> series title - episode title
title_candidates = [ n for n in after_epnum_in_same_pathgroup()
if n.clean_value.lower() not in non_episode_title ]
if ('title' not in mtree.info and # no title
before_epnum_in_same_pathgroup() == [] and # no groups before
len(title_candidates) == 2): # only 2 groups after
found_property(title_candidates[0], 'series', confidence=0.4)
found_property(title_candidates[1], 'title', confidence=0.4)
return
# if we have at least 1 valid group before the episodeNumber, then it's
# probably the series name
series_candidates = before_epnum_in_same_pathgroup()
if len(series_candidates) >= 1:
found_property(series_candidates[0], 'series', confidence=0.7)
# only 1 group after (in the same path group) and it's probably the
# episode title
title_candidates = [ n for n in after_epnum_in_same_pathgroup()
if n.clean_value.lower() not in non_episode_title ]
if len(title_candidates) == 1:
found_property(title_candidates[0], 'title', confidence=0.5)
return
else:
# try in the same explicit group, with lower confidence
title_candidates = [ n for n in after_epnum_in_same_explicitgroup()
if n.clean_value.lower() not in non_episode_title
]
if len(title_candidates) == 1:
found_property(title_candidates[0], 'title', confidence=0.4)
return
elif len(title_candidates) > 1:
found_property(title_candidates[0], 'title', confidence=0.3)
return
# get the one with the longest value
title_candidates = [ n for n in after_epnum_in_same_pathgroup()
if n.clean_value.lower() not in non_episode_title ]
if title_candidates:
maxidx = -1
maxv = -1
for i, c in enumerate(title_candidates):
if len(c.clean_value) > maxv:
maxidx = i
maxv = len(c.clean_value)
found_property(title_candidates[maxidx], 'title', confidence=0.3)
def process(mtree):
eps = [node for node in mtree.leaves() if 'episodeNumber' in node.guess]
if eps:
match_from_epnum_position(mtree, eps[0])
else:
# if we don't have the episode number, but at least 2 groups in the
# basename, then it's probably series - eptitle
basename = mtree.node_at((-2,))
title_candidates = [ n for n in basename.unidentified_leaves()
if n.clean_value.lower() not in non_episode_title
]
if len(title_candidates) >= 2:
found_property(title_candidates[0], 'series', 0.4)
found_property(title_candidates[1], 'title', 0.4)
# if we only have 1 remaining valid group in the folder containing the
# file, then it's likely that it is the series name
try:
series_candidates = mtree.node_at((-3,)).unidentified_leaves()
except ValueError:
series_candidates = []
if len(series_candidates) == 1:
found_property(series_candidates[0], 'series', 0.3)
# if there's a path group that only contains the season info, then the
# previous one is most likely the series title (ie: ../series/season X/..)
eps = [ node for node in mtree.nodes()
if 'season' in node.guess and 'episodeNumber' not in node.guess ]
if eps:
previous = [ node for node in mtree.unidentified_leaves()
if node.node_idx[0] == eps[0].node_idx[0] - 1 ]
if len(previous) == 1:
found_property(previous[0], 'series', 0.5)
# reduce the confidence of unlikely series
for node in mtree.nodes():
if 'series' in node.guess:
if node.guess['series'].lower() in unlikely_series:
new_confidence = node.guess.confidence('series') * 0.5
node.guess.set_confidence('series', new_confidence)
@@ -0,0 +1,42 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
from guessit.transfo import SingleNodeGuesser
from guessit.patterns import episode_rexps
import re
import logging
log = logging.getLogger("guessit.transfo.guess_episodes_rexps")
def guess_episodes_rexps(string):
for rexp, confidence, span_adjust in episode_rexps:
match = re.search(rexp, string, re.IGNORECASE)
if match:
return (Guess(match.groupdict(), confidence=confidence),
(match.start() + span_adjust[0],
match.end() + span_adjust[1]))
return None, None
def process(mtree):
SingleNodeGuesser(guess_episodes_rexps, None, log).process(mtree)
+113
View File
@@ -0,0 +1,113 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
from guessit.patterns import (subtitle_exts, video_exts, episode_rexps,
find_properties, canonical_form)
import os.path
import re
import mimetypes
import logging
log = logging.getLogger("guessit.transfo.guess_filetype")
def guess_filetype(filename, filetype):
other = {}
# look at the extension first
fileext = os.path.splitext(filename)[1][1:].lower()
if fileext in subtitle_exts:
if 'movie' in filetype:
filetype = 'moviesubtitle'
elif 'episode' in filetype:
filetype = 'episodesubtitle'
else:
filetype = 'subtitle'
other = { 'container': fileext }
elif fileext in video_exts:
if filetype == 'autodetect':
filetype = 'video'
other = { 'container': fileext }
else:
if filetype == 'autodetect':
filetype = 'unknown'
other = { 'extension': fileext }
# put the filetype inside a dummy container to be able to have the
# following functions work correctly as closures
# this is a workaround for python 2 which doesn't have the
# 'nonlocal' keyword (python 3 does have it)
filetype_container = [filetype]
def upgrade_episode():
if filetype_container[0] == 'video':
filetype_container[0] = 'episode'
elif filetype_container[0] == 'subtitle':
filetype_container[0] = 'episodesubtitle'
def upgrade_movie():
if filetype_container[0] == 'video':
filetype_container[0] = 'movie'
elif filetype_container[0] == 'subtitle':
filetype_container[0] = 'moviesubtitle'
# now look whether there are some specific hints for episode vs movie
if filetype in ('video', 'subtitle'):
for rexp, _, _ in episode_rexps:
match = re.search(rexp, filename, re.IGNORECASE)
if match:
upgrade_episode()
break
for prop, value, _, _ in find_properties(filename):
log.debug('prop: %s = %s' % (prop, value))
if prop == 'episodeFormat':
upgrade_episode()
break
elif canonical_form(value) == 'DVB':
upgrade_episode()
break
# if no episode info found, assume it's a movie
upgrade_movie()
filetype = filetype_container[0]
return filetype, other
def process(mtree, filetype='autodetect'):
filetype, other = guess_filetype(mtree.string, filetype)
mtree.guess.set('type', filetype, confidence=1.0)
log.debug('Found with confidence %.2f: %s' % (1.0, mtree.guess))
filetype_info = Guess(other, confidence=1.0)
# guess the mimetype of the filename
# TODO: handle other mimetypes not found on the default type_maps
# mimetypes.types_map['.srt']='text/subtitle'
mime, _ = mimetypes.guess_type(mtree.string, strict=False)
if mime is not None:
filetype_info.update({'mimetype': mime}, confidence=1.0)
node_ext = mtree.node_at((-1,))
node_ext.guess = filetype_info
log.debug('Found with confidence %.2f: %s' % (1.0, node_ext.guess))
+47
View File
@@ -0,0 +1,47 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
from guessit.transfo import SingleNodeGuesser
from guessit.language import search_language
from guessit.textutils import clean_string
import logging
log = logging.getLogger("guessit.transfo.guess_language")
def guess_language(string):
language, span, confidence = search_language(string)
if language:
# is it a subtitle language?
if 'sub' in clean_string(string[:span[0]]).lower().split(' '):
return (Guess({'subtitleLanguage': language},
confidence=confidence),
span)
else:
return (Guess({'language': language},
confidence=confidence),
span)
return None, None
def process(mtree):
SingleNodeGuesser(guess_language, None, log).process(mtree)
@@ -0,0 +1,171 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
import logging
log = logging.getLogger("guessit.transfo.guess_movie_title_from_position")
def process(mtree):
def found_property(node, name, value, confidence):
node.guess = Guess({ name: value },
confidence=confidence)
log.debug('Found with confidence %.2f: %s' % (confidence, node.guess))
def found_title(node, confidence):
found_property(node, 'title', node.clean_value, confidence)
basename = mtree.node_at((-2,))
all_valid = lambda leaf: len(leaf.clean_value) > 0
basename_leftover = basename.unidentified_leaves(valid=all_valid)
try:
folder = mtree.node_at((-3,))
folder_leftover = folder.unidentified_leaves()
except ValueError:
folder = None
folder_leftover = []
log.debug('folder: %s' % folder_leftover)
log.debug('basename: %s' % basename_leftover)
# specific cases:
# if we find the same group both in the folder name and the filename,
# it's a good candidate for title
if (folder_leftover and basename_leftover and
folder_leftover[0].clean_value == basename_leftover[0].clean_value):
found_title(folder_leftover[0], confidence=0.8)
return
# specific cases:
# if the basename contains a number first followed by an unidentified
# group, and the folder only contains 1 unidentified one, then we have
# a series
# ex: Millenium Trilogy (2009)/(1)The Girl With The Dragon Tattoo(2009).mkv
try:
series = folder_leftover[0]
filmNumber = basename_leftover[0]
title = basename_leftover[1]
basename_leaves = basename.leaves()
num = int(filmNumber.clean_value)
log.debug('series: %s' % series.clean_value)
log.debug('title: %s' % title.clean_value)
if (series.clean_value != title.clean_value and
series.clean_value != filmNumber.clean_value and
basename_leaves.index(filmNumber) == 0 and
basename_leaves.index(title) == 1):
found_title(title, confidence=0.6)
found_property(series, 'filmSeries',
series.clean_value, confidence=0.6)
found_property(filmNumber, 'filmNumber',
num, confidence=0.6)
return
except Exception:
pass
# specific cases:
# - movies/tttttt (yyyy)/tttttt.ccc
try:
if mtree.node_at((-4, 0)).value.lower() == 'movies':
folder = mtree.node_at((-3,))
# Note:too generic, might solve all the unittests as they all
# contain 'movies' in their path
#
#if containing_folder.is_leaf() and not containing_folder.guess:
# containing_folder.guess =
# Guess({ 'title': clean_string(containing_folder.value) },
# confidence=0.7)
year_group = folder.first_leaf_containing('year')
groups_before = folder.previous_unidentified_leaves(year_group)
found_title(groups_before[0], confidence=0.8)
return
except Exception:
pass
# if we have either format or videoCodec in the folder containing the file
# or one of its parents, then we should probably look for the title in
# there rather than in the basename
try:
props = mtree.previous_leaves_containing(mtree.children[-2],
[ 'videoCodec', 'format',
'language' ])
except IndexError:
props = []
if props:
group_idx = props[0].node_idx[0]
if all(g.node_idx[0] == group_idx for g in props):
# if they're all in the same group, take leftover info from there
leftover = mtree.node_at((group_idx,)).unidentified_leaves()
if leftover:
found_title(leftover[0], confidence=0.7)
return
# look for title in basename if there are some remaining undidentified
# groups there
if basename_leftover:
title_candidate = basename_leftover[0]
# if basename is only one word and the containing folder has at least
# 3 words in it, we should take the title from the folder name
# ex: Movies/Alice in Wonderland DVDRip.XviD-DiAMOND/dmd-aw.avi
# ex: Movies/Somewhere.2010.DVDRip.XviD-iLG/i-smwhr.avi <-- TODO: gets caught here?
if (title_candidate.clean_value.count(' ') == 0 and
folder_leftover and
folder_leftover[0].clean_value.count(' ') >= 2):
found_title(folder_leftover[0], confidence=0.7)
return
# if there are only 2 unidentified groups, the first of which is inside
# brackets or parentheses, we take the second one for the title:
# ex: Movies/[阿维达].Avida.2006.FRENCH.DVDRiP.XViD-PROD.avi
if len(basename_leftover) == 2 and basename_leftover[0].is_explicit():
found_title(basename_leftover[1], confidence=0.8)
return
# if all else fails, take the first remaining unidentified group in the
# basename as title
found_title(title_candidate, confidence=0.6)
return
# if there are no leftover groups in the basename, look in the folder name
if folder_leftover:
found_title(folder_leftover[0], confidence=0.5)
return
# if nothing worked, look if we have a very small group at the beginning
# of the basename
basename = mtree.node_at((-2,))
basename_leftover = basename.unidentified_leaves(valid=lambda leaf: True)
if basename_leftover:
found_title(basename_leftover[0], confidence=0.4)
return
+37
View File
@@ -0,0 +1,37 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.transfo import SingleNodeGuesser
from guessit.patterns import find_properties
import logging
log = logging.getLogger("guessit.transfo.guess_properties")
def guess_properties(string):
try:
prop, value, pos, end = find_properties(string)[0]
return { prop: value }, (pos, end)
except IndexError:
return None, None
def process(mtree):
SingleNodeGuesser(guess_properties, 1.0, log).process(mtree)
@@ -0,0 +1,44 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.transfo import SingleNodeGuesser
import re
import logging
log = logging.getLogger("guessit.transfo.guess_release_group")
def guess_release_group(string):
group_names = [ r'\.(Xvid)-(?P<releaseGroup>.*?)[ \.]',
r'\.(DivX)-(?P<releaseGroup>.*?)[\. ]',
r'\.(DVDivX)-(?P<releaseGroup>.*?)[\. ]',
]
for rexp in group_names:
match = re.search(rexp, string, re.IGNORECASE)
if match:
metadata = match.groupdict()
metadata.update({ 'videoCodec': match.group(1) })
return metadata, (match.start() + 1, match.end() - 1)
return None, None
def process(mtree):
SingleNodeGuesser(guess_release_group, 0.8, log).process(mtree)
+48
View File
@@ -0,0 +1,48 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
from guessit.transfo import SingleNodeGuesser
from guessit.patterns import video_rexps, sep
import re
import logging
log = logging.getLogger("guessit.transfo.guess_video_rexps")
def guess_video_rexps(string):
string = '-' + string + '-'
for rexp, confidence, span_adjust in video_rexps:
match = re.search(sep + rexp + sep, string, re.IGNORECASE)
if match:
metadata = match.groupdict()
# is this the better place to put it? (maybe, as it is at least
# the soonest that we can catch it)
if metadata.get('cdNumberTotal', -1) is None:
del metadata['cdNumberTotal']
return (Guess(metadata, confidence=confidence),
(match.start() + span_adjust[0],
match.end() + span_adjust[1] - 2))
return None, None
def process(mtree):
SingleNodeGuesser(guess_video_rexps, None, log).process(mtree)
@@ -0,0 +1,56 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import Guess
from guessit.transfo import SingleNodeGuesser
from guessit.patterns import weak_episode_rexps
import re
import logging
log = logging.getLogger("guessit.transfo.guess_weak_episodes_rexps")
def guess_weak_episodes_rexps(string, node):
if 'episodeNumber' in node.root.info:
return None, None
for rexp, span_adjust in weak_episode_rexps:
match = re.search(rexp, string, re.IGNORECASE)
if match:
metadata = match.groupdict()
span = (match.start() + span_adjust[0],
match.end() + span_adjust[1])
epnum = int(metadata['episodeNumber'])
if epnum > 100:
return Guess({ 'season': epnum // 100,
'episodeNumber': epnum % 100 },
confidence=0.6), span
else:
return Guess(metadata, confidence=0.3), span
return None, None
guess_weak_episodes_rexps.use_node = True
def process(mtree):
SingleNodeGuesser(guess_weak_episodes_rexps, 0.6, log).process(mtree)
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.transfo import SingleNodeGuesser
from guessit.patterns import websites
import logging
log = logging.getLogger("guessit.transfo.guess_website")
def guess_website(string):
low = string.lower()
for site in websites:
pos = low.find(site.lower())
if pos != -1:
return {'website': site}, (pos, pos + len(site))
return None, None
def process(mtree):
SingleNodeGuesser(guess_website, 1.0, log).process(mtree)
+37
View File
@@ -0,0 +1,37 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.transfo import SingleNodeGuesser
from guessit.date import search_year
import logging
log = logging.getLogger("guessit.transfo.guess_year")
def guess_year(string):
year, span = search_year(string)
if year:
return { 'year': year }, span
else:
return None, None
def process(mtree):
SingleNodeGuesser(guess_year, 1.0, log).process(mtree)
+69
View File
@@ -0,0 +1,69 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.patterns import subtitle_exts
import logging
log = logging.getLogger("guessit.transfo.post_process")
def process(mtree):
# 1- try to promote language to subtitle language where it makes sense
for node in mtree.nodes():
if 'language' not in node.guess:
continue
def promote_subtitle():
# pylint: disable=W0631
node.guess.set('subtitleLanguage', node.guess['language'],
confidence=node.guess.confidence('language'))
del node.guess['language']
# - if we matched a language in a file with a sub extension and that
# the group is the last group of the filename, it is probably the
# language of the subtitle
# (eg: 'xxx.english.srt')
if (mtree.node_at((-1,)).value.lower() in subtitle_exts and
node == mtree.leaves()[-2]):
promote_subtitle()
# - if a language is in an explicit group just preceded by "st",
# it is a subtitle language (eg: '...st[fr-eng]...')
try:
idx = node.node_idx
previous = mtree.node_at((idx[0], idx[1] - 1)).leaves()[-1]
if previous.value.lower()[-2:] == 'st':
promote_subtitle()
except IndexError:
pass
# 2- ", the" at the end of a series title should be prepended to it
for node in mtree.nodes():
if 'series' not in node.guess:
continue
series = node.guess['series']
lseries = series.lower()
if lseries[-4:] == ',the':
node.guess['series'] = 'The ' + series[:-4]
if lseries[-5:] == ', the':
node.guess['series'] = 'The ' + series[:-5]
@@ -0,0 +1,42 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.textutils import find_first_level_groups
from guessit.patterns import group_delimiters
import logging
log = logging.getLogger("guessit.transfo.split_explicit_groups")
def process(mtree):
"""return the string split into explicit groups, that is, those either
between parenthese, square brackets or curly braces, and those separated
by a dash."""
for c in mtree.children:
groups = find_first_level_groups(c.value, group_delimiters[0])
for delimiters in group_delimiters:
flatten = lambda l, x: l + find_first_level_groups(x, delimiters)
groups = reduce(flatten, groups, [])
# do not do this at this moment, it is not strong enough and can break other
# patterns, such as dates, etc...
#groups = reduce(lambda l, x: l + x.split('-'), groups, [])
c.split_on_components(groups)
+51
View File
@@ -0,0 +1,51 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.patterns import sep
import re
import logging
log = logging.getLogger("guessit.transfo.split_on_dash")
def process(mtree):
for node in mtree.unidentified_leaves():
indices = []
didx = 0
pattern = re.compile(sep + '-' + sep)
match = pattern.search(node.value)
while match:
span = match.span()
indices.extend([ span[0], span[1] ])
match = pattern.search(node.value, span[1])
didx = node.value.find('-')
while didx > 0:
if (didx > 10 and
(didx - 1 not in indices and
didx + 2 not in indices)):
indices.extend([ didx, didx + 1 ])
didx = node.value.find('-', didx + 1)
if indices:
node.partition(indices)
@@ -0,0 +1,35 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2012 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit import fileutils
import os.path
import logging
log = logging.getLogger("guessit.transfo.split_path_components")
def process(mtree):
"""Returns the filename split into [ dir*, basename, ext ]."""
components = fileutils.split_path(mtree.value)
basename = components.pop(-1)
components += list(os.path.splitext(basename))
components[-1] = components[-1][1:] # remove the '.' from the extension
mtree.split_on_components(components)