Update guessit with unicode fix

This commit is contained in:
Ruud
2013-06-22 00:34:58 +02:00
parent b2d9a7675d
commit bc8d8dcd04
35 changed files with 517 additions and 218 deletions
Regular → Executable
+1 -1
View File
@@ -45,7 +45,7 @@ def format_guess(guess):
elif isinstance(value, base_text_type):
if prop in ('edition',):
value = clean_string(value)
guess[prop] = canonical_form(value)
guess[prop] = canonical_form(value).replace('\\', '')
return guess
View File
+15 -9
View File
@@ -19,24 +19,30 @@
#
from __future__ import unicode_literals
#from guessit.transfo import SingleNodeGuesser
#from guessit.date import search_year
from guessit.country import Country
from guessit import Guess
import logging
log = logging.getLogger(__name__)
# list of common words which could be interpreted as countries, but which
# are far too common to be able to say they represent a country
country_common_words = frozenset([ 'bt', 'bb' ])
def process(mtree):
for node in mtree.unidentified_leaves():
# only keep explicit groups (enclosed in parentheses/brackets)
if len(node.node_idx) == 2:
try:
country = Country(node.value[1:-1], strict=True)
if node.value[0] + node.value[-1] not in ['()', '[]', '{}']:
continue
node.guess = Guess(country=country, confidence=1.0)
c = node.value[1:-1].lower()
if c in country_common_words:
continue
# only keep explicit groups (enclosed in parentheses/brackets)
if node.value[0] + node.value[-1] not in ['()', '[]', '{}']:
continue
try:
country = Country(c, strict=True)
except ValueError:
pass
continue
node.guess = Guess(country=country, confidence=1.0)
View File
View File
View File
+3 -3
View File
@@ -21,7 +21,7 @@
from __future__ import unicode_literals
from guessit import Guess
from guessit.patterns import (subtitle_exts, video_exts, episode_rexps,
find_properties, canonical_form)
find_properties, compute_canonical_form)
from guessit.date import valid_year
from guessit.textutils import clean_string
import os.path
@@ -89,7 +89,7 @@ def guess_filetype(mtree, filetype):
# check whether we are in a 'Movies', 'Tv Shows', ... folder
folder_rexps = [ (r'Movies?', upgrade_movie),
(r'Tv ?Shows?', upgrade_episode),
(r'Tv[ _-]?Shows?', upgrade_episode),
(r'Series', upgrade_episode)
]
for frexp, upgrade_func in folder_rexps:
@@ -142,7 +142,7 @@ def guess_filetype(mtree, filetype):
upgrade_episode()
break
elif canonical_form(value) == 'DVB':
elif compute_canonical_form('format', value) == 'DVB':
upgrade_episode()
break
+5 -10
View File
@@ -22,7 +22,7 @@ from __future__ import unicode_literals
from guessit import Guess
from guessit.transfo import SingleNodeGuesser
from guessit.language import search_language
from guessit.textutils import clean_string
from guessit.textutils import clean_string, find_words
import logging
log = logging.getLogger(__name__)
@@ -31,18 +31,13 @@ log = logging.getLogger(__name__)
def guess_language(string):
language, span, confidence = search_language(string)
if language:
# is it a subtitle language?
if 'sub' in clean_string(string[:span[0]]).lower().split(' '):
return (Guess({'subtitleLanguage': language},
confidence=confidence),
span)
else:
return (Guess({'language': language},
confidence=confidence),
span)
return (Guess({'language': language},
confidence=confidence),
span)
return None, None
def process(mtree):
SingleNodeGuesser(guess_language, None, log).process(mtree)
# Note: 'language' is promoted to 'subtitleLanguage' in the post_process transfo
+1
View File
@@ -20,6 +20,7 @@
from __future__ import unicode_literals
from guessit import Guess
import unicodedata
import logging
log = logging.getLogger(__name__)
View File
+26 -24
View File
@@ -20,49 +20,51 @@
from __future__ import unicode_literals
from guessit.transfo import SingleNodeGuesser
from guessit.patterns import properties, canonical_form
from guessit.patterns import prop_multi, compute_canonical_form, _dash, _psep
import re
import logging
log = logging.getLogger(__name__)
def get_patterns(property_name):
return [ p.replace(_dash, _psep) for patterns in prop_multi[property_name].values() for p in patterns ]
CODECS = properties['videoCodec']
FORMATS = properties['format']
CODECS = get_patterns('videoCodec')
FORMATS = get_patterns('format')
GROUP_NAMES = [ r'(?P<videoCodec>' + codec + r')-?(?P<releaseGroup>.*?)[ \.]'
for codec in CODECS ]
GROUP_NAMES += [ r'(?P<format>' + fmt + r')-?(?P<releaseGroup>.*?)[ \.]'
for fmt in FORMATS ]
GROUP_NAMES2 = [ r'\.(?P<videoCodec>' + codec + r')-(?P<releaseGroup>.*?)(-(.*?))?[ \.]'
for codec in CODECS ]
GROUP_NAMES2 += [ r'\.(?P<format>' + fmt + r')-(?P<releaseGroup>.*?)(-(.*?))?[ \.]'
for fmt in FORMATS ]
GROUP_NAMES = [ re.compile(r, re.IGNORECASE) for r in GROUP_NAMES ]
GROUP_NAMES2 = [ re.compile(r, re.IGNORECASE) for r in GROUP_NAMES2 ]
def adjust_metadata(md):
codec = canonical_form(md['videoCodec'])
if codec in FORMATS:
md['format'] = codec
del md['videoCodec']
return md
return dict((property_name, compute_canonical_form(property_name, value) or value)
for property_name, value in md.items())
def guess_release_group(string):
group_names = [ r'\.(Xvid)-(?P<releaseGroup>.*?)[ \.]',
r'\.(DivX)-(?P<releaseGroup>.*?)[\. ]',
r'\.(DVDivX)-(?P<releaseGroup>.*?)[\. ]',
]
# first try to see whether we have both a known codec and a known release group
group_names = [ r'\.(?P<videoCodec>' + codec + r')-(?P<releaseGroup>.*?)[ \.]'
for codec in (CODECS + FORMATS) ]
for rexp in group_names:
match = re.search(rexp, string, re.IGNORECASE)
for rexp in GROUP_NAMES:
match = rexp.search(string)
if match:
metadata = match.groupdict()
if canonical_form(metadata['releaseGroup']) in properties['releaseGroup']:
release_group = compute_canonical_form('releaseGroup', metadata['releaseGroup'])
if release_group:
return adjust_metadata(metadata), (match.start(1), match.end(2))
# pick anything as releaseGroup as long as we have a codec in front
# this doesn't include a potential dash ('-') ending the release group
# eg: [...].X264-HiS@SiLUHD-English.[...]
group_names = [ r'\.(?P<videoCodec>' + codec + r')-(?P<releaseGroup>.*?)(-(.*?))?[ \.]'
for codec in (CODECS + FORMATS) ]
for rexp in group_names:
match = re.search(rexp, string, re.IGNORECASE)
for rexp in GROUP_NAMES2:
match = rexp.search(string)
if match:
return adjust_metadata(match.groupdict()), (match.start(1), match.end(2))
View File
View File
View File
View File
+11 -8
View File
@@ -20,6 +20,7 @@
from __future__ import unicode_literals
from guessit.patterns import subtitle_exts
from guessit.textutils import reorder_title, find_words
import logging
log = logging.getLogger(__name__)
@@ -45,6 +46,15 @@ def process(mtree):
node == mtree.leaves()[-2]):
promote_subtitle()
# - if we find the word 'sub' before the language, and in the same explicit
# group, then upgrade the language
explicit_group = mtree.node_at(node.node_idx[:2])
group_str = explicit_group.value.lower()
if ('sub' in find_words(group_str) and
0 <= group_str.find('sub') < (node.span[0] - explicit_group.span[0])):
promote_subtitle()
# - if a language is in an explicit group just preceded by "st",
# it is a subtitle language (eg: '...st[fr-eng]...')
try:
@@ -60,11 +70,4 @@ def process(mtree):
if 'series' not in node.guess:
continue
series = node.guess['series']
lseries = series.lower()
if lseries[-4:] == ',the':
node.guess['series'] = 'The ' + series[:-4]
if lseries[-5:] == ', the':
node.guess['series'] = 'The ' + series[:-5]
node.guess['series'] = reorder_title(node.guess['series'])
View File
-10
View File
@@ -38,15 +38,5 @@ def process(mtree):
indices.extend([ span[0], span[1] ])
match = pattern.search(node.value, span[1])
didx = node.value.find('-')
while didx > 0:
if (didx > 10 and
(didx - 1 not in indices and
didx + 2 not in indices)):
indices.extend([ didx, didx + 1 ])
didx = node.value.find('-', didx + 1)
if indices:
node.partition(indices)
View File