Update guessit with unicode fix
This commit is contained in:
Regular → Executable
+1
-1
@@ -45,7 +45,7 @@ def format_guess(guess):
|
||||
elif isinstance(value, base_text_type):
|
||||
if prop in ('edition',):
|
||||
value = clean_string(value)
|
||||
guess[prop] = canonical_form(value)
|
||||
guess[prop] = canonical_form(value).replace('\\', '')
|
||||
|
||||
return guess
|
||||
|
||||
|
||||
Regular → Executable
Regular → Executable
+15
-9
@@ -19,24 +19,30 @@
|
||||
#
|
||||
|
||||
from __future__ import unicode_literals
|
||||
#from guessit.transfo import SingleNodeGuesser
|
||||
#from guessit.date import search_year
|
||||
from guessit.country import Country
|
||||
from guessit import Guess
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# list of common words which could be interpreted as countries, but which
|
||||
# are far too common to be able to say they represent a country
|
||||
country_common_words = frozenset([ 'bt', 'bb' ])
|
||||
|
||||
def process(mtree):
|
||||
for node in mtree.unidentified_leaves():
|
||||
# only keep explicit groups (enclosed in parentheses/brackets)
|
||||
if len(node.node_idx) == 2:
|
||||
try:
|
||||
country = Country(node.value[1:-1], strict=True)
|
||||
if node.value[0] + node.value[-1] not in ['()', '[]', '{}']:
|
||||
continue
|
||||
node.guess = Guess(country=country, confidence=1.0)
|
||||
c = node.value[1:-1].lower()
|
||||
if c in country_common_words:
|
||||
continue
|
||||
|
||||
# only keep explicit groups (enclosed in parentheses/brackets)
|
||||
if node.value[0] + node.value[-1] not in ['()', '[]', '{}']:
|
||||
continue
|
||||
|
||||
try:
|
||||
country = Country(c, strict=True)
|
||||
except ValueError:
|
||||
pass
|
||||
continue
|
||||
|
||||
node.guess = Guess(country=country, confidence=1.0)
|
||||
|
||||
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
+3
-3
@@ -21,7 +21,7 @@
|
||||
from __future__ import unicode_literals
|
||||
from guessit import Guess
|
||||
from guessit.patterns import (subtitle_exts, video_exts, episode_rexps,
|
||||
find_properties, canonical_form)
|
||||
find_properties, compute_canonical_form)
|
||||
from guessit.date import valid_year
|
||||
from guessit.textutils import clean_string
|
||||
import os.path
|
||||
@@ -89,7 +89,7 @@ def guess_filetype(mtree, filetype):
|
||||
|
||||
# check whether we are in a 'Movies', 'Tv Shows', ... folder
|
||||
folder_rexps = [ (r'Movies?', upgrade_movie),
|
||||
(r'Tv ?Shows?', upgrade_episode),
|
||||
(r'Tv[ _-]?Shows?', upgrade_episode),
|
||||
(r'Series', upgrade_episode)
|
||||
]
|
||||
for frexp, upgrade_func in folder_rexps:
|
||||
@@ -142,7 +142,7 @@ def guess_filetype(mtree, filetype):
|
||||
upgrade_episode()
|
||||
break
|
||||
|
||||
elif canonical_form(value) == 'DVB':
|
||||
elif compute_canonical_form('format', value) == 'DVB':
|
||||
upgrade_episode()
|
||||
break
|
||||
|
||||
|
||||
Regular → Executable
+5
-10
@@ -22,7 +22,7 @@ from __future__ import unicode_literals
|
||||
from guessit import Guess
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.language import search_language
|
||||
from guessit.textutils import clean_string
|
||||
from guessit.textutils import clean_string, find_words
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -31,18 +31,13 @@ log = logging.getLogger(__name__)
|
||||
def guess_language(string):
|
||||
language, span, confidence = search_language(string)
|
||||
if language:
|
||||
# is it a subtitle language?
|
||||
if 'sub' in clean_string(string[:span[0]]).lower().split(' '):
|
||||
return (Guess({'subtitleLanguage': language},
|
||||
confidence=confidence),
|
||||
span)
|
||||
else:
|
||||
return (Guess({'language': language},
|
||||
confidence=confidence),
|
||||
span)
|
||||
return (Guess({'language': language},
|
||||
confidence=confidence),
|
||||
span)
|
||||
|
||||
return None, None
|
||||
|
||||
|
||||
def process(mtree):
|
||||
SingleNodeGuesser(guess_language, None, log).process(mtree)
|
||||
# Note: 'language' is promoted to 'subtitleLanguage' in the post_process transfo
|
||||
|
||||
Regular → Executable
+1
@@ -20,6 +20,7 @@
|
||||
|
||||
from __future__ import unicode_literals
|
||||
from guessit import Guess
|
||||
import unicodedata
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
Regular → Executable
Regular → Executable
+26
-24
@@ -20,49 +20,51 @@
|
||||
|
||||
from __future__ import unicode_literals
|
||||
from guessit.transfo import SingleNodeGuesser
|
||||
from guessit.patterns import properties, canonical_form
|
||||
from guessit.patterns import prop_multi, compute_canonical_form, _dash, _psep
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
def get_patterns(property_name):
|
||||
return [ p.replace(_dash, _psep) for patterns in prop_multi[property_name].values() for p in patterns ]
|
||||
|
||||
CODECS = properties['videoCodec']
|
||||
FORMATS = properties['format']
|
||||
CODECS = get_patterns('videoCodec')
|
||||
FORMATS = get_patterns('format')
|
||||
|
||||
GROUP_NAMES = [ r'(?P<videoCodec>' + codec + r')-?(?P<releaseGroup>.*?)[ \.]'
|
||||
for codec in CODECS ]
|
||||
GROUP_NAMES += [ r'(?P<format>' + fmt + r')-?(?P<releaseGroup>.*?)[ \.]'
|
||||
for fmt in FORMATS ]
|
||||
|
||||
GROUP_NAMES2 = [ r'\.(?P<videoCodec>' + codec + r')-(?P<releaseGroup>.*?)(-(.*?))?[ \.]'
|
||||
for codec in CODECS ]
|
||||
GROUP_NAMES2 += [ r'\.(?P<format>' + fmt + r')-(?P<releaseGroup>.*?)(-(.*?))?[ \.]'
|
||||
for fmt in FORMATS ]
|
||||
|
||||
GROUP_NAMES = [ re.compile(r, re.IGNORECASE) for r in GROUP_NAMES ]
|
||||
GROUP_NAMES2 = [ re.compile(r, re.IGNORECASE) for r in GROUP_NAMES2 ]
|
||||
|
||||
def adjust_metadata(md):
|
||||
codec = canonical_form(md['videoCodec'])
|
||||
if codec in FORMATS:
|
||||
md['format'] = codec
|
||||
del md['videoCodec']
|
||||
return md
|
||||
return dict((property_name, compute_canonical_form(property_name, value) or value)
|
||||
for property_name, value in md.items())
|
||||
|
||||
|
||||
def guess_release_group(string):
|
||||
group_names = [ r'\.(Xvid)-(?P<releaseGroup>.*?)[ \.]',
|
||||
r'\.(DivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
r'\.(DVDivX)-(?P<releaseGroup>.*?)[\. ]',
|
||||
]
|
||||
|
||||
# first try to see whether we have both a known codec and a known release group
|
||||
group_names = [ r'\.(?P<videoCodec>' + codec + r')-(?P<releaseGroup>.*?)[ \.]'
|
||||
for codec in (CODECS + FORMATS) ]
|
||||
|
||||
for rexp in group_names:
|
||||
match = re.search(rexp, string, re.IGNORECASE)
|
||||
for rexp in GROUP_NAMES:
|
||||
match = rexp.search(string)
|
||||
if match:
|
||||
metadata = match.groupdict()
|
||||
if canonical_form(metadata['releaseGroup']) in properties['releaseGroup']:
|
||||
release_group = compute_canonical_form('releaseGroup', metadata['releaseGroup'])
|
||||
if release_group:
|
||||
return adjust_metadata(metadata), (match.start(1), match.end(2))
|
||||
|
||||
# pick anything as releaseGroup as long as we have a codec in front
|
||||
# this doesn't include a potential dash ('-') ending the release group
|
||||
# eg: [...].X264-HiS@SiLUHD-English.[...]
|
||||
group_names = [ r'\.(?P<videoCodec>' + codec + r')-(?P<releaseGroup>.*?)(-(.*?))?[ \.]'
|
||||
for codec in (CODECS + FORMATS) ]
|
||||
|
||||
for rexp in group_names:
|
||||
match = re.search(rexp, string, re.IGNORECASE)
|
||||
for rexp in GROUP_NAMES2:
|
||||
match = rexp.search(string)
|
||||
if match:
|
||||
return adjust_metadata(match.groupdict()), (match.start(1), match.end(2))
|
||||
|
||||
|
||||
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
Regular → Executable
+11
-8
@@ -20,6 +20,7 @@
|
||||
|
||||
from __future__ import unicode_literals
|
||||
from guessit.patterns import subtitle_exts
|
||||
from guessit.textutils import reorder_title, find_words
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -45,6 +46,15 @@ def process(mtree):
|
||||
node == mtree.leaves()[-2]):
|
||||
promote_subtitle()
|
||||
|
||||
# - if we find the word 'sub' before the language, and in the same explicit
|
||||
# group, then upgrade the language
|
||||
explicit_group = mtree.node_at(node.node_idx[:2])
|
||||
group_str = explicit_group.value.lower()
|
||||
|
||||
if ('sub' in find_words(group_str) and
|
||||
0 <= group_str.find('sub') < (node.span[0] - explicit_group.span[0])):
|
||||
promote_subtitle()
|
||||
|
||||
# - if a language is in an explicit group just preceded by "st",
|
||||
# it is a subtitle language (eg: '...st[fr-eng]...')
|
||||
try:
|
||||
@@ -60,11 +70,4 @@ def process(mtree):
|
||||
if 'series' not in node.guess:
|
||||
continue
|
||||
|
||||
series = node.guess['series']
|
||||
lseries = series.lower()
|
||||
|
||||
if lseries[-4:] == ',the':
|
||||
node.guess['series'] = 'The ' + series[:-4]
|
||||
|
||||
if lseries[-5:] == ', the':
|
||||
node.guess['series'] = 'The ' + series[:-5]
|
||||
node.guess['series'] = reorder_title(node.guess['series'])
|
||||
|
||||
Regular → Executable
Regular → Executable
-10
@@ -38,15 +38,5 @@ def process(mtree):
|
||||
indices.extend([ span[0], span[1] ])
|
||||
match = pattern.search(node.value, span[1])
|
||||
|
||||
didx = node.value.find('-')
|
||||
while didx > 0:
|
||||
if (didx > 10 and
|
||||
(didx - 1 not in indices and
|
||||
didx + 2 not in indices)):
|
||||
|
||||
indices.extend([ didx, didx + 1 ])
|
||||
|
||||
didx = node.value.find('-', didx + 1)
|
||||
|
||||
if indices:
|
||||
node.partition(indices)
|
||||
|
||||
Regular → Executable
Reference in New Issue
Block a user