New metadata scanner

This commit is contained in:
Ruud
2012-01-27 18:54:12 +01:00
parent fe7ac58bf5
commit 228f7e9e9e
192 changed files with 193 additions and 44918 deletions
+4 -1
View File
@@ -18,7 +18,7 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
__version__ = '0.2'
__version__ = '0.3-dev'
__all__ = [ 'Guess', 'Language',
'guess_file_info', 'guess_video_info',
'guess_movie_info', 'guess_episode_info' ]
@@ -52,6 +52,9 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
result = []
hashers = []
if isinstance(info, basestring):
info = [ info ]
for infotype in info:
if infotype == 'filename':
m = IterativeMatcher(filename, filetype = filetype)
+76
View File
@@ -0,0 +1,76 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
# the Free Software Foundation; either version 3 of the License, or
# (at your option) any later version.
#
# GuessIt is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# Lesser GNU General Public License for more details.
#
# You should have received a copy of the Lesser GNU General Public License
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
from guessit.patterns import subtitle_exts, video_exts, episode_rexps, find_properties, canonical_form
import os.path
import re
import logging
log = logging.getLogger("guessit.filetype")
def guess_filetype(filename, filetype = 'autodetect'):
other = {}
# look at the extension first
fileext = os.path.splitext(filename)[1][1:].lower()
if fileext in subtitle_exts:
if 'movie' in filetype:
filetype = 'moviesubtitle'
elif 'episode' in filetype:
filetype = 'episodesubtitle'
else:
filetype = 'subtitle'
other = { 'container': fileext }
elif fileext in video_exts:
if filetype == 'autodetect':
filetype = 'video'
other = { 'container': fileext }
else:
if filetype == 'autodetect':
filetype = 'unknown'
other = { 'extension': fileext }
# now look whether there are some specific hints for episode vs movie
if filetype in ('video', 'subtitle'):
for rexp, confidence, span_adjust in episode_rexps:
match = re.search(rexp, filename, re.IGNORECASE)
if match:
if filetype == 'video':
filetype = 'episode'
elif filetype == 'subtitle':
filetype = 'episodesubtitle'
break
for prop, value, start, end in find_properties(filename):
if canonical_form(value) == 'DVB':
if filetype == 'video':
filetype = 'episode'
elif filetype == 'subtitle':
filetype = 'episodesubtitle'
break
# if no episode info found, assume it's a movie
if filetype == 'video':
filetype = 'movie'
elif filetype == 'subtitle':
filetype = 'moviesubtitle'
return filetype, other
+16 -1
View File
@@ -18,7 +18,9 @@
# along with this program. If not, see <http://www.gnu.org/licenses/>.
#
import ntpath
import os.path
import zipfile
def split_path(path):
@@ -44,7 +46,7 @@ def split_path(path):
"""
result = []
while True:
head, tail = os.path.split(path)
head, tail = ntpath.split(path)
# on Unix systems, the root folder is '/'
if head == '/' and tail == '':
@@ -81,3 +83,16 @@ def file_in_same_dir(ref_file, desired_file):
"""
return os.path.join(*(split_path(ref_file)[:-1] + [ desired_file ]))
def load_file_in_same_dir(ref_file, filename):
"""Load a given file. Works even when the file is contained inside a zip."""
path = split_path(ref_file)[:-1] + [ filename ]
for i, p in enumerate(path):
if p.endswith('.zip'):
zfilename = os.path.join(*path[:i+1])
zfile = zipfile.ZipFile(zfilename)
return zfile.read('/'.join(path[i+1:]))
return open(os.path.join(*path)).read()
+2 -1
View File
@@ -33,7 +33,8 @@ log = logging.getLogger('guessit.language')
# "An alpha-3 (bibliographic) code, an alpha-3 (terminologic) code (when given),
# an alpha-2 code (when given), an English name, and a French name of a language
# are all separated by pipe (|) characters."
language_matrix = [ l.strip().decode('utf-8').split('|') for l in open(fileutils.file_in_same_dir(__file__, 'ISO-639-2_utf-8.txt')) ]
language_matrix = [ l.strip().decode('utf-8').split('|')
for l in fileutils.load_file_in_same_dir(__file__, 'ISO-639-2_utf-8.txt').split('\n') ]
lng3 = frozenset(filter(bool, (l[0] for l in language_matrix)))
lng3term = frozenset(filter(bool, (l[1] for l in language_matrix)))
+32 -55
View File
@@ -3,6 +3,7 @@
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
# Copyright (c) 2011 Ricard Marxer <ricardmp@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
@@ -22,7 +23,8 @@ from guessit import fileutils, textutils
from guessit.guess import Guess, merge_similar_guesses, merge_all, choose_int, choose_string
from guessit.date import search_date, search_year
from guessit.language import search_language
from guessit.patterns import video_exts, subtitle_exts, sep, deleted, video_rexps, websites, episode_rexps, weak_episode_rexps, non_episode_title, properties, canonical_form
from guessit.filetype import guess_filetype
from guessit.patterns import video_exts, subtitle_exts, sep, deleted, video_rexps, websites, episode_rexps, weak_episode_rexps, non_episode_title, find_properties, canonical_form, unlikely_series
from guessit.matchtree import get_group, find_group, leftover_valid_groups, tree_to_string
from guessit.textutils import find_first_level_groups, split_on_groups, blank_region, clean_string, to_utf8
from guessit.fileutils import split_path_components
@@ -31,6 +33,7 @@ import os.path
import re
import copy
import logging
import mimetypes
log = logging.getLogger("guessit.matcher")
@@ -148,22 +151,11 @@ def guess_groups(string, result, filetype):
# common well-defined words and regexps
clow = current.lower()
confidence = 1.0 # for all of them
for prop, values in properties.items():
for value in values:
pos = clow.find(value.lower())
if pos != -1:
end = pos + len(value)
# make sure our word is always surrounded by separators
if clow[pos-1] not in sep or clow[end] not in sep:
# note: sep is a regexp, but in this case using it as
# a sequence achieves the same goal
continue
for prop, value, pos, end in find_properties(current):
guess = guessed({ prop: value }, confidence = confidence)
current = update_found(current, guess, (pos, end))
guess = guessed({ prop: value }, confidence = confidence)
current = update_found(current, guess, (pos, end))
clow = current.lower()
# weak guesses for episode number, only run it if we don't have an estimate already
if filetype in ('episode', 'episodesubtitle'):
@@ -341,9 +333,10 @@ class IterativeMatcher(object):
resolution when they arise.
"""
if filetype not in ('autodetect', 'subtitle', 'movie', 'moviesubtitle',
if filetype not in ('autodetect', 'subtitle', 'video',
'movie', 'moviesubtitle',
'episode', 'episodesubtitle'):
raise ValueError, "filetype needs to be one of ('autodetect', 'subtitle', 'movie', 'moviesubtitle', 'episode', 'episodesubtitle')"
raise ValueError, "filetype needs to be one of ('autodetect', 'subtitle', 'video', 'movie', 'moviesubtitle', 'episode', 'episodesubtitle')"
if not isinstance(filename, unicode):
log.debug('WARNING: given filename to matcher is not unicode...')
@@ -368,43 +361,21 @@ class IterativeMatcher(object):
# 1- first split our path into dirs + basename + ext
match_tree = split_path_components(filename)
fileext = match_tree.pop(-1)[1:].lower()
if fileext in subtitle_exts:
if 'movie' in filetype:
filetype = 'moviesubtitle'
elif 'episode' in filetype:
filetype = 'episodesubtitle'
else:
filetype = 'subtitle'
extguess = guessed({ 'container': fileext }, confidence = 1.0)
elif fileext in video_exts:
extguess = guessed({ 'container': fileext }, confidence = 1.0)
else:
extguess = guessed({ 'extension': fileext}, confidence = 1.0)
# TODO: depending on the extension, we could already grab some info and maybe specialized
# guessers, eg: a lang parser for idx files, an automatic detection of the language
# for srt files, a video metadata extractor for avi, mkv, ...
# if we are on autodetect, try to do it now so we can tell the
# guess_groups function what type of info it should be looking for
if filetype in ('autodetect', 'subtitle'):
for rexp, confidence, span_adjust in episode_rexps:
match = re.search(rexp, filename, re.IGNORECASE)
if match:
if filetype == 'autodetect':
filetype = 'episode'
elif filetype == 'subtitle':
filetype = 'episodesubtitle'
break
# if no episode info found, assume it's a movie
if filetype == 'autodetect':
filetype = 'movie'
elif filetype == 'subtitle':
filetype = 'moviesubtitle'
# try to detect the file type
filetype, other = guess_filetype(filename, filetype)
guessed({ 'type': filetype }, confidence = 1.0)
extguess = guessed(other, confidence = 1.0)
# guess the mimetype of the filename
# TODO: handle other mimetypes not found on the default type_maps
# mimetypes.types_map['.srt']='text/subtitle'
mime, _ = mimetypes.guess_type(filename, strict=False)
if mime is not None:
guessed({ 'mimetype': mime }, confidence = 1.0)
# remove the extension from the match tree, as all indices relative
# the the filename groups assume the basename is the last one
fileext = match_tree.pop(-1)[1:].lower()
# 2- split each of those into explicit groups, if any
@@ -453,8 +424,14 @@ class IterativeMatcher(object):
if len(previous) == 1:
guess = guessed({ 'series': previous[0][0] }, confidence = 0.5)
leftover = update_found(leftover, previous[0][1], guess)
# reduce the confidence of unlikely series
for guess in result:
if 'series' in guess:
if guess['series'].lower() in unlikely_series:
guess.set_confidence('series', guess.confidence('series') * 0.5)
elif filetype in ('movie', 'moviesubtitle'):
leftover_all = leftover_valid_groups(match_tree)
+34 -7
View File
@@ -3,6 +3,7 @@
#
# GuessIt - A library for guessing information from filenames
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
# Copyright (c) 2011 Ricard Marxer <ricardmp@gmail.com>
#
# GuessIt is free software; you can redistribute it and/or modify it under
# the terms of the Lesser GNU General Public License as published by
@@ -19,9 +20,9 @@
#
subtitle_exts = [ 'srt', 'idx', 'sub' ]
subtitle_exts = [ 'srt', 'idx', 'sub', 'ssa', 'txt' ]
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'mov', 'ogg', 'ogm', 'ogv', 'wmv' ]
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'm4v', 'mov', 'ogg', 'ogm', 'ogv', 'wmv', 'divx' ]
# separator character regexp
sep = r'[][)(}{+ \._-]' # regexp art, hehe :D
@@ -34,6 +35,9 @@ episode_rexps = [ # ... Season 2 ...
(r'season (?P<season>[0-9]+)', 1.0, (0, 0)),
(r'saison (?P<season>[0-9]+)', 1.0, (0, 0)),
# ... s02-x01 ...
(r's(?P<season>[0-9]{1,2})-x(?P<bonusNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
# ... s02e13 ...
(r'[Ss](?P<season>[0-9]{1,2}).{,3}[EeXx](?P<episodeNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
@@ -41,7 +45,7 @@ episode_rexps = [ # ... Season 2 ...
(r'[^0-9](?P<season>[0-9]{1,2})[x\.](?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
# ... s02 ...
(sep + r's(?P<season>[0-9]{1,2})' + sep + '?', 0.6, (0, 0)),
(sep + r's(?P<season>[0-9]{1,2})' + sep + '?', 0.6, (1, -1)),
# v2 or v3 for some mangas which have multiples rips
(sep + r'(?P<episodeNumber>[0-9]{1,3})v[23]' + sep, 0.6, (0, 0)),
@@ -77,11 +81,11 @@ video_rexps = [ # cd number
websites = [ 'tvu.org.ru', 'emule-island.com', 'UsaBit.com', 'www.divx-overnet.com', 'sharethefiles.com' ]
properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'Blu-ray', 'BDRip', 'BRRip',
'HDRip', 'DVD', 'DVDivX', 'HDTV', 'DVB', 'WEBRip', 'DVDSCR', 'Screener', 'VHS',
'VIDEO_TS' ],
unlikely_series = ['series']
'container': [ 'avi', 'mkv', 'ogv', 'ogm', 'wmv', 'mp4', 'mov' ],
properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'Blu-ray', 'BDRip', 'BRRip',
'HDRip', 'DVD', 'DVDivX', 'HDTV', 'DVB', 'DVBRip', 'PDTV', 'WEBRip',
'DVDSCR', 'Screener', 'VHS', 'VIDEO_TS' ],
'screenSize': [ '720p', '720' ],
@@ -106,10 +110,29 @@ properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'B
],
}
def find_properties(filename):
result = []
clow = filename.lower()
for prop, values in properties.items():
for value in values:
pos = clow.find(value.lower())
if pos != -1:
end = pos + len(value)
# make sure our word is always surrounded by separators
if ((pos > 0 and clow[pos-1] not in sep) or
(end < len(clow) and clow[end] not in sep)):
# note: sep is a regexp, but in this case using it as
# a sequence achieves the same goal
continue
result.append((prop, value, pos, end))
return result
property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
'HD-DVD': [ 'HDDVD', 'HDDVDRip' ],
'BluRay': [ 'BDRip', 'BRRip', 'Blu-ray' ],
'DVB': [ 'DVBRip', 'PDTV' ],
'Screener': [ 'DVDSCR' ],
'DivX': [ 'DVDivX' ],
'h264': [ 'x264' ],
@@ -123,6 +146,10 @@ property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
reverse_synonyms = {}
for prop, values in properties.items():
for value in values:
reverse_synonyms[value.lower()] = value
for canonical, synonyms in property_synonyms.items():
for synonym in synonyms:
reverse_synonyms[synonym.lower()] = canonical