New metadata scanner
This commit is contained in:
@@ -18,7 +18,7 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
__version__ = '0.2'
|
||||
__version__ = '0.3-dev'
|
||||
__all__ = [ 'Guess', 'Language',
|
||||
'guess_file_info', 'guess_video_info',
|
||||
'guess_movie_info', 'guess_episode_info' ]
|
||||
@@ -52,6 +52,9 @@ def guess_file_info(filename, filetype, info = [ 'filename' ]):
|
||||
result = []
|
||||
hashers = []
|
||||
|
||||
if isinstance(info, basestring):
|
||||
info = [ info ]
|
||||
|
||||
for infotype in info:
|
||||
if infotype == 'filename':
|
||||
m = IterativeMatcher(filename, filetype = filetype)
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
#!/usr/bin/env python
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
# the Free Software Foundation; either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# GuessIt is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# Lesser GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the Lesser GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
from guessit.patterns import subtitle_exts, video_exts, episode_rexps, find_properties, canonical_form
|
||||
import os.path
|
||||
import re
|
||||
import logging
|
||||
|
||||
log = logging.getLogger("guessit.filetype")
|
||||
|
||||
|
||||
def guess_filetype(filename, filetype = 'autodetect'):
|
||||
other = {}
|
||||
|
||||
# look at the extension first
|
||||
fileext = os.path.splitext(filename)[1][1:].lower()
|
||||
if fileext in subtitle_exts:
|
||||
if 'movie' in filetype:
|
||||
filetype = 'moviesubtitle'
|
||||
elif 'episode' in filetype:
|
||||
filetype = 'episodesubtitle'
|
||||
else:
|
||||
filetype = 'subtitle'
|
||||
other = { 'container': fileext }
|
||||
elif fileext in video_exts:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'video'
|
||||
other = { 'container': fileext }
|
||||
else:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'unknown'
|
||||
other = { 'extension': fileext }
|
||||
|
||||
# now look whether there are some specific hints for episode vs movie
|
||||
if filetype in ('video', 'subtitle'):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, filename, re.IGNORECASE)
|
||||
if match:
|
||||
if filetype == 'video':
|
||||
filetype = 'episode'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'episodesubtitle'
|
||||
break
|
||||
|
||||
for prop, value, start, end in find_properties(filename):
|
||||
if canonical_form(value) == 'DVB':
|
||||
if filetype == 'video':
|
||||
filetype = 'episode'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'episodesubtitle'
|
||||
break
|
||||
|
||||
# if no episode info found, assume it's a movie
|
||||
if filetype == 'video':
|
||||
filetype = 'movie'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'moviesubtitle'
|
||||
|
||||
return filetype, other
|
||||
@@ -18,7 +18,9 @@
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
#
|
||||
|
||||
import ntpath
|
||||
import os.path
|
||||
import zipfile
|
||||
|
||||
|
||||
def split_path(path):
|
||||
@@ -44,7 +46,7 @@ def split_path(path):
|
||||
"""
|
||||
result = []
|
||||
while True:
|
||||
head, tail = os.path.split(path)
|
||||
head, tail = ntpath.split(path)
|
||||
|
||||
# on Unix systems, the root folder is '/'
|
||||
if head == '/' and tail == '':
|
||||
@@ -81,3 +83,16 @@ def file_in_same_dir(ref_file, desired_file):
|
||||
|
||||
"""
|
||||
return os.path.join(*(split_path(ref_file)[:-1] + [ desired_file ]))
|
||||
|
||||
|
||||
def load_file_in_same_dir(ref_file, filename):
|
||||
"""Load a given file. Works even when the file is contained inside a zip."""
|
||||
path = split_path(ref_file)[:-1] + [ filename ]
|
||||
|
||||
for i, p in enumerate(path):
|
||||
if p.endswith('.zip'):
|
||||
zfilename = os.path.join(*path[:i+1])
|
||||
zfile = zipfile.ZipFile(zfilename)
|
||||
return zfile.read('/'.join(path[i+1:]))
|
||||
|
||||
return open(os.path.join(*path)).read()
|
||||
|
||||
@@ -33,7 +33,8 @@ log = logging.getLogger('guessit.language')
|
||||
# "An alpha-3 (bibliographic) code, an alpha-3 (terminologic) code (when given),
|
||||
# an alpha-2 code (when given), an English name, and a French name of a language
|
||||
# are all separated by pipe (|) characters."
|
||||
language_matrix = [ l.strip().decode('utf-8').split('|') for l in open(fileutils.file_in_same_dir(__file__, 'ISO-639-2_utf-8.txt')) ]
|
||||
language_matrix = [ l.strip().decode('utf-8').split('|')
|
||||
for l in fileutils.load_file_in_same_dir(__file__, 'ISO-639-2_utf-8.txt').split('\n') ]
|
||||
|
||||
lng3 = frozenset(filter(bool, (l[0] for l in language_matrix)))
|
||||
lng3term = frozenset(filter(bool, (l[1] for l in language_matrix)))
|
||||
|
||||
+32
-55
@@ -3,6 +3,7 @@
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
# Copyright (c) 2011 Ricard Marxer <ricardmp@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
@@ -22,7 +23,8 @@ from guessit import fileutils, textutils
|
||||
from guessit.guess import Guess, merge_similar_guesses, merge_all, choose_int, choose_string
|
||||
from guessit.date import search_date, search_year
|
||||
from guessit.language import search_language
|
||||
from guessit.patterns import video_exts, subtitle_exts, sep, deleted, video_rexps, websites, episode_rexps, weak_episode_rexps, non_episode_title, properties, canonical_form
|
||||
from guessit.filetype import guess_filetype
|
||||
from guessit.patterns import video_exts, subtitle_exts, sep, deleted, video_rexps, websites, episode_rexps, weak_episode_rexps, non_episode_title, find_properties, canonical_form, unlikely_series
|
||||
from guessit.matchtree import get_group, find_group, leftover_valid_groups, tree_to_string
|
||||
from guessit.textutils import find_first_level_groups, split_on_groups, blank_region, clean_string, to_utf8
|
||||
from guessit.fileutils import split_path_components
|
||||
@@ -31,6 +33,7 @@ import os.path
|
||||
import re
|
||||
import copy
|
||||
import logging
|
||||
import mimetypes
|
||||
|
||||
log = logging.getLogger("guessit.matcher")
|
||||
|
||||
@@ -148,22 +151,11 @@ def guess_groups(string, result, filetype):
|
||||
|
||||
|
||||
# common well-defined words and regexps
|
||||
clow = current.lower()
|
||||
confidence = 1.0 # for all of them
|
||||
for prop, values in properties.items():
|
||||
for value in values:
|
||||
pos = clow.find(value.lower())
|
||||
if pos != -1:
|
||||
end = pos + len(value)
|
||||
# make sure our word is always surrounded by separators
|
||||
if clow[pos-1] not in sep or clow[end] not in sep:
|
||||
# note: sep is a regexp, but in this case using it as
|
||||
# a sequence achieves the same goal
|
||||
continue
|
||||
for prop, value, pos, end in find_properties(current):
|
||||
guess = guessed({ prop: value }, confidence = confidence)
|
||||
current = update_found(current, guess, (pos, end))
|
||||
|
||||
guess = guessed({ prop: value }, confidence = confidence)
|
||||
current = update_found(current, guess, (pos, end))
|
||||
clow = current.lower()
|
||||
|
||||
# weak guesses for episode number, only run it if we don't have an estimate already
|
||||
if filetype in ('episode', 'episodesubtitle'):
|
||||
@@ -341,9 +333,10 @@ class IterativeMatcher(object):
|
||||
resolution when they arise.
|
||||
"""
|
||||
|
||||
if filetype not in ('autodetect', 'subtitle', 'movie', 'moviesubtitle',
|
||||
if filetype not in ('autodetect', 'subtitle', 'video',
|
||||
'movie', 'moviesubtitle',
|
||||
'episode', 'episodesubtitle'):
|
||||
raise ValueError, "filetype needs to be one of ('autodetect', 'subtitle', 'movie', 'moviesubtitle', 'episode', 'episodesubtitle')"
|
||||
raise ValueError, "filetype needs to be one of ('autodetect', 'subtitle', 'video', 'movie', 'moviesubtitle', 'episode', 'episodesubtitle')"
|
||||
if not isinstance(filename, unicode):
|
||||
log.debug('WARNING: given filename to matcher is not unicode...')
|
||||
|
||||
@@ -368,43 +361,21 @@ class IterativeMatcher(object):
|
||||
# 1- first split our path into dirs + basename + ext
|
||||
match_tree = split_path_components(filename)
|
||||
|
||||
fileext = match_tree.pop(-1)[1:].lower()
|
||||
if fileext in subtitle_exts:
|
||||
if 'movie' in filetype:
|
||||
filetype = 'moviesubtitle'
|
||||
elif 'episode' in filetype:
|
||||
filetype = 'episodesubtitle'
|
||||
else:
|
||||
filetype = 'subtitle'
|
||||
extguess = guessed({ 'container': fileext }, confidence = 1.0)
|
||||
elif fileext in video_exts:
|
||||
extguess = guessed({ 'container': fileext }, confidence = 1.0)
|
||||
else:
|
||||
extguess = guessed({ 'extension': fileext}, confidence = 1.0)
|
||||
|
||||
# TODO: depending on the extension, we could already grab some info and maybe specialized
|
||||
# guessers, eg: a lang parser for idx files, an automatic detection of the language
|
||||
# for srt files, a video metadata extractor for avi, mkv, ...
|
||||
|
||||
# if we are on autodetect, try to do it now so we can tell the
|
||||
# guess_groups function what type of info it should be looking for
|
||||
if filetype in ('autodetect', 'subtitle'):
|
||||
for rexp, confidence, span_adjust in episode_rexps:
|
||||
match = re.search(rexp, filename, re.IGNORECASE)
|
||||
if match:
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'episode'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'episodesubtitle'
|
||||
break
|
||||
|
||||
# if no episode info found, assume it's a movie
|
||||
if filetype == 'autodetect':
|
||||
filetype = 'movie'
|
||||
elif filetype == 'subtitle':
|
||||
filetype = 'moviesubtitle'
|
||||
|
||||
# try to detect the file type
|
||||
filetype, other = guess_filetype(filename, filetype)
|
||||
guessed({ 'type': filetype }, confidence = 1.0)
|
||||
extguess = guessed(other, confidence = 1.0)
|
||||
|
||||
# guess the mimetype of the filename
|
||||
# TODO: handle other mimetypes not found on the default type_maps
|
||||
# mimetypes.types_map['.srt']='text/subtitle'
|
||||
mime, _ = mimetypes.guess_type(filename, strict=False)
|
||||
if mime is not None:
|
||||
guessed({ 'mimetype': mime }, confidence = 1.0)
|
||||
|
||||
# remove the extension from the match tree, as all indices relative
|
||||
# the the filename groups assume the basename is the last one
|
||||
fileext = match_tree.pop(-1)[1:].lower()
|
||||
|
||||
|
||||
# 2- split each of those into explicit groups, if any
|
||||
@@ -453,8 +424,14 @@ class IterativeMatcher(object):
|
||||
if len(previous) == 1:
|
||||
guess = guessed({ 'series': previous[0][0] }, confidence = 0.5)
|
||||
leftover = update_found(leftover, previous[0][1], guess)
|
||||
|
||||
|
||||
|
||||
# reduce the confidence of unlikely series
|
||||
for guess in result:
|
||||
if 'series' in guess:
|
||||
if guess['series'].lower() in unlikely_series:
|
||||
guess.set_confidence('series', guess.confidence('series') * 0.5)
|
||||
|
||||
|
||||
elif filetype in ('movie', 'moviesubtitle'):
|
||||
leftover_all = leftover_valid_groups(match_tree)
|
||||
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
#
|
||||
# GuessIt - A library for guessing information from filenames
|
||||
# Copyright (c) 2011 Nicolas Wack <wackou@gmail.com>
|
||||
# Copyright (c) 2011 Ricard Marxer <ricardmp@gmail.com>
|
||||
#
|
||||
# GuessIt is free software; you can redistribute it and/or modify it under
|
||||
# the terms of the Lesser GNU General Public License as published by
|
||||
@@ -19,9 +20,9 @@
|
||||
#
|
||||
|
||||
|
||||
subtitle_exts = [ 'srt', 'idx', 'sub' ]
|
||||
subtitle_exts = [ 'srt', 'idx', 'sub', 'ssa', 'txt' ]
|
||||
|
||||
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'mov', 'ogg', 'ogm', 'ogv', 'wmv' ]
|
||||
video_exts = [ 'avi', 'mkv', 'mpg', 'mp4', 'm4v', 'mov', 'ogg', 'ogm', 'ogv', 'wmv', 'divx' ]
|
||||
|
||||
# separator character regexp
|
||||
sep = r'[][)(}{+ \._-]' # regexp art, hehe :D
|
||||
@@ -34,6 +35,9 @@ episode_rexps = [ # ... Season 2 ...
|
||||
(r'season (?P<season>[0-9]+)', 1.0, (0, 0)),
|
||||
(r'saison (?P<season>[0-9]+)', 1.0, (0, 0)),
|
||||
|
||||
# ... s02-x01 ...
|
||||
(r's(?P<season>[0-9]{1,2})-x(?P<bonusNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
|
||||
|
||||
# ... s02e13 ...
|
||||
(r'[Ss](?P<season>[0-9]{1,2}).{,3}[EeXx](?P<episodeNumber>[0-9]{1,2})[^0-9]', 1.0, (0, -1)),
|
||||
|
||||
@@ -41,7 +45,7 @@ episode_rexps = [ # ... Season 2 ...
|
||||
(r'[^0-9](?P<season>[0-9]{1,2})[x\.](?P<episodeNumber>[0-9]{2})[^0-9]', 0.8, (1, -1)),
|
||||
|
||||
# ... s02 ...
|
||||
(sep + r's(?P<season>[0-9]{1,2})' + sep + '?', 0.6, (0, 0)),
|
||||
(sep + r's(?P<season>[0-9]{1,2})' + sep + '?', 0.6, (1, -1)),
|
||||
|
||||
# v2 or v3 for some mangas which have multiples rips
|
||||
(sep + r'(?P<episodeNumber>[0-9]{1,3})v[23]' + sep, 0.6, (0, 0)),
|
||||
@@ -77,11 +81,11 @@ video_rexps = [ # cd number
|
||||
|
||||
websites = [ 'tvu.org.ru', 'emule-island.com', 'UsaBit.com', 'www.divx-overnet.com', 'sharethefiles.com' ]
|
||||
|
||||
properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'Blu-ray', 'BDRip', 'BRRip',
|
||||
'HDRip', 'DVD', 'DVDivX', 'HDTV', 'DVB', 'WEBRip', 'DVDSCR', 'Screener', 'VHS',
|
||||
'VIDEO_TS' ],
|
||||
unlikely_series = ['series']
|
||||
|
||||
'container': [ 'avi', 'mkv', 'ogv', 'ogm', 'wmv', 'mp4', 'mov' ],
|
||||
properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'Blu-ray', 'BDRip', 'BRRip',
|
||||
'HDRip', 'DVD', 'DVDivX', 'HDTV', 'DVB', 'DVBRip', 'PDTV', 'WEBRip',
|
||||
'DVDSCR', 'Screener', 'VHS', 'VIDEO_TS' ],
|
||||
|
||||
'screenSize': [ '720p', '720' ],
|
||||
|
||||
@@ -106,10 +110,29 @@ properties = { 'format': [ 'DVDRip', 'HD-DVD', 'HDDVD', 'HDDVDRip', 'BluRay', 'B
|
||||
],
|
||||
}
|
||||
|
||||
def find_properties(filename):
|
||||
result = []
|
||||
clow = filename.lower()
|
||||
for prop, values in properties.items():
|
||||
for value in values:
|
||||
pos = clow.find(value.lower())
|
||||
if pos != -1:
|
||||
end = pos + len(value)
|
||||
# make sure our word is always surrounded by separators
|
||||
if ((pos > 0 and clow[pos-1] not in sep) or
|
||||
(end < len(clow) and clow[end] not in sep)):
|
||||
# note: sep is a regexp, but in this case using it as
|
||||
# a sequence achieves the same goal
|
||||
continue
|
||||
|
||||
result.append((prop, value, pos, end))
|
||||
return result
|
||||
|
||||
|
||||
property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
|
||||
'HD-DVD': [ 'HDDVD', 'HDDVDRip' ],
|
||||
'BluRay': [ 'BDRip', 'BRRip', 'Blu-ray' ],
|
||||
'DVB': [ 'DVBRip', 'PDTV' ],
|
||||
'Screener': [ 'DVDSCR' ],
|
||||
'DivX': [ 'DVDivX' ],
|
||||
'h264': [ 'x264' ],
|
||||
@@ -123,6 +146,10 @@ property_synonyms = { 'DVD': [ 'DVDRip', 'VIDEO_TS' ],
|
||||
|
||||
|
||||
reverse_synonyms = {}
|
||||
for prop, values in properties.items():
|
||||
for value in values:
|
||||
reverse_synonyms[value.lower()] = value
|
||||
|
||||
for canonical, synonyms in property_synonyms.items():
|
||||
for synonym in synonyms:
|
||||
reverse_synonyms[synonym.lower()] = canonical
|
||||
|
||||
Reference in New Issue
Block a user