Renamer, movie searcher, Library scanning

New theme for UI
and more..
This commit is contained in:
Ruud
2011-05-28 20:52:18 +02:00
parent e0b3c27957
commit 98aafeaec3
37 changed files with 2407 additions and 869 deletions
+205 -96
View File
@@ -1,10 +1,10 @@
from couchpotato import get_session
from couchpotato.core.event import fireEvent, addEvent
from couchpotato.core.helpers.encoding import toUnicode
from couchpotato.core.helpers.encoding import toUnicode, simplifyString
from couchpotato.core.helpers.variable import getExt
from couchpotato.core.logger import CPLog
from couchpotato.core.plugins.base import Plugin
from couchpotato.core.settings.model import File, Library, Release, Movie
from couchpotato.core.settings.model import File, Release, Movie
from couchpotato.environment import Env
from flask.helpers import json
from themoviedb.tmdb import opensubtitleHashFile
@@ -23,7 +23,7 @@ class Scanner(Plugin):
'trailer': 1048576, # 1MB
}
ignored_in_path = ['_unpack', '_failed_', '_unknown_', '_exists_', '.appledouble', '.appledb', '.appledesktop', os.path.sep + '._', '.ds_store', 'cp.cpnfo'] #unpacking, smb-crap, hidden files
ignore_names = ['extract', 'extracting', 'extracted', 'movie', 'movies', 'film', 'films']
ignore_names = ['extract', 'extracting', 'extracted', 'movie', 'movies', 'film', 'films', 'download', 'downloads']
extensions = {
'movie': ['mkv', 'wmv', 'avi', 'mpg', 'mpeg', 'mp4', 'm2ts', 'iso', 'img'],
'dvd': ['vts_*', 'vob'],
@@ -42,7 +42,7 @@ class Scanner(Plugin):
codecs = {
'audio': ['dts', 'ac3', 'ac3d', 'mp3'],
'video': ['x264', 'divx', 'xvid']
'video': ['x264', 'h264', 'divx', 'xvid']
}
source_media = {
@@ -52,12 +52,16 @@ class Scanner(Plugin):
'hdtv': ['hdtv']
}
clean = '(?i)[^\s](ac3|dts|custom|dc|divx|divx5|dsr|dsrip|dutch|dvd|dvdrip|dvdscr|dvdscreener|screener|dvdivx|cam|fragment|fs|hdtv|hdrip|hdtvrip|internal|limited|multisubs|ntsc|ogg|ogm|pal|pdtv|proper|repack|rerip|retail|r3|r5|bd5|se|svcd|swedish|german|read.nfo|nfofix|unrated|ws|telesync|ts|telecine|tc|brrip|bdrip|480p|480i|576p|576i|720p|720i|1080p|1080i|hrhd|hrhdtv|hddvd|bluray|x264|h264|xvid|xvidvd|xxx|www.www|cd[1-9]|\[.*\])[^\s]*'
clean = '[ _\,\.\(\)\[\]\-](french|swedisch|danish|dutch|swesub|spanish|german|ac3|dts|custom|dc|divx|divx5|dsr|dsrip|dutch|dvd|dvdrip|dvdscr|dvdscreener|screener|dvdivx|cam|fragment|fs|hdtv|hdrip|hdtvrip|internal|limited|multisubs|ntsc|ogg|ogm|pal|pdtv|proper|repack|rerip|retail|r3|r5|bd5|se|svcd|swedish|german|read.nfo|nfofix|unrated|ws|telesync|ts|telecine|tc|brrip|bdrip|video_ts|audio_ts|480p|480i|576p|576i|720p|720i|1080p|1080i|hrhd|hrhdtv|hddvd|bluray|x264|h264|xvid|xvidvd|xxx|www.www|cd[1-9]|\[.*\])([ _\,\.\(\)\[\]\-]|$)'
multipart_regex = [
'[ _\.-]+cd[ _\.-]*([0-9a-d]+)', #*cd1
'[ _\.-]+dvd[ _\.-]*([0-9a-d]+)', #*dvd1
'[ _\.-]+part[ _\.-]*([0-9a-d]+)', #*part1.mkv
'[ _\.-]+dis[ck][ _\.-]*([0-9a-d]+)', #*disk1.mkv
'[ _\.-]+part[ _\.-]*([0-9a-d]+)', #*part1
'[ _\.-]+dis[ck][ _\.-]*([0-9a-d]+)', #*disk1
'cd[ _\.-]*([0-9a-d]+)$', #cd1.ext
'dvd[ _\.-]*([0-9a-d]+)$', #dvd1.ext
'part[ _\.-]*([0-9a-d]+)$', #part1.mkv
'dis[ck][ _\.-]*([0-9a-d]+)$', #disk1.mkv
'()[ _\.-]+([0-9]*[abcd]+)(\.....?)$',
'([a-z])([0-9]+)(\.....?)$',
'()([ab])(\.....?)$' #*a.mkv
@@ -65,42 +69,54 @@ class Scanner(Plugin):
def __init__(self):
addEvent('app.load', self.scan)
#addEvent('app.load', self.scanLibrary)
def scan(self, folder = '/Volumes/Media/Test/'):
addEvent('scanner.scan', self.scan)
"""
Get all files
def scanLibrary(self):
For each file larger then 350MB
create movie "group", this is where all movie files will be grouped
group multipart together
check if its DVD (VIDEO_TS)
folder = '/Volumes/Media/Test/'
# This should work for non-folder based structure
for each moviegroup
groups = self.scan(folder = folder)
for each file smaller then 350MB, allfiles.filter(moviename*)
# Open up the db
db = get_session()
# Assuming the beginning of the filename is the same for this structure
Movie is masterfile, moviename-cd1.ext -> moviename
Find other files connected to moviename, moviename*.nfo, moviename*.sub, moviename*trailer.ext
# Mark all files as "offline" before a adding them to the database (again)
files_in_path = db.query(File).filter(File.path.like(toUnicode(folder) + u'%%'))
files_in_path.update({'available': 0}, synchronize_session = False)
db.commit()
Remove found file from allfiles
update_after = []
for group in groups.itervalues():
# This should work for folder based structure
for each leftover file
Loop over leftover files, use dirname as moviename
# Save to DB
if group['library']:
#library = db.query(Library).filter_by(id = library.get('id')).one()
# Add release
self.addRelease(group)
# Add identifier for library update
update_after.append(group['library'].get('identifier'))
for identifier in update_after:
fireEvent('library.update', identifier = identifier)
# If cleanup option is enabled, remove offline files from database
if self.conf('cleanup_offline'):
files_in_path = db.query(File).filter(File.path.like(folder + '%%')).filter_by(available = 0)
[db.delete(x) for x in files_in_path]
db.commit()
db.remove()
For each found movie
def scan(self, folder = None):
determine filetype
Check if it's already in the db
Add it to database
"""
if not folder or not os.path.isdir(folder):
log.error('Folder doesn\'t exists: %s' % folder)
return {}
# Get movie "master" files
movie_files = {}
@@ -153,22 +169,16 @@ class Scanner(Plugin):
# Remove the found files from the leftover stack
leftovers = leftovers - found_files
# Open up the db
db = get_session()
# Mark all files as "offline" before a adding them to the database (again)
files_in_path = db.query(File).filter(File.path.like(toUnicode(folder) + u'%%'))
files_in_path.update({'available': 0}, synchronize_session = False)
db.commit()
# Determine file types
update_after = []
for identifier, group in movie_files.iteritems():
for identifier in movie_files:
group = movie_files[identifier]
# Group extra (and easy) files first
images = self.getImages(group['unsorted_files'])
group['files'] = {
'subtitle': self.getSubtitles(group['unsorted_files']),
'subtitle_extra': self.getSubtitlesExtras(group['unsorted_files']),
'nfo': self.getNfo(group['unsorted_files']),
'trailer': self.getTrailers(group['unsorted_files']),
'backdrop': images['backdrop'],
@@ -180,7 +190,22 @@ class Scanner(Plugin):
group['files']['movie'] = self.getDVDFiles(group['unsorted_files'])
else:
group['files']['movie'] = self.getMediaFiles(group['unsorted_files'])
group['meta_data'] = self.getMetaData(group['files']['movie'])
group['meta_data'] = self.getMetaData(group)
# Get parent dir from movie files
for movie_file in group['files']['movie']:
group['parentdir'] = os.path.dirname(movie_file)
group['dirname'] = None
folders = group['parentdir'].replace(folder, '').split(os.path.sep)
# Try and get a proper dirname, so no "A", "Movie", "Download"
for folder in folders:
if folder.lower() in self.ignore_names or len(folder) < 2:
group['dirname'] = folder
break
break
# Leftover "sorted" files
for type in group['files']:
@@ -191,34 +216,16 @@ class Scanner(Plugin):
# Determine movie
group['library'] = self.determineMovie(group)
if not group['library']:
log.error('Unable to determin movie: %s' % group['identifiers'])
# Save to DB
if group['library']:
#library = db.query(Library).filter_by(id = library.get('id')).one()
# Add release
release = self.addRelease(group)
return
# Add identifier for library update
update_after.append(group['library'].get('identifier'))
for identifier in update_after:
fireEvent('library.update', identifier = identifier)
# If cleanup option is enabled, remove offline files from database
if self.conf('cleanup_offline'):
files_in_path = db.query(File).filter(File.path.like(folder + '%%')).filter_by(available = 0)
[db.delete(x) for x in files_in_path]
db.commit()
db.remove()
return movie_files
def addRelease(self, group):
db = get_session()
identifier = '%s.%s.%s' % (group['library']['identifier'], group['meta_data']['audio'], group['meta_data']['quality'])
identifier = '%s.%s.%s' % (group['library']['identifier'], group['meta_data'].get('audio', 'unknown'), group['meta_data']['quality']['identifier'])
# Add movie
done_status = fireEvent('status.get', 'done', single = True)
@@ -233,13 +240,13 @@ class Scanner(Plugin):
db.commit()
# Add release
quality = fireEvent('quality.single', group['meta_data']['quality'], single = True)
release = db.query(Release).filter_by(identifier = identifier).first()
if not release:
release = Release(
identifier = identifier,
movie = movie,
quality_id = quality.get('id'),
quality_id = group['meta_data']['quality'].get('id'),
status_id = done_status.get('id')
)
db.add(release)
@@ -259,18 +266,36 @@ class Scanner(Plugin):
db.remove()
def getMetaData(self, files):
def getMetaData(self, group):
return {
'audio': 'AC3',
'quality': '720p',
'quality_type': 'HD',
'resolution_width': 1280,
'resolution_height': 720
}
data = {}
files = group['files']['movie']
for file in files:
self.getMeta(file)
if os.path.getsize(file) < self.minimal_filesize['media']: continue # Ignore smaller files
meta = self.getMeta(file)
try:
data['video'] = self.getCodec(file, self.codecs['video'])
data['audio'] = meta['audio stream'][0]['compression']
data['resolution_width'] = meta['video stream'][0]['image width']
data['resolution_height'] = meta['video stream'][0]['image height']
except:
pass
if data.get('audio'): break
data['quality'] = fireEvent('quality.guess', files = files, extra = data, single = True)
if not data['quality']:
data['quality'] = fireEvent('quality.single', 'dvdr' if group['is_dvd'] else 'dvdrip', single = True)
data['quality_type'] = 'HD' if data.get('resolution_width', 0) >= 720 else 'SD'
data['group'] = self.getGroup(file[0])
data['source'] = self.getSourceMedia(file[0])
return data
def getMeta(self, filename):
lib_dir = os.path.join(Env.get('app_dir'), 'libs')
@@ -281,40 +306,58 @@ class Scanner(Plugin):
try:
meta = json.loads(z)
log.info('Retrieved metainfo: %s' % meta)
return meta
except Exception, e:
print e
log.error('Couldn\'t get metadata from file')
except Exception:
log.error('Couldn\'t get metadata from file: %s' % traceback.format_exc())
def determineMovie(self, group):
imdb_id = None
files = group['files']
# Check and see if nfo contains the imdb-id
try:
for nfo_file in files['nfo']:
imdb_id = self.getImdb(nfo_file)
if imdb_id: break
except:
pass
# Check if path is already in db
db = get_session()
# Check for CP(imdb_id) string in the file paths
for file in files['movie']:
f = db.query(File).filter_by(path = toUnicode(file)).first()
imdb_id = self.getCPImdb(file)
if imdb_id: break
# Check and see if nfo contains the imdb-id
if not imdb_id:
try:
imdb_id = f.library[0].identifier
break
for nfo_file in files['nfo']:
imdb_id = self.getImdb(nfo_file)
if imdb_id: break
except:
pass
db.remove()
# Check if path is already in db
if not imdb_id:
db = get_session()
for file in files['movie']:
f = db.query(File).filter_by(path = toUnicode(file)).first()
try:
imdb_id = f.library[0].identifier
break
except:
pass
db.remove()
# Search based on OpenSubtitleHash
if not imdb_id and not group['is_dvd']:
for file in files['movie']:
movie = fireEvent('provider.movie.by_hash', file = file, merge = True)
if len(movie) > 0:
imdb_id = movie[0]['imdb']
if imdb_id: break
# Search based on identifiers
if not imdb_id:
for identifier in group['identifiers']:
if len(identifier) > 2:
movie = fireEvent('provider.movie.search', q = identifier, merge = True, limit = 1)
if len(movie) > 0:
imdb_id = movie[0]['imdb']
if imdb_id: break
@@ -329,7 +372,7 @@ class Scanner(Plugin):
}, update_after = False, single = True)
log.error('No imdb_id found for %s.' % group['identifiers'])
return False
return {}
def saveFile(self, file, type = 'unknown', include_media_info = False):
@@ -342,6 +385,17 @@ class Scanner(Plugin):
# Check database and update/insert if necessary
return fireEvent('file.add', path = file, part = self.getPartNumber(file), type = self.file_types[type], properties = properties, single = True)
def getCPImdb(self, string):
try:
m = re.search('(cp\((?P<id>tt[0-9{7}]+)\))', string.lower())
id = m.group('id')
if id: return id
except AttributeError:
pass
return False
def getImdb(self, txt):
if os.path.isfile(txt):
@@ -375,6 +429,9 @@ class Scanner(Plugin):
def getSubtitles(self, files):
return set(filter(lambda s: getExt(s.lower()) in self.extensions['subtitle'], files))
def getSubtitlesExtras(self, files):
return set(filter(lambda s: getExt(s.lower()) in self.extensions['subtitle_extra'], files))
def getNfo(self, files):
return set(filter(lambda s: getExt(s.lower()) in self.extensions['nfo'], files))
@@ -447,20 +504,42 @@ class Scanner(Plugin):
return set(filter(lambda s:identifier in self.createFileIdentifier(s, folder), file_pile))
def createFileIdentifier(self, file_path, folder, exclude_filename = False):
identifier = file_path.replace(folder, '') # root folder
identifier = os.path.splitext(identifier)[0] # ext
if exclude_filename:
identifier = identifier[:len(identifier) - len(os.path.split(identifier)[-1])]
identifier = self.removeMultipart(identifier) # multipart
return identifier
# multipart
identifier = self.removeMultipart(identifier)
# groups, release tags, scenename cleaner, regex isn't correct
identifier = re.sub(self.clean, '::', simplifyString(identifier))
year = self.findYear(identifier)
if year:
identifier = '%s %s' % (identifier.split(year)[0].strip(), year)
else:
identifier = identifier.split('::')[0]
# Remove duplicates
out = []
for word in identifier.split():
if not word in out:
out.append(word)
identifier = ' '.join(out)
return simplifyString(identifier)
def removeMultipart(self, name):
for regex in self.multipart_regex:
try:
found = re.sub(regex, '', name)
if found != name:
return found
name = found
except:
pass
return name
@@ -475,3 +554,33 @@ class Scanner(Plugin):
except:
pass
return name
def getCodec(self, filename, codecs):
codecs = map(re.escape, codecs)
try:
codec = re.search('[^A-Z0-9](?P<codec>' + '|'.join(codecs) + ')[^A-Z0-9]', filename, re.I)
return (codec and codec.group('codec')) or ''
except:
return ''
def getGroup(self, file):
try:
group = re.search('-(?P<group>[A-Z0-9]+)$', file, re.I)
return group.group('group') or ''
except:
return ''
def getSourceMedia(self, file):
for media in self.source_media:
for alias in self.source_media[media]:
if alias in file.lower():
return media
return None
def findYear(self, text):
matches = re.search('(?P<year>[0-9]{4})', text)
if matches:
return matches.group('year')
return ''