From ea6b274f02efbf9d915ffd3c6b727c5b44608325 Mon Sep 17 00:00:00 2001 From: Ruud Date: Wed, 13 Jun 2012 13:08:42 +0200 Subject: [PATCH] Improved scoring --- couchpotato/core/plugins/score/main.py | 7 +- couchpotato/core/plugins/score/scores.py | 77 ++++++++++++++++++- .../core/providers/nzb/nzbclub/main.py | 9 ++- .../core/providers/nzb/nzbindex/main.py | 10 ++- 4 files changed, 93 insertions(+), 10 deletions(-) diff --git a/couchpotato/core/plugins/score/main.py b/couchpotato/core/plugins/score/main.py index 2506a473..f6f3232c 100644 --- a/couchpotato/core/plugins/score/main.py +++ b/couchpotato/core/plugins/score/main.py @@ -4,7 +4,8 @@ from couchpotato.core.helpers.variable import getTitle from couchpotato.core.logger import CPLog from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.score.scores import nameScore, nameRatioScore, \ - sizeScore, providerScore, duplicateScore, partialIgnoredScore + sizeScore, providerScore, duplicateScore, partialIgnoredScore, namePositionScore, \ + halfMultipartScore log = CPLog(__name__) @@ -21,6 +22,7 @@ class Score(Plugin): for movie_title in movie['library']['titles']: score += nameRatioScore(toUnicode(nzb['name']), toUnicode(movie_title['title'])) + score += namePositionScore(toUnicode(nzb['name']), toUnicode(movie_title['title'])) score += sizeScore(nzb['size']) @@ -41,6 +43,9 @@ class Score(Plugin): # Partial ignored words score += partialIgnoredScore(nzb['name'], getTitle(movie['library'])) + # Ignore single downloads from multipart + score += halfMultipartScore(nzb['name']) + # Extra provider specific check extra_score = nzb.get('extra_score') if extra_score: diff --git a/couchpotato/core/plugins/score/scores.py b/couchpotato/core/plugins/score/scores.py index 71fe1f07..ddff25b1 100644 --- a/couchpotato/core/plugins/score/scores.py +++ b/couchpotato/core/plugins/score/scores.py @@ -1,5 +1,7 @@ from couchpotato.core.event import fireEvent from couchpotato.core.helpers.encoding import simplifyString +from couchpotato.core.helpers.variable import tryInt +from couchpotato.core.plugins.scanner.main import Scanner from couchpotato.environment import Env import re @@ -16,11 +18,12 @@ name_scores = [ 'german:-10', 'french:-10', 'spanish:-10', 'swesub:-20', 'danish:-10', 'dutch:-10', # Release groups 'imbt:1', 'cocain:1', 'vomit:1', 'fico:1', 'arrow:1', 'pukka:1', 'prism:1', 'devise:1', 'esir:1', 'ctrlhd:1', - 'metis:1', 'diamond:1', 'wiki:1', 'cbgb:1', 'crossbow:1', 'sinners:1', 'amiable:1', 'refined:1', 'twizted:1', 'felony:1', 'hubris:1', 'machd:1', + 'metis:10', 'diamond:10', 'wiki:10', 'cbgb:10', 'crossbow:1', 'sinners:10', 'amiable:10', 'refined:1', 'twizted:1', 'felony:1', 'hubris:1', 'machd:1', # Extras 'extras:-40', 'trilogy:-40', ] + def nameScore(name, year): ''' Calculate score for words in the NZB name ''' @@ -47,8 +50,8 @@ def nameScore(name, year): return score -def nameRatioScore(nzb_name, movie_name): +def nameRatioScore(nzb_name, movie_name): nzb_words = re.split('\W+', fireEvent('scanner.create_file_identifier', nzb_name, single = True)) movie_words = re.split('\W+', simplifyString(movie_name)) @@ -56,15 +59,68 @@ def nameRatioScore(nzb_name, movie_name): return 10 - len(left_over) +def namePositionScore(nzb_name, movie_name): + score = 0 + + nzb_words = re.split('\W+', simplifyString(nzb_name)) + qualities = fireEvent('quality.all', single = True) + + try: + nzb_name = re.search(r'([\'"])[^\1]*\1', nzb_name).group(0) + except: + pass + + name_year = fireEvent('scanner.name_year', nzb_name, single = True) + + # Give points for movies beginning with the correct name + name_split = simplifyString(nzb_name).split(simplifyString(movie_name)) + if name_split[0].strip() == '': + score += 10 + + # If year is second in line, give more points + if len(name_split) > 1 and name_year: + after_name = name_split[1].strip() + if tryInt(after_name[:4]) == name_year.get('year', None): + score += 10 + after_name = after_name[4:] + + # Give -point to crap between year and quality + found_quality = None + for quality in qualities: + # Main in words + if quality['identifier'] in nzb_words: + found_quality = quality['identifier'] + + # Alt in words + for alt in quality['alternative']: + if alt in nzb_words: + found_quality = alt + break + + if not found_quality: + return score - 20 + + allowed = [] + for value in name_scores: + name, sc = value.split(':') + allowed.append(name) + + inbetween = re.split('\W+', after_name.split(found_quality)[0].strip()) + + score -= (10 * len(set(inbetween) - set(allowed))) + + return score + + def sizeScore(size): return 0 if size else -20 def providerScore(provider): if provider in ['NZBMatrix', 'Nzbs', 'Newzbin']: - return 30 + return 20 - if provider in ['Newznab', 'Moovee', 'X264']: + if provider in ['Newznab']: return 10 return 0 @@ -94,3 +150,16 @@ def partialIgnoredScore(nzb_name, movie_name): score -= 5 return score + +def halfMultipartScore(nzb_name): + + wrong_found = 0 + for nr in [1, 2, 3, 4, 5, 'i', 'ii', 'iii', 'iv', 'v', 'a', 'b', 'c', 'd', 'e']: + for wrong in ['cd', 'part', 'dis', 'disc', 'dvd']: + if '%s%s' % (wrong, nr) in nzb_name.lower(): + wrong_found += 1 + + if wrong_found == 1: + return -30 + + return 0 diff --git a/couchpotato/core/providers/nzb/nzbclub/main.py b/couchpotato/core/providers/nzb/nzbclub/main.py index 5eaf7b57..39835ded 100644 --- a/couchpotato/core/providers/nzb/nzbclub/main.py +++ b/couchpotato/core/providers/nzb/nzbclub/main.py @@ -28,7 +28,7 @@ class NZBClub(NZBProvider, RSS): if self.isDisabled(): return results - q = '"%s" %s %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier')) + q = '"%s %s" %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier')) for ignored in Env.setting('ignored_words', 'searcher').split(','): q = '%s -%s' % (q, ignored.strip()) @@ -62,9 +62,10 @@ class NZBClub(NZBProvider, RSS): def extra_check(item): full_description = self.getCache('nzbclub.%s' % nzbclub_id, item['detail_url'], cache_timeout = 25920000) - if 'ARCHIVE inside ARCHIVE' in full_description: - log.info('Wrong: Seems to be passworded files: %s', new['name']) - return False + for ignored in ['ARCHIVE inside ARCHIVE', 'Incomplete', 'repair impossible']: + if ignored in full_description: + log.info('Wrong: Seems to be passworded or corrupted files: %s', new['name']) + return False return True diff --git a/couchpotato/core/providers/nzb/nzbindex/main.py b/couchpotato/core/providers/nzb/nzbindex/main.py index 7d007bcb..89f68bee 100644 --- a/couchpotato/core/providers/nzb/nzbindex/main.py +++ b/couchpotato/core/providers/nzb/nzbindex/main.py @@ -30,7 +30,7 @@ class NzbIndex(NZBProvider, RSS): if self.isDisabled(): return results - q = '%s %s %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier')) + q = '"%s %s" %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier')) arguments = tryUrlencode({ 'q': q, 'age': Env.setting('retention', 'nzb'), @@ -67,6 +67,13 @@ class NzbIndex(NZBProvider, RSS): except: description = '' + def extra_check(new): + if '#c20000' in new['description'].lower(): + log.info('Wrong: Seems to be passworded: %s', new['name']) + return False + + return True + new = { 'id': nzbindex_id, 'type': 'nzb', @@ -79,6 +86,7 @@ class NzbIndex(NZBProvider, RSS): 'detail_url': enclosure['url'].replace('/download/', '/release/'), 'description': description, 'get_more_info': self.getMoreInfo, + 'extra_check': extra_check, 'check_nzb': True, }