Improved scoring

This commit is contained in:
Ruud
2012-06-13 13:08:42 +02:00
parent 138286722f
commit ea6b274f02
4 changed files with 93 additions and 10 deletions
+6 -1
View File
@@ -4,7 +4,8 @@ from couchpotato.core.helpers.variable import getTitle
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.base import Plugin
from couchpotato.core.plugins.score.scores import nameScore, nameRatioScore, \ from couchpotato.core.plugins.score.scores import nameScore, nameRatioScore, \
sizeScore, providerScore, duplicateScore, partialIgnoredScore sizeScore, providerScore, duplicateScore, partialIgnoredScore, namePositionScore, \
halfMultipartScore
log = CPLog(__name__) log = CPLog(__name__)
@@ -21,6 +22,7 @@ class Score(Plugin):
for movie_title in movie['library']['titles']: for movie_title in movie['library']['titles']:
score += nameRatioScore(toUnicode(nzb['name']), toUnicode(movie_title['title'])) score += nameRatioScore(toUnicode(nzb['name']), toUnicode(movie_title['title']))
score += namePositionScore(toUnicode(nzb['name']), toUnicode(movie_title['title']))
score += sizeScore(nzb['size']) score += sizeScore(nzb['size'])
@@ -41,6 +43,9 @@ class Score(Plugin):
# Partial ignored words # Partial ignored words
score += partialIgnoredScore(nzb['name'], getTitle(movie['library'])) score += partialIgnoredScore(nzb['name'], getTitle(movie['library']))
# Ignore single downloads from multipart
score += halfMultipartScore(nzb['name'])
# Extra provider specific check # Extra provider specific check
extra_score = nzb.get('extra_score') extra_score = nzb.get('extra_score')
if extra_score: if extra_score:
+73 -4
View File
@@ -1,5 +1,7 @@
from couchpotato.core.event import fireEvent from couchpotato.core.event import fireEvent
from couchpotato.core.helpers.encoding import simplifyString from couchpotato.core.helpers.encoding import simplifyString
from couchpotato.core.helpers.variable import tryInt
from couchpotato.core.plugins.scanner.main import Scanner
from couchpotato.environment import Env from couchpotato.environment import Env
import re import re
@@ -16,11 +18,12 @@ name_scores = [
'german:-10', 'french:-10', 'spanish:-10', 'swesub:-20', 'danish:-10', 'dutch:-10', 'german:-10', 'french:-10', 'spanish:-10', 'swesub:-20', 'danish:-10', 'dutch:-10',
# Release groups # Release groups
'imbt:1', 'cocain:1', 'vomit:1', 'fico:1', 'arrow:1', 'pukka:1', 'prism:1', 'devise:1', 'esir:1', 'ctrlhd:1', 'imbt:1', 'cocain:1', 'vomit:1', 'fico:1', 'arrow:1', 'pukka:1', 'prism:1', 'devise:1', 'esir:1', 'ctrlhd:1',
'metis:1', 'diamond:1', 'wiki:1', 'cbgb:1', 'crossbow:1', 'sinners:1', 'amiable:1', 'refined:1', 'twizted:1', 'felony:1', 'hubris:1', 'machd:1', 'metis:10', 'diamond:10', 'wiki:10', 'cbgb:10', 'crossbow:1', 'sinners:10', 'amiable:10', 'refined:1', 'twizted:1', 'felony:1', 'hubris:1', 'machd:1',
# Extras # Extras
'extras:-40', 'trilogy:-40', 'extras:-40', 'trilogy:-40',
] ]
def nameScore(name, year): def nameScore(name, year):
''' Calculate score for words in the NZB name ''' ''' Calculate score for words in the NZB name '''
@@ -47,8 +50,8 @@ def nameScore(name, year):
return score return score
def nameRatioScore(nzb_name, movie_name):
def nameRatioScore(nzb_name, movie_name):
nzb_words = re.split('\W+', fireEvent('scanner.create_file_identifier', nzb_name, single = True)) nzb_words = re.split('\W+', fireEvent('scanner.create_file_identifier', nzb_name, single = True))
movie_words = re.split('\W+', simplifyString(movie_name)) movie_words = re.split('\W+', simplifyString(movie_name))
@@ -56,15 +59,68 @@ def nameRatioScore(nzb_name, movie_name):
return 10 - len(left_over) return 10 - len(left_over)
def namePositionScore(nzb_name, movie_name):
score = 0
nzb_words = re.split('\W+', simplifyString(nzb_name))
qualities = fireEvent('quality.all', single = True)
try:
nzb_name = re.search(r'([\'"])[^\1]*\1', nzb_name).group(0)
except:
pass
name_year = fireEvent('scanner.name_year', nzb_name, single = True)
# Give points for movies beginning with the correct name
name_split = simplifyString(nzb_name).split(simplifyString(movie_name))
if name_split[0].strip() == '':
score += 10
# If year is second in line, give more points
if len(name_split) > 1 and name_year:
after_name = name_split[1].strip()
if tryInt(after_name[:4]) == name_year.get('year', None):
score += 10
after_name = after_name[4:]
# Give -point to crap between year and quality
found_quality = None
for quality in qualities:
# Main in words
if quality['identifier'] in nzb_words:
found_quality = quality['identifier']
# Alt in words
for alt in quality['alternative']:
if alt in nzb_words:
found_quality = alt
break
if not found_quality:
return score - 20
allowed = []
for value in name_scores:
name, sc = value.split(':')
allowed.append(name)
inbetween = re.split('\W+', after_name.split(found_quality)[0].strip())
score -= (10 * len(set(inbetween) - set(allowed)))
return score
def sizeScore(size): def sizeScore(size):
return 0 if size else -20 return 0 if size else -20
def providerScore(provider): def providerScore(provider):
if provider in ['NZBMatrix', 'Nzbs', 'Newzbin']: if provider in ['NZBMatrix', 'Nzbs', 'Newzbin']:
return 30 return 20
if provider in ['Newznab', 'Moovee', 'X264']: if provider in ['Newznab']:
return 10 return 10
return 0 return 0
@@ -94,3 +150,16 @@ def partialIgnoredScore(nzb_name, movie_name):
score -= 5 score -= 5
return score return score
def halfMultipartScore(nzb_name):
wrong_found = 0
for nr in [1, 2, 3, 4, 5, 'i', 'ii', 'iii', 'iv', 'v', 'a', 'b', 'c', 'd', 'e']:
for wrong in ['cd', 'part', 'dis', 'disc', 'dvd']:
if '%s%s' % (wrong, nr) in nzb_name.lower():
wrong_found += 1
if wrong_found == 1:
return -30
return 0
@@ -28,7 +28,7 @@ class NZBClub(NZBProvider, RSS):
if self.isDisabled(): if self.isDisabled():
return results return results
q = '"%s" %s %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier')) q = '"%s %s" %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier'))
for ignored in Env.setting('ignored_words', 'searcher').split(','): for ignored in Env.setting('ignored_words', 'searcher').split(','):
q = '%s -%s' % (q, ignored.strip()) q = '%s -%s' % (q, ignored.strip())
@@ -62,9 +62,10 @@ class NZBClub(NZBProvider, RSS):
def extra_check(item): def extra_check(item):
full_description = self.getCache('nzbclub.%s' % nzbclub_id, item['detail_url'], cache_timeout = 25920000) full_description = self.getCache('nzbclub.%s' % nzbclub_id, item['detail_url'], cache_timeout = 25920000)
if 'ARCHIVE inside ARCHIVE' in full_description: for ignored in ['ARCHIVE inside ARCHIVE', 'Incomplete', 'repair impossible']:
log.info('Wrong: Seems to be passworded files: %s', new['name']) if ignored in full_description:
return False log.info('Wrong: Seems to be passworded or corrupted files: %s', new['name'])
return False
return True return True
@@ -30,7 +30,7 @@ class NzbIndex(NZBProvider, RSS):
if self.isDisabled(): if self.isDisabled():
return results return results
q = '%s %s %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier')) q = '"%s %s" %s' % (simplifyString(getTitle(movie['library'])), movie['library']['year'], quality.get('identifier'))
arguments = tryUrlencode({ arguments = tryUrlencode({
'q': q, 'q': q,
'age': Env.setting('retention', 'nzb'), 'age': Env.setting('retention', 'nzb'),
@@ -67,6 +67,13 @@ class NzbIndex(NZBProvider, RSS):
except: except:
description = '' description = ''
def extra_check(new):
if '#c20000' in new['description'].lower():
log.info('Wrong: Seems to be passworded: %s', new['name'])
return False
return True
new = { new = {
'id': nzbindex_id, 'id': nzbindex_id,
'type': 'nzb', 'type': 'nzb',
@@ -79,6 +86,7 @@ class NzbIndex(NZBProvider, RSS):
'detail_url': enclosure['url'].replace('/download/', '/release/'), 'detail_url': enclosure['url'].replace('/download/', '/release/'),
'description': description, 'description': description,
'get_more_info': self.getMoreInfo, 'get_more_info': self.getMoreInfo,
'extra_check': extra_check,
'check_nzb': True, 'check_nzb': True,
} }