Merge branch 'develop' into tv

This commit is contained in:
seedzero
2014-10-28 17:37:56 +11:00
224 changed files with 7151 additions and 15308 deletions
-1
View File
@@ -10,7 +10,6 @@ import socket
import subprocess import subprocess
import sys import sys
import traceback import traceback
import time
# Root path # Root path
base_path = dirname(os.path.abspath(__file__)) base_path = dirname(os.path.abspath(__file__))
+2
View File
@@ -46,6 +46,8 @@ Linux:
* (systemd) Enable it at boot with `sudo systemctl enable couchpotato` * (systemd) Enable it at boot with `sudo systemctl enable couchpotato`
* Open your browser and go to `http://localhost:5050/` * Open your browser and go to `http://localhost:5050/`
Docker:
* You can use [razorgirl's Dockerfile](https://github.com/razorgirl/docker-couchpotato) to quickly build your own isolated app container. It's based on the Linux instructions above. For more info about Docker check out the [official website](https://www.docker.com).
FreeBSD : FreeBSD :
+2
View File
@@ -143,6 +143,8 @@ class ApiHandler(RequestHandler):
else: else:
self.write(result) self.write(result)
self.finish() self.finish()
except UnicodeDecodeError:
log.error('Failed proper encode: %s', traceback.format_exc())
except: except:
log.debug('Failed doing request, probably already closed: %s', (traceback.format_exc())) log.debug('Failed doing request, probably already closed: %s', (traceback.format_exc()))
try: self.finish({'success': False, 'error': 'Failed returning results'}) try: self.finish({'success': False, 'error': 'Failed returning results'})
+5 -5
View File
@@ -181,13 +181,13 @@ class Core(Plugin):
return '%sapi/%s' % (self.createBaseUrl(), Env.setting('api_key')) return '%sapi/%s' % (self.createBaseUrl(), Env.setting('api_key'))
def version(self): def version(self):
ver = fireEvent('updater.info', single = True) ver = fireEvent('updater.info', single = True) or {'version': {}}
if os.name == 'nt': platf = 'windows' if os.name == 'nt': platf = 'windows'
elif 'Darwin' in platform.platform(): platf = 'osx' elif 'Darwin' in platform.platform(): platf = 'osx'
else: platf = 'linux' else: platf = 'linux'
return '%s - %s-%s - v2' % (platf, ver.get('version')['type'], ver.get('version')['hash']) return '%s - %s-%s - v2' % (platf, ver.get('version').get('type') or 'unknown', ver.get('version').get('hash') or 'unknown')
def versionView(self, **kwargs): def versionView(self, **kwargs):
return { return {
@@ -286,13 +286,13 @@ config = [{
'name': 'permission_folder', 'name': 'permission_folder',
'default': '0755', 'default': '0755',
'label': 'Folder CHMOD', 'label': 'Folder CHMOD',
'description': 'Can be either decimal (493) or octal (leading zero: 0755)', 'description': 'Can be either decimal (493) or octal (leading zero: 0755). <a target="_blank" href="http://permissions-calculator.org/">Calculate the correct value</a>',
}, },
{ {
'name': 'permission_file', 'name': 'permission_file',
'default': '0755', 'default': '0644',
'label': 'File CHMOD', 'label': 'File CHMOD',
'description': 'Same as Folder CHMOD but for files', 'description': 'See Folder CHMOD description, but for files',
}, },
], ],
}, },
+17 -8
View File
@@ -205,19 +205,28 @@ class GitUpdater(BaseUpdater):
def getVersion(self): def getVersion(self):
if not self.version: if not self.version:
hash = None
date = None
branch = self.branch
try: try:
output = self.repo.getHead() # Yes, please output = self.repo.getHead() # Yes, please
log.debug('Git version output: %s', output.hash) log.debug('Git version output: %s', output.hash)
self.version = {
'repr': 'git:(%s:%s % s) %s (%s)' % (self.repo_user, self.repo_name, self.repo.getCurrentBranch().name or self.branch, output.hash[:8], datetime.fromtimestamp(output.getDate())), hash = output.hash[:8]
'hash': output.hash[:8], date = output.getDate()
'date': output.getDate(), branch = self.repo.getCurrentBranch().name
'type': 'git',
'branch': self.repo.getCurrentBranch().name
}
except Exception as e: except Exception as e:
log.error('Failed using GIT updater, running from source, you need to have GIT installed. %s', e) log.error('Failed using GIT updater, running from source, you need to have GIT installed. %s', e)
return 'No GIT'
self.version = {
'repr': 'git:(%s:%s % s) %s (%s)' % (self.repo_user, self.repo_name, branch, hash or 'unknown_hash', datetime.fromtimestamp(date) if date else 'unknown_date'),
'hash': hash,
'date': date,
'type': 'git',
'branch': branch
}
return self.version return self.version
+306 -274
View File
@@ -2,6 +2,7 @@ import json
import os import os
import time import time
import traceback import traceback
from sqlite3 import OperationalError
from CodernityDB.database import RecordNotFound from CodernityDB.database import RecordNotFound
from CodernityDB.index import IndexException, IndexNotFoundException, IndexConflict from CodernityDB.index import IndexException, IndexNotFoundException, IndexConflict
@@ -9,7 +10,7 @@ from couchpotato import CPLog
from couchpotato.api import addApiView from couchpotato.api import addApiView
from couchpotato.core.event import addEvent, fireEvent, fireEventAsync from couchpotato.core.event import addEvent, fireEvent, fireEventAsync
from couchpotato.core.helpers.encoding import toUnicode, sp from couchpotato.core.helpers.encoding import toUnicode, sp
from couchpotato.core.helpers.variable import getImdb, tryInt from couchpotato.core.helpers.variable import getImdb, tryInt, randomString
log = CPLog(__name__) log = CPLog(__name__)
@@ -32,6 +33,7 @@ class Database(object):
addEvent('database.setup.after', self.startup_compact) addEvent('database.setup.after', self.startup_compact)
addEvent('database.setup_index', self.setupIndex) addEvent('database.setup_index', self.setupIndex)
addEvent('database.delete_corrupted', self.deleteCorrupted)
addEvent('app.migrate', self.migrate) addEvent('app.migrate', self.migrate)
addEvent('app.after_shutdown', self.close) addEvent('app.after_shutdown', self.close)
@@ -147,6 +149,17 @@ class Database(object):
return results return results
def deleteCorrupted(self, _id, traceback_error = ''):
db = self.getDB()
try:
log.debug('Deleted corrupted document "%s": %s', (_id, traceback_error))
corrupted = db.get('id', _id, with_storage = False)
db._delete_id_index(corrupted.get('_id'), corrupted.get('_rev'), None)
except:
log.debug('Failed deleting corrupted: %s', traceback.format_exc())
def reindex(self, **kwargs): def reindex(self, **kwargs):
success = True success = True
@@ -299,309 +312,328 @@ class Database(object):
} }
migrate_data = {} migrate_data = {}
rename_old = False
c = conn.cursor() try:
for ml in migrate_list: c = conn.cursor()
migrate_data[ml] = {}
rows = migrate_list[ml]
try: for ml in migrate_list:
c.execute('SELECT %s FROM `%s`' % ('`' + '`,`'.join(rows) + '`', ml)) migrate_data[ml] = {}
except: rows = migrate_list[ml]
# ignore faulty destination_id database
if ml == 'category': try:
migrate_data[ml] = {} c.execute('SELECT %s FROM `%s`' % ('`' + '`,`'.join(rows) + '`', ml))
except:
# ignore faulty destination_id database
if ml == 'category':
migrate_data[ml] = {}
else:
rename_old = True
raise
for p in c.fetchall():
columns = {}
for row in migrate_list[ml]:
columns[row] = p[rows.index(row)]
if not migrate_data[ml].get(p[0]):
migrate_data[ml][p[0]] = columns
else:
if not isinstance(migrate_data[ml][p[0]], list):
migrate_data[ml][p[0]] = [migrate_data[ml][p[0]]]
migrate_data[ml][p[0]].append(columns)
conn.close()
log.info('Getting data took %s', time.time() - migrate_start)
db = self.getDB()
if not db.opened:
return
# Use properties
properties = migrate_data['properties']
log.info('Importing %s properties', len(properties))
for x in properties:
property = properties[x]
Env.prop(property.get('identifier'), property.get('value'))
# Categories
categories = migrate_data.get('category', [])
log.info('Importing %s categories', len(categories))
category_link = {}
for x in categories:
c = categories[x]
new_c = db.insert({
'_t': 'category',
'order': c.get('order', 999),
'label': toUnicode(c.get('label', '')),
'ignored': toUnicode(c.get('ignored', '')),
'preferred': toUnicode(c.get('preferred', '')),
'required': toUnicode(c.get('required', '')),
'destination': toUnicode(c.get('destination', '')),
})
category_link[x] = new_c.get('_id')
# Profiles
log.info('Importing profiles')
new_profiles = db.all('profile', with_doc = True)
new_profiles_by_label = {}
for x in new_profiles:
# Remove default non core profiles
if not x['doc'].get('core'):
db.delete(x['doc'])
else: else:
raise new_profiles_by_label[x['doc']['label']] = x['_id']
for p in c.fetchall(): profiles = migrate_data['profile']
columns = {} profile_link = {}
for row in migrate_list[ml]: for x in profiles:
columns[row] = p[rows.index(row)] p = profiles[x]
if not migrate_data[ml].get(p[0]): exists = new_profiles_by_label.get(p.get('label'))
migrate_data[ml][p[0]] = columns
# Update existing with order only
if exists and p.get('core'):
profile = db.get('id', exists)
profile['order'] = tryInt(p.get('order'))
profile['hide'] = p.get('hide') in [1, True, 'true', 'True']
db.update(profile)
profile_link[x] = profile.get('_id')
else: else:
if not isinstance(migrate_data[ml][p[0]], list):
migrate_data[ml][p[0]] = [migrate_data[ml][p[0]]]
migrate_data[ml][p[0]].append(columns)
conn.close() new_profile = {
'_t': 'profile',
'label': p.get('label'),
'order': int(p.get('order', 999)),
'core': p.get('core', False),
'qualities': [],
'wait_for': [],
'finish': []
}
log.info('Getting data took %s', time.time() - migrate_start) types = migrate_data['profiletype']
for profile_type in types:
p_type = types[profile_type]
if types[profile_type]['profile_id'] == p['id']:
if p_type['quality_id']:
new_profile['finish'].append(p_type['finish'])
new_profile['wait_for'].append(p_type['wait_for'])
new_profile['qualities'].append(migrate_data['quality'][p_type['quality_id']]['identifier'])
db = self.getDB() if len(new_profile['qualities']) > 0:
if not db.opened: new_profile.update(db.insert(new_profile))
return profile_link[x] = new_profile.get('_id')
else:
log.error('Corrupt profile list for "%s", using default.', p.get('label'))
# Use properties # Qualities
properties = migrate_data['properties'] log.info('Importing quality sizes')
log.info('Importing %s properties', len(properties)) new_qualities = db.all('quality', with_doc = True)
for x in properties: new_qualities_by_identifier = {}
property = properties[x] for x in new_qualities:
Env.prop(property.get('identifier'), property.get('value')) new_qualities_by_identifier[x['doc']['identifier']] = x['_id']
# Categories qualities = migrate_data['quality']
categories = migrate_data.get('category', []) quality_link = {}
log.info('Importing %s categories', len(categories)) for x in qualities:
category_link = {} q = qualities[x]
for x in categories: q_id = new_qualities_by_identifier[q.get('identifier')]
c = categories[x]
new_c = db.insert({ quality = db.get('id', q_id)
'_t': 'category', quality['order'] = q.get('order')
'order': c.get('order', 999), quality['size_min'] = tryInt(q.get('size_min'))
'label': toUnicode(c.get('label', '')), quality['size_max'] = tryInt(q.get('size_max'))
'ignored': toUnicode(c.get('ignored', '')), db.update(quality)
'preferred': toUnicode(c.get('preferred', '')),
'required': toUnicode(c.get('required', '')),
'destination': toUnicode(c.get('destination', '')),
})
category_link[x] = new_c.get('_id') quality_link[x] = quality
# Profiles # Titles
log.info('Importing profiles') titles = migrate_data['librarytitle']
new_profiles = db.all('profile', with_doc = True) titles_by_library = {}
new_profiles_by_label = {} for x in titles:
for x in new_profiles: title = titles[x]
if title.get('default'):
titles_by_library[title.get('libraries_id')] = title.get('title')
# Remove default non core profiles # Releases
if not x['doc'].get('core'): releaseinfos = migrate_data['releaseinfo']
db.delete(x['doc']) for x in releaseinfos:
else: info = releaseinfos[x]
new_profiles_by_label[x['doc']['label']] = x['_id']
profiles = migrate_data['profile'] # Skip if release doesn't exist for this info
profile_link = {} if not migrate_data['release'].get(info.get('release_id')):
for x in profiles:
p = profiles[x]
exists = new_profiles_by_label.get(p.get('label'))
# Update existing with order only
if exists and p.get('core'):
profile = db.get('id', exists)
profile['order'] = tryInt(p.get('order'))
profile['hide'] = p.get('hide') in [1, True, 'true', 'True']
db.update(profile)
profile_link[x] = profile.get('_id')
else:
new_profile = {
'_t': 'profile',
'label': p.get('label'),
'order': int(p.get('order', 999)),
'core': p.get('core', False),
'qualities': [],
'wait_for': [],
'finish': []
}
types = migrate_data['profiletype']
for profile_type in types:
p_type = types[profile_type]
if types[profile_type]['profile_id'] == p['id']:
if p_type['quality_id']:
new_profile['finish'].append(p_type['finish'])
new_profile['wait_for'].append(p_type['wait_for'])
new_profile['qualities'].append(migrate_data['quality'][p_type['quality_id']]['identifier'])
if len(new_profile['qualities']) > 0:
new_profile.update(db.insert(new_profile))
profile_link[x] = new_profile.get('_id')
else:
log.error('Corrupt profile list for "%s", using default.', p.get('label'))
# Qualities
log.info('Importing quality sizes')
new_qualities = db.all('quality', with_doc = True)
new_qualities_by_identifier = {}
for x in new_qualities:
new_qualities_by_identifier[x['doc']['identifier']] = x['_id']
qualities = migrate_data['quality']
quality_link = {}
for x in qualities:
q = qualities[x]
q_id = new_qualities_by_identifier[q.get('identifier')]
quality = db.get('id', q_id)
quality['order'] = q.get('order')
quality['size_min'] = tryInt(q.get('size_min'))
quality['size_max'] = tryInt(q.get('size_max'))
db.update(quality)
quality_link[x] = quality
# Titles
titles = migrate_data['librarytitle']
titles_by_library = {}
for x in titles:
title = titles[x]
if title.get('default'):
titles_by_library[title.get('libraries_id')] = title.get('title')
# Releases
releaseinfos = migrate_data['releaseinfo']
for x in releaseinfos:
info = releaseinfos[x]
# Skip if release doesn't exist for this info
if not migrate_data['release'].get(info.get('release_id')):
continue
if not migrate_data['release'][info.get('release_id')].get('info'):
migrate_data['release'][info.get('release_id')]['info'] = {}
migrate_data['release'][info.get('release_id')]['info'][info.get('identifier')] = info.get('value')
releases = migrate_data['release']
releases_by_media = {}
for x in releases:
release = releases[x]
if not releases_by_media.get(release.get('movie_id')):
releases_by_media[release.get('movie_id')] = []
releases_by_media[release.get('movie_id')].append(release)
# Type ids
types = migrate_data['filetype']
type_by_id = {}
for t in types:
type = types[t]
type_by_id[type.get('id')] = type
# Media
log.info('Importing %s media items', len(migrate_data['movie']))
statuses = migrate_data['status']
libraries = migrate_data['library']
library_files = migrate_data['library_files__file_library']
releases_files = migrate_data['release_files__file_release']
all_files = migrate_data['file']
poster_type = migrate_data['filetype']['poster']
medias = migrate_data['movie']
for x in medias:
m = medias[x]
status = statuses.get(m['status_id']).get('identifier')
l = libraries.get(m['library_id'])
# Only migrate wanted movies, Skip if no identifier present
if not l or not getImdb(l.get('identifier')): continue
profile_id = profile_link.get(m['profile_id'])
category_id = category_link.get(m['category_id'])
title = titles_by_library.get(m['library_id'])
releases = releases_by_media.get(x, [])
info = json.loads(l.get('info', ''))
files = library_files.get(m['library_id'], [])
if not isinstance(files, list):
files = [files]
added_media = fireEvent('movie.add', {
'info': info,
'identifier': l.get('identifier'),
'profile_id': profile_id,
'category_id': category_id,
'title': title
}, force_readd = False, search_after = False, update_after = False, notify_after = False, status = status, single = True)
if not added_media:
log.error('Failed adding media %s: %s', (l.get('identifier'), info))
continue
added_media['files'] = added_media.get('files', {})
for f in files:
ffile = all_files[f.get('file_id')]
# Only migrate posters
if ffile.get('type_id') == poster_type.get('id'):
if ffile.get('path') not in added_media['files'].get('image_poster', []) and os.path.isfile(ffile.get('path')):
added_media['files']['image_poster'] = [ffile.get('path')]
break
if 'image_poster' in added_media['files']:
db.update(added_media)
for rel in releases:
empty_info = False
if not rel.get('info'):
empty_info = True
rel['info'] = {}
quality = quality_link.get(rel.get('quality_id'))
if not quality:
continue continue
release_status = statuses.get(rel.get('status_id')).get('identifier') if not migrate_data['release'][info.get('release_id')].get('info'):
migrate_data['release'][info.get('release_id')]['info'] = {}
if rel['info'].get('download_id'): migrate_data['release'][info.get('release_id')]['info'][info.get('identifier')] = info.get('value')
status_support = rel['info'].get('download_status_support', False) in [True, 'true', 'True']
rel['info']['download_info'] = {
'id': rel['info'].get('download_id'),
'downloader': rel['info'].get('download_downloader'),
'status_support': status_support,
}
# Add status to keys releases = migrate_data['release']
rel['info']['status'] = release_status releases_by_media = {}
if not empty_info: for x in releases:
fireEvent('release.create_from_search', [rel['info']], added_media, quality, single = True) release = releases[x]
else: if not releases_by_media.get(release.get('movie_id')):
release = { releases_by_media[release.get('movie_id')] = []
'_t': 'release',
'identifier': rel.get('identifier'),
'media_id': added_media.get('_id'),
'quality': quality.get('identifier'),
'status': release_status,
'last_edit': int(time.time()),
'files': {}
}
# Add downloader info if provided releases_by_media[release.get('movie_id')].append(release)
try:
release['download_info'] = rel['info']['download_info']
del rel['download_info']
except:
pass
# Add files # Type ids
release_files = releases_files.get(rel.get('id'), []) types = migrate_data['filetype']
if not isinstance(release_files, list): type_by_id = {}
release_files = [release_files] for t in types:
type = types[t]
type_by_id[type.get('id')] = type
if len(release_files) == 0: # Media
log.info('Importing %s media items', len(migrate_data['movie']))
statuses = migrate_data['status']
libraries = migrate_data['library']
library_files = migrate_data['library_files__file_library']
releases_files = migrate_data['release_files__file_release']
all_files = migrate_data['file']
poster_type = migrate_data['filetype']['poster']
medias = migrate_data['movie']
for x in medias:
m = medias[x]
status = statuses.get(m['status_id']).get('identifier')
l = libraries.get(m['library_id'])
# Only migrate wanted movies, Skip if no identifier present
if not l or not getImdb(l.get('identifier')): continue
profile_id = profile_link.get(m['profile_id'])
category_id = category_link.get(m['category_id'])
title = titles_by_library.get(m['library_id'])
releases = releases_by_media.get(x, [])
info = json.loads(l.get('info', ''))
files = library_files.get(m['library_id'], [])
if not isinstance(files, list):
files = [files]
added_media = fireEvent('movie.add', {
'info': info,
'identifier': l.get('identifier'),
'profile_id': profile_id,
'category_id': category_id,
'title': title
}, force_readd = False, search_after = False, update_after = False, notify_after = False, status = status, single = True)
if not added_media:
log.error('Failed adding media %s: %s', (l.get('identifier'), info))
continue
added_media['files'] = added_media.get('files', {})
for f in files:
ffile = all_files[f.get('file_id')]
# Only migrate posters
if ffile.get('type_id') == poster_type.get('id'):
if ffile.get('path') not in added_media['files'].get('image_poster', []) and os.path.isfile(ffile.get('path')):
added_media['files']['image_poster'] = [ffile.get('path')]
break
if 'image_poster' in added_media['files']:
db.update(added_media)
for rel in releases:
empty_info = False
if not rel.get('info'):
empty_info = True
rel['info'] = {}
quality = quality_link.get(rel.get('quality_id'))
if not quality:
continue continue
for f in release_files: release_status = statuses.get(rel.get('status_id')).get('identifier')
rfile = all_files[f.get('file_id')]
file_type = type_by_id.get(rfile.get('type_id')).get('identifier')
if not release['files'].get(file_type): if rel['info'].get('download_id'):
release['files'][file_type] = [] status_support = rel['info'].get('download_status_support', False) in [True, 'true', 'True']
rel['info']['download_info'] = {
'id': rel['info'].get('download_id'),
'downloader': rel['info'].get('download_downloader'),
'status_support': status_support,
}
release['files'][file_type].append(rfile.get('path')) # Add status to keys
rel['info']['status'] = release_status
if not empty_info:
fireEvent('release.create_from_search', [rel['info']], added_media, quality, single = True)
else:
release = {
'_t': 'release',
'identifier': rel.get('identifier'),
'media_id': added_media.get('_id'),
'quality': quality.get('identifier'),
'status': release_status,
'last_edit': int(time.time()),
'files': {}
}
try: # Add downloader info if provided
rls = db.get('release_identifier', rel.get('identifier'), with_doc = True)['doc'] try:
rls.update(release) release['download_info'] = rel['info']['download_info']
db.update(rls) del rel['download_info']
except: except:
db.insert(release) pass
# Add files
release_files = releases_files.get(rel.get('id'), [])
if not isinstance(release_files, list):
release_files = [release_files]
if len(release_files) == 0:
continue
for f in release_files:
rfile = all_files.get(f.get('file_id'))
if not rfile:
continue
file_type = type_by_id.get(rfile.get('type_id')).get('identifier')
if not release['files'].get(file_type):
release['files'][file_type] = []
release['files'][file_type].append(rfile.get('path'))
try:
rls = db.get('release_identifier', rel.get('identifier'), with_doc = True)['doc']
rls.update(release)
db.update(rls)
except:
db.insert(release)
log.info('Total migration took %s', time.time() - migrate_start)
log.info('=' * 30)
rename_old = True
except OperationalError:
log.error('Migrating from faulty database, probably a (too) old version: %s', traceback.format_exc())
rename_old = True
except:
log.error('Migration failed: %s', traceback.format_exc())
log.info('Total migration took %s', time.time() - migrate_start)
log.info('=' * 30)
# rename old database # rename old database
log.info('Renaming old database to %s ', old_db + '.old') if rename_old:
os.rename(old_db, old_db + '.old') random = randomString()
log.info('Renaming old database to %s ', '%s.%s_old' % (old_db, random))
os.rename(old_db, '%s.%s_old' % (old_db, random))
if os.path.isfile(old_db + '-wal'): if os.path.isfile(old_db + '-wal'):
os.rename(old_db + '-wal', old_db + '-wal.old') os.rename(old_db + '-wal', '%s-wal.%s_old' % (old_db, random))
if os.path.isfile(old_db + '-shm'): if os.path.isfile(old_db + '-shm'):
os.rename(old_db + '-shm', old_db + '-shm.old') os.rename(old_db + '-shm', '%s-shm.%s_old' % (old_db, random))
+5
View File
@@ -27,6 +27,11 @@ class Deluge(DownloaderBase):
def connect(self, reconnect = False): def connect(self, reconnect = False):
# Load host from config and split out port. # Load host from config and split out port.
host = cleanHost(self.conf('host'), protocol = False).split(':') host = cleanHost(self.conf('host'), protocol = False).split(':')
# Force host assignment
if len(host) == 1:
host.append(80)
if not isInt(host[1]): if not isInt(host[1]):
log.error('Config properties are not filled in correctly, port is missing.') log.error('Config properties are not filled in correctly, port is missing.')
return False return False
+39 -64
View File
@@ -1,16 +1,10 @@
from base64 import b64encode from base64 import b64encode
from urllib2 import URLError import os
from uuid import uuid4 from uuid import uuid4
import hashlib import hashlib
import httplib
import json
import os
import socket
import ssl
import sys
import time
import traceback import traceback
import urllib2
from requests import HTTPError
from couchpotato.core._base.downloader.main import DownloaderBase, ReleaseDownloadList from couchpotato.core._base.downloader.main import DownloaderBase, ReleaseDownloadList
from couchpotato.core.helpers.encoding import tryUrlencode, sp from couchpotato.core.helpers.encoding import tryUrlencode, sp
@@ -35,13 +29,17 @@ class NZBVortex(DownloaderBase):
# Send the nzb # Send the nzb
try: try:
nzb_filename = self.createFileName(data, filedata, media) nzb_filename = self.createFileName(data, filedata, media, unique_tag = True)
self.call('nzb/add', files = {'file': (nzb_filename, filedata)}) response = self.call('nzb/add', files = {'file': (nzb_filename, filedata, 'application/octet-stream')}, parameters = {
'name': nzb_filename,
'groupname': self.conf('group')
})
time.sleep(10) if response and response.get('result', '').lower() == 'ok':
raw_statuses = self.call('nzb') return self.downloadReturnId(nzb_filename)
nzb_id = [nzb['id'] for nzb in raw_statuses.get('nzbs', []) if os.path.basename(nzb['nzbFileName']) == nzb_filename][0]
return self.downloadReturnId(nzb_id) log.error('Something went wrong sending the NZB file. Response: %s', response)
return False
except: except:
log.error('Something went wrong sending the NZB file: %s', traceback.format_exc()) log.error('Something went wrong sending the NZB file: %s', traceback.format_exc())
return False return False
@@ -60,7 +58,8 @@ class NZBVortex(DownloaderBase):
release_downloads = ReleaseDownloadList(self) release_downloads = ReleaseDownloadList(self)
for nzb in raw_statuses.get('nzbs', []): for nzb in raw_statuses.get('nzbs', []):
if nzb['id'] in ids: nzb_id = os.path.basename(nzb['nzbFileName'])
if nzb_id in ids:
# Check status # Check status
status = 'busy' status = 'busy'
@@ -70,7 +69,8 @@ class NZBVortex(DownloaderBase):
status = 'failed' status = 'failed'
release_downloads.append({ release_downloads.append({
'id': nzb['id'], 'temp_id': nzb['id'],
'id': nzb_id,
'name': nzb['uiTitle'], 'name': nzb['uiTitle'],
'status': status, 'status': status,
'original_status': nzb['state'], 'original_status': nzb['state'],
@@ -85,7 +85,7 @@ class NZBVortex(DownloaderBase):
log.info('%s failed downloading, deleting...', release_download['name']) log.info('%s failed downloading, deleting...', release_download['name'])
try: try:
self.call('nzb/%s/cancel' % release_download['id']) self.call('nzb/%s/cancel' % release_download['temp_id'])
except: except:
log.error('Failed deleting: %s', traceback.format_exc(0)) log.error('Failed deleting: %s', traceback.format_exc(0))
return False return False
@@ -114,7 +114,7 @@ class NZBVortex(DownloaderBase):
log.error('Login failed, please check you api-key') log.error('Login failed, please check you api-key')
return False return False
def call(self, call, parameters = None, repeat = False, auth = True, *args, **kwargs): def call(self, call, parameters = None, is_repeat = False, auth = True, *args, **kwargs):
# Login first # Login first
if not parameters: parameters = {} if not parameters: parameters = {}
@@ -127,19 +127,20 @@ class NZBVortex(DownloaderBase):
params = tryUrlencode(parameters) params = tryUrlencode(parameters)
url = cleanHost(self.conf('host'), ssl = self.conf('ssl')) + 'api/' + call url = cleanHost(self.conf('host')) + 'api/' + call
try: try:
data = self.urlopen('%s?%s' % (url, params), *args, **kwargs) data = self.getJsonData('%s%s' % (url, '?' + params if params else ''), *args, cache_timeout = 0, show_error = False, **kwargs)
if data: if data:
return json.loads(data) return data
except URLError as e: except HTTPError as e:
if hasattr(e, 'code') and e.code == 403: sc = e.response.status_code
if sc == 403:
# Try login and do again # Try login and do again
if not repeat: if not is_repeat:
self.login() self.login()
return self.call(call, parameters = parameters, repeat = True, **kwargs) return self.call(call, parameters = parameters, is_repeat = True, **kwargs)
log.error('Failed to parsing %s: %s', (self.getName(), traceback.format_exc())) log.error('Failed to parsing %s: %s', (self.getName(), traceback.format_exc()))
except: except:
@@ -151,13 +152,12 @@ class NZBVortex(DownloaderBase):
if not self.api_level: if not self.api_level:
url = cleanHost(self.conf('host')) + 'api/app/apilevel'
try: try:
data = self.urlopen(url, show_error = False) data = self.call('app/apilevel', auth = False)
self.api_level = float(json.loads(data).get('apilevel')) self.api_level = float(data.get('apilevel'))
except URLError as e: except HTTPError as e:
if hasattr(e, 'code') and e.code == 403: sc = e.response.status_code
if sc == 403:
log.error('This version of NZBVortex isn\'t supported. Please update to 2.8.6 or higher') log.error('This version of NZBVortex isn\'t supported. Please update to 2.8.6 or higher')
else: else:
log.error('NZBVortex doesn\'t seem to be running or maybe the remote option isn\'t enabled yet: %s', traceback.format_exc(1)) log.error('NZBVortex doesn\'t seem to be running or maybe the remote option isn\'t enabled yet: %s', traceback.format_exc(1))
@@ -169,29 +169,6 @@ class NZBVortex(DownloaderBase):
return super(NZBVortex, self).isEnabled(manual, data) and self.getApiLevel() return super(NZBVortex, self).isEnabled(manual, data) and self.getApiLevel()
class HTTPSConnection(httplib.HTTPSConnection):
def __init__(self, *args, **kwargs):
httplib.HTTPSConnection.__init__(self, *args, **kwargs)
def connect(self):
sock = socket.create_connection((self.host, self.port), self.timeout)
if sys.version_info < (2, 6, 7):
if hasattr(self, '_tunnel_host'):
self.sock = sock
self._tunnel()
else:
if self._tunnel_host:
self.sock = sock
self._tunnel()
self.sock = ssl.wrap_socket(sock, self.key_file, self.cert_file, ssl_version = ssl.PROTOCOL_TLSv1)
class HTTPSHandler(urllib2.HTTPSHandler):
def https_open(self, req):
return self.do_open(HTTPSConnection, req)
config = [{ config = [{
'name': 'nzbvortex', 'name': 'nzbvortex',
'groups': [ 'groups': [
@@ -211,20 +188,18 @@ config = [{
}, },
{ {
'name': 'host', 'name': 'host',
'default': 'localhost:4321', 'default': 'https://localhost:4321',
'description': 'Hostname with port. Usually <strong>localhost:4321</strong>', 'description': 'Hostname with port. Usually <strong>https://localhost:4321</strong>',
},
{
'name': 'ssl',
'default': 1,
'type': 'bool',
'advanced': True,
'description': 'Use HyperText Transfer Protocol Secure, or <strong>https</strong>',
}, },
{ {
'name': 'api_key', 'name': 'api_key',
'label': 'Api Key', 'label': 'Api Key',
}, },
{
'name': 'group',
'label': 'Group',
'description': 'The group CP places the nzb in. Make sure to create it in NZBVortex.',
},
{ {
'name': 'manual', 'name': 'manual',
'default': False, 'default': False,
+9 -7
View File
@@ -25,7 +25,7 @@ class Transmission(DownloaderBase):
def connect(self): def connect(self):
# Load host from config and split out port. # Load host from config and split out port.
host = cleanHost(self.conf('host'), protocol = False).split(':') host = cleanHost(self.conf('host')).rstrip('/').rsplit(':', 1)
if not isInt(host[1]): if not isInt(host[1]):
log.error('Config properties are not filled in correctly, port is missing.') log.error('Config properties are not filled in correctly, port is missing.')
return False return False
@@ -78,12 +78,14 @@ class Transmission(DownloaderBase):
log.error('Failed sending torrent to Transmission') log.error('Failed sending torrent to Transmission')
return False return False
data = remote_torrent.get('torrent-added') or remote_torrent.get('torrent-duplicate')
# Change settings of added torrents # Change settings of added torrents
if torrent_params: if torrent_params:
self.trpc.set_torrent(remote_torrent['torrent-added']['hashString'], torrent_params) self.trpc.set_torrent(data['hashString'], torrent_params)
log.info('Torrent sent to Transmission successfully.') log.info('Torrent sent to Transmission successfully.')
return self.downloadReturnId(remote_torrent['torrent-added']['hashString']) return self.downloadReturnId(data['hashString'])
def test(self): def test(self):
if self.connect() and self.trpc.get_session(): if self.connect() and self.trpc.get_session():
@@ -162,11 +164,11 @@ class Transmission(DownloaderBase):
class TransmissionRPC(object): class TransmissionRPC(object):
"""TransmissionRPC lite library""" """TransmissionRPC lite library"""
def __init__(self, host = 'localhost', port = 9091, rpc_url = 'transmission', username = None, password = None): def __init__(self, host = 'http://localhost', port = 9091, rpc_url = 'transmission', username = None, password = None):
super(TransmissionRPC, self).__init__() super(TransmissionRPC, self).__init__()
self.url = 'http://' + host + ':' + str(port) + '/' + rpc_url + '/rpc' self.url = host + ':' + str(port) + '/' + rpc_url + '/rpc'
self.tag = 0 self.tag = 0
self.session_id = 0 self.session_id = 0
self.session = {} self.session = {}
@@ -274,8 +276,8 @@ config = [{
}, },
{ {
'name': 'host', 'name': 'host',
'default': 'localhost:9091', 'default': 'http://localhost:9091',
'description': 'Hostname with port. Usually <strong>localhost:9091</strong>', 'description': 'Hostname with port. Usually <strong>http://localhost:9091</strong>',
}, },
{ {
'name': 'rpc_url', 'name': 'rpc_url',
+1 -1
View File
@@ -90,7 +90,7 @@ def fireEvent(name, *args, **kwargs):
else: else:
e = Event(name = name, threads = 10, exc_info = True, traceback = True, lock = threading.RLock()) e = Event(name = name, threads = 10, exc_info = True, traceback = True)
for event in events[name]: for event in events[name]:
e.handle(event['handler'], priority = event['priority']) e.handle(event['handler'], priority = event['priority'])
+8 -1
View File
@@ -5,6 +5,7 @@ import re
import traceback import traceback
import unicodedata import unicodedata
from chardet import detect
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
import six import six
@@ -35,6 +36,9 @@ def toUnicode(original, *args):
return six.text_type(original, *args) return six.text_type(original, *args)
except: except:
try: try:
detected = detect(original)
if detected.get('encoding') == 'utf-8':
return original.decode('utf-8')
return ek(original, *args) return ek(original, *args)
except: except:
raise raise
@@ -52,7 +56,10 @@ def ss(original, *args):
return u_original.encode(Env.get('encoding')) return u_original.encode(Env.get('encoding'))
except Exception as e: except Exception as e:
log.debug('Failed ss encoding char, force UTF8: %s', e) log.debug('Failed ss encoding char, force UTF8: %s', e)
return u_original.encode('UTF-8') try:
return u_original.encode(Env.get('encoding'), 'replace')
except:
return u_original.encode('utf-8', 'replace')
def sp(path, *args): def sp(path, *args):
+24 -2
View File
@@ -41,11 +41,11 @@ def symlink(src, dst):
def getUserDir(): def getUserDir():
try: try:
import pwd import pwd
os.environ['HOME'] = pwd.getpwuid(os.geteuid()).pw_dir os.environ['HOME'] = sp(pwd.getpwuid(os.geteuid()).pw_dir)
except: except:
pass pass
return os.path.expanduser('~') return sp(os.path.expanduser('~'))
def getDownloadDir(): def getDownloadDir():
@@ -382,6 +382,28 @@ def getFreeSpace(directories):
return free_space return free_space
def getSize(paths):
single = not isinstance(paths, (tuple, list))
if single:
paths = [paths]
total_size = 0
for path in paths:
path = sp(path)
if os.path.isdir(path):
total_size = 0
for dirpath, _, filenames in os.walk(path):
for f in filenames:
total_size += os.path.getsize(sp(os.path.join(dirpath, f)))
elif os.path.isfile(path):
total_size += os.path.getsize(path)
return total_size / 1048576 # MB
def find(func, iterable): def find(func, iterable):
for item in iterable: for item in iterable:
if func(item): if func(item):
+8 -9
View File
@@ -59,15 +59,14 @@ class CPLog(object):
msg = ss(msg) msg = ss(msg)
try: try:
msg = msg % replace_tuple if isinstance(replace_tuple, tuple):
except: msg = msg % tuple([ss(x) if not isinstance(x, (int, float)) else x for x in list(replace_tuple)])
try: elif isinstance(replace_tuple, dict):
if isinstance(replace_tuple, tuple): msg = msg % dict((k, ss(v) if not isinstance(v, (int, float)) else v) for k, v in replace_tuple.iteritems())
msg = msg % tuple([ss(x) for x in list(replace_tuple)]) else:
else: msg = msg % ss(replace_tuple)
msg = msg % ss(replace_tuple) except Exception as e:
except Exception as e: self.logger.error('Failed encoding stuff to log "%s": %s' % (msg, e))
self.logger.error('Failed encoding stuff to log "%s": %s' % (msg, e))
self.setup() self.setup()
if not self.is_develop: if not self.is_develop:
+7 -7
View File
@@ -26,9 +26,9 @@ class MediaBase(Plugin):
def onComplete(): def onComplete():
try: try:
media = fireEvent('media.get', media_id, single = True) media = fireEvent('media.get', media_id, single = True)
event_name = '%s.searcher.single' % media.get('type') if media:
event_name = '%s.searcher.single' % media.get('type')
fireEventAsync(event_name, media, on_complete = self.createNotifyFront(media_id), manual = True) fireEventAsync(event_name, media, on_complete = self.createNotifyFront(media_id), manual = True)
except: except:
log.error('Failed creating onComplete: %s', traceback.format_exc()) log.error('Failed creating onComplete: %s', traceback.format_exc())
@@ -39,9 +39,9 @@ class MediaBase(Plugin):
def notifyFront(): def notifyFront():
try: try:
media = fireEvent('media.get', media_id, single = True) media = fireEvent('media.get', media_id, single = True)
event_name = '%s.update' % media.get('type') if media:
event_name = '%s.update' % media.get('type')
fireEvent('notify.frontend', type = event_name, data = media) fireEvent('notify.frontend', type = event_name, data = media)
except: except:
log.error('Failed creating onComplete: %s', traceback.format_exc()) log.error('Failed creating onComplete: %s', traceback.format_exc())
@@ -95,7 +95,7 @@ class MediaBase(Plugin):
if file_type not in existing_files or len(existing_files.get(file_type, [])) == 0: if file_type not in existing_files or len(existing_files.get(file_type, [])) == 0:
file_path = fireEvent('file.download', url = image, single = True) file_path = fireEvent('file.download', url = image, single = True)
if file_path: if file_path:
existing_files[file_type] = [file_path] existing_files[file_type] = [toUnicode(file_path)]
break break
else: else:
break break
+40 -14
View File
@@ -1,5 +1,4 @@
from datetime import timedelta from datetime import timedelta
from operator import itemgetter
import time import time
import traceback import traceback
from string import ascii_lowercase from string import ascii_lowercase
@@ -78,6 +77,7 @@ class MediaPlugin(MediaBase):
addEvent('app.load', self.addSingleListView, priority = 100) addEvent('app.load', self.addSingleListView, priority = 100)
addEvent('app.load', self.addSingleCharView, priority = 100) addEvent('app.load', self.addSingleCharView, priority = 100)
addEvent('app.load', self.addSingleDeleteView, priority = 100) addEvent('app.load', self.addSingleDeleteView, priority = 100)
addEvent('app.load', self.cleanupFaults)
addEvent('media.get', self.get) addEvent('media.get', self.get)
addEvent('media.with_status', self.withStatus) addEvent('media.with_status', self.withStatus)
@@ -88,6 +88,18 @@ class MediaPlugin(MediaBase):
addEvent('media.tag', self.tag) addEvent('media.tag', self.tag)
addEvent('media.untag', self.unTag) addEvent('media.untag', self.unTag)
# Wrongly tagged media files
def cleanupFaults(self):
medias = fireEvent('media.with_status', 'ignored', single = True) or []
db = get_db()
for media in medias:
try:
media['status'] = 'done'
db.update(media)
except:
pass
def refresh(self, id = '', **kwargs): def refresh(self, id = '', **kwargs):
handlers = [] handlers = []
ids = splitString(id) ids = splitString(id)
@@ -179,8 +191,10 @@ class MediaPlugin(MediaBase):
continue continue
yield doc yield doc
except RecordNotFound: except (RecordDeleted, RecordNotFound):
log.debug('Record not found, skipping: %s', ms['_id']) log.debug('Record not found, skipping: %s', ms['_id'])
except (ValueError, EOFError):
fireEvent('database.delete_corrupted', ms.get('_id'), traceback_error = traceback.format_exc(0))
else: else:
yield ms yield ms
@@ -193,6 +207,7 @@ class MediaPlugin(MediaBase):
except: except:
pass pass
log.debug('No media found with identifiers: %s', identifiers)
return False return False
def list(self, types = None, status = None, release_status = None, status_or = False, limit_offset = None, with_tags = None, starts_with = None, search = None): def list(self, types = None, status = None, release_status = None, status_or = False, limit_offset = None, with_tags = None, starts_with = None, search = None):
@@ -276,6 +291,10 @@ class MediaPlugin(MediaBase):
media = fireEvent('media.get', media_id, single = True) media = fireEvent('media.get', media_id, single = True)
# Skip if no media has been found
if not media:
continue
# Merge releases with movie dict # Merge releases with movie dict
medias.append(media) medias.append(media)
@@ -327,7 +346,7 @@ class MediaPlugin(MediaBase):
def addSingleListView(self): def addSingleListView(self):
for media_type in fireEvent('media.types', merge = True): for media_type in fireEvent('media.types', merge = True):
tempList = lambda media_type = media_type, *args, **kwargs : self.listView(type = media_type, **kwargs) tempList = lambda *args, **kwargs : self.listView(type = media_type, **kwargs)
addApiView('%s.list' % media_type, tempList, docs = { addApiView('%s.list' % media_type, tempList, docs = {
'desc': 'List media', 'desc': 'List media',
'params': { 'params': {
@@ -388,7 +407,7 @@ class MediaPlugin(MediaBase):
if x['_id'] in media_ids: if x['_id'] in media_ids:
chars.add(x['key']) chars.add(x['key'])
if len(chars) == 25: if len(chars) == 27:
break break
return list(chars) return list(chars)
@@ -409,7 +428,7 @@ class MediaPlugin(MediaBase):
def addSingleCharView(self): def addSingleCharView(self):
for media_type in fireEvent('media.types', merge = True): for media_type in fireEvent('media.types', merge = True):
tempChar = lambda media_type = media_type, *args, **kwargs : self.charView(type = media_type, **kwargs) tempChar = lambda *args, **kwargs : self.charView(type = media_type, **kwargs)
addApiView('%s.available_chars' % media_type, tempChar) addApiView('%s.available_chars' % media_type, tempChar)
def delete(self, media_id, delete_from = None): def delete(self, media_id, delete_from = None):
@@ -447,11 +466,16 @@ class MediaPlugin(MediaBase):
db.delete(release) db.delete(release)
total_deleted += 1 total_deleted += 1
if (total_releases == total_deleted and media['status'] != 'active') or (total_releases == 0 and not new_media_status) or (not new_media_status and delete_from == 'late'): if (total_releases == total_deleted) or (total_releases == 0 and not new_media_status) or (not new_media_status and delete_from == 'late'):
db.delete(media) db.delete(media)
deleted = True deleted = True
elif new_media_status: elif new_media_status:
media['status'] = new_media_status media['status'] = new_media_status
# Remove profile (no use for in manage)
if new_media_status == 'done':
media['profile_id'] = None
db.update(media) db.update(media)
fireEvent('media.untag', media['_id'], 'recent', single = True) fireEvent('media.untag', media['_id'], 'recent', single = True)
@@ -478,7 +502,7 @@ class MediaPlugin(MediaBase):
def addSingleDeleteView(self): def addSingleDeleteView(self):
for media_type in fireEvent('media.types', merge = True): for media_type in fireEvent('media.types', merge = True):
tempDelete = lambda media_type = media_type, *args, **kwargs : self.deleteView(type = media_type, **kwargs) tempDelete = lambda *args, **kwargs : self.deleteView(type = media_type, **kwargs)
addApiView('%s.delete' % media_type, tempDelete, docs = { addApiView('%s.delete' % media_type, tempDelete, docs = {
'desc': 'Delete a ' + media_type + ' from the wanted list', 'desc': 'Delete a ' + media_type + ' from the wanted list',
'params': { 'params': {
@@ -487,7 +511,7 @@ class MediaPlugin(MediaBase):
} }
}) })
def restatus(self, media_id): def restatus(self, media_id, tag_recent = True, allowed_restatus = None):
try: try:
db = get_db() db = get_db()
@@ -507,12 +531,13 @@ class MediaPlugin(MediaBase):
done_releases = [release for release in media_releases if release.get('status') == 'done'] done_releases = [release for release in media_releases if release.get('status') == 'done']
if done_releases: if done_releases:
# Only look at latest added release
release = sorted(done_releases, key = itemgetter('last_edit'), reverse = True)[0]
# Check if we are finished with the media # Check if we are finished with the media
if fireEvent('quality.isfinish', {'identifier': release['quality'], 'is_3d': release.get('is_3d', False)}, profile, timedelta(seconds = time.time() - release['last_edit']).days, single = True): for release in done_releases:
m['status'] = 'done' if fireEvent('quality.isfinish', {'identifier': release['quality'], 'is_3d': release.get('is_3d', False)}, profile, timedelta(seconds = time.time() - release['last_edit']).days, single = True):
m['status'] = 'done'
break
elif previous_status == 'done': elif previous_status == 'done':
m['status'] = 'done' m['status'] = 'done'
@@ -521,11 +546,12 @@ class MediaPlugin(MediaBase):
m['status'] = previous_status m['status'] = previous_status
# Only update when status has changed # Only update when status has changed
if previous_status != m['status']: if previous_status != m['status'] and (not allowed_restatus or m['status'] in allowed_restatus):
db.update(m) db.update(m)
# Tag media as recent # Tag media as recent
self.tag(media_id, 'recent', update_edited = True) if tag_recent:
self.tag(media_id, 'recent', update_edited = True)
return m['status'] return m['status']
except: except:
@@ -45,7 +45,7 @@ class Base(NZBProvider, RSS):
def _searchOnHost(self, host, media, quality, results): def _searchOnHost(self, host, media, quality, results):
query = self.buildUrl(media, host) query = self.buildUrl(media, host)
url = '%s&%s' % (self.getUrl(host['host']), query) url = '%s%s' % (self.getUrl(host['host']), query)
nzbs = self.getRSSData(url, cache_timeout = 1800, headers = {'User-Agent': Env.getIdentifier()}) nzbs = self.getRSSData(url, cache_timeout = 1800, headers = {'User-Agent': Env.getIdentifier()})
for nzb in nzbs: for nzb in nzbs:
@@ -83,7 +83,7 @@ class Base(NZBProvider, RSS):
try: try:
# Get details for extended description to retrieve passwords # Get details for extended description to retrieve passwords
query = self.buildDetailsUrl(nzb_id, host['api_key']) query = self.buildDetailsUrl(nzb_id, host['api_key'])
url = '%s&%s' % (self.getUrl(host['host']), query) url = '%s%s' % (self.getUrl(host['host']), query)
nzb_details = self.getRSSData(url, cache_timeout = 1800, headers = {'User-Agent': Env.getIdentifier()})[0] nzb_details = self.getRSSData(url, cache_timeout = 1800, headers = {'User-Agent': Env.getIdentifier()})[0]
description = self.getTextElement(nzb_details, 'description') description = self.getTextElement(nzb_details, 'description')
@@ -187,11 +187,12 @@ class Base(NZBProvider, RSS):
self.limits_reached[host] = False self.limits_reached[host] = False
return data return data
except HTTPError as e: except HTTPError as e:
if e.code == 503: sc = e.response.status_code
if sc in [503, 429]:
response = e.read().lower() response = e.read().lower()
if 'maximum api' in response or 'download limit' in response: if sc == 429 or 'maximum api' in response or 'download limit' in response:
if not self.limits_reached.get(host): if not self.limits_reached.get(host):
log.error('Limit reached for newznab provider: %s', host) log.error('Limit reached / to many requests for newznab provider: %s', host)
self.limits_reached[host] = time.time() self.limits_reached[host] = time.time()
return 'try_next' return 'try_next'
@@ -1,126 +0,0 @@
import re
import time
from bs4 import BeautifulSoup
from couchpotato.core.helpers.encoding import toUnicode
from couchpotato.core.helpers.rss import RSS
from couchpotato.core.helpers.variable import tryInt
from couchpotato.core.logger import CPLog
from couchpotato.core.event import fireEvent
from couchpotato.core.media._base.providers.nzb.base import NZBProvider
from dateutil.parser import parse
log = CPLog(__name__)
class Base(NZBProvider, RSS):
urls = {
'download': 'https://www.nzbindex.com/download/',
'search': 'https://www.nzbindex.com/rss/?%s',
}
http_time_between_calls = 1 # Seconds
def _search(self, media, quality, results):
nzbs = self.getRSSData(self.urls['search'] % self.buildUrl(media, quality))
for nzb in nzbs:
enclosure = self.getElement(nzb, 'enclosure').attrib
nzbindex_id = int(self.getTextElement(nzb, "link").split('/')[4])
title = self.getTextElement(nzb, "title")
match = fireEvent('matcher.parse', title, parser='usenet', single = True)
if not match.chains:
log.info('Unable to parse release with title "%s"', title)
continue
# TODO should we consider other lower-weight chains here?
info = fireEvent('matcher.flatten_info', match.chains[0].info, single = True)
release_name = fireEvent('matcher.construct_from_raw', info.get('release_name'), single = True)
file_name = info.get('detail', {}).get('file_name')
file_name = file_name[0] if file_name else None
title = release_name or file_name
# Strip extension from parsed title (if one exists)
ext_pos = title.rfind('.')
# Assume extension if smaller than 4 characters
# TODO this should probably be done a better way
if len(title[ext_pos + 1:]) <= 4:
title = title[:ext_pos]
if not title:
log.info('Unable to find release name from match')
continue
try:
description = self.getTextElement(nzb, "description")
except:
description = ''
def extra_check(item):
if '#c20000' in item['description'].lower():
log.info('Wrong: Seems to be passworded: %s', item['name'])
return False
return True
results.append({
'id': nzbindex_id,
'name': title,
'age': self.calculateAge(int(time.mktime(parse(self.getTextElement(nzb, "pubDate")).timetuple()))),
'size': tryInt(enclosure['length']) / 1024 / 1024,
'url': enclosure['url'],
'detail_url': enclosure['url'].replace('/download/', '/release/'),
'description': description,
'get_more_info': self.getMoreInfo,
'extra_check': extra_check,
})
def getMoreInfo(self, item):
try:
if '/nfo/' in item['description'].lower():
nfo_url = re.search('href=\"(?P<nfo>.+)\" ', item['description']).group('nfo')
full_description = self.getCache('nzbindex.%s' % item['id'], url = nfo_url, cache_timeout = 25920000)
html = BeautifulSoup(full_description)
item['description'] = toUnicode(html.find('pre', attrs = {'id': 'nfo0'}).text)
except:
pass
config = [{
'name': 'nzbindex',
'groups': [
{
'tab': 'searcher',
'list': 'nzb_providers',
'name': 'nzbindex',
'description': 'Free provider, less accurate. See <a href="https://www.nzbindex.com/">NZBIndex</a>',
'wizard': True,
'icon': 'iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAYAAAAf8/9hAAAAo0lEQVR42t2SQQ2AMBAEcUCwUAv94QMLfHliAQtYqIVawEItYAG6yZFMLkUANNlk79Kbbtp2P1j9uKxVV9VWFeStl+Wh3fWK9hNwEoADZkJtMD49AqS5AUjWGx6A+m+ARICGrM5W+wSTB0gETKzdHZwCEZAJ8PGZQN4AiQAmkR9s06EBAugJiBoAAPFfAQcBgZcIHzwA6TYP4JsXeSg3P9L31w3eksbH3zMb/wAAAABJRU5ErkJggg==',
'options': [
{
'name': 'enabled',
'type': 'enabler',
'default': True,
},
{
'name': 'extra_score',
'advanced': True,
'label': 'Extra Score',
'type': 'int',
'default': 0,
'description': 'Starting score for each release found via this provider.',
}
],
},
],
}]
@@ -22,6 +22,9 @@ class Base(TorrentProvider):
http_time_between_calls = 1 # Seconds http_time_between_calls = 1 # Seconds
only_tables_tags = SoupStrainer('table') only_tables_tags = SoupStrainer('table')
torrent_name_cell = 1
torrent_download_cell = 2
def _searchOnTitle(self, title, movie, quality, results): def _searchOnTitle(self, title, movie, quality, results):
url = self.urls['search'] % self.buildUrl(title, movie, quality) url = self.urls['search'] % self.buildUrl(title, movie, quality)
@@ -40,8 +43,8 @@ class Base(TorrentProvider):
all_cells = result.find_all('td') all_cells = result.find_all('td')
torrent = all_cells[1].find('a') torrent = all_cells[self.torrent_name_cell].find('a')
download = all_cells[3].find('a') download = all_cells[self.torrent_download_cell].find('a')
torrent_id = torrent['href'] torrent_id = torrent['href']
torrent_id = torrent_id.replace('details.php?id=', '') torrent_id = torrent_id.replace('details.php?id=', '')
@@ -49,9 +52,9 @@ class Base(TorrentProvider):
torrent_name = torrent.getText() torrent_name = torrent.getText()
torrent_size = self.parseSize(all_cells[7].getText()) torrent_size = self.parseSize(all_cells[8].getText())
torrent_seeders = tryInt(all_cells[9].getText()) torrent_seeders = tryInt(all_cells[10].getText())
torrent_leechers = tryInt(all_cells[10].getText()) torrent_leechers = tryInt(all_cells[11].getText())
torrent_url = self.urls['baseurl'] % download['href'] torrent_url = self.urls['baseurl'] % download['href']
torrent_detail_url = self.urls['baseurl'] % torrent['href'] torrent_detail_url = self.urls['baseurl'] % torrent['href']
@@ -34,8 +34,7 @@ class Base(TorrentMagnetProvider):
'http://kickass.pw', 'http://kickass.pw',
'http://kickassto.come.in', 'http://kickassto.come.in',
'http://katproxy.ws', 'http://katproxy.ws',
'http://www.kickassunblock.info', 'http://kickass.bitproxy.eu',
'http://www.kickassproxy.info',
'http://katph.eu', 'http://katph.eu',
'http://kickassto.come.in', 'http://kickassto.come.in',
] ]
@@ -64,6 +64,10 @@ class Base(TorrentProvider):
torrentdesc += ' HQ' torrentdesc += ' HQ'
if self.conf('prefer_golden'): if self.conf('prefer_golden'):
torrentscore += 5000 torrentscore += 5000
if 'FreeleechType' in torrent:
torrentdesc += ' Freeleech'
if self.conf('prefer_freeleech'):
torrentscore += 7000
if 'Scene' in torrent and torrent['Scene']: if 'Scene' in torrent and torrent['Scene']:
torrentdesc += ' Scene' torrentdesc += ' Scene'
if self.conf('prefer_scene'): if self.conf('prefer_scene'):
@@ -223,6 +227,14 @@ config = [{
'default': 1, 'default': 1,
'description': 'Favors Golden Popcorn-releases over all other releases.' 'description': 'Favors Golden Popcorn-releases over all other releases.'
}, },
{
'name': 'prefer_freeleech',
'advanced': True,
'type': 'bool',
'label': 'Prefer Freeleech',
'default': 1,
'description': 'Favors torrents marked as freeleech over all other releases.'
},
{ {
'name': 'prefer_scene', 'name': 'prefer_scene',
'advanced': True, 'advanced': True,
@@ -24,16 +24,16 @@ class Base(TorrentMagnetProvider):
http_time_between_calls = 0 http_time_between_calls = 0
proxy_list = [ proxy_list = [
'https://nobay.net', 'https://dieroschtibay.org',
'https://thebay.al', 'https://thebay.al',
'https://thepiratebay.se', 'https://thepiratebay.se',
'http://thepiratebay.cd', 'http://thepiratebay.se.net',
'http://thebootlegbay.com', 'http://thebootlegbay.com',
'http://www.tpb.gr', 'http://tpb.ninja.so',
'http://tpbproxy.co.uk', 'http://proxybay.fr',
'http://pirateproxy.in', 'http://pirateproxy.in',
'http://www.getpirate.com', 'http://piratebay.skey.sk',
'http://piratebay.io', 'http://pirateproxy.be',
'http://bayproxy.li', 'http://bayproxy.li',
'http://proxybay.pw', 'http://proxybay.pw',
] ]
@@ -1,126 +0,0 @@
import traceback
from bs4 import BeautifulSoup
from couchpotato.core.helpers.variable import tryInt
from couchpotato.core.logger import CPLog
from couchpotato.core.media._base.providers.torrent.base import TorrentProvider
import six
log = CPLog(__name__)
class Base(TorrentProvider):
urls = {
'test': 'http://www.torrentleech.org/',
'login': 'http://www.torrentleech.org/user/account/login/',
'login_check': 'http://torrentleech.org/user/messages',
'detail': 'http://www.torrentleech.org/torrent/%s',
'search': 'http://www.torrentleech.org/torrents/browse/index/query/%s/categories/%d',
'download': 'http://www.torrentleech.org%s',
}
http_time_between_calls = 1 # Seconds
cat_backup_id = None
def _searchOnTitle(self, title, media, quality, results):
url = self.urls['search'] % self.buildUrl(title, media, quality)
data = self.getHTMLData(url)
if data:
html = BeautifulSoup(data)
try:
result_table = html.find('table', attrs = {'id': 'torrenttable'})
if not result_table:
return
entries = result_table.find_all('tr')
for result in entries[1:]:
link = result.find('td', attrs = {'class': 'name'}).find('a')
url = result.find('td', attrs = {'class': 'quickdownload'}).find('a')
details = result.find('td', attrs = {'class': 'name'}).find('a')
results.append({
'id': link['href'].replace('/torrent/', ''),
'name': six.text_type(link.string),
'url': self.urls['download'] % url['href'],
'detail_url': self.urls['download'] % details['href'],
'size': self.parseSize(result.find_all('td')[4].string),
'seeders': tryInt(result.find('td', attrs = {'class': 'seeders'}).string),
'leechers': tryInt(result.find('td', attrs = {'class': 'leechers'}).string),
})
except:
log.error('Failed to parsing %s: %s', (self.getName(), traceback.format_exc()))
def getLoginParams(self):
return {
'username': self.conf('username'),
'password': self.conf('password'),
'remember_me': 'on',
'login': 'submit',
}
def loginSuccess(self, output):
return '/user/account/logout' in output.lower() or 'welcome back' in output.lower()
loginCheckSuccess = loginSuccess
config = [{
'name': 'torrentleech',
'groups': [
{
'tab': 'searcher',
'list': 'torrent_providers',
'name': 'TorrentLeech',
'description': '<a href="http://torrentleech.org">TorrentLeech</a>',
'wizard': True,
'icon': 'iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAIAAACQkWg2AAACHUlEQVR4AZVSO48SYRSdGTCBEMKzILLAWiybkKAGMZRUUJEoDZX7B9zsbuQPYEEjNLTQkYgJDwsoSaxspEBsCITXjjNAIKi8AkzceXgmbHQ1NJ5iMufmO9/9zrmXlCSJ+B8o75J8Pp/NZj0eTzweBy0Wi4PBYD6f12o1r9ebTCZx+22HcrnMsuxms7m6urTZ7LPZDMVYLBZ8ZV3yo8aq9Pq0wzCMTqe77dDv9y8uLyAWBH6xWOyL0K/56fcb+rrPgPZ6PZfLRe1fsl6vCUmGKIqoqNXqdDr9Dbjps9znUV0uTqdTjuPkDoVCIfcuJ4gizjMMm8u9vW+1nr04czqdK56c37CbKY9j2+1WEARZ0Gq1RFHAz2q1qlQqXxoN69HRcDjUarW8ZD6QUigUOnY8uKYH8N1sNkul9yiGw+F6vS4Rxn8EsodEIqHRaOSnq9T7ajQazWQycEIR1AEBYDabSZJyHDucJyegwWBQr9ebTCaKvHd4cCQANUU9evwQ1Ofz4YvUKUI43GE8HouSiFiNRhOowWBIpVLyHITJkuW3PwgAEf3pgIwxF5r+OplMEsk3CPT5szCMnY7EwUdhwUh/CXiej0Qi3idPz89fdrpdbsfBzH7S3Q9K5pP4c0sAKpVKoVAQGO1ut+t0OoFAQHkH2Da/3/+but3uarWK0ZMQoNdyucRutdttmqZxMTzY7XaYxsrgtUjEZrNhkSwWyy/0NCatZumrNQAAAABJRU5ErkJggg==',
'options': [
{
'name': 'enabled',
'type': 'enabler',
'default': False,
},
{
'name': 'username',
'default': '',
},
{
'name': 'password',
'default': '',
'type': 'password',
},
{
'name': 'seed_ratio',
'label': 'Seed ratio',
'type': 'float',
'default': 1,
'description': 'Will not be (re)moved until this seed ratio is met.',
},
{
'name': 'seed_time',
'label': 'Seed time',
'type': 'int',
'default': 40,
'description': 'Will not be (re)moved until this seed time (in hours) is met.',
},
{
'name': 'extra_score',
'advanced': True,
'label': 'Extra Score',
'type': 'int',
'default': 20,
'description': 'Starting score for each release found via this provider.',
}
],
},
],
}]
@@ -13,12 +13,12 @@ log = CPLog(__name__)
class Base(TorrentProvider): class Base(TorrentProvider):
urls = { urls = {
'test': 'https://torrentshack.net/', 'test': 'http://torrentshack.eu/',
'login': 'https://torrentshack.net/login.php', 'login': 'http://torrentshack.eu/login.php',
'login_check': 'https://torrentshack.net/inbox.php', 'login_check': 'http://torrentshack.eu/inbox.php',
'detail': 'https://torrentshack.net/torrent/%s', 'detail': 'http://torrentshack.eu/torrent/%s',
'search': 'https://torrentshack.net/torrents.php?action=advanced&searchstr=%s&scene=%s&filter_cat[%d]=1', 'search': 'http://torrentshack.eu/torrents.php?action=advanced&searchstr=%s&scene=%s&filter_cat[%d]=1',
'download': 'https://torrentshack.net/%s', 'download': 'http://torrentshack.eu/%s',
} }
http_time_between_calls = 1 # Seconds http_time_between_calls = 1 # Seconds
@@ -42,6 +42,7 @@ class Base(TorrentProvider):
link = result.find('span', attrs = {'class': 'torrent_name_link'}).parent link = result.find('span', attrs = {'class': 'torrent_name_link'}).parent
url = result.find('td', attrs = {'class': 'torrent_td'}).find('a') url = result.find('td', attrs = {'class': 'torrent_td'}).find('a')
tds = result.find_all('td')
results.append({ results.append({
'id': link['href'].replace('torrents.php?torrentid=', ''), 'id': link['href'].replace('torrents.php?torrentid=', ''),
@@ -49,8 +50,8 @@ class Base(TorrentProvider):
'url': self.urls['download'] % url['href'], 'url': self.urls['download'] % url['href'],
'detail_url': self.urls['download'] % link['href'], 'detail_url': self.urls['download'] % link['href'],
'size': self.parseSize(result.find_all('td')[5].string), 'size': self.parseSize(result.find_all('td')[5].string),
'seeders': tryInt(result.find_all('td')[7].string), 'seeders': tryInt(tds[len(tds)-2].string),
'leechers': tryInt(result.find_all('td')[8].string), 'leechers': tryInt(tds[len(tds)-1].string),
}) })
except: except:
@@ -80,7 +81,7 @@ config = [{
'tab': 'searcher', 'tab': 'searcher',
'list': 'torrent_providers', 'list': 'torrent_providers',
'name': 'TorrentShack', 'name': 'TorrentShack',
'description': '<a href="https://www.torrentshack.net/">TorrentShack</a>', 'description': '<a href="http://torrentshack.eu/">TorrentShack</a>',
'wizard': True, 'wizard': True,
'icon': 'iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAIAAACQkWg2AAABmElEQVQoFQXBzY2cVRiE0afqvd84CQiAnxWWtyxsS6ThINBYg2Dc7mZBMEjE4mzs6e9WcY5+ePNuVFJJodQAoLo+SaWCy9rcV8cmjah3CI6iYu7oRU30kE5xxELRfamklY3k1NL19sSm7vPzP/ZdNZzKVDaY2sPZJBh9fv5ITrmG2+Vp4e1sPchVqTCQZJnVXi+/L4uuAJGly1+Pw8CprLbi8Om7tbT19/XRqJUk11JP9uHj9ulxhXbvJbI9qJvr5YkGXFG2IBT8tXczt+sfzDZCp3765f3t9tHEHGEDACma77+8o4oATKk+/PfW9YmHruRFjWoVSFsVsGu1YSKq6Oc37+n98unPZSRlY7vsKDqN+92X3yR9+PdXee3iJNKMStqdcZqoTJbUSi5JOkpfRlhSI0mSpEmCFKoU7FqSNOLAk54uGwCStMUCgLrVic62g7oDoFmmdI+P3S0pDe1xvDqb6XrZqbtzShWNoh9fv/XQHaDdM9OqrZi2M7M3UrB2vlkPS1IbdEBk7UiSoD6VlZ6aKWer4aH4f/AvKoHUTjuyAAAAAElFTkSuQmCC', 'icon': 'iVBORw0KGgoAAAANSUhEUgAAABAAAAAQCAIAAACQkWg2AAABmElEQVQoFQXBzY2cVRiE0afqvd84CQiAnxWWtyxsS6ThINBYg2Dc7mZBMEjE4mzs6e9WcY5+ePNuVFJJodQAoLo+SaWCy9rcV8cmjah3CI6iYu7oRU30kE5xxELRfamklY3k1NL19sSm7vPzP/ZdNZzKVDaY2sPZJBh9fv5ITrmG2+Vp4e1sPchVqTCQZJnVXi+/L4uuAJGly1+Pw8CprLbi8Om7tbT19/XRqJUk11JP9uHj9ulxhXbvJbI9qJvr5YkGXFG2IBT8tXczt+sfzDZCp3765f3t9tHEHGEDACma77+8o4oATKk+/PfW9YmHruRFjWoVSFsVsGu1YSKq6Oc37+n98unPZSRlY7vsKDqN+92X3yR9+PdXee3iJNKMStqdcZqoTJbUSi5JOkpfRlhSI0mSpEmCFKoU7FqSNOLAk54uGwCStMUCgLrVic62g7oDoFmmdI+P3S0pDe1xvDqb6XrZqbtzShWNoh9fv/XQHaDdM9OqrZi2M7M3UrB2vlkPS1IbdEBk7UiSoD6VlZ6aKWer4aH4f/AvKoHUTjuyAAAAAElFTkSuQmCC',
'options': [ 'options': [
@@ -73,4 +73,24 @@ config = [{
], ],
}, },
], ],
}, {
'name': 'torrent',
'groups': [
{
'tab': 'searcher',
'name': 'searcher',
'wizard': True,
'options': [
{
'name': 'minimum_seeders',
'advanced': True,
'label': 'Minimum seeders',
'description': 'Ignore torrents with seeders below this number',
'default': 1,
'type': 'int',
'unit': 'seeders'
},
],
},
],
}] }]
+28 -23
View File
@@ -129,7 +129,11 @@ class Searcher(SearcherBase):
# Try guessing via quality tags # Try guessing via quality tags
guess = fireEvent('quality.guess', [nzb.get('name')], single = True) guess = fireEvent('quality.guess', [nzb.get('name')], single = True)
return threed == guess.get('is_3d') if guess:
return threed == guess.get('is_3d')
# If no quality guess, assume not 3d
else:
return threed == False
def correctYear(self, haystack, year, year_range): def correctYear(self, haystack, year, year_range):
@@ -174,6 +178,25 @@ class Searcher(SearcherBase):
return False return False
def containsWords(self, rel_name, rel_words, conf, media):
# Make sure it has required words
words = splitString(self.conf('%s_words' % conf, section = 'searcher').lower())
try: words = removeDuplicate(words + splitString(media['category'][conf].lower()))
except: pass
req_match = 0
for req_set in words:
if len(req_set) >= 2 and (req_set[:1] + req_set[-1:]) == '//':
if re.search(req_set[1:-1], rel_name):
log.debug('Regex match: %s', req_set[1:-1])
req_match += 1
else:
req = splitString(req_set, '&')
req_match += len(list(set(rel_words) & set(req))) == len(req)
return words, req_match > 0
def correctWords(self, rel_name, media): def correctWords(self, rel_name, media):
media_title = fireEvent('searcher.get_search_title', media, single = True) media_title = fireEvent('searcher.get_search_title', media, single = True)
media_words = re.split('\W+', simplifyString(media_title)) media_words = re.split('\W+', simplifyString(media_title))
@@ -181,31 +204,13 @@ class Searcher(SearcherBase):
rel_name = simplifyString(rel_name) rel_name = simplifyString(rel_name)
rel_words = re.split('\W+', rel_name) rel_words = re.split('\W+', rel_name)
# Make sure it has required words required_words, contains_required = self.containsWords(rel_name, rel_words, 'required', media)
required_words = splitString(self.conf('required_words', section = 'searcher').lower()) if len(required_words) > 0 and not contains_required:
try: required_words = removeDuplicate(required_words + splitString(media['category']['required'].lower()))
except: pass
req_match = 0
for req_set in required_words:
req = splitString(req_set, '&')
req_match += len(list(set(rel_words) & set(req))) == len(req)
if len(required_words) > 0 and req_match == 0:
log.info2('Wrong: Required word missing: %s', rel_name) log.info2('Wrong: Required word missing: %s', rel_name)
return False return False
# Ignore releases ignored_words, contains_ignored = self.containsWords(rel_name, rel_words, 'ignored', media)
ignored_words = splitString(self.conf('ignored_words', section = 'searcher').lower()) if len(ignored_words) > 0 and contains_ignored:
try: ignored_words = removeDuplicate(ignored_words + splitString(media['category']['ignored'].lower()))
except: pass
ignored_match = 0
for ignored_set in ignored_words:
ignored = splitString(ignored_set, '&')
ignored_match += len(list(set(rel_words) & set(ignored))) == len(ignored)
if len(ignored_words) > 0 and ignored_match:
log.info2("Wrong: '%s' contains 'ignored words'", rel_name) log.info2("Wrong: '%s' contains 'ignored words'", rel_name)
return False return False
+14 -5
View File
@@ -1,4 +1,3 @@
import os
import traceback import traceback
import time import time
@@ -28,6 +27,10 @@ class MovieBase(MovieTypeBase):
addApiView('movie.add', self.addView, docs = { addApiView('movie.add', self.addView, docs = {
'desc': 'Add new movie to the wanted list', 'desc': 'Add new movie to the wanted list',
'return': {'type': 'object', 'example': """{
'success': True,
'movie': object
}"""},
'params': { 'params': {
'identifier': {'desc': 'IMDB id of the movie your want to add.'}, 'identifier': {'desc': 'IMDB id of the movie your want to add.'},
'profile_id': {'desc': 'ID of quality profile you want the add the movie in. If empty will use the default profile.'}, 'profile_id': {'desc': 'ID of quality profile you want the add the movie in. If empty will use the default profile.'},
@@ -151,8 +154,7 @@ class MovieBase(MovieTypeBase):
for release in fireEvent('release.for_media', m['_id'], single = True): for release in fireEvent('release.for_media', m['_id'], single = True):
if release.get('status') in ['downloaded', 'snatched', 'seeding', 'done']: if release.get('status') in ['downloaded', 'snatched', 'seeding', 'done']:
if params.get('ignore_previous', False): if params.get('ignore_previous', False):
release['status'] = 'ignored' fireEvent('release.update_status', release['_id'], status = 'ignored')
db.update(release)
else: else:
fireEvent('release.delete', release['_id'], single = True) fireEvent('release.delete', release['_id'], single = True)
@@ -180,6 +182,9 @@ class MovieBase(MovieTypeBase):
db.delete(rel) db.delete(rel)
movie_dict = fireEvent('media.get', m['_id'], single = True) movie_dict = fireEvent('media.get', m['_id'], single = True)
if not movie_dict:
log.debug('Failed adding media, can\'t find it anymore')
return False
if do_search and search_after: if do_search and search_after:
onComplete = self.createOnComplete(m['_id']) onComplete = self.createOnComplete(m['_id'])
@@ -269,6 +274,10 @@ class MovieBase(MovieTypeBase):
if self.shuttingDown(): if self.shuttingDown():
return return
lock_key = 'media.get.%s' % media_id if media_id else identifier
self.acquireLock(lock_key)
media = {}
try: try:
db = get_db() db = get_db()
@@ -317,11 +326,11 @@ class MovieBase(MovieTypeBase):
self.getPoster(media, image_urls) self.getPoster(media, image_urls)
db.update(media) db.update(media)
return media
except: except:
log.error('Failed update media: %s', traceback.format_exc()) log.error('Failed update media: %s', traceback.format_exc())
return {} self.releaseLock(lock_key)
return media
def updateReleaseDate(self, media_id): def updateReleaseDate(self, media_id):
""" """
@@ -115,8 +115,15 @@ MA.Release = new Class({
self.releases = null; self.releases = null;
if(self.options_container){ if(self.options_container){
self.options_container.destroy(); // Releases are currently displayed
self.options_container = null; if(self.options_container.isDisplayed()){
self.options_container.destroy();
self.createReleases();
}
else {
self.options_container.destroy();
self.options_container = null;
}
} }
}); });
@@ -131,10 +138,10 @@ MA.Release = new Class({
}, },
createReleases: function(){ createReleases: function(refresh){
var self = this; var self = this;
if(!self.options_container){ if(!self.options_container || refresh){
self.options_container = new Element('div.options').grab( self.options_container = new Element('div.options').grab(
self.release_container = new Element('div.releases.table') self.release_container = new Element('div.releases.table')
); );
@@ -54,13 +54,21 @@ var Movie = new Class({
// Reload when releases have updated // Reload when releases have updated
self.global_events['release.update_status'] = function(notification){ self.global_events['release.update_status'] = function(notification){
var data = notification.data; var data = notification.data;
if(data && self.data._id == data.movie_id){ if(data && self.data._id == data.media_id){
if(!self.data.releases) if(!self.data.releases)
self.data.releases = []; self.data.releases = [];
self.data.releases.push({'quality': data.quality, 'status': data.status}); var updated = false;
self.updateReleases(); self.data.releases.each(function(release){
if(release._id == data._id){
release['status'] = data.status;
updated = true;
}
});
if(updated)
self.updateReleases();
} }
}; };
@@ -159,7 +167,7 @@ var Movie = new Class({
} }
} }
}), }),
self.thumbnail = (self.data.files && self.data.files.image_poster) ? new Element('img', { self.thumbnail = (self.data.files && self.data.files.image_poster && self.data.files.image_poster.length > 0) ? new Element('img', {
'class': 'type_image poster', 'class': 'type_image poster',
'src': Api.createUrl('file.cache') + self.data.files.image_poster[0].split(Api.getOption('path_sep')).pop() 'src': Api.createUrl('file.cache') + self.data.files.image_poster[0].split(Api.getOption('path_sep')).pop()
}): null, }): null,
@@ -21,13 +21,6 @@ config = [{
'type': 'int', 'type': 'int',
'description': 'Maximum number of items displayed from each chart.', 'description': 'Maximum number of items displayed from each chart.',
}, },
{
'name': 'update_interval',
'default': 12,
'type': 'int',
'advanced': True,
'description': '(hours)',
},
{ {
'name': 'hide_wanted', 'name': 'hide_wanted',
'default': False, 'default': False,
+3 -3
View File
@@ -1,6 +1,5 @@
import time import time
from couchpotato import tryInt
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.api import addApiView from couchpotato.api import addApiView
from couchpotato.core.event import addEvent,fireEvent from couchpotato.core.event import addEvent,fireEvent
@@ -13,13 +12,14 @@ log = CPLog(__name__)
class Charts(Plugin): class Charts(Plugin):
update_in_progress = False update_in_progress = False
update_interval = 72 # hours
def __init__(self): def __init__(self):
addApiView('charts.view', self.automationView) addApiView('charts.view', self.automationView)
addEvent('app.load', self.setCrons) addEvent('app.load', self.setCrons)
def setCrons(self): def setCrons(self):
fireEvent('schedule.interval', 'charts.update_cache', self.updateViewCache, hours = self.conf('update_interval', default = 12)) fireEvent('schedule.interval', 'charts.update_cache', self.updateViewCache, hours = self.update_interval)
def automationView(self, force_update = False, **kwargs): def automationView(self, force_update = False, **kwargs):
@@ -52,7 +52,7 @@ class Charts(Plugin):
for chart in charts: for chart in charts:
chart['hide_wanted'] = self.conf('hide_wanted') chart['hide_wanted'] = self.conf('hide_wanted')
chart['hide_library'] = self.conf('hide_library') chart['hide_library'] = self.conf('hide_library')
self.setCache('charts_cached', charts, timeout = 7200 * tryInt(self.conf('update_interval', default = 12))) self.setCache('charts_cached', charts, timeout = self.update_interval * 3600)
except: except:
log.error('Failed refreshing charts') log.error('Failed refreshing charts')
@@ -264,3 +264,11 @@
height: 40px; height: 40px;
} }
@media all and (max-width: 480px) {
.toggle_menu h2 {
font-size: 16px;
text-align: center;
height: 30px;
}
}
@@ -2,6 +2,8 @@ var Charts = new Class({
Implements: [Options, Events], Implements: [Options, Events],
shown_once: false,
initialize: function(options){ initialize: function(options){
var self = this; var self = this;
self.setOptions(options); self.setOptions(options);
@@ -40,17 +42,13 @@ var Charts = new Class({
) )
); );
if( Cookie.read('suggestions_charts_menu_selected') === 'charts') if( Cookie.read('suggestions_charts_menu_selected') === 'charts'){
self.el.show(); self.show();
self.fireEvent.delay(0, self, 'created');
}
else else
self.el.hide(); self.el.hide();
self.api_request = Api.request('charts.view', {
'onComplete': self.fill.bind(self)
});
self.fireEvent.delay(0, self, 'created');
}, },
fill: function(json){ fill: function(json){
@@ -157,6 +155,24 @@ var Charts = new Class({
}, },
show: function(){
var self = this;
self.el.show();
if(!self.shown_once){
self.api_request = Api.request('charts.view', {
'onComplete': self.fill.bind(self)
});
self.shown_once = true;
}
},
hide: function(){
this.el.hide();
},
afterAdded: function(m){ afterAdded: function(m){
$(m).getElement('div.chart_number') $(m).getElement('div.chart_number')
@@ -1,3 +1,5 @@
import traceback
from bs4 import BeautifulSoup from bs4 import BeautifulSoup
from couchpotato import fireEvent from couchpotato import fireEvent
from couchpotato.core.helpers.rss import RSS from couchpotato.core.helpers.rss import RSS
@@ -5,6 +7,7 @@ from couchpotato.core.helpers.variable import tryInt
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.media.movie.providers.automation.base import Automation from couchpotato.core.media.movie.providers.automation.base import Automation
log = CPLog(__name__) log = CPLog(__name__)
autoload = 'Bluray' autoload = 'Bluray'
@@ -34,27 +37,49 @@ class Bluray(Automation, RSS):
try: try:
# Stop if the release year is before the minimal year # Stop if the release year is before the minimal year
page_year = soup.body.find_all('center')[3].table.tr.find_all('td', recursive = False)[3].h3.get_text().split(', ')[1] brk = False
if tryInt(page_year) < self.getMinimal('year'): h3s = soup.body.find_all('h3')
for h3 in h3s:
if h3.parent.name != 'a':
try:
page_year = tryInt(h3.get_text()[-4:])
if page_year > 0 and page_year < self.getMinimal('year'):
brk = True
except:
log.error('Failed determining page year: %s', traceback.format_exc())
brk = True
break
if brk:
break break
for table in soup.body.find_all('center')[3].table.tr.find_all('td', recursive = False)[3].find_all('table')[1:20]: for h3 in h3s:
name = table.h3.get_text().lower().split('blu-ray')[0].strip() try:
year = table.small.get_text().split('|')[1].strip() if h3.parent.name == 'a':
name = h3.get_text().lower().split('blu-ray')[0].strip()
if not name.find('/') == -1: # make sure it is not a double movie release if not name.find('/') == -1: # make sure it is not a double movie release
continue continue
if tryInt(year) < self.getMinimal('year'): if not h3.parent.parent.small: # ignore non-movie tables
continue continue
imdb = self.search(name, year) year = h3.parent.parent.small.get_text().split('|')[1].strip()
if imdb: if tryInt(year) < self.getMinimal('year'):
if self.isMinimalMovie(imdb): continue
movies.append(imdb['imdb'])
imdb = self.search(name, year)
if imdb:
if self.isMinimalMovie(imdb):
movies.append(imdb['imdb'])
except:
log.debug('Error parsing movie html: %s', traceback.format_exc())
break
except: except:
log.debug('Error loading page: %s', page) log.debug('Error loading page %s: %s', (page, traceback.format_exc()))
break break
self.conf('backlog', value = False) self.conf('backlog', value = False)
@@ -134,7 +159,7 @@ config = [{
{ {
'name': 'backlog', 'name': 'backlog',
'advanced': True, 'advanced': True,
'description': 'Parses the history until the minimum movie year is reached. (Will be disabled once it has completed)', 'description': ('Parses the history until the minimum movie year is reached. (Takes a while)', 'Will be disabled once it has completed'),
'default': False, 'default': False,
'type': 'bool', 'type': 'bool',
}, },
@@ -2,7 +2,7 @@ import base64
import time import time
from couchpotato.core.event import addEvent, fireEvent from couchpotato.core.event import addEvent, fireEvent
from couchpotato.core.helpers.encoding import tryUrlencode from couchpotato.core.helpers.encoding import tryUrlencode, ss
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.media.movie.providers.base import MovieProvider from couchpotato.core.media.movie.providers.base import MovieProvider
from couchpotato.environment import Env from couchpotato.environment import Env
@@ -66,7 +66,7 @@ class CouchPotatoApi(MovieProvider):
if not name: if not name:
return return
name_enc = base64.b64encode(name) name_enc = base64.b64encode(ss(name))
return self.getJsonData(self.urls['validate'] % name_enc, headers = self.getRequestHeaders()) return self.getJsonData(self.urls['validate'] % name_enc, headers = self.getRequestHeaders())
def isMovie(self, identifier = None): def isMovie(self, identifier = None):
@@ -4,6 +4,7 @@ from couchpotato import tryInt
from couchpotato.core.event import addEvent from couchpotato.core.event import addEvent
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.media.movie.providers.base import MovieProvider from couchpotato.core.media.movie.providers.base import MovieProvider
from requests import HTTPError
log = CPLog(__name__) log = CPLog(__name__)
@@ -23,22 +24,23 @@ class FanartTV(MovieProvider):
def __init__(self): def __init__(self):
addEvent('movie.info', self.getArt, priority = 1) addEvent('movie.info', self.getArt, priority = 1)
def getArt(self, identifier = None, **kwargs): def getArt(self, identifier = None, extended = True, **kwargs):
log.debug("Getting Extra Artwork from Fanart.tv...") if not identifier or not extended:
if not identifier:
return {} return {}
images = {} images = {}
try: try:
url = self.urls['api'] % identifier url = self.urls['api'] % identifier
fanart_data = self.getJsonData(url) fanart_data = self.getJsonData(url, show_error = False)
if fanart_data: if fanart_data:
log.debug('Found images for %s', fanart_data.get('name')) log.debug('Found images for %s', fanart_data.get('name'))
images = self._parseMovie(fanart_data) images = self._parseMovie(fanart_data)
except HTTPError as e:
log.debug('Failed getting extra art for %s: %s',
(identifier, e))
except: except:
log.error('Failed getting extra art for %s: %s', log.error('Failed getting extra art for %s: %s',
(identifier, traceback.format_exc())) (identifier, traceback.format_exc()))
@@ -1,11 +1,10 @@
import traceback import traceback
from couchpotato.core.event import addEvent from couchpotato.core.event import addEvent, fireEvent
from couchpotato.core.helpers.encoding import simplifyString, toUnicode, ss from couchpotato.core.helpers.encoding import toUnicode, ss, tryUrlencode
from couchpotato.core.helpers.variable import tryInt from couchpotato.core.helpers.variable import tryInt
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.media.movie.providers.base import MovieProvider from couchpotato.core.media.movie.providers.base import MovieProvider
import tmdb3
log = CPLog(__name__) log = CPLog(__name__)
@@ -13,54 +12,66 @@ autoload = 'TheMovieDb'
class TheMovieDb(MovieProvider): class TheMovieDb(MovieProvider):
MAX_EXTRATHUMBS = 4
http_time_between_calls = .35
configuration = {
'images': {
'secure_base_url': 'https://image.tmdb.org/t/p/',
},
}
def __init__(self): def __init__(self):
addEvent('info.search', self.search, priority = 3)
addEvent('movie.search', self.search, priority = 3)
addEvent('movie.info', self.getInfo, priority = 3) addEvent('movie.info', self.getInfo, priority = 3)
addEvent('movie.info_by_tmdb', self.getInfo) addEvent('movie.info_by_tmdb', self.getInfo)
addEvent('app.load', self.config)
# Configure TMDB settings def config(self):
tmdb3.set_key(self.conf('api_key')) configuration = self.request('configuration')
tmdb3.set_cache('null') if configuration:
self.configuration = configuration
def search(self, q, limit = 12): def search(self, q, limit = 3):
""" Find movie by name """ """ Find movie by name """
if self.isDisabled(): if self.isDisabled():
return False return False
search_string = simplifyString(q) log.debug('Searching for movie: %s', q)
cache_key = 'tmdb.cache.%s.%s' % (search_string, limit)
results = self.getCache(cache_key)
if not results: raw = None
log.debug('Searching for movie: %s', q) try:
name_year = fireEvent('scanner.name_year', q, single = True)
raw = self.request('search/movie', {
'query': name_year.get('name', q),
'year': name_year.get('year'),
'search_type': 'ngram' if limit > 1 else 'phrase'
}, return_key = 'results')
except:
log.error('Failed searching TMDB for "%s": %s', (q, traceback.format_exc()))
raw = None results = []
if raw:
try: try:
raw = tmdb3.searchMovie(search_string) nr = 0
except:
log.error('Failed searching TMDB for "%s": %s', (search_string, traceback.format_exc()))
results = [] for movie in raw:
if raw: parsed_movie = self.parseMovie(movie, extended = False)
try: if parsed_movie:
nr = 0 results.append(parsed_movie)
for movie in raw: nr += 1
results.append(self.parseMovie(movie, extended = False)) if nr == limit:
break
nr += 1 log.info('Found: %s', [result['titles'][0] + ' (' + str(result.get('year', 0)) + ')' for result in results])
if nr == limit:
break
log.info('Found: %s', [result['titles'][0] + ' (' + str(result.get('year', 0)) + ')' for result in results]) return results
except SyntaxError as e:
self.setCache(cache_key, results) log.error('Failed to parse XML response: %s', e)
return results return False
except SyntaxError as e:
log.error('Failed to parse XML response: %s', e)
return False
return results return results
@@ -69,101 +80,91 @@ class TheMovieDb(MovieProvider):
if not identifier: if not identifier:
return {} return {}
cache_key = 'tmdb.cache.%s%s' % (identifier, '.ex' if extended else '') result = self.parseMovie({
result = self.getCache(cache_key) 'id': identifier
}, extended = extended)
if not result: return result or {}
try:
log.debug('Getting info: %s', cache_key)
# noinspection PyArgumentList
movie = tmdb3.Movie(identifier)
try: exists = movie.title is not None
except: exists = False
if exists:
result = self.parseMovie(movie, extended = extended)
self.setCache(cache_key, result)
else:
result = {}
except:
log.error('Failed getting info for %s: %s', (identifier, traceback.format_exc()))
return result
def parseMovie(self, movie, extended = True): def parseMovie(self, movie, extended = True):
cache_key = 'tmdb.cache.%s%s' % (movie.id, '.ex' if extended else '') # Do request, append other items
movie_data = self.getCache(cache_key) movie = self.request('movie/%s' % movie.get('id'), {
'append_to_response': 'alternative_titles' + (',images,casts' if extended else '')
})
if not movie:
return
if not movie_data: # Images
poster = self.getImage(movie, type = 'poster', size = 'w154')
poster_original = self.getImage(movie, type = 'poster', size = 'original')
backdrop_original = self.getImage(movie, type = 'backdrop', size = 'original')
extra_thumbs = self.getMultImages(movie, type = 'backdrops', size = 'original') if extended else []
# Images images = {
poster = self.getImage(movie, type = 'poster', size = 'w154') 'poster': [poster] if poster else [],
poster_original = self.getImage(movie, type = 'poster', size = 'original') #'backdrop': [backdrop] if backdrop else [],
backdrop_original = self.getImage(movie, type = 'backdrop', size = 'original') 'poster_original': [poster_original] if poster_original else [],
extra_thumbs = self.getMultImages(movie, type = 'backdrops', size = 'original', n = self.MAX_EXTRATHUMBS, skipfirst = True) 'backdrop_original': [backdrop_original] if backdrop_original else [],
'actors': {},
'extra_thumbs': extra_thumbs
}
images = { # Genres
'poster': [poster] if poster else [], try:
#'backdrop': [backdrop] if backdrop else [], genres = [genre.get('name') for genre in movie.get('genres', [])]
'poster_original': [poster_original] if poster_original else [], except:
'backdrop_original': [backdrop_original] if backdrop_original else [], genres = []
'actors': {},
'extra_thumbs': extra_thumbs
}
# Genres # 1900 is the same as None
try: year = str(movie.get('release_date') or '')[:4]
genres = [genre.name for genre in movie.genres] if not movie.get('release_date') or year == '1900' or year.lower() == 'none':
except: year = None
genres = []
# 1900 is the same as None # Gather actors data
year = str(movie.releasedate or '')[:4] actors = {}
if not movie.releasedate or year == '1900' or year.lower() == 'none': if extended:
year = None
# Gather actors data # Full data
actors = {} cast = movie.get('casts', {}).get('cast', [])
if extended:
for cast_item in movie.cast:
try:
actors[toUnicode(cast_item.name)] = toUnicode(cast_item.character)
images['actors'][toUnicode(cast_item.name)] = self.getImage(cast_item, type = 'profile', size = 'original')
except:
log.debug('Error getting cast info for %s: %s', (cast_item, traceback.format_exc()))
movie_data = { for cast_item in cast:
'type': 'movie', try:
'via_tmdb': True, actors[toUnicode(cast_item.get('name'))] = toUnicode(cast_item.get('character'))
'tmdb_id': movie.id, images['actors'][toUnicode(cast_item.get('name'))] = self.getImage(cast_item, type = 'profile', size = 'original')
'titles': [toUnicode(movie.title)], except:
'original_title': movie.originaltitle, log.debug('Error getting cast info for %s: %s', (cast_item, traceback.format_exc()))
'images': images,
'imdb': movie.imdb,
'runtime': movie.runtime,
'released': str(movie.releasedate),
'year': tryInt(year, None),
'plot': movie.overview,
'genres': genres,
'collection': getattr(movie.collection, 'name', None),
'actor_roles': actors
}
movie_data = dict((k, v) for k, v in movie_data.items() if v) movie_data = {
'type': 'movie',
'via_tmdb': True,
'tmdb_id': movie.get('id'),
'titles': [toUnicode(movie.get('title'))],
'original_title': movie.get('original_title'),
'images': images,
'imdb': movie.get('imdb_id'),
'runtime': movie.get('runtime'),
'released': str(movie.get('release_date')),
'year': tryInt(year, None),
'plot': movie.get('overview'),
'genres': genres,
'collection': getattr(movie.get('belongs_to_collection'), 'name', None),
'actor_roles': actors
}
# Add alternative names movie_data = dict((k, v) for k, v in movie_data.items() if v)
if movie_data['original_title'] and movie_data['original_title'] not in movie_data['titles']:
movie_data['titles'].append(movie_data['original_title'])
if extended: # Add alternative names
for alt in movie.alternate_titles: if movie_data['original_title'] and movie_data['original_title'] not in movie_data['titles']:
alt_name = alt.title movie_data['titles'].append(movie_data['original_title'])
if alt_name and alt_name not in movie_data['titles'] and alt_name.lower() != 'none' and alt_name is not None:
movie_data['titles'].append(alt_name)
# Cache movie parsed # Add alternative titles
self.setCache(cache_key, movie_data) alternate_titles = movie.get('alternative_titles', {}).get('titles', [])
for alt in alternate_titles:
alt_name = alt.get('title')
if alt_name and alt_name not in movie_data['titles'] and alt_name.lower() != 'none' and alt_name is not None:
movie_data['titles'].append(alt_name)
return movie_data return movie_data
@@ -171,36 +172,41 @@ class TheMovieDb(MovieProvider):
image_url = '' image_url = ''
try: try:
image_url = getattr(movie, type).geturl(size = size) path = movie.get('%s_path' % type)
image_url = '%s%s%s' % (self.configuration['images']['secure_base_url'], size, path)
except: except:
log.debug('Failed getting %s.%s for "%s"', (type, size, ss(str(movie)))) log.debug('Failed getting %s.%s for "%s"', (type, size, ss(str(movie))))
return image_url return image_url
def getMultImages(self, movie, type = 'backdrops', size = 'original', n = -1, skipfirst = False): def getMultImages(self, movie, type = 'backdrops', size = 'original'):
"""
If n < 0, return all images. Otherwise return n images.
If n > len(getattr(movie, type)), then return all images.
If skipfirst is True, then it will skip getattr(movie, type)[0]. This
is because backdrops[0] is typically backdrop.
"""
image_urls = [] image_urls = []
try: try:
images = getattr(movie, type) for image in movie.get('images', {}).get(type, [])[1:5]:
if n < 0 or n > len(images): image_urls.append(self.getImage(image, 'file', size))
num_images = len(images)
else:
num_images = n
for i in range(int(skipfirst), num_images + int(skipfirst)):
image_urls.append(images[i].geturl(size = size))
except: except:
log.debug('Failed getting %i %s.%s for "%s"', (n, type, size, ss(str(movie)))) log.debug('Failed getting %s.%s for "%s"', (type, size, ss(str(movie))))
return image_urls return image_urls
def request(self, call = '', params = {}, return_key = None):
params = dict((k, v) for k, v in params.items() if v)
params = tryUrlencode(params)
try:
url = 'http://api.themoviedb.org/3/%s?api_key=%s%s' % (call, self.conf('api_key'), '&%s' % params if params else '')
data = self.getJsonData(url, show_error = False)
except:
log.debug('Movie not found: %s, %s', (call, params))
data = None
if data and return_key and return_key in data:
data = data.get(return_key)
return data
def isDisabled(self): def isDisabled(self):
if self.conf('api_key') == '': if self.conf('api_key') == '':
log.error('No API key provided.') log.error('No API key provided.')
@@ -1,30 +0,0 @@
from couchpotato.core.helpers.encoding import tryUrlencode
from couchpotato.core.logger import CPLog
from couchpotato.core.event import fireEvent
from couchpotato.core.media._base.providers.nzb.nzbindex import Base
from couchpotato.core.media.movie.providers.base import MovieProvider
from couchpotato.environment import Env
log = CPLog(__name__)
autoload = 'NzbIndex'
class NzbIndex(MovieProvider, Base):
def buildUrl(self, media, quality):
title = fireEvent('library.query', media, include_year = False, single = True)
year = media['info']['year']
query = tryUrlencode({
'q': '"%s %s" | "%s (%s)"' % (title, year, title, year),
'age': Env.setting('retention', 'nzb'),
'sort': 'agedesc',
'minsize': quality.get('size_min'),
'maxsize': quality.get('size_max'),
'rating': 1,
'max': 250,
'more': 1,
'complete': 1,
})
return query
@@ -13,7 +13,7 @@ class IPTorrents(MovieProvider, Base):
([87], ['3d']), ([87], ['3d']),
([48], ['720p', '1080p', 'bd50']), ([48], ['720p', '1080p', 'bd50']),
([72], ['cam', 'ts', 'tc', 'r5', 'scr']), ([72], ['cam', 'ts', 'tc', 'r5', 'scr']),
([7], ['dvdrip', 'brrip']), ([7,48], ['dvdrip', 'brrip']),
([6], ['dvdr']), ([6], ['dvdr']),
] ]
@@ -13,7 +13,7 @@ class PassThePopcorn(MovieProvider, Base):
'bd50': {'media': 'Blu-ray', 'format': 'BD50'}, 'bd50': {'media': 'Blu-ray', 'format': 'BD50'},
'1080p': {'resolution': '1080p'}, '1080p': {'resolution': '1080p'},
'720p': {'resolution': '720p'}, '720p': {'resolution': '720p'},
'brrip': {'media': 'Blu-ray'}, 'brrip': {'resolution': 'anyhd'},
'dvdr': {'resolution': 'anysd'}, 'dvdr': {'resolution': 'anysd'},
'dvdrip': {'media': 'DVD'}, 'dvdrip': {'media': 'DVD'},
'scr': {'media': 'DVD-Screener'}, 'scr': {'media': 'DVD-Screener'},
@@ -27,7 +27,7 @@ class PassThePopcorn(MovieProvider, Base):
'bd50': {'Codec': ['BD50']}, 'bd50': {'Codec': ['BD50']},
'1080p': {'Resolution': ['1080p']}, '1080p': {'Resolution': ['1080p']},
'720p': {'Resolution': ['720p']}, '720p': {'Resolution': ['720p']},
'brrip': {'Source': ['Blu-ray'], 'Quality': ['High Definition'], 'Container': ['!ISO']}, 'brrip': {'Quality': ['High Definition'], 'Container': ['!ISO']},
'dvdr': {'Codec': ['DVD5', 'DVD9']}, 'dvdr': {'Codec': ['DVD5', 'DVD9']},
'dvdrip': {'Source': ['DVD'], 'Codec': ['!DVD5', '!DVD9']}, 'dvdrip': {'Source': ['DVD'], 'Codec': ['!DVD5', '!DVD9']},
'scr': {'Source': ['DVD-Screener']}, 'scr': {'Source': ['DVD-Screener']},
@@ -1,27 +0,0 @@
from couchpotato.core.helpers.encoding import tryUrlencode
from couchpotato.core.logger import CPLog
from couchpotato.core.media._base.providers.torrent.torrentleech import Base
from couchpotato.core.media.movie.providers.base import MovieProvider
log = CPLog(__name__)
autoload = 'TorrentLeech'
class TorrentLeech(MovieProvider, Base):
cat_ids = [
([13], ['720p', '1080p', 'bd50']),
([8], ['cam']),
([9], ['ts', 'tc']),
([10], ['r5', 'scr']),
([11], ['dvdrip']),
([14], ['brrip']),
([12], ['dvdr']),
]
def buildUrl(self, title, media, quality):
return (
tryUrlencode(title.replace(':', '')),
self.getCatId(quality)[0]
)
@@ -3,7 +3,7 @@ import re
from bs4 import SoupStrainer, BeautifulSoup from bs4 import SoupStrainer, BeautifulSoup
from couchpotato.core.helpers.encoding import tryUrlencode from couchpotato.core.helpers.encoding import tryUrlencode
from couchpotato.core.helpers.variable import mergeDicts, getTitle from couchpotato.core.helpers.variable import mergeDicts, getTitle, getIdentifier
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.media.movie.providers.trailer.base import TrailerProvider from couchpotato.core.media.movie.providers.trailer.base import TrailerProvider
from requests import HTTPError from requests import HTTPError
@@ -29,7 +29,7 @@ class HDTrailers(TrailerProvider):
url = self.urls['api'] % self.movieUrlName(movie_name) url = self.urls['api'] % self.movieUrlName(movie_name)
try: try:
data = self.getCache('hdtrailers.%s' % group['identifier'], url, show_error = False) data = self.getCache('hdtrailers.%s' % getIdentifier(group), url, show_error = False)
except HTTPError: except HTTPError:
log.debug('No page found for: %s', movie_name) log.debug('No page found for: %s', movie_name)
data = None data = None
@@ -59,7 +59,7 @@ class HDTrailers(TrailerProvider):
url = "%s?%s" % (self.urls['backup'], tryUrlencode({'s':movie_name})) url = "%s?%s" % (self.urls['backup'], tryUrlencode({'s':movie_name}))
try: try:
data = self.getCache('hdtrailers.alt.%s' % group['identifier'], url, show_error = False) data = self.getCache('hdtrailers.alt.%s' % getIdentifier(group), url, show_error = False)
except HTTPError: except HTTPError:
log.debug('No alternative page found for: %s', movie_name) log.debug('No alternative page found for: %s', movie_name)
data = None data = None
@@ -68,7 +68,7 @@ class HDTrailers(TrailerProvider):
return results return results
try: try:
html = BeautifulSoup(data, 'html.parser', parse_only = self.only_tables_tags) html = BeautifulSoup(data, parse_only = self.only_tables_tags)
result_table = html.find_all('h2', text = re.compile(movie_name)) result_table = html.find_all('h2', text = re.compile(movie_name))
for h2 in result_table: for h2 in result_table:
@@ -90,7 +90,7 @@ class HDTrailers(TrailerProvider):
results = {'480p':[], '720p':[], '1080p':[]} results = {'480p':[], '720p':[], '1080p':[]}
try: try:
html = BeautifulSoup(data, 'html.parser', parse_only = self.only_tables_tags) html = BeautifulSoup(data, parse_only = self.only_tables_tags)
result_table = html.find('table', attrs = {'class':'bottomTable'}) result_table = html.find('table', attrs = {'class':'bottomTable'})
for tr in result_table.find_all('tr'): for tr in result_table.find_all('tr'):
@@ -25,6 +25,6 @@ class Filmstarts(UserscriptBase):
name = html.find("meta", {"property":"og:title"})['content'] name = html.find("meta", {"property":"og:title"})['content']
# Year of production is not available in the meta data, so get it from the table # Year of production is not available in the meta data, so get it from the table
year = table.find("tr", text="Produktionsjahr").parent.parent.parent.td.text year = table.find(text="Produktionsjahr").parent.parent.next_sibling.text
return self.search(name, year) return self.search(name, year)
+28 -26
View File
@@ -74,7 +74,7 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
self.in_progress = True self.in_progress = True
fireEvent('notify.frontend', type = 'movie.searcher.started', data = True, message = 'Full search started') fireEvent('notify.frontend', type = 'movie.searcher.started', data = True, message = 'Full search started')
medias = [x['_id'] for x in fireEvent('media.with_status', 'active', 'movie', single = True)] medias = [x['_id'] for x in fireEvent('media.with_status', 'active', types = 'movie', with_doc = False, single = True)]
random.shuffle(medias) random.shuffle(medias)
total = len(medias) total = len(medias)
@@ -89,6 +89,7 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
for media_id in medias: for media_id in medias:
media = fireEvent('media.get', media_id, single = True) media = fireEvent('media.get', media_id, single = True)
if not media: continue
try: try:
self.single(media, search_protocols, manual = manual) self.single(media, search_protocols, manual = manual)
@@ -140,17 +141,17 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
previous_releases = movie.get('releases', []) previous_releases = movie.get('releases', [])
too_early_to_search = [] too_early_to_search = []
outside_eta_results = 0 outside_eta_results = 0
alway_search = self.conf('always_search') always_search = self.conf('always_search')
ignore_eta = manual ignore_eta = manual
total_result_count = 0 total_result_count = 0
fireEvent('notify.frontend', type = 'movie.searcher.started', data = {'_id': movie['_id']}, message = 'Searching for "%s"' % default_title) fireEvent('notify.frontend', type = 'movie.searcher.started', data = {'_id': movie['_id']}, message = 'Searching for "%s"' % default_title)
# Ignore eta once every 7 days # Ignore eta once every 7 days
if not alway_search: if not always_search:
prop_name = 'last_ignored_eta.%s' % movie['_id'] prop_name = 'last_ignored_eta.%s' % movie['_id']
last_ignored_eta = float(Env.prop(prop_name, default = 0)) last_ignored_eta = float(Env.prop(prop_name, default = 0))
if last_ignored_eta > time.time() - 604800: if last_ignored_eta < time.time() - 604800:
ignore_eta = True ignore_eta = True
Env.prop(prop_name, value = time.time()) Env.prop(prop_name, value = time.time())
@@ -165,11 +166,12 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
'quality': q_identifier, 'quality': q_identifier,
'finish': profile['finish'][index], 'finish': profile['finish'][index],
'wait_for': tryInt(profile['wait_for'][index]), 'wait_for': tryInt(profile['wait_for'][index]),
'3d': profile['3d'][index] if profile.get('3d') else False '3d': profile['3d'][index] if profile.get('3d') else False,
'minimum_score': profile.get('minimum_score', 1),
} }
could_not_be_released = not self.couldBeReleased(q_identifier in pre_releases, release_dates, movie['info']['year']) could_not_be_released = not self.couldBeReleased(q_identifier in pre_releases, release_dates, movie['info']['year'])
if not alway_search and could_not_be_released: if not always_search and could_not_be_released:
too_early_to_search.append(q_identifier) too_early_to_search.append(q_identifier)
# Skip release, if ETA isn't ignored # Skip release, if ETA isn't ignored
@@ -195,19 +197,12 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
break break
quality = fireEvent('quality.single', identifier = q_identifier, single = True) quality = fireEvent('quality.single', identifier = q_identifier, single = True)
log.info('Search for %s in %s%s', (default_title, quality['label'], ' ignoring ETA' if alway_search or ignore_eta else '')) log.info('Search for %s in %s%s', (default_title, quality['label'], ' ignoring ETA' if always_search or ignore_eta else ''))
# Extend quality with profile customs # Extend quality with profile customs
quality['custom'] = quality_custom quality['custom'] = quality_custom
results = fireEvent('searcher.search', search_protocols, movie, quality, single = True) or [] results = fireEvent('searcher.search', search_protocols, movie, quality, single = True) or []
results_count = len(results)
total_result_count += results_count
if results_count == 0:
log.debug('Nothing found for %s in %s', (default_title, quality['label']))
# Keep track of releases found outside ETA window
outside_eta_results += results_count if could_not_be_released else 0
# Check if movie isn't deleted while searching # Check if movie isn't deleted while searching
if not fireEvent('media.get', movie.get('_id'), single = True): if not fireEvent('media.get', movie.get('_id'), single = True):
@@ -215,14 +210,20 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
# Add them to this movie releases list # Add them to this movie releases list
found_releases += fireEvent('release.create_from_search', results, movie, quality, single = True) found_releases += fireEvent('release.create_from_search', results, movie, quality, single = True)
results_count = len(found_releases)
total_result_count += results_count
if results_count == 0:
log.debug('Nothing found for %s in %s', (default_title, quality['label']))
# Keep track of releases found outside ETA window
outside_eta_results += results_count if could_not_be_released else 0
# Don't trigger download, but notify user of available releases # Don't trigger download, but notify user of available releases
if could_not_be_released: if could_not_be_released and results_count > 0:
if results_count > 0: log.debug('Found %s releases for "%s", but ETA isn\'t correct yet.', (results_count, default_title))
log.debug('Found %s releases for "%s", but ETA isn\'t correct yet.', (results_count, default_title))
# Try find a valid result and download it # Try find a valid result and download it
if (force_download or not could_not_be_released or alway_search) and fireEvent('release.try_download_result', results, movie, quality_custom, single = True): if (force_download or not could_not_be_released or always_search) and fireEvent('release.try_download_result', results, movie, quality_custom, single = True):
ret = True ret = True
# Remove releases that aren't found anymore # Remove releases that aren't found anymore
@@ -277,7 +278,7 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
# Contains lower quality string # Contains lower quality string
contains_other = fireEvent('searcher.contains_other_quality', nzb, movie_year = media['info']['year'], preferred_quality = preferred_quality, single = True) contains_other = fireEvent('searcher.contains_other_quality', nzb, movie_year = media['info']['year'], preferred_quality = preferred_quality, single = True)
if contains_other != False: if contains_other and isinstance(contains_other, dict):
log.info2('Wrong: %s, looking for %s, found %s', (nzb['name'], quality['label'], [x for x in contains_other] if contains_other else 'no quality')) log.info2('Wrong: %s, looking for %s, found %s', (nzb['name'], quality['label'], [x for x in contains_other] if contains_other else 'no quality'))
return False return False
@@ -381,16 +382,17 @@ class MovieSearcher(SearcherBase, MovieTypeBase):
def tryNextRelease(self, media_id, manual = False, force_download = False): def tryNextRelease(self, media_id, manual = False, force_download = False):
try: try:
db = get_db()
rels = fireEvent('media.with_status', ['snatched', 'done'], single = True) rels = fireEvent('release.for_media', media_id, single = True)
for rel in rels: for rel in rels:
rel['status'] = 'ignored' if rel.get('status') in ['snatched', 'done']:
db.update(rel) fireEvent('release.update_status', rel.get('_id'), status = 'ignored')
movie_dict = fireEvent('media.get', media_id, single = True) media = fireEvent('media.get', media_id, single = True)
log.info('Trying next release for: %s', getTitle(movie_dict)) if media:
self.single(movie_dict, manual = manual, force_download = force_download) log.info('Trying next release for: %s', getTitle(media))
self.single(media, manual = manual, force_download = force_download)
return True return True
@@ -27,7 +27,7 @@ class Suggestion(Plugin):
else: else:
if not movies or len(movies) == 0: if not movies or len(movies) == 0:
active_movies = fireEvent('media.with_status', ['active', 'done'], 'movie', single = True) active_movies = fireEvent('media.with_status', ['active', 'done'], types = 'movie', single = True)
movies = [getIdentifier(x) for x in active_movies] movies = [getIdentifier(x) for x in active_movies]
if not ignored or len(ignored) == 0: if not ignored or len(ignored) == 0:
@@ -2,6 +2,8 @@ var SuggestList = new Class({
Implements: [Options, Events], Implements: [Options, Events],
shown_once: false,
initialize: function(options){ initialize: function(options){
var self = this; var self = this;
self.setOptions(options); self.setOptions(options);
@@ -44,12 +46,13 @@ var SuggestList = new Class({
} }
}); });
var cookie_menu_select = Cookie.read('suggestions_charts_menu_selected'); var cookie_menu_select = Cookie.read('suggestions_charts_menu_selected') || 'suggestions';
if( cookie_menu_select === 'suggestions' || cookie_menu_select === null ) self.el.show(); else self.el.hide(); if( cookie_menu_select === 'suggestions')
self.show();
else
self.hide();
self.api_request = Api.request('suggestion.view', { self.fireEvent('created');
'onComplete': self.fill.bind(self)
});
}, },
@@ -145,6 +148,24 @@ var SuggestList = new Class({
}, },
show: function(){
var self = this;
self.el.show();
if(!self.shown_once){
self.api_request = Api.request('suggestion.view', {
'onComplete': self.fill.bind(self)
});
self.shown_once = true;
}
},
hide: function(){
this.el.hide();
},
toElement: function(){ toElement: function(){
return this.el; return this.el;
} }
+19 -6
View File
@@ -3,6 +3,7 @@ import threading
import time import time
import traceback import traceback
import uuid import uuid
from CodernityDB.database import RecordDeleted
from couchpotato import get_db from couchpotato import get_db
from couchpotato.api import addApiView, addNonBlockApiView from couchpotato.api import addApiView, addNonBlockApiView
@@ -66,7 +67,9 @@ class CoreNotifier(Notification):
fireEvent('schedule.interval', 'core.clean_messages', self.cleanMessages, seconds = 15, single = True) fireEvent('schedule.interval', 'core.clean_messages', self.cleanMessages, seconds = 15, single = True)
addEvent('app.load', self.clean) addEvent('app.load', self.clean)
addEvent('app.load', self.checkMessages)
if not Env.get('dev'):
addEvent('app.load', self.checkMessages)
self.messages = [] self.messages = []
self.listeners = [] self.listeners = []
@@ -153,9 +156,14 @@ class CoreNotifier(Notification):
n = { n = {
'_t': 'notification', '_t': 'notification',
'time': int(time.time()), 'time': int(time.time()),
'message': toUnicode(message), 'message': toUnicode(message)
'data': data
} }
if data.get('sticky'):
n['sticky'] = True
if data.get('important'):
n['important'] = True
db.insert(n) db.insert(n)
self.frontend(type = listener, data = n) self.frontend(type = listener, data = n)
@@ -263,11 +271,16 @@ class CoreNotifier(Notification):
if init: if init:
db = get_db() db = get_db()
notifications = db.all('notification', with_doc = True) notifications = db.all('notification')
for n in notifications: for n in notifications:
if n['doc'].get('time') > (time.time() - 604800):
messages.append(n['doc']) try:
doc = db.get('id', n.get('_id'))
if doc.get('time') > (time.time() - 604800):
messages.append(doc)
except RecordDeleted:
pass
return { return {
'success': True, 'success': True,
@@ -50,7 +50,7 @@ var NotificationBase = new Class({
, 'top'); , 'top');
self.notifications.include(result); self.notifications.include(result);
if((result.data.important !== undefined || result.data.sticky !== undefined) && !result.read){ if((result.important !== undefined || result.sticky !== undefined) && !result.read){
var sticky = true; var sticky = true;
App.trigger('message', [result.message, sticky, result]) App.trigger('message', [result.message, sticky, result])
} }
@@ -72,7 +72,7 @@ var NotificationBase = new Class({
if(!force_ids) { if(!force_ids) {
var rn = self.notifications.filter(function(n){ var rn = self.notifications.filter(function(n){
return !n.read && n.data.important === undefined return !n.read && n.important === undefined
}); });
var ids = []; var ids = [];
+1 -1
View File
@@ -42,7 +42,7 @@ class Email(Notification):
# Open the SMTP connection, via SSL if requested # Open the SMTP connection, via SSL if requested
log.debug("Connecting to host %s on port %s" % (smtp_server, smtp_port)) log.debug("Connecting to host %s on port %s" % (smtp_server, smtp_port))
log.debug("SMTP over SSL %s", ("enabled" if ssl == 1 else "disabled")) log.debug("SMTP over SSL %s", ("enabled" if ssl == 1 else "disabled"))
mailserver = smtplib.SMTP_SSL(smtp_server) if ssl == 1 else smtplib.SMTP(smtp_server) mailserver = smtplib.SMTP_SSL(smtp_server, smtp_port) if ssl == 1 else smtplib.SMTP(smtp_server, smtp_port)
if starttls: if starttls:
log.debug("Using StartTLS to initiate the connection with the SMTP server") log.debug("Using StartTLS to initiate the connection with the SMTP server")
+4 -4
View File
@@ -34,9 +34,9 @@ class Growl(Notification):
self.growl = notifier.GrowlNotifier( self.growl = notifier.GrowlNotifier(
applicationName = Env.get('appname'), applicationName = Env.get('appname'),
notifications = ["Updates"], notifications = ['Updates'],
defaultNotifications = ["Updates"], defaultNotifications = ['Updates'],
applicationIcon = '%s/static/images/couch.png' % fireEvent('app.api_url', single = True), applicationIcon = self.getNotificationImage('medium'),
hostname = hostname if hostname else 'localhost', hostname = hostname if hostname else 'localhost',
password = password if password else None, password = password if password else None,
port = port if port else 23053 port = port if port else 23053
@@ -56,7 +56,7 @@ class Growl(Notification):
try: try:
self.growl.notify( self.growl.notify(
noteType = "Updates", noteType = 'Updates',
title = self.default_title, title = self.default_title,
description = message, description = message,
sticky = False, sticky = False,
@@ -1,68 +0,0 @@
from couchpotato.core.helpers.variable import splitString
from couchpotato.core.logger import CPLog
from couchpotato.core.notifications.base import Notification
from pynmwp import PyNMWP
import six
log = CPLog(__name__)
autoload = 'NotifyMyWP'
class NotifyMyWP(Notification):
def notify(self, message = '', data = None, listener = None):
if not data: data = {}
keys = splitString(self.conf('api_key'))
p = PyNMWP(keys, self.conf('dev_key'))
response = p.push(application = self.default_title, event = message, description = message, priority = self.conf('priority'), batch_mode = len(keys) > 1)
for key in keys:
if not response[key]['Code'] == six.u('200'):
log.error('Could not send notification to NotifyMyWindowsPhone (%s). %s', (key, response[key]['message']))
return False
return response
config = [{
'name': 'notifymywp',
'groups': [
{
'tab': 'notifications',
'list': 'notification_providers',
'name': 'notifymywp',
'label': 'Windows Phone',
'options': [
{
'name': 'enabled',
'default': 0,
'type': 'enabler',
},
{
'name': 'api_key',
'description': 'Multiple keys seperated by a comma. Maximum of 5.'
},
{
'name': 'dev_key',
'advanced': True,
},
{
'name': 'priority',
'default': 0,
'type': 'dropdown',
'values': [('Very Low', -2), ('Moderate', -1), ('Normal', 0), ('High', 1), ('Emergency', 2)],
},
{
'name': 'on_snatch',
'default': 0,
'type': 'bool',
'advanced': True,
'description': 'Also send message when movie is snatched.',
},
],
}
],
}]
+4 -3
View File
@@ -1,4 +1,4 @@
from couchpotato.core.helpers.variable import getTitle from couchpotato.core.helpers.variable import getTitle, getIdentifier
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.notifications.base import Notification from couchpotato.core.notifications.base import Notification
@@ -16,7 +16,8 @@ class Trakt(Notification):
'test': 'account/test/%s', 'test': 'account/test/%s',
} }
listen_to = ['movie.downloaded'] listen_to = ['movie.snatched']
enabled_option = 'notification_enabled'
def notify(self, message = '', data = None, listener = None): def notify(self, message = '', data = None, listener = None):
if not data: data = {} if not data: data = {}
@@ -38,7 +39,7 @@ class Trakt(Notification):
'username': self.conf('automation_username'), 'username': self.conf('automation_username'),
'password': self.conf('automation_password'), 'password': self.conf('automation_password'),
'movies': [{ 'movies': [{
'imdb_id': data['identifier'], 'imdb_id': getIdentifier(data),
'title': getTitle(data), 'title': getTitle(data),
'year': data['info']['year'] 'year': data['info']['year']
}] if data else [] }] if data else []
+4 -4
View File
@@ -7,8 +7,8 @@ import urllib
from couchpotato.core.helpers.variable import splitString, getTitle from couchpotato.core.helpers.variable import splitString, getTitle
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.notifications.base import Notification from couchpotato.core.notifications.base import Notification
import requests from requests.exceptions import ConnectionError, Timeout
from requests.packages.urllib3.exceptions import MaxRetryError, ConnectionError from requests.packages.urllib3.exceptions import MaxRetryError
log = CPLog(__name__) log = CPLog(__name__)
@@ -172,7 +172,7 @@ class XBMC(Notification):
# manually fake expected response array # manually fake expected response array
return [{'result': 'Error'}] return [{'result': 'Error'}]
except (MaxRetryError, requests.exceptions.Timeout, ConnectionError): except (MaxRetryError, Timeout, ConnectionError):
log.info2('Couldn\'t send request to XBMC, assuming it\'s turned off') log.info2('Couldn\'t send request to XBMC, assuming it\'s turned off')
return [{'result': 'Error'}] return [{'result': 'Error'}]
except: except:
@@ -208,7 +208,7 @@ class XBMC(Notification):
log.debug('Returned from request %s: %s', (host, response)) log.debug('Returned from request %s: %s', (host, response))
return response return response
except (MaxRetryError, requests.exceptions.Timeout, ConnectionError): except (MaxRetryError, Timeout, ConnectionError):
log.info2('Couldn\'t send request to XBMC, assuming it\'s turned off') log.info2('Couldn\'t send request to XBMC, assuming it\'s turned off')
return [] return []
except: except:
+2 -1
View File
@@ -46,7 +46,8 @@ class Automation(Plugin):
break break
movie_dict = fireEvent('media.get', movie_id, single = True) movie_dict = fireEvent('media.get', movie_id, single = True)
fireEvent('movie.searcher.single', movie_dict) if movie_dict:
fireEvent('movie.searcher.single', movie_dict)
return True return True
+85 -36
View File
@@ -1,3 +1,4 @@
import threading
from urllib import quote from urllib import quote
from urlparse import urlparse from urlparse import urlparse
import glob import glob
@@ -10,7 +11,8 @@ import traceback
from couchpotato.core.event import fireEvent, addEvent from couchpotato.core.event import fireEvent, addEvent
from couchpotato.core.helpers.encoding import ss, toSafeString, \ from couchpotato.core.helpers.encoding import ss, toSafeString, \
toUnicode, sp toUnicode, sp
from couchpotato.core.helpers.variable import getExt, md5, isLocalIP, scanForPassword, tryInt, getIdentifier from couchpotato.core.helpers.variable import getExt, md5, isLocalIP, scanForPassword, tryInt, getIdentifier, \
randomString
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.environment import Env from couchpotato.environment import Env
import requests import requests
@@ -35,6 +37,8 @@ class Plugin(object):
_needs_shutdown = False _needs_shutdown = False
_running = None _running = None
_locks = {}
user_agent = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10.8; rv:24.0) Gecko/20130519 Firefox/24.0' user_agent = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10.8; rv:24.0) Gecko/20130519 Firefox/24.0'
http_last_use = {} http_last_use = {}
http_time_between_calls = 0 http_time_between_calls = 0
@@ -118,15 +122,31 @@ class Plugin(object):
if os.path.exists(path): if os.path.exists(path):
log.debug('%s already exists, overwriting file with new version', path) log.debug('%s already exists, overwriting file with new version', path)
try: write_type = 'w+' if not binary else 'w+b'
f = open(path, 'w+' if not binary else 'w+b')
f.write(content) # Stream file using response object
f.close() if isinstance(content, requests.models.Response):
os.chmod(path, Env.getPermission('file'))
except: # Write file to temp
log.error('Unable writing to file "%s": %s', (path, traceback.format_exc())) with open('%s.tmp' % path, write_type) as f:
if os.path.isfile(path): for chunk in content.iter_content(chunk_size = 1048576):
os.remove(path) if chunk: # filter out keep-alive new chunks
f.write(chunk)
f.flush()
# Rename to destination
os.rename('%s.tmp' % path, path)
else:
try:
f = open(path, write_type)
f.write(content)
f.close()
os.chmod(path, Env.getPermission('file'))
except:
log.error('Unable writing to file "%s": %s', (path, traceback.format_exc()))
if os.path.isfile(path):
os.remove(path)
def makeDir(self, path): def makeDir(self, path):
path = sp(path) path = sp(path)
@@ -143,21 +163,17 @@ class Plugin(object):
folder = sp(folder) folder = sp(folder)
for item in os.listdir(folder): for item in os.listdir(folder):
full_folder = os.path.join(folder, item) full_folder = sp(os.path.join(folder, item))
if not only_clean or (item in only_clean and os.path.isdir(full_folder)): if not only_clean or (item in only_clean and os.path.isdir(full_folder)):
for root, dirs, files in os.walk(full_folder): for subfolder, dirs, files in os.walk(full_folder, topdown = False):
for dir_name in dirs: try:
full_path = os.path.join(root, dir_name) os.rmdir(subfolder)
except:
if len(os.listdir(full_path)) == 0: if show_error:
try: log.info2('Couldn\'t remove directory %s: %s', (subfolder, traceback.format_exc()))
os.rmdir(full_path)
except:
if show_error:
log.info2('Couldn\'t remove directory %s: %s', (full_path, traceback.format_exc()))
try: try:
os.rmdir(folder) os.rmdir(folder)
@@ -166,7 +182,7 @@ class Plugin(object):
log.error('Couldn\'t remove empty directory %s: %s', (folder, traceback.format_exc())) log.error('Couldn\'t remove empty directory %s: %s', (folder, traceback.format_exc()))
# http request # http request
def urlopen(self, url, timeout = 30, data = None, headers = None, files = None, show_error = True): def urlopen(self, url, timeout = 30, data = None, headers = None, files = None, show_error = True, stream = False):
url = quote(ss(url), safe = "%/:=&?~#+!$,;'@()*[]") url = quote(ss(url), safe = "%/:=&?~#+!$,;'@()*[]")
if not headers: headers = {} if not headers: headers = {}
@@ -177,10 +193,10 @@ class Plugin(object):
host = '%s%s' % (parsed_url.hostname, (':' + str(parsed_url.port) if parsed_url.port else '')) host = '%s%s' % (parsed_url.hostname, (':' + str(parsed_url.port) if parsed_url.port else ''))
headers['Referer'] = headers.get('Referer', '%s://%s' % (parsed_url.scheme, host)) headers['Referer'] = headers.get('Referer', '%s://%s' % (parsed_url.scheme, host))
headers['Host'] = headers.get('Host', host) headers['Host'] = headers.get('Host', None)
headers['User-Agent'] = headers.get('User-Agent', self.user_agent) headers['User-Agent'] = headers.get('User-Agent', self.user_agent)
headers['Accept-encoding'] = headers.get('Accept-encoding', 'gzip') headers['Accept-encoding'] = headers.get('Accept-encoding', 'gzip')
headers['Connection'] = headers.get('Connection', 'keep-alive') headers['Connection'] = headers.get('Connection', 'close')
headers['Cache-Control'] = headers.get('Cache-Control', 'max-age=0') headers['Cache-Control'] = headers.get('Cache-Control', 'max-age=0')
r = Env.get('http_opener') r = Env.get('http_opener')
@@ -198,6 +214,7 @@ class Plugin(object):
del self.http_failed_disabled[host] del self.http_failed_disabled[host]
self.wait(host) self.wait(host)
status_code = None
try: try:
kwargs = { kwargs = {
@@ -206,14 +223,16 @@ class Plugin(object):
'timeout': timeout, 'timeout': timeout,
'files': files, 'files': files,
'verify': False, #verify_ssl, Disable for now as to many wrongly implemented certificates.. 'verify': False, #verify_ssl, Disable for now as to many wrongly implemented certificates..
'stream': stream,
} }
method = 'post' if len(data) > 0 or files else 'get' method = 'post' if len(data) > 0 or files else 'get'
log.info('Opening url: %s %s, data: %s', (method, url, [x for x in data.keys()] if isinstance(data, dict) else 'with data')) log.info('Opening url: %s %s, data: %s', (method, url, [x for x in data.keys()] if isinstance(data, dict) else 'with data'))
response = r.request(method, url, **kwargs) response = r.request(method, url, **kwargs)
status_code = response.status_code
if response.status_code == requests.codes.ok: if response.status_code == requests.codes.ok:
data = response.content data = response if stream else response.content
else: else:
response.raise_for_status() response.raise_for_status()
@@ -224,6 +243,12 @@ class Plugin(object):
# Save failed requests by hosts # Save failed requests by hosts
try: try:
# To many requests
if status_code in [429]:
self.http_failed_request[host] = 1
self.http_failed_disabled[host] = time.time()
if not self.http_failed_request.get(host): if not self.http_failed_request.get(host):
self.http_failed_request[host] = 1 self.http_failed_request[host] = 1
else: else:
@@ -254,8 +279,8 @@ class Plugin(object):
wait = (last_use - now) + self.http_time_between_calls wait = (last_use - now) + self.http_time_between_calls
if wait > 0: if wait > 0:
log.debug('Waiting for %s, %d seconds', (self.getName(), wait)) log.debug('Waiting for %s, %d seconds', (self.getName(), max(1, wait)))
time.sleep(wait) time.sleep(min(wait, 30))
def beforeCall(self, handler): def beforeCall(self, handler):
self.isRunning('%s.%s' % (self.getName(), handler.__name__)) self.isRunning('%s.%s' % (self.getName(), handler.__name__))
@@ -322,9 +347,9 @@ class Plugin(object):
Env.get('cache').set(cache_key_md5, value, timeout) Env.get('cache').set(cache_key_md5, value, timeout)
return value return value
def createNzbName(self, data, media): def createNzbName(self, data, media, unique_tag = False):
release_name = data.get('name') release_name = data.get('name')
tag = self.cpTag(media) tag = self.cpTag(media, unique_tag = unique_tag)
# Check if password is filename # Check if password is filename
name_password = scanForPassword(data.get('name')) name_password = scanForPassword(data.get('name'))
@@ -337,18 +362,26 @@ class Plugin(object):
max_length = 127 - len(tag) # Some filesystems don't support 128+ long filenames max_length = 127 - len(tag) # Some filesystems don't support 128+ long filenames
return '%s%s' % (toSafeString(toUnicode(release_name)[:max_length]), tag) return '%s%s' % (toSafeString(toUnicode(release_name)[:max_length]), tag)
def createFileName(self, data, filedata, media): def createFileName(self, data, filedata, media, unique_tag = False):
name = self.createNzbName(data, media) name = self.createNzbName(data, media, unique_tag = unique_tag)
if data.get('protocol') == 'nzb' and 'DOCTYPE nzb' not in filedata and '</nzb>' not in filedata: if data.get('protocol') == 'nzb' and 'DOCTYPE nzb' not in filedata and '</nzb>' not in filedata:
return '%s.%s' % (name, 'rar') return '%s.%s' % (name, 'rar')
return '%s.%s' % (name, data.get('protocol')) return '%s.%s' % (name, data.get('protocol'))
def cpTag(self, media): def cpTag(self, media, unique_tag = False):
if Env.setting('enabled', 'renamer'):
identifier = getIdentifier(media)
return '.cp(' + identifier + ')' if identifier else ''
return '' tag = ''
if Env.setting('enabled', 'renamer') or unique_tag:
identifier = getIdentifier(media) or ''
unique_tag = ', ' + randomString() if unique_tag else ''
tag = '.cp('
tag += identifier
tag += ', ' if unique_tag and identifier else ''
tag += randomString() if unique_tag else ''
tag += ')'
return tag if len(tag) > 7 else ''
def checkFilesChanged(self, files, unchanged_for = 60): def checkFilesChanged(self, files, unchanged_for = 60):
now = time.time() now = time.time()
@@ -393,3 +426,19 @@ class Plugin(object):
def isEnabled(self): def isEnabled(self):
return self.conf(self.enabled_option) or self.conf(self.enabled_option) is None return self.conf(self.enabled_option) or self.conf(self.enabled_option) is None
def acquireLock(self, key):
lock = self._locks.get(key)
if not lock:
self._locks[key] = threading.RLock()
log.debug('Acquiring lock: %s', key)
self._locks.get(key).acquire()
def releaseLock(self, key):
lock = self._locks.get(key)
if lock:
log.debug('Releasing lock: %s', key)
self._locks.get(key).release()
+20 -9
View File
@@ -1,12 +1,18 @@
import ctypes import ctypes
import os import os
import string import string
import traceback
import time
from couchpotato import CPLog
from couchpotato.api import addApiView from couchpotato.api import addApiView
from couchpotato.core.helpers.encoding import sp from couchpotato.core.event import addEvent
from couchpotato.core.helpers.encoding import sp, ss, toUnicode
from couchpotato.core.helpers.variable import getUserDir from couchpotato.core.helpers.variable import getUserDir
from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.base import Plugin
import six
log = CPLog(__name__)
if os.name == 'nt': if os.name == 'nt':
@@ -53,9 +59,9 @@ class FileBrowser(Plugin):
dirs = [] dirs = []
path = sp(path) path = sp(path)
for f in os.listdir(path): for f in os.listdir(path):
p = os.path.join(path, f) p = sp(os.path.join(path, f))
if os.path.isdir(p) and ((self.is_hidden(p) and bool(int(show_hidden))) or not self.is_hidden(p)): if os.path.isdir(p) and ((self.is_hidden(p) and bool(int(show_hidden))) or not self.is_hidden(p)):
dirs.append(p + os.path.sep) dirs.append(toUnicode('%s%s' % (p, os.path.sep)))
return sorted(dirs) return sorted(dirs)
@@ -66,8 +72,8 @@ class FileBrowser(Plugin):
driveletters = [] driveletters = []
for drive in string.ascii_uppercase: for drive in string.ascii_uppercase:
if win32file.GetDriveType(drive + ":") in [win32file.DRIVE_FIXED, win32file.DRIVE_REMOTE, win32file.DRIVE_RAMDISK, win32file.DRIVE_REMOVABLE]: if win32file.GetDriveType(drive + ':') in [win32file.DRIVE_FIXED, win32file.DRIVE_REMOTE, win32file.DRIVE_RAMDISK, win32file.DRIVE_REMOVABLE]:
driveletters.append(drive + ":\\") driveletters.append(drive + ':\\')
return driveletters return driveletters
@@ -100,14 +106,19 @@ class FileBrowser(Plugin):
def is_hidden(self, filepath): def is_hidden(self, filepath):
name = os.path.basename(os.path.abspath(filepath)) name = ss(os.path.basename(os.path.abspath(filepath)))
return name.startswith('.') or self.has_hidden_attribute(filepath) return name.startswith('.') or self.has_hidden_attribute(filepath)
def has_hidden_attribute(self, filepath): def has_hidden_attribute(self, filepath):
result = False
try: try:
attrs = ctypes.windll.kernel32.GetFileAttributesW(six.text_type(filepath)) #@UndefinedVariable attrs = ctypes.windll.kernel32.GetFileAttributesW(sp(filepath)) #@UndefinedVariable
assert attrs != -1 assert attrs != -1
result = bool(attrs & 2) result = bool(attrs & 2)
except (AttributeError, AssertionError): except (AttributeError, AssertionError):
result = False pass
except:
log.error('Failed getting hidden attribute: %s', traceback.format_exc())
return result return result
+1 -1
View File
@@ -27,7 +27,7 @@ class CategoryPlugin(Plugin):
'desc': 'List all available categories', 'desc': 'List all available categories',
'return': {'type': 'object', 'example': """{ 'return': {'type': 'object', 'example': """{
'success': True, 'success': True,
'list': array, categories 'categories': array, categories
}"""} }"""}
}) })
+16 -10
View File
@@ -1,6 +1,6 @@
from datetime import date
import random as rndm import random as rndm
import time import time
from CodernityDB.database import RecordDeleted
from couchpotato import get_db from couchpotato import get_db
from couchpotato.api import addApiView from couchpotato.api import addApiView
@@ -48,7 +48,6 @@ class Dashboard(Plugin):
active_ids = [x['_id'] for x in fireEvent('media.with_status', 'active', with_doc = False, single = True)] active_ids = [x['_id'] for x in fireEvent('media.with_status', 'active', with_doc = False, single = True)]
medias = [] medias = []
now_year = date.today().year
if len(active_ids) > 0: if len(active_ids) > 0:
@@ -60,7 +59,11 @@ class Dashboard(Plugin):
rndm.shuffle(active_ids) rndm.shuffle(active_ids)
for media_id in active_ids: for media_id in active_ids:
media = db.get('id', media_id) try:
media = db.get('id', media_id)
except RecordDeleted:
log.debug('Record already deleted: %s', media_id)
continue
pp = profile_pre.get(media.get('profile_id')) pp = profile_pre.get(media.get('profile_id'))
if not pp: continue if not pp: continue
@@ -69,23 +72,26 @@ class Dashboard(Plugin):
coming_soon = False coming_soon = False
# Theater quality # Theater quality
if pp.get('theater') and fireEvent('movie.searcher.could_be_released', True, eta, media['info'].get('year'), single = True): if pp.get('theater') and fireEvent('movie.searcher.could_be_released', True, eta, media['info']['year'], single = True):
coming_soon = True coming_soon = 'theater'
elif pp.get('dvd') and fireEvent('movie.searcher.could_be_released', False, eta, media['info'].get('year'), single = True): elif pp.get('dvd') and fireEvent('movie.searcher.could_be_released', False, eta, media['info']['year'], single = True):
coming_soon = True coming_soon = 'dvd'
if coming_soon: if coming_soon:
# Don't list older movies # Don't list older movies
if ((not late and (media['info'].get('year') >= now_year - 1) and (not eta.get('dvd') and not eta.get('theater') or eta.get('dvd') and eta.get('dvd') > (now - 2419200))) or eta_date = eta.get(coming_soon)
(late and (media['info'].get('year') < now_year - 1 or (eta.get('dvd', 0) > 0 or eta.get('theater')) and eta.get('dvd') < (now - 2419200)))): eta_3month_passed = eta_date < (now - 7862400) # Release was more than 3 months ago
if (not late and not eta_3month_passed) or \
(late and eta_3month_passed):
add = True add = True
# Check if it doesn't have any releases # Check if it doesn't have any releases
if late: if late:
media['releases'] = fireEvent('release.for_media', media['_id'], single = True) media['releases'] = fireEvent('release.for_media', media['_id'], single = True)
for release in media.get('releases'): for release in media.get('releases'):
if release.get('status') in ['snatched', 'available', 'seeding', 'downloaded']: if release.get('status') in ['snatched', 'available', 'seeding', 'downloaded']:
add = False add = False
+9 -4
View File
@@ -4,7 +4,7 @@ import traceback
from couchpotato import get_db from couchpotato import get_db
from couchpotato.api import addApiView from couchpotato.api import addApiView
from couchpotato.core.event import addEvent, fireEvent from couchpotato.core.event import addEvent, fireEvent
from couchpotato.core.helpers.encoding import toUnicode from couchpotato.core.helpers.encoding import toUnicode, ss, sp
from couchpotato.core.helpers.variable import md5, getExt, isSubFolder from couchpotato.core.helpers.variable import md5, getExt, isSubFolder
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.base import Plugin
@@ -59,13 +59,18 @@ class FileManager(Plugin):
log.error('Failed removing unused file: %s', traceback.format_exc()) log.error('Failed removing unused file: %s', traceback.format_exc())
def showCacheFile(self, route, **kwargs): def showCacheFile(self, route, **kwargs):
Env.get('app').add_handlers(".*$", [('%s%s' % (Env.get('api_base'), route), StaticFileHandler, {'path': Env.get('cache_dir')})]) Env.get('app').add_handlers(".*$", [('%s%s' % (Env.get('api_base'), route), StaticFileHandler, {'path': toUnicode(Env.get('cache_dir'))})])
def download(self, url = '', dest = None, overwrite = False, urlopen_kwargs = None): def download(self, url = '', dest = None, overwrite = False, urlopen_kwargs = None):
if not urlopen_kwargs: urlopen_kwargs = {} if not urlopen_kwargs: urlopen_kwargs = {}
# Return response object to stream download
urlopen_kwargs['stream'] = True
if not dest: # to Cache if not dest: # to Cache
dest = os.path.join(Env.get('cache_dir'), '%s.%s' % (md5(url), getExt(url))) dest = os.path.join(Env.get('cache_dir'), ss('%s.%s' % (md5(url), getExt(url))))
dest = sp(dest)
if not overwrite and os.path.isfile(dest): if not overwrite and os.path.isfile(dest):
return dest return dest
@@ -107,4 +112,4 @@ class FileManager(Plugin):
else: else:
log.info('Subfolder test succeeded') log.info('Subfolder test succeeded')
return failed == 0 return failed == 0
+1 -1
View File
@@ -241,7 +241,7 @@ Running on: ...\n\
'href': 'https://github.com/RuudBurger/CouchPotatoServer/blob/develop/contributing.md' 'href': 'https://github.com/RuudBurger/CouchPotatoServer/blob/develop/contributing.md'
}), }),
new Element('span', { new Element('span', {
'text': ' before posting (kittens die if you don\'t), then copy the text below.' 'html': ' before posting, then copy the text below and <strong>FILL IN</strong> the dots.'
}) })
), ),
textarea = new Element('textarea', { textarea = new Element('textarea', {
+5 -3
View File
@@ -123,7 +123,7 @@ class Manage(Plugin):
fireEvent('notify.frontend', type = 'manage.update', data = True, message = 'Scanning for movies in "%s"' % folder) fireEvent('notify.frontend', type = 'manage.update', data = True, message = 'Scanning for movies in "%s"' % folder)
onFound = self.createAddToLibrary(folder, added_identifiers) onFound = self.createAddToLibrary(folder, added_identifiers)
fireEvent('scanner.scan', folder = folder, simple = True, newer_than = last_update if not full else 0, on_found = onFound, single = True) fireEvent('scanner.scan', folder = folder, simple = True, newer_than = last_update if not full else 0, check_file_date = False, on_found = onFound, single = True)
# Break if CP wants to shut down # Break if CP wants to shut down
if self.shuttingDown(): if self.shuttingDown():
@@ -165,7 +165,7 @@ class Manage(Plugin):
already_used = used_files.get(release_file) already_used = used_files.get(release_file)
if already_used: if already_used:
release_id = release['_id'] if already_used.get('last_edit', 0) < release.get('last_edit', 0) else already_used['_id'] release_id = release['_id'] if already_used.get('last_edit', 0) > release.get('last_edit', 0) else already_used['_id']
if release_id not in deleted_releases: if release_id not in deleted_releases:
fireEvent('release.delete', release_id, single = True) fireEvent('release.delete', release_id, single = True)
deleted_releases.append(release_id) deleted_releases.append(release_id)
@@ -190,6 +190,7 @@ class Manage(Plugin):
delete_me = {} delete_me = {}
# noinspection PyTypeChecker
for folder in self.in_progress: for folder in self.in_progress:
if self.in_progress[folder]['to_go'] <= 0: if self.in_progress[folder]['to_go'] <= 0:
delete_me[folder] = True delete_me[folder] = True
@@ -233,7 +234,8 @@ class Manage(Plugin):
total = self.in_progress[folder]['total'] total = self.in_progress[folder]['total']
movie_dict = fireEvent('media.get', identifier, single = True) movie_dict = fireEvent('media.get', identifier, single = True)
fireEvent('notify.frontend', type = 'movie.added', data = movie_dict, message = None if total > 5 else 'Added "%s" to manage.' % getTitle(movie_dict)) if movie_dict:
fireEvent('notify.frontend', type = 'movie.added', data = movie_dict, message = None if total > 5 else 'Added "%s" to manage.' % getTitle(movie_dict))
return afterUpdate return afterUpdate
+2
View File
@@ -86,6 +86,7 @@ class ProfilePlugin(Plugin):
'label': toUnicode(kwargs.get('label')), 'label': toUnicode(kwargs.get('label')),
'order': tryInt(kwargs.get('order', 999)), 'order': tryInt(kwargs.get('order', 999)),
'core': kwargs.get('core', False), 'core': kwargs.get('core', False),
'minimum_score': tryInt(kwargs.get('minimum_score', 1)),
'qualities': [], 'qualities': [],
'wait_for': [], 'wait_for': [],
'stop_after': [], 'stop_after': [],
@@ -217,6 +218,7 @@ class ProfilePlugin(Plugin):
'label': toUnicode(profile.get('label')), 'label': toUnicode(profile.get('label')),
'order': order, 'order': order,
'qualities': profile.get('qualities'), 'qualities': profile.get('qualities'),
'minimum_score': 1,
'finish': [], 'finish': [],
'wait_for': [], 'wait_for': [],
'stop_after': [], 'stop_after': [],
@@ -51,6 +51,11 @@
margin: 0 5px !important; margin: 0 5px !important;
} }
.profile .wait_for .minimum_score_input {
width: 40px !important;
text-align: left;
}
.profile .types { .profile .types {
padding: 0; padding: 0;
margin: 0 20px 0 -4px; margin: 0 20px 0 -4px;
@@ -53,12 +53,21 @@ var Profile = new Class({
}), }),
new Element('span', {'text':'day(s) for a better quality '}), new Element('span', {'text':'day(s) for a better quality '}),
new Element('span.advanced', {'text':'and keep searching'}), new Element('span.advanced', {'text':'and keep searching'}),
// "After a checked quality is found and downloaded, continue searching for even better quality releases for the entered number of days." // "After a checked quality is found and downloaded, continue searching for even better quality releases for the entered number of days."
new Element('input.inlay.xsmall.stop_after_input.advanced', { new Element('input.inlay.xsmall.stop_after_input.advanced', {
'type':'text', 'type':'text',
'value': data.stop_after && data.stop_after.length > 0 ? data.stop_after[0] : 0 'value': data.stop_after && data.stop_after.length > 0 ? data.stop_after[0] : 0
}), }),
new Element('span.advanced', {'text':'day(s) for a better (checked) quality.'}) new Element('span.advanced', {'text':'day(s) for a better (checked) quality.'}),
// Minimum score of
new Element('span.advanced', {'html':'<br/>Releases need a minimum score of'}),
new Element('input.advanced.inlay.xsmall.minimum_score_input', {
'size': 4,
'type':'text',
'value': data.minimum_score || 1
})
) )
); );
@@ -126,6 +135,7 @@ var Profile = new Class({
'label' : self.el.getElement('.quality_label input').get('value'), 'label' : self.el.getElement('.quality_label input').get('value'),
'wait_for' : self.el.getElement('.wait_for_input').get('value'), 'wait_for' : self.el.getElement('.wait_for_input').get('value'),
'stop_after' : self.el.getElement('.stop_after_input').get('value'), 'stop_after' : self.el.getElement('.stop_after_input').get('value'),
'minimum_score' : self.el.getElement('.minimum_score_input').get('value'),
'types': [] 'types': []
}; };
+78 -44
View File
@@ -1,3 +1,4 @@
from math import fabs, ceil
import traceback import traceback
import re import re
@@ -6,7 +7,7 @@ from couchpotato import get_db
from couchpotato.api import addApiView from couchpotato.api import addApiView
from couchpotato.core.event import addEvent, fireEvent from couchpotato.core.event import addEvent, fireEvent
from couchpotato.core.helpers.encoding import toUnicode, ss from couchpotato.core.helpers.encoding import toUnicode, ss
from couchpotato.core.helpers.variable import mergeDicts, getExt, tryInt, splitString from couchpotato.core.helpers.variable import mergeDicts, getExt, tryInt, splitString, tryFloat
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.base import Plugin
from couchpotato.core.plugins.quality.index import QualityIndex from couchpotato.core.plugins.quality.index import QualityIndex
@@ -22,17 +23,17 @@ class QualityPlugin(Plugin):
} }
qualities = [ qualities = [
{'identifier': 'bd50', 'hd': True, 'allow_3d': True, 'size': (20000, 60000), 'label': 'BR-Disk', 'alternative': ['bd25', ('br', 'disk')], 'allow': ['1080p'], 'ext':['iso', 'img'], 'tags': ['bdmv', 'certificate', ('complete', 'bluray'), 'avc', 'mvc']}, {'identifier': 'bd50', 'hd': True, 'allow_3d': True, 'size': (20000, 60000), 'median_size': 40000, 'label': 'BR-Disk', 'alternative': ['bd25', ('br', 'disk')], 'allow': ['1080p'], 'ext':['iso', 'img'], 'tags': ['bdmv', 'certificate', ('complete', 'bluray'), 'avc', 'mvc']},
{'identifier': '1080p', 'hd': True, 'allow_3d': True, 'size': (4000, 20000), 'label': '1080p', 'width': 1920, 'height': 1080, 'alternative': [], 'allow': [], 'ext':['mkv', 'm2ts', 'ts'], 'tags': ['m2ts', 'x264', 'h264']}, {'identifier': '1080p', 'hd': True, 'allow_3d': True, 'size': (4000, 20000), 'median_size': 10000, 'label': '1080p', 'width': 1920, 'height': 1080, 'alternative': [], 'allow': [], 'ext':['mkv', 'm2ts', 'ts'], 'tags': ['m2ts', 'x264', 'h264']},
{'identifier': '720p', 'hd': True, 'allow_3d': True, 'size': (3000, 10000), 'label': '720p', 'width': 1280, 'height': 720, 'alternative': [], 'allow': [], 'ext':['mkv', 'ts'], 'tags': ['x264', 'h264']}, {'identifier': '720p', 'hd': True, 'allow_3d': True, 'size': (3000, 10000), 'median_size': 5500, 'label': '720p', 'width': 1280, 'height': 720, 'alternative': [], 'allow': [], 'ext':['mkv', 'ts'], 'tags': ['x264', 'h264']},
{'identifier': 'brrip', 'hd': True, 'allow_3d': True, 'size': (700, 7000), 'label': 'BR-Rip', 'alternative': ['bdrip', ('br', 'rip')], 'allow': ['720p', '1080p'], 'ext':['mp4', 'avi'], 'tags': ['hdtv', 'hdrip', 'webdl', ('web', 'dl')]}, {'identifier': 'brrip', 'hd': True, 'allow_3d': True, 'size': (700, 7000), 'median_size': 2000, 'label': 'BR-Rip', 'alternative': ['bdrip', ('br', 'rip'), 'hdtv', 'hdrip'], 'allow': ['720p', '1080p'], 'ext':['mp4', 'avi'], 'tags': ['webdl', ('web', 'dl')]},
{'identifier': 'dvdr', 'size': (3000, 10000), 'label': 'DVD-R', 'alternative': ['br2dvd', ('dvd', 'r')], 'allow': [], 'ext':['iso', 'img', 'vob'], 'tags': ['pal', 'ntsc', 'video_ts', 'audio_ts', ('dvd', 'r'), 'dvd9']}, {'identifier': 'dvdr', 'size': (3000, 10000), 'median_size': 4500, 'label': 'DVD-R', 'alternative': ['br2dvd', ('dvd', 'r')], 'allow': [], 'ext':['iso', 'img', 'vob'], 'tags': ['pal', 'ntsc', 'video_ts', 'audio_ts', ('dvd', 'r'), 'dvd9']},
{'identifier': 'dvdrip', 'size': (600, 2400), 'label': 'DVD-Rip', 'width': 720, 'alternative': [('dvd', 'rip')], 'allow': [], 'ext':['avi'], 'tags': [('dvd', 'rip'), ('dvd', 'xvid'), ('dvd', 'divx')]}, {'identifier': 'dvdrip', 'size': (600, 2400), 'median_size': 1500, 'label': 'DVD-Rip', 'width': 720, 'alternative': [('dvd', 'rip')], 'allow': [], 'ext':['avi'], 'tags': [('dvd', 'rip'), ('dvd', 'xvid'), ('dvd', 'divx')]},
{'identifier': 'scr', 'size': (600, 1600), 'label': 'Screener', 'alternative': ['screener', 'dvdscr', 'ppvrip', 'dvdscreener', 'hdscr'], 'allow': ['dvdr', 'dvdrip', '720p', '1080p'], 'ext':[], 'tags': ['webrip', ('web', 'rip')]}, {'identifier': 'scr', 'size': (600, 1600), 'median_size': 700, 'label': 'Screener', 'alternative': ['screener', 'dvdscr', 'ppvrip', 'dvdscreener', 'hdscr', 'webrip', ('web', 'rip')], 'allow': ['dvdr', 'dvdrip', '720p', '1080p'], 'ext':[], 'tags': []},
{'identifier': 'r5', 'size': (600, 1000), 'label': 'R5', 'alternative': ['r6'], 'allow': ['dvdr', '720p'], 'ext':[]}, {'identifier': 'r5', 'size': (600, 1000), 'median_size': 700, 'label': 'R5', 'alternative': ['r6'], 'allow': ['dvdr', '720p', '1080p'], 'ext':[]},
{'identifier': 'tc', 'size': (600, 1000), 'label': 'TeleCine', 'alternative': ['telecine'], 'allow': ['720p'], 'ext':[]}, {'identifier': 'tc', 'size': (600, 1000), 'median_size': 700, 'label': 'TeleCine', 'alternative': ['telecine'], 'allow': ['720p', '1080p'], 'ext':[]},
{'identifier': 'ts', 'size': (600, 1000), 'label': 'TeleSync', 'alternative': ['telesync', 'hdts'], 'allow': ['720p'], 'ext':[]}, {'identifier': 'ts', 'size': (600, 1000), 'median_size': 700, 'label': 'TeleSync', 'alternative': ['telesync', 'hdts'], 'allow': ['720p', '1080p'], 'ext':[]},
{'identifier': 'cam', 'size': (600, 1000), 'label': 'Cam', 'alternative': ['camrip', 'hdcam'], 'allow': ['720p'], 'ext':[]}, {'identifier': 'cam', 'size': (600, 1000), 'median_size': 700, 'label': 'Cam', 'alternative': ['camrip', 'hdcam'], 'allow': ['720p', '1080p'], 'ext':[]},
# TODO come back to this later, think this could be handled better, this is starting to get out of hand.... # TODO come back to this later, think this could be handled better, this is starting to get out of hand....
# BluRay # BluRay
@@ -205,14 +206,15 @@ class QualityPlugin(Plugin):
return False return False
def guess(self, files, extra = None, size = None): def guess(self, files, extra = None, size = None, use_cache = True):
if not extra: extra = {} if not extra: extra = {}
# Create hash for cache # Create hash for cache
cache_key = str([f.replace('.' + getExt(f), '') if len(getExt(f)) < 4 else f for f in files]) cache_key = str([f.replace('.' + getExt(f), '') if len(getExt(f)) < 4 else f for f in files])
cached = self.getCache(cache_key) if use_cache:
if cached and len(extra) == 0: cached = self.getCache(cache_key)
return cached if cached and len(extra) == 0:
return cached
qualities = self.all() qualities = self.all()
@@ -224,6 +226,10 @@ class QualityPlugin(Plugin):
'3d': {} '3d': {}
} }
# Use metadata titles as extra check
if extra and extra.get('titles'):
files.extend(extra.get('titles'))
for cur_file in files: for cur_file in files:
words = re.split('\W+', cur_file.lower()) words = re.split('\W+', cur_file.lower())
name_year = fireEvent('scanner.name_year', cur_file, file_name = cur_file, single = True) name_year = fireEvent('scanner.name_year', cur_file, file_name = cur_file, single = True)
@@ -236,7 +242,7 @@ class QualityPlugin(Plugin):
contains_score = self.containsTagScore(quality, words, cur_file) contains_score = self.containsTagScore(quality, words, cur_file)
threedscore = self.contains3D(quality, threed_words, cur_file) if quality.get('allow_3d') else (0, None) threedscore = self.contains3D(quality, threed_words, cur_file) if quality.get('allow_3d') else (0, None)
self.calcScore(score, quality, contains_score, threedscore) self.calcScore(score, quality, contains_score, threedscore, penalty = contains_score)
size_scores = [] size_scores = []
for quality in qualities: for quality in qualities:
@@ -248,11 +254,11 @@ class QualityPlugin(Plugin):
if size_score > 0: if size_score > 0:
size_scores.append(quality) size_scores.append(quality)
self.calcScore(score, quality, size_score + loose_score, penalty = False) self.calcScore(score, quality, size_score + loose_score)
# Add additional size score if only 1 size validated # Add additional size score if only 1 size validated
if len(size_scores) == 1: if len(size_scores) == 1:
self.calcScore(score, size_scores[0], 10, penalty = False) self.calcScore(score, size_scores[0], 8)
del size_scores del size_scores
# Return nothing if all scores are <= 0 # Return nothing if all scores are <= 0
@@ -277,19 +283,21 @@ class QualityPlugin(Plugin):
def containsTagScore(self, quality, words, cur_file = ''): def containsTagScore(self, quality, words, cur_file = ''):
cur_file = ss(cur_file) cur_file = ss(cur_file)
score = 0 score = 0.0
extension = words[-1] extension = words[-1]
words = words[:-1] words = words[:-1]
points = { points = {
'identifier': 10, 'identifier': 20,
'label': 10, 'label': 20,
'alternative': 9, 'alternative': 20,
'tags': 9, 'tags': 11,
'ext': 3, 'ext': 5,
} }
scored_on = []
# Check alt and tags # Check alt and tags
for tag_type in ['identifier', 'alternative', 'tags', 'label']: for tag_type in ['identifier', 'alternative', 'tags', 'label']:
qualities = quality.get(tag_type, []) qualities = quality.get(tag_type, [])
@@ -301,13 +309,12 @@ class QualityPlugin(Plugin):
log.debug('Found %s via %s %s in %s', (quality['identifier'], tag_type, quality.get(tag_type), cur_file)) log.debug('Found %s via %s %s in %s', (quality['identifier'], tag_type, quality.get(tag_type), cur_file))
score += points.get(tag_type) score += points.get(tag_type)
if isinstance(alt, (str, unicode)) and ss(alt.lower()) in words: if isinstance(alt, (str, unicode)) and ss(alt.lower()) in words and ss(alt.lower()) not in scored_on:
log.debug('Found %s via %s %s in %s', (quality['identifier'], tag_type, quality.get(tag_type), cur_file)) log.debug('Found %s via %s %s in %s', (quality['identifier'], tag_type, quality.get(tag_type), cur_file))
score += points.get(tag_type) / 2 score += points.get(tag_type)
if list(set(qualities) & set(words)): # Don't score twice on same tag
log.debug('Found %s via %s %s in %s', (quality['identifier'], tag_type, quality.get(tag_type), cur_file)) scored_on.append(ss(alt).lower())
score += points.get(tag_type)
# Check extention # Check extention
for ext in quality.get('ext', []): for ext in quality.get('ext', []):
@@ -343,7 +350,7 @@ class QualityPlugin(Plugin):
# Check width resolution, range 20 # Check width resolution, range 20
if quality.get('width') and (quality.get('width') - 20) <= extra.get('resolution_width', 0) <= (quality.get('width') + 20): if quality.get('width') and (quality.get('width') - 20) <= extra.get('resolution_width', 0) <= (quality.get('width') + 20):
log.debug('Found %s via resolution_width: %s == %s', (quality['identifier'], quality.get('width'), extra.get('resolution_width', 0))) log.debug('Found %s via resolution_width: %s == %s', (quality['identifier'], quality.get('width'), extra.get('resolution_width', 0)))
score += 5 score += 10
# Check height resolution, range 20 # Check height resolution, range 20
if quality.get('height') and (quality.get('height') - 20) <= extra.get('resolution_height', 0) <= (quality.get('height') + 20): if quality.get('height') and (quality.get('height') - 20) <= extra.get('resolution_height', 0) <= (quality.get('height') + 20):
@@ -363,15 +370,28 @@ class QualityPlugin(Plugin):
if size: if size:
if tryInt(quality['size_min']) <= tryInt(size) <= tryInt(quality['size_max']): size = tryFloat(size)
log.debug('Found %s via release size: %s MB < %s MB < %s MB', (quality['identifier'], quality['size_min'], size, quality['size_max'])) size_min = tryFloat(quality['size_min'])
score += 5 size_max = tryFloat(quality['size_max'])
if size_min <= size <= size_max:
log.debug('Found %s via release size: %s MB < %s MB < %s MB', (quality['identifier'], size_min, size, size_max))
proc_range = size_max - size_min
size_diff = size - size_min
size_proc = (size_diff / proc_range)
median_diff = quality['median_size'] - size_min
median_proc = (median_diff / proc_range)
max_points = 8
score += ceil(max_points - (fabs(size_proc - median_proc) * max_points))
else: else:
score -= 5 score -= 5
return score return score
def calcScore(self, score, quality, add_score, threedscore = (0, None), penalty = True): def calcScore(self, score, quality, add_score, threedscore = (0, None), penalty = 0):
score[quality['identifier']]['score'] += add_score score[quality['identifier']]['score'] += add_score
@@ -390,11 +410,11 @@ class QualityPlugin(Plugin):
if penalty and add_score != 0: if penalty and add_score != 0:
for allow in quality.get('allow', []): for allow in quality.get('allow', []):
score[allow]['score'] -= 40 if self.cached_order[allow] < self.cached_order[quality['identifier']] else 5 score[allow]['score'] -= ((penalty * 2) if self.cached_order[allow] < self.cached_order[quality['identifier']] else penalty) * 2
# Give panelty for all lower qualities # Give panelty for all other qualities
for q in self.qualities[self.order.index(quality.get('identifier'))+1:]: for q in self.qualities:
if score.get(q.get('identifier')): if quality.get('identifier') != q.get('identifier') and score.get(q.get('identifier')):
score[q.get('identifier')]['score'] -= 1 score[q.get('identifier')]['score'] -= 1
def isFinish(self, quality, profile, release_age = 0): def isFinish(self, quality, profile, release_age = 0):
@@ -462,10 +482,12 @@ class QualityPlugin(Plugin):
'Movie Monuments 2013 BrRip 1080p': {'size': 1800, 'quality': 'brrip'}, 'Movie Monuments 2013 BrRip 1080p': {'size': 1800, 'quality': 'brrip'},
'Movie Monuments 2013 BrRip 720p': {'size': 1300, 'quality': 'brrip'}, 'Movie Monuments 2013 BrRip 720p': {'size': 1300, 'quality': 'brrip'},
'The.Movie.2014.3D.1080p.BluRay.AVC.DTS-HD.MA.5.1-GroupName': {'size': 30000, 'quality': 'bd50', 'is_3d': True}, 'The.Movie.2014.3D.1080p.BluRay.AVC.DTS-HD.MA.5.1-GroupName': {'size': 30000, 'quality': 'bd50', 'is_3d': True},
'/home/namehou/Movie Monuments (2013)/Movie Monuments.mkv': {'size': 4500, 'quality': '1080p', 'is_3d': False}, '/home/namehou/Movie Monuments (2012)/Movie Monuments.mkv': {'size': 5500, 'quality': '720p', 'is_3d': False},
'/home/namehou/Movie Monuments (2013)/Movie Monuments Full-OU.mkv': {'size': 4500, 'quality': '1080p', 'is_3d': True}, '/home/namehou/Movie Monuments (2012)/Movie Monuments Full-OU.mkv': {'size': 5500, 'quality': '720p', 'is_3d': True},
'/home/namehou/Movie Monuments (2013)/Movie Monuments.mkv': {'size': 10000, 'quality': '1080p', 'is_3d': False},
'/home/namehou/Movie Monuments (2013)/Movie Monuments Full-OU.mkv': {'size': 10000, 'quality': '1080p', 'is_3d': True},
'/volume1/Public/3D/Moviename/Moviename (2009).3D.SBS.ts': {'size': 7500, 'quality': '1080p', 'is_3d': True}, '/volume1/Public/3D/Moviename/Moviename (2009).3D.SBS.ts': {'size': 7500, 'quality': '1080p', 'is_3d': True},
'/volume1/Public/Moviename/Moviename (2009).ts': {'size': 5500, 'quality': '1080p'}, '/volume1/Public/Moviename/Moviename (2009).ts': {'size': 7500, 'quality': '1080p'},
'/movies/BluRay HDDVD H.264 MKV 720p EngSub/QuiQui le fou (criterion collection #123, 1915)/QuiQui le fou (1915) 720p x264 BluRay.mkv': {'size': 5500, 'quality': '720p'}, '/movies/BluRay HDDVD H.264 MKV 720p EngSub/QuiQui le fou (criterion collection #123, 1915)/QuiQui le fou (1915) 720p x264 BluRay.mkv': {'size': 5500, 'quality': '720p'},
'C:\\movies\QuiQui le fou (collection #123, 1915)\QuiQui le fou (1915) 720p x264 BluRay.mkv': {'size': 5500, 'quality': '720p'}, 'C:\\movies\QuiQui le fou (collection #123, 1915)\QuiQui le fou (1915) 720p x264 BluRay.mkv': {'size': 5500, 'quality': '720p'},
'C:\\movies\QuiQui le fou (collection #123, 1915)\QuiQui le fou (1915) half-sbs 720p x264 BluRay.mkv': {'size': 5500, 'quality': '720p', 'is_3d': True}, 'C:\\movies\QuiQui le fou (collection #123, 1915)\QuiQui le fou (1915) half-sbs 720p x264 BluRay.mkv': {'size': 5500, 'quality': '720p', 'is_3d': True},
@@ -474,12 +496,24 @@ class QualityPlugin(Plugin):
'Movie Name (2014).mp4': {'size': 750, 'quality': 'brrip'}, 'Movie Name (2014).mp4': {'size': 750, 'quality': 'brrip'},
'Moviename.2014.720p.R6.WEB-DL.x264.AC3-xyz': {'size': 750, 'quality': 'r5'}, 'Moviename.2014.720p.R6.WEB-DL.x264.AC3-xyz': {'size': 750, 'quality': 'r5'},
'Movie name 2014 New Source 720p HDCAM x264 AC3 xyz': {'size': 750, 'quality': 'cam'}, 'Movie name 2014 New Source 720p HDCAM x264 AC3 xyz': {'size': 750, 'quality': 'cam'},
'Movie.Name.2014.720p.HD.TS.AC3.x264': {'size': 750, 'quality': 'ts'} 'Movie.Name.2014.720p.HD.TS.AC3.x264': {'size': 750, 'quality': 'ts'},
'Movie.Name.2014.1080p.HDrip.x264.aac-ReleaseGroup': {'size': 7000, 'quality': 'brrip'},
'Movie.Name.2014.HDCam.Chinese.Subs-ReleaseGroup': {'size': 15000, 'quality': 'cam'},
'Movie Name 2014 HQ DVDRip X264 AC3 (bla)': {'size': 0, 'quality': 'dvdrip'},
'Movie Name1 (2012).mkv': {'size': 4500, 'quality': '720p'},
'Movie Name (2013).mkv': {'size': 8500, 'quality': '1080p'},
'Movie Name (2014).mkv': {'size': 4500, 'quality': '720p', 'extra': {'titles': ['Movie Name 2014 720p Bluray']}},
'Movie Name (2015).mkv': {'size': 500, 'quality': '1080p', 'extra': {'resolution_width': 1920}},
'Movie Name (2015).mp4': {'size': 6500, 'quality': 'brrip'},
'Movie Name (2015).mp4': {'size': 6500, 'quality': 'brrip'},
'Movie Name.2014.720p Web-Dl Aac2.0 h264-ReleaseGroup': {'size': 3800, 'quality': 'brrip'},
'Movie Name.2014.720p.WEBRip.x264.AC3-ReleaseGroup': {'size': 3000, 'quality': 'scr'},
'Movie.Name.2014.1080p.HDCAM.-.ReleaseGroup': {'size': 5300, 'quality': 'cam'},
} }
correct = 0 correct = 0
for name in tests: for name in tests:
test_quality = self.guess(files = [name], extra = tests[name].get('extra', None), size = tests[name].get('size', None)) or {} test_quality = self.guess(files = [name], extra = tests[name].get('extra', None), size = tests[name].get('size', None), use_cache = False) or {}
success = test_quality.get('identifier') == tests[name]['quality'] and test_quality.get('is_3d') == tests[name].get('is_3d', False) success = test_quality.get('identifier') == tests[name]['quality'] and test_quality.get('is_3d') == tests[name].get('is_3d', False)
if not success: if not success:
log.error('%s failed check, thinks it\'s "%s" expecting "%s"', (name, log.error('%s failed check, thinks it\'s "%s" expecting "%s"', (name,
+42 -19
View File
@@ -8,7 +8,7 @@ from couchpotato import md5, get_db
from couchpotato.api import addApiView from couchpotato.api import addApiView
from couchpotato.core.event import fireEvent, addEvent from couchpotato.core.event import fireEvent, addEvent
from couchpotato.core.helpers.encoding import toUnicode, sp from couchpotato.core.helpers.encoding import toUnicode, sp
from couchpotato.core.helpers.variable import getTitle from couchpotato.core.helpers.variable import getTitle, tryInt
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.base import Plugin
from .index import ReleaseIndex, ReleaseStatusIndex, ReleaseIDIndex, ReleaseDownloadIndex from .index import ReleaseIndex, ReleaseStatusIndex, ReleaseIDIndex, ReleaseDownloadIndex
@@ -65,43 +65,58 @@ class Release(Plugin):
log.debug('Removing releases from dashboard') log.debug('Removing releases from dashboard')
now = time.time() now = time.time()
week = 262080 week = 604800
db = get_db() db = get_db()
# Get (and remove) parentless releases # Get (and remove) parentless releases
releases = db.all('release', with_doc = True) releases = db.all('release', with_doc = False)
media_exist = [] media_exist = []
reindex = 0
for release in releases: for release in releases:
if release.get('key') in media_exist: if release.get('key') in media_exist:
continue continue
try: try:
try:
doc = db.get('id', release.get('_id'))
except RecordDeleted:
reindex += 1
continue
db.get('id', release.get('key')) db.get('id', release.get('key'))
media_exist.append(release.get('key')) media_exist.append(release.get('key'))
try: try:
if release['doc'].get('status') == 'ignore': if doc.get('status') == 'ignore':
release['doc']['status'] = 'ignored' doc['status'] = 'ignored'
db.update(release['doc']) db.update(doc)
except: except:
log.error('Failed fixing mis-status tag: %s', traceback.format_exc()) log.error('Failed fixing mis-status tag: %s', traceback.format_exc())
except ValueError:
fireEvent('database.delete_corrupted', release.get('key'), traceback_error = traceback.format_exc(0))
reindex += 1
except RecordDeleted: except RecordDeleted:
db.delete(release['doc']) db.delete(doc)
log.debug('Deleted orphaned release: %s', release['doc']) log.debug('Deleted orphaned release: %s', doc)
reindex += 1
except: except:
log.debug('Failed cleaning up orphaned releases: %s', traceback.format_exc()) log.debug('Failed cleaning up orphaned releases: %s', traceback.format_exc())
if reindex > 0:
db.reindex()
del media_exist del media_exist
# get movies last_edit more than a week ago # get movies last_edit more than a week ago
medias = fireEvent('media.with_status', 'done', single = True) medias = fireEvent('media.with_status', ['done', 'active'], single = True)
for media in medias: for media in medias:
if media.get('last_edit', 0) > (now - week): if media.get('last_edit', 0) > (now - week):
continue continue
for rel in fireEvent('release.for_media', media['_id'], single = True): for rel in self.forMedia(media['_id']):
# Remove all available releases # Remove all available releases
if rel['status'] in ['available']: if rel['status'] in ['available']:
@@ -111,7 +126,8 @@ class Release(Plugin):
elif rel['status'] in ['snatched', 'downloaded']: elif rel['status'] in ['snatched', 'downloaded']:
self.updateStatus(rel['_id'], status = 'ignored') self.updateStatus(rel['_id'], status = 'ignored')
fireEvent('media.untag', media.get('_id'), 'recent', single = True) if 'recent' in media.get('tags', []):
fireEvent('media.untag', media.get('_id'), 'recent', single = True)
def add(self, group, update_info = True, update_id = None): def add(self, group, update_info = True, update_id = None):
@@ -171,7 +187,7 @@ class Release(Plugin):
release['files'] = dict((k, [toUnicode(x) for x in v]) for k, v in group['files'].items() if v) release['files'] = dict((k, [toUnicode(x) for x in v]) for k, v in group['files'].items() if v)
db.update(release) db.update(release)
fireEvent('media.restatus', media['_id'], single = True) fireEvent('media.restatus', media['_id'], allowed_restatus = ['done'], single = True)
return True return True
except: except:
@@ -364,6 +380,7 @@ class Release(Plugin):
wait_for = False wait_for = False
let_through = False let_through = False
filtered_results = [] filtered_results = []
minimum_seeders = tryInt(Env.setting('minimum_seeders', section = 'torrent', default = 1))
# Filter out ignored and other releases we don't want # Filter out ignored and other releases we don't want
for rel in results: for rel in results:
@@ -372,16 +389,16 @@ class Release(Plugin):
log.info('Ignored: %s', rel['name']) log.info('Ignored: %s', rel['name'])
continue continue
if rel['score'] <= 0: if rel['score'] < quality_custom.get('minimum_score'):
log.info('Ignored, score "%s" to low: %s', (rel['score'], rel['name'])) log.info('Ignored, score "%s" to low, need at least "%s": %s', (rel['score'], quality_custom.get('minimum_score'), rel['name']))
continue continue
if rel['size'] <= 50: if rel['size'] <= 50:
log.info('Ignored, size "%sMB" to low: %s', (rel['size'], rel['name'])) log.info('Ignored, size "%sMB" to low: %s', (rel['size'], rel['name']))
continue continue
if 'seeders' in rel and rel.get('seeders') <= 0: if 'seeders' in rel and rel.get('seeders') < minimum_seeders:
log.info('Ignored, no seeders: %s', (rel['name'])) log.info('Ignored, not enough seeders, has %s needs %s: %s', (rel.get('seeders'), minimum_seeders, rel['name']))
continue continue
# If a single release comes through the "wait for", let through all # If a single release comes through the "wait for", let through all
@@ -424,7 +441,6 @@ class Release(Plugin):
for rel in search_results: for rel in search_results:
rel_identifier = md5(rel['url']) rel_identifier = md5(rel['url'])
found_releases.append(rel_identifier)
release = { release = {
'_t': 'release', '_t': 'release',
@@ -465,6 +481,9 @@ class Release(Plugin):
# Update release in search_results # Update release in search_results
rel['status'] = rls.get('status') rel['status'] = rls.get('status')
if rel['status'] == 'available':
found_releases.append(rel_identifier)
return found_releases return found_releases
except: except:
log.error('Failed: %s', traceback.format_exc()) log.error('Failed: %s', traceback.format_exc())
@@ -527,11 +546,15 @@ class Release(Plugin):
def forMedia(self, media_id): def forMedia(self, media_id):
db = get_db() db = get_db()
raw_releases = list(db.get_many('release', media_id, with_doc = True)) raw_releases = db.get_many('release', media_id)
releases = [] releases = []
for r in raw_releases: for r in raw_releases:
releases.append(r['doc']) try:
doc = db.get('id', r.get('_id'))
releases.append(doc)
except RecordDeleted:
pass
releases = sorted(releases, key = lambda k: k.get('info', {}).get('score', 0), reverse = True) releases = sorted(releases, key = lambda k: k.get('info', {}).get('score', 0), reverse = True)
+62 -17
View File
@@ -10,7 +10,8 @@ from couchpotato.api import addApiView
from couchpotato.core.event import addEvent, fireEvent, fireEventAsync from couchpotato.core.event import addEvent, fireEvent, fireEventAsync
from couchpotato.core.helpers.encoding import toUnicode, ss, sp from couchpotato.core.helpers.encoding import toUnicode, ss, sp
from couchpotato.core.helpers.variable import getExt, mergeDicts, getTitle, \ from couchpotato.core.helpers.variable import getExt, mergeDicts, getTitle, \
getImdb, link, symlink, tryInt, splitString, fnEscape, isSubFolder, getIdentifier getImdb, link, symlink, tryInt, splitString, fnEscape, isSubFolder, \
getIdentifier, randomString, getFreeSpace, getSize
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.base import Plugin
from couchpotato.environment import Env from couchpotato.environment import Env
@@ -219,6 +220,16 @@ class Renamer(Plugin):
nfo_name = self.conf('nfo_name') nfo_name = self.conf('nfo_name')
separator = self.conf('separator') separator = self.conf('separator')
if len(file_name) == 0:
log.error('Please fill in the filename option under renamer settings. Forcing it on <original>.<ext> to keep the same name as source file.')
file_name = '<original>.<ext>'
cd_keys = ['<cd>','<cd_nr>', '<original>']
if not any(x in folder_name for x in cd_keys) and not any(x in file_name for x in cd_keys):
log.error('Missing `cd` or `cd_nr` in the renamer. This will cause multi-file releases of being renamed to the same file. '
'Please add it in the renamer settings. Force adding it for now.')
file_name = '%s %s' % ('<cd>', file_name)
# Tag release folder as failed_rename in case no groups were found. This prevents check_snatched from removing the release from the downloader. # Tag release folder as failed_rename in case no groups were found. This prevents check_snatched from removing the release from the downloader.
if not groups and self.statusInfoComplete(release_download): if not groups and self.statusInfoComplete(release_download):
self.tagRelease(release_download = release_download, tag = 'failed_rename') self.tagRelease(release_download = release_download, tag = 'failed_rename')
@@ -266,13 +277,14 @@ class Renamer(Plugin):
category_label = category['label'] category_label = category['label']
if category['destination'] and len(category['destination']) > 0 and category['destination'] != 'None': if category['destination'] and len(category['destination']) > 0 and category['destination'] != 'None':
destination = category['destination'] destination = sp(category['destination'])
log.debug('Setting category destination for "%s": %s' % (media_title, destination)) log.debug('Setting category destination for "%s": %s' % (media_title, destination))
else: else:
log.debug('No category destination found for "%s"' % media_title) log.debug('No category destination found for "%s"' % media_title)
except: except:
log.error('Failed getting category label: %s', traceback.format_exc()) log.error('Failed getting category label: %s', traceback.format_exc())
# Find subtitle for renaming # Find subtitle for renaming
group['before_rename'] = [] group['before_rename'] = []
fireEvent('renamer.before', group) fireEvent('renamer.before', group)
@@ -344,6 +356,9 @@ class Renamer(Plugin):
replacements['original'] = os.path.splitext(os.path.basename(current_file))[0] replacements['original'] = os.path.splitext(os.path.basename(current_file))[0]
replacements['original_folder'] = fireEvent('scanner.remove_cptag', group['dirname'], single = True) replacements['original_folder'] = fireEvent('scanner.remove_cptag', group['dirname'], single = True)
if not replacements['original_folder'] or len(replacements['original_folder']) == 0:
replacements['original_folder'] = replacements['original']
# Extension # Extension
replacements['ext'] = getExt(current_file) replacements['ext'] = getExt(current_file)
@@ -362,10 +377,6 @@ class Renamer(Plugin):
elif file_type is 'nfo': elif file_type is 'nfo':
final_file_name = self.doReplace(nfo_name, replacements, remove_multiple = True) final_file_name = self.doReplace(nfo_name, replacements, remove_multiple = True)
# Seperator replace
if separator:
final_file_name = final_file_name.replace(' ', separator)
# Move DVD files (no structure renaming) # Move DVD files (no structure renaming)
if group['is_dvd'] and file_type is 'movie': if group['is_dvd'] and file_type is 'movie':
found = False found = False
@@ -526,7 +537,7 @@ class Renamer(Plugin):
# Remove leftover files # Remove leftover files
if not remove_leftovers: # Don't remove anything if not remove_leftovers: # Don't remove anything
break continue
log.debug('Removing leftover files') log.debug('Removing leftover files')
for current_file in group['files']['leftover']: for current_file in group['files']['leftover']:
@@ -534,6 +545,14 @@ class Renamer(Plugin):
(not keep_original or self.fileIsAdded(current_file, group)): (not keep_original or self.fileIsAdded(current_file, group)):
remove_files.append(current_file) remove_files.append(current_file)
if self.conf('check_space'):
total_space, available_space = getFreeSpace(destination)
renaming_size = getSize(rename_files.keys())
if renaming_size > available_space:
log.error('Not enough space left, need %s MB but only %s MB available', (renaming_size, available_space))
self.tagRelease(group = group, tag = 'not_enough_space')
continue
# Remove files # Remove files
delete_folders = [] delete_folders = []
for src in remove_files: for src in remove_files:
@@ -549,9 +568,9 @@ class Renamer(Plugin):
os.remove(src) os.remove(src)
parent_dir = os.path.dirname(src) parent_dir = os.path.dirname(src)
if delete_folders.count(parent_dir) == 0 and os.path.isdir(parent_dir) and \ if parent_dir not in delete_folders and os.path.isdir(parent_dir) and \
not isSubFolder(destination, parent_dir) and not isSubFolder(media_folder, parent_dir) and \ not isSubFolder(destination, parent_dir) and not isSubFolder(media_folder, parent_dir) and \
not isSubFolder(parent_dir, base_folder): isSubFolder(parent_dir, base_folder):
delete_folders.append(parent_dir) delete_folders.append(parent_dir)
@@ -560,6 +579,7 @@ class Renamer(Plugin):
self.tagRelease(group = group, tag = 'failed_remove') self.tagRelease(group = group, tag = 'failed_remove')
# Delete leftover folder from older releases # Delete leftover folder from older releases
delete_folders = sorted(delete_folders, key = len, reverse = True)
for delete_folder in delete_folders: for delete_folder in delete_folders:
try: try:
self.deleteEmptyFolder(delete_folder, show_error = False) self.deleteEmptyFolder(delete_folder, show_error = False)
@@ -572,7 +592,10 @@ class Renamer(Plugin):
for src in rename_files: for src in rename_files:
if rename_files[src]: if rename_files[src]:
dst = rename_files[src] dst = rename_files[src]
log.info('Renaming "%s" to "%s"', (src, dst))
if dst in group['renamed_files']:
log.error('File "%s" already renamed once, adding random string at the end to prevent data loss', dst)
dst = '%s.random-%s' % (dst, randomString())
# Create dir # Create dir
self.makeDir(os.path.dirname(dst)) self.makeDir(os.path.dirname(dst))
@@ -614,8 +637,9 @@ class Renamer(Plugin):
group_folder = sp(os.path.join(base_folder, os.path.relpath(group['parentdir'], base_folder).split(os.path.sep)[0])) group_folder = sp(os.path.join(base_folder, os.path.relpath(group['parentdir'], base_folder).split(os.path.sep)[0]))
try: try:
log.info('Deleting folder: %s', group_folder) if self.conf('cleanup') or self.conf('move_leftover'):
self.deleteEmptyFolder(group_folder) log.info('Deleting folder: %s', group_folder)
self.deleteEmptyFolder(group_folder)
except: except:
log.error('Failed removing %s: %s', (group_folder, traceback.format_exc())) log.error('Failed removing %s: %s', (group_folder, traceback.format_exc()))
@@ -771,22 +795,32 @@ Remove it if you want it to be renamed (again, or at least let it try again)
dest = sp(dest) dest = sp(dest)
try: try:
if os.path.exists(dest) and os.path.isfile(dest):
raise Exception('Destination "%s" already exists' % dest)
move_type = self.conf('file_action') move_type = self.conf('file_action')
if use_default: if use_default:
move_type = self.conf('default_file_action') move_type = self.conf('default_file_action')
if move_type not in ['copy', 'link']: if move_type not in ['copy', 'link']:
try: try:
log.info('Moving "%s" to "%s"', (old, dest))
shutil.move(old, dest) shutil.move(old, dest)
except: except:
if os.path.exists(dest): exists = os.path.exists(dest)
if exists and os.path.getsize(old) == os.path.getsize(dest):
log.error('Successfully moved file "%s", but something went wrong: %s', (dest, traceback.format_exc())) log.error('Successfully moved file "%s", but something went wrong: %s', (dest, traceback.format_exc()))
os.unlink(old) os.unlink(old)
else: else:
# remove faultly copied file
if exists:
os.unlink(dest)
raise raise
elif move_type == 'copy': elif move_type == 'copy':
log.info('Copying "%s" to "%s"', (old, dest))
shutil.copy(old, dest) shutil.copy(old, dest)
else: else:
log.info('Linking "%s" to "%s"', (old, dest))
# First try to hardlink # First try to hardlink
try: try:
log.debug('Hardlinking file "%s" to "%s"...', (old, dest)) log.debug('Hardlinking file "%s" to "%s"...', (old, dest))
@@ -796,9 +830,10 @@ Remove it if you want it to be renamed (again, or at least let it try again)
log.debug('Couldn\'t hardlink file "%s" to "%s". Symlinking instead. Error: %s.', (old, dest, traceback.format_exc())) log.debug('Couldn\'t hardlink file "%s" to "%s". Symlinking instead. Error: %s.', (old, dest, traceback.format_exc()))
shutil.copy(old, dest) shutil.copy(old, dest)
try: try:
symlink(dest, old + '.link') old_link = '%s.link' % sp(old)
symlink(dest, old_link)
os.unlink(old) os.unlink(old)
os.rename(old + '.link', old) os.rename(old_link, old)
except: except:
log.error('Couldn\'t symlink file "%s" to "%s". Copied instead. Error: %s. ', (old, dest, traceback.format_exc())) log.error('Couldn\'t symlink file "%s" to "%s". Copied instead. Error: %s. ', (old, dest, traceback.format_exc()))
@@ -841,7 +876,7 @@ Remove it if you want it to be renamed (again, or at least let it try again)
replaced = re.sub(r"[\x00:\*\?\"<>\|]", '', replaced) replaced = re.sub(r"[\x00:\*\?\"<>\|]", '', replaced)
sep = self.conf('foldersep') if folder else self.conf('separator') sep = self.conf('foldersep') if folder else self.conf('separator')
return replaced.replace(' ', ' ' if not sep else sep) return ss(replaced.replace(' ', ' ' if not sep else sep))
def replaceDoubles(self, string): def replaceDoubles(self, string):
@@ -854,6 +889,8 @@ Remove it if you want it to be renamed (again, or at least let it try again)
reg, replace_with = r reg, replace_with = r
string = re.sub(reg, replace_with, string) string = re.sub(reg, replace_with, string)
string = string.rstrip(',_-/\\ ')
return string return string
def checkSnatched(self, fire_scan = True): def checkSnatched(self, fire_scan = True):
@@ -1187,7 +1224,7 @@ Remove it if you want it to be renamed (again, or at least let it try again)
except Exception as e: except Exception as e:
log.error('Failed moving left over file %s to %s: %s %s', (leftoverfile, move_to, e, traceback.format_exc())) log.error('Failed moving left over file %s to %s: %s %s', (leftoverfile, move_to, e, traceback.format_exc()))
# As we probably tried to overwrite the nfo file, check if it exists and then remove the original # As we probably tried to overwrite the nfo file, check if it exists and then remove the original
if os.path.isfile(move_to): if os.path.isfile(move_to) and os.path.getsize(leftoverfile) == os.path.getsize(move_to):
if cleanup: if cleanup:
log.info('Deleting left over file %s instead...', leftoverfile) log.info('Deleting left over file %s instead...', leftoverfile)
os.unlink(leftoverfile) os.unlink(leftoverfile)
@@ -1357,6 +1394,14 @@ config = [{
'label': 'Folder-Separator', 'label': 'Folder-Separator',
'description': ('Replace all the spaces with a character.', 'Example: ".", "-" (without quotes). Leave empty to use spaces.'), 'description': ('Replace all the spaces with a character.', 'Example: ".", "-" (without quotes). Leave empty to use spaces.'),
}, },
{
'name': 'check_space',
'label': 'Check space',
'default': True,
'type': 'bool',
'description': ('Check if there\'s enough available space to rename the files', 'Disable when the filesystem doesn\'t return the proper value'),
'advanced': True,
},
{ {
'name': 'default_file_action', 'name': 'default_file_action',
'label': 'Default File Action', 'label': 'Default File Action',
+24 -8
View File
@@ -11,7 +11,6 @@ from couchpotato.core.helpers.variable import getExt, getImdb, tryInt, \
splitString, getIdentifier splitString, getIdentifier
from couchpotato.core.logger import CPLog from couchpotato.core.logger import CPLog
from couchpotato.core.plugins.base import Plugin from couchpotato.core.plugins.base import Plugin
from enzyme.exceptions import NoParserError, ParseError
from guessit import guess_movie_info from guessit import guess_movie_info
from subliminal.videos import Video from subliminal.videos import Video
import enzyme import enzyme
@@ -121,7 +120,7 @@ class Scanner(Plugin):
'()([ab])(\.....?)$' #*a.mkv '()([ab])(\.....?)$' #*a.mkv
] ]
cp_imdb = '(.cp.(?P<id>tt[0-9{7}]+).)' cp_imdb = '\.cp\((?P<id>tt[0-9]+),?\s?(?P<random>[A-Za-z0-9]+)?\)'
def __init__(self): def __init__(self):
@@ -132,7 +131,7 @@ class Scanner(Plugin):
addEvent('scanner.name_year', self.getReleaseNameYear) addEvent('scanner.name_year', self.getReleaseNameYear)
addEvent('scanner.partnumber', self.getPartNumber) addEvent('scanner.partnumber', self.getPartNumber)
def scan(self, folder = None, files = None, release_download = None, simple = False, newer_than = 0, return_ignored = True, on_found = None): def scan(self, folder = None, files = None, release_download = None, simple = False, newer_than = 0, return_ignored = True, check_file_date = True, on_found = None):
folder = sp(folder) folder = sp(folder)
@@ -146,7 +145,6 @@ class Scanner(Plugin):
# Scan all files of the folder if no files are set # Scan all files of the folder if no files are set
if not files: if not files:
check_file_date = True
try: try:
files = [] files = []
for root, dirs, walk_files in os.walk(folder, followlinks=True): for root, dirs, walk_files in os.walk(folder, followlinks=True):
@@ -457,6 +455,7 @@ class Scanner(Plugin):
meta = self.getMeta(cur_file) meta = self.getMeta(cur_file)
try: try:
data['titles'] = meta.get('titles', [])
data['video'] = meta.get('video', self.getCodec(cur_file, self.codecs['video'])) data['video'] = meta.get('video', self.getCodec(cur_file, self.codecs['video']))
data['audio'] = meta.get('audio', self.getCodec(cur_file, self.codecs['audio'])) data['audio'] = meta.get('audio', self.getCodec(cur_file, self.codecs['audio']))
data['audio_channels'] = meta.get('audio_channels', 2.0) data['audio_channels'] = meta.get('audio_channels', 2.0)
@@ -492,7 +491,7 @@ class Scanner(Plugin):
data['quality_type'] = 'HD' if data.get('resolution_width', 0) >= 1280 or data['quality'].get('hd') else 'SD' data['quality_type'] = 'HD' if data.get('resolution_width', 0) >= 1280 or data['quality'].get('hd') else 'SD'
filename = re.sub('(.cp\(tt[0-9{7}]+\))', '', files[0]) filename = re.sub(self.cp_imdb, '', files[0])
data['group'] = self.getGroup(filename[len(folder):]) data['group'] = self.getGroup(filename[len(folder):])
data['source'] = self.getSourceMedia(filename) data['source'] = self.getSourceMedia(filename)
if data['quality'].get('is_3d', 0): if data['quality'].get('is_3d', 0):
@@ -527,16 +526,33 @@ class Scanner(Plugin):
try: ac = self.audio_codec_map.get(p.audio[0].codec) try: ac = self.audio_codec_map.get(p.audio[0].codec)
except: pass except: pass
# Find title in video headers
titles = []
try:
if p.title and self.findYear(p.title):
titles.append(ss(p.title))
except:
log.error('Failed getting title from meta: %s', traceback.format_exc())
for video in p.video:
try:
if video.title and self.findYear(video.title):
titles.append(ss(video.title))
except:
log.error('Failed getting title from meta: %s', traceback.format_exc())
return { return {
'titles': list(set(titles)),
'video': vc, 'video': vc,
'audio': ac, 'audio': ac,
'resolution_width': tryInt(p.video[0].width), 'resolution_width': tryInt(p.video[0].width),
'resolution_height': tryInt(p.video[0].height), 'resolution_height': tryInt(p.video[0].height),
'audio_channels': p.audio[0].channels, 'audio_channels': p.audio[0].channels,
} }
except ParseError: except enzyme.exceptions.ParseError:
log.debug('Failed to parse meta for %s', filename) log.debug('Failed to parse meta for %s', filename)
except NoParserError: except enzyme.exceptions.NoParserError:
log.debug('No parser found for %s', filename) log.debug('No parser found for %s', filename)
except: except:
log.debug('Failed parsing %s', filename) log.debug('Failed parsing %s', filename)
@@ -677,7 +693,7 @@ class Scanner(Plugin):
def removeCPTag(self, name): def removeCPTag(self, name):
try: try:
return re.sub(self.cp_imdb, '', name) return re.sub(self.cp_imdb, '', name).strip()
except: except:
pass pass
return name return name
+63 -38
View File
@@ -33,33 +33,43 @@ name_scores = [
def nameScore(name, year, preferred_words): def nameScore(name, year, preferred_words):
""" Calculate score for words in the NZB name """ """ Calculate score for words in the NZB name """
score = 0 try:
name = name.lower() score = 0
name = name.lower()
# give points for the cool stuff # give points for the cool stuff
for value in name_scores: for value in name_scores:
v = value.split(':') v = value.split(':')
add = int(v.pop()) add = int(v.pop())
if v.pop() in name: if v.pop() in name:
score += add score += add
# points if the year is correct # points if the year is correct
if year and str(year) in name: if str(year) in name:
score += 5 score += 5
# Contains preferred word # Contains preferred word
nzb_words = re.split('\W+', simplifyString(name)) nzb_words = re.split('\W+', simplifyString(name))
score += 100 * len(list(set(nzb_words) & set(preferred_words))) score += 100 * len(list(set(nzb_words) & set(preferred_words)))
return score return score
except:
log.error('Failed doing nameScore: %s', traceback.format_exc())
return 0
def nameRatioScore(nzb_name, movie_name): def nameRatioScore(nzb_name, movie_name):
nzb_words = re.split('\W+', fireEvent('scanner.create_file_identifier', nzb_name, single = True)) try:
movie_words = re.split('\W+', simplifyString(movie_name)) nzb_words = re.split('\W+', fireEvent('scanner.create_file_identifier', nzb_name, single = True))
movie_words = re.split('\W+', simplifyString(movie_name))
left_over = set(nzb_words) - set(movie_words) left_over = set(nzb_words) - set(movie_words)
return 10 - len(left_over) return 10 - len(left_over)
except:
log.error('Failed doing nameRatioScore: %s', traceback.format_exc())
return 0
def namePositionScore(nzb_name, movie_name): def namePositionScore(nzb_name, movie_name):
@@ -134,38 +144,53 @@ def providerScore(provider):
def duplicateScore(nzb_name, movie_name): def duplicateScore(nzb_name, movie_name):
nzb_words = re.split('\W+', simplifyString(nzb_name)) try:
movie_words = re.split('\W+', simplifyString(movie_name)) nzb_words = re.split('\W+', simplifyString(nzb_name))
movie_words = re.split('\W+', simplifyString(movie_name))
# minus for duplicates # minus for duplicates
duplicates = [x for i, x in enumerate(nzb_words) if nzb_words[i:].count(x) > 1] duplicates = [x for i, x in enumerate(nzb_words) if nzb_words[i:].count(x) > 1]
return len(list(set(duplicates) - set(movie_words))) * -4 return len(list(set(duplicates) - set(movie_words))) * -4
except:
log.error('Failed doing duplicateScore: %s', traceback.format_exc())
return 0
def partialIgnoredScore(nzb_name, movie_name, ignored_words): def partialIgnoredScore(nzb_name, movie_name, ignored_words):
nzb_name = nzb_name.lower() try:
movie_name = movie_name.lower() nzb_name = nzb_name.lower()
movie_name = movie_name.lower()
score = 0 score = 0
for ignored_word in ignored_words: for ignored_word in ignored_words:
if ignored_word in nzb_name and ignored_word not in movie_name: if ignored_word in nzb_name and ignored_word not in movie_name:
score -= 5 score -= 5
return score return score
except:
log.error('Failed doing partialIgnoredScore: %s', traceback.format_exc())
return 0
def halfMultipartScore(nzb_name): def halfMultipartScore(nzb_name):
wrong_found = 0 try:
for nr in [1, 2, 3, 4, 5, 'i', 'ii', 'iii', 'iv', 'v', 'a', 'b', 'c', 'd', 'e']: wrong_found = 0
for wrong in ['cd', 'part', 'dis', 'disc', 'dvd']: for nr in [1, 2, 3, 4, 5, 'i', 'ii', 'iii', 'iv', 'v', 'a', 'b', 'c', 'd', 'e']:
if '%s%s' % (wrong, nr) in nzb_name.lower(): for wrong in ['cd', 'part', 'dis', 'disc', 'dvd']:
wrong_found += 1 if '%s%s' % (wrong, nr) in nzb_name.lower():
wrong_found += 1
if wrong_found == 1: if wrong_found == 1:
return -30 return -30
return 0
except:
log.error('Failed doing halfMultipartScore: %s', traceback.format_exc())
return 0 return 0
+22 -8
View File
@@ -9,6 +9,7 @@ import traceback
import warnings import warnings
import re import re
import tarfile import tarfile
import shutil
from CodernityDB.database_super_thread_safe import SuperThreadSafeDatabase from CodernityDB.database_super_thread_safe import SuperThreadSafeDatabase
from argparse import ArgumentParser from argparse import ArgumentParser
@@ -19,6 +20,7 @@ from couchpotato.core.event import fireEventAsync, fireEvent
from couchpotato.core.helpers.encoding import sp from couchpotato.core.helpers.encoding import sp
from couchpotato.core.helpers.variable import getDataDir, tryInt, getFreeSpace from couchpotato.core.helpers.variable import getDataDir, tryInt, getFreeSpace
import requests import requests
from requests.packages.urllib3 import disable_warnings
from tornado.httpserver import HTTPServer from tornado.httpserver import HTTPServer
from tornado.web import Application, StaticFileHandler, RedirectHandler from tornado.web import Application, StaticFileHandler, RedirectHandler
@@ -107,14 +109,20 @@ def runCouchPotato(options, base_path, args, data_dir = None, log_dir = None, En
if not os.path.isdir(backup_path): os.makedirs(backup_path) if not os.path.isdir(backup_path): os.makedirs(backup_path)
for root, dirs, files in os.walk(backup_path): for root, dirs, files in os.walk(backup_path):
for backup_file in sorted(files): # Only consider files being a direct child of the backup_path
ints = re.findall('\d+', backup_file) if root == backup_path:
for backup_file in sorted(files):
ints = re.findall('\d+', backup_file)
# Delete non zip files # Delete non zip files
if len(ints) != 1: if len(ints) != 1:
os.remove(os.path.join(backup_path, backup_file)) try: os.remove(os.path.join(root, backup_file))
else: except: pass
existing_backups.append((int(ints[0]), backup_file)) else:
existing_backups.append((int(ints[0]), backup_file))
else:
# Delete stray directories.
shutil.rmtree(root)
# Remove all but the last 5 # Remove all but the last 5
for eb in existing_backups[:-backup_count]: for eb in existing_backups[:-backup_count]:
@@ -144,12 +152,15 @@ def runCouchPotato(options, base_path, args, data_dir = None, log_dir = None, En
if not os.path.exists(python_cache): if not os.path.exists(python_cache):
os.mkdir(python_cache) os.mkdir(python_cache)
session = requests.Session()
session.max_redirects = 5
# Register environment settings # Register environment settings
Env.set('app_dir', sp(base_path)) Env.set('app_dir', sp(base_path))
Env.set('data_dir', sp(data_dir)) Env.set('data_dir', sp(data_dir))
Env.set('log_path', sp(os.path.join(log_dir, 'CouchPotato.log'))) Env.set('log_path', sp(os.path.join(log_dir, 'CouchPotato.log')))
Env.set('db', db) Env.set('db', db)
Env.set('http_opener', requests.Session()) Env.set('http_opener', session)
Env.set('cache_dir', cache_dir) Env.set('cache_dir', cache_dir)
Env.set('cache', FileSystemCache(python_cache)) Env.set('cache', FileSystemCache(python_cache))
Env.set('console_log', options.console_log) Env.set('console_log', options.console_log)
@@ -174,6 +185,9 @@ def runCouchPotato(options, base_path, args, data_dir = None, log_dir = None, En
for logger_name in ['gntp']: for logger_name in ['gntp']:
logging.getLogger(logger_name).setLevel(logging.WARNING) logging.getLogger(logger_name).setLevel(logging.WARNING)
# Disable SSL warning
disable_warnings()
# Use reloader # Use reloader
reloader = debug is True and development and not Env.get('desktop') and not options.daemon reloader = debug is True and development and not Env.get('desktop') and not options.daemon
+11 -4
View File
@@ -54,16 +54,22 @@
}, },
pushState: function(e){ pushState: function(e){
if((!e.meta && Browser.platform.mac) || (!e.control && !Browser.platform.mac)){ var self = this;
if((!e.meta && self.isMac()) || (!e.control && !self.isMac())){
(e).preventDefault(); (e).preventDefault();
var url = e.target.get('href'); var url = e.target.get('href');
if(History.getPath() != url)
// Middle click
if(e.event && e.event.button == 1)
window.open(url);
else if(History.getPath() != url)
History.push(url); History.push(url);
} }
}, },
isMac: function(){ isMac: function(){
return Browser.platform.mac return Browser.platform == 'mac'
}, },
createLayout: function(){ createLayout: function(){
@@ -325,11 +331,12 @@
}, },
openDerefered: function(e, el){ openDerefered: function(e, el){
var self = this;
(e).stop(); (e).stop();
var url = 'http://www.dereferer.org/?' + el.get('href'); var url = 'http://www.dereferer.org/?' + el.get('href');
if(el.get('target') == '_blank' || (e.meta && Browser.platform.mac) || (e.control && !Browser.platform.mac)) if(el.get('target') == '_blank' || (e.meta && self.isMac()) || (e.control && !self.isMac()))
window.open(url); window.open(url);
else else
window.location = url; window.location = url;
+35 -47
View File
@@ -146,13 +146,13 @@ Page.Home = new Class({
var self = this; var self = this;
// Suggest // Suggest
self.suggestion_list = new SuggestList({ self.suggestions_list = new SuggestList({
'onLoaded': function(){ 'onCreated': function(){
self.chain.callChain(); self.chain.callChain();
} }
}); });
$(self.suggestion_list).inject(self.el); $(self.suggestions_list).inject(self.el);
}, },
@@ -160,46 +160,38 @@ Page.Home = new Class({
var self = this; var self = this;
// Charts // Charts
self.charts = new Charts({ self.charts_list = new Charts({
'onCreated': function(){ 'onCreated': function(){
self.chain.callChain(); self.chain.callChain();
} }
}); });
$(self.charts).inject(self.el); $(self.charts_list).inject(self.el);
}, },
createSuggestionsChartsMenu: function(){ createSuggestionsChartsMenu: function(){
var self = this; var self = this,
suggestion_tab, charts_tab;
self.el_toggle_menu_suggestions = new Element('a.toggle_suggestions.active', { self.el_toggle_menu = new Element('div.toggle_menu', {
'href': '#', 'events': {
'events': { 'click': function(e) { 'click:relay(a)': function(e, el) {
e.preventDefault(); e.preventDefault();
self.toggleSuggestionsCharts('suggestions'); self.toggleSuggestionsCharts(el.get('data-container'), el);
} }
} }
}).grab( new Element('h2', {'text': 'Suggestions'})); }).adopt(
suggestion_tab = new Element('a.toggle_suggestions', {
'data-container': 'suggestions'
}).grab(new Element('h2', {'text': 'Suggestions'})),
charts_tab = new Element('a.toggle_charts', {
'data-container': 'charts'
}).grab( new Element('h2', {'text': 'Charts'}))
);
self.el_toggle_menu_charts = new Element('a.toggle_charts', { var menu_selected = Cookie.read('suggestions_charts_menu_selected') || 'suggestions';
'href': '#', self.toggleSuggestionsCharts(menu_selected, menu_selected == 'suggestions' ? suggestion_tab : charts_tab);
'events': { 'click': function(e) {
e.preventDefault();
self.toggleSuggestionsCharts('charts');
}
}
}).grab( new Element('h2', {'text': 'Charts'}));
self.el_toggle_menu = new Element('div.toggle_menu').grab(
self.el_toggle_menu_suggestions
).grab(
self.el_toggle_menu_charts
);
var menu_selected = Cookie.read('suggestions_charts_menu_selected');
if( menu_selected === null ) menu_selected = 'suggestions';
self.toggleSuggestionsCharts( menu_selected );
self.el_toggle_menu.inject(self.el); self.el_toggle_menu.inject(self.el);
@@ -207,23 +199,19 @@ Page.Home = new Class({
}, },
toggleSuggestionsCharts: function(menu_id){ toggleSuggestionsCharts: function(menu_id, el){
var self = this; var self = this;
switch(menu_id) { // Toggle ta
case 'suggestions': self.el_toggle_menu.getElements('.active').removeClass('active');
if($(self.suggestion_list)) $(self.suggestion_list).show(); if(el) el.addClass('active');
self.el_toggle_menu_suggestions.addClass('active');
if($(self.charts)) $(self.charts).hide(); // Hide both
self.el_toggle_menu_charts.removeClass('active'); if(self.suggestions_list) self.suggestions_list.hide();
break; if(self.charts_list) self.charts_list.hide();
case 'charts':
if($(self.charts)) $(self.charts).show(); var toggle_to = self[menu_id + '_list'];
self.el_toggle_menu_charts.addClass('active'); if(toggle_to) toggle_to.show();
if($(self.suggestion_list)) $(self.suggestion_list).hide();
self.el_toggle_menu_suggestions.removeClass('active');
break;
}
Cookie.write('suggestions_charts_menu_selected', menu_id, {'duration': 365}); Cookie.write('suggestions_charts_menu_selected', menu_id, {'duration': 365});
}, },
+112 -24
View File
@@ -560,11 +560,19 @@ Option.Password = new Class({
create: function(){ create: function(){
var self = this; var self = this;
self.parent(); self.el.adopt(
self.input.set('type', 'password'); self.createLabel(),
self.input = new Element('input.inlay', {
'type': 'text',
'name': self.postName(),
'value': self.getSettingValue() ? '********' : '',
'placeholder': self.getPlaceholder()
})
);
self.input.addEvent('focus', function(){ self.input.addEvent('focus', function(){
self.input.set('value', '') self.input.set('value', '');
self.input.set('type', 'password');
}) })
} }
@@ -634,6 +642,7 @@ Option.Directory = new Class({
browser: null, browser: null,
save_on_change: false, save_on_change: false,
use_cache: false, use_cache: false,
current_dir: '',
create: function(){ create: function(){
var self = this; var self = this;
@@ -645,8 +654,17 @@ Option.Directory = new Class({
'click': self.showBrowser.bind(self) 'click': self.showBrowser.bind(self)
} }
}).adopt( }).adopt(
self.input = new Element('span', { self.input = new Element('input', {
'text': self.getSettingValue() 'value': self.getSettingValue(),
'events': {
'change': self.filterDirectory.bind(self),
'keydown': function(e){
if(e.key == 'enter' || e.key == 'tab')
(e).stop();
},
'keyup': self.filterDirectory.bind(self),
'paste': self.filterDirectory.bind(self)
}
}) })
) )
); );
@@ -654,10 +672,55 @@ Option.Directory = new Class({
self.cached = {}; self.cached = {};
}, },
filterDirectory: function(e){
var self = this,
value = self.getValue(),
path_sep = Api.getOption('path_sep'),
active_selector = 'li:not(.blur):not(.empty)';
if(e.key == 'enter' || e.key == 'tab'){
(e).stop();
var first = self.dir_list.getElement(active_selector);
if(first){
self.selectDirectory(first.get('data-value'));
}
}
else {
// New folder
if(value.substr(-1) == path_sep){
if(self.current_dir != value)
self.selectDirectory(value)
}
else {
var pd = self.getParentDir(value);
if(self.current_dir != pd)
self.getDirs(pd);
var folder_filter = value.split(path_sep).getLast()
self.dir_list.getElements('li').each(function(li){
var valid = li.get('text').substr(0, folder_filter.length).toLowerCase() != folder_filter.toLowerCase()
li[valid ? 'addClass' : 'removeClass']('blur')
});
var first = self.dir_list.getElement(active_selector);
if(first){
if(!self.dir_list_scroll)
self.dir_list_scroll = new Fx.Scroll(self.dir_list, {
'transition': 'quint:in:out'
});
self.dir_list_scroll.toElement(first);
}
}
}
},
selectDirectory: function(dir){ selectDirectory: function(dir){
var self = this; var self = this;
self.input.set('text', dir); self.input.set('value', dir);
self.getDirs() self.getDirs()
}, },
@@ -668,9 +731,28 @@ Option.Directory = new Class({
self.selectDirectory(self.getParentDir()) self.selectDirectory(self.getParentDir())
}, },
caretAtEnd: function(){
var self = this;
self.input.focus();
if (typeof self.input.selectionStart == "number") {
self.input.selectionStart = self.input.selectionEnd = self.input.get('value').length;
} else if (typeof el.createTextRange != "undefined") {
self.input.focus();
var range = self.input.createTextRange();
range.collapse(false);
range.select();
}
},
showBrowser: function(){ showBrowser: function(){
var self = this; var self = this;
// Move caret to back of the input
if(!self.browser || self.browser && !self.browser.isVisible())
self.caretAtEnd()
if(!self.browser){ if(!self.browser){
self.browser = new Element('div.directory_list').adopt( self.browser = new Element('div.directory_list').adopt(
new Element('div.pointer'), new Element('div.pointer'),
@@ -686,7 +768,9 @@ Option.Directory = new Class({
}).adopt( }).adopt(
self.show_hidden = new Element('input[type=checkbox].inlay', { self.show_hidden = new Element('input[type=checkbox].inlay', {
'events': { 'events': {
'change': self.getDirs.bind(self) 'change': function(){
self.getDirs()
}
} }
}) })
) )
@@ -707,7 +791,7 @@ Option.Directory = new Class({
'text': 'Clear', 'text': 'Clear',
'events': { 'events': {
'click': function(e){ 'click': function(e){
self.input.set('text', ''); self.input.set('value', '');
self.hideBrowser(e, true); self.hideBrowser(e, true);
} }
} }
@@ -735,7 +819,7 @@ Option.Directory = new Class({
new Form.Check(self.show_hidden); new Form.Check(self.show_hidden);
} }
self.initial_directory = self.input.get('text'); self.initial_directory = self.input.get('value');
self.getDirs(); self.getDirs();
self.browser.show(); self.browser.show();
@@ -749,7 +833,7 @@ Option.Directory = new Class({
if(save) if(save)
self.save(); self.save();
else else
self.input.set('text', self.initial_directory); self.input.set('value', self.initial_directory);
self.browser.hide(); self.browser.hide();
self.el.removeEvents('outerClick') self.el.removeEvents('outerClick')
@@ -757,21 +841,21 @@ Option.Directory = new Class({
}, },
fillBrowser: function(json){ fillBrowser: function(json){
var self = this; var self = this,
v = self.getValue();
self.data = json; self.data = json;
var v = self.getValue(); var previous_dir = json.parent;
var previous_dir = self.getParentDir();
if(v == '') if(v == '')
self.input.set('text', json.home); self.input.set('value', json.home);
if(previous_dir != v && previous_dir.length >= 1 && !json.is_root){ if(previous_dir.length >= 1 && !json.is_root){
var prev_dirname = self.getCurrentDirname(previous_dir); var prev_dirname = self.getCurrentDirname(previous_dir);
if(previous_dir == json.home) if(previous_dir == json.home)
prev_dirname = 'Home'; prev_dirname = 'Home Folder';
else if(previous_dir == '/' && json.platform == 'nt') else if(previous_dir == '/' && json.platform == 'nt')
prev_dirname = 'Computer'; prev_dirname = 'Computer';
@@ -801,12 +885,13 @@ Option.Directory = new Class({
new Element('li.empty', { new Element('li.empty', {
'text': 'Selected folder is empty' 'text': 'Selected folder is empty'
}).inject(self.dir_list) }).inject(self.dir_list)
self.caretAtEnd();
}, },
getDirs: function(){ getDirs: function(dir){
var self = this; var self = this,
c = dir || self.getValue();
var c = self.getValue();
if(self.cached[c] && self.use_cache){ if(self.cached[c] && self.use_cache){
self.fillBrowser() self.fillBrowser()
@@ -817,7 +902,10 @@ Option.Directory = new Class({
'path': c, 'path': c,
'show_hidden': +self.show_hidden.checked 'show_hidden': +self.show_hidden.checked
}, },
'onComplete': self.fillBrowser.bind(self) 'onComplete': function(json){
self.current_dir = c;
self.fillBrowser(json);
}
}) })
} }
}, },
@@ -831,8 +919,8 @@ Option.Directory = new Class({
var v = dir || self.getValue(); var v = dir || self.getValue();
var sep = Api.getOption('path_sep'); var sep = Api.getOption('path_sep');
var dirs = v.split(sep); var dirs = v.split(sep);
if(dirs.pop() == '') if(dirs.pop() == '')
dirs.pop(); dirs.pop();
return dirs.join(sep) + sep return dirs.join(sep) + sep
}, },
@@ -845,7 +933,7 @@ Option.Directory = new Class({
getValue: function(){ getValue: function(){
var self = this; var self = this;
return self.input.get('text'); return self.input.get('value');
} }
}); });
+11 -2
View File
@@ -302,15 +302,19 @@
font-family: 'Elusive-Icons'; font-family: 'Elusive-Icons';
color: #f5e39c; color: #f5e39c;
} }
.page form .directory > span { .page form .directory > input {
height: 25px; height: 25px;
display: inline-block; display: inline-block;
float: right; float: right;
text-align: right; text-align: right;
white-space: nowrap; white-space: nowrap;
cursor: pointer; cursor: pointer;
background: none;
border: 0;
color: #FFF;
width: 100%;
} }
.page form .directory span:empty:before { .page form .directory input:empty:before {
content: 'No folder selected'; content: 'No folder selected';
font-style: italic; font-style: italic;
opacity: .3; opacity: .3;
@@ -353,6 +357,11 @@
white-space: nowrap; white-space: nowrap;
text-overflow: ellipsis; text-overflow: ellipsis;
} }
.page .directory_list li.blur {
opacity: .3;
}
.page .directory_list li:last-child { .page .directory_list li:last-child {
border-bottom: 1px solid rgba(255,255,255,0.1); border-bottom: 1px solid rgba(255,255,255,0.1);
} }
+3 -3
View File
@@ -46,7 +46,7 @@ DESC=CouchPotato
# Run CP as username # Run CP as username
RUN_AS=${CP_USER-couchpotato} RUN_AS=${CP_USER-couchpotato}
# Path to app # Path to app
# CP_HOME=path_to_app_CouchPotato.py # CP_HOME=path_to_app_CouchPotato.py
APP_PATH=${CP_HOME-/opt/couchpotato/} APP_PATH=${CP_HOME-/opt/couchpotato/}
@@ -100,12 +100,12 @@ case "$1" in
;; ;;
stop) stop)
echo "Stopping $DESC" echo "Stopping $DESC"
start-stop-daemon --stop --pidfile $PID_FILE --retry 15 start-stop-daemon --stop --pidfile $PID_FILE --retry 15 --oknodo
;; ;;
restart|force-reload) restart|force-reload)
echo "Restarting $DESC" echo "Restarting $DESC"
start-stop-daemon --stop --pidfile $PID_FILE --retry 15 start-stop-daemon --stop --pidfile $PID_FILE --retry 15 --oknodo
start-stop-daemon -d $APP_PATH -c $RUN_AS $EXTRA_SSD_OPTS --start --pidfile $PID_FILE --exec $DAEMON -- $DAEMON_OPTS start-stop-daemon -d $APP_PATH -c $RUN_AS $EXTRA_SSD_OPTS --start --pidfile $PID_FILE --exec $DAEMON -- $DAEMON_OPTS
;; ;;
+3 -3
View File
@@ -235,12 +235,12 @@ class Event(object):
self.error_handler(sys.exc_info()) self.error_handler(sys.exc_info())
finally: finally:
if not self.asynchronous:
self.queue.task_done()
if order_lock: if order_lock:
order_lock.release() order_lock.release()
if not self.asynchronous:
self.queue.task_done()
if self.queue.empty(): if self.queue.empty():
raise Empty raise Empty
+10 -4
View File
@@ -3,22 +3,28 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
__version__ = "1.0.1" __version__ = "2.2.1"
from sys import version_info
def detect(aBuf): def detect(aBuf):
import universaldetector if ((version_info < (3, 0) and isinstance(aBuf, unicode)) or
(version_info >= (3, 0) and not isinstance(aBuf, bytes))):
raise ValueError('Expected a bytes object, not a unicode object')
from . import universaldetector
u = universaldetector.UniversalDetector() u = universaldetector.UniversalDetector()
u.reset() u.reset()
u.feed(aBuf) u.feed(aBuf)
+11 -9
View File
@@ -1,11 +1,11 @@
######################## BEGIN LICENSE BLOCK ######################## ######################## BEGIN LICENSE BLOCK ########################
# The Original Code is Mozilla Communicator client code. # The Original Code is Mozilla Communicator client code.
# #
# The Initial Developer of the Original Code is # The Initial Developer of the Original Code is
# Netscape Communications Corporation. # Netscape Communications Corporation.
# Portions created by the Initial Developer are Copyright (C) 1998 # Portions created by the Initial Developer are Copyright (C) 1998
# the Initial Developer. All Rights Reserved. # the Initial Developer. All Rights Reserved.
# #
# Contributor(s): # Contributor(s):
# Mark Pilgrim - port to Python # Mark Pilgrim - port to Python
# #
@@ -13,12 +13,12 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
@@ -26,18 +26,18 @@
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
# Big5 frequency table # Big5 frequency table
# by Taiwan's Mandarin Promotion Council # by Taiwan's Mandarin Promotion Council
# <http://www.edu.tw:81/mandr/> # <http://www.edu.tw:81/mandr/>
# #
# 128 --> 0.42261 # 128 --> 0.42261
# 256 --> 0.57851 # 256 --> 0.57851
# 512 --> 0.74851 # 512 --> 0.74851
# 1024 --> 0.89384 # 1024 --> 0.89384
# 2048 --> 0.97583 # 2048 --> 0.97583
# #
# Ideal Distribution Ratio = 0.74851/(1-0.74851) =2.98 # Ideal Distribution Ratio = 0.74851/(1-0.74851) =2.98
# Random Distribution Ration = 512/(5401-512)=0.105 # Random Distribution Ration = 512/(5401-512)=0.105
# #
# Typical Distribution Ratio about 25% of Ideal one, still much higher than RDR # Typical Distribution Ratio about 25% of Ideal one, still much higher than RDR
BIG5_TYPICAL_DISTRIBUTION_RATIO = 0.75 BIG5_TYPICAL_DISTRIBUTION_RATIO = 0.75
@@ -45,7 +45,7 @@ BIG5_TYPICAL_DISTRIBUTION_RATIO = 0.75
#Char to FreqOrder table #Char to FreqOrder table
BIG5_TABLE_SIZE = 5376 BIG5_TABLE_SIZE = 5376
Big5CharToFreqOrder = ( \ Big5CharToFreqOrder = (
1,1801,1506, 255,1431, 198, 9, 82, 6,5008, 177, 202,3681,1256,2821, 110, # 16 1,1801,1506, 255,1431, 198, 9, 82, 6,5008, 177, 202,3681,1256,2821, 110, # 16
3814, 33,3274, 261, 76, 44,2114, 16,2946,2187,1176, 659,3971, 26,3451,2653, # 32 3814, 33,3274, 261, 76, 44,2114, 16,2946,2187,1176, 659,3971, 26,3451,2653, # 32
1198,3972,3350,4202, 410,2215, 302, 590, 361,1964, 8, 204, 58,4510,5009,1932, # 48 1198,3972,3350,4202, 410,2215, 302, 590, 361,1964, 8, 204, 58,4510,5009,1932, # 48
@@ -921,3 +921,5 @@ Big5CharToFreqOrder = ( \
13936,13937,13938,13939,13940,13941,13942,13943,13944,13945,13946,13947,13948,13949,13950,13951, #13952 13936,13937,13938,13939,13940,13941,13942,13943,13944,13945,13946,13947,13948,13949,13950,13951, #13952
13952,13953,13954,13955,13956,13957,13958,13959,13960,13961,13962,13963,13964,13965,13966,13967, #13968 13952,13953,13954,13955,13956,13957,13958,13959,13960,13961,13962,13963,13964,13965,13966,13967, #13968
13968,13969,13970,13971,13972) #13973 13968,13969,13970,13971,13972) #13973
# flake8: noqa
+9 -8
View File
@@ -1,11 +1,11 @@
######################## BEGIN LICENSE BLOCK ######################## ######################## BEGIN LICENSE BLOCK ########################
# The Original Code is Mozilla Communicator client code. # The Original Code is Mozilla Communicator client code.
# #
# The Initial Developer of the Original Code is # The Initial Developer of the Original Code is
# Netscape Communications Corporation. # Netscape Communications Corporation.
# Portions created by the Initial Developer are Copyright (C) 1998 # Portions created by the Initial Developer are Copyright (C) 1998
# the Initial Developer. All Rights Reserved. # the Initial Developer. All Rights Reserved.
# #
# Contributor(s): # Contributor(s):
# Mark Pilgrim - port to Python # Mark Pilgrim - port to Python
# #
@@ -13,22 +13,23 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
from mbcharsetprober import MultiByteCharSetProber from .mbcharsetprober import MultiByteCharSetProber
from codingstatemachine import CodingStateMachine from .codingstatemachine import CodingStateMachine
from chardistribution import Big5DistributionAnalysis from .chardistribution import Big5DistributionAnalysis
from mbcssm import Big5SMModel from .mbcssm import Big5SMModel
class Big5Prober(MultiByteCharSetProber): class Big5Prober(MultiByteCharSetProber):
def __init__(self): def __init__(self):
+94 -63
View File
@@ -1,11 +1,11 @@
######################## BEGIN LICENSE BLOCK ######################## ######################## BEGIN LICENSE BLOCK ########################
# The Original Code is Mozilla Communicator client code. # The Original Code is Mozilla Communicator client code.
# #
# The Initial Developer of the Original Code is # The Initial Developer of the Original Code is
# Netscape Communications Corporation. # Netscape Communications Corporation.
# Portions created by the Initial Developer are Copyright (C) 1998 # Portions created by the Initial Developer are Copyright (C) 1998
# the Initial Developer. All Rights Reserved. # the Initial Developer. All Rights Reserved.
# #
# Contributor(s): # Contributor(s):
# Mark Pilgrim - port to Python # Mark Pilgrim - port to Python
# #
@@ -13,47 +13,63 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants from .euctwfreq import (EUCTWCharToFreqOrder, EUCTW_TABLE_SIZE,
from euctwfreq import EUCTWCharToFreqOrder, EUCTW_TABLE_SIZE, EUCTW_TYPICAL_DISTRIBUTION_RATIO EUCTW_TYPICAL_DISTRIBUTION_RATIO)
from euckrfreq import EUCKRCharToFreqOrder, EUCKR_TABLE_SIZE, EUCKR_TYPICAL_DISTRIBUTION_RATIO from .euckrfreq import (EUCKRCharToFreqOrder, EUCKR_TABLE_SIZE,
from gb2312freq import GB2312CharToFreqOrder, GB2312_TABLE_SIZE, GB2312_TYPICAL_DISTRIBUTION_RATIO EUCKR_TYPICAL_DISTRIBUTION_RATIO)
from big5freq import Big5CharToFreqOrder, BIG5_TABLE_SIZE, BIG5_TYPICAL_DISTRIBUTION_RATIO from .gb2312freq import (GB2312CharToFreqOrder, GB2312_TABLE_SIZE,
from jisfreq import JISCharToFreqOrder, JIS_TABLE_SIZE, JIS_TYPICAL_DISTRIBUTION_RATIO GB2312_TYPICAL_DISTRIBUTION_RATIO)
from .big5freq import (Big5CharToFreqOrder, BIG5_TABLE_SIZE,
BIG5_TYPICAL_DISTRIBUTION_RATIO)
from .jisfreq import (JISCharToFreqOrder, JIS_TABLE_SIZE,
JIS_TYPICAL_DISTRIBUTION_RATIO)
from .compat import wrap_ord
ENOUGH_DATA_THRESHOLD = 1024 ENOUGH_DATA_THRESHOLD = 1024
SURE_YES = 0.99 SURE_YES = 0.99
SURE_NO = 0.01 SURE_NO = 0.01
MINIMUM_DATA_THRESHOLD = 3
class CharDistributionAnalysis: class CharDistributionAnalysis:
def __init__(self): def __init__(self):
self._mCharToFreqOrder = None # Mapping table to get frequency order from char order (get from GetOrder()) # Mapping table to get frequency order from char order (get from
self._mTableSize = None # Size of above table # GetOrder())
self._mTypicalDistributionRatio = None # This is a constant value which varies from language to language, used in calculating confidence. See http://www.mozilla.org/projects/intl/UniversalCharsetDetection.html for further detail. self._mCharToFreqOrder = None
self._mTableSize = None # Size of above table
# This is a constant value which varies from language to language,
# used in calculating confidence. See
# http://www.mozilla.org/projects/intl/UniversalCharsetDetection.html
# for further detail.
self._mTypicalDistributionRatio = None
self.reset() self.reset()
def reset(self): def reset(self):
"""reset analyser, clear any state""" """reset analyser, clear any state"""
self._mDone = constants.False # If this flag is set to constants.True, detection is done and conclusion has been made # If this flag is set to True, detection is done and conclusion has
self._mTotalChars = 0 # Total characters encountered # been made
self._mFreqChars = 0 # The number of characters whose frequency order is less than 512 self._mDone = False
self._mTotalChars = 0 # Total characters encountered
# The number of characters whose frequency order is less than 512
self._mFreqChars = 0
def feed(self, aStr, aCharLen): def feed(self, aBuf, aCharLen):
"""feed a character with known length""" """feed a character with known length"""
if aCharLen == 2: if aCharLen == 2:
# we only care about 2-bytes character in our distribution analysis # we only care about 2-bytes character in our distribution analysis
order = self.get_order(aStr) order = self.get_order(aBuf)
else: else:
order = -1 order = -1
if order >= 0: if order >= 0:
@@ -65,12 +81,14 @@ class CharDistributionAnalysis:
def get_confidence(self): def get_confidence(self):
"""return confidence based on existing data""" """return confidence based on existing data"""
# if we didn't receive any character in our consideration range, return negative answer # if we didn't receive any character in our consideration range,
if self._mTotalChars <= 0: # return negative answer
if self._mTotalChars <= 0 or self._mFreqChars <= MINIMUM_DATA_THRESHOLD:
return SURE_NO return SURE_NO
if self._mTotalChars != self._mFreqChars: if self._mTotalChars != self._mFreqChars:
r = self._mFreqChars / ((self._mTotalChars - self._mFreqChars) * self._mTypicalDistributionRatio) r = (self._mFreqChars / ((self._mTotalChars - self._mFreqChars)
* self._mTypicalDistributionRatio))
if r < SURE_YES: if r < SURE_YES:
return r return r
@@ -78,16 +96,18 @@ class CharDistributionAnalysis:
return SURE_YES return SURE_YES
def got_enough_data(self): def got_enough_data(self):
# It is not necessary to receive all data to draw conclusion. For charset detection, # It is not necessary to receive all data to draw conclusion.
# certain amount of data is enough # For charset detection, certain amount of data is enough
return self._mTotalChars > ENOUGH_DATA_THRESHOLD return self._mTotalChars > ENOUGH_DATA_THRESHOLD
def get_order(self, aStr): def get_order(self, aBuf):
# We do not handle characters based on the original encoding string, but # We do not handle characters based on the original encoding string,
# convert this encoding string to a number, here called order. # but convert this encoding string to a number, here called order.
# This allows multiple encodings of a language to share one frequency table. # This allows multiple encodings of a language to share one frequency
# table.
return -1 return -1
class EUCTWDistributionAnalysis(CharDistributionAnalysis): class EUCTWDistributionAnalysis(CharDistributionAnalysis):
def __init__(self): def __init__(self):
CharDistributionAnalysis.__init__(self) CharDistributionAnalysis.__init__(self)
@@ -95,16 +115,18 @@ class EUCTWDistributionAnalysis(CharDistributionAnalysis):
self._mTableSize = EUCTW_TABLE_SIZE self._mTableSize = EUCTW_TABLE_SIZE
self._mTypicalDistributionRatio = EUCTW_TYPICAL_DISTRIBUTION_RATIO self._mTypicalDistributionRatio = EUCTW_TYPICAL_DISTRIBUTION_RATIO
def get_order(self, aStr): def get_order(self, aBuf):
# for euc-TW encoding, we are interested # for euc-TW encoding, we are interested
# first byte range: 0xc4 -- 0xfe # first byte range: 0xc4 -- 0xfe
# second byte range: 0xa1 -- 0xfe # second byte range: 0xa1 -- 0xfe
# no validation needed here. State machine has done that # no validation needed here. State machine has done that
if aStr[0] >= '\xC4': first_char = wrap_ord(aBuf[0])
return 94 * (ord(aStr[0]) - 0xC4) + ord(aStr[1]) - 0xA1 if first_char >= 0xC4:
return 94 * (first_char - 0xC4) + wrap_ord(aBuf[1]) - 0xA1
else: else:
return -1 return -1
class EUCKRDistributionAnalysis(CharDistributionAnalysis): class EUCKRDistributionAnalysis(CharDistributionAnalysis):
def __init__(self): def __init__(self):
CharDistributionAnalysis.__init__(self) CharDistributionAnalysis.__init__(self)
@@ -112,15 +134,17 @@ class EUCKRDistributionAnalysis(CharDistributionAnalysis):
self._mTableSize = EUCKR_TABLE_SIZE self._mTableSize = EUCKR_TABLE_SIZE
self._mTypicalDistributionRatio = EUCKR_TYPICAL_DISTRIBUTION_RATIO self._mTypicalDistributionRatio = EUCKR_TYPICAL_DISTRIBUTION_RATIO
def get_order(self, aStr): def get_order(self, aBuf):
# for euc-KR encoding, we are interested # for euc-KR encoding, we are interested
# first byte range: 0xb0 -- 0xfe # first byte range: 0xb0 -- 0xfe
# second byte range: 0xa1 -- 0xfe # second byte range: 0xa1 -- 0xfe
# no validation needed here. State machine has done that # no validation needed here. State machine has done that
if aStr[0] >= '\xB0': first_char = wrap_ord(aBuf[0])
return 94 * (ord(aStr[0]) - 0xB0) + ord(aStr[1]) - 0xA1 if first_char >= 0xB0:
return 94 * (first_char - 0xB0) + wrap_ord(aBuf[1]) - 0xA1
else: else:
return -1; return -1
class GB2312DistributionAnalysis(CharDistributionAnalysis): class GB2312DistributionAnalysis(CharDistributionAnalysis):
def __init__(self): def __init__(self):
@@ -129,15 +153,17 @@ class GB2312DistributionAnalysis(CharDistributionAnalysis):
self._mTableSize = GB2312_TABLE_SIZE self._mTableSize = GB2312_TABLE_SIZE
self._mTypicalDistributionRatio = GB2312_TYPICAL_DISTRIBUTION_RATIO self._mTypicalDistributionRatio = GB2312_TYPICAL_DISTRIBUTION_RATIO
def get_order(self, aStr): def get_order(self, aBuf):
# for GB2312 encoding, we are interested # for GB2312 encoding, we are interested
# first byte range: 0xb0 -- 0xfe # first byte range: 0xb0 -- 0xfe
# second byte range: 0xa1 -- 0xfe # second byte range: 0xa1 -- 0xfe
# no validation needed here. State machine has done that # no validation needed here. State machine has done that
if (aStr[0] >= '\xB0') and (aStr[1] >= '\xA1'): first_char, second_char = wrap_ord(aBuf[0]), wrap_ord(aBuf[1])
return 94 * (ord(aStr[0]) - 0xB0) + ord(aStr[1]) - 0xA1 if (first_char >= 0xB0) and (second_char >= 0xA1):
return 94 * (first_char - 0xB0) + second_char - 0xA1
else: else:
return -1; return -1
class Big5DistributionAnalysis(CharDistributionAnalysis): class Big5DistributionAnalysis(CharDistributionAnalysis):
def __init__(self): def __init__(self):
@@ -146,19 +172,21 @@ class Big5DistributionAnalysis(CharDistributionAnalysis):
self._mTableSize = BIG5_TABLE_SIZE self._mTableSize = BIG5_TABLE_SIZE
self._mTypicalDistributionRatio = BIG5_TYPICAL_DISTRIBUTION_RATIO self._mTypicalDistributionRatio = BIG5_TYPICAL_DISTRIBUTION_RATIO
def get_order(self, aStr): def get_order(self, aBuf):
# for big5 encoding, we are interested # for big5 encoding, we are interested
# first byte range: 0xa4 -- 0xfe # first byte range: 0xa4 -- 0xfe
# second byte range: 0x40 -- 0x7e , 0xa1 -- 0xfe # second byte range: 0x40 -- 0x7e , 0xa1 -- 0xfe
# no validation needed here. State machine has done that # no validation needed here. State machine has done that
if aStr[0] >= '\xA4': first_char, second_char = wrap_ord(aBuf[0]), wrap_ord(aBuf[1])
if aStr[1] >= '\xA1': if first_char >= 0xA4:
return 157 * (ord(aStr[0]) - 0xA4) + ord(aStr[1]) - 0xA1 + 63 if second_char >= 0xA1:
return 157 * (first_char - 0xA4) + second_char - 0xA1 + 63
else: else:
return 157 * (ord(aStr[0]) - 0xA4) + ord(aStr[1]) - 0x40 return 157 * (first_char - 0xA4) + second_char - 0x40
else: else:
return -1 return -1
class SJISDistributionAnalysis(CharDistributionAnalysis): class SJISDistributionAnalysis(CharDistributionAnalysis):
def __init__(self): def __init__(self):
CharDistributionAnalysis.__init__(self) CharDistributionAnalysis.__init__(self)
@@ -166,22 +194,24 @@ class SJISDistributionAnalysis(CharDistributionAnalysis):
self._mTableSize = JIS_TABLE_SIZE self._mTableSize = JIS_TABLE_SIZE
self._mTypicalDistributionRatio = JIS_TYPICAL_DISTRIBUTION_RATIO self._mTypicalDistributionRatio = JIS_TYPICAL_DISTRIBUTION_RATIO
def get_order(self, aStr): def get_order(self, aBuf):
# for sjis encoding, we are interested # for sjis encoding, we are interested
# first byte range: 0x81 -- 0x9f , 0xe0 -- 0xfe # first byte range: 0x81 -- 0x9f , 0xe0 -- 0xfe
# second byte range: 0x40 -- 0x7e, 0x81 -- oxfe # second byte range: 0x40 -- 0x7e, 0x81 -- oxfe
# no validation needed here. State machine has done that # no validation needed here. State machine has done that
if (aStr[0] >= '\x81') and (aStr[0] <= '\x9F'): first_char, second_char = wrap_ord(aBuf[0]), wrap_ord(aBuf[1])
order = 188 * (ord(aStr[0]) - 0x81) if (first_char >= 0x81) and (first_char <= 0x9F):
elif (aStr[0] >= '\xE0') and (aStr[0] <= '\xEF'): order = 188 * (first_char - 0x81)
order = 188 * (ord(aStr[0]) - 0xE0 + 31) elif (first_char >= 0xE0) and (first_char <= 0xEF):
order = 188 * (first_char - 0xE0 + 31)
else: else:
return -1; return -1
order = order + ord(aStr[1]) - 0x40 order = order + second_char - 0x40
if aStr[1] > '\x7F': if second_char > 0x7F:
order =- 1 order = -1
return order return order
class EUCJPDistributionAnalysis(CharDistributionAnalysis): class EUCJPDistributionAnalysis(CharDistributionAnalysis):
def __init__(self): def __init__(self):
CharDistributionAnalysis.__init__(self) CharDistributionAnalysis.__init__(self)
@@ -189,12 +219,13 @@ class EUCJPDistributionAnalysis(CharDistributionAnalysis):
self._mTableSize = JIS_TABLE_SIZE self._mTableSize = JIS_TABLE_SIZE
self._mTypicalDistributionRatio = JIS_TYPICAL_DISTRIBUTION_RATIO self._mTypicalDistributionRatio = JIS_TYPICAL_DISTRIBUTION_RATIO
def get_order(self, aStr): def get_order(self, aBuf):
# for euc-JP encoding, we are interested # for euc-JP encoding, we are interested
# first byte range: 0xa0 -- 0xfe # first byte range: 0xa0 -- 0xfe
# second byte range: 0xa1 -- 0xfe # second byte range: 0xa1 -- 0xfe
# no validation needed here. State machine has done that # no validation needed here. State machine has done that
if aStr[0] >= '\xA0': char = wrap_ord(aBuf[0])
return 94 * (ord(aStr[0]) - 0xA1) + ord(aStr[1]) - 0xa1 if char >= 0xA0:
return 94 * (char - 0xA1) + wrap_ord(aBuf[1]) - 0xa1
else: else:
return -1 return -1
+23 -13
View File
@@ -25,8 +25,10 @@
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants, sys from . import constants
from charsetprober import CharSetProber import sys
from .charsetprober import CharSetProber
class CharSetGroupProber(CharSetProber): class CharSetGroupProber(CharSetProber):
def __init__(self): def __init__(self):
@@ -34,35 +36,39 @@ class CharSetGroupProber(CharSetProber):
self._mActiveNum = 0 self._mActiveNum = 0
self._mProbers = [] self._mProbers = []
self._mBestGuessProber = None self._mBestGuessProber = None
def reset(self): def reset(self):
CharSetProber.reset(self) CharSetProber.reset(self)
self._mActiveNum = 0 self._mActiveNum = 0
for prober in self._mProbers: for prober in self._mProbers:
if prober: if prober:
prober.reset() prober.reset()
prober.active = constants.True prober.active = True
self._mActiveNum += 1 self._mActiveNum += 1
self._mBestGuessProber = None self._mBestGuessProber = None
def get_charset_name(self): def get_charset_name(self):
if not self._mBestGuessProber: if not self._mBestGuessProber:
self.get_confidence() self.get_confidence()
if not self._mBestGuessProber: return None if not self._mBestGuessProber:
return None
# self._mBestGuessProber = self._mProbers[0] # self._mBestGuessProber = self._mProbers[0]
return self._mBestGuessProber.get_charset_name() return self._mBestGuessProber.get_charset_name()
def feed(self, aBuf): def feed(self, aBuf):
for prober in self._mProbers: for prober in self._mProbers:
if not prober: continue if not prober:
if not prober.active: continue continue
if not prober.active:
continue
st = prober.feed(aBuf) st = prober.feed(aBuf)
if not st: continue if not st:
continue
if st == constants.eFoundIt: if st == constants.eFoundIt:
self._mBestGuessProber = prober self._mBestGuessProber = prober
return self.get_state() return self.get_state()
elif st == constants.eNotMe: elif st == constants.eNotMe:
prober.active = constants.False prober.active = False
self._mActiveNum -= 1 self._mActiveNum -= 1
if self._mActiveNum <= 0: if self._mActiveNum <= 0:
self._mState = constants.eNotMe self._mState = constants.eNotMe
@@ -78,18 +84,22 @@ class CharSetGroupProber(CharSetProber):
bestConf = 0.0 bestConf = 0.0
self._mBestGuessProber = None self._mBestGuessProber = None
for prober in self._mProbers: for prober in self._mProbers:
if not prober: continue if not prober:
continue
if not prober.active: if not prober.active:
if constants._debug: if constants._debug:
sys.stderr.write(prober.get_charset_name() + ' not active\n') sys.stderr.write(prober.get_charset_name()
+ ' not active\n')
continue continue
cf = prober.get_confidence() cf = prober.get_confidence()
if constants._debug: if constants._debug:
sys.stderr.write('%s confidence = %s\n' % (prober.get_charset_name(), cf)) sys.stderr.write('%s confidence = %s\n' %
(prober.get_charset_name(), cf))
if bestConf < cf: if bestConf < cf:
bestConf = cf bestConf = cf
self._mBestGuessProber = prober self._mBestGuessProber = prober
if not self._mBestGuessProber: return 0.0 if not self._mBestGuessProber:
return 0.0
return bestConf return bestConf
# else: # else:
# self._mBestGuessProber = self._mProbers[0] # self._mBestGuessProber = self._mProbers[0]
+13 -11
View File
@@ -1,11 +1,11 @@
######################## BEGIN LICENSE BLOCK ######################## ######################## BEGIN LICENSE BLOCK ########################
# The Original Code is Mozilla Universal charset detector code. # The Original Code is Mozilla Universal charset detector code.
# #
# The Initial Developer of the Original Code is # The Initial Developer of the Original Code is
# Netscape Communications Corporation. # Netscape Communications Corporation.
# Portions created by the Initial Developer are Copyright (C) 2001 # Portions created by the Initial Developer are Copyright (C) 2001
# the Initial Developer. All Rights Reserved. # the Initial Developer. All Rights Reserved.
# #
# Contributor(s): # Contributor(s):
# Mark Pilgrim - port to Python # Mark Pilgrim - port to Python
# Shy Shalom - original C code # Shy Shalom - original C code
@@ -14,27 +14,29 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants, re from . import constants
import re
class CharSetProber: class CharSetProber:
def __init__(self): def __init__(self):
pass pass
def reset(self): def reset(self):
self._mState = constants.eDetecting self._mState = constants.eDetecting
def get_charset_name(self): def get_charset_name(self):
return None return None
@@ -48,13 +50,13 @@ class CharSetProber:
return 0.0 return 0.0
def filter_high_bit_only(self, aBuf): def filter_high_bit_only(self, aBuf):
aBuf = re.sub(r'([\x00-\x7F])+', ' ', aBuf) aBuf = re.sub(b'([\x00-\x7F])+', b' ', aBuf)
return aBuf return aBuf
def filter_without_english_letters(self, aBuf): def filter_without_english_letters(self, aBuf):
aBuf = re.sub(r'([A-Za-z])+', ' ', aBuf) aBuf = re.sub(b'([A-Za-z])+', b' ', aBuf)
return aBuf return aBuf
def filter_with_english_letters(self, aBuf): def filter_with_english_letters(self, aBuf):
# TODO # TODO
return aBuf return aBuf
+10 -5
View File
@@ -13,19 +13,21 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
from constants import eStart, eError, eItsMe from .constants import eStart
from .compat import wrap_ord
class CodingStateMachine: class CodingStateMachine:
def __init__(self, sm): def __init__(self, sm):
@@ -40,12 +42,15 @@ class CodingStateMachine:
def next_state(self, c): def next_state(self, c):
# for each byte we get its class # for each byte we get its class
# if it is first byte, we also get byte length # if it is first byte, we also get byte length
byteCls = self._mModel['classTable'][ord(c)] # PY3K: aBuf is a byte stream, so c is an int, not a byte
byteCls = self._mModel['classTable'][wrap_ord(c)]
if self._mCurrentState == eStart: if self._mCurrentState == eStart:
self._mCurrentBytePos = 0 self._mCurrentBytePos = 0
self._mCurrentCharLen = self._mModel['charLenTable'][byteCls] self._mCurrentCharLen = self._mModel['charLenTable'][byteCls]
# from byte's class and stateTable, we get its next state # from byte's class and stateTable, we get its next state
self._mCurrentState = self._mModel['stateTable'][self._mCurrentState * self._mModel['classFactor'] + byteCls] curr_state = (self._mCurrentState * self._mModel['classFactor']
+ byteCls)
self._mCurrentState = self._mModel['stateTable'][curr_state]
self._mCurrentBytePos += 1 self._mCurrentBytePos += 1
return self._mCurrentState return self._mCurrentState
-8
View File
@@ -37,11 +37,3 @@ eError = 1
eItsMe = 2 eItsMe = 2
SHORTCUT_THRESHOLD = 0.95 SHORTCUT_THRESHOLD = 0.95
import __builtin__
if not hasattr(__builtin__, 'False'):
False = 0
True = 1
else:
False = __builtin__.False
True = __builtin__.True
+23 -16
View File
@@ -13,39 +13,43 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants, sys from . import constants
from escsm import HZSMModel, ISO2022CNSMModel, ISO2022JPSMModel, ISO2022KRSMModel from .escsm import (HZSMModel, ISO2022CNSMModel, ISO2022JPSMModel,
from charsetprober import CharSetProber ISO2022KRSMModel)
from codingstatemachine import CodingStateMachine from .charsetprober import CharSetProber
from .codingstatemachine import CodingStateMachine
from .compat import wrap_ord
class EscCharSetProber(CharSetProber): class EscCharSetProber(CharSetProber):
def __init__(self): def __init__(self):
CharSetProber.__init__(self) CharSetProber.__init__(self)
self._mCodingSM = [ \ self._mCodingSM = [
CodingStateMachine(HZSMModel), CodingStateMachine(HZSMModel),
CodingStateMachine(ISO2022CNSMModel), CodingStateMachine(ISO2022CNSMModel),
CodingStateMachine(ISO2022JPSMModel), CodingStateMachine(ISO2022JPSMModel),
CodingStateMachine(ISO2022KRSMModel) CodingStateMachine(ISO2022KRSMModel)
] ]
self.reset() self.reset()
def reset(self): def reset(self):
CharSetProber.reset(self) CharSetProber.reset(self)
for codingSM in self._mCodingSM: for codingSM in self._mCodingSM:
if not codingSM: continue if not codingSM:
codingSM.active = constants.True continue
codingSM.active = True
codingSM.reset() codingSM.reset()
self._mActiveSM = len(self._mCodingSM) self._mActiveSM = len(self._mCodingSM)
self._mDetectedCharset = None self._mDetectedCharset = None
@@ -61,19 +65,22 @@ class EscCharSetProber(CharSetProber):
def feed(self, aBuf): def feed(self, aBuf):
for c in aBuf: for c in aBuf:
# PY3K: aBuf is a byte array, so c is an int, not a byte
for codingSM in self._mCodingSM: for codingSM in self._mCodingSM:
if not codingSM: continue if not codingSM:
if not codingSM.active: continue continue
codingState = codingSM.next_state(c) if not codingSM.active:
continue
codingState = codingSM.next_state(wrap_ord(c))
if codingState == constants.eError: if codingState == constants.eError:
codingSM.active = constants.False codingSM.active = False
self._mActiveSM -= 1 self._mActiveSM -= 1
if self._mActiveSM <= 0: if self._mActiveSM <= 0:
self._mState = constants.eNotMe self._mState = constants.eNotMe
return self.get_state() return self.get_state()
elif codingState == constants.eItsMe: elif codingState == constants.eItsMe:
self._mState = constants.eFoundIt self._mState = constants.eFoundIt
self._mDetectedCharset = codingSM.get_coding_state_machine() self._mDetectedCharset = codingSM.get_coding_state_machine() # nopep8
return self.get_state() return self.get_state()
return self.get_state() return self.get_state()
+169 -167
View File
@@ -13,62 +13,62 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
from constants import eStart, eError, eItsMe from .constants import eStart, eError, eItsMe
HZ_cls = ( \ HZ_cls = (
1,0,0,0,0,0,0,0, # 00 - 07 1,0,0,0,0,0,0,0, # 00 - 07
0,0,0,0,0,0,0,0, # 08 - 0f 0,0,0,0,0,0,0,0, # 08 - 0f
0,0,0,0,0,0,0,0, # 10 - 17 0,0,0,0,0,0,0,0, # 10 - 17
0,0,0,1,0,0,0,0, # 18 - 1f 0,0,0,1,0,0,0,0, # 18 - 1f
0,0,0,0,0,0,0,0, # 20 - 27 0,0,0,0,0,0,0,0, # 20 - 27
0,0,0,0,0,0,0,0, # 28 - 2f 0,0,0,0,0,0,0,0, # 28 - 2f
0,0,0,0,0,0,0,0, # 30 - 37 0,0,0,0,0,0,0,0, # 30 - 37
0,0,0,0,0,0,0,0, # 38 - 3f 0,0,0,0,0,0,0,0, # 38 - 3f
0,0,0,0,0,0,0,0, # 40 - 47 0,0,0,0,0,0,0,0, # 40 - 47
0,0,0,0,0,0,0,0, # 48 - 4f 0,0,0,0,0,0,0,0, # 48 - 4f
0,0,0,0,0,0,0,0, # 50 - 57 0,0,0,0,0,0,0,0, # 50 - 57
0,0,0,0,0,0,0,0, # 58 - 5f 0,0,0,0,0,0,0,0, # 58 - 5f
0,0,0,0,0,0,0,0, # 60 - 67 0,0,0,0,0,0,0,0, # 60 - 67
0,0,0,0,0,0,0,0, # 68 - 6f 0,0,0,0,0,0,0,0, # 68 - 6f
0,0,0,0,0,0,0,0, # 70 - 77 0,0,0,0,0,0,0,0, # 70 - 77
0,0,0,4,0,5,2,0, # 78 - 7f 0,0,0,4,0,5,2,0, # 78 - 7f
1,1,1,1,1,1,1,1, # 80 - 87 1,1,1,1,1,1,1,1, # 80 - 87
1,1,1,1,1,1,1,1, # 88 - 8f 1,1,1,1,1,1,1,1, # 88 - 8f
1,1,1,1,1,1,1,1, # 90 - 97 1,1,1,1,1,1,1,1, # 90 - 97
1,1,1,1,1,1,1,1, # 98 - 9f 1,1,1,1,1,1,1,1, # 98 - 9f
1,1,1,1,1,1,1,1, # a0 - a7 1,1,1,1,1,1,1,1, # a0 - a7
1,1,1,1,1,1,1,1, # a8 - af 1,1,1,1,1,1,1,1, # a8 - af
1,1,1,1,1,1,1,1, # b0 - b7 1,1,1,1,1,1,1,1, # b0 - b7
1,1,1,1,1,1,1,1, # b8 - bf 1,1,1,1,1,1,1,1, # b8 - bf
1,1,1,1,1,1,1,1, # c0 - c7 1,1,1,1,1,1,1,1, # c0 - c7
1,1,1,1,1,1,1,1, # c8 - cf 1,1,1,1,1,1,1,1, # c8 - cf
1,1,1,1,1,1,1,1, # d0 - d7 1,1,1,1,1,1,1,1, # d0 - d7
1,1,1,1,1,1,1,1, # d8 - df 1,1,1,1,1,1,1,1, # d8 - df
1,1,1,1,1,1,1,1, # e0 - e7 1,1,1,1,1,1,1,1, # e0 - e7
1,1,1,1,1,1,1,1, # e8 - ef 1,1,1,1,1,1,1,1, # e8 - ef
1,1,1,1,1,1,1,1, # f0 - f7 1,1,1,1,1,1,1,1, # f0 - f7
1,1,1,1,1,1,1,1, # f8 - ff 1,1,1,1,1,1,1,1, # f8 - ff
) )
HZ_st = ( \ HZ_st = (
eStart,eError, 3,eStart,eStart,eStart,eError,eError,# 00-07 eStart,eError, 3,eStart,eStart,eStart,eError,eError,# 00-07
eError,eError,eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,# 08-0f eError,eError,eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,# 08-0f
eItsMe,eItsMe,eError,eError,eStart,eStart, 4,eError,# 10-17 eItsMe,eItsMe,eError,eError,eStart,eStart, 4,eError,# 10-17
5,eError, 6,eError, 5, 5, 4,eError,# 18-1f 5,eError, 6,eError, 5, 5, 4,eError,# 18-1f
4,eError, 4, 4, 4,eError, 4,eError,# 20-27 4,eError, 4, 4, 4,eError, 4,eError,# 20-27
4,eItsMe,eStart,eStart,eStart,eStart,eStart,eStart,# 28-2f 4,eItsMe,eStart,eStart,eStart,eStart,eStart,eStart,# 28-2f
) )
HZCharLenTable = (0, 0, 0, 0, 0, 0) HZCharLenTable = (0, 0, 0, 0, 0, 0)
@@ -79,50 +79,50 @@ HZSMModel = {'classTable': HZ_cls,
'charLenTable': HZCharLenTable, 'charLenTable': HZCharLenTable,
'name': "HZ-GB-2312"} 'name': "HZ-GB-2312"}
ISO2022CN_cls = ( \ ISO2022CN_cls = (
2,0,0,0,0,0,0,0, # 00 - 07 2,0,0,0,0,0,0,0, # 00 - 07
0,0,0,0,0,0,0,0, # 08 - 0f 0,0,0,0,0,0,0,0, # 08 - 0f
0,0,0,0,0,0,0,0, # 10 - 17 0,0,0,0,0,0,0,0, # 10 - 17
0,0,0,1,0,0,0,0, # 18 - 1f 0,0,0,1,0,0,0,0, # 18 - 1f
0,0,0,0,0,0,0,0, # 20 - 27 0,0,0,0,0,0,0,0, # 20 - 27
0,3,0,0,0,0,0,0, # 28 - 2f 0,3,0,0,0,0,0,0, # 28 - 2f
0,0,0,0,0,0,0,0, # 30 - 37 0,0,0,0,0,0,0,0, # 30 - 37
0,0,0,0,0,0,0,0, # 38 - 3f 0,0,0,0,0,0,0,0, # 38 - 3f
0,0,0,4,0,0,0,0, # 40 - 47 0,0,0,4,0,0,0,0, # 40 - 47
0,0,0,0,0,0,0,0, # 48 - 4f 0,0,0,0,0,0,0,0, # 48 - 4f
0,0,0,0,0,0,0,0, # 50 - 57 0,0,0,0,0,0,0,0, # 50 - 57
0,0,0,0,0,0,0,0, # 58 - 5f 0,0,0,0,0,0,0,0, # 58 - 5f
0,0,0,0,0,0,0,0, # 60 - 67 0,0,0,0,0,0,0,0, # 60 - 67
0,0,0,0,0,0,0,0, # 68 - 6f 0,0,0,0,0,0,0,0, # 68 - 6f
0,0,0,0,0,0,0,0, # 70 - 77 0,0,0,0,0,0,0,0, # 70 - 77
0,0,0,0,0,0,0,0, # 78 - 7f 0,0,0,0,0,0,0,0, # 78 - 7f
2,2,2,2,2,2,2,2, # 80 - 87 2,2,2,2,2,2,2,2, # 80 - 87
2,2,2,2,2,2,2,2, # 88 - 8f 2,2,2,2,2,2,2,2, # 88 - 8f
2,2,2,2,2,2,2,2, # 90 - 97 2,2,2,2,2,2,2,2, # 90 - 97
2,2,2,2,2,2,2,2, # 98 - 9f 2,2,2,2,2,2,2,2, # 98 - 9f
2,2,2,2,2,2,2,2, # a0 - a7 2,2,2,2,2,2,2,2, # a0 - a7
2,2,2,2,2,2,2,2, # a8 - af 2,2,2,2,2,2,2,2, # a8 - af
2,2,2,2,2,2,2,2, # b0 - b7 2,2,2,2,2,2,2,2, # b0 - b7
2,2,2,2,2,2,2,2, # b8 - bf 2,2,2,2,2,2,2,2, # b8 - bf
2,2,2,2,2,2,2,2, # c0 - c7 2,2,2,2,2,2,2,2, # c0 - c7
2,2,2,2,2,2,2,2, # c8 - cf 2,2,2,2,2,2,2,2, # c8 - cf
2,2,2,2,2,2,2,2, # d0 - d7 2,2,2,2,2,2,2,2, # d0 - d7
2,2,2,2,2,2,2,2, # d8 - df 2,2,2,2,2,2,2,2, # d8 - df
2,2,2,2,2,2,2,2, # e0 - e7 2,2,2,2,2,2,2,2, # e0 - e7
2,2,2,2,2,2,2,2, # e8 - ef 2,2,2,2,2,2,2,2, # e8 - ef
2,2,2,2,2,2,2,2, # f0 - f7 2,2,2,2,2,2,2,2, # f0 - f7
2,2,2,2,2,2,2,2, # f8 - ff 2,2,2,2,2,2,2,2, # f8 - ff
) )
ISO2022CN_st = ( \ ISO2022CN_st = (
eStart, 3,eError,eStart,eStart,eStart,eStart,eStart,# 00-07 eStart, 3,eError,eStart,eStart,eStart,eStart,eStart,# 00-07
eStart,eError,eError,eError,eError,eError,eError,eError,# 08-0f eStart,eError,eError,eError,eError,eError,eError,eError,# 08-0f
eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,# 10-17 eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,# 10-17
eItsMe,eItsMe,eItsMe,eError,eError,eError, 4,eError,# 18-1f eItsMe,eItsMe,eItsMe,eError,eError,eError, 4,eError,# 18-1f
eError,eError,eError,eItsMe,eError,eError,eError,eError,# 20-27 eError,eError,eError,eItsMe,eError,eError,eError,eError,# 20-27
5, 6,eError,eError,eError,eError,eError,eError,# 28-2f 5, 6,eError,eError,eError,eError,eError,eError,# 28-2f
eError,eError,eError,eItsMe,eError,eError,eError,eError,# 30-37 eError,eError,eError,eItsMe,eError,eError,eError,eError,# 30-37
eError,eError,eError,eError,eError,eItsMe,eError,eStart,# 38-3f eError,eError,eError,eError,eError,eItsMe,eError,eStart,# 38-3f
) )
ISO2022CNCharLenTable = (0, 0, 0, 0, 0, 0, 0, 0, 0) ISO2022CNCharLenTable = (0, 0, 0, 0, 0, 0, 0, 0, 0)
@@ -133,51 +133,51 @@ ISO2022CNSMModel = {'classTable': ISO2022CN_cls,
'charLenTable': ISO2022CNCharLenTable, 'charLenTable': ISO2022CNCharLenTable,
'name': "ISO-2022-CN"} 'name': "ISO-2022-CN"}
ISO2022JP_cls = ( \ ISO2022JP_cls = (
2,0,0,0,0,0,0,0, # 00 - 07 2,0,0,0,0,0,0,0, # 00 - 07
0,0,0,0,0,0,2,2, # 08 - 0f 0,0,0,0,0,0,2,2, # 08 - 0f
0,0,0,0,0,0,0,0, # 10 - 17 0,0,0,0,0,0,0,0, # 10 - 17
0,0,0,1,0,0,0,0, # 18 - 1f 0,0,0,1,0,0,0,0, # 18 - 1f
0,0,0,0,7,0,0,0, # 20 - 27 0,0,0,0,7,0,0,0, # 20 - 27
3,0,0,0,0,0,0,0, # 28 - 2f 3,0,0,0,0,0,0,0, # 28 - 2f
0,0,0,0,0,0,0,0, # 30 - 37 0,0,0,0,0,0,0,0, # 30 - 37
0,0,0,0,0,0,0,0, # 38 - 3f 0,0,0,0,0,0,0,0, # 38 - 3f
6,0,4,0,8,0,0,0, # 40 - 47 6,0,4,0,8,0,0,0, # 40 - 47
0,9,5,0,0,0,0,0, # 48 - 4f 0,9,5,0,0,0,0,0, # 48 - 4f
0,0,0,0,0,0,0,0, # 50 - 57 0,0,0,0,0,0,0,0, # 50 - 57
0,0,0,0,0,0,0,0, # 58 - 5f 0,0,0,0,0,0,0,0, # 58 - 5f
0,0,0,0,0,0,0,0, # 60 - 67 0,0,0,0,0,0,0,0, # 60 - 67
0,0,0,0,0,0,0,0, # 68 - 6f 0,0,0,0,0,0,0,0, # 68 - 6f
0,0,0,0,0,0,0,0, # 70 - 77 0,0,0,0,0,0,0,0, # 70 - 77
0,0,0,0,0,0,0,0, # 78 - 7f 0,0,0,0,0,0,0,0, # 78 - 7f
2,2,2,2,2,2,2,2, # 80 - 87 2,2,2,2,2,2,2,2, # 80 - 87
2,2,2,2,2,2,2,2, # 88 - 8f 2,2,2,2,2,2,2,2, # 88 - 8f
2,2,2,2,2,2,2,2, # 90 - 97 2,2,2,2,2,2,2,2, # 90 - 97
2,2,2,2,2,2,2,2, # 98 - 9f 2,2,2,2,2,2,2,2, # 98 - 9f
2,2,2,2,2,2,2,2, # a0 - a7 2,2,2,2,2,2,2,2, # a0 - a7
2,2,2,2,2,2,2,2, # a8 - af 2,2,2,2,2,2,2,2, # a8 - af
2,2,2,2,2,2,2,2, # b0 - b7 2,2,2,2,2,2,2,2, # b0 - b7
2,2,2,2,2,2,2,2, # b8 - bf 2,2,2,2,2,2,2,2, # b8 - bf
2,2,2,2,2,2,2,2, # c0 - c7 2,2,2,2,2,2,2,2, # c0 - c7
2,2,2,2,2,2,2,2, # c8 - cf 2,2,2,2,2,2,2,2, # c8 - cf
2,2,2,2,2,2,2,2, # d0 - d7 2,2,2,2,2,2,2,2, # d0 - d7
2,2,2,2,2,2,2,2, # d8 - df 2,2,2,2,2,2,2,2, # d8 - df
2,2,2,2,2,2,2,2, # e0 - e7 2,2,2,2,2,2,2,2, # e0 - e7
2,2,2,2,2,2,2,2, # e8 - ef 2,2,2,2,2,2,2,2, # e8 - ef
2,2,2,2,2,2,2,2, # f0 - f7 2,2,2,2,2,2,2,2, # f0 - f7
2,2,2,2,2,2,2,2, # f8 - ff 2,2,2,2,2,2,2,2, # f8 - ff
) )
ISO2022JP_st = ( \ ISO2022JP_st = (
eStart, 3,eError,eStart,eStart,eStart,eStart,eStart,# 00-07 eStart, 3,eError,eStart,eStart,eStart,eStart,eStart,# 00-07
eStart,eStart,eError,eError,eError,eError,eError,eError,# 08-0f eStart,eStart,eError,eError,eError,eError,eError,eError,# 08-0f
eError,eError,eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,# 10-17 eError,eError,eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,# 10-17
eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,eError,eError,# 18-1f eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,eItsMe,eError,eError,# 18-1f
eError, 5,eError,eError,eError, 4,eError,eError,# 20-27 eError, 5,eError,eError,eError, 4,eError,eError,# 20-27
eError,eError,eError, 6,eItsMe,eError,eItsMe,eError,# 28-2f eError,eError,eError, 6,eItsMe,eError,eItsMe,eError,# 28-2f
eError,eError,eError,eError,eError,eError,eItsMe,eItsMe,# 30-37 eError,eError,eError,eError,eError,eError,eItsMe,eItsMe,# 30-37
eError,eError,eError,eItsMe,eError,eError,eError,eError,# 38-3f eError,eError,eError,eItsMe,eError,eError,eError,eError,# 38-3f
eError,eError,eError,eError,eItsMe,eError,eStart,eStart,# 40-47 eError,eError,eError,eError,eItsMe,eError,eStart,eStart,# 40-47
) )
ISO2022JPCharLenTable = (0, 0, 0, 0, 0, 0, 0, 0, 0, 0) ISO2022JPCharLenTable = (0, 0, 0, 0, 0, 0, 0, 0, 0, 0)
@@ -188,47 +188,47 @@ ISO2022JPSMModel = {'classTable': ISO2022JP_cls,
'charLenTable': ISO2022JPCharLenTable, 'charLenTable': ISO2022JPCharLenTable,
'name': "ISO-2022-JP"} 'name': "ISO-2022-JP"}
ISO2022KR_cls = ( \ ISO2022KR_cls = (
2,0,0,0,0,0,0,0, # 00 - 07 2,0,0,0,0,0,0,0, # 00 - 07
0,0,0,0,0,0,0,0, # 08 - 0f 0,0,0,0,0,0,0,0, # 08 - 0f
0,0,0,0,0,0,0,0, # 10 - 17 0,0,0,0,0,0,0,0, # 10 - 17
0,0,0,1,0,0,0,0, # 18 - 1f 0,0,0,1,0,0,0,0, # 18 - 1f
0,0,0,0,3,0,0,0, # 20 - 27 0,0,0,0,3,0,0,0, # 20 - 27
0,4,0,0,0,0,0,0, # 28 - 2f 0,4,0,0,0,0,0,0, # 28 - 2f
0,0,0,0,0,0,0,0, # 30 - 37 0,0,0,0,0,0,0,0, # 30 - 37
0,0,0,0,0,0,0,0, # 38 - 3f 0,0,0,0,0,0,0,0, # 38 - 3f
0,0,0,5,0,0,0,0, # 40 - 47 0,0,0,5,0,0,0,0, # 40 - 47
0,0,0,0,0,0,0,0, # 48 - 4f 0,0,0,0,0,0,0,0, # 48 - 4f
0,0,0,0,0,0,0,0, # 50 - 57 0,0,0,0,0,0,0,0, # 50 - 57
0,0,0,0,0,0,0,0, # 58 - 5f 0,0,0,0,0,0,0,0, # 58 - 5f
0,0,0,0,0,0,0,0, # 60 - 67 0,0,0,0,0,0,0,0, # 60 - 67
0,0,0,0,0,0,0,0, # 68 - 6f 0,0,0,0,0,0,0,0, # 68 - 6f
0,0,0,0,0,0,0,0, # 70 - 77 0,0,0,0,0,0,0,0, # 70 - 77
0,0,0,0,0,0,0,0, # 78 - 7f 0,0,0,0,0,0,0,0, # 78 - 7f
2,2,2,2,2,2,2,2, # 80 - 87 2,2,2,2,2,2,2,2, # 80 - 87
2,2,2,2,2,2,2,2, # 88 - 8f 2,2,2,2,2,2,2,2, # 88 - 8f
2,2,2,2,2,2,2,2, # 90 - 97 2,2,2,2,2,2,2,2, # 90 - 97
2,2,2,2,2,2,2,2, # 98 - 9f 2,2,2,2,2,2,2,2, # 98 - 9f
2,2,2,2,2,2,2,2, # a0 - a7 2,2,2,2,2,2,2,2, # a0 - a7
2,2,2,2,2,2,2,2, # a8 - af 2,2,2,2,2,2,2,2, # a8 - af
2,2,2,2,2,2,2,2, # b0 - b7 2,2,2,2,2,2,2,2, # b0 - b7
2,2,2,2,2,2,2,2, # b8 - bf 2,2,2,2,2,2,2,2, # b8 - bf
2,2,2,2,2,2,2,2, # c0 - c7 2,2,2,2,2,2,2,2, # c0 - c7
2,2,2,2,2,2,2,2, # c8 - cf 2,2,2,2,2,2,2,2, # c8 - cf
2,2,2,2,2,2,2,2, # d0 - d7 2,2,2,2,2,2,2,2, # d0 - d7
2,2,2,2,2,2,2,2, # d8 - df 2,2,2,2,2,2,2,2, # d8 - df
2,2,2,2,2,2,2,2, # e0 - e7 2,2,2,2,2,2,2,2, # e0 - e7
2,2,2,2,2,2,2,2, # e8 - ef 2,2,2,2,2,2,2,2, # e8 - ef
2,2,2,2,2,2,2,2, # f0 - f7 2,2,2,2,2,2,2,2, # f0 - f7
2,2,2,2,2,2,2,2, # f8 - ff 2,2,2,2,2,2,2,2, # f8 - ff
) )
ISO2022KR_st = ( \ ISO2022KR_st = (
eStart, 3,eError,eStart,eStart,eStart,eError,eError,# 00-07 eStart, 3,eError,eStart,eStart,eStart,eError,eError,# 00-07
eError,eError,eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,# 08-0f eError,eError,eError,eError,eItsMe,eItsMe,eItsMe,eItsMe,# 08-0f
eItsMe,eItsMe,eError,eError,eError, 4,eError,eError,# 10-17 eItsMe,eItsMe,eError,eError,eError, 4,eError,eError,# 10-17
eError,eError,eError,eError, 5,eError,eError,eError,# 18-1f eError,eError,eError,eError, 5,eError,eError,eError,# 18-1f
eError,eError,eError,eItsMe,eStart,eStart,eStart,eStart,# 20-27 eError,eError,eError,eItsMe,eStart,eStart,eStart,eStart,# 20-27
) )
ISO2022KRCharLenTable = (0, 0, 0, 0, 0, 0) ISO2022KRCharLenTable = (0, 0, 0, 0, 0, 0)
@@ -238,3 +238,5 @@ ISO2022KRSMModel = {'classTable': ISO2022KR_cls,
'stateTable': ISO2022KR_st, 'stateTable': ISO2022KR_st,
'charLenTable': ISO2022KRCharLenTable, 'charLenTable': ISO2022KRCharLenTable,
'name': "ISO-2022-KR"} 'name': "ISO-2022-KR"}
# flake8: noqa
+25 -20
View File
@@ -13,25 +13,26 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants, sys import sys
from constants import eStart, eError, eItsMe from . import constants
from mbcharsetprober import MultiByteCharSetProber from .mbcharsetprober import MultiByteCharSetProber
from codingstatemachine import CodingStateMachine from .codingstatemachine import CodingStateMachine
from chardistribution import EUCJPDistributionAnalysis from .chardistribution import EUCJPDistributionAnalysis
from jpcntx import EUCJPContextAnalysis from .jpcntx import EUCJPContextAnalysis
from mbcssm import EUCJPSMModel from .mbcssm import EUCJPSMModel
class EUCJPProber(MultiByteCharSetProber): class EUCJPProber(MultiByteCharSetProber):
def __init__(self): def __init__(self):
@@ -44,37 +45,41 @@ class EUCJPProber(MultiByteCharSetProber):
def reset(self): def reset(self):
MultiByteCharSetProber.reset(self) MultiByteCharSetProber.reset(self)
self._mContextAnalyzer.reset() self._mContextAnalyzer.reset()
def get_charset_name(self): def get_charset_name(self):
return "EUC-JP" return "EUC-JP"
def feed(self, aBuf): def feed(self, aBuf):
aLen = len(aBuf) aLen = len(aBuf)
for i in range(0, aLen): for i in range(0, aLen):
# PY3K: aBuf is a byte array, so aBuf[i] is an int, not a byte
codingState = self._mCodingSM.next_state(aBuf[i]) codingState = self._mCodingSM.next_state(aBuf[i])
if codingState == eError: if codingState == constants.eError:
if constants._debug: if constants._debug:
sys.stderr.write(self.get_charset_name() + ' prober hit error at byte ' + str(i) + '\n') sys.stderr.write(self.get_charset_name()
+ ' prober hit error at byte ' + str(i)
+ '\n')
self._mState = constants.eNotMe self._mState = constants.eNotMe
break break
elif codingState == eItsMe: elif codingState == constants.eItsMe:
self._mState = constants.eFoundIt self._mState = constants.eFoundIt
break break
elif codingState == eStart: elif codingState == constants.eStart:
charLen = self._mCodingSM.get_current_charlen() charLen = self._mCodingSM.get_current_charlen()
if i == 0: if i == 0:
self._mLastChar[1] = aBuf[0] self._mLastChar[1] = aBuf[0]
self._mContextAnalyzer.feed(self._mLastChar, charLen) self._mContextAnalyzer.feed(self._mLastChar, charLen)
self._mDistributionAnalyzer.feed(self._mLastChar, charLen) self._mDistributionAnalyzer.feed(self._mLastChar, charLen)
else: else:
self._mContextAnalyzer.feed(aBuf[i-1:i+1], charLen) self._mContextAnalyzer.feed(aBuf[i - 1:i + 1], charLen)
self._mDistributionAnalyzer.feed(aBuf[i-1:i+1], charLen) self._mDistributionAnalyzer.feed(aBuf[i - 1:i + 1],
charLen)
self._mLastChar[0] = aBuf[aLen - 1] self._mLastChar[0] = aBuf[aLen - 1]
if self.get_state() == constants.eDetecting: if self.get_state() == constants.eDetecting:
if self._mContextAnalyzer.got_enough_data() and \ if (self._mContextAnalyzer.got_enough_data() and
(self.get_confidence() > constants.SHORTCUT_THRESHOLD): (self.get_confidence() > constants.SHORTCUT_THRESHOLD)):
self._mState = constants.eFoundIt self._mState = constants.eFoundIt
return self.get_state() return self.get_state()
+2
View File
@@ -592,3 +592,5 @@ EUCKRCharToFreqOrder = ( \
8704,8705,8706,8707,8708,8709,8710,8711,8712,8713,8714,8715,8716,8717,8718,8719, 8704,8705,8706,8707,8708,8709,8710,8711,8712,8713,8714,8715,8716,8717,8718,8719,
8720,8721,8722,8723,8724,8725,8726,8727,8728,8729,8730,8731,8732,8733,8734,8735, 8720,8721,8722,8723,8724,8725,8726,8727,8728,8729,8730,8731,8732,8733,8734,8735,
8736,8737,8738,8739,8740,8741) 8736,8737,8738,8739,8740,8741)
# flake8: noqa
+7 -6
View File
@@ -13,22 +13,23 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
from mbcharsetprober import MultiByteCharSetProber from .mbcharsetprober import MultiByteCharSetProber
from codingstatemachine import CodingStateMachine from .codingstatemachine import CodingStateMachine
from chardistribution import EUCKRDistributionAnalysis from .chardistribution import EUCKRDistributionAnalysis
from mbcssm import EUCKRSMModel from .mbcssm import EUCKRSMModel
class EUCKRProber(MultiByteCharSetProber): class EUCKRProber(MultiByteCharSetProber):
def __init__(self): def __init__(self):
+9 -7
View File
@@ -13,12 +13,12 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
@@ -26,8 +26,8 @@
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
# EUCTW frequency table # EUCTW frequency table
# Converted from big5 work # Converted from big5 work
# by Taiwan's Mandarin Promotion Council # by Taiwan's Mandarin Promotion Council
# <http:#www.edu.tw:81/mandr/> # <http:#www.edu.tw:81/mandr/>
# 128 --> 0.42261 # 128 --> 0.42261
@@ -38,15 +38,15 @@
# #
# Idea Distribution Ratio = 0.74851/(1-0.74851) =2.98 # Idea Distribution Ratio = 0.74851/(1-0.74851) =2.98
# Random Distribution Ration = 512/(5401-512)=0.105 # Random Distribution Ration = 512/(5401-512)=0.105
# #
# Typical Distribution Ratio about 25% of Ideal one, still much higher than RDR # Typical Distribution Ratio about 25% of Ideal one, still much higher than RDR
EUCTW_TYPICAL_DISTRIBUTION_RATIO = 0.75 EUCTW_TYPICAL_DISTRIBUTION_RATIO = 0.75
# Char to FreqOrder table , # Char to FreqOrder table ,
EUCTW_TABLE_SIZE = 8102 EUCTW_TABLE_SIZE = 8102
EUCTWCharToFreqOrder = ( \ EUCTWCharToFreqOrder = (
1,1800,1506, 255,1431, 198, 9, 82, 6,7310, 177, 202,3615,1256,2808, 110, # 2742 1,1800,1506, 255,1431, 198, 9, 82, 6,7310, 177, 202,3615,1256,2808, 110, # 2742
3735, 33,3241, 261, 76, 44,2113, 16,2931,2184,1176, 659,3868, 26,3404,2643, # 2758 3735, 33,3241, 261, 76, 44,2113, 16,2931,2184,1176, 659,3868, 26,3404,2643, # 2758
1198,3869,3313,4060, 410,2211, 302, 590, 361,1963, 8, 204, 58,4296,7311,1931, # 2774 1198,3869,3313,4060, 410,2211, 302, 590, 361,1963, 8, 204, 58,4296,7311,1931, # 2774
@@ -424,3 +424,5 @@ EUCTWCharToFreqOrder = ( \
8694,8695,8696,8697,8698,8699,8700,8701,8702,8703,8704,8705,8706,8707,8708,8709, # 8710 8694,8695,8696,8697,8698,8699,8700,8701,8702,8703,8704,8705,8706,8707,8708,8709, # 8710
8710,8711,8712,8713,8714,8715,8716,8717,8718,8719,8720,8721,8722,8723,8724,8725, # 8726 8710,8711,8712,8713,8714,8715,8716,8717,8718,8719,8720,8721,8722,8723,8724,8725, # 8726
8726,8727,8728,8729,8730,8731,8732,8733,8734,8735,8736,8737,8738,8739,8740,8741) # 8742 8726,8727,8728,8729,8730,8731,8732,8733,8734,8735,8736,8737,8738,8739,8740,8741) # 8742
# flake8: noqa
+4 -4
View File
@@ -25,10 +25,10 @@
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
from mbcharsetprober import MultiByteCharSetProber from .mbcharsetprober import MultiByteCharSetProber
from codingstatemachine import CodingStateMachine from .codingstatemachine import CodingStateMachine
from chardistribution import EUCTWDistributionAnalysis from .chardistribution import EUCTWDistributionAnalysis
from mbcssm import EUCTWSMModel from .mbcssm import EUCTWSMModel
class EUCTWProber(MultiByteCharSetProber): class EUCTWProber(MultiByteCharSetProber):
def __init__(self): def __init__(self):
+5 -4
View File
@@ -13,12 +13,12 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
@@ -36,14 +36,14 @@
# #
# Ideal Distribution Ratio = 0.79135/(1-0.79135) = 3.79 # Ideal Distribution Ratio = 0.79135/(1-0.79135) = 3.79
# Random Distribution Ration = 512 / (3755 - 512) = 0.157 # Random Distribution Ration = 512 / (3755 - 512) = 0.157
# #
# Typical Distribution Ratio about 25% of Ideal one, still much higher that RDR # Typical Distribution Ratio about 25% of Ideal one, still much higher that RDR
GB2312_TYPICAL_DISTRIBUTION_RATIO = 0.9 GB2312_TYPICAL_DISTRIBUTION_RATIO = 0.9
GB2312_TABLE_SIZE = 3760 GB2312_TABLE_SIZE = 3760
GB2312CharToFreqOrder = ( \ GB2312CharToFreqOrder = (
1671, 749,1443,2364,3924,3807,2330,3921,1704,3463,2691,1511,1515, 572,3191,2205, 1671, 749,1443,2364,3924,3807,2330,3921,1704,3463,2691,1511,1515, 572,3191,2205,
2361, 224,2558, 479,1711, 963,3162, 440,4060,1905,2966,2947,3580,2647,3961,3842, 2361, 224,2558, 479,1711, 963,3162, 440,4060,1905,2966,2947,3580,2647,3961,3842,
2204, 869,4207, 970,2678,5626,2944,2956,1479,4048, 514,3595, 588,1346,2820,3409, 2204, 869,4207, 970,2678,5626,2944,2956,1479,4048, 514,3595, 588,1346,2820,3409,
@@ -469,3 +469,4 @@ GB2312CharToFreqOrder = ( \
5867,5507,6273,4206,6274,4789,6098,6764,3619,3646,3833,3804,2394,3788,4936,3978, 5867,5507,6273,4206,6274,4789,6098,6764,3619,3646,3833,3804,2394,3788,4936,3978,
4866,4899,6099,6100,5559,6478,6765,3599,5868,6101,5869,5870,6275,6766,4527,6767) 4866,4899,6099,6100,5559,6478,6765,3599,5868,6101,5869,5870,6275,6766,4527,6767)
# flake8: noqa
+4 -4
View File
@@ -25,10 +25,10 @@
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
from mbcharsetprober import MultiByteCharSetProber from .mbcharsetprober import MultiByteCharSetProber
from codingstatemachine import CodingStateMachine from .codingstatemachine import CodingStateMachine
from chardistribution import GB2312DistributionAnalysis from .chardistribution import GB2312DistributionAnalysis
from mbcssm import GB2312SMModel from .mbcssm import GB2312SMModel
class GB2312Prober(MultiByteCharSetProber): class GB2312Prober(MultiByteCharSetProber):
def __init__(self): def __init__(self):
+99 -85
View File
@@ -13,20 +13,21 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
from charsetprober import CharSetProber from .charsetprober import CharSetProber
import constants from .constants import eNotMe, eDetecting
from .compat import wrap_ord
# This prober doesn't actually recognize a language or a charset. # This prober doesn't actually recognize a language or a charset.
# It is a helper prober for the use of the Hebrew model probers # It is a helper prober for the use of the Hebrew model probers
@@ -35,40 +36,40 @@ import constants
# #
# Four main charsets exist in Hebrew: # Four main charsets exist in Hebrew:
# "ISO-8859-8" - Visual Hebrew # "ISO-8859-8" - Visual Hebrew
# "windows-1255" - Logical Hebrew # "windows-1255" - Logical Hebrew
# "ISO-8859-8-I" - Logical Hebrew # "ISO-8859-8-I" - Logical Hebrew
# "x-mac-hebrew" - ?? Logical Hebrew ?? # "x-mac-hebrew" - ?? Logical Hebrew ??
# #
# Both "ISO" charsets use a completely identical set of code points, whereas # Both "ISO" charsets use a completely identical set of code points, whereas
# "windows-1255" and "x-mac-hebrew" are two different proper supersets of # "windows-1255" and "x-mac-hebrew" are two different proper supersets of
# these code points. windows-1255 defines additional characters in the range # these code points. windows-1255 defines additional characters in the range
# 0x80-0x9F as some misc punctuation marks as well as some Hebrew-specific # 0x80-0x9F as some misc punctuation marks as well as some Hebrew-specific
# diacritics and additional 'Yiddish' ligature letters in the range 0xc0-0xd6. # diacritics and additional 'Yiddish' ligature letters in the range 0xc0-0xd6.
# x-mac-hebrew defines similar additional code points but with a different # x-mac-hebrew defines similar additional code points but with a different
# mapping. # mapping.
# #
# As far as an average Hebrew text with no diacritics is concerned, all four # As far as an average Hebrew text with no diacritics is concerned, all four
# charsets are identical with respect to code points. Meaning that for the # charsets are identical with respect to code points. Meaning that for the
# main Hebrew alphabet, all four map the same values to all 27 Hebrew letters # main Hebrew alphabet, all four map the same values to all 27 Hebrew letters
# (including final letters). # (including final letters).
# #
# The dominant difference between these charsets is their directionality. # The dominant difference between these charsets is their directionality.
# "Visual" directionality means that the text is ordered as if the renderer is # "Visual" directionality means that the text is ordered as if the renderer is
# not aware of a BIDI rendering algorithm. The renderer sees the text and # not aware of a BIDI rendering algorithm. The renderer sees the text and
# draws it from left to right. The text itself when ordered naturally is read # draws it from left to right. The text itself when ordered naturally is read
# backwards. A buffer of Visual Hebrew generally looks like so: # backwards. A buffer of Visual Hebrew generally looks like so:
# "[last word of first line spelled backwards] [whole line ordered backwards # "[last word of first line spelled backwards] [whole line ordered backwards
# and spelled backwards] [first word of first line spelled backwards] # and spelled backwards] [first word of first line spelled backwards]
# [end of line] [last word of second line] ... etc' " # [end of line] [last word of second line] ... etc' "
# adding punctuation marks, numbers and English text to visual text is # adding punctuation marks, numbers and English text to visual text is
# naturally also "visual" and from left to right. # naturally also "visual" and from left to right.
# #
# "Logical" directionality means the text is ordered "naturally" according to # "Logical" directionality means the text is ordered "naturally" according to
# the order it is read. It is the responsibility of the renderer to display # the order it is read. It is the responsibility of the renderer to display
# the text from right to left. A BIDI algorithm is used to place general # the text from right to left. A BIDI algorithm is used to place general
# punctuation marks, numbers and English text in the text. # punctuation marks, numbers and English text in the text.
# #
# Texts in x-mac-hebrew are almost impossible to find on the Internet. From # Texts in x-mac-hebrew are almost impossible to find on the Internet. From
# what little evidence I could find, it seems that its general directionality # what little evidence I could find, it seems that its general directionality
# is Logical. # is Logical.
# #
@@ -76,17 +77,17 @@ import constants
# charsets: # charsets:
# Visual Hebrew - "ISO-8859-8" - backwards text - Words and sentences are # Visual Hebrew - "ISO-8859-8" - backwards text - Words and sentences are
# backwards while line order is natural. For charset recognition purposes # backwards while line order is natural. For charset recognition purposes
# the line order is unimportant (In fact, for this implementation, even # the line order is unimportant (In fact, for this implementation, even
# word order is unimportant). # word order is unimportant).
# Logical Hebrew - "windows-1255" - normal, naturally ordered text. # Logical Hebrew - "windows-1255" - normal, naturally ordered text.
# #
# "ISO-8859-8-I" is a subset of windows-1255 and doesn't need to be # "ISO-8859-8-I" is a subset of windows-1255 and doesn't need to be
# specifically identified. # specifically identified.
# "x-mac-hebrew" is also identified as windows-1255. A text in x-mac-hebrew # "x-mac-hebrew" is also identified as windows-1255. A text in x-mac-hebrew
# that contain special punctuation marks or diacritics is displayed with # that contain special punctuation marks or diacritics is displayed with
# some unconverted characters showing as question marks. This problem might # some unconverted characters showing as question marks. This problem might
# be corrected using another model prober for x-mac-hebrew. Due to the fact # be corrected using another model prober for x-mac-hebrew. Due to the fact
# that x-mac-hebrew texts are so rare, writing another model prober isn't # that x-mac-hebrew texts are so rare, writing another model prober isn't
# worth the effort and performance hit. # worth the effort and performance hit.
# #
#### The Prober #### #### The Prober ####
@@ -126,28 +127,31 @@ import constants
# charset identified, either "windows-1255" or "ISO-8859-8". # charset identified, either "windows-1255" or "ISO-8859-8".
# windows-1255 / ISO-8859-8 code points of interest # windows-1255 / ISO-8859-8 code points of interest
FINAL_KAF = '\xea' FINAL_KAF = 0xea
NORMAL_KAF = '\xeb' NORMAL_KAF = 0xeb
FINAL_MEM = '\xed' FINAL_MEM = 0xed
NORMAL_MEM = '\xee' NORMAL_MEM = 0xee
FINAL_NUN = '\xef' FINAL_NUN = 0xef
NORMAL_NUN = '\xf0' NORMAL_NUN = 0xf0
FINAL_PE = '\xf3' FINAL_PE = 0xf3
NORMAL_PE = '\xf4' NORMAL_PE = 0xf4
FINAL_TSADI = '\xf5' FINAL_TSADI = 0xf5
NORMAL_TSADI = '\xf6' NORMAL_TSADI = 0xf6
# Minimum Visual vs Logical final letter score difference. # Minimum Visual vs Logical final letter score difference.
# If the difference is below this, don't rely solely on the final letter score distance. # If the difference is below this, don't rely solely on the final letter score
# distance.
MIN_FINAL_CHAR_DISTANCE = 5 MIN_FINAL_CHAR_DISTANCE = 5
# Minimum Visual vs Logical model score difference. # Minimum Visual vs Logical model score difference.
# If the difference is below this, don't rely at all on the model score distance. # If the difference is below this, don't rely at all on the model score
# distance.
MIN_MODEL_DISTANCE = 0.01 MIN_MODEL_DISTANCE = 0.01
VISUAL_HEBREW_NAME = "ISO-8859-8" VISUAL_HEBREW_NAME = "ISO-8859-8"
LOGICAL_HEBREW_NAME = "windows-1255" LOGICAL_HEBREW_NAME = "windows-1255"
class HebrewProber(CharSetProber): class HebrewProber(CharSetProber):
def __init__(self): def __init__(self):
CharSetProber.__init__(self) CharSetProber.__init__(self)
@@ -159,84 +163,91 @@ class HebrewProber(CharSetProber):
self._mFinalCharLogicalScore = 0 self._mFinalCharLogicalScore = 0
self._mFinalCharVisualScore = 0 self._mFinalCharVisualScore = 0
# The two last characters seen in the previous buffer, # The two last characters seen in the previous buffer,
# mPrev and mBeforePrev are initialized to space in order to simulate a word # mPrev and mBeforePrev are initialized to space in order to simulate
# delimiter at the beginning of the data # a word delimiter at the beginning of the data
self._mPrev = ' ' self._mPrev = ' '
self._mBeforePrev = ' ' self._mBeforePrev = ' '
# These probers are owned by the group prober. # These probers are owned by the group prober.
def set_model_probers(self, logicalProber, visualProber): def set_model_probers(self, logicalProber, visualProber):
self._mLogicalProber = logicalProber self._mLogicalProber = logicalProber
self._mVisualProber = visualProber self._mVisualProber = visualProber
def is_final(self, c): def is_final(self, c):
return c in [FINAL_KAF, FINAL_MEM, FINAL_NUN, FINAL_PE, FINAL_TSADI] return wrap_ord(c) in [FINAL_KAF, FINAL_MEM, FINAL_NUN, FINAL_PE,
FINAL_TSADI]
def is_non_final(self, c): def is_non_final(self, c):
# The normal Tsadi is not a good Non-Final letter due to words like # The normal Tsadi is not a good Non-Final letter due to words like
# 'lechotet' (to chat) containing an apostrophe after the tsadi. This # 'lechotet' (to chat) containing an apostrophe after the tsadi. This
# apostrophe is converted to a space in FilterWithoutEnglishLetters causing # apostrophe is converted to a space in FilterWithoutEnglishLetters
# the Non-Final tsadi to appear at an end of a word even though this is not # causing the Non-Final tsadi to appear at an end of a word even
# the case in the original text. # though this is not the case in the original text.
# The letters Pe and Kaf rarely display a related behavior of not being a # The letters Pe and Kaf rarely display a related behavior of not being
# good Non-Final letter. Words like 'Pop', 'Winamp' and 'Mubarak' for # a good Non-Final letter. Words like 'Pop', 'Winamp' and 'Mubarak'
# example legally end with a Non-Final Pe or Kaf. However, the benefit of # for example legally end with a Non-Final Pe or Kaf. However, the
# these letters as Non-Final letters outweighs the damage since these words # benefit of these letters as Non-Final letters outweighs the damage
# are quite rare. # since these words are quite rare.
return c in [NORMAL_KAF, NORMAL_MEM, NORMAL_NUN, NORMAL_PE] return wrap_ord(c) in [NORMAL_KAF, NORMAL_MEM, NORMAL_NUN, NORMAL_PE]
def feed(self, aBuf): def feed(self, aBuf):
# Final letter analysis for logical-visual decision. # Final letter analysis for logical-visual decision.
# Look for evidence that the received buffer is either logical Hebrew or # Look for evidence that the received buffer is either logical Hebrew
# visual Hebrew. # or visual Hebrew.
# The following cases are checked: # The following cases are checked:
# 1) A word longer than 1 letter, ending with a final letter. This is an # 1) A word longer than 1 letter, ending with a final letter. This is
# indication that the text is laid out "naturally" since the final letter # an indication that the text is laid out "naturally" since the
# really appears at the end. +1 for logical score. # final letter really appears at the end. +1 for logical score.
# 2) A word longer than 1 letter, ending with a Non-Final letter. In normal # 2) A word longer than 1 letter, ending with a Non-Final letter. In
# Hebrew, words ending with Kaf, Mem, Nun, Pe or Tsadi, should not end with # normal Hebrew, words ending with Kaf, Mem, Nun, Pe or Tsadi,
# the Non-Final form of that letter. Exceptions to this rule are mentioned # should not end with the Non-Final form of that letter. Exceptions
# above in isNonFinal(). This is an indication that the text is laid out # to this rule are mentioned above in isNonFinal(). This is an
# backwards. +1 for visual score # indication that the text is laid out backwards. +1 for visual
# 3) A word longer than 1 letter, starting with a final letter. Final letters # score
# should not appear at the beginning of a word. This is an indication that # 3) A word longer than 1 letter, starting with a final letter. Final
# the text is laid out backwards. +1 for visual score. # letters should not appear at the beginning of a word. This is an
# # indication that the text is laid out backwards. +1 for visual
# The visual score and logical score are accumulated throughout the text and # score.
# are finally checked against each other in GetCharSetName(). #
# No checking for final letters in the middle of words is done since that case # The visual score and logical score are accumulated throughout the
# is not an indication for either Logical or Visual text. # text and are finally checked against each other in GetCharSetName().
# # No checking for final letters in the middle of words is done since
# We automatically filter out all 7-bit characters (replace them with spaces) # that case is not an indication for either Logical or Visual text.
# so the word boundary detection works properly. [MAP] #
# We automatically filter out all 7-bit characters (replace them with
# spaces) so the word boundary detection works properly. [MAP]
if self.get_state() == constants.eNotMe: if self.get_state() == eNotMe:
# Both model probers say it's not them. No reason to continue. # Both model probers say it's not them. No reason to continue.
return constants.eNotMe return eNotMe
aBuf = self.filter_high_bit_only(aBuf) aBuf = self.filter_high_bit_only(aBuf)
for cur in aBuf: for cur in aBuf:
if cur == ' ': if cur == ' ':
# We stand on a space - a word just ended # We stand on a space - a word just ended
if self._mBeforePrev != ' ': if self._mBeforePrev != ' ':
# next-to-last char was not a space so self._mPrev is not a 1 letter word # next-to-last char was not a space so self._mPrev is not a
# 1 letter word
if self.is_final(self._mPrev): if self.is_final(self._mPrev):
# case (1) [-2:not space][-1:final letter][cur:space] # case (1) [-2:not space][-1:final letter][cur:space]
self._mFinalCharLogicalScore += 1 self._mFinalCharLogicalScore += 1
elif self.is_non_final(self._mPrev): elif self.is_non_final(self._mPrev):
# case (2) [-2:not space][-1:Non-Final letter][cur:space] # case (2) [-2:not space][-1:Non-Final letter][
# cur:space]
self._mFinalCharVisualScore += 1 self._mFinalCharVisualScore += 1
else: else:
# Not standing on a space # Not standing on a space
if (self._mBeforePrev == ' ') and (self.is_final(self._mPrev)) and (cur != ' '): if ((self._mBeforePrev == ' ') and
(self.is_final(self._mPrev)) and (cur != ' ')):
# case (3) [-2:space][-1:final letter][cur:not space] # case (3) [-2:space][-1:final letter][cur:not space]
self._mFinalCharVisualScore += 1 self._mFinalCharVisualScore += 1
self._mBeforePrev = self._mPrev self._mBeforePrev = self._mPrev
self._mPrev = cur self._mPrev = cur
# Forever detecting, till the end or until both model probers return eNotMe (handled above) # Forever detecting, till the end or until both model probers return
return constants.eDetecting # eNotMe (handled above)
return eDetecting
def get_charset_name(self): def get_charset_name(self):
# Make the decision: is it Logical or Visual? # Make the decision: is it Logical or Visual?
@@ -248,22 +259,25 @@ class HebrewProber(CharSetProber):
return VISUAL_HEBREW_NAME return VISUAL_HEBREW_NAME
# It's not dominant enough, try to rely on the model scores instead. # It's not dominant enough, try to rely on the model scores instead.
modelsub = self._mLogicalProber.get_confidence() - self._mVisualProber.get_confidence() modelsub = (self._mLogicalProber.get_confidence()
- self._mVisualProber.get_confidence())
if modelsub > MIN_MODEL_DISTANCE: if modelsub > MIN_MODEL_DISTANCE:
return LOGICAL_HEBREW_NAME return LOGICAL_HEBREW_NAME
if modelsub < -MIN_MODEL_DISTANCE: if modelsub < -MIN_MODEL_DISTANCE:
return VISUAL_HEBREW_NAME return VISUAL_HEBREW_NAME
# Still no good, back to final letter distance, maybe it'll save the day. # Still no good, back to final letter distance, maybe it'll save the
# day.
if finalsub < 0.0: if finalsub < 0.0:
return VISUAL_HEBREW_NAME return VISUAL_HEBREW_NAME
# (finalsub > 0 - Logical) or (don't know what to do) default to Logical. # (finalsub > 0 - Logical) or (don't know what to do) default to
# Logical.
return LOGICAL_HEBREW_NAME return LOGICAL_HEBREW_NAME
def get_state(self): def get_state(self):
# Remain active as long as any of the model probers are active. # Remain active as long as any of the model probers are active.
if (self._mLogicalProber.get_state() == constants.eNotMe) and \ if (self._mLogicalProber.get_state() == eNotMe) and \
(self._mVisualProber.get_state() == constants.eNotMe): (self._mVisualProber.get_state() == eNotMe):
return constants.eNotMe return eNotMe
return constants.eDetecting return eDetecting
+9 -7
View File
@@ -13,12 +13,12 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
@@ -28,7 +28,7 @@
# Sampling from about 20M text materials include literature and computer technology # Sampling from about 20M text materials include literature and computer technology
# #
# Japanese frequency table, applied to both S-JIS and EUC-JP # Japanese frequency table, applied to both S-JIS and EUC-JP
# They are sorted in order. # They are sorted in order.
# 128 --> 0.77094 # 128 --> 0.77094
# 256 --> 0.85710 # 256 --> 0.85710
@@ -38,15 +38,15 @@
# #
# Ideal Distribution Ratio = 0.92635 / (1-0.92635) = 12.58 # Ideal Distribution Ratio = 0.92635 / (1-0.92635) = 12.58
# Random Distribution Ration = 512 / (2965+62+83+86-512) = 0.191 # Random Distribution Ration = 512 / (2965+62+83+86-512) = 0.191
# #
# Typical Distribution Ratio, 25% of IDR # Typical Distribution Ratio, 25% of IDR
JIS_TYPICAL_DISTRIBUTION_RATIO = 3.0 JIS_TYPICAL_DISTRIBUTION_RATIO = 3.0
# Char to FreqOrder table , # Char to FreqOrder table ,
JIS_TABLE_SIZE = 4368 JIS_TABLE_SIZE = 4368
JISCharToFreqOrder = ( \ JISCharToFreqOrder = (
40, 1, 6, 182, 152, 180, 295,2127, 285, 381,3295,4304,3068,4606,3165,3510, # 16 40, 1, 6, 182, 152, 180, 295,2127, 285, 381,3295,4304,3068,4606,3165,3510, # 16
3511,1822,2785,4607,1193,2226,5070,4608, 171,2996,1247, 18, 179,5071, 856,1661, # 32 3511,1822,2785,4607,1193,2226,5070,4608, 171,2996,1247, 18, 179,5071, 856,1661, # 32
1262,5072, 619, 127,3431,3512,3230,1899,1700, 232, 228,1294,1298, 284, 283,2041, # 48 1262,5072, 619, 127,3431,3512,3230,1899,1700, 232, 228,1294,1298, 284, 283,2041, # 48
@@ -565,3 +565,5 @@ JISCharToFreqOrder = ( \
8224,8225,8226,8227,8228,8229,8230,8231,8232,8233,8234,8235,8236,8237,8238,8239, # 8240 8224,8225,8226,8227,8228,8229,8230,8231,8232,8233,8234,8235,8236,8237,8238,8239, # 8240
8240,8241,8242,8243,8244,8245,8246,8247,8248,8249,8250,8251,8252,8253,8254,8255, # 8256 8240,8241,8242,8243,8244,8245,8246,8247,8248,8249,8250,8251,8252,8253,8254,8255, # 8256
8256,8257,8258,8259,8260,8261,8262,8263,8264,8265,8266,8267,8268,8269,8270,8271) # 8272 8256,8257,8258,8259,8260,8261,8262,8263,8264,8265,8266,8267,8268,8269,8270,8271) # 8272
# flake8: noqa
+49 -40
View File
@@ -13,19 +13,19 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants from .compat import wrap_ord
NUM_OF_CATEGORY = 6 NUM_OF_CATEGORY = 6
DONT_KNOW = -1 DONT_KNOW = -1
@@ -34,7 +34,7 @@ MAX_REL_THRESHOLD = 1000
MINIMUM_DATA_THRESHOLD = 4 MINIMUM_DATA_THRESHOLD = 4
# This is hiragana 2-char sequence table, the number in each cell represents its frequency category # This is hiragana 2-char sequence table, the number in each cell represents its frequency category
jp2CharContext = ( \ jp2CharContext = (
(0,0,0,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,1,0,0,0,0,0,0,0,0,0,0,1), (0,0,0,2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,1,0,0,0,0,0,0,0,0,0,0,1),
(2,4,0,4,0,3,0,4,0,3,4,4,4,2,4,3,3,4,3,2,3,3,4,2,3,3,3,2,4,1,4,3,3,1,5,4,3,4,3,4,3,5,3,0,3,5,4,2,0,3,1,0,3,3,0,3,3,0,1,1,0,4,3,0,3,3,0,4,0,2,0,3,5,5,5,5,4,0,4,1,0,3,4), (2,4,0,4,0,3,0,4,0,3,4,4,4,2,4,3,3,4,3,2,3,3,4,2,3,3,3,2,4,1,4,3,3,1,5,4,3,4,3,4,3,5,3,0,3,5,4,2,0,3,1,0,3,3,0,3,3,0,1,1,0,4,3,0,3,3,0,4,0,2,0,3,5,5,5,5,4,0,4,1,0,3,4),
(0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,2), (0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,2),
@@ -123,26 +123,33 @@ jp2CharContext = ( \
class JapaneseContextAnalysis: class JapaneseContextAnalysis:
def __init__(self): def __init__(self):
self.reset() self.reset()
def reset(self): def reset(self):
self._mTotalRel = 0 # total sequence received self._mTotalRel = 0 # total sequence received
self._mRelSample = [0] * NUM_OF_CATEGORY # category counters, each interger counts sequence in its category # category counters, each interger counts sequence in its category
self._mNeedToSkipCharNum = 0 # if last byte in current buffer is not the last byte of a character, we need to know how many bytes to skip in next buffer self._mRelSample = [0] * NUM_OF_CATEGORY
self._mLastCharOrder = -1 # The order of previous char # if last byte in current buffer is not the last byte of a character,
self._mDone = constants.False # If this flag is set to constants.True, detection is done and conclusion has been made # we need to know how many bytes to skip in next buffer
self._mNeedToSkipCharNum = 0
self._mLastCharOrder = -1 # The order of previous char
# If this flag is set to True, detection is done and conclusion has
# been made
self._mDone = False
def feed(self, aBuf, aLen): def feed(self, aBuf, aLen):
if self._mDone: return if self._mDone:
return
# The buffer we got is byte oriented, and a character may span in more than one # The buffer we got is byte oriented, and a character may span in more than one
# buffers. In case the last one or two byte in last buffer is not complete, we # buffers. In case the last one or two byte in last buffer is not
# record how many byte needed to complete that character and skip these bytes here. # complete, we record how many byte needed to complete that character
# We can choose to record those bytes as well and analyse the character once it # and skip these bytes here. We can choose to record those bytes as
# is complete, but since a character will not make much difference, by simply skipping # well and analyse the character once it is complete, but since a
# character will not make much difference, by simply skipping
# this character will simply our logic and improve performance. # this character will simply our logic and improve performance.
i = self._mNeedToSkipCharNum i = self._mNeedToSkipCharNum
while i < aLen: while i < aLen:
order, charLen = self.get_order(aBuf[i:i+2]) order, charLen = self.get_order(aBuf[i:i + 2])
i += charLen i += charLen
if i > aLen: if i > aLen:
self._mNeedToSkipCharNum = i - aLen self._mNeedToSkipCharNum = i - aLen
@@ -151,14 +158,14 @@ class JapaneseContextAnalysis:
if (order != -1) and (self._mLastCharOrder != -1): if (order != -1) and (self._mLastCharOrder != -1):
self._mTotalRel += 1 self._mTotalRel += 1
if self._mTotalRel > MAX_REL_THRESHOLD: if self._mTotalRel > MAX_REL_THRESHOLD:
self._mDone = constants.True self._mDone = True
break break
self._mRelSample[jp2CharContext[self._mLastCharOrder][order]] += 1 self._mRelSample[jp2CharContext[self._mLastCharOrder][order]] += 1
self._mLastCharOrder = order self._mLastCharOrder = order
def got_enough_data(self): def got_enough_data(self):
return self._mTotalRel > ENOUGH_REL_THRESHOLD return self._mTotalRel > ENOUGH_REL_THRESHOLD
def get_confidence(self): def get_confidence(self):
# This is just one way to calculate confidence. It works well for me. # This is just one way to calculate confidence. It works well for me.
if self._mTotalRel > MINIMUM_DATA_THRESHOLD: if self._mTotalRel > MINIMUM_DATA_THRESHOLD:
@@ -166,45 +173,47 @@ class JapaneseContextAnalysis:
else: else:
return DONT_KNOW return DONT_KNOW
def get_order(self, aStr): def get_order(self, aBuf):
return -1, 1 return -1, 1
class SJISContextAnalysis(JapaneseContextAnalysis): class SJISContextAnalysis(JapaneseContextAnalysis):
def get_order(self, aStr): def get_order(self, aBuf):
if not aStr: return -1, 1 if not aBuf:
return -1, 1
# find out current char's byte length # find out current char's byte length
if ((aStr[0] >= '\x81') and (aStr[0] <= '\x9F')) or \ first_char = wrap_ord(aBuf[0])
((aStr[0] >= '\xE0') and (aStr[0] <= '\xFC')): if ((0x81 <= first_char <= 0x9F) or (0xE0 <= first_char <= 0xFC)):
charLen = 2 charLen = 2
else: else:
charLen = 1 charLen = 1
# return its order if it is hiragana # return its order if it is hiragana
if len(aStr) > 1: if len(aBuf) > 1:
if (aStr[0] == '\202') and \ second_char = wrap_ord(aBuf[1])
(aStr[1] >= '\x9F') and \ if (first_char == 202) and (0x9F <= second_char <= 0xF1):
(aStr[1] <= '\xF1'): return second_char - 0x9F, charLen
return ord(aStr[1]) - 0x9F, charLen
return -1, charLen return -1, charLen
class EUCJPContextAnalysis(JapaneseContextAnalysis): class EUCJPContextAnalysis(JapaneseContextAnalysis):
def get_order(self, aStr): def get_order(self, aBuf):
if not aStr: return -1, 1 if not aBuf:
return -1, 1
# find out current char's byte length # find out current char's byte length
if (aStr[0] == '\x8E') or \ first_char = wrap_ord(aBuf[0])
((aStr[0] >= '\xA1') and (aStr[0] <= '\xFE')): if (first_char == 0x8E) or (0xA1 <= first_char <= 0xFE):
charLen = 2 charLen = 2
elif aStr[0] == '\x8F': elif first_char == 0x8F:
charLen = 3 charLen = 3
else: else:
charLen = 1 charLen = 1
# return its order if it is hiragana # return its order if it is hiragana
if len(aStr) > 1: if len(aBuf) > 1:
if (aStr[0] == '\xA4') and \ second_char = wrap_ord(aBuf[1])
(aStr[1] >= '\xA1') and \ if (first_char == 0xA4) and (0xA1 <= second_char <= 0xF3):
(aStr[1] <= '\xF3'): return second_char - 0xA1, charLen
return ord(aStr[1]) - 0xA1, charLen
return -1, charLen return -1, charLen
# flake8: noqa
+15 -14
View File
@@ -13,30 +13,28 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants
# 255: Control characters that usually does not exist in any text # 255: Control characters that usually does not exist in any text
# 254: Carriage/Return # 254: Carriage/Return
# 253: symbol (punctuation) that does not belong to word # 253: symbol (punctuation) that does not belong to word
# 252: 0 - 9 # 252: 0 - 9
# Character Mapping Table: # Character Mapping Table:
# this table is modified base on win1251BulgarianCharToOrderMap, so # this table is modified base on win1251BulgarianCharToOrderMap, so
# only number <64 is sure valid # only number <64 is sure valid
Latin5_BulgarianCharToOrderMap = ( \ Latin5_BulgarianCharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -55,7 +53,7 @@ Latin5_BulgarianCharToOrderMap = ( \
62,242,243,244, 58,245, 98,246,247,248,249,250,251, 91,252,253, # f0 62,242,243,244, 58,245, 98,246,247,248,249,250,251, 91,252,253, # f0
) )
win1251BulgarianCharToOrderMap = ( \ win1251BulgarianCharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -74,13 +72,13 @@ win1251BulgarianCharToOrderMap = ( \
7, 8, 5, 19, 29, 25, 22, 21, 27, 24, 17, 75, 52,253, 42, 16, # f0 7, 8, 5, 19, 29, 25, 22, 21, 27, 24, 17, 75, 52,253, 42, 16, # f0
) )
# Model Table: # Model Table:
# total sequences: 100% # total sequences: 100%
# first 512 sequences: 96.9392% # first 512 sequences: 96.9392%
# first 1024 sequences:3.0618% # first 1024 sequences:3.0618%
# rest sequences: 0.2992% # rest sequences: 0.2992%
# negative sequences: 0.0020% # negative sequences: 0.0020%
BulgarianLangModel = ( \ BulgarianLangModel = (
0,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,2,3,3,3,3,3,3,3,3,2,3,3,3,3,3, 0,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,2,3,3,3,3,3,3,3,3,2,3,3,3,3,3,
3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,0,3,3,3,2,2,3,2,2,1,2,2, 3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,0,3,3,3,2,2,3,2,2,1,2,2,
3,1,3,3,2,3,3,3,3,3,3,3,3,3,3,3,3,0,3,3,3,3,3,3,3,3,3,3,0,3,0,1, 3,1,3,3,2,3,3,3,3,3,3,3,3,3,3,3,3,0,3,3,3,3,3,3,3,3,3,3,0,3,0,1,
@@ -211,18 +209,21 @@ BulgarianLangModel = ( \
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,
) )
Latin5BulgarianModel = { \ Latin5BulgarianModel = {
'charToOrderMap': Latin5_BulgarianCharToOrderMap, 'charToOrderMap': Latin5_BulgarianCharToOrderMap,
'precedenceMatrix': BulgarianLangModel, 'precedenceMatrix': BulgarianLangModel,
'mTypicalPositiveRatio': 0.969392, 'mTypicalPositiveRatio': 0.969392,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "ISO-8859-5" 'charsetName': "ISO-8859-5"
} }
Win1251BulgarianModel = { \ Win1251BulgarianModel = {
'charToOrderMap': win1251BulgarianCharToOrderMap, 'charToOrderMap': win1251BulgarianCharToOrderMap,
'precedenceMatrix': BulgarianLangModel, 'precedenceMatrix': BulgarianLangModel,
'mTypicalPositiveRatio': 0.969392, 'mTypicalPositiveRatio': 0.969392,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "windows-1251" 'charsetName': "windows-1251"
} }
# flake8: noqa
+25 -25
View File
@@ -13,23 +13,21 @@
# modify it under the terms of the GNU Lesser General Public # modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either # License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version. # version 2.1 of the License, or (at your option) any later version.
# #
# This library is distributed in the hope that it will be useful, # This library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of # but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
# Lesser General Public License for more details. # Lesser General Public License for more details.
# #
# You should have received a copy of the GNU Lesser General Public # You should have received a copy of the GNU Lesser General Public
# License along with this library; if not, write to the Free Software # License along with this library; if not, write to the Free Software
# Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA
# 02110-1301 USA # 02110-1301 USA
######################### END LICENSE BLOCK ######################### ######################### END LICENSE BLOCK #########################
import constants
# KOI8-R language model # KOI8-R language model
# Character Mapping Table: # Character Mapping Table:
KOI8R_CharToOrderMap = ( \ KOI8R_CharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -48,7 +46,7 @@ KOI8R_CharToOrderMap = ( \
35, 43, 45, 32, 40, 52, 56, 33, 61, 62, 51, 57, 47, 63, 50, 70, # f0 35, 43, 45, 32, 40, 52, 56, 33, 61, 62, 51, 57, 47, 63, 50, 70, # f0
) )
win1251_CharToOrderMap = ( \ win1251_CharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -67,7 +65,7 @@ win1251_CharToOrderMap = ( \
9, 7, 6, 14, 39, 26, 28, 22, 25, 29, 54, 18, 17, 30, 27, 16, 9, 7, 6, 14, 39, 26, 28, 22, 25, 29, 54, 18, 17, 30, 27, 16,
) )
latin5_CharToOrderMap = ( \ latin5_CharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -86,7 +84,7 @@ latin5_CharToOrderMap = ( \
239, 68,240,241,242,243,244,245,246,247,248,249,250,251,252,255, 239, 68,240,241,242,243,244,245,246,247,248,249,250,251,252,255,
) )
macCyrillic_CharToOrderMap = ( \ macCyrillic_CharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -105,7 +103,7 @@ macCyrillic_CharToOrderMap = ( \
9, 7, 6, 14, 39, 26, 28, 22, 25, 29, 54, 18, 17, 30, 27,255, 9, 7, 6, 14, 39, 26, 28, 22, 25, 29, 54, 18, 17, 30, 27,255,
) )
IBM855_CharToOrderMap = ( \ IBM855_CharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -124,7 +122,7 @@ IBM855_CharToOrderMap = ( \
250, 18, 62, 20, 51, 25, 57, 30, 47, 29, 63, 22, 50,251,252,255, 250, 18, 62, 20, 51, 25, 57, 30, 47, 29, 63, 22, 50,251,252,255,
) )
IBM866_CharToOrderMap = ( \ IBM866_CharToOrderMap = (
255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00 255,255,255,255,255,255,255,255,255,255,254,255,255,254,255,255, # 00
255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10 255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255, # 10
253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20 253,253,253,253,253,253,253,253,253,253,253,253,253,253,253,253, # 20
@@ -143,13 +141,13 @@ IBM866_CharToOrderMap = ( \
239, 68,240,241,242,243,244,245,246,247,248,249,250,251,252,255, 239, 68,240,241,242,243,244,245,246,247,248,249,250,251,252,255,
) )
# Model Table: # Model Table:
# total sequences: 100% # total sequences: 100%
# first 512 sequences: 97.6601% # first 512 sequences: 97.6601%
# first 1024 sequences: 2.3389% # first 1024 sequences: 2.3389%
# rest sequences: 0.1237% # rest sequences: 0.1237%
# negative sequences: 0.0009% # negative sequences: 0.0009%
RussianLangModel = ( \ RussianLangModel = (
0,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,1,1,3,3,3,3,1,3,3,3,2,3,2,3,3, 0,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,1,1,3,3,3,3,1,3,3,3,2,3,2,3,3,
3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,0,3,2,2,2,2,2,0,0,2, 3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,3,0,3,2,2,2,2,2,0,0,2,
3,3,3,2,3,3,3,3,3,3,3,3,3,3,2,3,3,0,0,3,3,3,3,3,3,3,3,3,2,3,2,0, 3,3,3,2,3,3,3,3,3,3,3,3,3,3,2,3,3,0,0,3,3,3,3,3,3,3,3,3,2,3,2,0,
@@ -280,50 +278,52 @@ RussianLangModel = ( \
0,0,0,0,0,1,0,0,0,0,1,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0, 0,0,0,0,0,1,0,0,0,0,1,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,
) )
Koi8rModel = { \ Koi8rModel = {
'charToOrderMap': KOI8R_CharToOrderMap, 'charToOrderMap': KOI8R_CharToOrderMap,
'precedenceMatrix': RussianLangModel, 'precedenceMatrix': RussianLangModel,
'mTypicalPositiveRatio': 0.976601, 'mTypicalPositiveRatio': 0.976601,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "KOI8-R" 'charsetName': "KOI8-R"
} }
Win1251CyrillicModel = { \ Win1251CyrillicModel = {
'charToOrderMap': win1251_CharToOrderMap, 'charToOrderMap': win1251_CharToOrderMap,
'precedenceMatrix': RussianLangModel, 'precedenceMatrix': RussianLangModel,
'mTypicalPositiveRatio': 0.976601, 'mTypicalPositiveRatio': 0.976601,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "windows-1251" 'charsetName': "windows-1251"
} }
Latin5CyrillicModel = { \ Latin5CyrillicModel = {
'charToOrderMap': latin5_CharToOrderMap, 'charToOrderMap': latin5_CharToOrderMap,
'precedenceMatrix': RussianLangModel, 'precedenceMatrix': RussianLangModel,
'mTypicalPositiveRatio': 0.976601, 'mTypicalPositiveRatio': 0.976601,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "ISO-8859-5" 'charsetName': "ISO-8859-5"
} }
MacCyrillicModel = { \ MacCyrillicModel = {
'charToOrderMap': macCyrillic_CharToOrderMap, 'charToOrderMap': macCyrillic_CharToOrderMap,
'precedenceMatrix': RussianLangModel, 'precedenceMatrix': RussianLangModel,
'mTypicalPositiveRatio': 0.976601, 'mTypicalPositiveRatio': 0.976601,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "MacCyrillic" 'charsetName': "MacCyrillic"
}; };
Ibm866Model = { \ Ibm866Model = {
'charToOrderMap': IBM866_CharToOrderMap, 'charToOrderMap': IBM866_CharToOrderMap,
'precedenceMatrix': RussianLangModel, 'precedenceMatrix': RussianLangModel,
'mTypicalPositiveRatio': 0.976601, 'mTypicalPositiveRatio': 0.976601,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "IBM866" 'charsetName': "IBM866"
} }
Ibm855Model = { \ Ibm855Model = {
'charToOrderMap': IBM855_CharToOrderMap, 'charToOrderMap': IBM855_CharToOrderMap,
'precedenceMatrix': RussianLangModel, 'precedenceMatrix': RussianLangModel,
'mTypicalPositiveRatio': 0.976601, 'mTypicalPositiveRatio': 0.976601,
'keepEnglishLetter': constants.False, 'keepEnglishLetter': False,
'charsetName': "IBM855" 'charsetName': "IBM855"
} }
# flake8: noqa

Some files were not shown because too many files have changed in this diff Show More