imdbPy update
This commit is contained in:
@@ -109,6 +109,10 @@ _old_cookie_uu = '3M3AXsquTU5Gur/Svik+ewflPm5Rk2ieY3BIPlLjyK3C0Dp9F8UoPgbTyKiGtZ
|
||||
_cookie_id = 'rH1jNAkjTlNXvHolvBVBsgaPICNZbNdjVjzFwzas9JRmusdjVoqBs/Hs12NR+1WFxEoR9bGKEDUg6sNlADqXwkas12N131Rwdb+UQNGKN8PWrNdjcdqBQVLq8mbGDHP3hqzxhbD692NQi9D0JjpBtRaPIbP1zNdjUOqENQYv1ADWrNcT9vyXU1'
|
||||
_cookie_uu = 'su4/m8cho4c6HP+W1qgq6wchOmhnF0w+lIWvHjRUPJ6nRA9sccEafjGADJ6hQGrMd4GKqLcz2X4z5+w+M4OIKnRn7FpENH7dxDQu3bQEHyx0ZEyeRFTPHfQEX03XF+yeN1dsPpcXaqjUZAw+lGRfXRQEfz3RIX9IgVEffdBAHw2wQXyf9xdMPrQELw0QNB8dsffsqcdQemjPB0w+moLcPh0JrKrHJ9hjBzdMPpcXTH7XRwwOk='
|
||||
|
||||
# imdbpy2010 account.
|
||||
#_cookie_id = 'QrCdxVi+L+WgqOLrQJJgBgRRXGInphxiBPU/YXSFDyExMFzCp6YcYgSVXyEUhS/xMID8wqemHGID4DlntwZ49vemP5UXsAxiJ4D6goSmHGIgNT9hMXBaRSF2vMS3phxB0bVfQiQlP1RxdrzhB6YcRHFASyIhQVowwXCKtDSlD2YhgRvxBsCKtGemHBKH9mxSI='
|
||||
#_cookie_uu = 'oiEo2yoJFCA2Zbn/o7Z1LAPIwotAu6QdALv3foDb1x5F/tdrFY63XkSfty4kntS8Y8jkHSDLt3406+d+JThEilPI0mtTaOQdA/t2/iErp22jaLdeVU5ya4PIREpj7HFdpzhEHadcIAngSER50IoHDpD6Bz4Qy3b+UIhE/hBbhz5Q63ceA2hEvhPo5B0FnrL9Q8jkWjDIbA0Au3d+AOtnXoCIRL4Q28c+UOtnXpP4RL4T6OQdA+6ijUCI5B0AW2d+UOtnXpPYRL4T6OQdA8jkTUOYlC0A=='
|
||||
|
||||
|
||||
class _FakeURLOpener(object):
|
||||
"""Fake URLOpener object, used to return empty strings instead of
|
||||
|
||||
@@ -225,8 +225,8 @@ class DOMHTMLMovieParser(DOMParserBase):
|
||||
postprocess=lambda x: x.strip()),
|
||||
Attribute(key="countries",
|
||||
path="./h5[starts-with(text(), " \
|
||||
"'Countr')]/..//a/text()",
|
||||
postprocess=makeSplitter(sep='\n')),
|
||||
"'Countr')]/../div[@class='info-content']//text()",
|
||||
postprocess=makeSplitter('|')),
|
||||
Attribute(key="language",
|
||||
path="./h5[starts-with(text(), " \
|
||||
"'Language')]/..//text()",
|
||||
@@ -541,11 +541,13 @@ class DOMHTMLPlotParser(DOMParserBase):
|
||||
|
||||
def _process_award(x):
|
||||
award = {}
|
||||
award['award'] = x.get('award').strip()
|
||||
if not award['award']:
|
||||
return {}
|
||||
award['year'] = x.get('year').strip()
|
||||
if award['year'] and award['year'].isdigit():
|
||||
award['year'] = int(award['year'])
|
||||
award['result'] = x.get('result').strip()
|
||||
award['award'] = x.get('award').strip()
|
||||
category = x.get('category').strip()
|
||||
if category:
|
||||
award['category'] = category
|
||||
@@ -649,6 +651,8 @@ class DOMHTMLAwardsParser(DOMParserBase):
|
||||
assigner = self.xpath(dom, "//a/text()")[0]
|
||||
for entry in data[key]:
|
||||
if not entry.has_key('name'):
|
||||
if not entry:
|
||||
continue
|
||||
# this is an award, not a recipient
|
||||
entry['assigner'] = assigner.strip()
|
||||
# find the recipients
|
||||
@@ -996,8 +1000,10 @@ class DOMHTMLRatingsParser(DOMParserBase):
|
||||
if votes:
|
||||
nd['number of votes'] = {}
|
||||
for i in xrange(1, 11):
|
||||
nd['number of votes'][int(votes[i]['ordinal'])] = \
|
||||
int(votes[i]['votes'].replace(',', ''))
|
||||
_ordinal = int(votes[i]['ordinal'])
|
||||
_strvts = votes[i]['votes'] or '0'
|
||||
nd['number of votes'][_ordinal] = \
|
||||
int(_strvts.replace(',', ''))
|
||||
mean = data.get('mean and median', '')
|
||||
if mean:
|
||||
means = self.re_means.findall(mean)
|
||||
@@ -1699,10 +1705,14 @@ class DOMHTMLEpisodesParser(DOMParserBase):
|
||||
try: season_key = int(season_key)
|
||||
except: pass
|
||||
nd[season_key] = {}
|
||||
ep_counter = 1
|
||||
for episode in data[key]:
|
||||
if not episode: continue
|
||||
episode_key = episode.get('episode')
|
||||
if episode_key is None: continue
|
||||
if not isinstance(episode_key, int):
|
||||
episode_key = ep_counter
|
||||
ep_counter += 1
|
||||
cast_key = 'Season %s, Episode %s:' % (season_key,
|
||||
episode_key)
|
||||
if data.has_key(cast_key):
|
||||
|
||||
@@ -63,85 +63,131 @@ class DOMHTMLMaindetailsParser(DOMParserBase):
|
||||
|
||||
_birth_attrs = [Attribute(key='birth date',
|
||||
path={
|
||||
'day': "./div/a[starts-with(@href, " \
|
||||
'day': ".//a[starts-with(@href, " \
|
||||
"'/date/')]/text()",
|
||||
'year': "./div/a[starts-with(@href, " \
|
||||
'year': ".//a[starts-with(@href, " \
|
||||
"'/search/name?birth_year=')]/text()"
|
||||
},
|
||||
postprocess=build_date),
|
||||
Attribute(key='birth notes',
|
||||
path="./div/a[starts-with(@href, " \
|
||||
Attribute(key='birth place',
|
||||
path=".//a[starts-with(@href, " \
|
||||
"'/search/name?birth_place=')]/text()")]
|
||||
_death_attrs = [Attribute(key='death date',
|
||||
path={
|
||||
'day': "./div/a[starts-with(@href, " \
|
||||
'day': ".//a[starts-with(@href, " \
|
||||
"'/date/')]/text()",
|
||||
'year': "./div/a[starts-with(@href, " \
|
||||
"'/search/name?death_date=')]/text()"
|
||||
'year': ".//a[starts-with(@href, " \
|
||||
"'/search/name?death_year=')]/text()"
|
||||
},
|
||||
postprocess=build_date),
|
||||
Attribute(key='death notes',
|
||||
path="./div/text()",
|
||||
# TODO: check if this slicing is always correct
|
||||
postprocess=lambda x: x.strip()[2:])]
|
||||
Attribute(key='death place',
|
||||
path=".//a[starts-with(@href, " \
|
||||
"'/search/name?death_place=')]/text()")]
|
||||
_film_attrs = [Attribute(key=None,
|
||||
multi=True,
|
||||
path={
|
||||
'link': "./a[1]/@href",
|
||||
'title': ".//text()",
|
||||
'status': "./i/a//text()",
|
||||
'roleID': "./div[@class='_imdbpyrole']/@roleid"
|
||||
'link': "./b/a[1]/@href",
|
||||
'title': "./b/a[1]/text()",
|
||||
'notes': "./b/following-sibling::text()",
|
||||
'year': "./span[@class='year_column']/text()",
|
||||
'status': "./a[@class='in_production']/text()",
|
||||
'rolesNoChar': './/br/following-sibling::text()',
|
||||
'chrRoles': "./a[@imdbpyname]/@imdbpyname",
|
||||
'roleID': "./a[starts-with(@href, '/character/')]/@href"
|
||||
},
|
||||
postprocess=lambda x:
|
||||
build_movie(x.get('title') or u'',
|
||||
year=x.get('year'),
|
||||
movieID=analyze_imdbid(x.get('link') or u''),
|
||||
roleID=(x.get('roleID') or u'').split('/'),
|
||||
rolesNoChar=(x.get('rolesNoChar') or u'').strip(),
|
||||
chrRoles=(x.get('chrRoles') or u'').strip(),
|
||||
additionalNotes=x.get('notes'),
|
||||
roleID=(x.get('roleID') or u''),
|
||||
status=x.get('status') or None))]
|
||||
|
||||
extractors = [
|
||||
Extractor(label='page title',
|
||||
path="//title",
|
||||
Extractor(label='name',
|
||||
path="//h1[@class='header']",
|
||||
attrs=Attribute(key='name',
|
||||
path="./text()",
|
||||
path=".//text()",
|
||||
postprocess=lambda x: analyze_name(x,
|
||||
canonical=1))),
|
||||
canonical=1))),
|
||||
|
||||
Extractor(label='birth info',
|
||||
path="//div[h5='Date of Birth:']",
|
||||
path="//div[h4='Born:']",
|
||||
attrs=_birth_attrs),
|
||||
|
||||
Extractor(label='death info',
|
||||
path="//div[h5='Date of Death:']",
|
||||
path="//div[h4='Died:']",
|
||||
attrs=_death_attrs),
|
||||
|
||||
Extractor(label='headshot',
|
||||
path="//a[@name='headshot']",
|
||||
path="//td[@id='img_primary']/a",
|
||||
attrs=Attribute(key='headshot',
|
||||
path="./img/@src")),
|
||||
|
||||
Extractor(label='akas',
|
||||
path="//div[h5='Alternate Names:']",
|
||||
path="//div[h4='Alternate Names:']",
|
||||
attrs=Attribute(key='akas',
|
||||
path="./div/text()",
|
||||
postprocess=lambda x: x.strip().split(' | '))),
|
||||
path="./text()",
|
||||
postprocess=lambda x: x.strip().split(' '))),
|
||||
|
||||
Extractor(label='filmography',
|
||||
group="//div[@class='filmo'][h5]",
|
||||
group_key="./h5/a[@name]/text()",
|
||||
group_key_normalize=lambda x: x.lower()[:-1],
|
||||
path="./ol/li",
|
||||
attrs=_film_attrs)
|
||||
group="//div[starts-with(@id, 'filmo-head-')]",
|
||||
group_key="./a[@name]/text()",
|
||||
group_key_normalize=lambda x: x.lower().replace(': ', ' '),
|
||||
path="./following-sibling::div[1]" \
|
||||
"/div[starts-with(@class, 'filmo-row')]",
|
||||
attrs=_film_attrs),
|
||||
|
||||
Extractor(label='indevelopment',
|
||||
path="//div[starts-with(@class,'devitem')]",
|
||||
attrs=Attribute(key='in development',
|
||||
multi=True,
|
||||
path={
|
||||
'link': './a/@href',
|
||||
'title': './a/text()'
|
||||
},
|
||||
postprocess=lambda x:
|
||||
build_movie(x.get('title') or u'',
|
||||
movieID=analyze_imdbid(x.get('link') or u''),
|
||||
roleID=(x.get('roleID') or u'').split('/'),
|
||||
status=x.get('status') or None)))
|
||||
]
|
||||
preprocessors = [
|
||||
# XXX: check that this doesn't cut "status" or other info...
|
||||
(re.compile(r'<br>(\.\.\.| ?).+?</li>', re.I | re.M | re.S),
|
||||
'</li>'),
|
||||
(_reRoles, _manageRoles)]
|
||||
|
||||
preprocessors = [('<div class="clear"/> </div>', ''),
|
||||
('<br/>', '<br />'),
|
||||
(re.compile(r'(<a href="/character/ch[0-9]{7}")>(.*?)</a>'),
|
||||
r'\1 imdbpyname="\2@@">\2</a>')]
|
||||
|
||||
def postprocess_data(self, data):
|
||||
for what in 'birth date', 'death date':
|
||||
if what in data and not data[what]:
|
||||
del data[what]
|
||||
# XXX: the code below is for backwards compatibility
|
||||
# probably could be removed
|
||||
for key in data.keys():
|
||||
if key.startswith('actor '):
|
||||
if not data.has_key('actor'):
|
||||
data['actor'] = []
|
||||
data['actor'].extend(data[key])
|
||||
del data[key]
|
||||
if key.startswith('actress '):
|
||||
if not data.has_key('actress'):
|
||||
data['actress'] = []
|
||||
data['actress'].extend(data[key])
|
||||
del data[key]
|
||||
if key.startswith('self '):
|
||||
if not data.has_key('self'):
|
||||
data['self'] = []
|
||||
data['self'].extend(data[key])
|
||||
del data[key]
|
||||
if key == 'birth place':
|
||||
data['birth notes'] = data[key]
|
||||
del data[key]
|
||||
if key == 'death place':
|
||||
data['death notes'] = data[key]
|
||||
del data[key]
|
||||
return data
|
||||
|
||||
|
||||
@@ -181,6 +227,10 @@ class DOMHTMLBioParser(DOMParserBase):
|
||||
# TODO: check if this slicing is always correct
|
||||
postprocess=lambda x: u''.join(x).strip()[2:])]
|
||||
extractors = [
|
||||
Extractor(label='headshot',
|
||||
path="//a[@name='headshot']",
|
||||
attrs=Attribute(key='headshot',
|
||||
path="./img/@src")),
|
||||
Extractor(label='birth info',
|
||||
path="//div[h5='Date of Birth']",
|
||||
attrs=_birth_attrs),
|
||||
|
||||
@@ -262,14 +262,20 @@ def build_person(txt, personID=None, billingPos=None,
|
||||
return person
|
||||
|
||||
|
||||
_re_chrIDs = re.compile('[0-9]{7}')
|
||||
|
||||
_b_m_logger = logging.getLogger('imdbpy.parser.http.build_movie')
|
||||
# To shrink spaces.
|
||||
re_spaces = re.compile(r'\s+')
|
||||
def build_movie(txt, movieID=None, roleID=None, status=None,
|
||||
accessSystem='http', modFunct=None, _parsingCharacter=False,
|
||||
_parsingCompany=False):
|
||||
_parsingCompany=False, year=None, chrRoles=None,
|
||||
rolesNoChar=None, additionalNotes=None):
|
||||
"""Given a string as normally seen on the "categorized" page of
|
||||
a person on the IMDb's web site, returns a Movie instance."""
|
||||
# FIXME: Oook, lets face it: build_movie and build_person are now
|
||||
# two horrible sets of patches to support the new IMDb design. They
|
||||
# must be rewritten from scratch.
|
||||
if _parsingCharacter:
|
||||
_defSep = ' Played by '
|
||||
elif _parsingCompany:
|
||||
@@ -291,6 +297,8 @@ def build_movie(txt, movieID=None, roleID=None, status=None,
|
||||
title = title[:-14] + ' (mini)'
|
||||
# Try to understand where the movie title ends.
|
||||
while True:
|
||||
if year:
|
||||
break
|
||||
if title[-1:] != ')':
|
||||
# Ignore the silly "TV Series" notice.
|
||||
if title[-9:] == 'TV Series':
|
||||
@@ -319,12 +327,24 @@ def build_movie(txt, movieID=None, roleID=None, status=None,
|
||||
if notes: notes = '%s %s' % (title[nidx:], notes)
|
||||
else: notes = title[nidx:]
|
||||
title = title[:nidx].rstrip()
|
||||
if year:
|
||||
year = year.strip()
|
||||
if title[-1] == ')':
|
||||
fpIdx = title.rfind('(')
|
||||
if fpIdx != -1:
|
||||
if notes: notes = '%s %s' % (title[fpIdx:], notes)
|
||||
else: notes = title[fpIdx:]
|
||||
title = title[:fpIdx].rstrip()
|
||||
title = u'%s (%s)' % (title, year)
|
||||
if _parsingCharacter and roleID and not role:
|
||||
roleID = None
|
||||
if not roleID:
|
||||
roleID = None
|
||||
elif len(roleID) == 1:
|
||||
roleID = roleID[0]
|
||||
if not role and chrRoles and isinstance(roleID, (str, unicode)):
|
||||
roleID = _re_chrIDs.findall(roleID)
|
||||
role = ' / '.join(filter(None, chrRoles.split('@@')))
|
||||
# Manages multiple roleIDs.
|
||||
if isinstance(roleID, list):
|
||||
tmprole = role.split('/')
|
||||
@@ -355,13 +375,29 @@ def build_movie(txt, movieID=None, roleID=None, status=None,
|
||||
movieID = str(movieID)
|
||||
if (not title) or (movieID is None):
|
||||
_b_m_logger.error('empty title or movieID for "%s"', txt)
|
||||
if rolesNoChar:
|
||||
rolesNoChar = filter(None, [x.strip() for x in rolesNoChar.split('/')])
|
||||
if not role:
|
||||
role = []
|
||||
elif not isinstance(role, list):
|
||||
role = [role]
|
||||
role += rolesNoChar
|
||||
notes = notes.strip()
|
||||
if additionalNotes:
|
||||
additionalNotes = re_spaces.sub(' ', additionalNotes).strip()
|
||||
if notes:
|
||||
notes += u' '
|
||||
notes += additionalNotes
|
||||
m = Movie(title=title, movieID=movieID, notes=notes, currentRole=role,
|
||||
roleID=roleID, roleIsPerson=_parsingCharacter,
|
||||
modFunct=modFunct, accessSystem=accessSystem)
|
||||
if roleNotes and len(roleNotes) == len(roleID):
|
||||
for idx, role in enumerate(m.currentRole):
|
||||
if roleNotes[idx]:
|
||||
role.notes = roleNotes[idx]
|
||||
try:
|
||||
if roleNotes[idx]:
|
||||
role.notes = roleNotes[idx]
|
||||
except IndexError:
|
||||
break
|
||||
# Status can't be checked here, and must be detected by the parser.
|
||||
if status:
|
||||
m['status'] = status
|
||||
@@ -468,8 +504,10 @@ class DOMParserBase(object):
|
||||
# converted to title=""Family Guy"" and this confuses BeautifulSoup.
|
||||
if self.usingModule == 'beautifulsoup':
|
||||
html_string = html_string.replace('""', '"')
|
||||
#print html_string.encode('utf8')
|
||||
if html_string:
|
||||
dom = self.get_dom(html_string)
|
||||
#print self.tostring(dom).encode('utf8')
|
||||
try:
|
||||
dom = self.preprocess_dom(dom)
|
||||
except Exception, e:
|
||||
|
||||
Reference in New Issue
Block a user