python-ox/ox/web/rottentomatoes.py

# -*- coding: UTF-8 -*-
# vi:si:et:sw=4:sts=4:ts=4
import re

from ox.cache import read_url
from ox import find_re, strip_tags


def get_url(id=None, imdb=None):
    #this would also wor but does not cache:
    '''
    from urllib2 import urlopen
    u = urlopen(url)
    return u.url
    '''
    if imdb:
        url = "http://www.rottentomatoes.com/alias?type=imdbid&s=%s" % imdb
        data = read_url(url)
        if "movie_title" in data:
            movies = re.compile('(/m/.*?/)').findall(data)
            if movies:
                return "http://www.rottentomatoes.com" + movies[0]
    return None

def get_og(data, key):
    return find_re(data, '<meta property="og:%s".*?content="(.*?)"' % key)

def get_data(url):
    data = read_url(url)
    r = {}
    r['title'] = find_re(data, '<h1 class="movie_title">(.*?)</h1>')
    if '(' in r['title']:
        r['year'] = find_re(r['title'], '\((\d*?)\)')
        r['title'] = strip_tags(re.sub('\((\d*?)\)', '', r['title'])).strip()
    r['summary'] = strip_tags(find_re(data, '<p id="movieSynopsis" class="movie_synopsis" itemprop="description">(.*?)</p>')).strip()
    r['summary'] = r['summary'].replace('\t', ' ').replace('\n', ' ').replace('  ', ' ').replace('  ', ' ')
    if not r['summary']:
        r['summary'] = get_og(data, 'description')

    meter = re.compile('<span id="all-critics-meter" class="meter(.*?)">(.*?)</span>').findall(data)
    meter = filter(lambda m: m[1].isdigit(), meter)
    if meter:
        r['tomatometer'] = meter[0][1]
    r['rating'] = find_re(data, 'Average Rating: <span>([\d.]+)/10</span>')
    r['user_score'] = find_re(data, '<span class="meter popcorn numeric ">(\d+)</span>')
    r['user_rating'] = find_re(data, 'Average Rating: ([\d.]+)/5')
    poster = get_og(data, 'image')
    if poster and not 'poster_default.gif' in poster:
        r['posters'] = [poster]
    for key in r.keys():
        if not r[key]:
            del r[key]
    return r
add ox.web to this repos 2010-07-07 23:25:57 +00:00			`# -- coding: UTF-8 --`
			`# vi:si:et:sw=4:sts=4:ts=4`
			`import re`

ox.web under_score api rewrite 2012-08-15 15:15:40 +00:00			`from ox.cache import read_url`
replace all CammelCase with under_score in ox 2012-08-14 14:12:43 +00:00			`from ox import find_re, strip_tags`
add ox.web to this repos 2010-07-07 23:25:57 +00:00

ox.web under_score api rewrite 2012-08-15 15:15:40 +00:00			`def get_url(id=None, imdb=None):`
add ox.web to this repos 2010-07-07 23:25:57 +00:00			`#this would also wor but does not cache:`
			`'''`
			`from urllib2 import urlopen`
			`u = urlopen(url)`
			`return u.url`
			`'''`
ox.web under_score api rewrite 2012-08-15 15:15:40 +00:00			`if imdb:`
			`url = "http://www.rottentomatoes.com/alias?type=imdbid&s=%s" % imdb`
			`data = read_url(url)`
			`if "movie_title" in data:`
			`movies = re.compile('(/m/.*?/)').findall(data)`
			`if movies:`
			`return "http://www.rottentomatoes.com" + movies[0]`
add ox.web to this repos 2010-07-07 23:25:57 +00:00			`return None`

update rottentomatoes and metacritic 2012-07-08 11:09:58 +00:00			`def get_og(data, key):`
replace all CammelCase with under_score in ox 2012-08-14 14:12:43 +00:00			`return find_re(data, '<meta property="og:%s".?content="(.?)"' % key)`
update rottentomatoes and metacritic 2012-07-08 11:09:58 +00:00
ox.web under_score api rewrite 2012-08-15 15:15:40 +00:00			`def get_data(url):`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`data = read_url(url)`
add ox.web to this repos 2010-07-07 23:25:57 +00:00			`r = {}`
replace all CammelCase with under_score in ox 2012-08-14 14:12:43 +00:00			`r['title'] = find_re(data, '<h1 class="movie_title">(.*?)</h1>')`
add ox.web to this repos 2010-07-07 23:25:57 +00:00			`if '(' in r['title']:`
replace all CammelCase with under_score in ox 2012-08-14 14:12:43 +00:00			`r['year'] = find_re(r['title'], '\((\d*?)\)')`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`r['title'] = strip_tags(re.sub('\((\d*?)\)', '', r['title'])).strip()`
replace all CammelCase with under_score in ox 2012-08-14 14:12:43 +00:00			`r['summary'] = strip_tags(find_re(data, '<p id="movieSynopsis" class="movie_synopsis" itemprop="description">(.*?)</p>')).strip()`
update rottentomatoes and metacritic 2012-07-08 11:09:58 +00:00			`r['summary'] = r['summary'].replace('\t', ' ').replace('\n', ' ').replace(' ', ' ').replace(' ', ' ')`
			`if not r['summary']:`
			`r['summary'] = get_og(data, 'description')`

			`meter = re.compile('<span id="all-critics-meter" class="meter(.?)">(.?)</span>').findall(data)`
			`meter = filter(lambda m: m[1].isdigit(), meter)`
			`if meter:`
			`r['tomatometer'] = meter[0][1]`
replace all CammelCase with under_score in ox 2012-08-14 14:12:43 +00:00			`r['rating'] = find_re(data, 'Average Rating: <span>([\d.]+)/10</span>')`
			`r['user_score'] = find_re(data, '<span class="meter popcorn numeric ">(\d+)</span>')`
			`r['user_rating'] = find_re(data, 'Average Rating: ([\d.]+)/5')`
update rottentomatoes and metacritic 2012-07-08 11:09:58 +00:00			`poster = get_og(data, 'image')`
			`if poster and not 'poster_default.gif' in poster:`
			`r['posters'] = [poster]`
			`for key in r.keys():`
			`if not r[key]:`
			`del r[key]`
add ox.web to this repos 2010-07-07 23:25:57 +00:00			`return r`