python-ox/ox/web/ubu.py

# -*- coding: utf-8 -*-
# vi:si:et:sw=4:sts=4:ts=4
from __future__ import print_function
import re

import lxml.html

from ox import strip_tags, decode_html
from ox.cache import read_url


def get_id(url):
    return url.replace('http://www.ubu.com/', '').split('.html')[0].replace('/./', '/')

def get_url(id):
    return 'http://www.ubu.com/%s.html' % id

def get_data(url):
    if not url.startswith('http:'):
        url = get_url(url)
    data = read_url(url, unicode=True)
    m = {
        'id': get_id(url),
        'url': url,
        'type': re.compile('ubu.com/(.*?)/').findall(url)[0]
    }
    if m['type'] == 'sound':
        m['tracks'] = [{
            'title': strip_tags(decode_html(t[1])).strip(),
            'url': t[0]
        } for t in re.compile('"(http.*?.mp3)"[^>]*>(.+)</a', re.IGNORECASE).findall(data)]
    else:
        for videourl, title in re.compile('href="(http://ubumexico.centro.org.mx/.*?)">(.*?)</a>').findall(data):
            if videourl.endswith('.srt'):
                m['srt'] = videourl
            elif not 'video' in m:
                m['video'] = videourl
                m['video'] = m['video'].replace('/video/ ', '/video/').replace(' ', '%20')
                if m['video'] == 'http://ubumexico.centro.org.mx/video/':
                    del m['video']
            if not 'title' in m:
                m['title'] = strip_tags(decode_html(title)).strip()
        if not 'url' in m:
            print(url, 'missing')
        if 'title' in m:
            m['title'] = re.sub(r'(.*?) \(\d{4}\)$', '\\1', m['title'])

        if not 'title' in m:
            match = re.compile('<span id="ubuwork">(.*?)</span>').findall(data)
            if match:
                m['title'] = strip_tags(decode_html(match[0])).strip()
        if not 'title' in m:
            match = re.compile("<title>.*?&amp;(.*?)</title>", re.DOTALL).findall(data)
            if match:
                m['title'] = re.sub(r'\s+', ' ', match[0]).strip()
                if ' - ' in m['title']:
                    m['title'] = m['title'].split(' - ', 1)[-1]
        if 'title' in m:
            m['title'] = strip_tags(decode_html(m['title']).strip())
        match = re.compile("flashvars','file=(.*?.flv)'").findall(data)
        if match:
            m['flv'] = match[0]
            m['flv'] = m['flv'].replace('/video/ ', '/video/').replace(' ', '%20')

        match = re.compile('''src=(.*?) type="video/mp4"''').findall(data)
        if match:
            m['mp4'] = match[0].strip('"').strip("'").replace(' ', '%20')
            if not m['mp4'].startswith('http'):
                m['mp4'] = 'http://ubumexico.centro.org.mx/video/' + m['mp4']
        elif 'video' in m and (m['video'].endswith('.mp4') or m['video'].endswith('.m4v')):
            m['mp4'] = m['video']

        doc = lxml.html.document_fromstring(read_url(url))
        desc = doc.xpath("//div[contains(@id, 'ubudesc')]")
        if len(desc):
            txt = []
            for part in desc[0].text_content().split('\n\n'):
                if part == 'RESOURCES:':
                    break
                if part.strip():
                    txt.append(part)
            if txt:
                if len(txt) > 1 and txt[0].strip() == m.get('title'):
                    txt = txt[1:]
                m['description'] = '\n\n'.join(txt).split('RESOURCES')[0].split('RELATED')[0].strip()
        y = re.compile(r'\((\d{4})\)').findall(data)
        if y:
            m['year'] = int(y[0])
        d = re.compile('Director: (.+)').findall(data)
        if d:
            m['director'] = strip_tags(decode_html(d[0])).strip()

        a = re.compile('<a href="(.*?)">Back to (.*?)</a>', re.DOTALL).findall(data)
        if a:
            m['artist'] = strip_tags(decode_html(a[0][1])).strip()
        else:
            a = re.compile('<a href="(.*?)">(.*?) in UbuWeb Film').findall(data)
            if a:
                m['artist'] = strip_tags(decode_html(a[0][1])).strip()
            else:
                a = re.compile(r'<b>(.*?)\(b\..*?\d{4}\)').findall(data)
                if a:
                    m['artist'] = strip_tags(decode_html(a[0])).strip()
                elif m['id'] == 'film/lawder_color':
                    m['artist'] = 'Standish Lawder'

        if 'artist' in m:
            m['artist'] = m['artist'].replace('in UbuWeb Film', '')
            m['artist'] = m['artist'].replace('on UbuWeb Film', '').strip()
        if m['id'] == 'film/coulibeuf':
            m['title'] = 'Balkan Baroque'
            m['year'] = 1999
    return m

def get_films():
    ids = get_ids()
    films = []
    for id in ids:
        info = get_data(id)
        if info['type'] == 'film' and ('flv' in info or 'video' in info):
            films.append(info)
    return films

def get_ids():
    data = read_url('http://www.ubu.com/film/')
    ids = []
    author_urls = []
    for url, author in re.compile(r'<a href="(\./.*?)">(.*?)</a>').findall(data):
        url = 'http://www.ubu.com/film' + url[1:]
        data = read_url(url)
        author_urls.append(url)
        for u, title in re.compile(r'<a href="(.*?)">(.*?)</a>').findall(data):
            if not u.startswith('http'):
                if u == '../../sound/burroughs.html':
                    u = 'http://www.ubu.com/sound/burroughs.html'
                elif u.startswith('../'):
                    u = 'http://www.ubu.com/' + u[3:]
                else:
                    u = 'http://www.ubu.com/film/' + u
                if u not in author_urls and u.endswith('.html'):
                    ids.append(u)
    ids = [get_id(url) for url in list(set(ids))]
    return ids

def get_sound_ids():
    data = read_url('http://www.ubu.com/sound/')
    ids = []
    for url, author in re.compile(r'<a href="(\./.*?)">(.*?)</a>').findall(data):
        url = 'http://www.ubu.com/sound' + url[1:]
        ids.append(url)
    ids = [get_id(url) for url in sorted(set(ids))]
    return ids
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`# -- coding: utf-8 --`
			`# vi:si:et:sw=4:sts=4:ts=4`
from __futre__ import print_function 2014-09-30 19:27:26 +00:00			`from __future__ import print_function`
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`import re`

update ubu/archive 2015-03-14 19:37:34 +00:00			`import lxml.html`

			`from ox import strip_tags, decode_html`
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`from ox.cache import read_url`


			`def get_id(url):`
update ubu/archive 2015-03-14 19:37:34 +00:00			`return url.replace('http://www.ubu.com/', '').split('.html')[0].replace('/./', '/')`
add ox.web.ubu 2012-12-29 12:04:24 +00:00
			`def get_url(id):`
			`return 'http://www.ubu.com/%s.html' % id`

			`def get_data(url):`
			`if not url.startswith('http:'):`
			`url = get_url(url)`
			`data = read_url(url, unicode=True)`
			`m = {`
			`'id': get_id(url),`
			`'url': url,`
			`'type': re.compile('ubu.com/(.*?)/').findall(url)[0]`
			`}`
update crawler 2015-11-03 22:16:34 +00:00			`if m['type'] == 'sound':`
			`m['tracks'] = [{`
			`'title': strip_tags(decode_html(t[1])).strip(),`
			`'url': t[0]`
			`} for t in re.compile('"(http.?.mp3)"[^>]>(.+)</a', re.IGNORECASE).findall(data)]`
			`else:`
			`for videourl, title in re.compile('href="(http://ubumexico.centro.org.mx/.?)">(.?)</a>').findall(data):`
			`if videourl.endswith('.srt'):`
			`m['srt'] = videourl`
			`elif not 'video' in m:`
			`m['video'] = videourl`
			`m['video'] = m['video'].replace('/video/ ', '/video/').replace(' ', '%20')`
			`if m['video'] == 'http://ubumexico.centro.org.mx/video/':`
			`del m['video']`
			`if not 'title' in m:`
			`m['title'] = strip_tags(decode_html(title)).strip()`
			`if not 'url' in m:`
			`print(url, 'missing')`
			`if 'title' in m:`
escape strings 2024-09-11 21:52:01 +00:00			`m['title'] = re.sub(r'(.*?) \(\d{4}\)$', '\\1', m['title'])`
add ox.web.ubu 2012-12-29 12:04:24 +00:00
update crawler 2015-11-03 22:16:34 +00:00			`if not 'title' in m:`
			`match = re.compile('<span id="ubuwork">(.*?)</span>').findall(data)`
			`if match:`
			`m['title'] = strip_tags(decode_html(match[0])).strip()`
			`if not 'title' in m:`
			`match = re.compile("<title>.?&(.?)</title>", re.DOTALL).findall(data)`
			`if match:`
escape strings 2024-09-11 21:52:01 +00:00			`m['title'] = re.sub(r'\s+', ' ', match[0]).strip()`
update crawler 2015-11-03 22:16:34 +00:00			`if ' - ' in m['title']:`
			`m['title'] = m['title'].split(' - ', 1)[-1]`
			`if 'title' in m:`
			`m['title'] = strip_tags(decode_html(m['title']).strip())`
			`match = re.compile("flashvars','file=(.*?.flv)'").findall(data)`
better title 2015-03-14 21:08:38 +00:00			`if match:`
update crawler 2015-11-03 22:16:34 +00:00			`m['flv'] = match[0]`
			`m['flv'] = m['flv'].replace('/video/ ', '/video/').replace(' ', '%20')`
add ox.web.ubu 2012-12-29 12:04:24 +00:00
update crawler 2015-11-03 22:16:34 +00:00			`match = re.compile('''src=(.*?) type="video/mp4"''').findall(data)`
			`if match:`
			`m['mp4'] = match[0].strip('"').strip("'").replace(' ', '%20')`
			`if not m['mp4'].startswith('http'):`
			`m['mp4'] = 'http://ubumexico.centro.org.mx/video/' + m['mp4']`
			`elif 'video' in m and (m['video'].endswith('.mp4') or m['video'].endswith('.m4v')):`
			`m['mp4'] = m['video']`
update ubu/archive 2015-03-14 19:37:34 +00:00
update crawler 2015-11-03 22:16:34 +00:00			`doc = lxml.html.document_fromstring(read_url(url))`
			`desc = doc.xpath("//div[contains(@id, 'ubudesc')]")`
			`if len(desc):`
			`txt = []`
			`for part in desc[0].text_content().split('\n\n'):`
			`if part == 'RESOURCES:':`
			`break`
			`if part.strip():`
			`txt.append(part)`
			`if txt:`
			`if len(txt) > 1 and txt[0].strip() == m.get('title'):`
			`txt = txt[1:]`
			`m['description'] = '\n\n'.join(txt).split('RESOURCES')[0].split('RELATED')[0].strip()`
escape strings 2024-09-11 21:52:01 +00:00			`y = re.compile(r'\((\d{4})\)').findall(data)`
update crawler 2015-11-03 22:16:34 +00:00			`if y:`
			`m['year'] = int(y[0])`
			`d = re.compile('Director: (.+)').findall(data)`
			`if d:`
			`m['director'] = strip_tags(decode_html(d[0])).strip()`
add ox.web.ubu 2012-12-29 12:04:24 +00:00
update crawler 2015-11-03 22:16:34 +00:00			`a = re.compile('<a href="(.?)">Back to (.?)</a>', re.DOTALL).findall(data)`
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`if a:`
			`m['artist'] = strip_tags(decode_html(a[0][1])).strip()`
			`else:`
update crawler 2015-11-03 22:16:34 +00:00			`a = re.compile('<a href="(.?)">(.?) in UbuWeb Film').findall(data)`
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`if a:`
update crawler 2015-11-03 22:16:34 +00:00			`m['artist'] = strip_tags(decode_html(a[0][1])).strip()`
			`else:`
escape strings 2024-09-11 21:52:01 +00:00			`a = re.compile(r'<b>(.?)\(b\..?\d{4}\)').findall(data)`
update crawler 2015-11-03 22:16:34 +00:00			`if a:`
			`m['artist'] = strip_tags(decode_html(a[0])).strip()`
			`elif m['id'] == 'film/lawder_color':`
			`m['artist'] = 'Standish Lawder'`
update ubu/archive 2015-03-14 19:37:34 +00:00
update crawler 2015-11-03 22:16:34 +00:00			`if 'artist' in m:`
			`m['artist'] = m['artist'].replace('in UbuWeb Film', '')`
			`m['artist'] = m['artist'].replace('on UbuWeb Film', '').strip()`
			`if m['id'] == 'film/coulibeuf':`
			`m['title'] = 'Balkan Baroque'`
			`m['year'] = 1999`
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`return m`

			`def get_films():`
			`ids = get_ids()`
			`films = []`
			`for id in ids:`
			`info = get_data(id)`
			`if info['type'] == 'film' and ('flv' in info or 'video' in info):`
			`films.append(info)`
			`return films`

			`def get_ids():`
			`data = read_url('http://www.ubu.com/film/')`
			`ids = []`
			`author_urls = []`
escape strings 2024-09-11 21:52:01 +00:00			`for url, author in re.compile(r'<a href="(\./.?)">(.?)</a>').findall(data):`
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`url = 'http://www.ubu.com/film' + url[1:]`
			`data = read_url(url)`
			`author_urls.append(url)`
escape strings 2024-09-11 21:52:01 +00:00			`for u, title in re.compile(r'<a href="(.?)">(.?)</a>').findall(data):`
add ox.web.ubu 2012-12-29 12:04:24 +00:00			`if not u.startswith('http'):`
			`if u == '../../sound/burroughs.html':`
			`u = 'http://www.ubu.com/sound/burroughs.html'`
			`elif u.startswith('../'):`
			`u = 'http://www.ubu.com/' + u[3:]`
			`else:`
			`u = 'http://www.ubu.com/film/' + u`
			`if u not in author_urls and u.endswith('.html'):`
			`ids.append(u)`
			`ids = [get_id(url) for url in list(set(ids))]`
			`return ids`
update crawler 2015-11-03 22:16:34 +00:00
			`def get_sound_ids():`
			`data = read_url('http://www.ubu.com/sound/')`
			`ids = []`
escape strings 2024-09-11 21:52:01 +00:00			`for url, author in re.compile(r'<a href="(\./.?)">(.?)</a>').findall(data):`
update crawler 2015-11-03 22:16:34 +00:00			`url = 'http://www.ubu.com/sound' + url[1:]`
			`ids.append(url)`
			`ids = [get_id(url) for url in sorted(set(ids))]`
			`return ids`