python-ox/ox/net.py

# -*- coding: utf-8 -*-
# vi:si:et:sw=4:sts=4:ts=4
# GPL 2008
from __future__ import print_function
import gzip
import json
import os
import re
import struct

import requests

from io import BytesIO
import urllib
from chardet.universaldetector import UniversalDetector


DEBUG = False
# Default headers for HTTP requests.
DEFAULT_HEADERS = {
    'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:78.0) Gecko/20100101 Firefox/78.0',
    'Accept-Charset': 'ISO-8859-1,utf-8;q=0.7,*;q=0.7',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'en-US,en;q=0.8,fr;q=0.6,de;q=0.4',
    'Accept-Encoding': 'gzip',
}

def status(url, data=None, headers=None):
    try:
        f = open_url(url, data, headers)
        s = f.code
    except urllib.error.HTTPError as e:
        s = e.code
    return s

def exists(url, data=None, headers=None):
    s = status(url, data, headers)
    if s >= 200 and s < 400:
        return True
    return False

def get_headers(url, data=None, headers=None):
    try:
        f = open_url(url, data, headers)
        f.headers['Status'] = "%s" % f.code
        headers = f.headers
        f.close()
    except urllib.error.HTTPError as e:
        e.headers['Status'] = "%s" % e.code
        headers = e.headers
    return dict(headers)

def get_json(url, data=None, headers=None):
    return json.loads(read_url(url, data, headers).decode('utf-8'))  # pylint: disable=no-member

def open_url(url, data=None, headers=None):
    if headers is None:
        headers = DEFAULT_HEADERS.copy()
    if isinstance(url, bytes):
        url = url.decode('utf-8')
    url = url.replace(' ', '%20')
    if data and not isinstance(data, bytes):
        data = data.encode('utf-8')
    req = urllib.request.Request(url, data, headers)
    return urllib.request.urlopen(req)

def read_url(url, data=None, headers=None, return_headers=False, unicode=False):
    if DEBUG:
        print('ox.net.read_url', url)
    f = open_url(url, data, headers)
    result = f.read()
    f.close()
    if f.headers.get('content-encoding', None) == 'gzip':
        result = gzip.GzipFile(fileobj=BytesIO(result)).read()
    if unicode:
        ctype = f.headers.get('content-type', '').lower()
        if 'charset' in ctype:
            encoding = ctype.split('charset=')[-1]
        else:
            encoding = detect_encoding(result)
        if not encoding:
            encoding = 'latin-1'
        result = result.decode(encoding)
    if return_headers:
        f.headers['Status'] = "%s" % f.code
        headers = {}
        for key in f.headers:
            headers[key.lower()] = f.headers[key]
        return headers, result
    return result

def detect_encoding(data):
    data_lower = data.lower().decode('utf-8', 'ignore')
    charset = re.compile('content="text/html; charset=(.*?)"').findall(data_lower)
    if not charset:
        charset = re.compile('meta charset="(.*?)"').findall(data_lower)
    if charset:
        return charset[0].lower()
    detector = UniversalDetector()
    p = 0
    l = len(data)
    s = 1024
    while p < l:
        detector.feed(data[p:p+s])
        if detector.done:
            break
        p += s
    detector.close()
    return detector.result['encoding']

get_url = read_url

def save_url(url, filename, overwrite=False):
    if not os.path.exists(filename) or overwrite:
        dirname = os.path.dirname(filename)
        os.makedirs(dirname, exist_ok=True)
        headers = DEFAULT_HEADERS.copy()
        r = requests.get(url, headers=headers, stream=True)
        filename_tmp = filename + '~'
        with open(filename_tmp, 'wb') as f:
            for chunk in r.iter_content(chunk_size=1024):
                if chunk:  # filter out keep-alive new chunks
                    f.write(chunk)
        os.rename(filename_tmp, filename)

def _get_size(url):
    req = urllib.request.Request(url, headers=DEFAULT_HEADERS.copy())
    req.get_method = lambda: 'HEAD'
    u = urllib.request.urlopen(req)
    if u.code != 200 or 'Content-Length' not in u.headers:
        raise IOError
    return int(u.headers['Content-Length'])

def _get_range(url, start, end):
    headers = DEFAULT_HEADERS.copy()
    headers['Range'] = 'bytes=%s-%s' % (start, end)
    req = urllib.request.Request(url, headers=headers)
    u = urllib.request.urlopen(req)
    return u.read()

def oshash(url):
    try:
        longlongformat = 'q'  # long long
        bytesize = struct.calcsize(longlongformat)

        filesize = _get_size(url)
        hash_ = filesize
        head = _get_range(url, 0, min(filesize, 65536))
        if filesize > 65536:
            tail = _get_range(url, filesize-65536, filesize)
        if filesize < 65536:
            f = BytesIO(head)
            for _ in range(int(filesize/bytesize)):
                buffer = f.read(bytesize)
                (l_value,) = struct.unpack(longlongformat, buffer)
                hash_ += l_value
                hash_ = hash_ & 0xFFFFFFFFFFFFFFFF  # cut off 64bit overflow
        else:
            for offset in range(0, 65536, bytesize):
                buffer = head[offset:offset+bytesize]
                (l_value,) = struct.unpack(longlongformat, buffer)
                hash_ += l_value
                hash_ = hash_ & 0xFFFFFFFFFFFFFFFF  # cut of 64bit overflow
            for offset in range(0, 65536, bytesize):
                buffer = tail[offset:offset+bytesize]
                (l_value,) = struct.unpack(longlongformat, buffer)
                hash_ += l_value
                hash_ = hash_ & 0xFFFFFFFFFFFFFFFF
        returnedhash = "%016x" % hash_
        return returnedhash
    except IOError:
        return "IOError"
add some functions 2008-04-27 16:54:37 +00:00			`# -- coding: utf-8 --`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`# vi:si:et:sw=4:sts=4:ts=4`
move and rename some 2008-07-06 13:00:06 +00:00			`# GPL 2008`
remove with_statement(for 2.5) from __future__ 2016-08-23 16:12:46 +00:00			`from __future__ import print_function`
adding gzip support to getUrl 2008-04-28 09:55:30 +00:00			`import gzip`
add ox.cache.get_json/ox.net.get_json, fixes #2451 2014-10-05 11:24:14 +00:00			`import json`
			`import os`
faster and more reliable encoding detection of html content 2013-06-01 11:29:24 +00:00			`import re`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`import struct`
add some functions 2008-04-27 16:54:37 +00:00
requests is always required now 2023-07-27 16:07:49 +00:00			`import requests`

drop six and python2 support 2023-07-27 11:07:13 +00:00			`from io import BytesIO`
			`import urllib`
faster way to detect encoding, speeds up getUrlUnicode on large pages 2008-06-17 10:53:29 +00:00			`from chardet.universaldetector import UniversalDetector`
add some functions 2008-04-27 16:54:37 +00:00

add read_url debug output 2012-08-21 06:41:25 +00:00			`DEBUG = False`
add some functions 2008-04-27 16:54:37 +00:00			`# Default headers for HTTP requests.`
adding encoding to default headers (itunes.py may supply others) 2008-04-29 08:45:14 +00:00			`DEFAULT_HEADERS = {`
use default headers with requests backend 2020-06-06 09:41:43 +00:00			`'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:78.0) Gecko/20100101 Firefox/78.0',`
update useragent 2010-08-12 21:35:31 +00:00			`'Accept-Charset': 'ISO-8859-1,utf-8;q=0.7,*;q=0.7',`
			`'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,/;q=0.8',`
simple title detection (imdb) 2017-08-02 14:48:22 +00:00			`'Accept-Language': 'en-US,en;q=0.8,fr;q=0.6,de;q=0.4',`
			`'Accept-Encoding': 'gzip',`
adding encoding to default headers (itunes.py may supply others) 2008-04-29 08:45:14 +00:00			`}`
add some functions 2008-04-27 16:54:37 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def status(url, data=None, headers=None):`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`try:`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`f = open_url(url, data, headers)`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`s = f.code`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`except urllib.error.HTTPError as e:`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`s = e.code`
			`return s`
add status, exists and getHeaders to net and cache, fix timeout bug in cache, now timeouts other then default actually work 2008-04-30 12:43:14 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def exists(url, data=None, headers=None):`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`s = status(url, data, headers)`
			`if s >= 200 and s < 400:`
			`return True`
			`return False`
add status, exists and getHeaders to net and cache, fix timeout bug in cache, now timeouts other then default actually work 2008-04-30 12:43:14 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def get_headers(url, data=None, headers=None):`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`try:`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`f = open_url(url, data, headers)`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`f.headers['Status'] = "%s" % f.code`
			`headers = f.headers`
			`f.close()`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`except urllib.error.HTTPError as e:`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`e.headers['Status'] = "%s" % e.code`
			`headers = e.headers`
			`return dict(headers)`
add status, exists and getHeaders to net and cache, fix timeout bug in cache, now timeouts other then default actually work 2008-04-30 12:43:14 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def get_json(url, data=None, headers=None):`
			`return json.loads(read_url(url, data, headers).decode('utf-8')) # pylint: disable=no-member`
add ox.cache.get_json/ox.net.get_json, fixes #2451 2014-10-05 11:24:14 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def open_url(url, data=None, headers=None):`
			`if headers is None:`
			`headers = DEFAULT_HEADERS.copy()`
drop six and python2 support 2023-07-27 11:07:13 +00:00			`if isinstance(url, bytes):`
			`url = url.decode('utf-8')`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`url = url.replace(' ', '%20')`
drop six and python2 support 2023-07-27 11:07:13 +00:00			`if data and not isinstance(data, bytes):`
fix POST in py3 2014-10-04 19:04:55 +00:00			`data = data.encode('utf-8')`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`req = urllib.request.Request(url, data, headers)`
			`return urllib.request.urlopen(req)`
add some functions 2008-04-27 16:54:37 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def read_url(url, data=None, headers=None, return_headers=False, unicode=False):`
add read_url debug output 2012-08-21 06:41:25 +00:00			`if DEBUG:`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`print('ox.net.read_url', url)`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`f = open_url(url, data, headers)`
fix ox.cache.read_url 2012-08-17 20:20:35 +00:00			`result = f.read()`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`f.close()`
			`if f.headers.get('content-encoding', None) == 'gzip':`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`result = gzip.GzipFile(fileobj=BytesIO(result)).read()`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`if unicode:`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`ctype = f.headers.get('content-type', '').lower()`
			`if 'charset' in ctype:`
			`encoding = ctype.split('charset=')[-1]`
			`else:`
			`encoding = detect_encoding(result)`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`if not encoding:`
			`encoding = 'latin-1'`
fix ox.cache.read_url 2012-08-17 20:20:35 +00:00			`result = result.decode(encoding)`
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`if return_headers:`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`f.headers['Status'] = "%s" % f.code`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`headers = {}`
			`for key in f.headers:`
			`headers[key.lower()] = f.headers[key]`
			`return headers, result`
fix ox.cache.read_url 2012-08-17 20:20:35 +00:00			`return result`
add some functions 2008-04-27 16:54:37 +00:00
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`def detect_encoding(data):`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`data_lower = data.lower().decode('utf-8', 'ignore')`
			`charset = re.compile('content="text/html; charset=(.*?)"').findall(data_lower)`
faster and more reliable encoding detection of html content 2013-06-01 11:29:24 +00:00			`if not charset:`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`charset = re.compile('meta charset="(.*?)"').findall(data_lower)`
faster and more reliable encoding detection of html content 2013-06-01 11:29:24 +00:00			`if charset:`
			`return charset[0].lower()`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`detector = UniversalDetector()`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`p = 0`
			`l = len(data)`
			`s = 1024`
			`while p < l:`
			`detector.feed(data[p:p+s])`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`if detector.done:`
			`break`
use six to support python 2 and 3 2014-09-30 19:04:46 +00:00			`p += s`
vi:si:et:sw=4:sts=4:ts=4 2008-06-19 09:21:21 +00:00			`detector.close()`
			`return detector.result['encoding']`
faster way to detect encoding, speeds up getUrlUnicode on large pages 2008-06-17 10:53:29 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`get_url = read_url`
add ox.cache.get_json/ox.net.get_json, fixes #2451 2014-10-05 11:24:14 +00:00
net/cache readUrl->read_url / Unicode -> unicode=True format replace all CammelCase with under_score 2012-08-14 13:58:05 +00:00			`def save_url(url, filename, overwrite=False):`
detecht iso-8859-1 in html header 2009-07-15 13:53:40 +00:00			`if not os.path.exists(filename) or overwrite:`
			`dirname = os.path.dirname(filename)`
download to tmp file and rename once download is complete 2024-03-22 07:48:48 +00:00			`os.makedirs(dirname, exist_ok=True)`
avoid reading file to ram in ox.net.save_url 2018-12-29 10:36:47 +00:00			`headers = DEFAULT_HEADERS.copy()`
requests is always required now 2023-07-27 16:07:49 +00:00			`r = requests.get(url, headers=headers, stream=True)`
download to tmp file and rename once download is complete 2024-03-22 07:48:48 +00:00			`filename_tmp = filename + '~'`
			`with open(filename_tmp, 'wb') as f:`
requests is always required now 2023-07-27 16:07:49 +00:00			`for chunk in r.iter_content(chunk_size=1024):`
			`if chunk: # filter out keep-alive new chunks`
			`f.write(chunk)`
download to tmp file and rename once download is complete 2024-03-22 07:48:48 +00:00			`os.rename(filename_tmp, filename)`
detecht iso-8859-1 in html header 2009-07-15 13:53:40 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def _get_size(url):`
			`req = urllib.request.Request(url, headers=DEFAULT_HEADERS.copy())`
			`req.get_method = lambda: 'HEAD'`
			`u = urllib.request.urlopen(req)`
			`if u.code != 200 or 'Content-Length' not in u.headers:`
			`raise IOError`
			`return int(u.headers['Content-Length'])`

			`def _get_range(url, start, end):`
			`headers = DEFAULT_HEADERS.copy()`
			`headers['Range'] = 'bytes=%s-%s' % (start, end)`
			`req = urllib.request.Request(url, headers=headers)`
			`u = urllib.request.urlopen(req)`
			`return u.read()`
add ox.net.oshash 2012-04-30 08:40:10 +00:00
avoid dict as default value 2016-06-08 13:30:25 +00:00			`def oshash(url):`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`try:`
			`longlongformat = 'q' # long long`
			`bytesize = struct.calcsize(longlongformat)`

avoid dict as default value 2016-06-08 13:30:25 +00:00			`filesize = _get_size(url)`
			`hash_ = filesize`
			`head = _get_range(url, 0, min(filesize, 65536))`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`if filesize > 65536:`
avoid dict as default value 2016-06-08 13:30:25 +00:00			`tail = _get_range(url, filesize-65536, filesize)`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`if filesize < 65536:`
fix net.oshash for small files 2015-03-19 13:27:50 +00:00			`f = BytesIO(head)`
avoid dict as default value 2016-06-08 13:30:25 +00:00			`for _ in range(int(filesize/bytesize)):`
fix net.oshash for small files 2015-03-19 13:27:50 +00:00			`buffer = f.read(bytesize)`
avoid dict as default value 2016-06-08 13:30:25 +00:00			`(l_value,) = struct.unpack(longlongformat, buffer)`
			`hash_ += l_value`
			`hash_ = hash_ & 0xFFFFFFFFFFFFFFFF # cut off 64bit overflow`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`else:`
			`for offset in range(0, 65536, bytesize):`
			`buffer = head[offset:offset+bytesize]`
avoid dict as default value 2016-06-08 13:30:25 +00:00			`(l_value,) = struct.unpack(longlongformat, buffer)`
			`hash_ += l_value`
			`hash_ = hash_ & 0xFFFFFFFFFFFFFFFF # cut of 64bit overflow`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`for offset in range(0, 65536, bytesize):`
			`buffer = tail[offset:offset+bytesize]`
avoid dict as default value 2016-06-08 13:30:25 +00:00			`(l_value,) = struct.unpack(longlongformat, buffer)`
			`hash_ += l_value`
			`hash_ = hash_ & 0xFFFFFFFFFFFFFFFF`
			`returnedhash = "%016x" % hash_`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`return returnedhash`
avoid dict as default value 2016-06-08 13:30:25 +00:00			`except IOError:`
add ox.net.oshash 2012-04-30 08:40:10 +00:00			`return "IOError"`