bitbot-3.11-fork/src/utils/http.py

import ipaddress, re, signal, socket, traceback, typing
import urllib.error, urllib.parse
import json as _json
import bs4, netifaces, requests, tornado.gen, tornado.httpclient, tornado.ioloop
from src import utils

REGEX_URL = re.compile("https?://[A-Z0-9{}]+".format(re.escape("-._~:/%?#[]@!$&'()*+,;=")), re.I)

# best-effort tidying up of URLs
def url_sanitise(url: str):
    if url.endswith(")"):
        # trim ")" from the end only if there's not a "(" to match it
        # google.com/) -> google.com/
        # google.com/() -> google.com/()
        # google.com/()) -> google.com/()

        if "(" in url:
            open_index = url.rfind("(")
            other_index = url.rfind(")", 0, len(url)-1)
            if other_index == -1 or other_index < open_index:
                return url
        return url[:-1]
    return url

USER_AGENT = ("Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 "
    "(KHTML, like Gecko) Chrome/49.0.2623.87 Safari/537.36")

RESPONSE_MAX = (1024*1024)*100
SOUP_CONTENT_TYPES = ["text/html", "text/xml", "application/xml"]

class HTTPException(Exception):
    pass
class HTTPTimeoutException(HTTPException):
    def __init__(self):
        Exception.__init__(self, "HTTP request timed out")
class HTTPParsingException(HTTPException):
    def __init__(self, message: str=None):
        Exception.__init__(self, message or "HTTP parsing failed")
class HTTPWrongContentTypeException(HTTPException):
    def __init__(self, message: str=None):
        Exception.__init__(self,
            message or "HTTP request gave wrong content type")

def throw_timeout():
    raise HTTPTimeoutException()

class Response(object):
    def __init__(self, code: int, data: typing.Any,
            headers: typing.Dict[str, str]):
        self.code = code
        self.data = data
        self.headers = headers

def request(url: str, method: str="GET", get_params: dict={},
        post_data: typing.Any=None, headers: dict={},
        json_data: typing.Any=None, code: bool=False, json: bool=False,
        soup: bool=False, parser: str="lxml", fallback_encoding: str="utf8",
        allow_redirects: bool=True
        ) -> Response:

    if not urllib.parse.urlparse(url).scheme:
        url = "http://%s" % url

    if not "Accept-Language" in headers:
        headers["Accept-Language"] = "en-GB"
    if not "User-Agent" in headers:
        headers["User-Agent"] = USER_AGENT

    signal.signal(signal.SIGALRM, lambda _1, _2: throw_timeout())
    signal.alarm(5)
    try:
        response = requests.request(
            method.upper(),
            url,
            headers=headers,
            params=get_params,
            data=post_data,
            json=json_data,
            allow_redirects=allow_redirects,
            stream=True
        )
        response_content = response.raw.read(RESPONSE_MAX, decode_content=True)
    except TimeoutError:
        raise HTTPTimeoutException()
    finally:
        signal.signal(signal.SIGALRM, signal.SIG_IGN)

    response_headers = utils.CaseInsensitiveDict(dict(response.headers))
    content_type = response.headers["Content-Type"].split(";", 1)[0]

    def _decode_data():
        return response_content.decode(response.encoding or fallback_encoding)

    if soup:
        if content_type in SOUP_CONTENT_TYPES:
            soup = bs4.BeautifulSoup(_decode_data(), parser)
            return Response(response.status_code, soup, response_headers)
        else:
            raise HTTPWrongContentTypeException(
                "Tried to soup non-html/non-xml data")

    data = _decode_data()
    if json and data:
        try:
            return Response(response.status_code, _json.loads(data),
                response_headers)
        except _json.decoder.JSONDecodeError as e:
            raise HTTPParsingException(str(e))

    return Response(response.status_code, data, response_headers)

def request_many(urls: typing.List[str]) -> typing.Dict[str, Response]:
    responses = {}

    @tornado.gen.coroutine
    def _request():
        for url in urls:
            client = tornado.httpclient.AsyncHTTPClient()
            request = tornado.httpclient.HTTPRequest(url, method="GET",
                connect_timeout=2, request_timeout=2)
            response = yield client.fetch(request)

            headers = utils.CaseInsensitiveDict(dict(response.headers))
            data = response.body.decode("utf8")
            responses[url] = Response(response.code, data, headers)

    tornado.ioloop.IOLoop.current().run_sync(_request)
    return responses

def strip_html(s: str) -> str:
    return bs4.BeautifulSoup(s, "lxml").get_text()

def resolve_hostname(hostname: str) -> typing.List[str]:
    try:
        addresses = socket.getaddrinfo(hostname, None, 0, socket.SOCK_STREAM)
    except:
        return []
    return [address[-1][0] for address in addresses]

def is_ip(addr: str) -> bool:
    try:
        ipaddress.ip_address(addr)
    except ValueError:
        return False
    return True

def is_localhost(hostname: str) -> bool:
    if is_ip(hostname):
        ips = [ipaddress.ip_address(hostname)]
    else:
        ips = [ipaddress.ip_address(ip) for ip in resolve_hostname(hostname)]

    for interface in netifaces.interfaces():
        links = netifaces.ifaddresses(interface)

        for link in links.get(netifaces.AF_INET, []
                )+links.get(netifaces.AF_INET6, []):
            address = ipaddress.ip_address(link["addr"].split("%", 1)[0])
            if address in ips:
                return True

    return False
Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`import ipaddress, re, signal, socket, traceback, typing`
			`import urllib.error, urllib.parse`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00			`import json as _json`
implement utils.http.request_many as a tonado ioloop yield 2019-07-08 10:43:09 +00:00			`import bs4, netifaces, requests, tornado.gen, tornado.httpclient, tornado.ioloop`
Add missing `utils` import in utils.http 2018-12-11 22:30:05 +00:00			`from src import utils`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00
forgot the beautiful % 2019-05-03 03:50:51 +00:00			`REGEX_URL = re.compile("https?://[A-Z0-9{}]+".format(re.escape("-._~:/%?#[]@!$&'()*+,;=")), re.I)`
Move REGEX_URL out of isgd.py and title.py in to utils.http 2019-04-24 14:46:54 +00:00
Add utils.http.url_validate() for best-effort url tidying 2019-07-02 13:10:18 +00:00			`# best-effort tidying up of URLs`
url_validate() -> url_sanitise() 2019-07-02 13:15:49 +00:00			`def url_sanitise(url: str):`
Add utils.http.url_validate() for best-effort url tidying 2019-07-02 13:10:18 +00:00			`if url.endswith(")"):`
			`# trim ")" from the end only if there's not a "(" to match it`
			`# google.com/) -> google.com/`
			`# google.com/() -> google.com/()`
			`# google.com/()) -> google.com/()`

			`if "(" in url:`
			`open_index = url.rfind("(")`
			`other_index = url.rfind(")", 0, len(url)-1)`
			`if other_index == -1 or other_index < open_index:`
			`return url`
			`return url[:-1]`
			`return url`

Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00			`USER_AGENT = ("Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 "`
			`"(KHTML, like Gecko) Chrome/49.0.2623.87 Safari/537.36")`

Set a max size of 100mb for utils.http.get_url 2018-10-10 13:05:15 +00:00			`RESPONSE_MAX = (10241024)100`
Don't try to parse non-html/xml stuff with BeautifulSoup 2019-02-26 11:18:50 +00:00			`SOUP_CONTENT_TYPES = ["text/html", "text/xml", "application/xml"]`
Set a max size of 100mb for utils.http.get_url 2018-10-10 13:05:15 +00:00
Fix/refactor issues brought up by type hint linting 2018-10-30 17:49:35 +00:00			`class HTTPException(Exception):`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`pass`
			`class HTTPTimeoutException(HTTPException):`
Give descriptions to utils.http.HTTPException subclasses 2019-06-27 17:28:08 +00:00			`def __init__(self):`
			`Exception.__init__(self, "HTTP request timed out")`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`class HTTPParsingException(HTTPException):`
message arg for HTTPWrongContentTypeException/HTTPParsingException 2019-06-28 22:00:48 +00:00			`def __init__(self, message: str=None):`
			`Exception.__init__(self, message or "HTTP parsing failed")`
Raise a specific exception in utils.http.request for "wrong content type" 2019-02-28 23:28:45 +00:00			`class HTTPWrongContentTypeException(HTTPException):`
message arg for HTTPWrongContentTypeException/HTTPParsingException 2019-06-28 22:00:48 +00:00			`def __init__(self, message: str=None):`
			`Exception.__init__(self,`
			`message or "HTTP request gave wrong content type")`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00
Fix syntax error for throwing a timeout when signal.alarm fires 2018-10-10 14:07:04 +00:00			`def throw_timeout():`
			`raise HTTPTimeoutException()`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00			`class Response(object):`
			`def __init__(self, code: int, data: typing.Any,`
			`headers: typing.Dict[str, str]):`
			`self.code = code`
			`self.data = data`
			`self.headers = headers`

			`def request(url: str, method: str="GET", get_params: dict={},`
Add type/return hints throughout src/ and, in doing so, fix some cyclical references. 2018-10-30 14:58:48 +00:00			`post_data: typing.Any=None, headers: dict={},`
			`json_data: typing.Any=None, code: bool=False, json: bool=False,`
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00			`soup: bool=False, parser: str="lxml", fallback_encoding: str="utf8",`
add `allow_redirects` kwarg to utils.http.request() 2019-06-26 16:53:16 +00:00			`allow_redirects: bool=True`
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00			`) -> Response:`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00			`if not urllib.parse.urlparse(url).scheme:`
			`url = "http://%s" % url`

Change utils.http to use requests 2018-10-10 12:41:58 +00:00			`if not "Accept-Language" in headers:`
			`headers["Accept-Language"] = "en-GB"`
			`if not "User-Agent" in headers:`
			`headers["User-Agent"] = USER_AGENT`

signal.signal timer callback takes 2 args 2018-10-25 13:09:19 +00:00			`signal.signal(signal.SIGALRM, lambda _1, _2: throw_timeout())`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`signal.alarm(5)`
			`try:`
			`response = requests.request(`
			`method.upper(),`
			`url,`
			`headers=headers,`
			`params=get_params,`
			`data=post_data,`
			`json=json_data,`
add `allow_redirects` kwarg to utils.http.request() 2019-06-26 16:53:16 +00:00			`allow_redirects=allow_redirects,`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`stream=True`
			`)`
			`response_content = response.raw.read(RESPONSE_MAX, decode_content=True)`
			`except TimeoutError:`
			`raise HTTPTimeoutException()`
			`finally:`
			`signal.signal(signal.SIGALRM, signal.SIG_IGN)`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00
Pass a `dict` to utils.CaseInsensitiveDict, not a MutableMapping 2018-12-11 22:30:57 +00:00			`response_headers = utils.CaseInsensitiveDict(dict(response.headers))`
Don't try to parse non-html/xml stuff with BeautifulSoup 2019-02-26 11:18:50 +00:00			`content_type = response.headers["Content-Type"].split(";", 1)[0]`
Pass str object to BeautifulSoup, not bytes. closes #56 2019-05-28 09:22:35 +00:00
Defer decoding http payload bytestring until after checking ContentType 2019-06-04 12:47:03 +00:00			`def _decode_data():`
			`return response_content.decode(response.encoding or fallback_encoding)`

Throw ValueError when utils.http.request tries to soup non-html/xml data 2019-02-27 15:16:08 +00:00			`if soup:`
			`if content_type in SOUP_CONTENT_TYPES:`
Defer decoding http payload bytestring until after checking ContentType 2019-06-04 12:47:03 +00:00			`soup = bs4.BeautifulSoup(_decode_data(), parser)`
Throw ValueError when utils.http.request tries to soup non-html/xml data 2019-02-27 15:16:08 +00:00			`return Response(response.status_code, soup, response_headers)`
			`else:`
Raise a specific exception in utils.http.request for "wrong content type" 2019-02-28 23:28:45 +00:00			`raise HTTPWrongContentTypeException(`
			`"Tried to soup non-html/non-xml data")`
Return response code from utils.http.get_url when code=True and soup=True 2018-10-09 21:16:04 +00:00
Defer decoding http payload bytestring until after checking ContentType 2019-06-04 12:47:03 +00:00			`data = _decode_data()`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00			`if json and data:`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00			`try:`
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00			`return Response(response.status_code, _json.loads(data),`
			`response_headers)`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`except _json.decoder.JSONDecodeError as e:`
			`raise HTTPParsingException(str(e))`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00			`return Response(response.status_code, data, response_headers)`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00
implement utils.http.request_many as a tonado ioloop yield 2019-07-08 10:43:09 +00:00			`def request_many(urls: typing.List[str]) -> typing.Dict[str, Response]:`
			`responses = {}`

			`@tornado.gen.coroutine`
			`def _request():`
			`for url in urls:`
			`client = tornado.httpclient.AsyncHTTPClient()`
			`request = tornado.httpclient.HTTPRequest(url, method="GET",`
			`connect_timeout=2, request_timeout=2)`
			`response = yield client.fetch(request)`

			`headers = utils.CaseInsensitiveDict(dict(response.headers))`
			`data = response.body.decode("utf8")`
			`responses[url] = Response(response.code, data, headers)`

			`tornado.ioloop.IOLoop.current().run_sync(_request)`
			`return responses`

Add type/return hints throughout src/ and, in doing so, fix some cyclical references. 2018-10-30 14:58:48 +00:00			`def strip_html(s: str) -> str:`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00			`return bs4.BeautifulSoup(s, "lxml").get_text()`

Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`def resolve_hostname(hostname: str) -> typing.List[str]:`
			`try:`
			`addresses = socket.getaddrinfo(hostname, None, 0, socket.SOCK_STREAM)`
			`except:`
			`return []`
			`return [address[-1][0] for address in addresses]`

			`def is_ip(addr: str) -> bool:`
			`try:`
			`ipaddress.ip_address(addr)`
			`except ValueError:`
			`return False`
			`return True`

			`def is_localhost(hostname: str) -> bool:`
			`if is_ip(hostname):`
			`ips = [ipaddress.ip_address(hostname)]`
			`else:`
			`ips = [ipaddress.ip_address(ip) for ip in resolve_hostname(hostname)]`

			`for interface in netifaces.interfaces():`
			`links = netifaces.ifaddresses(interface)`
Support interfaces that don't have AF_INET and/or AF_INET6 2019-04-25 16:48:51 +00:00
			`for link in links.get(netifaces.AF_INET, []`
Add missing ":" 2019-04-25 16:50:41 +00:00			`)+links.get(netifaces.AF_INET6, []):`
Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`address = ipaddress.ip_address(link["addr"].split("%", 1)[0])`
			`if address in ips:`
			`return True`
Support interfaces that don't have AF_INET and/or AF_INET6 2019-04-25 16:48:51 +00:00
Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`return False`