bitbot-3.11-fork/src/utils/http.py

import asyncio, ipaddress, re, signal, socket, traceback, typing
import urllib.error, urllib.parse
import json as _json
import bs4, netifaces, requests
import tornado.httpclient
from src import utils

REGEX_URL = re.compile("https?://\S+", re.I)

PAIRED_CHARACTERS = ["<>", "()"]

# best-effort tidying up of URLs
def url_sanitise(url: str):
    if not urllib.parse.urlparse(url).scheme:
        url = "http://%s" % url

    for pair_start, pair_end in PAIRED_CHARACTERS:
        # trim ")" from the end only if there's not a "(" to match it
        # google.com/) -> google.com/
        # google.com/() -> google.com/()
        # google.com/()) -> google.com/()
        if url.endswith(pair_end):
            if pair_start in url:
                open_index = url.rfind("(")
                other_index = url.rfind(")", 0, len(url)-1)
                if not other_index == -1 and other_index < open_index:
                    url = url[:-1]
            else:
                url = url[:-1]
    return url

DEFAULT_USERAGENT = ("Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 "
    "(KHTML, like Gecko) Chrome/49.0.2623.87 Safari/537.36")

RESPONSE_MAX = (1024*1024)*100
SOUP_CONTENT_TYPES = ["text/html", "text/xml", "application/xml"]
DECODE_CONTENT_TYPES = ["text/plain"]+SOUP_CONTENT_TYPES

class HTTPException(Exception):
    pass
class HTTPTimeoutException(HTTPException):
    def __init__(self):
        Exception.__init__(self, "HTTP request timed out")
class HTTPParsingException(HTTPException):
    def __init__(self, message: str, data: str=None):
        Exception.__init__(self, message or "HTTP parsing failed")
class HTTPWrongContentTypeException(HTTPException):
    def __init__(self, message: str=None):
        Exception.__init__(self,
            message or "HTTP request gave wrong content type")

def throw_timeout():
    raise HTTPTimeoutException()

class Request(object):
    def __init__(self, url: str, method: str="GET",
            get_params: typing.Dict[str, str]={}, post_data: typing.Any=None,
            headers: typing.Dict[str, str]={},

            json: bool=False, json_body: bool=False, allow_redirects: bool=True,
            check_content_type: bool=True, parse: bool=False,
            detect_encoding: bool=True,

            parser: str="lxml", fallback_encoding="iso-8859-1",
            content_type: str=None, proxy: str=None, useragent: str=None,

            **kwargs):
        self.set_url(url)
        self.method = method.upper()
        self.get_params = get_params
        self.post_data = post_data
        self.headers = headers

        self.json = json
        self.json_body = json_body
        self.allow_redirects = allow_redirects
        self.check_content_type = check_content_type
        self.parse = parse
        self.detect_encoding = detect_encoding

        self.parser = parser
        self.fallback_encoding = fallback_encoding
        self.content_type = content_type
        self.proxy = proxy
        self.useragent = useragent

        if kwargs:
            if method == "POST":
                self.post_data = kwargs
            else:
                self.get_params.update(kwargs)

    def set_url(self, url: str):
        if not urllib.parse.urlparse(url).scheme:
            url = "http://%s" % url
        self.url = url

    def get_headers(self) -> typing.Dict[str, str]:
        headers = self.headers.copy()
        if not "Accept-Language" in headers:
            headers["Accept-Language"] = "en-GB"
        if not "User-Agent" in headers:
            headers["User-Agent"] = self.useragent or DEFAULT_USERAGENT
        if not "Content-Type" in headers and self.content_type:
            headers["Content-Type"] = self.content_type
        return headers

    def get_body(self) -> typing.Any:
        if not self.post_data == None:
            if self.content_type == "application/json" or self.json_body:
                return _json.dumps(self.post_data)
            else:
                return self.post_data
        else:
            return None

class Response(object):
    def __init__(self, code: int, data: typing.Any,
            headers: typing.Dict[str, str], encoding: str):
        self.code = code
        self.data = data
        self.headers = headers
        self.encoding = encoding

def _meta_content(s: str) -> typing.Dict[str, str]:
    out = {}
    for keyvalue in s.split(";"):
        key, _, value = keyvalue.strip().partition("=")
        out[key] = value
    return out

def _find_encoding(soup: bs4.BeautifulSoup) -> typing.Optional[str]:
    if not soup.meta == None:
        meta_charset = soup.meta.get("charset")
        if not meta_charset == None:
            return meta_charset

        meta_content_type = soup.findAll("meta",
            {"http-equiv": lambda v: (v or "").lower() == "content-type"})
        if meta_content_type:
            return _meta_content(meta_content_type[0].get("content"))["charset"]

    doctype = [item for item in soup.contents if isinstance(item,
        bs4.Doctype)] or None
    if doctype and doctype[0] == "html":
        return "utf8"

    return None

def request(request_obj: typing.Union[str, Request], **kwargs) -> Response:
    if type(request_obj) == str:
        request_obj = Request(request_obj, **kwargs)
    return _request(request_obj)

def _request(request_obj: Request) -> Response:
    headers = request_obj.get_headers()

    def _wrap():
        response = requests.request(
            request_obj.method,
            request_obj.url,
            headers=headers,
            params=request_obj.get_params,
            data=request_obj.get_body(),
            allow_redirects=request_obj.allow_redirects,
            stream=True
        )
        response_content = response.raw.read(RESPONSE_MAX,
            decode_content=True)
        if not response_content or not response.raw.read(1) == b"":
            raise ValueError("Response too large")

        our_response = Response(response.status_code, response_content,
            headers=utils.CaseInsensitiveDict(dict(response.headers)),
            encoding=response.encoding)
        return our_response

    try:
        response = utils.deadline_process(_wrap, seconds=5)
    except utils.DeadlineExceededException:
        raise HTTPTimeoutException()

    content_type = response.headers.get("Content-Type", "").split(";", 1)[0]
    encoding = response.encoding or request_obj.fallback_encoding

    if (request_obj.detect_encoding and
            content_type and content_type in SOUP_CONTENT_TYPES):
        souped = bs4.BeautifulSoup(response.data, request_obj.parser)
        encoding = _find_encoding(souped) or encoding

    def _decode_data():
        return response.data.decode(encoding)

    if request_obj.parse:
        if (not request_obj.check_content_type or
                content_type in SOUP_CONTENT_TYPES):
            souped = bs4.BeautifulSoup(_decode_data(), request_obj.parser)
            response.data = souped
            return response
        else:
            raise HTTPWrongContentTypeException(
                "Tried to soup non-html/non-xml data (%s)" % content_type)

    if request_obj.json and response.data:
        data = _decode_data()
        try:
            response.data = _json.loads(data)
            return response
        except _json.decoder.JSONDecodeError as e:
            raise HTTPParsingException(str(e), data)

    if content_type in DECODE_CONTENT_TYPES:
        response.data = _decode_data()
        return response
    else:
        return response

def request_many(urls: typing.List[str]) -> typing.Dict[str, Response]:
    responses = {}

    async def _request(url):
        client = tornado.httpclient.AsyncHTTPClient()
        request = tornado.httpclient.HTTPRequest(url, method="GET",
            connect_timeout=2, request_timeout=2)

        response = await client.fetch(request)

        headers = utils.CaseInsensitiveDict(dict(response.headers))
        data = response.body.decode("utf8")
        responses[url] = Response(response.code, data, headers, "utf8")

    loop = asyncio.new_event_loop()
    awaits = []
    for url in urls:
        awaits.append(_request(url))
    task = asyncio.wait(awaits, loop=loop, timeout=5)
    loop.run_until_complete(task)
    loop.close()

    return responses

class Client(object):
    request = request
    request_many = request_many

def strip_html(s: str) -> str:
    return bs4.BeautifulSoup(s, "lxml").get_text()

def resolve_hostname(hostname: str) -> typing.List[str]:
    try:
        addresses = socket.getaddrinfo(hostname, None, 0, socket.SOCK_STREAM)
    except:
        return []
    return [address[-1][0] for address in addresses]

def is_ip(addr: str) -> bool:
    try:
        ipaddress.ip_address(addr)
    except ValueError:
        return False
    return True

def is_localhost(hostname: str) -> bool:
    if is_ip(hostname):
        ips = [ipaddress.ip_address(hostname)]
    else:
        ips = [ipaddress.ip_address(ip) for ip in resolve_hostname(hostname)]

    for interface in netifaces.interfaces():
        links = netifaces.ifaddresses(interface)

        for link in links.get(netifaces.AF_INET, []
                )+links.get(netifaces.AF_INET6, []):
            address = ipaddress.ip_address(link["addr"].split("%", 1)[0])
            if address in ips:
                return True

    return False
switch to using asyncio's event loop 2019-07-08 11:45:10 +00:00			`import asyncio, ipaddress, re, signal, socket, traceback, typing`
Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`import urllib.error, urllib.parse`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00			`import json as _json`
switch to using asyncio's event loop 2019-07-08 11:45:10 +00:00			`import bs4, netifaces, requests`
			`import tornado.httpclient`
Add missing `utils` import in utils.http 2018-12-11 22:30:05 +00:00			`from src import utils`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00
use \S+ for url regex (for non-ascii chars), use url_sanitize to catch <> 2019-09-02 12:25:48 +00:00			`REGEX_URL = re.compile("https?://\S+", re.I)`

			`PAIRED_CHARACTERS = ["<>", "()"]`
Move REGEX_URL out of isgd.py and title.py in to utils.http 2019-04-24 14:46:54 +00:00
Add utils.http.url_validate() for best-effort url tidying 2019-07-02 13:10:18 +00:00			`# best-effort tidying up of URLs`
url_validate() -> url_sanitise() 2019-07-02 13:15:49 +00:00			`def url_sanitise(url: str):`
add missing schema in utils.http.sanitise_url, use in rss.py 2019-07-08 11:54:06 +00:00			`if not urllib.parse.urlparse(url).scheme:`
			`url = "http://%s" % url`

use \S+ for url regex (for non-ascii chars), use url_sanitize to catch <> 2019-09-02 12:25:48 +00:00			`for pair_start, pair_end in PAIRED_CHARACTERS:`
Add utils.http.url_validate() for best-effort url tidying 2019-07-02 13:10:18 +00:00			`# trim ")" from the end only if there's not a "(" to match it`
			`# google.com/) -> google.com/`
			`# google.com/() -> google.com/()`
			`# google.com/()) -> google.com/()`
use \S+ for url regex (for non-ascii chars), use url_sanitize to catch <> 2019-09-02 12:25:48 +00:00			`if url.endswith(pair_end):`
			`if pair_start in url:`
			`open_index = url.rfind("(")`
			`other_index = url.rfind(")", 0, len(url)-1)`
			`if not other_index == -1 and other_index < open_index:`
			`url = url[:-1]`
			`else:`
			`url = url[:-1]`
Add utils.http.url_validate() for best-effort url tidying 2019-07-02 13:10:18 +00:00			`return url`

allow Requests to specify a useragent 2019-09-12 09:41:50 +00:00			`DEFAULT_USERAGENT = ("Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 "`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00			`"(KHTML, like Gecko) Chrome/49.0.2623.87 Safari/537.36")`

Set a max size of 100mb for utils.http.get_url 2018-10-10 13:05:15 +00:00			`RESPONSE_MAX = (10241024)100`
Don't try to parse non-html/xml stuff with BeautifulSoup 2019-02-26 11:18:50 +00:00			`SOUP_CONTENT_TYPES = ["text/html", "text/xml", "application/xml"]`
automatically decode certain http content types 2019-09-11 14:28:13 +00:00			`DECODE_CONTENT_TYPES = ["text/plain"]+SOUP_CONTENT_TYPES`
Set a max size of 100mb for utils.http.get_url 2018-10-10 13:05:15 +00:00
Fix/refactor issues brought up by type hint linting 2018-10-30 17:49:35 +00:00			`class HTTPException(Exception):`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`pass`
			`class HTTPTimeoutException(HTTPException):`
Give descriptions to utils.http.HTTPException subclasses 2019-06-27 17:28:08 +00:00			`def __init__(self):`
			`Exception.__init__(self, "HTTP request timed out")`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`class HTTPParsingException(HTTPException):`
Pass the content of a webpage to HTTPParsingException 2019-09-02 12:27:44 +00:00			`def __init__(self, message: str, data: str=None):`
message arg for HTTPWrongContentTypeException/HTTPParsingException 2019-06-28 22:00:48 +00:00			`Exception.__init__(self, message or "HTTP parsing failed")`
Raise a specific exception in utils.http.request for "wrong content type" 2019-02-28 23:28:45 +00:00			`class HTTPWrongContentTypeException(HTTPException):`
message arg for HTTPWrongContentTypeException/HTTPParsingException 2019-06-28 22:00:48 +00:00			`def __init__(self, message: str=None):`
			`Exception.__init__(self,`
			`message or "HTTP request gave wrong content type")`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00
Fix syntax error for throwing a timeout when signal.alarm fires 2018-10-10 14:07:04 +00:00			`def throw_timeout():`
			`raise HTTPTimeoutException()`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`class Request(object):`
			`def __init__(self, url: str, method: str="GET",`
			`get_params: typing.Dict[str, str]={}, post_data: typing.Any=None,`
			`headers: typing.Dict[str, str]={},`

add `json_body` arg to Request to json-encode body, only return from `body` if not null 2019-09-16 09:57:18 +00:00			`json: bool=False, json_body: bool=False, allow_redirects: bool=True,`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`check_content_type: bool=True, parse: bool=False,`
			`detect_encoding: bool=True,`

			`parser: str="lxml", fallback_encoding="iso-8859-1",`
allow Requests to specify a useragent 2019-09-12 09:41:50 +00:00			`content_type: str=None, proxy: str=None, useragent: str=None,`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00
			`**kwargs):`
			`self.set_url(url)`
			`self.method = method.upper()`
			`self.get_params = get_params`
			`self.post_data = post_data`
			`self.headers = headers`

			`self.json = json`
add `json_body` arg to Request to json-encode body, only return from `body` if not null 2019-09-16 09:57:18 +00:00			`self.json_body = json_body`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`self.allow_redirects = allow_redirects`
			`self.check_content_type = check_content_type`
			`self.parse = parse`
			`self.detect_encoding = detect_encoding`

			`self.parser = parser`
			`self.fallback_encoding = fallback_encoding`
			`self.content_type = content_type`
add `proxy` to Request objects 2019-09-11 16:53:37 +00:00			`self.proxy = proxy`
allow Requests to specify a useragent 2019-09-12 09:41:50 +00:00			`self.useragent = useragent`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00
			`if kwargs:`
			`if method == "POST":`
			`self.post_data = kwargs`
			`else:`
			`self.get_params.update(kwargs)`

			`def set_url(self, url: str):`
			`if not urllib.parse.urlparse(url).scheme:`
			`url = "http://%s" % url`
			`self.url = url`

			`def get_headers(self) -> typing.Dict[str, str]:`
			`headers = self.headers.copy()`
			`if not "Accept-Language" in headers:`
			`headers["Accept-Language"] = "en-GB"`
			`if not "User-Agent" in headers:`
allow Requests to specify a useragent 2019-09-12 09:41:50 +00:00			`headers["User-Agent"] = self.useragent or DEFAULT_USERAGENT`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`if not "Content-Type" in headers and self.content_type:`
			`headers["Content-Type"] = self.content_type`
			`return headers`

			`def get_body(self) -> typing.Any:`
add `json_body` arg to Request to json-encode body, only return from `body` if not null 2019-09-16 09:57:18 +00:00			`if not self.post_data == None:`
			`if self.content_type == "application/json" or self.json_body:`
			`return _json.dumps(self.post_data)`
			`else:`
			`return self.post_data`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`else:`
add `json_body` arg to Request to json-encode body, only return from `body` if not null 2019-09-16 09:57:18 +00:00			`return None`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00			`class Response(object):`
			`def __init__(self, code: int, data: typing.Any,`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`headers: typing.Dict[str, str], encoding: str):`
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00			`self.code = code`
			`self.data = data`
			`self.headers = headers`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`self.encoding = encoding`
'utils.http.get_url' -> 'utils.http.request', return a Response object from utils.http.request 2018-12-11 22:26:38 +00:00
change utils.http.request to best-effort detect on-page encoding closes #113 2019-09-09 13:10:58 +00:00			`def _meta_content(s: str) -> typing.Dict[str, str]:`
			`out = {}`
'str.split' -> 's.split' 2019-09-09 13:53:11 +00:00			`for keyvalue in s.split(";"):`
change utils.http.request to best-effort detect on-page encoding closes #113 2019-09-09 13:10:58 +00:00			`key, _, value = keyvalue.strip().partition("=")`
			`out[key] = value`
			`return out`

			`def _find_encoding(soup: bs4.BeautifulSoup) -> typing.Optional[str]:`
only look for <meta>-related tags when there are meta tags 2019-09-09 13:39:19 +00:00			`if not soup.meta == None:`
			`meta_charset = soup.meta.get("charset")`
			`if not meta_charset == None:`
			`return meta_charset`

change utils.http.request to best-effort detect on-page encoding closes #113 2019-09-09 13:10:58 +00:00			`meta_content_type = soup.findAll("meta",`
			`{"http-equiv": lambda v: (v or "").lower() == "content-type"})`
			`if meta_content_type:`
			`return _meta_content(meta_content_type[0].get("content"))["charset"]`
only look for <meta>-related tags when there are meta tags 2019-09-09 13:39:19 +00:00
			`doctype = [item for item in soup.contents if isinstance(item,`
			`bs4.Doctype)] or None`
			`if doctype and doctype[0] == "html":`
			`return "utf8"`

add explicit None return for _find_encoding (mypy) 2019-09-09 13:25:01 +00:00			`return None`
change utils.http.request to best-effort detect on-page encoding closes #113 2019-09-09 13:10:58 +00:00
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`def request(request_obj: typing.Union[str, Request], **kwargs) -> Response:`
			`if type(request_obj) == str:`
			`request_obj = Request(request_obj, **kwargs)`
			`return _request(request_obj)`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`def _request(request_obj: Request) -> Response:`
			`headers = request_obj.get_headers()`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`def _wrap():`
			`response = requests.request(`
			`request_obj.method,`
			`request_obj.url,`
			`headers=headers,`
			`params=request_obj.get_params,`
			`data=request_obj.get_body(),`
			`allow_redirects=request_obj.allow_redirects,`
			`stream=True`
			`)`
			`response_content = response.raw.read(RESPONSE_MAX,`
			`decode_content=True)`
			`if not response_content or not response.raw.read(1) == b"":`
			`raise ValueError("Response too large")`

			`our_response = Response(response.status_code, response_content,`
			`headers=utils.CaseInsensitiveDict(dict(response.headers)),`
			`encoding=response.encoding)`
			`return our_response`

			`try:`
restore 5 second (instead of default 10) deadline for http.request 2019-09-17 12:44:14 +00:00			`response = utils.deadline_process(_wrap, seconds=5)`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`except utils.DeadlineExceededException:`
			`raise HTTPTimeoutException()`
Pass str object to BeautifulSoup, not bytes. closes #56 2019-05-28 09:22:35 +00:00
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`content_type = response.headers.get("Content-Type", "").split(";", 1)[0]`
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`encoding = response.encoding or request_obj.fallback_encoding`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`if (request_obj.detect_encoding and`
			`content_type and content_type in SOUP_CONTENT_TYPES):`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`souped = bs4.BeautifulSoup(response.data, request_obj.parser)`
Don't try to .decode non-html things, default iso-lat-1 for non-html too 2019-09-09 15:17:26 +00:00			`encoding = _find_encoding(souped) or encoding`
change utils.http.request to best-effort detect on-page encoding closes #113 2019-09-09 13:10:58 +00:00
Defer decoding http payload bytestring until after checking ContentType 2019-06-04 12:47:03 +00:00			`def _decode_data():`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`return response.data.decode(encoding)`
Defer decoding http payload bytestring until after checking ContentType 2019-06-04 12:47:03 +00:00
refactor utils.http.requests to support a Request object 2019-09-11 16:44:07 +00:00			`if request_obj.parse:`
			`if (not request_obj.check_content_type or`
			`content_type in SOUP_CONTENT_TYPES):`
			`souped = bs4.BeautifulSoup(_decode_data(), request_obj.parser)`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`response.data = souped`
			`return response`
Throw ValueError when utils.http.request tries to soup non-html/xml data 2019-02-27 15:16:08 +00:00			`else:`
Raise a specific exception in utils.http.request for "wrong content type" 2019-02-28 23:28:45 +00:00			`raise HTTPWrongContentTypeException(`
Allow bypass of content-type check in utils.http.request 2019-08-05 14:41:02 +00:00			`"Tried to soup non-html/non-xml data (%s)" % content_type)`
Return response code from utils.http.get_url when code=True and soup=True 2018-10-09 21:16:04 +00:00
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`if request_obj.json and response.data:`
Don't try to .decode non-html things, default iso-lat-1 for non-html too 2019-09-09 15:17:26 +00:00			`data = _decode_data()`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00			`try:`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`response.data = _json.loads(data)`
			`return response`
Use signal.alarm to Deadline utils.http.get_url and throw useful exceptions 2018-10-10 13:25:44 +00:00			`except _json.decoder.JSONDecodeError as e:`
Pass the content of a webpage to HTTPParsingException 2019-09-02 12:27:44 +00:00			`raise HTTPParsingException(str(e), data)`
Change utils.http to use requests 2018-10-10 12:41:58 +00:00
automatically decode certain http content types 2019-09-11 14:28:13 +00:00			`if content_type in DECODE_CONTENT_TYPES:`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`response.data = _decode_data()`
			`return response`
automatically decode certain http content types 2019-09-11 14:28:13 +00:00			`else:`
use utils.deadline_process() in utils.http._request() so background threads can call _request() 2019-09-17 12:41:11 +00:00			`return response`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00
implement utils.http.request_many as a tonado ioloop yield 2019-07-08 10:43:09 +00:00			`def request_many(urls: typing.List[str]) -> typing.Dict[str, Response]:`
			`responses = {}`

switch request_many() to use asyncio.gather 2019-07-08 12:46:27 +00:00			`async def _request(url):`
			`client = tornado.httpclient.AsyncHTTPClient()`
			`request = tornado.httpclient.HTTPRequest(url, method="GET",`
			`connect_timeout=2, request_timeout=2)`

Don't try/except async http exceptions 2019-07-08 12:51:02 +00:00			`response = await client.fetch(request)`
switch request_many() to use asyncio.gather 2019-07-08 12:46:27 +00:00
			`headers = utils.CaseInsensitiveDict(dict(response.headers))`
			`data = response.body.decode("utf8")`
Response.__init__() needs `encoding` now 2019-09-17 13:11:12 +00:00			`responses[url] = Response(response.code, data, headers, "utf8")`
implement utils.http.request_many as a tonado ioloop yield 2019-07-08 10:43:09 +00:00
close event loop when we're done with it (request_many()) 2019-07-08 12:59:48 +00:00			`loop = asyncio.new_event_loop()`
switch request_many() to use asyncio.gather 2019-07-08 12:46:27 +00:00			`awaits = []`
			`for url in urls:`
			`awaits.append(_request(url))`
asyncio.gather -> asyncio.wait (with timeout) 2019-07-08 13:50:11 +00:00			`task = asyncio.wait(awaits, loop=loop, timeout=5)`
switch request_many() to use asyncio.gather 2019-07-08 12:46:27 +00:00			`loop.run_until_complete(task)`
close event loop when we're done with it (request_many()) 2019-07-08 12:59:48 +00:00			`loop.close()`
switch request_many() to use asyncio.gather 2019-07-08 12:46:27 +00:00
implement utils.http.request_many as a tonado ioloop yield 2019-07-08 10:43:09 +00:00			`return responses`

add a helper utils.http.Client static object 2019-09-11 16:53:49 +00:00			`class Client(object):`
			`request = request`
			`request_many = request_many`

Add type/return hints throughout src/ and, in doing so, fix some cyclical references. 2018-10-30 14:58:48 +00:00			`def strip_html(s: str) -> str:`
Move src/Utils.py in to src/utils/, splitting functionality out in to modules of related functionality 2018-10-03 12:22:37 +00:00			`return bs4.BeautifulSoup(s, "lxml").get_text()`

Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`def resolve_hostname(hostname: str) -> typing.List[str]:`
			`try:`
			`addresses = socket.getaddrinfo(hostname, None, 0, socket.SOCK_STREAM)`
			`except:`
			`return []`
			`return [address[-1][0] for address in addresses]`

			`def is_ip(addr: str) -> bool:`
			`try:`
			`ipaddress.ip_address(addr)`
			`except ValueError:`
			`return False`
			`return True`

			`def is_localhost(hostname: str) -> bool:`
			`if is_ip(hostname):`
			`ips = [ipaddress.ip_address(hostname)]`
			`else:`
			`ips = [ipaddress.ip_address(ip) for ip in resolve_hostname(hostname)]`

			`for interface in netifaces.interfaces():`
			`links = netifaces.ifaddresses(interface)`
Support interfaces that don't have AF_INET and/or AF_INET6 2019-04-25 16:48:51 +00:00
			`for link in links.get(netifaces.AF_INET, []`
Add missing ":" 2019-04-25 16:50:41 +00:00			`)+links.get(netifaces.AF_INET6, []):`
Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`address = ipaddress.ip_address(link["addr"].split("%", 1)[0])`
			`if address in ips:`
			`return True`
Support interfaces that don't have AF_INET and/or AF_INET6 2019-04-25 16:48:51 +00:00
Refuse to get the title for any url that points locall 2019-04-25 14:58:58 +00:00			`return False`