mews.page/tests/test_fetch.py

175 lines
5 KiB
Python

"""The fetcher's refusals.
The submit endpoint fetches addresses strangers type, so this is the test that
keeps the service from being used as a probe against its own machine.
"""
import pytest
from mews import fetch
from mews.fetch import Fetcher, FetchError, UrlError, normalise_url, vet_address
PRIVATE = [
"127.0.0.1",
"127.1.2.3",
"10.0.0.1",
"172.16.0.1",
"192.168.1.1",
"169.254.169.254",
"0.0.0.0",
"100.64.0.1",
"192.0.0.1",
"198.18.0.1",
"192.0.2.1",
"203.0.113.1",
"240.0.0.1",
"255.255.255.255",
"::1",
"::",
"fd00::1",
"fe80::1",
"::ffff:127.0.0.1",
"2002:7f00:1::",
"64:ff9b::7f00:1",
"2001:db8::1",
]
PUBLIC = ["93.184.216.34", "1.1.1.1", "194.68.44.29", "2606:2800:220:1::"]
REFUSED = [
"file:///etc/passwd",
"javascript:alert(1)",
"gopher://example.com/",
"http://user:pw@example.com/",
"http://example.com:22/",
"http://127.0.0.1/",
"http://2130706433/",
"http://0x7f.1/",
"http://[::1]/",
"http://localhost/",
"http://box.internal/",
"http://printer.local/",
"http://not-a-tld/",
"https://exa mple.com/",
"https://a_b.example.com/",
"",
]
@pytest.mark.parametrize("address", PRIVATE)
def test_addresses_off_the_public_internet_are_refused(address):
with pytest.raises(UrlError):
vet_address(address)
@pytest.mark.parametrize("address", PUBLIC)
def test_public_addresses_pass(address):
vet_address(address)
@pytest.mark.parametrize("url", REFUSED)
def test_addresses_the_checker_will_not_read(url):
with pytest.raises(UrlError):
normalise_url(url)
def test_a_bare_domain_becomes_https():
assert normalise_url("example.com") == "https://example.com/"
def test_the_fragment_is_dropped():
assert normalise_url("https://example.com/a#b") == "https://example.com/a"
def test_the_host_is_lowercased_and_punycoded():
assert normalise_url("https://EXAMPLE.com/") == "https://example.com/"
def test_the_extra_blocked_ranges_come_from_the_environment(monkeypatch):
"""The deployment blocks its own public address this way."""
monkeypatch.setattr(
fetch, "_EXTRA_NETS", [__import__("ipaddress").ip_network("194.68.44.28/32")]
)
with pytest.raises(UrlError):
vet_address("194.68.44.28")
vet_address("194.68.44.29")
def test_loopback_is_allowed_only_when_asked():
with pytest.raises(UrlError):
vet_address("127.0.0.1")
vet_address("127.0.0.1", allow_loopback=True)
# --- against a real server on loopback ---------------------------------
def test_a_page_is_read(site):
url = site.add("/a.html", b"<p>hi</p>")
with Fetcher(allow_loopback=True) as fetcher:
response = fetcher.get(url)
assert response.status == 200
assert response.body == b"<p>hi</p>"
def test_the_host_header_carries_the_name_not_the_address(site):
url = site.add("/a.html", b"x")
with Fetcher(allow_loopback=True) as fetcher:
fetcher.get(url)
_, headers = site.requests[-1]
assert headers["Host"] == url.split("//")[1].split("/")[0]
def test_a_redirect_is_followed(site):
target = site.add("/b.html", b"there")
site.add("/a.html", status=302, headers={"Location": target})
with Fetcher(allow_loopback=True) as fetcher:
response = fetcher.get(site.base + "/a.html")
assert response.body == b"there"
assert response.url == target
def test_a_redirect_to_a_private_address_is_refused(site):
site.add("/a.html", status=302, headers={"Location": "http://10.0.0.1/"})
with Fetcher(allow_loopback=True) as fetcher, pytest.raises(UrlError):
fetcher.get(site.base + "/a.html")
def test_a_redirect_loop_stops(site):
site.add("/a.html", status=302, headers={"Location": site.base + "/a.html"})
with Fetcher(allow_loopback=True) as fetcher, pytest.raises(FetchError):
fetcher.get(site.base + "/a.html")
def test_a_body_over_the_cap_is_cut_short(site):
url = site.add("/big.html", b"x" * 400_000, declare_length=False)
with Fetcher(allow_loopback=True) as fetcher:
response = fetcher.get(url)
assert response.truncated
assert len(response.body) == fetch.PAGE_CAP
def test_a_declared_length_over_the_cap_is_refused_unread(site):
url = site.add("/big.html", b"x" * 100, headers={"Content-Length": "9999999"})
with Fetcher(allow_loopback=True) as fetcher, pytest.raises(FetchError):
fetcher.get(url)
def test_a_slow_page_times_out(site):
url = site.add("/slow.html", b"x")
site.delay = 0.3
with (
Fetcher(allow_loopback=True, budget=0.05) as fetcher,
pytest.raises(FetchError),
):
fetcher.get(url)
def test_a_missing_page_is_reported_by_status(site):
with Fetcher(allow_loopback=True) as fetcher:
assert fetcher.get(site.base + "/nope.html").status == 404
def test_a_name_that_does_not_resolve_is_reported():
with Fetcher() as fetcher, pytest.raises(FetchError):
fetcher.get("https://this-name-does-not-exist.example/")