"""The fetcher's refusals. The submit endpoint fetches addresses strangers type, so this is the test that keeps the service from being used as a probe against its own machine. """ import pytest from mews import fetch from mews.fetch import Fetcher, FetchError, UrlError, normalise_url, vet_address PRIVATE = [ "127.0.0.1", "127.1.2.3", "10.0.0.1", "172.16.0.1", "192.168.1.1", "169.254.169.254", "0.0.0.0", "100.64.0.1", "192.0.0.1", "198.18.0.1", "192.0.2.1", "203.0.113.1", "240.0.0.1", "255.255.255.255", "::1", "::", "fd00::1", "fe80::1", "::ffff:127.0.0.1", "2002:7f00:1::", "64:ff9b::7f00:1", "2001:db8::1", ] PUBLIC = ["93.184.216.34", "1.1.1.1", "194.68.44.29", "2606:2800:220:1::"] REFUSED = [ "file:///etc/passwd", "javascript:alert(1)", "gopher://example.com/", "http://user:pw@example.com/", "http://example.com:22/", "http://127.0.0.1/", "http://2130706433/", "http://0x7f.1/", "http://[::1]/", "http://localhost/", "http://box.internal/", "http://printer.local/", "http://not-a-tld/", "https://exa mple.com/", "https://a_b.example.com/", "", ] @pytest.mark.parametrize("address", PRIVATE) def test_addresses_off_the_public_internet_are_refused(address): with pytest.raises(UrlError): vet_address(address) @pytest.mark.parametrize("address", PUBLIC) def test_public_addresses_pass(address): vet_address(address) @pytest.mark.parametrize("url", REFUSED) def test_addresses_the_checker_will_not_read(url): with pytest.raises(UrlError): normalise_url(url) def test_a_bare_domain_becomes_https(): assert normalise_url("example.com") == "https://example.com/" def test_the_fragment_is_dropped(): assert normalise_url("https://example.com/a#b") == "https://example.com/a" def test_the_host_is_lowercased_and_punycoded(): assert normalise_url("https://EXAMPLE.com/") == "https://example.com/" def test_the_extra_blocked_ranges_come_from_the_environment(monkeypatch): """The deployment blocks its own public address this way.""" monkeypatch.setattr( fetch, "_EXTRA_NETS", [__import__("ipaddress").ip_network("194.68.44.28/32")] ) with pytest.raises(UrlError): vet_address("194.68.44.28") vet_address("194.68.44.29") def test_loopback_is_allowed_only_when_asked(): with pytest.raises(UrlError): vet_address("127.0.0.1") vet_address("127.0.0.1", allow_loopback=True) # --- against a real server on loopback --------------------------------- def test_a_page_is_read(site): url = site.add("/a.html", b"
hi
") with Fetcher(allow_loopback=True) as fetcher: response = fetcher.get(url) assert response.status == 200 assert response.body == b"hi
" def test_the_host_header_carries_the_name_not_the_address(site): url = site.add("/a.html", b"x") with Fetcher(allow_loopback=True) as fetcher: fetcher.get(url) _, headers = site.requests[-1] assert headers["Host"] == url.split("//")[1].split("/")[0] def test_a_redirect_is_followed(site): target = site.add("/b.html", b"there") site.add("/a.html", status=302, headers={"Location": target}) with Fetcher(allow_loopback=True) as fetcher: response = fetcher.get(site.base + "/a.html") assert response.body == b"there" assert response.url == target def test_a_redirect_to_a_private_address_is_refused(site): site.add("/a.html", status=302, headers={"Location": "http://10.0.0.1/"}) with Fetcher(allow_loopback=True) as fetcher, pytest.raises(UrlError): fetcher.get(site.base + "/a.html") def test_a_redirect_loop_stops(site): site.add("/a.html", status=302, headers={"Location": site.base + "/a.html"}) with Fetcher(allow_loopback=True) as fetcher, pytest.raises(FetchError): fetcher.get(site.base + "/a.html") def test_a_body_over_the_cap_is_cut_short(site): url = site.add("/big.html", b"x" * 400_000, declare_length=False) with Fetcher(allow_loopback=True) as fetcher: response = fetcher.get(url) assert response.truncated assert len(response.body) == fetch.PAGE_CAP def test_a_declared_length_over_the_cap_is_refused_unread(site): url = site.add("/big.html", b"x" * 100, headers={"Content-Length": "9999999"}) with Fetcher(allow_loopback=True) as fetcher, pytest.raises(FetchError): fetcher.get(url) def test_a_slow_page_times_out(site): url = site.add("/slow.html", b"x") site.delay = 0.3 with ( Fetcher(allow_loopback=True, budget=0.05) as fetcher, pytest.raises(FetchError), ): fetcher.get(url) def test_a_missing_page_is_reported_by_status(site): with Fetcher(allow_loopback=True) as fetcher: assert fetcher.get(site.base + "/nope.html").status == 404 def test_a_name_that_does_not_resolve_is_reported(): with Fetcher() as fetcher, pytest.raises(FetchError): fetcher.get("https://this-name-does-not-exist.example/")