mews.page/mews/check.py

269 lines
8 KiB
Python
Raw Permalink Normal View History

"""Checking a page and recording what came of it.
Both paths into the directory run through here: a submission from the form and
a recheck from the timer. Keeping them together means a site is listed and
relisted on exactly the same evidence.
"""
from dataclasses import dataclass
import sqlite3
import time
from urllib.parse import urlsplit
from mews import db, pages
from mews.fetch import (
Fetched,
Fetcher,
FetchError,
UrlError,
normalise_url,
registered_domain,
)
from mews.lint import Report, validate_bytes
OK = 200
NOT_MODIFIED = 304
@dataclass
class Outcome:
"""What a submission produced: either a report, or a reason there isn't one."""
report: Report | None = None
listed: bool = False
message: str | None = None
title: str = ""
def _findings_rows(report: Report) -> list[tuple[str, str, str, str, str]]:
return [
(f.section, f.level, f.code, f.message, f.location) for f in report.findings
]
def check_page(
url: str, *, fetcher: Fetcher
) -> tuple[Report | None, Fetched | None, str | None]:
"""Fetch and validate one page.
Returns the report, the response it came from, and a plain message when
there was no page to check.
"""
try:
response = fetcher.get(url)
except (UrlError, FetchError) as error:
return None, None, str(error)
if response.status != OK:
return (
None,
response,
f"That page returned {response.status}. Check the address and try again.",
)
report = validate_bytes(
response.body, url=response.url, response=response, fetcher=fetcher
)
return report, response, None
def submit(
connection: sqlite3.Connection,
raw_url: str,
*,
client: str,
listing: bool,
fetcher: Fetcher,
directory: str | None = None,
) -> Outcome:
"""Check a page, and list its site when the author asked and it conforms."""
try:
url = normalise_url(raw_url, allow_loopback=fetcher.allow_loopback)
except UrlError as error:
return Outcome(message=str(error))
host = urlsplit(url).hostname or ""
domain = registered_domain(host)
if domain is None:
return Outcome(message="That domain name isn't one the checker can reach.")
reason = db.blocked_reason(connection, domain)
if reason is not None:
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="blocked"
)
return Outcome(
message="This site was taken out of the directory. Write to the "
"address on the home page if that looks wrong."
)
limited = db.rate_limited(connection, client=client, domain=domain, listing=listing)
if limited is not None:
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="rate_limited"
)
return Outcome(message=limited)
started = time.monotonic()
report, response, failure = check_page(url, fetcher=fetcher)
elapsed = int((time.monotonic() - started) * 1000)
status = response.status if response else None
if report is None:
db.record_check(
connection,
site_id=None,
kind="submission",
result="unreachable",
http_status=status,
duration_ms=elapsed,
)
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="unreachable"
)
return Outcome(message=failure)
if not listing:
db.record_check(
connection,
site_id=None,
kind="submission",
result="pass" if report.conforms else "fail",
findings=_findings_rows(report),
http_status=status,
size=report.size,
duration_ms=elapsed,
)
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="checked"
)
return Outcome(report=report)
if not report.conforms:
db.record_check(
connection,
site_id=None,
kind="submission",
result="fail",
findings=_findings_rows(report),
http_status=status,
size=report.size,
duration_ms=elapsed,
)
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="rejected"
)
return Outcome(report=report)
site_id, existed = db.upsert_site(
connection,
domain=domain,
url=report.url or url,
title=report.title or domain,
description=report.description,
language=report.language,
etag=response.headers.get("etag") if response else None,
last_modified=response.headers.get("last-modified") if response else None,
)
db.record_check(
connection,
site_id=site_id,
kind="submission",
result="pass",
findings=_findings_rows(report),
http_status=status,
size=report.size,
duration_ms=elapsed,
)
db.record_submission(
connection,
url=url,
domain=domain,
client=client,
outcome="updated" if existed else "listed",
site_id=site_id,
)
if directory:
pages.write_directory(directory, db.listed(connection))
return Outcome(report=report, listed=True, title=report.title)
def recheck(
connection: sqlite3.Connection, row: sqlite3.Row, *, fetcher: Fetcher
) -> str:
"""Check one listed site again, and drop it if its page stopped conforming.
A page that no longer conforms goes immediately: there is no contact address
to warn, and resubmitting after a fix lists the site again at once. A site
we simply could not reach is retried, because downtime is not a conformance
failure.
"""
conditional: dict[str, str] = {}
if row["etag"]:
conditional["If-None-Match"] = row["etag"]
if row["last_modified"]:
conditional["If-Modified-Since"] = row["last_modified"]
started = time.monotonic()
try:
response = fetcher.get(row["url"], headers=conditional or None)
except (UrlError, FetchError):
return _transient(connection, row, "Your site didn't answer.")
elapsed = int((time.monotonic() - started) * 1000)
if response.status == NOT_MODIFIED:
db.mark_pass(
connection,
row["id"],
etag=row["etag"],
last_modified=row["last_modified"],
)
db.record_check(
connection,
site_id=row["id"],
kind="recheck",
result="pass",
http_status=304,
duration_ms=elapsed,
)
return "pass"
if response.status != OK:
return _transient(connection, row, f"Your page returned {response.status}.")
report = validate_bytes(
response.body, url=response.url, response=response, fetcher=fetcher
)
db.record_check(
connection,
site_id=row["id"],
kind="recheck",
result="pass" if report.conforms else "fail",
findings=_findings_rows(report),
http_status=response.status,
size=report.size,
duration_ms=elapsed,
)
if report.conforms:
db.mark_pass(
connection,
row["id"],
etag=response.headers.get("etag"),
last_modified=response.headers.get("last-modified"),
)
return "pass"
db.remove(connection, row["id"], str(report.failures[0]))
return "fail"
def _transient(connection: sqlite3.Connection, row: sqlite3.Row, note: str) -> str:
"""Count a failure that wasn't the page's fault, dropping the site if it sticks."""
count = db.mark_transient(connection, row["id"])
db.record_check(connection, site_id=row["id"], kind="recheck", result="unreachable")
if count >= db.TRANSIENT_LIMIT:
db.remove(
connection,
row["id"],
f"{note} It stayed that way for {db.TRANSIENT_LIMIT} days.",
)
return "fail"
return "unreachable"