"""Checking a page and recording what came of it. Both paths into the directory run through here: a submission from the form and a recheck from the timer. Keeping them together means a site is listed and relisted on exactly the same evidence. """ from dataclasses import dataclass import sqlite3 import time from urllib.parse import urlsplit from mews import db, pages from mews.fetch import ( Fetched, Fetcher, FetchError, UrlError, normalise_url, registered_domain, ) from mews.lint import Report, validate_bytes OK = 200 NOT_MODIFIED = 304 @dataclass class Outcome: """What a submission produced: either a report, or a reason there isn't one.""" report: Report | None = None listed: bool = False message: str | None = None title: str = "" def _findings_rows(report: Report) -> list[tuple[str, str, str, str, str]]: return [ (f.section, f.level, f.code, f.message, f.location) for f in report.findings ] def check_page( url: str, *, fetcher: Fetcher ) -> tuple[Report | None, Fetched | None, str | None]: """Fetch and validate one page. Returns the report, the response it came from, and a plain message when there was no page to check. """ try: response = fetcher.get(url) except (UrlError, FetchError) as error: return None, None, str(error) if response.status != OK: return ( None, response, f"We couldn't read that page (it returned {response.status}). " "Check the address and try again.", ) report = validate_bytes( response.body, url=response.url, response=response, fetcher=fetcher ) return report, response, None def submit( connection: sqlite3.Connection, raw_url: str, *, client: str, listing: bool, fetcher: Fetcher, directory: str | None = None, ) -> Outcome: """Check a page, and list its site when the author asked and it conforms.""" try: url = normalise_url(raw_url, allow_loopback=fetcher.allow_loopback) except UrlError as error: return Outcome(message=str(error)) host = urlsplit(url).hostname or "" domain = registered_domain(host) if domain is None: return Outcome(message="That domain name isn't one the checker can reach.") reason = db.blocked_reason(connection, domain) if reason is not None: db.record_submission( connection, url=url, domain=domain, client=client, outcome="blocked" ) return Outcome( message="This site was taken out of the directory. Write to the " "address on the about page if that looks wrong." ) limited = db.rate_limited(connection, client=client, domain=domain, listing=listing) if limited is not None: db.record_submission( connection, url=url, domain=domain, client=client, outcome="rate_limited" ) return Outcome(message=limited) started = time.monotonic() report, response, failure = check_page(url, fetcher=fetcher) elapsed = int((time.monotonic() - started) * 1000) status = response.status if response else None if report is None: db.record_check( connection, site_id=None, kind="submission", result="unreachable", http_status=status, duration_ms=elapsed, ) db.record_submission( connection, url=url, domain=domain, client=client, outcome="unreachable" ) return Outcome(message=failure) if not listing: db.record_check( connection, site_id=None, kind="submission", result="pass" if report.conforms else "fail", findings=_findings_rows(report), http_status=status, size=report.size, duration_ms=elapsed, ) db.record_submission( connection, url=url, domain=domain, client=client, outcome="checked" ) return Outcome(report=report) if not report.conforms: db.record_check( connection, site_id=None, kind="submission", result="fail", findings=_findings_rows(report), http_status=status, size=report.size, duration_ms=elapsed, ) db.record_submission( connection, url=url, domain=domain, client=client, outcome="rejected" ) return Outcome(report=report) site_id, existed = db.upsert_site( connection, domain=domain, url=report.url or url, title=report.title or domain, description=report.description, language=report.language, etag=response.headers.get("etag") if response else None, last_modified=response.headers.get("last-modified") if response else None, ) db.record_check( connection, site_id=site_id, kind="submission", result="pass", findings=_findings_rows(report), http_status=status, size=report.size, duration_ms=elapsed, ) db.record_submission( connection, url=url, domain=domain, client=client, outcome="updated" if existed else "listed", site_id=site_id, ) if directory: pages.write_directory(directory, db.listed(connection)) return Outcome(report=report, listed=True, title=report.title) def recheck( connection: sqlite3.Connection, row: sqlite3.Row, *, fetcher: Fetcher ) -> str: """Check one listed site again, and drop it if its page stopped conforming. A page that no longer conforms goes immediately: there is no contact address to warn, and resubmitting after a fix lists the site again at once. A site we simply could not reach is retried, because downtime is not a conformance failure. """ conditional: dict[str, str] = {} if row["etag"]: conditional["If-None-Match"] = row["etag"] if row["last_modified"]: conditional["If-Modified-Since"] = row["last_modified"] started = time.monotonic() try: response = fetcher.get(row["url"], headers=conditional or None) except (UrlError, FetchError): return _transient(connection, row, "We couldn't reach your site.") elapsed = int((time.monotonic() - started) * 1000) if response.status == NOT_MODIFIED: db.mark_pass( connection, row["id"], etag=row["etag"], last_modified=row["last_modified"], ) db.record_check( connection, site_id=row["id"], kind="recheck", result="pass", http_status=304, duration_ms=elapsed, ) return "pass" if response.status != OK: return _transient(connection, row, f"Your page returned {response.status}.") report = validate_bytes( response.body, url=response.url, response=response, fetcher=fetcher ) db.record_check( connection, site_id=row["id"], kind="recheck", result="pass" if report.conforms else "fail", findings=_findings_rows(report), http_status=response.status, size=report.size, duration_ms=elapsed, ) if report.conforms: db.mark_pass( connection, row["id"], etag=response.headers.get("etag"), last_modified=response.headers.get("last-modified"), ) return "pass" db.remove(connection, row["id"], str(report.failures[0])) return "fail" def _transient(connection: sqlite3.Connection, row: sqlite3.Row, note: str) -> str: """Count a failure that wasn't the page's fault, dropping the site if it sticks.""" count = db.mark_transient(connection, row["id"]) db.record_check(connection, site_id=row["id"], kind="recheck", result="unreachable") if count >= db.TRANSIENT_LIMIT: db.remove( connection, row["id"], f"{note} We tried for {db.TRANSIENT_LIMIT} days.", ) return "fail" return "unreachable"