diff --git a/mews/app.py b/mews/app.py new file mode 100644 index 0000000..4b829dc --- /dev/null +++ b/mews/app.py @@ -0,0 +1,165 @@ +"""The one dynamic page on mews.page. + +The site is static apart from this: a form posts a page address to /result, the +page is fetched and checked, and the author gets the findings back. Nothing here +runs JavaScript or sets a cookie, because the response is itself a Mews page. +""" + +from collections.abc import Iterable +import os +import threading +from urllib.parse import parse_qs + +from mews import db, pages +from mews.check import submit +from mews.fetch import Fetcher + +# Outbound checks are slow and this runs on a small machine, so only a couple +# happen at once; the rest of the queue is turned away rather than piled up. +_CHECKS = threading.BoundedSemaphore(2) +_REQUESTS = threading.BoundedSemaphore(8) +_SLOT_WAIT = 5.0 + +MAX_BODY = 4096 + +BUSY = "The checker is busy right now. Try again in a few minutes." + + +def directory_path() -> str: + """Where the directory page is written.""" + return os.environ.get("MEWS_DIRECTORY", "directory.html") + + +def _start(start_response, status: str, body: bytes) -> list[bytes]: + start_response( + status, + [ + ("Content-Type", "text/html; charset=utf-8"), + ("Content-Length", str(len(body))), + ("Cache-Control", "no-store"), + ], + ) + return [body] + + +def _client(environ: dict) -> str: + """Return the reader's address, as the proxy reports it. + + Caddy sets X-Real-IP from the connection it accepted. The server binds to + the loopback address only, so nothing else can set that header. + """ + return environ.get("HTTP_X_REAL_IP") or environ.get("REMOTE_ADDR", "") + + +def _form(environ: dict) -> dict[str, list[str]]: + try: + length = min(int(environ.get("CONTENT_LENGTH") or 0), MAX_BODY) + except ValueError: + return {} + if length <= 0: + return {} + raw = environ["wsgi.input"].read(length) + return parse_qs(raw.decode("utf-8", "replace"), keep_blank_values=True) + + +def application(environ: dict, start_response) -> Iterable[bytes]: + """Handle one request.""" + path = environ.get("PATH_INFO", "/") + method = environ.get("REQUEST_METHOD", "GET") + + if path != "/result": + return _start( + start_response, + "404 Not Found", + pages.message( + "Page not found", + ["That address isn't part of this site."], + [("/", "Mews"), ("/directory", "The directory")], + ).encode(), + ) + + if method != "POST": + return _start( + start_response, + "405 Method Not Allowed", + pages.message( + "Nothing to show", + ["Results appear here after you check a page."], + [("/check", "Check a page")], + ).encode(), + ) + + form = _form(environ) + raw_url = (form.get("url") or [""])[0].strip() + listing = "list" in form + + if not _REQUESTS.acquire(blocking=False): + return _start(start_response, "503 Service Unavailable", _busy()) + try: + if not _CHECKS.acquire(timeout=_SLOT_WAIT): + return _start(start_response, "503 Service Unavailable", _busy()) + try: + body = _check(raw_url, _client(environ), listing) + finally: + _CHECKS.release() + finally: + _REQUESTS.release() + return _start(start_response, "200 OK", body) + + +def _busy() -> bytes: + return pages.message("Busy", [BUSY], [("/check", "Try again")]).encode() + + +def _check(raw_url: str, client: str, listing: bool) -> bytes: + """Run one check and render the page the author gets back.""" + connection = db.connect() + try: + with Fetcher() as fetcher: + outcome = submit( + connection, + raw_url, + client=db.ip_hash(connection, client), + listing=listing, + fetcher=fetcher, + directory=directory_path() if listing else None, + ) + except ValueError as error: + # The directory page refused to publish itself. The site is listed; the + # page will be rebuilt by the next recheck pass. + return pages.message( + "Listed, but the directory didn't rebuild", + [str(error), "Your site is listed and will appear shortly."], + [("/directory", "The directory")], + ).encode() + finally: + connection.close() + + if outcome.report is None: + return pages.message( + "We couldn't check that page", + [outcome.message or "Something went wrong. Try again."], + [("/check", "Try again"), ("/spec/0.1", "The spec")], + ).encode() + + command = f"uv run mewslint.py {outcome.report.url or raw_url}" + return pages.report(outcome.report, listed=outcome.listed, command=command).encode() + + +def serve() -> None: + """Run the app for real, on the loopback address only.""" + # Imported here so the validator and the command line tools do not pull in + # a web server. + from waitress import serve as waitress_serve # noqa: PLC0415 + + listen = os.environ.get("MEWS_LISTEN", "127.0.0.1:8394") + host = listen.rsplit(":", 1)[0] + if host not in ("127.0.0.1", "::1", "localhost"): + raise SystemExit( + "mewsd serves on the loopback address only, behind a reverse proxy. " + f"Set MEWS_LISTEN to 127.0.0.1 with a port, not {listen}." + ) + waitress_serve(application, listen=listen, threads=4, ident="mews.page") + + +app = application diff --git a/mews/check.py b/mews/check.py new file mode 100644 index 0000000..17c7d45 --- /dev/null +++ b/mews/check.py @@ -0,0 +1,269 @@ +"""Checking a page and recording what came of it. + +Both paths into the directory run through here: a submission from the form and +a recheck from the timer. Keeping them together means a site is listed and +relisted on exactly the same evidence. +""" + +from dataclasses import dataclass +import sqlite3 +import time +from urllib.parse import urlsplit + +from mews import db, pages +from mews.fetch import ( + Fetched, + Fetcher, + FetchError, + UrlError, + normalise_url, + registered_domain, +) +from mews.lint import Report, validate_bytes + +OK = 200 +NOT_MODIFIED = 304 + + +@dataclass +class Outcome: + """What a submission produced: either a report, or a reason there isn't one.""" + + report: Report | None = None + listed: bool = False + message: str | None = None + title: str = "" + + +def _findings_rows(report: Report) -> list[tuple[str, str, str, str, str]]: + return [ + (f.section, f.level, f.code, f.message, f.location) for f in report.findings + ] + + +def check_page( + url: str, *, fetcher: Fetcher +) -> tuple[Report | None, Fetched | None, str | None]: + """Fetch and validate one page. + + Returns the report, the response it came from, and a plain message when + there was no page to check. + """ + try: + response = fetcher.get(url) + except (UrlError, FetchError) as error: + return None, None, str(error) + if response.status != OK: + return ( + None, + response, + f"We couldn't read that page (it returned {response.status}). " + "Check the address and try again.", + ) + report = validate_bytes( + response.body, url=response.url, response=response, fetcher=fetcher + ) + return report, response, None + + +def submit( + connection: sqlite3.Connection, + raw_url: str, + *, + client: str, + listing: bool, + fetcher: Fetcher, + directory: str | None = None, +) -> Outcome: + """Check a page, and list its site when the author asked and it conforms.""" + try: + url = normalise_url(raw_url, allow_loopback=fetcher.allow_loopback) + except UrlError as error: + return Outcome(message=str(error)) + + host = urlsplit(url).hostname or "" + domain = registered_domain(host) + if domain is None: + return Outcome(message="That domain name isn't one the checker can reach.") + + reason = db.blocked_reason(connection, domain) + if reason is not None: + db.record_submission( + connection, url=url, domain=domain, client=client, outcome="blocked" + ) + return Outcome( + message="This site was taken out of the directory. Write to the " + "address on the about page if that looks wrong." + ) + + limited = db.rate_limited(connection, client=client, domain=domain, listing=listing) + if limited is not None: + db.record_submission( + connection, url=url, domain=domain, client=client, outcome="rate_limited" + ) + return Outcome(message=limited) + + started = time.monotonic() + report, response, failure = check_page(url, fetcher=fetcher) + elapsed = int((time.monotonic() - started) * 1000) + status = response.status if response else None + if report is None: + db.record_check( + connection, + site_id=None, + kind="submission", + result="unreachable", + http_status=status, + duration_ms=elapsed, + ) + db.record_submission( + connection, url=url, domain=domain, client=client, outcome="unreachable" + ) + return Outcome(message=failure) + + if not listing: + db.record_check( + connection, + site_id=None, + kind="submission", + result="pass" if report.conforms else "fail", + findings=_findings_rows(report), + http_status=status, + size=report.size, + duration_ms=elapsed, + ) + db.record_submission( + connection, url=url, domain=domain, client=client, outcome="checked" + ) + return Outcome(report=report) + + if not report.conforms: + db.record_check( + connection, + site_id=None, + kind="submission", + result="fail", + findings=_findings_rows(report), + http_status=status, + size=report.size, + duration_ms=elapsed, + ) + db.record_submission( + connection, url=url, domain=domain, client=client, outcome="rejected" + ) + return Outcome(report=report) + + site_id, existed = db.upsert_site( + connection, + domain=domain, + url=report.url or url, + title=report.title or domain, + description=report.description, + language=report.language, + etag=response.headers.get("etag") if response else None, + last_modified=response.headers.get("last-modified") if response else None, + ) + db.record_check( + connection, + site_id=site_id, + kind="submission", + result="pass", + findings=_findings_rows(report), + http_status=status, + size=report.size, + duration_ms=elapsed, + ) + db.record_submission( + connection, + url=url, + domain=domain, + client=client, + outcome="updated" if existed else "listed", + site_id=site_id, + ) + if directory: + pages.write_directory(directory, db.listed(connection)) + return Outcome(report=report, listed=True, title=report.title) + + +def recheck( + connection: sqlite3.Connection, row: sqlite3.Row, *, fetcher: Fetcher +) -> str: + """Check one listed site again, and drop it if its page stopped conforming. + + A page that no longer conforms goes immediately: there is no contact address + to warn, and resubmitting after a fix lists the site again at once. A site + we simply could not reach is retried, because downtime is not a conformance + failure. + """ + conditional: dict[str, str] = {} + if row["etag"]: + conditional["If-None-Match"] = row["etag"] + if row["last_modified"]: + conditional["If-Modified-Since"] = row["last_modified"] + + started = time.monotonic() + try: + response = fetcher.get(row["url"], headers=conditional or None) + except (UrlError, FetchError): + return _transient(connection, row, "We couldn't reach your site.") + + elapsed = int((time.monotonic() - started) * 1000) + if response.status == NOT_MODIFIED: + db.mark_pass( + connection, + row["id"], + etag=row["etag"], + last_modified=row["last_modified"], + ) + db.record_check( + connection, + site_id=row["id"], + kind="recheck", + result="pass", + http_status=304, + duration_ms=elapsed, + ) + return "pass" + + if response.status != OK: + return _transient(connection, row, f"Your page returned {response.status}.") + + report = validate_bytes( + response.body, url=response.url, response=response, fetcher=fetcher + ) + db.record_check( + connection, + site_id=row["id"], + kind="recheck", + result="pass" if report.conforms else "fail", + findings=_findings_rows(report), + http_status=response.status, + size=report.size, + duration_ms=elapsed, + ) + if report.conforms: + db.mark_pass( + connection, + row["id"], + etag=response.headers.get("etag"), + last_modified=response.headers.get("last-modified"), + ) + return "pass" + + db.remove(connection, row["id"], str(report.failures[0])) + return "fail" + + +def _transient(connection: sqlite3.Connection, row: sqlite3.Row, note: str) -> str: + """Count a failure that wasn't the page's fault, dropping the site if it sticks.""" + count = db.mark_transient(connection, row["id"]) + db.record_check(connection, site_id=row["id"], kind="recheck", result="unreachable") + if count >= db.TRANSIENT_LIMIT: + db.remove( + connection, + row["id"], + f"{note} We tried for {db.TRANSIENT_LIMIT} days.", + ) + return "fail" + return "unreachable" diff --git a/mews/cli.py b/mews/cli.py new file mode 100644 index 0000000..72bbbd4 --- /dev/null +++ b/mews/cli.py @@ -0,0 +1,138 @@ +"""mewsd: run the service, and look after the directory from the shell.""" + +import argparse +import sys +import time + +from mews import check as checker, db, pages +from mews.app import directory_path, serve +from mews.fetch import Fetcher, registered_domain + +RECHECK_BATCH = 50 +PAUSE_SECONDS = 2.0 + + +def main(argv: list[str] | None = None) -> int: + """Run one command. Returns an exit status.""" + parser = argparse.ArgumentParser( + prog="mewsd", description="Run and look after the mews.page directory." + ) + commands = parser.add_subparsers(dest="command", required=True) + commands.add_parser("serve", help="serve the check endpoint") + commands.add_parser("init-db", help="create the database if it isn't there") + commands.add_parser("recheck", help="check the sites that are due") + commands.add_parser("build-directory", help="write the directory page again") + drop = commands.add_parser("remove", help="take a site out and refuse it later") + drop.add_argument("domain") + drop.add_argument("--reason", required=True, help="shown in the log and to you") + allow = commands.add_parser("unblock", help="allow a removed domain again") + allow.add_argument("domain") + show = commands.add_parser("show", help="what the last check of a site found") + show.add_argument("domain") + args = parser.parse_args(argv) + + if args.command == "serve": + serve() + return 0 + + connection = db.connect() + try: + db.init(connection) + if args.command == "init-db": + print(f"database ready at {db.path()}") + return 0 + if args.command == "recheck": + return _recheck(connection) + if args.command == "build-directory": + return _build(connection) + if args.command == "remove": + return _remove(connection, args.domain, args.reason) + if args.command == "unblock": + return _unblock(connection, args.domain) + if args.command == "show": + return _show(connection, args.domain) + finally: + connection.close() + return 0 + + +def _domain(name: str) -> str: + return registered_domain(name) or name.strip().lower() + + +def _recheck(connection) -> int: + """Check every site that is due, then write the directory page again.""" + sites = db.due(connection, RECHECK_BATCH) + counts = {"pass": 0, "fail": 0, "unreachable": 0} + for index, row in enumerate(sites): + if index: + # One site at a time, with a gap, so a recheck pass is invisible to + # the sites it visits. + time.sleep(PAUSE_SECONDS) + with Fetcher() as fetcher: + counts[checker.recheck(connection, row, fetcher=fetcher)] += 1 + db.prune(connection) + _build(connection) + print( + f"checked {len(sites)}: {counts['pass']} passed, {counts['fail']} dropped, " + f"{counts['unreachable']} unreachable" + ) + return 0 + + +def _build(connection) -> int: + """Write the directory page from what is listed now.""" + try: + warnings = pages.write_directory(directory_path(), db.listed(connection)) + except ValueError as error: + print(f"the directory page wasn't written: {error}", file=sys.stderr) + return 1 + for warning in warnings: + print(f"note: {warning}", file=sys.stderr) + print(f"wrote {directory_path()}") + return 0 + + +def _remove(connection, domain: str, reason: str) -> int: + name = _domain(domain) + row = db.site(connection, name) + if row is not None: + db.remove(connection, row["id"], reason) + db.block(connection, name, reason) + _build(connection) + print(f"removed {name}") + return 0 + + +def _unblock(connection, domain: str) -> int: + name = _domain(domain) + db.unblock(connection, name) + print(f"{name} can be submitted again") + return 0 + + +def _show(connection, domain: str) -> int: + name = _domain(domain) + row = db.site(connection, name) + if row is None: + print(f"{name} isn't in the directory", file=sys.stderr) + return 1 + print(f"{row['domain']} — {row['state']}") + print(f" page {row['url']}") + print(f" title {row['title']}") + print(f" listed {row['listed_at']}") + print(f" checked {row['last_check_at']} (last passed {row['last_ok_at']})") + print(f" next {row['next_check_at']}") + if row["reason"]: + print(f" dropped {row['reason']}") + findings = db.last_findings(connection, row["id"]) + for finding in findings: + print( + f" {finding['level']:6} section {finding['section']} — " + f"{finding['message']}" + ) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/mews/db.py b/mews/db.py new file mode 100644 index 0000000..5cb1299 --- /dev/null +++ b/mews/db.py @@ -0,0 +1,486 @@ +"""The directory's SQLite store. + +One row per site, keyed by registrable domain, plus the submission and check +history the rate limits and the recheck schedule read. Client addresses are +never stored: a submission keeps a salted hash, and the salt is replaced every +month so old hashes stop being comparable to new ones. +""" + +from datetime import UTC, datetime, timedelta +import hashlib +import os +import secrets +import sqlite3 + +SCHEMA_VERSION = "1" + +RECHECK_DAYS = 7 +RETRY_HOURS = 24 +TRANSIENT_LIMIT = 3 +RETENTION_DAYS = 90 + +# Rate limits. Listing is slower than checking because it writes to the +# directory; checking only costs a fetch. +IP_LISTINGS_PER_HOUR = 5 +IP_LISTINGS_PER_DAY = 20 +IP_CHECKS_PER_HOUR = 20 +DOMAIN_COOLDOWN_MINUTES = 10 +DOMAIN_REJECTS_BEFORE_SLOWDOWN = 3 +DOMAIN_SLOW_COOLDOWN_HOURS = 24 +LISTINGS_PER_HOUR = 60 + +SCHEMA = """ +CREATE TABLE IF NOT EXISTS meta ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS sites ( + id INTEGER PRIMARY KEY, + domain TEXT NOT NULL UNIQUE, + url TEXT NOT NULL, + title TEXT NOT NULL, + description TEXT, + language TEXT, + state TEXT NOT NULL CHECK (state IN ('listed', 'removed')), + reason TEXT, + listed_at TEXT NOT NULL, + last_check_at TEXT, + last_ok_at TEXT, + next_check_at TEXT, + transient_failures INTEGER NOT NULL DEFAULT 0, + etag TEXT, + last_modified TEXT +); +CREATE INDEX IF NOT EXISTS sites_due + ON sites(next_check_at) WHERE state = 'listed'; +CREATE INDEX IF NOT EXISTS sites_state ON sites(state, domain); + +CREATE TABLE IF NOT EXISTS submissions ( + id INTEGER PRIMARY KEY, + created_at TEXT NOT NULL, + url TEXT NOT NULL, + domain TEXT, + ip_hash TEXT NOT NULL, + outcome TEXT NOT NULL CHECK (outcome IN ( + 'listed', 'updated', 'rejected', 'unreachable', + 'rate_limited', 'blocked', 'checked')), + site_id INTEGER REFERENCES sites(id) ON DELETE SET NULL +); +CREATE INDEX IF NOT EXISTS submissions_ip ON submissions(ip_hash, created_at); +CREATE INDEX IF NOT EXISTS submissions_domain ON submissions(domain, created_at); + +CREATE TABLE IF NOT EXISTS checks ( + id INTEGER PRIMARY KEY, + site_id INTEGER REFERENCES sites(id) ON DELETE CASCADE, + checked_at TEXT NOT NULL, + kind TEXT NOT NULL CHECK (kind IN ('submission', 'recheck')), + result TEXT NOT NULL CHECK (result IN ('pass', 'fail', 'unreachable')), + http_status INTEGER, + bytes INTEGER, + duration_ms INTEGER +); +CREATE INDEX IF NOT EXISTS checks_site ON checks(site_id, checked_at DESC); + +CREATE TABLE IF NOT EXISTS findings ( + check_id INTEGER NOT NULL REFERENCES checks(id) ON DELETE CASCADE, + section TEXT NOT NULL, + level TEXT NOT NULL CHECK (level IN ('must', 'should')), + code TEXT NOT NULL, + message TEXT NOT NULL, + location TEXT +); +CREATE INDEX IF NOT EXISTS findings_check ON findings(check_id); + +CREATE TABLE IF NOT EXISTS blocked ( + domain TEXT PRIMARY KEY, + blocked_at TEXT NOT NULL, + reason TEXT NOT NULL +); +""" + + +def now() -> datetime: + """Return the current time, in UTC.""" + return datetime.now(UTC) + + +def path() -> str: + """Return where the database lives.""" + return os.environ.get("MEWS_DB", "mews.db") + + +def connect(database: str | None = None) -> sqlite3.Connection: + """Open the database with the settings every caller wants.""" + connection = sqlite3.connect(database or path(), timeout=5.0) + connection.row_factory = sqlite3.Row + connection.execute("PRAGMA journal_mode = WAL") + connection.execute("PRAGMA foreign_keys = ON") + connection.execute("PRAGMA busy_timeout = 5000") + return connection + + +def init(connection: sqlite3.Connection) -> None: + """Create the schema if it isn't there yet.""" + connection.executescript(SCHEMA) + connection.execute( + "INSERT OR IGNORE INTO meta (key, value) VALUES ('schema_version', ?)", + (SCHEMA_VERSION,), + ) + connection.commit() + + +def ip_hash(connection: sqlite3.Connection, address: str) -> str: + """Hash a client address with a salt that is replaced each month.""" + stamp = now().strftime("%Y-%m") + row = connection.execute( + "SELECT value FROM meta WHERE key = 'ip_salt_month'" + ).fetchone() + if row is None or row["value"] != stamp: + connection.execute( + "INSERT INTO meta (key, value) VALUES ('ip_salt', ?) " + "ON CONFLICT(key) DO UPDATE SET value = excluded.value", + (secrets.token_hex(16),), + ) + connection.execute( + "INSERT INTO meta (key, value) VALUES ('ip_salt_month', ?) " + "ON CONFLICT(key) DO UPDATE SET value = excluded.value", + (stamp,), + ) + connection.commit() + salt = connection.execute( + "SELECT value FROM meta WHERE key = 'ip_salt'" + ).fetchone()["value"] + return hashlib.sha256(f"{salt}{address}".encode()).hexdigest() + + +def _since(hours: float) -> str: + return (now() - timedelta(hours=hours)).isoformat() + + +def _count(connection: sqlite3.Connection, sql: str, *args: object) -> int: + return connection.execute(sql, args).fetchone()[0] + + +def blocked_reason(connection: sqlite3.Connection, domain: str) -> str | None: + """Return why a domain was removed for good, or None if it wasn't.""" + row = connection.execute( + "SELECT reason FROM blocked WHERE domain = ?", (domain,) + ).fetchone() + return row["reason"] if row else None + + +def rate_limited( + connection: sqlite3.Connection, *, client: str, domain: str, listing: bool +) -> str | None: + """Return why this request can't go ahead now, as a sentence for the author. + + Checking a page is cheap and gets a looser limit than listing a site. + """ + if not listing: + checks = _count( + connection, + "SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ?", + client, + _since(1), + ) + if checks >= IP_CHECKS_PER_HOUR: + return "You've checked several pages already. Try again in an hour." + return None + + if ( + _count( + connection, + "SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? " + "AND outcome IN ('listed', 'updated', 'rejected')", + client, + _since(1), + ) + >= IP_LISTINGS_PER_HOUR + ): + return "You've submitted several sites already. Try again in an hour." + if ( + _count( + connection, + "SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? " + "AND outcome IN ('listed', 'updated', 'rejected')", + client, + _since(24), + ) + >= IP_LISTINGS_PER_DAY + ): + return "You've submitted several sites already. Try again tomorrow." + + rejects = _count( + connection, + "SELECT COUNT(*) FROM submissions WHERE domain = ? AND outcome = 'rejected' " + "AND created_at > ?", + domain, + _since(24), + ) + cooldown = ( + DOMAIN_SLOW_COOLDOWN_HOURS + if rejects >= DOMAIN_REJECTS_BEFORE_SLOWDOWN + else DOMAIN_COOLDOWN_MINUTES / 60 + ) + if _count( + connection, + "SELECT COUNT(*) FROM submissions WHERE domain = ? AND created_at > ?", + domain, + _since(cooldown), + ): + if cooldown > 1: + return ( + f"{domain} didn't pass the checks a few times today. Fix the " + "page and try again tomorrow." + ) + return f"{domain} was submitted a moment ago. Try again in ten minutes." + + if ( + _count( + connection, + "SELECT COUNT(*) FROM submissions WHERE created_at > ? " + "AND outcome IN ('listed', 'updated')", + _since(1), + ) + >= LISTINGS_PER_HOUR + ): + return "The directory is busy right now. Try again in an hour." + return None + + +def record_submission( + connection: sqlite3.Connection, + *, + url: str, + domain: str | None, + client: str, + outcome: str, + site_id: int | None = None, +) -> int: + """Store one submission and return its id.""" + cursor = connection.execute( + "INSERT INTO submissions (created_at, url, domain, ip_hash, outcome, site_id) " + "VALUES (?, ?, ?, ?, ?, ?)", + (now().isoformat(), url, domain, client, outcome, site_id), + ) + connection.commit() + return int(cursor.lastrowid or 0) + + +def record_check( + connection: sqlite3.Connection, + *, + site_id: int | None, + kind: str, + result: str, + findings: list[tuple[str, str, str, str, str]] = [], # noqa: B006 + http_status: int | None = None, + size: int | None = None, + duration_ms: int | None = None, +) -> int: + """Store one check and its findings, returning the check id.""" + cursor = connection.execute( + "INSERT INTO checks (site_id, checked_at, kind, result, http_status, bytes, " + "duration_ms) VALUES (?, ?, ?, ?, ?, ?, ?)", + (site_id, now().isoformat(), kind, result, http_status, size, duration_ms), + ) + check_id = int(cursor.lastrowid or 0) + connection.executemany( + "INSERT INTO findings (check_id, section, level, code, message, location) " + "VALUES (?, ?, ?, ?, ?, ?)", + [(check_id, *finding) for finding in findings], + ) + connection.commit() + return check_id + + +def next_check(moment: datetime | None = None) -> str: + """Return when a site that just passed should be looked at again. + + The interval is jittered so submissions made together do not come back as a + burst a week later. + """ + base = (moment or now()) + timedelta(days=RECHECK_DAYS) + offset = secrets.randbelow(24 * 60) - 12 * 60 + return (base + timedelta(minutes=offset)).isoformat() + + +def upsert_site( + connection: sqlite3.Connection, + *, + domain: str, + url: str, + title: str, + description: str, + language: str, + etag: str | None, + last_modified: str | None, +) -> tuple[int, bool]: + """List a site, or refresh the row of one already listed. + + Returns the site id and whether the row already existed. + """ + moment = now().isoformat() + row = connection.execute( + "SELECT id FROM sites WHERE domain = ?", (domain,) + ).fetchone() + if row is None: + cursor = connection.execute( + "INSERT INTO sites (domain, url, title, description, language, state, " + "listed_at, last_check_at, last_ok_at, next_check_at, etag, last_modified) " + "VALUES (?, ?, ?, ?, ?, 'listed', ?, ?, ?, ?, ?, ?)", + ( + domain, + url, + title, + description, + language, + moment, + moment, + moment, + next_check(), + etag, + last_modified, + ), + ) + connection.commit() + return int(cursor.lastrowid or 0), False + + connection.execute( + "UPDATE sites SET url = ?, title = ?, description = ?, language = ?, " + "state = 'listed', reason = NULL, last_check_at = ?, last_ok_at = ?, " + "next_check_at = ?, transient_failures = 0, etag = ?, last_modified = ? " + "WHERE id = ?", + ( + url, + title, + description, + language, + moment, + moment, + next_check(), + etag, + last_modified, + row["id"], + ), + ) + connection.commit() + return int(row["id"]), True + + +def listed(connection: sqlite3.Connection) -> list[sqlite3.Row]: + """Return every listed site, in the order the directory page shows them.""" + return list( + connection.execute( + "SELECT * FROM sites WHERE state = 'listed' ORDER BY domain" + ).fetchall() + ) + + +def due(connection: sqlite3.Connection, limit: int = 50) -> list[sqlite3.Row]: + """Return the listed sites whose next check has come round.""" + return list( + connection.execute( + "SELECT * FROM sites WHERE state = 'listed' AND next_check_at <= ? " + "ORDER BY next_check_at LIMIT ?", + (now().isoformat(), limit), + ).fetchall() + ) + + +def site(connection: sqlite3.Connection, domain: str) -> sqlite3.Row | None: + """Return one site by registrable domain.""" + return connection.execute( + "SELECT * FROM sites WHERE domain = ?", (domain,) + ).fetchone() + + +def mark_pass(connection: sqlite3.Connection, site_id: int, **columns: object) -> None: + """Record a check a site passed.""" + moment = now().isoformat() + connection.execute( + "UPDATE sites SET last_check_at = ?, last_ok_at = ?, next_check_at = ?, " + "transient_failures = 0, etag = ?, last_modified = ? WHERE id = ?", + ( + moment, + moment, + next_check(), + columns.get("etag"), + columns.get("last_modified"), + site_id, + ), + ) + connection.commit() + + +def mark_transient(connection: sqlite3.Connection, site_id: int) -> int: + """Count one failure that wasn't the page's fault, and return the new count. + + A site is only dropped once these pile up: a weekend of downtime should not + empty the directory. + """ + connection.execute( + "UPDATE sites SET transient_failures = transient_failures + 1, " + "last_check_at = ?, next_check_at = ? WHERE id = ?", + ( + now().isoformat(), + (now() + timedelta(hours=RETRY_HOURS)).isoformat(), + site_id, + ), + ) + connection.commit() + return int( + connection.execute( + "SELECT transient_failures FROM sites WHERE id = ?", (site_id,) + ).fetchone()[0] + ) + + +def remove(connection: sqlite3.Connection, site_id: int, reason: str) -> None: + """Drop a site from the directory, keeping the row for the record.""" + connection.execute( + "UPDATE sites SET state = 'removed', reason = ?, last_check_at = ?, " + "next_check_at = NULL WHERE id = ?", + (reason, now().isoformat(), site_id), + ) + connection.commit() + + +def block(connection: sqlite3.Connection, domain: str, reason: str) -> None: + """Refuse future submissions from a domain.""" + connection.execute( + "INSERT INTO blocked (domain, blocked_at, reason) VALUES (?, ?, ?) " + "ON CONFLICT(domain) DO UPDATE SET reason = excluded.reason", + (domain, now().isoformat(), reason), + ) + connection.commit() + + +def unblock(connection: sqlite3.Connection, domain: str) -> None: + """Allow submissions from a domain again.""" + connection.execute("DELETE FROM blocked WHERE domain = ?", (domain,)) + connection.commit() + + +def last_findings(connection: sqlite3.Connection, site_id: int) -> list[sqlite3.Row]: + """Return the findings from a site's most recent check.""" + row = connection.execute( + "SELECT id FROM checks WHERE site_id = ? ORDER BY checked_at DESC LIMIT 1", + (site_id,), + ).fetchone() + if row is None: + return [] + return list( + connection.execute( + "SELECT * FROM findings WHERE check_id = ?", (row["id"],) + ).fetchall() + ) + + +def prune(connection: sqlite3.Connection, days: int = RETENTION_DAYS) -> None: + """Delete submission and check history past the retention window.""" + cutoff = (now() - timedelta(days=days)).isoformat() + connection.execute("DELETE FROM submissions WHERE created_at < ?", (cutoff,)) + connection.execute("DELETE FROM checks WHERE checked_at < ?", (cutoff,)) + connection.commit() diff --git a/mews/pages.py b/mews/pages.py new file mode 100644 index 0000000..f3db756 --- /dev/null +++ b/mews/pages.py @@ -0,0 +1,221 @@ +"""Emitting the pages mews.page serves. + +Every page the service produces is itself a Mews page, so this module writes the +same prologue, head and body shape as md2mews.py. The emitter is repeated here +rather than imported because md2mews.py is a standalone script that pulls in a +Markdown parser; tests/test_pages.py fails if the two drift apart. +""" + +from datetime import UTC, datetime +import fcntl +from html import escape +import os +from pathlib import Path +import sqlite3 + +from mews import VERSION +from mews.lint import validate_bytes + +DOCTYPE = ( + '' +) +VARIANTS = ("mews-warm", "mews-cool", "mews-green", "mews-mono") +STYLESHEET = "mews-0.1.css" +SITE_VARIANT = "mews-cool" + +# SPEC.md 4.3 caps a page at 256 KB and asks for 64 KB. The directory refuses to +# publish a page over the first and warns past the second. +SIZE_MUST = 256 * 1024 +SIZE_SHOULD = 64 * 1024 + + +def _text(value: str) -> str: + """Escape text content, leaving apostrophes alone so prose reads as prose.""" + return escape(value, quote=False) + + +def _attr(value: str) -> str: + """Escape a value going into an attribute.""" + return escape(value, quote=True) + + +def page( + title: str, + body: list[str], + language: str = "en", + direction: str | None = None, + variant: str | None = None, + description: str | None = None, +) -> str: + """Wrap body lines in a complete Mews page. Mirrors md2mews.page().""" + attrs = f'xml:lang="{language}" lang="{language}"' + if direction: + attrs += f' dir="{direction}"' + head = [ + f"
{_text(line)}
" for line in paragraphs] + body.append("{_text(opening)}
"] + if findings_report.url: + body.append( + f'Checked ' + f"{_text(findings_report.url)}.
" + ) + + if findings_report.failures: + body.append("{_text(note)}
") + + body.append("The rules are all in the spec. Checking ' + "on your own machine is quicker while you are still fixing things:
" + ) + body.append(f"{_text(command)}")
+ body.append(
+ ' Check another page ' + '· The directory
' + ) + return site_page(title, body) + + +def directory(rows: list[sqlite3.Row]) -> str: + """Build the directory page: every site whose page passes the checks.""" + body = ["{count} listed. Every one of these pages passed the checks " + 'against Mews Profile 0.1, and gets looked ' + "at again every week.
" + ) + body.append("Nothing here yet. Yours could be the first one in!
") + body.append("Add your site ' + '· Mews
' + ) + body.append( + f" Updated {datetime.now(UTC).strftime('%Y-%m-%d %H:%M')} " + "UTC" + ) + return site_page( + "Mews directory", + body, + description="Sites whose pages follow Mews Profile 0.1.", + ) + + +def write_directory(target: str, rows: list[sqlite3.Row]) -> list[str]: + """Write the directory page, replacing it only if it conforms. + + The page is validated before it is published, because the one page that has + to be a good example is this one. Returns any warnings about it. + """ + destination = Path(target) + destination.parent.mkdir(parents=True, exist_ok=True) + text = directory(rows).encode() + if len(text) > SIZE_MUST: + raise ValueError( + "The directory page would be over 256 KB, which section 4.3 " + "forbids. Split it before listing more sites." + ) + + lock = destination.with_name(destination.name + ".lock") + with open(lock, "w") as handle: + fcntl.flock(handle, fcntl.LOCK_EX) + result = validate_bytes(text) + if not result.conforms: + raise ValueError( + "The directory page doesn't conform: " + + "; ".join(str(f) for f in result.failures) + ) + temporary = destination.with_name("." + destination.name + ".tmp") + temporary.write_bytes(text) + os.replace(temporary, destination) + + warnings = [str(f) for f in result.warnings] + if len(text) > SIZE_SHOULD: + warnings.append( + f"The directory page is {len(text) // 1024} KB, over the 64 KB " + "section 4.3 asks for." + ) + return warnings