feat: site directory with a public check endpoint

This commit is contained in:
randogoth 2026-10-11 15:05:38 +03:00
parent b3c436bf1a
commit 5fa14bdf1b
5 changed files with 1279 additions and 0 deletions

165
mews/app.py Normal file
View file

@ -0,0 +1,165 @@
"""The one dynamic page on mews.page.
The site is static apart from this: a form posts a page address to /result, the
page is fetched and checked, and the author gets the findings back. Nothing here
runs JavaScript or sets a cookie, because the response is itself a Mews page.
"""
from collections.abc import Iterable
import os
import threading
from urllib.parse import parse_qs
from mews import db, pages
from mews.check import submit
from mews.fetch import Fetcher
# Outbound checks are slow and this runs on a small machine, so only a couple
# happen at once; the rest of the queue is turned away rather than piled up.
_CHECKS = threading.BoundedSemaphore(2)
_REQUESTS = threading.BoundedSemaphore(8)
_SLOT_WAIT = 5.0
MAX_BODY = 4096
BUSY = "The checker is busy right now. Try again in a few minutes."
def directory_path() -> str:
"""Where the directory page is written."""
return os.environ.get("MEWS_DIRECTORY", "directory.html")
def _start(start_response, status: str, body: bytes) -> list[bytes]:
start_response(
status,
[
("Content-Type", "text/html; charset=utf-8"),
("Content-Length", str(len(body))),
("Cache-Control", "no-store"),
],
)
return [body]
def _client(environ: dict) -> str:
"""Return the reader's address, as the proxy reports it.
Caddy sets X-Real-IP from the connection it accepted. The server binds to
the loopback address only, so nothing else can set that header.
"""
return environ.get("HTTP_X_REAL_IP") or environ.get("REMOTE_ADDR", "")
def _form(environ: dict) -> dict[str, list[str]]:
try:
length = min(int(environ.get("CONTENT_LENGTH") or 0), MAX_BODY)
except ValueError:
return {}
if length <= 0:
return {}
raw = environ["wsgi.input"].read(length)
return parse_qs(raw.decode("utf-8", "replace"), keep_blank_values=True)
def application(environ: dict, start_response) -> Iterable[bytes]:
"""Handle one request."""
path = environ.get("PATH_INFO", "/")
method = environ.get("REQUEST_METHOD", "GET")
if path != "/result":
return _start(
start_response,
"404 Not Found",
pages.message(
"Page not found",
["That address isn't part of this site."],
[("/", "Mews"), ("/directory", "The directory")],
).encode(),
)
if method != "POST":
return _start(
start_response,
"405 Method Not Allowed",
pages.message(
"Nothing to show",
["Results appear here after you check a page."],
[("/check", "Check a page")],
).encode(),
)
form = _form(environ)
raw_url = (form.get("url") or [""])[0].strip()
listing = "list" in form
if not _REQUESTS.acquire(blocking=False):
return _start(start_response, "503 Service Unavailable", _busy())
try:
if not _CHECKS.acquire(timeout=_SLOT_WAIT):
return _start(start_response, "503 Service Unavailable", _busy())
try:
body = _check(raw_url, _client(environ), listing)
finally:
_CHECKS.release()
finally:
_REQUESTS.release()
return _start(start_response, "200 OK", body)
def _busy() -> bytes:
return pages.message("Busy", [BUSY], [("/check", "Try again")]).encode()
def _check(raw_url: str, client: str, listing: bool) -> bytes:
"""Run one check and render the page the author gets back."""
connection = db.connect()
try:
with Fetcher() as fetcher:
outcome = submit(
connection,
raw_url,
client=db.ip_hash(connection, client),
listing=listing,
fetcher=fetcher,
directory=directory_path() if listing else None,
)
except ValueError as error:
# The directory page refused to publish itself. The site is listed; the
# page will be rebuilt by the next recheck pass.
return pages.message(
"Listed, but the directory didn't rebuild",
[str(error), "Your site is listed and will appear shortly."],
[("/directory", "The directory")],
).encode()
finally:
connection.close()
if outcome.report is None:
return pages.message(
"We couldn't check that page",
[outcome.message or "Something went wrong. Try again."],
[("/check", "Try again"), ("/spec/0.1", "The spec")],
).encode()
command = f"uv run mewslint.py {outcome.report.url or raw_url}"
return pages.report(outcome.report, listed=outcome.listed, command=command).encode()
def serve() -> None:
"""Run the app for real, on the loopback address only."""
# Imported here so the validator and the command line tools do not pull in
# a web server.
from waitress import serve as waitress_serve # noqa: PLC0415
listen = os.environ.get("MEWS_LISTEN", "127.0.0.1:8394")
host = listen.rsplit(":", 1)[0]
if host not in ("127.0.0.1", "::1", "localhost"):
raise SystemExit(
"mewsd serves on the loopback address only, behind a reverse proxy. "
f"Set MEWS_LISTEN to 127.0.0.1 with a port, not {listen}."
)
waitress_serve(application, listen=listen, threads=4, ident="mews.page")
app = application

269
mews/check.py Normal file
View file

@ -0,0 +1,269 @@
"""Checking a page and recording what came of it.
Both paths into the directory run through here: a submission from the form and
a recheck from the timer. Keeping them together means a site is listed and
relisted on exactly the same evidence.
"""
from dataclasses import dataclass
import sqlite3
import time
from urllib.parse import urlsplit
from mews import db, pages
from mews.fetch import (
Fetched,
Fetcher,
FetchError,
UrlError,
normalise_url,
registered_domain,
)
from mews.lint import Report, validate_bytes
OK = 200
NOT_MODIFIED = 304
@dataclass
class Outcome:
"""What a submission produced: either a report, or a reason there isn't one."""
report: Report | None = None
listed: bool = False
message: str | None = None
title: str = ""
def _findings_rows(report: Report) -> list[tuple[str, str, str, str, str]]:
return [
(f.section, f.level, f.code, f.message, f.location) for f in report.findings
]
def check_page(
url: str, *, fetcher: Fetcher
) -> tuple[Report | None, Fetched | None, str | None]:
"""Fetch and validate one page.
Returns the report, the response it came from, and a plain message when
there was no page to check.
"""
try:
response = fetcher.get(url)
except (UrlError, FetchError) as error:
return None, None, str(error)
if response.status != OK:
return (
None,
response,
f"We couldn't read that page (it returned {response.status}). "
"Check the address and try again.",
)
report = validate_bytes(
response.body, url=response.url, response=response, fetcher=fetcher
)
return report, response, None
def submit(
connection: sqlite3.Connection,
raw_url: str,
*,
client: str,
listing: bool,
fetcher: Fetcher,
directory: str | None = None,
) -> Outcome:
"""Check a page, and list its site when the author asked and it conforms."""
try:
url = normalise_url(raw_url, allow_loopback=fetcher.allow_loopback)
except UrlError as error:
return Outcome(message=str(error))
host = urlsplit(url).hostname or ""
domain = registered_domain(host)
if domain is None:
return Outcome(message="That domain name isn't one the checker can reach.")
reason = db.blocked_reason(connection, domain)
if reason is not None:
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="blocked"
)
return Outcome(
message="This site was taken out of the directory. Write to the "
"address on the about page if that looks wrong."
)
limited = db.rate_limited(connection, client=client, domain=domain, listing=listing)
if limited is not None:
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="rate_limited"
)
return Outcome(message=limited)
started = time.monotonic()
report, response, failure = check_page(url, fetcher=fetcher)
elapsed = int((time.monotonic() - started) * 1000)
status = response.status if response else None
if report is None:
db.record_check(
connection,
site_id=None,
kind="submission",
result="unreachable",
http_status=status,
duration_ms=elapsed,
)
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="unreachable"
)
return Outcome(message=failure)
if not listing:
db.record_check(
connection,
site_id=None,
kind="submission",
result="pass" if report.conforms else "fail",
findings=_findings_rows(report),
http_status=status,
size=report.size,
duration_ms=elapsed,
)
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="checked"
)
return Outcome(report=report)
if not report.conforms:
db.record_check(
connection,
site_id=None,
kind="submission",
result="fail",
findings=_findings_rows(report),
http_status=status,
size=report.size,
duration_ms=elapsed,
)
db.record_submission(
connection, url=url, domain=domain, client=client, outcome="rejected"
)
return Outcome(report=report)
site_id, existed = db.upsert_site(
connection,
domain=domain,
url=report.url or url,
title=report.title or domain,
description=report.description,
language=report.language,
etag=response.headers.get("etag") if response else None,
last_modified=response.headers.get("last-modified") if response else None,
)
db.record_check(
connection,
site_id=site_id,
kind="submission",
result="pass",
findings=_findings_rows(report),
http_status=status,
size=report.size,
duration_ms=elapsed,
)
db.record_submission(
connection,
url=url,
domain=domain,
client=client,
outcome="updated" if existed else "listed",
site_id=site_id,
)
if directory:
pages.write_directory(directory, db.listed(connection))
return Outcome(report=report, listed=True, title=report.title)
def recheck(
connection: sqlite3.Connection, row: sqlite3.Row, *, fetcher: Fetcher
) -> str:
"""Check one listed site again, and drop it if its page stopped conforming.
A page that no longer conforms goes immediately: there is no contact address
to warn, and resubmitting after a fix lists the site again at once. A site
we simply could not reach is retried, because downtime is not a conformance
failure.
"""
conditional: dict[str, str] = {}
if row["etag"]:
conditional["If-None-Match"] = row["etag"]
if row["last_modified"]:
conditional["If-Modified-Since"] = row["last_modified"]
started = time.monotonic()
try:
response = fetcher.get(row["url"], headers=conditional or None)
except (UrlError, FetchError):
return _transient(connection, row, "We couldn't reach your site.")
elapsed = int((time.monotonic() - started) * 1000)
if response.status == NOT_MODIFIED:
db.mark_pass(
connection,
row["id"],
etag=row["etag"],
last_modified=row["last_modified"],
)
db.record_check(
connection,
site_id=row["id"],
kind="recheck",
result="pass",
http_status=304,
duration_ms=elapsed,
)
return "pass"
if response.status != OK:
return _transient(connection, row, f"Your page returned {response.status}.")
report = validate_bytes(
response.body, url=response.url, response=response, fetcher=fetcher
)
db.record_check(
connection,
site_id=row["id"],
kind="recheck",
result="pass" if report.conforms else "fail",
findings=_findings_rows(report),
http_status=response.status,
size=report.size,
duration_ms=elapsed,
)
if report.conforms:
db.mark_pass(
connection,
row["id"],
etag=response.headers.get("etag"),
last_modified=response.headers.get("last-modified"),
)
return "pass"
db.remove(connection, row["id"], str(report.failures[0]))
return "fail"
def _transient(connection: sqlite3.Connection, row: sqlite3.Row, note: str) -> str:
"""Count a failure that wasn't the page's fault, dropping the site if it sticks."""
count = db.mark_transient(connection, row["id"])
db.record_check(connection, site_id=row["id"], kind="recheck", result="unreachable")
if count >= db.TRANSIENT_LIMIT:
db.remove(
connection,
row["id"],
f"{note} We tried for {db.TRANSIENT_LIMIT} days.",
)
return "fail"
return "unreachable"

138
mews/cli.py Normal file
View file

@ -0,0 +1,138 @@
"""mewsd: run the service, and look after the directory from the shell."""
import argparse
import sys
import time
from mews import check as checker, db, pages
from mews.app import directory_path, serve
from mews.fetch import Fetcher, registered_domain
RECHECK_BATCH = 50
PAUSE_SECONDS = 2.0
def main(argv: list[str] | None = None) -> int:
"""Run one command. Returns an exit status."""
parser = argparse.ArgumentParser(
prog="mewsd", description="Run and look after the mews.page directory."
)
commands = parser.add_subparsers(dest="command", required=True)
commands.add_parser("serve", help="serve the check endpoint")
commands.add_parser("init-db", help="create the database if it isn't there")
commands.add_parser("recheck", help="check the sites that are due")
commands.add_parser("build-directory", help="write the directory page again")
drop = commands.add_parser("remove", help="take a site out and refuse it later")
drop.add_argument("domain")
drop.add_argument("--reason", required=True, help="shown in the log and to you")
allow = commands.add_parser("unblock", help="allow a removed domain again")
allow.add_argument("domain")
show = commands.add_parser("show", help="what the last check of a site found")
show.add_argument("domain")
args = parser.parse_args(argv)
if args.command == "serve":
serve()
return 0
connection = db.connect()
try:
db.init(connection)
if args.command == "init-db":
print(f"database ready at {db.path()}")
return 0
if args.command == "recheck":
return _recheck(connection)
if args.command == "build-directory":
return _build(connection)
if args.command == "remove":
return _remove(connection, args.domain, args.reason)
if args.command == "unblock":
return _unblock(connection, args.domain)
if args.command == "show":
return _show(connection, args.domain)
finally:
connection.close()
return 0
def _domain(name: str) -> str:
return registered_domain(name) or name.strip().lower()
def _recheck(connection) -> int:
"""Check every site that is due, then write the directory page again."""
sites = db.due(connection, RECHECK_BATCH)
counts = {"pass": 0, "fail": 0, "unreachable": 0}
for index, row in enumerate(sites):
if index:
# One site at a time, with a gap, so a recheck pass is invisible to
# the sites it visits.
time.sleep(PAUSE_SECONDS)
with Fetcher() as fetcher:
counts[checker.recheck(connection, row, fetcher=fetcher)] += 1
db.prune(connection)
_build(connection)
print(
f"checked {len(sites)}: {counts['pass']} passed, {counts['fail']} dropped, "
f"{counts['unreachable']} unreachable"
)
return 0
def _build(connection) -> int:
"""Write the directory page from what is listed now."""
try:
warnings = pages.write_directory(directory_path(), db.listed(connection))
except ValueError as error:
print(f"the directory page wasn't written: {error}", file=sys.stderr)
return 1
for warning in warnings:
print(f"note: {warning}", file=sys.stderr)
print(f"wrote {directory_path()}")
return 0
def _remove(connection, domain: str, reason: str) -> int:
name = _domain(domain)
row = db.site(connection, name)
if row is not None:
db.remove(connection, row["id"], reason)
db.block(connection, name, reason)
_build(connection)
print(f"removed {name}")
return 0
def _unblock(connection, domain: str) -> int:
name = _domain(domain)
db.unblock(connection, name)
print(f"{name} can be submitted again")
return 0
def _show(connection, domain: str) -> int:
name = _domain(domain)
row = db.site(connection, name)
if row is None:
print(f"{name} isn't in the directory", file=sys.stderr)
return 1
print(f"{row['domain']} — {row['state']}")
print(f" page {row['url']}")
print(f" title {row['title']}")
print(f" listed {row['listed_at']}")
print(f" checked {row['last_check_at']} (last passed {row['last_ok_at']})")
print(f" next {row['next_check_at']}")
if row["reason"]:
print(f" dropped {row['reason']}")
findings = db.last_findings(connection, row["id"])
for finding in findings:
print(
f" {finding['level']:6} section {finding['section']} — "
f"{finding['message']}"
)
return 0
if __name__ == "__main__":
sys.exit(main())

486
mews/db.py Normal file
View file

@ -0,0 +1,486 @@
"""The directory's SQLite store.
One row per site, keyed by registrable domain, plus the submission and check
history the rate limits and the recheck schedule read. Client addresses are
never stored: a submission keeps a salted hash, and the salt is replaced every
month so old hashes stop being comparable to new ones.
"""
from datetime import UTC, datetime, timedelta
import hashlib
import os
import secrets
import sqlite3
SCHEMA_VERSION = "1"
RECHECK_DAYS = 7
RETRY_HOURS = 24
TRANSIENT_LIMIT = 3
RETENTION_DAYS = 90
# Rate limits. Listing is slower than checking because it writes to the
# directory; checking only costs a fetch.
IP_LISTINGS_PER_HOUR = 5
IP_LISTINGS_PER_DAY = 20
IP_CHECKS_PER_HOUR = 20
DOMAIN_COOLDOWN_MINUTES = 10
DOMAIN_REJECTS_BEFORE_SLOWDOWN = 3
DOMAIN_SLOW_COOLDOWN_HOURS = 24
LISTINGS_PER_HOUR = 60
SCHEMA = """
CREATE TABLE IF NOT EXISTS meta (
key TEXT PRIMARY KEY,
value TEXT NOT NULL
);
CREATE TABLE IF NOT EXISTS sites (
id INTEGER PRIMARY KEY,
domain TEXT NOT NULL UNIQUE,
url TEXT NOT NULL,
title TEXT NOT NULL,
description TEXT,
language TEXT,
state TEXT NOT NULL CHECK (state IN ('listed', 'removed')),
reason TEXT,
listed_at TEXT NOT NULL,
last_check_at TEXT,
last_ok_at TEXT,
next_check_at TEXT,
transient_failures INTEGER NOT NULL DEFAULT 0,
etag TEXT,
last_modified TEXT
);
CREATE INDEX IF NOT EXISTS sites_due
ON sites(next_check_at) WHERE state = 'listed';
CREATE INDEX IF NOT EXISTS sites_state ON sites(state, domain);
CREATE TABLE IF NOT EXISTS submissions (
id INTEGER PRIMARY KEY,
created_at TEXT NOT NULL,
url TEXT NOT NULL,
domain TEXT,
ip_hash TEXT NOT NULL,
outcome TEXT NOT NULL CHECK (outcome IN (
'listed', 'updated', 'rejected', 'unreachable',
'rate_limited', 'blocked', 'checked')),
site_id INTEGER REFERENCES sites(id) ON DELETE SET NULL
);
CREATE INDEX IF NOT EXISTS submissions_ip ON submissions(ip_hash, created_at);
CREATE INDEX IF NOT EXISTS submissions_domain ON submissions(domain, created_at);
CREATE TABLE IF NOT EXISTS checks (
id INTEGER PRIMARY KEY,
site_id INTEGER REFERENCES sites(id) ON DELETE CASCADE,
checked_at TEXT NOT NULL,
kind TEXT NOT NULL CHECK (kind IN ('submission', 'recheck')),
result TEXT NOT NULL CHECK (result IN ('pass', 'fail', 'unreachable')),
http_status INTEGER,
bytes INTEGER,
duration_ms INTEGER
);
CREATE INDEX IF NOT EXISTS checks_site ON checks(site_id, checked_at DESC);
CREATE TABLE IF NOT EXISTS findings (
check_id INTEGER NOT NULL REFERENCES checks(id) ON DELETE CASCADE,
section TEXT NOT NULL,
level TEXT NOT NULL CHECK (level IN ('must', 'should')),
code TEXT NOT NULL,
message TEXT NOT NULL,
location TEXT
);
CREATE INDEX IF NOT EXISTS findings_check ON findings(check_id);
CREATE TABLE IF NOT EXISTS blocked (
domain TEXT PRIMARY KEY,
blocked_at TEXT NOT NULL,
reason TEXT NOT NULL
);
"""
def now() -> datetime:
"""Return the current time, in UTC."""
return datetime.now(UTC)
def path() -> str:
"""Return where the database lives."""
return os.environ.get("MEWS_DB", "mews.db")
def connect(database: str | None = None) -> sqlite3.Connection:
"""Open the database with the settings every caller wants."""
connection = sqlite3.connect(database or path(), timeout=5.0)
connection.row_factory = sqlite3.Row
connection.execute("PRAGMA journal_mode = WAL")
connection.execute("PRAGMA foreign_keys = ON")
connection.execute("PRAGMA busy_timeout = 5000")
return connection
def init(connection: sqlite3.Connection) -> None:
"""Create the schema if it isn't there yet."""
connection.executescript(SCHEMA)
connection.execute(
"INSERT OR IGNORE INTO meta (key, value) VALUES ('schema_version', ?)",
(SCHEMA_VERSION,),
)
connection.commit()
def ip_hash(connection: sqlite3.Connection, address: str) -> str:
"""Hash a client address with a salt that is replaced each month."""
stamp = now().strftime("%Y-%m")
row = connection.execute(
"SELECT value FROM meta WHERE key = 'ip_salt_month'"
).fetchone()
if row is None or row["value"] != stamp:
connection.execute(
"INSERT INTO meta (key, value) VALUES ('ip_salt', ?) "
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
(secrets.token_hex(16),),
)
connection.execute(
"INSERT INTO meta (key, value) VALUES ('ip_salt_month', ?) "
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
(stamp,),
)
connection.commit()
salt = connection.execute(
"SELECT value FROM meta WHERE key = 'ip_salt'"
).fetchone()["value"]
return hashlib.sha256(f"{salt}{address}".encode()).hexdigest()
def _since(hours: float) -> str:
return (now() - timedelta(hours=hours)).isoformat()
def _count(connection: sqlite3.Connection, sql: str, *args: object) -> int:
return connection.execute(sql, args).fetchone()[0]
def blocked_reason(connection: sqlite3.Connection, domain: str) -> str | None:
"""Return why a domain was removed for good, or None if it wasn't."""
row = connection.execute(
"SELECT reason FROM blocked WHERE domain = ?", (domain,)
).fetchone()
return row["reason"] if row else None
def rate_limited(
connection: sqlite3.Connection, *, client: str, domain: str, listing: bool
) -> str | None:
"""Return why this request can't go ahead now, as a sentence for the author.
Checking a page is cheap and gets a looser limit than listing a site.
"""
if not listing:
checks = _count(
connection,
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ?",
client,
_since(1),
)
if checks >= IP_CHECKS_PER_HOUR:
return "You've checked several pages already. Try again in an hour."
return None
if (
_count(
connection,
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? "
"AND outcome IN ('listed', 'updated', 'rejected')",
client,
_since(1),
)
>= IP_LISTINGS_PER_HOUR
):
return "You've submitted several sites already. Try again in an hour."
if (
_count(
connection,
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? "
"AND outcome IN ('listed', 'updated', 'rejected')",
client,
_since(24),
)
>= IP_LISTINGS_PER_DAY
):
return "You've submitted several sites already. Try again tomorrow."
rejects = _count(
connection,
"SELECT COUNT(*) FROM submissions WHERE domain = ? AND outcome = 'rejected' "
"AND created_at > ?",
domain,
_since(24),
)
cooldown = (
DOMAIN_SLOW_COOLDOWN_HOURS
if rejects >= DOMAIN_REJECTS_BEFORE_SLOWDOWN
else DOMAIN_COOLDOWN_MINUTES / 60
)
if _count(
connection,
"SELECT COUNT(*) FROM submissions WHERE domain = ? AND created_at > ?",
domain,
_since(cooldown),
):
if cooldown > 1:
return (
f"{domain} didn't pass the checks a few times today. Fix the "
"page and try again tomorrow."
)
return f"{domain} was submitted a moment ago. Try again in ten minutes."
if (
_count(
connection,
"SELECT COUNT(*) FROM submissions WHERE created_at > ? "
"AND outcome IN ('listed', 'updated')",
_since(1),
)
>= LISTINGS_PER_HOUR
):
return "The directory is busy right now. Try again in an hour."
return None
def record_submission(
connection: sqlite3.Connection,
*,
url: str,
domain: str | None,
client: str,
outcome: str,
site_id: int | None = None,
) -> int:
"""Store one submission and return its id."""
cursor = connection.execute(
"INSERT INTO submissions (created_at, url, domain, ip_hash, outcome, site_id) "
"VALUES (?, ?, ?, ?, ?, ?)",
(now().isoformat(), url, domain, client, outcome, site_id),
)
connection.commit()
return int(cursor.lastrowid or 0)
def record_check(
connection: sqlite3.Connection,
*,
site_id: int | None,
kind: str,
result: str,
findings: list[tuple[str, str, str, str, str]] = [], # noqa: B006
http_status: int | None = None,
size: int | None = None,
duration_ms: int | None = None,
) -> int:
"""Store one check and its findings, returning the check id."""
cursor = connection.execute(
"INSERT INTO checks (site_id, checked_at, kind, result, http_status, bytes, "
"duration_ms) VALUES (?, ?, ?, ?, ?, ?, ?)",
(site_id, now().isoformat(), kind, result, http_status, size, duration_ms),
)
check_id = int(cursor.lastrowid or 0)
connection.executemany(
"INSERT INTO findings (check_id, section, level, code, message, location) "
"VALUES (?, ?, ?, ?, ?, ?)",
[(check_id, *finding) for finding in findings],
)
connection.commit()
return check_id
def next_check(moment: datetime | None = None) -> str:
"""Return when a site that just passed should be looked at again.
The interval is jittered so submissions made together do not come back as a
burst a week later.
"""
base = (moment or now()) + timedelta(days=RECHECK_DAYS)
offset = secrets.randbelow(24 * 60) - 12 * 60
return (base + timedelta(minutes=offset)).isoformat()
def upsert_site(
connection: sqlite3.Connection,
*,
domain: str,
url: str,
title: str,
description: str,
language: str,
etag: str | None,
last_modified: str | None,
) -> tuple[int, bool]:
"""List a site, or refresh the row of one already listed.
Returns the site id and whether the row already existed.
"""
moment = now().isoformat()
row = connection.execute(
"SELECT id FROM sites WHERE domain = ?", (domain,)
).fetchone()
if row is None:
cursor = connection.execute(
"INSERT INTO sites (domain, url, title, description, language, state, "
"listed_at, last_check_at, last_ok_at, next_check_at, etag, last_modified) "
"VALUES (?, ?, ?, ?, ?, 'listed', ?, ?, ?, ?, ?, ?)",
(
domain,
url,
title,
description,
language,
moment,
moment,
moment,
next_check(),
etag,
last_modified,
),
)
connection.commit()
return int(cursor.lastrowid or 0), False
connection.execute(
"UPDATE sites SET url = ?, title = ?, description = ?, language = ?, "
"state = 'listed', reason = NULL, last_check_at = ?, last_ok_at = ?, "
"next_check_at = ?, transient_failures = 0, etag = ?, last_modified = ? "
"WHERE id = ?",
(
url,
title,
description,
language,
moment,
moment,
next_check(),
etag,
last_modified,
row["id"],
),
)
connection.commit()
return int(row["id"]), True
def listed(connection: sqlite3.Connection) -> list[sqlite3.Row]:
"""Return every listed site, in the order the directory page shows them."""
return list(
connection.execute(
"SELECT * FROM sites WHERE state = 'listed' ORDER BY domain"
).fetchall()
)
def due(connection: sqlite3.Connection, limit: int = 50) -> list[sqlite3.Row]:
"""Return the listed sites whose next check has come round."""
return list(
connection.execute(
"SELECT * FROM sites WHERE state = 'listed' AND next_check_at <= ? "
"ORDER BY next_check_at LIMIT ?",
(now().isoformat(), limit),
).fetchall()
)
def site(connection: sqlite3.Connection, domain: str) -> sqlite3.Row | None:
"""Return one site by registrable domain."""
return connection.execute(
"SELECT * FROM sites WHERE domain = ?", (domain,)
).fetchone()
def mark_pass(connection: sqlite3.Connection, site_id: int, **columns: object) -> None:
"""Record a check a site passed."""
moment = now().isoformat()
connection.execute(
"UPDATE sites SET last_check_at = ?, last_ok_at = ?, next_check_at = ?, "
"transient_failures = 0, etag = ?, last_modified = ? WHERE id = ?",
(
moment,
moment,
next_check(),
columns.get("etag"),
columns.get("last_modified"),
site_id,
),
)
connection.commit()
def mark_transient(connection: sqlite3.Connection, site_id: int) -> int:
"""Count one failure that wasn't the page's fault, and return the new count.
A site is only dropped once these pile up: a weekend of downtime should not
empty the directory.
"""
connection.execute(
"UPDATE sites SET transient_failures = transient_failures + 1, "
"last_check_at = ?, next_check_at = ? WHERE id = ?",
(
now().isoformat(),
(now() + timedelta(hours=RETRY_HOURS)).isoformat(),
site_id,
),
)
connection.commit()
return int(
connection.execute(
"SELECT transient_failures FROM sites WHERE id = ?", (site_id,)
).fetchone()[0]
)
def remove(connection: sqlite3.Connection, site_id: int, reason: str) -> None:
"""Drop a site from the directory, keeping the row for the record."""
connection.execute(
"UPDATE sites SET state = 'removed', reason = ?, last_check_at = ?, "
"next_check_at = NULL WHERE id = ?",
(reason, now().isoformat(), site_id),
)
connection.commit()
def block(connection: sqlite3.Connection, domain: str, reason: str) -> None:
"""Refuse future submissions from a domain."""
connection.execute(
"INSERT INTO blocked (domain, blocked_at, reason) VALUES (?, ?, ?) "
"ON CONFLICT(domain) DO UPDATE SET reason = excluded.reason",
(domain, now().isoformat(), reason),
)
connection.commit()
def unblock(connection: sqlite3.Connection, domain: str) -> None:
"""Allow submissions from a domain again."""
connection.execute("DELETE FROM blocked WHERE domain = ?", (domain,))
connection.commit()
def last_findings(connection: sqlite3.Connection, site_id: int) -> list[sqlite3.Row]:
"""Return the findings from a site's most recent check."""
row = connection.execute(
"SELECT id FROM checks WHERE site_id = ? ORDER BY checked_at DESC LIMIT 1",
(site_id,),
).fetchone()
if row is None:
return []
return list(
connection.execute(
"SELECT * FROM findings WHERE check_id = ?", (row["id"],)
).fetchall()
)
def prune(connection: sqlite3.Connection, days: int = RETENTION_DAYS) -> None:
"""Delete submission and check history past the retention window."""
cutoff = (now() - timedelta(days=days)).isoformat()
connection.execute("DELETE FROM submissions WHERE created_at < ?", (cutoff,))
connection.execute("DELETE FROM checks WHERE checked_at < ?", (cutoff,))
connection.commit()

221
mews/pages.py Normal file
View file

@ -0,0 +1,221 @@
"""Emitting the pages mews.page serves.
Every page the service produces is itself a Mews page, so this module writes the
same prologue, head and body shape as md2mews.py. The emitter is repeated here
rather than imported because md2mews.py is a standalone script that pulls in a
Markdown parser; tests/test_pages.py fails if the two drift apart.
"""
from datetime import UTC, datetime
import fcntl
from html import escape
import os
from pathlib import Path
import sqlite3
from mews import VERSION
from mews.lint import validate_bytes
DOCTYPE = (
'<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"\n'
' "http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">'
)
VARIANTS = ("mews-warm", "mews-cool", "mews-green", "mews-mono")
STYLESHEET = "mews-0.1.css"
SITE_VARIANT = "mews-cool"
# SPEC.md 4.3 caps a page at 256 KB and asks for 64 KB. The directory refuses to
# publish a page over the first and warns past the second.
SIZE_MUST = 256 * 1024
SIZE_SHOULD = 64 * 1024
def _text(value: str) -> str:
"""Escape text content, leaving apostrophes alone so prose reads as prose."""
return escape(value, quote=False)
def _attr(value: str) -> str:
"""Escape a value going into an attribute."""
return escape(value, quote=True)
def page(
title: str,
body: list[str],
language: str = "en",
direction: str | None = None,
variant: str | None = None,
description: str | None = None,
) -> str:
"""Wrap body lines in a complete Mews page. Mirrors md2mews.page()."""
attrs = f'xml:lang="{language}" lang="{language}"'
if direction:
attrs += f' dir="{direction}"'
head = [
f" <title>{_text(title)}</title>",
f' <meta name="mews-profile" content="{VERSION}" />',
' <meta name="viewport" content="width=device-width" />',
]
if description:
head.append(f' <meta name="description" content="{_attr(description)}" />')
head.append(f' <link rel="stylesheet" type="text/css" href="{STYLESHEET}" />')
body_attrs = f' class="{variant}"' if variant else ""
return "\n".join(
[
'<?xml version="1.0" encoding="UTF-8"?>',
DOCTYPE,
f'<html xmlns="http://www.w3.org/1999/xhtml" {attrs}>',
" <head>",
*head,
" </head>",
f" <body{body_attrs}>",
*body,
" </body>",
"</html>",
"",
]
)
def site_page(title: str, body: list[str], description: str | None = None) -> str:
"""Wrap body lines in a page with the mews.page look."""
return page(title, body, variant=SITE_VARIANT, description=description)
def message(title: str, paragraphs: list[str], links: list[tuple[str, str]]) -> str:
"""Build a short page that says one thing, with a way onwards."""
body = [f" <h1>{_text(title)}</h1>"]
body += [f" <p>{_text(line)}</p>" for line in paragraphs]
body.append(" <hr />")
body += [
f' <p><a href="{_attr(href)}">{_text(label)}</a></p>'
for href, label in links
]
return site_page(title, body, description=paragraphs[0] if paragraphs else None)
def report(findings_report, *, listed: bool, command: str) -> str:
"""Build the page an author gets back after a check."""
conforms = findings_report.conforms
if listed:
title = "Your site is listed"
opening = (
"That page follows Mews Profile 0.1, so your site is in the "
"directory now. Thank you for adding it."
)
elif conforms:
title = "This page conforms"
opening = "That page follows Mews Profile 0.1. There is nothing to fix."
else:
title = "This page doesn't conform yet"
opening = (
"Here is what to fix. Each line names the section of the spec that "
"explains the rule, so you can go and read why."
)
body = [f" <h1>{_text(title)}</h1>", f" <p>{_text(opening)}</p>"]
if findings_report.url:
body.append(
f' <p>Checked <a href="{_attr(findings_report.url)}">'
f"{_text(findings_report.url)}</a>.</p>"
)
if findings_report.failures:
body.append(" <h2>To fix</h2>")
body.append(" <ul>")
body += [f" <li>{_text(str(f))}</li>" for f in findings_report.failures]
body.append(" </ul>")
if findings_report.warnings:
body.append(" <h2>Worth fixing</h2>")
body.append(" <ul>")
body += [f" <li>{_text(str(f))}</li>" for f in findings_report.warnings]
body.append(" </ul>")
for note in findings_report.notes:
body.append(f" <p>{_text(note)}</p>")
body.append(" <hr />")
body.append(
' <p>The rules are all in the <a href="/spec/0.1">spec</a>. Checking '
"on your own machine is quicker while you are still fixing things:</p>"
)
body.append(f" <pre>{_text(command)}</pre>")
body.append(
' <p><a href="/check" accesskey="1">Check another page</a> '
'· <a href="/directory" accesskey="2">The directory</a></p>'
)
return site_page(title, body)
def directory(rows: list[sqlite3.Row]) -> str:
"""Build the directory page: every site whose page passes the checks."""
body = [" <h1>Mews directory</h1>"]
if rows:
count = f"{len(rows)} site" + ("" if len(rows) == 1 else "s")
body.append(
f" <p>{count} listed. Every one of these pages passed the checks "
'against <a href="/spec/0.1">Mews Profile 0.1</a>, and gets looked '
"at again every week.</p>"
)
body.append(" <ul>")
for row in rows:
label = _text(row["title"] or row["domain"])
line = f' <li><a href="{_attr(row["url"])}">{label}</a>'
line += f" — {_text(row['domain'])}"
if row["description"]:
line += f". {_text(row['description'])}"
body.append(line + "</li>")
body.append(" </ul>")
else:
body.append(" <p>Nothing here yet. Yours could be the first one in!</p>")
body.append(" <hr />")
body.append(
' <p><a href="/check" accesskey="1">Add your site</a> '
'· <a href="/" accesskey="0">Mews</a></p>'
)
body.append(
f" <address>Updated {datetime.now(UTC).strftime('%Y-%m-%d %H:%M')} "
"UTC</address>"
)
return site_page(
"Mews directory",
body,
description="Sites whose pages follow Mews Profile 0.1.",
)
def write_directory(target: str, rows: list[sqlite3.Row]) -> list[str]:
"""Write the directory page, replacing it only if it conforms.
The page is validated before it is published, because the one page that has
to be a good example is this one. Returns any warnings about it.
"""
destination = Path(target)
destination.parent.mkdir(parents=True, exist_ok=True)
text = directory(rows).encode()
if len(text) > SIZE_MUST:
raise ValueError(
"The directory page would be over 256 KB, which section 4.3 "
"forbids. Split it before listing more sites."
)
lock = destination.with_name(destination.name + ".lock")
with open(lock, "w") as handle:
fcntl.flock(handle, fcntl.LOCK_EX)
result = validate_bytes(text)
if not result.conforms:
raise ValueError(
"The directory page doesn't conform: "
+ "; ".join(str(f) for f in result.failures)
)
temporary = destination.with_name("." + destination.name + ".tmp")
temporary.write_bytes(text)
os.replace(temporary, destination)
warnings = [str(f) for f in result.warnings]
if len(text) > SIZE_SHOULD:
warnings.append(
f"The directory page is {len(text) // 1024} KB, over the 64 KB "
"section 4.3 asks for."
)
return warnings