feat: site directory with a public check endpoint
This commit is contained in:
parent
b3c436bf1a
commit
5fa14bdf1b
5 changed files with 1279 additions and 0 deletions
165
mews/app.py
Normal file
165
mews/app.py
Normal file
|
|
@ -0,0 +1,165 @@
|
|||
"""The one dynamic page on mews.page.
|
||||
|
||||
The site is static apart from this: a form posts a page address to /result, the
|
||||
page is fetched and checked, and the author gets the findings back. Nothing here
|
||||
runs JavaScript or sets a cookie, because the response is itself a Mews page.
|
||||
"""
|
||||
|
||||
from collections.abc import Iterable
|
||||
import os
|
||||
import threading
|
||||
from urllib.parse import parse_qs
|
||||
|
||||
from mews import db, pages
|
||||
from mews.check import submit
|
||||
from mews.fetch import Fetcher
|
||||
|
||||
# Outbound checks are slow and this runs on a small machine, so only a couple
|
||||
# happen at once; the rest of the queue is turned away rather than piled up.
|
||||
_CHECKS = threading.BoundedSemaphore(2)
|
||||
_REQUESTS = threading.BoundedSemaphore(8)
|
||||
_SLOT_WAIT = 5.0
|
||||
|
||||
MAX_BODY = 4096
|
||||
|
||||
BUSY = "The checker is busy right now. Try again in a few minutes."
|
||||
|
||||
|
||||
def directory_path() -> str:
|
||||
"""Where the directory page is written."""
|
||||
return os.environ.get("MEWS_DIRECTORY", "directory.html")
|
||||
|
||||
|
||||
def _start(start_response, status: str, body: bytes) -> list[bytes]:
|
||||
start_response(
|
||||
status,
|
||||
[
|
||||
("Content-Type", "text/html; charset=utf-8"),
|
||||
("Content-Length", str(len(body))),
|
||||
("Cache-Control", "no-store"),
|
||||
],
|
||||
)
|
||||
return [body]
|
||||
|
||||
|
||||
def _client(environ: dict) -> str:
|
||||
"""Return the reader's address, as the proxy reports it.
|
||||
|
||||
Caddy sets X-Real-IP from the connection it accepted. The server binds to
|
||||
the loopback address only, so nothing else can set that header.
|
||||
"""
|
||||
return environ.get("HTTP_X_REAL_IP") or environ.get("REMOTE_ADDR", "")
|
||||
|
||||
|
||||
def _form(environ: dict) -> dict[str, list[str]]:
|
||||
try:
|
||||
length = min(int(environ.get("CONTENT_LENGTH") or 0), MAX_BODY)
|
||||
except ValueError:
|
||||
return {}
|
||||
if length <= 0:
|
||||
return {}
|
||||
raw = environ["wsgi.input"].read(length)
|
||||
return parse_qs(raw.decode("utf-8", "replace"), keep_blank_values=True)
|
||||
|
||||
|
||||
def application(environ: dict, start_response) -> Iterable[bytes]:
|
||||
"""Handle one request."""
|
||||
path = environ.get("PATH_INFO", "/")
|
||||
method = environ.get("REQUEST_METHOD", "GET")
|
||||
|
||||
if path != "/result":
|
||||
return _start(
|
||||
start_response,
|
||||
"404 Not Found",
|
||||
pages.message(
|
||||
"Page not found",
|
||||
["That address isn't part of this site."],
|
||||
[("/", "Mews"), ("/directory", "The directory")],
|
||||
).encode(),
|
||||
)
|
||||
|
||||
if method != "POST":
|
||||
return _start(
|
||||
start_response,
|
||||
"405 Method Not Allowed",
|
||||
pages.message(
|
||||
"Nothing to show",
|
||||
["Results appear here after you check a page."],
|
||||
[("/check", "Check a page")],
|
||||
).encode(),
|
||||
)
|
||||
|
||||
form = _form(environ)
|
||||
raw_url = (form.get("url") or [""])[0].strip()
|
||||
listing = "list" in form
|
||||
|
||||
if not _REQUESTS.acquire(blocking=False):
|
||||
return _start(start_response, "503 Service Unavailable", _busy())
|
||||
try:
|
||||
if not _CHECKS.acquire(timeout=_SLOT_WAIT):
|
||||
return _start(start_response, "503 Service Unavailable", _busy())
|
||||
try:
|
||||
body = _check(raw_url, _client(environ), listing)
|
||||
finally:
|
||||
_CHECKS.release()
|
||||
finally:
|
||||
_REQUESTS.release()
|
||||
return _start(start_response, "200 OK", body)
|
||||
|
||||
|
||||
def _busy() -> bytes:
|
||||
return pages.message("Busy", [BUSY], [("/check", "Try again")]).encode()
|
||||
|
||||
|
||||
def _check(raw_url: str, client: str, listing: bool) -> bytes:
|
||||
"""Run one check and render the page the author gets back."""
|
||||
connection = db.connect()
|
||||
try:
|
||||
with Fetcher() as fetcher:
|
||||
outcome = submit(
|
||||
connection,
|
||||
raw_url,
|
||||
client=db.ip_hash(connection, client),
|
||||
listing=listing,
|
||||
fetcher=fetcher,
|
||||
directory=directory_path() if listing else None,
|
||||
)
|
||||
except ValueError as error:
|
||||
# The directory page refused to publish itself. The site is listed; the
|
||||
# page will be rebuilt by the next recheck pass.
|
||||
return pages.message(
|
||||
"Listed, but the directory didn't rebuild",
|
||||
[str(error), "Your site is listed and will appear shortly."],
|
||||
[("/directory", "The directory")],
|
||||
).encode()
|
||||
finally:
|
||||
connection.close()
|
||||
|
||||
if outcome.report is None:
|
||||
return pages.message(
|
||||
"We couldn't check that page",
|
||||
[outcome.message or "Something went wrong. Try again."],
|
||||
[("/check", "Try again"), ("/spec/0.1", "The spec")],
|
||||
).encode()
|
||||
|
||||
command = f"uv run mewslint.py {outcome.report.url or raw_url}"
|
||||
return pages.report(outcome.report, listed=outcome.listed, command=command).encode()
|
||||
|
||||
|
||||
def serve() -> None:
|
||||
"""Run the app for real, on the loopback address only."""
|
||||
# Imported here so the validator and the command line tools do not pull in
|
||||
# a web server.
|
||||
from waitress import serve as waitress_serve # noqa: PLC0415
|
||||
|
||||
listen = os.environ.get("MEWS_LISTEN", "127.0.0.1:8394")
|
||||
host = listen.rsplit(":", 1)[0]
|
||||
if host not in ("127.0.0.1", "::1", "localhost"):
|
||||
raise SystemExit(
|
||||
"mewsd serves on the loopback address only, behind a reverse proxy. "
|
||||
f"Set MEWS_LISTEN to 127.0.0.1 with a port, not {listen}."
|
||||
)
|
||||
waitress_serve(application, listen=listen, threads=4, ident="mews.page")
|
||||
|
||||
|
||||
app = application
|
||||
269
mews/check.py
Normal file
269
mews/check.py
Normal file
|
|
@ -0,0 +1,269 @@
|
|||
"""Checking a page and recording what came of it.
|
||||
|
||||
Both paths into the directory run through here: a submission from the form and
|
||||
a recheck from the timer. Keeping them together means a site is listed and
|
||||
relisted on exactly the same evidence.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
import sqlite3
|
||||
import time
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from mews import db, pages
|
||||
from mews.fetch import (
|
||||
Fetched,
|
||||
Fetcher,
|
||||
FetchError,
|
||||
UrlError,
|
||||
normalise_url,
|
||||
registered_domain,
|
||||
)
|
||||
from mews.lint import Report, validate_bytes
|
||||
|
||||
OK = 200
|
||||
NOT_MODIFIED = 304
|
||||
|
||||
|
||||
@dataclass
|
||||
class Outcome:
|
||||
"""What a submission produced: either a report, or a reason there isn't one."""
|
||||
|
||||
report: Report | None = None
|
||||
listed: bool = False
|
||||
message: str | None = None
|
||||
title: str = ""
|
||||
|
||||
|
||||
def _findings_rows(report: Report) -> list[tuple[str, str, str, str, str]]:
|
||||
return [
|
||||
(f.section, f.level, f.code, f.message, f.location) for f in report.findings
|
||||
]
|
||||
|
||||
|
||||
def check_page(
|
||||
url: str, *, fetcher: Fetcher
|
||||
) -> tuple[Report | None, Fetched | None, str | None]:
|
||||
"""Fetch and validate one page.
|
||||
|
||||
Returns the report, the response it came from, and a plain message when
|
||||
there was no page to check.
|
||||
"""
|
||||
try:
|
||||
response = fetcher.get(url)
|
||||
except (UrlError, FetchError) as error:
|
||||
return None, None, str(error)
|
||||
if response.status != OK:
|
||||
return (
|
||||
None,
|
||||
response,
|
||||
f"We couldn't read that page (it returned {response.status}). "
|
||||
"Check the address and try again.",
|
||||
)
|
||||
report = validate_bytes(
|
||||
response.body, url=response.url, response=response, fetcher=fetcher
|
||||
)
|
||||
return report, response, None
|
||||
|
||||
|
||||
def submit(
|
||||
connection: sqlite3.Connection,
|
||||
raw_url: str,
|
||||
*,
|
||||
client: str,
|
||||
listing: bool,
|
||||
fetcher: Fetcher,
|
||||
directory: str | None = None,
|
||||
) -> Outcome:
|
||||
"""Check a page, and list its site when the author asked and it conforms."""
|
||||
try:
|
||||
url = normalise_url(raw_url, allow_loopback=fetcher.allow_loopback)
|
||||
except UrlError as error:
|
||||
return Outcome(message=str(error))
|
||||
|
||||
host = urlsplit(url).hostname or ""
|
||||
domain = registered_domain(host)
|
||||
if domain is None:
|
||||
return Outcome(message="That domain name isn't one the checker can reach.")
|
||||
|
||||
reason = db.blocked_reason(connection, domain)
|
||||
if reason is not None:
|
||||
db.record_submission(
|
||||
connection, url=url, domain=domain, client=client, outcome="blocked"
|
||||
)
|
||||
return Outcome(
|
||||
message="This site was taken out of the directory. Write to the "
|
||||
"address on the about page if that looks wrong."
|
||||
)
|
||||
|
||||
limited = db.rate_limited(connection, client=client, domain=domain, listing=listing)
|
||||
if limited is not None:
|
||||
db.record_submission(
|
||||
connection, url=url, domain=domain, client=client, outcome="rate_limited"
|
||||
)
|
||||
return Outcome(message=limited)
|
||||
|
||||
started = time.monotonic()
|
||||
report, response, failure = check_page(url, fetcher=fetcher)
|
||||
elapsed = int((time.monotonic() - started) * 1000)
|
||||
status = response.status if response else None
|
||||
if report is None:
|
||||
db.record_check(
|
||||
connection,
|
||||
site_id=None,
|
||||
kind="submission",
|
||||
result="unreachable",
|
||||
http_status=status,
|
||||
duration_ms=elapsed,
|
||||
)
|
||||
db.record_submission(
|
||||
connection, url=url, domain=domain, client=client, outcome="unreachable"
|
||||
)
|
||||
return Outcome(message=failure)
|
||||
|
||||
if not listing:
|
||||
db.record_check(
|
||||
connection,
|
||||
site_id=None,
|
||||
kind="submission",
|
||||
result="pass" if report.conforms else "fail",
|
||||
findings=_findings_rows(report),
|
||||
http_status=status,
|
||||
size=report.size,
|
||||
duration_ms=elapsed,
|
||||
)
|
||||
db.record_submission(
|
||||
connection, url=url, domain=domain, client=client, outcome="checked"
|
||||
)
|
||||
return Outcome(report=report)
|
||||
|
||||
if not report.conforms:
|
||||
db.record_check(
|
||||
connection,
|
||||
site_id=None,
|
||||
kind="submission",
|
||||
result="fail",
|
||||
findings=_findings_rows(report),
|
||||
http_status=status,
|
||||
size=report.size,
|
||||
duration_ms=elapsed,
|
||||
)
|
||||
db.record_submission(
|
||||
connection, url=url, domain=domain, client=client, outcome="rejected"
|
||||
)
|
||||
return Outcome(report=report)
|
||||
|
||||
site_id, existed = db.upsert_site(
|
||||
connection,
|
||||
domain=domain,
|
||||
url=report.url or url,
|
||||
title=report.title or domain,
|
||||
description=report.description,
|
||||
language=report.language,
|
||||
etag=response.headers.get("etag") if response else None,
|
||||
last_modified=response.headers.get("last-modified") if response else None,
|
||||
)
|
||||
db.record_check(
|
||||
connection,
|
||||
site_id=site_id,
|
||||
kind="submission",
|
||||
result="pass",
|
||||
findings=_findings_rows(report),
|
||||
http_status=status,
|
||||
size=report.size,
|
||||
duration_ms=elapsed,
|
||||
)
|
||||
db.record_submission(
|
||||
connection,
|
||||
url=url,
|
||||
domain=domain,
|
||||
client=client,
|
||||
outcome="updated" if existed else "listed",
|
||||
site_id=site_id,
|
||||
)
|
||||
if directory:
|
||||
pages.write_directory(directory, db.listed(connection))
|
||||
return Outcome(report=report, listed=True, title=report.title)
|
||||
|
||||
|
||||
def recheck(
|
||||
connection: sqlite3.Connection, row: sqlite3.Row, *, fetcher: Fetcher
|
||||
) -> str:
|
||||
"""Check one listed site again, and drop it if its page stopped conforming.
|
||||
|
||||
A page that no longer conforms goes immediately: there is no contact address
|
||||
to warn, and resubmitting after a fix lists the site again at once. A site
|
||||
we simply could not reach is retried, because downtime is not a conformance
|
||||
failure.
|
||||
"""
|
||||
conditional: dict[str, str] = {}
|
||||
if row["etag"]:
|
||||
conditional["If-None-Match"] = row["etag"]
|
||||
if row["last_modified"]:
|
||||
conditional["If-Modified-Since"] = row["last_modified"]
|
||||
|
||||
started = time.monotonic()
|
||||
try:
|
||||
response = fetcher.get(row["url"], headers=conditional or None)
|
||||
except (UrlError, FetchError):
|
||||
return _transient(connection, row, "We couldn't reach your site.")
|
||||
|
||||
elapsed = int((time.monotonic() - started) * 1000)
|
||||
if response.status == NOT_MODIFIED:
|
||||
db.mark_pass(
|
||||
connection,
|
||||
row["id"],
|
||||
etag=row["etag"],
|
||||
last_modified=row["last_modified"],
|
||||
)
|
||||
db.record_check(
|
||||
connection,
|
||||
site_id=row["id"],
|
||||
kind="recheck",
|
||||
result="pass",
|
||||
http_status=304,
|
||||
duration_ms=elapsed,
|
||||
)
|
||||
return "pass"
|
||||
|
||||
if response.status != OK:
|
||||
return _transient(connection, row, f"Your page returned {response.status}.")
|
||||
|
||||
report = validate_bytes(
|
||||
response.body, url=response.url, response=response, fetcher=fetcher
|
||||
)
|
||||
db.record_check(
|
||||
connection,
|
||||
site_id=row["id"],
|
||||
kind="recheck",
|
||||
result="pass" if report.conforms else "fail",
|
||||
findings=_findings_rows(report),
|
||||
http_status=response.status,
|
||||
size=report.size,
|
||||
duration_ms=elapsed,
|
||||
)
|
||||
if report.conforms:
|
||||
db.mark_pass(
|
||||
connection,
|
||||
row["id"],
|
||||
etag=response.headers.get("etag"),
|
||||
last_modified=response.headers.get("last-modified"),
|
||||
)
|
||||
return "pass"
|
||||
|
||||
db.remove(connection, row["id"], str(report.failures[0]))
|
||||
return "fail"
|
||||
|
||||
|
||||
def _transient(connection: sqlite3.Connection, row: sqlite3.Row, note: str) -> str:
|
||||
"""Count a failure that wasn't the page's fault, dropping the site if it sticks."""
|
||||
count = db.mark_transient(connection, row["id"])
|
||||
db.record_check(connection, site_id=row["id"], kind="recheck", result="unreachable")
|
||||
if count >= db.TRANSIENT_LIMIT:
|
||||
db.remove(
|
||||
connection,
|
||||
row["id"],
|
||||
f"{note} We tried for {db.TRANSIENT_LIMIT} days.",
|
||||
)
|
||||
return "fail"
|
||||
return "unreachable"
|
||||
138
mews/cli.py
Normal file
138
mews/cli.py
Normal file
|
|
@ -0,0 +1,138 @@
|
|||
"""mewsd: run the service, and look after the directory from the shell."""
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
import time
|
||||
|
||||
from mews import check as checker, db, pages
|
||||
from mews.app import directory_path, serve
|
||||
from mews.fetch import Fetcher, registered_domain
|
||||
|
||||
RECHECK_BATCH = 50
|
||||
PAUSE_SECONDS = 2.0
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
"""Run one command. Returns an exit status."""
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="mewsd", description="Run and look after the mews.page directory."
|
||||
)
|
||||
commands = parser.add_subparsers(dest="command", required=True)
|
||||
commands.add_parser("serve", help="serve the check endpoint")
|
||||
commands.add_parser("init-db", help="create the database if it isn't there")
|
||||
commands.add_parser("recheck", help="check the sites that are due")
|
||||
commands.add_parser("build-directory", help="write the directory page again")
|
||||
drop = commands.add_parser("remove", help="take a site out and refuse it later")
|
||||
drop.add_argument("domain")
|
||||
drop.add_argument("--reason", required=True, help="shown in the log and to you")
|
||||
allow = commands.add_parser("unblock", help="allow a removed domain again")
|
||||
allow.add_argument("domain")
|
||||
show = commands.add_parser("show", help="what the last check of a site found")
|
||||
show.add_argument("domain")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
if args.command == "serve":
|
||||
serve()
|
||||
return 0
|
||||
|
||||
connection = db.connect()
|
||||
try:
|
||||
db.init(connection)
|
||||
if args.command == "init-db":
|
||||
print(f"database ready at {db.path()}")
|
||||
return 0
|
||||
if args.command == "recheck":
|
||||
return _recheck(connection)
|
||||
if args.command == "build-directory":
|
||||
return _build(connection)
|
||||
if args.command == "remove":
|
||||
return _remove(connection, args.domain, args.reason)
|
||||
if args.command == "unblock":
|
||||
return _unblock(connection, args.domain)
|
||||
if args.command == "show":
|
||||
return _show(connection, args.domain)
|
||||
finally:
|
||||
connection.close()
|
||||
return 0
|
||||
|
||||
|
||||
def _domain(name: str) -> str:
|
||||
return registered_domain(name) or name.strip().lower()
|
||||
|
||||
|
||||
def _recheck(connection) -> int:
|
||||
"""Check every site that is due, then write the directory page again."""
|
||||
sites = db.due(connection, RECHECK_BATCH)
|
||||
counts = {"pass": 0, "fail": 0, "unreachable": 0}
|
||||
for index, row in enumerate(sites):
|
||||
if index:
|
||||
# One site at a time, with a gap, so a recheck pass is invisible to
|
||||
# the sites it visits.
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
with Fetcher() as fetcher:
|
||||
counts[checker.recheck(connection, row, fetcher=fetcher)] += 1
|
||||
db.prune(connection)
|
||||
_build(connection)
|
||||
print(
|
||||
f"checked {len(sites)}: {counts['pass']} passed, {counts['fail']} dropped, "
|
||||
f"{counts['unreachable']} unreachable"
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
def _build(connection) -> int:
|
||||
"""Write the directory page from what is listed now."""
|
||||
try:
|
||||
warnings = pages.write_directory(directory_path(), db.listed(connection))
|
||||
except ValueError as error:
|
||||
print(f"the directory page wasn't written: {error}", file=sys.stderr)
|
||||
return 1
|
||||
for warning in warnings:
|
||||
print(f"note: {warning}", file=sys.stderr)
|
||||
print(f"wrote {directory_path()}")
|
||||
return 0
|
||||
|
||||
|
||||
def _remove(connection, domain: str, reason: str) -> int:
|
||||
name = _domain(domain)
|
||||
row = db.site(connection, name)
|
||||
if row is not None:
|
||||
db.remove(connection, row["id"], reason)
|
||||
db.block(connection, name, reason)
|
||||
_build(connection)
|
||||
print(f"removed {name}")
|
||||
return 0
|
||||
|
||||
|
||||
def _unblock(connection, domain: str) -> int:
|
||||
name = _domain(domain)
|
||||
db.unblock(connection, name)
|
||||
print(f"{name} can be submitted again")
|
||||
return 0
|
||||
|
||||
|
||||
def _show(connection, domain: str) -> int:
|
||||
name = _domain(domain)
|
||||
row = db.site(connection, name)
|
||||
if row is None:
|
||||
print(f"{name} isn't in the directory", file=sys.stderr)
|
||||
return 1
|
||||
print(f"{row['domain']} — {row['state']}")
|
||||
print(f" page {row['url']}")
|
||||
print(f" title {row['title']}")
|
||||
print(f" listed {row['listed_at']}")
|
||||
print(f" checked {row['last_check_at']} (last passed {row['last_ok_at']})")
|
||||
print(f" next {row['next_check_at']}")
|
||||
if row["reason"]:
|
||||
print(f" dropped {row['reason']}")
|
||||
findings = db.last_findings(connection, row["id"])
|
||||
for finding in findings:
|
||||
print(
|
||||
f" {finding['level']:6} section {finding['section']} — "
|
||||
f"{finding['message']}"
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
486
mews/db.py
Normal file
486
mews/db.py
Normal file
|
|
@ -0,0 +1,486 @@
|
|||
"""The directory's SQLite store.
|
||||
|
||||
One row per site, keyed by registrable domain, plus the submission and check
|
||||
history the rate limits and the recheck schedule read. Client addresses are
|
||||
never stored: a submission keeps a salted hash, and the salt is replaced every
|
||||
month so old hashes stop being comparable to new ones.
|
||||
"""
|
||||
|
||||
from datetime import UTC, datetime, timedelta
|
||||
import hashlib
|
||||
import os
|
||||
import secrets
|
||||
import sqlite3
|
||||
|
||||
SCHEMA_VERSION = "1"
|
||||
|
||||
RECHECK_DAYS = 7
|
||||
RETRY_HOURS = 24
|
||||
TRANSIENT_LIMIT = 3
|
||||
RETENTION_DAYS = 90
|
||||
|
||||
# Rate limits. Listing is slower than checking because it writes to the
|
||||
# directory; checking only costs a fetch.
|
||||
IP_LISTINGS_PER_HOUR = 5
|
||||
IP_LISTINGS_PER_DAY = 20
|
||||
IP_CHECKS_PER_HOUR = 20
|
||||
DOMAIN_COOLDOWN_MINUTES = 10
|
||||
DOMAIN_REJECTS_BEFORE_SLOWDOWN = 3
|
||||
DOMAIN_SLOW_COOLDOWN_HOURS = 24
|
||||
LISTINGS_PER_HOUR = 60
|
||||
|
||||
SCHEMA = """
|
||||
CREATE TABLE IF NOT EXISTS meta (
|
||||
key TEXT PRIMARY KEY,
|
||||
value TEXT NOT NULL
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS sites (
|
||||
id INTEGER PRIMARY KEY,
|
||||
domain TEXT NOT NULL UNIQUE,
|
||||
url TEXT NOT NULL,
|
||||
title TEXT NOT NULL,
|
||||
description TEXT,
|
||||
language TEXT,
|
||||
state TEXT NOT NULL CHECK (state IN ('listed', 'removed')),
|
||||
reason TEXT,
|
||||
listed_at TEXT NOT NULL,
|
||||
last_check_at TEXT,
|
||||
last_ok_at TEXT,
|
||||
next_check_at TEXT,
|
||||
transient_failures INTEGER NOT NULL DEFAULT 0,
|
||||
etag TEXT,
|
||||
last_modified TEXT
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS sites_due
|
||||
ON sites(next_check_at) WHERE state = 'listed';
|
||||
CREATE INDEX IF NOT EXISTS sites_state ON sites(state, domain);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS submissions (
|
||||
id INTEGER PRIMARY KEY,
|
||||
created_at TEXT NOT NULL,
|
||||
url TEXT NOT NULL,
|
||||
domain TEXT,
|
||||
ip_hash TEXT NOT NULL,
|
||||
outcome TEXT NOT NULL CHECK (outcome IN (
|
||||
'listed', 'updated', 'rejected', 'unreachable',
|
||||
'rate_limited', 'blocked', 'checked')),
|
||||
site_id INTEGER REFERENCES sites(id) ON DELETE SET NULL
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS submissions_ip ON submissions(ip_hash, created_at);
|
||||
CREATE INDEX IF NOT EXISTS submissions_domain ON submissions(domain, created_at);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS checks (
|
||||
id INTEGER PRIMARY KEY,
|
||||
site_id INTEGER REFERENCES sites(id) ON DELETE CASCADE,
|
||||
checked_at TEXT NOT NULL,
|
||||
kind TEXT NOT NULL CHECK (kind IN ('submission', 'recheck')),
|
||||
result TEXT NOT NULL CHECK (result IN ('pass', 'fail', 'unreachable')),
|
||||
http_status INTEGER,
|
||||
bytes INTEGER,
|
||||
duration_ms INTEGER
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS checks_site ON checks(site_id, checked_at DESC);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS findings (
|
||||
check_id INTEGER NOT NULL REFERENCES checks(id) ON DELETE CASCADE,
|
||||
section TEXT NOT NULL,
|
||||
level TEXT NOT NULL CHECK (level IN ('must', 'should')),
|
||||
code TEXT NOT NULL,
|
||||
message TEXT NOT NULL,
|
||||
location TEXT
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS findings_check ON findings(check_id);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS blocked (
|
||||
domain TEXT PRIMARY KEY,
|
||||
blocked_at TEXT NOT NULL,
|
||||
reason TEXT NOT NULL
|
||||
);
|
||||
"""
|
||||
|
||||
|
||||
def now() -> datetime:
|
||||
"""Return the current time, in UTC."""
|
||||
return datetime.now(UTC)
|
||||
|
||||
|
||||
def path() -> str:
|
||||
"""Return where the database lives."""
|
||||
return os.environ.get("MEWS_DB", "mews.db")
|
||||
|
||||
|
||||
def connect(database: str | None = None) -> sqlite3.Connection:
|
||||
"""Open the database with the settings every caller wants."""
|
||||
connection = sqlite3.connect(database or path(), timeout=5.0)
|
||||
connection.row_factory = sqlite3.Row
|
||||
connection.execute("PRAGMA journal_mode = WAL")
|
||||
connection.execute("PRAGMA foreign_keys = ON")
|
||||
connection.execute("PRAGMA busy_timeout = 5000")
|
||||
return connection
|
||||
|
||||
|
||||
def init(connection: sqlite3.Connection) -> None:
|
||||
"""Create the schema if it isn't there yet."""
|
||||
connection.executescript(SCHEMA)
|
||||
connection.execute(
|
||||
"INSERT OR IGNORE INTO meta (key, value) VALUES ('schema_version', ?)",
|
||||
(SCHEMA_VERSION,),
|
||||
)
|
||||
connection.commit()
|
||||
|
||||
|
||||
def ip_hash(connection: sqlite3.Connection, address: str) -> str:
|
||||
"""Hash a client address with a salt that is replaced each month."""
|
||||
stamp = now().strftime("%Y-%m")
|
||||
row = connection.execute(
|
||||
"SELECT value FROM meta WHERE key = 'ip_salt_month'"
|
||||
).fetchone()
|
||||
if row is None or row["value"] != stamp:
|
||||
connection.execute(
|
||||
"INSERT INTO meta (key, value) VALUES ('ip_salt', ?) "
|
||||
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
||||
(secrets.token_hex(16),),
|
||||
)
|
||||
connection.execute(
|
||||
"INSERT INTO meta (key, value) VALUES ('ip_salt_month', ?) "
|
||||
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
||||
(stamp,),
|
||||
)
|
||||
connection.commit()
|
||||
salt = connection.execute(
|
||||
"SELECT value FROM meta WHERE key = 'ip_salt'"
|
||||
).fetchone()["value"]
|
||||
return hashlib.sha256(f"{salt}{address}".encode()).hexdigest()
|
||||
|
||||
|
||||
def _since(hours: float) -> str:
|
||||
return (now() - timedelta(hours=hours)).isoformat()
|
||||
|
||||
|
||||
def _count(connection: sqlite3.Connection, sql: str, *args: object) -> int:
|
||||
return connection.execute(sql, args).fetchone()[0]
|
||||
|
||||
|
||||
def blocked_reason(connection: sqlite3.Connection, domain: str) -> str | None:
|
||||
"""Return why a domain was removed for good, or None if it wasn't."""
|
||||
row = connection.execute(
|
||||
"SELECT reason FROM blocked WHERE domain = ?", (domain,)
|
||||
).fetchone()
|
||||
return row["reason"] if row else None
|
||||
|
||||
|
||||
def rate_limited(
|
||||
connection: sqlite3.Connection, *, client: str, domain: str, listing: bool
|
||||
) -> str | None:
|
||||
"""Return why this request can't go ahead now, as a sentence for the author.
|
||||
|
||||
Checking a page is cheap and gets a looser limit than listing a site.
|
||||
"""
|
||||
if not listing:
|
||||
checks = _count(
|
||||
connection,
|
||||
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ?",
|
||||
client,
|
||||
_since(1),
|
||||
)
|
||||
if checks >= IP_CHECKS_PER_HOUR:
|
||||
return "You've checked several pages already. Try again in an hour."
|
||||
return None
|
||||
|
||||
if (
|
||||
_count(
|
||||
connection,
|
||||
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? "
|
||||
"AND outcome IN ('listed', 'updated', 'rejected')",
|
||||
client,
|
||||
_since(1),
|
||||
)
|
||||
>= IP_LISTINGS_PER_HOUR
|
||||
):
|
||||
return "You've submitted several sites already. Try again in an hour."
|
||||
if (
|
||||
_count(
|
||||
connection,
|
||||
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? "
|
||||
"AND outcome IN ('listed', 'updated', 'rejected')",
|
||||
client,
|
||||
_since(24),
|
||||
)
|
||||
>= IP_LISTINGS_PER_DAY
|
||||
):
|
||||
return "You've submitted several sites already. Try again tomorrow."
|
||||
|
||||
rejects = _count(
|
||||
connection,
|
||||
"SELECT COUNT(*) FROM submissions WHERE domain = ? AND outcome = 'rejected' "
|
||||
"AND created_at > ?",
|
||||
domain,
|
||||
_since(24),
|
||||
)
|
||||
cooldown = (
|
||||
DOMAIN_SLOW_COOLDOWN_HOURS
|
||||
if rejects >= DOMAIN_REJECTS_BEFORE_SLOWDOWN
|
||||
else DOMAIN_COOLDOWN_MINUTES / 60
|
||||
)
|
||||
if _count(
|
||||
connection,
|
||||
"SELECT COUNT(*) FROM submissions WHERE domain = ? AND created_at > ?",
|
||||
domain,
|
||||
_since(cooldown),
|
||||
):
|
||||
if cooldown > 1:
|
||||
return (
|
||||
f"{domain} didn't pass the checks a few times today. Fix the "
|
||||
"page and try again tomorrow."
|
||||
)
|
||||
return f"{domain} was submitted a moment ago. Try again in ten minutes."
|
||||
|
||||
if (
|
||||
_count(
|
||||
connection,
|
||||
"SELECT COUNT(*) FROM submissions WHERE created_at > ? "
|
||||
"AND outcome IN ('listed', 'updated')",
|
||||
_since(1),
|
||||
)
|
||||
>= LISTINGS_PER_HOUR
|
||||
):
|
||||
return "The directory is busy right now. Try again in an hour."
|
||||
return None
|
||||
|
||||
|
||||
def record_submission(
|
||||
connection: sqlite3.Connection,
|
||||
*,
|
||||
url: str,
|
||||
domain: str | None,
|
||||
client: str,
|
||||
outcome: str,
|
||||
site_id: int | None = None,
|
||||
) -> int:
|
||||
"""Store one submission and return its id."""
|
||||
cursor = connection.execute(
|
||||
"INSERT INTO submissions (created_at, url, domain, ip_hash, outcome, site_id) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?)",
|
||||
(now().isoformat(), url, domain, client, outcome, site_id),
|
||||
)
|
||||
connection.commit()
|
||||
return int(cursor.lastrowid or 0)
|
||||
|
||||
|
||||
def record_check(
|
||||
connection: sqlite3.Connection,
|
||||
*,
|
||||
site_id: int | None,
|
||||
kind: str,
|
||||
result: str,
|
||||
findings: list[tuple[str, str, str, str, str]] = [], # noqa: B006
|
||||
http_status: int | None = None,
|
||||
size: int | None = None,
|
||||
duration_ms: int | None = None,
|
||||
) -> int:
|
||||
"""Store one check and its findings, returning the check id."""
|
||||
cursor = connection.execute(
|
||||
"INSERT INTO checks (site_id, checked_at, kind, result, http_status, bytes, "
|
||||
"duration_ms) VALUES (?, ?, ?, ?, ?, ?, ?)",
|
||||
(site_id, now().isoformat(), kind, result, http_status, size, duration_ms),
|
||||
)
|
||||
check_id = int(cursor.lastrowid or 0)
|
||||
connection.executemany(
|
||||
"INSERT INTO findings (check_id, section, level, code, message, location) "
|
||||
"VALUES (?, ?, ?, ?, ?, ?)",
|
||||
[(check_id, *finding) for finding in findings],
|
||||
)
|
||||
connection.commit()
|
||||
return check_id
|
||||
|
||||
|
||||
def next_check(moment: datetime | None = None) -> str:
|
||||
"""Return when a site that just passed should be looked at again.
|
||||
|
||||
The interval is jittered so submissions made together do not come back as a
|
||||
burst a week later.
|
||||
"""
|
||||
base = (moment or now()) + timedelta(days=RECHECK_DAYS)
|
||||
offset = secrets.randbelow(24 * 60) - 12 * 60
|
||||
return (base + timedelta(minutes=offset)).isoformat()
|
||||
|
||||
|
||||
def upsert_site(
|
||||
connection: sqlite3.Connection,
|
||||
*,
|
||||
domain: str,
|
||||
url: str,
|
||||
title: str,
|
||||
description: str,
|
||||
language: str,
|
||||
etag: str | None,
|
||||
last_modified: str | None,
|
||||
) -> tuple[int, bool]:
|
||||
"""List a site, or refresh the row of one already listed.
|
||||
|
||||
Returns the site id and whether the row already existed.
|
||||
"""
|
||||
moment = now().isoformat()
|
||||
row = connection.execute(
|
||||
"SELECT id FROM sites WHERE domain = ?", (domain,)
|
||||
).fetchone()
|
||||
if row is None:
|
||||
cursor = connection.execute(
|
||||
"INSERT INTO sites (domain, url, title, description, language, state, "
|
||||
"listed_at, last_check_at, last_ok_at, next_check_at, etag, last_modified) "
|
||||
"VALUES (?, ?, ?, ?, ?, 'listed', ?, ?, ?, ?, ?, ?)",
|
||||
(
|
||||
domain,
|
||||
url,
|
||||
title,
|
||||
description,
|
||||
language,
|
||||
moment,
|
||||
moment,
|
||||
moment,
|
||||
next_check(),
|
||||
etag,
|
||||
last_modified,
|
||||
),
|
||||
)
|
||||
connection.commit()
|
||||
return int(cursor.lastrowid or 0), False
|
||||
|
||||
connection.execute(
|
||||
"UPDATE sites SET url = ?, title = ?, description = ?, language = ?, "
|
||||
"state = 'listed', reason = NULL, last_check_at = ?, last_ok_at = ?, "
|
||||
"next_check_at = ?, transient_failures = 0, etag = ?, last_modified = ? "
|
||||
"WHERE id = ?",
|
||||
(
|
||||
url,
|
||||
title,
|
||||
description,
|
||||
language,
|
||||
moment,
|
||||
moment,
|
||||
next_check(),
|
||||
etag,
|
||||
last_modified,
|
||||
row["id"],
|
||||
),
|
||||
)
|
||||
connection.commit()
|
||||
return int(row["id"]), True
|
||||
|
||||
|
||||
def listed(connection: sqlite3.Connection) -> list[sqlite3.Row]:
|
||||
"""Return every listed site, in the order the directory page shows them."""
|
||||
return list(
|
||||
connection.execute(
|
||||
"SELECT * FROM sites WHERE state = 'listed' ORDER BY domain"
|
||||
).fetchall()
|
||||
)
|
||||
|
||||
|
||||
def due(connection: sqlite3.Connection, limit: int = 50) -> list[sqlite3.Row]:
|
||||
"""Return the listed sites whose next check has come round."""
|
||||
return list(
|
||||
connection.execute(
|
||||
"SELECT * FROM sites WHERE state = 'listed' AND next_check_at <= ? "
|
||||
"ORDER BY next_check_at LIMIT ?",
|
||||
(now().isoformat(), limit),
|
||||
).fetchall()
|
||||
)
|
||||
|
||||
|
||||
def site(connection: sqlite3.Connection, domain: str) -> sqlite3.Row | None:
|
||||
"""Return one site by registrable domain."""
|
||||
return connection.execute(
|
||||
"SELECT * FROM sites WHERE domain = ?", (domain,)
|
||||
).fetchone()
|
||||
|
||||
|
||||
def mark_pass(connection: sqlite3.Connection, site_id: int, **columns: object) -> None:
|
||||
"""Record a check a site passed."""
|
||||
moment = now().isoformat()
|
||||
connection.execute(
|
||||
"UPDATE sites SET last_check_at = ?, last_ok_at = ?, next_check_at = ?, "
|
||||
"transient_failures = 0, etag = ?, last_modified = ? WHERE id = ?",
|
||||
(
|
||||
moment,
|
||||
moment,
|
||||
next_check(),
|
||||
columns.get("etag"),
|
||||
columns.get("last_modified"),
|
||||
site_id,
|
||||
),
|
||||
)
|
||||
connection.commit()
|
||||
|
||||
|
||||
def mark_transient(connection: sqlite3.Connection, site_id: int) -> int:
|
||||
"""Count one failure that wasn't the page's fault, and return the new count.
|
||||
|
||||
A site is only dropped once these pile up: a weekend of downtime should not
|
||||
empty the directory.
|
||||
"""
|
||||
connection.execute(
|
||||
"UPDATE sites SET transient_failures = transient_failures + 1, "
|
||||
"last_check_at = ?, next_check_at = ? WHERE id = ?",
|
||||
(
|
||||
now().isoformat(),
|
||||
(now() + timedelta(hours=RETRY_HOURS)).isoformat(),
|
||||
site_id,
|
||||
),
|
||||
)
|
||||
connection.commit()
|
||||
return int(
|
||||
connection.execute(
|
||||
"SELECT transient_failures FROM sites WHERE id = ?", (site_id,)
|
||||
).fetchone()[0]
|
||||
)
|
||||
|
||||
|
||||
def remove(connection: sqlite3.Connection, site_id: int, reason: str) -> None:
|
||||
"""Drop a site from the directory, keeping the row for the record."""
|
||||
connection.execute(
|
||||
"UPDATE sites SET state = 'removed', reason = ?, last_check_at = ?, "
|
||||
"next_check_at = NULL WHERE id = ?",
|
||||
(reason, now().isoformat(), site_id),
|
||||
)
|
||||
connection.commit()
|
||||
|
||||
|
||||
def block(connection: sqlite3.Connection, domain: str, reason: str) -> None:
|
||||
"""Refuse future submissions from a domain."""
|
||||
connection.execute(
|
||||
"INSERT INTO blocked (domain, blocked_at, reason) VALUES (?, ?, ?) "
|
||||
"ON CONFLICT(domain) DO UPDATE SET reason = excluded.reason",
|
||||
(domain, now().isoformat(), reason),
|
||||
)
|
||||
connection.commit()
|
||||
|
||||
|
||||
def unblock(connection: sqlite3.Connection, domain: str) -> None:
|
||||
"""Allow submissions from a domain again."""
|
||||
connection.execute("DELETE FROM blocked WHERE domain = ?", (domain,))
|
||||
connection.commit()
|
||||
|
||||
|
||||
def last_findings(connection: sqlite3.Connection, site_id: int) -> list[sqlite3.Row]:
|
||||
"""Return the findings from a site's most recent check."""
|
||||
row = connection.execute(
|
||||
"SELECT id FROM checks WHERE site_id = ? ORDER BY checked_at DESC LIMIT 1",
|
||||
(site_id,),
|
||||
).fetchone()
|
||||
if row is None:
|
||||
return []
|
||||
return list(
|
||||
connection.execute(
|
||||
"SELECT * FROM findings WHERE check_id = ?", (row["id"],)
|
||||
).fetchall()
|
||||
)
|
||||
|
||||
|
||||
def prune(connection: sqlite3.Connection, days: int = RETENTION_DAYS) -> None:
|
||||
"""Delete submission and check history past the retention window."""
|
||||
cutoff = (now() - timedelta(days=days)).isoformat()
|
||||
connection.execute("DELETE FROM submissions WHERE created_at < ?", (cutoff,))
|
||||
connection.execute("DELETE FROM checks WHERE checked_at < ?", (cutoff,))
|
||||
connection.commit()
|
||||
221
mews/pages.py
Normal file
221
mews/pages.py
Normal file
|
|
@ -0,0 +1,221 @@
|
|||
"""Emitting the pages mews.page serves.
|
||||
|
||||
Every page the service produces is itself a Mews page, so this module writes the
|
||||
same prologue, head and body shape as md2mews.py. The emitter is repeated here
|
||||
rather than imported because md2mews.py is a standalone script that pulls in a
|
||||
Markdown parser; tests/test_pages.py fails if the two drift apart.
|
||||
"""
|
||||
|
||||
from datetime import UTC, datetime
|
||||
import fcntl
|
||||
from html import escape
|
||||
import os
|
||||
from pathlib import Path
|
||||
import sqlite3
|
||||
|
||||
from mews import VERSION
|
||||
from mews.lint import validate_bytes
|
||||
|
||||
DOCTYPE = (
|
||||
'<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"\n'
|
||||
' "http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">'
|
||||
)
|
||||
VARIANTS = ("mews-warm", "mews-cool", "mews-green", "mews-mono")
|
||||
STYLESHEET = "mews-0.1.css"
|
||||
SITE_VARIANT = "mews-cool"
|
||||
|
||||
# SPEC.md 4.3 caps a page at 256 KB and asks for 64 KB. The directory refuses to
|
||||
# publish a page over the first and warns past the second.
|
||||
SIZE_MUST = 256 * 1024
|
||||
SIZE_SHOULD = 64 * 1024
|
||||
|
||||
|
||||
def _text(value: str) -> str:
|
||||
"""Escape text content, leaving apostrophes alone so prose reads as prose."""
|
||||
return escape(value, quote=False)
|
||||
|
||||
|
||||
def _attr(value: str) -> str:
|
||||
"""Escape a value going into an attribute."""
|
||||
return escape(value, quote=True)
|
||||
|
||||
|
||||
def page(
|
||||
title: str,
|
||||
body: list[str],
|
||||
language: str = "en",
|
||||
direction: str | None = None,
|
||||
variant: str | None = None,
|
||||
description: str | None = None,
|
||||
) -> str:
|
||||
"""Wrap body lines in a complete Mews page. Mirrors md2mews.page()."""
|
||||
attrs = f'xml:lang="{language}" lang="{language}"'
|
||||
if direction:
|
||||
attrs += f' dir="{direction}"'
|
||||
head = [
|
||||
f" <title>{_text(title)}</title>",
|
||||
f' <meta name="mews-profile" content="{VERSION}" />',
|
||||
' <meta name="viewport" content="width=device-width" />',
|
||||
]
|
||||
if description:
|
||||
head.append(f' <meta name="description" content="{_attr(description)}" />')
|
||||
head.append(f' <link rel="stylesheet" type="text/css" href="{STYLESHEET}" />')
|
||||
body_attrs = f' class="{variant}"' if variant else ""
|
||||
return "\n".join(
|
||||
[
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
DOCTYPE,
|
||||
f'<html xmlns="http://www.w3.org/1999/xhtml" {attrs}>',
|
||||
" <head>",
|
||||
*head,
|
||||
" </head>",
|
||||
f" <body{body_attrs}>",
|
||||
*body,
|
||||
" </body>",
|
||||
"</html>",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def site_page(title: str, body: list[str], description: str | None = None) -> str:
|
||||
"""Wrap body lines in a page with the mews.page look."""
|
||||
return page(title, body, variant=SITE_VARIANT, description=description)
|
||||
|
||||
|
||||
def message(title: str, paragraphs: list[str], links: list[tuple[str, str]]) -> str:
|
||||
"""Build a short page that says one thing, with a way onwards."""
|
||||
body = [f" <h1>{_text(title)}</h1>"]
|
||||
body += [f" <p>{_text(line)}</p>" for line in paragraphs]
|
||||
body.append(" <hr />")
|
||||
body += [
|
||||
f' <p><a href="{_attr(href)}">{_text(label)}</a></p>'
|
||||
for href, label in links
|
||||
]
|
||||
return site_page(title, body, description=paragraphs[0] if paragraphs else None)
|
||||
|
||||
|
||||
def report(findings_report, *, listed: bool, command: str) -> str:
|
||||
"""Build the page an author gets back after a check."""
|
||||
conforms = findings_report.conforms
|
||||
if listed:
|
||||
title = "Your site is listed"
|
||||
opening = (
|
||||
"That page follows Mews Profile 0.1, so your site is in the "
|
||||
"directory now. Thank you for adding it."
|
||||
)
|
||||
elif conforms:
|
||||
title = "This page conforms"
|
||||
opening = "That page follows Mews Profile 0.1. There is nothing to fix."
|
||||
else:
|
||||
title = "This page doesn't conform yet"
|
||||
opening = (
|
||||
"Here is what to fix. Each line names the section of the spec that "
|
||||
"explains the rule, so you can go and read why."
|
||||
)
|
||||
|
||||
body = [f" <h1>{_text(title)}</h1>", f" <p>{_text(opening)}</p>"]
|
||||
if findings_report.url:
|
||||
body.append(
|
||||
f' <p>Checked <a href="{_attr(findings_report.url)}">'
|
||||
f"{_text(findings_report.url)}</a>.</p>"
|
||||
)
|
||||
|
||||
if findings_report.failures:
|
||||
body.append(" <h2>To fix</h2>")
|
||||
body.append(" <ul>")
|
||||
body += [f" <li>{_text(str(f))}</li>" for f in findings_report.failures]
|
||||
body.append(" </ul>")
|
||||
if findings_report.warnings:
|
||||
body.append(" <h2>Worth fixing</h2>")
|
||||
body.append(" <ul>")
|
||||
body += [f" <li>{_text(str(f))}</li>" for f in findings_report.warnings]
|
||||
body.append(" </ul>")
|
||||
for note in findings_report.notes:
|
||||
body.append(f" <p>{_text(note)}</p>")
|
||||
|
||||
body.append(" <hr />")
|
||||
body.append(
|
||||
' <p>The rules are all in the <a href="/spec/0.1">spec</a>. Checking '
|
||||
"on your own machine is quicker while you are still fixing things:</p>"
|
||||
)
|
||||
body.append(f" <pre>{_text(command)}</pre>")
|
||||
body.append(
|
||||
' <p><a href="/check" accesskey="1">Check another page</a> '
|
||||
'· <a href="/directory" accesskey="2">The directory</a></p>'
|
||||
)
|
||||
return site_page(title, body)
|
||||
|
||||
|
||||
def directory(rows: list[sqlite3.Row]) -> str:
|
||||
"""Build the directory page: every site whose page passes the checks."""
|
||||
body = [" <h1>Mews directory</h1>"]
|
||||
if rows:
|
||||
count = f"{len(rows)} site" + ("" if len(rows) == 1 else "s")
|
||||
body.append(
|
||||
f" <p>{count} listed. Every one of these pages passed the checks "
|
||||
'against <a href="/spec/0.1">Mews Profile 0.1</a>, and gets looked '
|
||||
"at again every week.</p>"
|
||||
)
|
||||
body.append(" <ul>")
|
||||
for row in rows:
|
||||
label = _text(row["title"] or row["domain"])
|
||||
line = f' <li><a href="{_attr(row["url"])}">{label}</a>'
|
||||
line += f" — {_text(row['domain'])}"
|
||||
if row["description"]:
|
||||
line += f". {_text(row['description'])}"
|
||||
body.append(line + "</li>")
|
||||
body.append(" </ul>")
|
||||
else:
|
||||
body.append(" <p>Nothing here yet. Yours could be the first one in!</p>")
|
||||
body.append(" <hr />")
|
||||
body.append(
|
||||
' <p><a href="/check" accesskey="1">Add your site</a> '
|
||||
'· <a href="/" accesskey="0">Mews</a></p>'
|
||||
)
|
||||
body.append(
|
||||
f" <address>Updated {datetime.now(UTC).strftime('%Y-%m-%d %H:%M')} "
|
||||
"UTC</address>"
|
||||
)
|
||||
return site_page(
|
||||
"Mews directory",
|
||||
body,
|
||||
description="Sites whose pages follow Mews Profile 0.1.",
|
||||
)
|
||||
|
||||
|
||||
def write_directory(target: str, rows: list[sqlite3.Row]) -> list[str]:
|
||||
"""Write the directory page, replacing it only if it conforms.
|
||||
|
||||
The page is validated before it is published, because the one page that has
|
||||
to be a good example is this one. Returns any warnings about it.
|
||||
"""
|
||||
destination = Path(target)
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
text = directory(rows).encode()
|
||||
if len(text) > SIZE_MUST:
|
||||
raise ValueError(
|
||||
"The directory page would be over 256 KB, which section 4.3 "
|
||||
"forbids. Split it before listing more sites."
|
||||
)
|
||||
|
||||
lock = destination.with_name(destination.name + ".lock")
|
||||
with open(lock, "w") as handle:
|
||||
fcntl.flock(handle, fcntl.LOCK_EX)
|
||||
result = validate_bytes(text)
|
||||
if not result.conforms:
|
||||
raise ValueError(
|
||||
"The directory page doesn't conform: "
|
||||
+ "; ".join(str(f) for f in result.failures)
|
||||
)
|
||||
temporary = destination.with_name("." + destination.name + ".tmp")
|
||||
temporary.write_bytes(text)
|
||||
os.replace(temporary, destination)
|
||||
|
||||
warnings = [str(f) for f in result.warnings]
|
||||
if len(text) > SIZE_SHOULD:
|
||||
warnings.append(
|
||||
f"The directory page is {len(text) // 1024} KB, over the 64 KB "
|
||||
"section 4.3 asks for."
|
||||
)
|
||||
return warnings
|
||||
Loading…
Add table
Add a link
Reference in a new issue