feat: site directory with a public check endpoint
This commit is contained in:
parent
b3c436bf1a
commit
5fa14bdf1b
5 changed files with 1279 additions and 0 deletions
165
mews/app.py
Normal file
165
mews/app.py
Normal file
|
|
@ -0,0 +1,165 @@
|
||||||
|
"""The one dynamic page on mews.page.
|
||||||
|
|
||||||
|
The site is static apart from this: a form posts a page address to /result, the
|
||||||
|
page is fetched and checked, and the author gets the findings back. Nothing here
|
||||||
|
runs JavaScript or sets a cookie, because the response is itself a Mews page.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from collections.abc import Iterable
|
||||||
|
import os
|
||||||
|
import threading
|
||||||
|
from urllib.parse import parse_qs
|
||||||
|
|
||||||
|
from mews import db, pages
|
||||||
|
from mews.check import submit
|
||||||
|
from mews.fetch import Fetcher
|
||||||
|
|
||||||
|
# Outbound checks are slow and this runs on a small machine, so only a couple
|
||||||
|
# happen at once; the rest of the queue is turned away rather than piled up.
|
||||||
|
_CHECKS = threading.BoundedSemaphore(2)
|
||||||
|
_REQUESTS = threading.BoundedSemaphore(8)
|
||||||
|
_SLOT_WAIT = 5.0
|
||||||
|
|
||||||
|
MAX_BODY = 4096
|
||||||
|
|
||||||
|
BUSY = "The checker is busy right now. Try again in a few minutes."
|
||||||
|
|
||||||
|
|
||||||
|
def directory_path() -> str:
|
||||||
|
"""Where the directory page is written."""
|
||||||
|
return os.environ.get("MEWS_DIRECTORY", "directory.html")
|
||||||
|
|
||||||
|
|
||||||
|
def _start(start_response, status: str, body: bytes) -> list[bytes]:
|
||||||
|
start_response(
|
||||||
|
status,
|
||||||
|
[
|
||||||
|
("Content-Type", "text/html; charset=utf-8"),
|
||||||
|
("Content-Length", str(len(body))),
|
||||||
|
("Cache-Control", "no-store"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
return [body]
|
||||||
|
|
||||||
|
|
||||||
|
def _client(environ: dict) -> str:
|
||||||
|
"""Return the reader's address, as the proxy reports it.
|
||||||
|
|
||||||
|
Caddy sets X-Real-IP from the connection it accepted. The server binds to
|
||||||
|
the loopback address only, so nothing else can set that header.
|
||||||
|
"""
|
||||||
|
return environ.get("HTTP_X_REAL_IP") or environ.get("REMOTE_ADDR", "")
|
||||||
|
|
||||||
|
|
||||||
|
def _form(environ: dict) -> dict[str, list[str]]:
|
||||||
|
try:
|
||||||
|
length = min(int(environ.get("CONTENT_LENGTH") or 0), MAX_BODY)
|
||||||
|
except ValueError:
|
||||||
|
return {}
|
||||||
|
if length <= 0:
|
||||||
|
return {}
|
||||||
|
raw = environ["wsgi.input"].read(length)
|
||||||
|
return parse_qs(raw.decode("utf-8", "replace"), keep_blank_values=True)
|
||||||
|
|
||||||
|
|
||||||
|
def application(environ: dict, start_response) -> Iterable[bytes]:
|
||||||
|
"""Handle one request."""
|
||||||
|
path = environ.get("PATH_INFO", "/")
|
||||||
|
method = environ.get("REQUEST_METHOD", "GET")
|
||||||
|
|
||||||
|
if path != "/result":
|
||||||
|
return _start(
|
||||||
|
start_response,
|
||||||
|
"404 Not Found",
|
||||||
|
pages.message(
|
||||||
|
"Page not found",
|
||||||
|
["That address isn't part of this site."],
|
||||||
|
[("/", "Mews"), ("/directory", "The directory")],
|
||||||
|
).encode(),
|
||||||
|
)
|
||||||
|
|
||||||
|
if method != "POST":
|
||||||
|
return _start(
|
||||||
|
start_response,
|
||||||
|
"405 Method Not Allowed",
|
||||||
|
pages.message(
|
||||||
|
"Nothing to show",
|
||||||
|
["Results appear here after you check a page."],
|
||||||
|
[("/check", "Check a page")],
|
||||||
|
).encode(),
|
||||||
|
)
|
||||||
|
|
||||||
|
form = _form(environ)
|
||||||
|
raw_url = (form.get("url") or [""])[0].strip()
|
||||||
|
listing = "list" in form
|
||||||
|
|
||||||
|
if not _REQUESTS.acquire(blocking=False):
|
||||||
|
return _start(start_response, "503 Service Unavailable", _busy())
|
||||||
|
try:
|
||||||
|
if not _CHECKS.acquire(timeout=_SLOT_WAIT):
|
||||||
|
return _start(start_response, "503 Service Unavailable", _busy())
|
||||||
|
try:
|
||||||
|
body = _check(raw_url, _client(environ), listing)
|
||||||
|
finally:
|
||||||
|
_CHECKS.release()
|
||||||
|
finally:
|
||||||
|
_REQUESTS.release()
|
||||||
|
return _start(start_response, "200 OK", body)
|
||||||
|
|
||||||
|
|
||||||
|
def _busy() -> bytes:
|
||||||
|
return pages.message("Busy", [BUSY], [("/check", "Try again")]).encode()
|
||||||
|
|
||||||
|
|
||||||
|
def _check(raw_url: str, client: str, listing: bool) -> bytes:
|
||||||
|
"""Run one check and render the page the author gets back."""
|
||||||
|
connection = db.connect()
|
||||||
|
try:
|
||||||
|
with Fetcher() as fetcher:
|
||||||
|
outcome = submit(
|
||||||
|
connection,
|
||||||
|
raw_url,
|
||||||
|
client=db.ip_hash(connection, client),
|
||||||
|
listing=listing,
|
||||||
|
fetcher=fetcher,
|
||||||
|
directory=directory_path() if listing else None,
|
||||||
|
)
|
||||||
|
except ValueError as error:
|
||||||
|
# The directory page refused to publish itself. The site is listed; the
|
||||||
|
# page will be rebuilt by the next recheck pass.
|
||||||
|
return pages.message(
|
||||||
|
"Listed, but the directory didn't rebuild",
|
||||||
|
[str(error), "Your site is listed and will appear shortly."],
|
||||||
|
[("/directory", "The directory")],
|
||||||
|
).encode()
|
||||||
|
finally:
|
||||||
|
connection.close()
|
||||||
|
|
||||||
|
if outcome.report is None:
|
||||||
|
return pages.message(
|
||||||
|
"We couldn't check that page",
|
||||||
|
[outcome.message or "Something went wrong. Try again."],
|
||||||
|
[("/check", "Try again"), ("/spec/0.1", "The spec")],
|
||||||
|
).encode()
|
||||||
|
|
||||||
|
command = f"uv run mewslint.py {outcome.report.url or raw_url}"
|
||||||
|
return pages.report(outcome.report, listed=outcome.listed, command=command).encode()
|
||||||
|
|
||||||
|
|
||||||
|
def serve() -> None:
|
||||||
|
"""Run the app for real, on the loopback address only."""
|
||||||
|
# Imported here so the validator and the command line tools do not pull in
|
||||||
|
# a web server.
|
||||||
|
from waitress import serve as waitress_serve # noqa: PLC0415
|
||||||
|
|
||||||
|
listen = os.environ.get("MEWS_LISTEN", "127.0.0.1:8394")
|
||||||
|
host = listen.rsplit(":", 1)[0]
|
||||||
|
if host not in ("127.0.0.1", "::1", "localhost"):
|
||||||
|
raise SystemExit(
|
||||||
|
"mewsd serves on the loopback address only, behind a reverse proxy. "
|
||||||
|
f"Set MEWS_LISTEN to 127.0.0.1 with a port, not {listen}."
|
||||||
|
)
|
||||||
|
waitress_serve(application, listen=listen, threads=4, ident="mews.page")
|
||||||
|
|
||||||
|
|
||||||
|
app = application
|
||||||
269
mews/check.py
Normal file
269
mews/check.py
Normal file
|
|
@ -0,0 +1,269 @@
|
||||||
|
"""Checking a page and recording what came of it.
|
||||||
|
|
||||||
|
Both paths into the directory run through here: a submission from the form and
|
||||||
|
a recheck from the timer. Keeping them together means a site is listed and
|
||||||
|
relisted on exactly the same evidence.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from dataclasses import dataclass
|
||||||
|
import sqlite3
|
||||||
|
import time
|
||||||
|
from urllib.parse import urlsplit
|
||||||
|
|
||||||
|
from mews import db, pages
|
||||||
|
from mews.fetch import (
|
||||||
|
Fetched,
|
||||||
|
Fetcher,
|
||||||
|
FetchError,
|
||||||
|
UrlError,
|
||||||
|
normalise_url,
|
||||||
|
registered_domain,
|
||||||
|
)
|
||||||
|
from mews.lint import Report, validate_bytes
|
||||||
|
|
||||||
|
OK = 200
|
||||||
|
NOT_MODIFIED = 304
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Outcome:
|
||||||
|
"""What a submission produced: either a report, or a reason there isn't one."""
|
||||||
|
|
||||||
|
report: Report | None = None
|
||||||
|
listed: bool = False
|
||||||
|
message: str | None = None
|
||||||
|
title: str = ""
|
||||||
|
|
||||||
|
|
||||||
|
def _findings_rows(report: Report) -> list[tuple[str, str, str, str, str]]:
|
||||||
|
return [
|
||||||
|
(f.section, f.level, f.code, f.message, f.location) for f in report.findings
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def check_page(
|
||||||
|
url: str, *, fetcher: Fetcher
|
||||||
|
) -> tuple[Report | None, Fetched | None, str | None]:
|
||||||
|
"""Fetch and validate one page.
|
||||||
|
|
||||||
|
Returns the report, the response it came from, and a plain message when
|
||||||
|
there was no page to check.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
response = fetcher.get(url)
|
||||||
|
except (UrlError, FetchError) as error:
|
||||||
|
return None, None, str(error)
|
||||||
|
if response.status != OK:
|
||||||
|
return (
|
||||||
|
None,
|
||||||
|
response,
|
||||||
|
f"We couldn't read that page (it returned {response.status}). "
|
||||||
|
"Check the address and try again.",
|
||||||
|
)
|
||||||
|
report = validate_bytes(
|
||||||
|
response.body, url=response.url, response=response, fetcher=fetcher
|
||||||
|
)
|
||||||
|
return report, response, None
|
||||||
|
|
||||||
|
|
||||||
|
def submit(
|
||||||
|
connection: sqlite3.Connection,
|
||||||
|
raw_url: str,
|
||||||
|
*,
|
||||||
|
client: str,
|
||||||
|
listing: bool,
|
||||||
|
fetcher: Fetcher,
|
||||||
|
directory: str | None = None,
|
||||||
|
) -> Outcome:
|
||||||
|
"""Check a page, and list its site when the author asked and it conforms."""
|
||||||
|
try:
|
||||||
|
url = normalise_url(raw_url, allow_loopback=fetcher.allow_loopback)
|
||||||
|
except UrlError as error:
|
||||||
|
return Outcome(message=str(error))
|
||||||
|
|
||||||
|
host = urlsplit(url).hostname or ""
|
||||||
|
domain = registered_domain(host)
|
||||||
|
if domain is None:
|
||||||
|
return Outcome(message="That domain name isn't one the checker can reach.")
|
||||||
|
|
||||||
|
reason = db.blocked_reason(connection, domain)
|
||||||
|
if reason is not None:
|
||||||
|
db.record_submission(
|
||||||
|
connection, url=url, domain=domain, client=client, outcome="blocked"
|
||||||
|
)
|
||||||
|
return Outcome(
|
||||||
|
message="This site was taken out of the directory. Write to the "
|
||||||
|
"address on the about page if that looks wrong."
|
||||||
|
)
|
||||||
|
|
||||||
|
limited = db.rate_limited(connection, client=client, domain=domain, listing=listing)
|
||||||
|
if limited is not None:
|
||||||
|
db.record_submission(
|
||||||
|
connection, url=url, domain=domain, client=client, outcome="rate_limited"
|
||||||
|
)
|
||||||
|
return Outcome(message=limited)
|
||||||
|
|
||||||
|
started = time.monotonic()
|
||||||
|
report, response, failure = check_page(url, fetcher=fetcher)
|
||||||
|
elapsed = int((time.monotonic() - started) * 1000)
|
||||||
|
status = response.status if response else None
|
||||||
|
if report is None:
|
||||||
|
db.record_check(
|
||||||
|
connection,
|
||||||
|
site_id=None,
|
||||||
|
kind="submission",
|
||||||
|
result="unreachable",
|
||||||
|
http_status=status,
|
||||||
|
duration_ms=elapsed,
|
||||||
|
)
|
||||||
|
db.record_submission(
|
||||||
|
connection, url=url, domain=domain, client=client, outcome="unreachable"
|
||||||
|
)
|
||||||
|
return Outcome(message=failure)
|
||||||
|
|
||||||
|
if not listing:
|
||||||
|
db.record_check(
|
||||||
|
connection,
|
||||||
|
site_id=None,
|
||||||
|
kind="submission",
|
||||||
|
result="pass" if report.conforms else "fail",
|
||||||
|
findings=_findings_rows(report),
|
||||||
|
http_status=status,
|
||||||
|
size=report.size,
|
||||||
|
duration_ms=elapsed,
|
||||||
|
)
|
||||||
|
db.record_submission(
|
||||||
|
connection, url=url, domain=domain, client=client, outcome="checked"
|
||||||
|
)
|
||||||
|
return Outcome(report=report)
|
||||||
|
|
||||||
|
if not report.conforms:
|
||||||
|
db.record_check(
|
||||||
|
connection,
|
||||||
|
site_id=None,
|
||||||
|
kind="submission",
|
||||||
|
result="fail",
|
||||||
|
findings=_findings_rows(report),
|
||||||
|
http_status=status,
|
||||||
|
size=report.size,
|
||||||
|
duration_ms=elapsed,
|
||||||
|
)
|
||||||
|
db.record_submission(
|
||||||
|
connection, url=url, domain=domain, client=client, outcome="rejected"
|
||||||
|
)
|
||||||
|
return Outcome(report=report)
|
||||||
|
|
||||||
|
site_id, existed = db.upsert_site(
|
||||||
|
connection,
|
||||||
|
domain=domain,
|
||||||
|
url=report.url or url,
|
||||||
|
title=report.title or domain,
|
||||||
|
description=report.description,
|
||||||
|
language=report.language,
|
||||||
|
etag=response.headers.get("etag") if response else None,
|
||||||
|
last_modified=response.headers.get("last-modified") if response else None,
|
||||||
|
)
|
||||||
|
db.record_check(
|
||||||
|
connection,
|
||||||
|
site_id=site_id,
|
||||||
|
kind="submission",
|
||||||
|
result="pass",
|
||||||
|
findings=_findings_rows(report),
|
||||||
|
http_status=status,
|
||||||
|
size=report.size,
|
||||||
|
duration_ms=elapsed,
|
||||||
|
)
|
||||||
|
db.record_submission(
|
||||||
|
connection,
|
||||||
|
url=url,
|
||||||
|
domain=domain,
|
||||||
|
client=client,
|
||||||
|
outcome="updated" if existed else "listed",
|
||||||
|
site_id=site_id,
|
||||||
|
)
|
||||||
|
if directory:
|
||||||
|
pages.write_directory(directory, db.listed(connection))
|
||||||
|
return Outcome(report=report, listed=True, title=report.title)
|
||||||
|
|
||||||
|
|
||||||
|
def recheck(
|
||||||
|
connection: sqlite3.Connection, row: sqlite3.Row, *, fetcher: Fetcher
|
||||||
|
) -> str:
|
||||||
|
"""Check one listed site again, and drop it if its page stopped conforming.
|
||||||
|
|
||||||
|
A page that no longer conforms goes immediately: there is no contact address
|
||||||
|
to warn, and resubmitting after a fix lists the site again at once. A site
|
||||||
|
we simply could not reach is retried, because downtime is not a conformance
|
||||||
|
failure.
|
||||||
|
"""
|
||||||
|
conditional: dict[str, str] = {}
|
||||||
|
if row["etag"]:
|
||||||
|
conditional["If-None-Match"] = row["etag"]
|
||||||
|
if row["last_modified"]:
|
||||||
|
conditional["If-Modified-Since"] = row["last_modified"]
|
||||||
|
|
||||||
|
started = time.monotonic()
|
||||||
|
try:
|
||||||
|
response = fetcher.get(row["url"], headers=conditional or None)
|
||||||
|
except (UrlError, FetchError):
|
||||||
|
return _transient(connection, row, "We couldn't reach your site.")
|
||||||
|
|
||||||
|
elapsed = int((time.monotonic() - started) * 1000)
|
||||||
|
if response.status == NOT_MODIFIED:
|
||||||
|
db.mark_pass(
|
||||||
|
connection,
|
||||||
|
row["id"],
|
||||||
|
etag=row["etag"],
|
||||||
|
last_modified=row["last_modified"],
|
||||||
|
)
|
||||||
|
db.record_check(
|
||||||
|
connection,
|
||||||
|
site_id=row["id"],
|
||||||
|
kind="recheck",
|
||||||
|
result="pass",
|
||||||
|
http_status=304,
|
||||||
|
duration_ms=elapsed,
|
||||||
|
)
|
||||||
|
return "pass"
|
||||||
|
|
||||||
|
if response.status != OK:
|
||||||
|
return _transient(connection, row, f"Your page returned {response.status}.")
|
||||||
|
|
||||||
|
report = validate_bytes(
|
||||||
|
response.body, url=response.url, response=response, fetcher=fetcher
|
||||||
|
)
|
||||||
|
db.record_check(
|
||||||
|
connection,
|
||||||
|
site_id=row["id"],
|
||||||
|
kind="recheck",
|
||||||
|
result="pass" if report.conforms else "fail",
|
||||||
|
findings=_findings_rows(report),
|
||||||
|
http_status=response.status,
|
||||||
|
size=report.size,
|
||||||
|
duration_ms=elapsed,
|
||||||
|
)
|
||||||
|
if report.conforms:
|
||||||
|
db.mark_pass(
|
||||||
|
connection,
|
||||||
|
row["id"],
|
||||||
|
etag=response.headers.get("etag"),
|
||||||
|
last_modified=response.headers.get("last-modified"),
|
||||||
|
)
|
||||||
|
return "pass"
|
||||||
|
|
||||||
|
db.remove(connection, row["id"], str(report.failures[0]))
|
||||||
|
return "fail"
|
||||||
|
|
||||||
|
|
||||||
|
def _transient(connection: sqlite3.Connection, row: sqlite3.Row, note: str) -> str:
|
||||||
|
"""Count a failure that wasn't the page's fault, dropping the site if it sticks."""
|
||||||
|
count = db.mark_transient(connection, row["id"])
|
||||||
|
db.record_check(connection, site_id=row["id"], kind="recheck", result="unreachable")
|
||||||
|
if count >= db.TRANSIENT_LIMIT:
|
||||||
|
db.remove(
|
||||||
|
connection,
|
||||||
|
row["id"],
|
||||||
|
f"{note} We tried for {db.TRANSIENT_LIMIT} days.",
|
||||||
|
)
|
||||||
|
return "fail"
|
||||||
|
return "unreachable"
|
||||||
138
mews/cli.py
Normal file
138
mews/cli.py
Normal file
|
|
@ -0,0 +1,138 @@
|
||||||
|
"""mewsd: run the service, and look after the directory from the shell."""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
|
||||||
|
from mews import check as checker, db, pages
|
||||||
|
from mews.app import directory_path, serve
|
||||||
|
from mews.fetch import Fetcher, registered_domain
|
||||||
|
|
||||||
|
RECHECK_BATCH = 50
|
||||||
|
PAUSE_SECONDS = 2.0
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: list[str] | None = None) -> int:
|
||||||
|
"""Run one command. Returns an exit status."""
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
prog="mewsd", description="Run and look after the mews.page directory."
|
||||||
|
)
|
||||||
|
commands = parser.add_subparsers(dest="command", required=True)
|
||||||
|
commands.add_parser("serve", help="serve the check endpoint")
|
||||||
|
commands.add_parser("init-db", help="create the database if it isn't there")
|
||||||
|
commands.add_parser("recheck", help="check the sites that are due")
|
||||||
|
commands.add_parser("build-directory", help="write the directory page again")
|
||||||
|
drop = commands.add_parser("remove", help="take a site out and refuse it later")
|
||||||
|
drop.add_argument("domain")
|
||||||
|
drop.add_argument("--reason", required=True, help="shown in the log and to you")
|
||||||
|
allow = commands.add_parser("unblock", help="allow a removed domain again")
|
||||||
|
allow.add_argument("domain")
|
||||||
|
show = commands.add_parser("show", help="what the last check of a site found")
|
||||||
|
show.add_argument("domain")
|
||||||
|
args = parser.parse_args(argv)
|
||||||
|
|
||||||
|
if args.command == "serve":
|
||||||
|
serve()
|
||||||
|
return 0
|
||||||
|
|
||||||
|
connection = db.connect()
|
||||||
|
try:
|
||||||
|
db.init(connection)
|
||||||
|
if args.command == "init-db":
|
||||||
|
print(f"database ready at {db.path()}")
|
||||||
|
return 0
|
||||||
|
if args.command == "recheck":
|
||||||
|
return _recheck(connection)
|
||||||
|
if args.command == "build-directory":
|
||||||
|
return _build(connection)
|
||||||
|
if args.command == "remove":
|
||||||
|
return _remove(connection, args.domain, args.reason)
|
||||||
|
if args.command == "unblock":
|
||||||
|
return _unblock(connection, args.domain)
|
||||||
|
if args.command == "show":
|
||||||
|
return _show(connection, args.domain)
|
||||||
|
finally:
|
||||||
|
connection.close()
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def _domain(name: str) -> str:
|
||||||
|
return registered_domain(name) or name.strip().lower()
|
||||||
|
|
||||||
|
|
||||||
|
def _recheck(connection) -> int:
|
||||||
|
"""Check every site that is due, then write the directory page again."""
|
||||||
|
sites = db.due(connection, RECHECK_BATCH)
|
||||||
|
counts = {"pass": 0, "fail": 0, "unreachable": 0}
|
||||||
|
for index, row in enumerate(sites):
|
||||||
|
if index:
|
||||||
|
# One site at a time, with a gap, so a recheck pass is invisible to
|
||||||
|
# the sites it visits.
|
||||||
|
time.sleep(PAUSE_SECONDS)
|
||||||
|
with Fetcher() as fetcher:
|
||||||
|
counts[checker.recheck(connection, row, fetcher=fetcher)] += 1
|
||||||
|
db.prune(connection)
|
||||||
|
_build(connection)
|
||||||
|
print(
|
||||||
|
f"checked {len(sites)}: {counts['pass']} passed, {counts['fail']} dropped, "
|
||||||
|
f"{counts['unreachable']} unreachable"
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def _build(connection) -> int:
|
||||||
|
"""Write the directory page from what is listed now."""
|
||||||
|
try:
|
||||||
|
warnings = pages.write_directory(directory_path(), db.listed(connection))
|
||||||
|
except ValueError as error:
|
||||||
|
print(f"the directory page wasn't written: {error}", file=sys.stderr)
|
||||||
|
return 1
|
||||||
|
for warning in warnings:
|
||||||
|
print(f"note: {warning}", file=sys.stderr)
|
||||||
|
print(f"wrote {directory_path()}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def _remove(connection, domain: str, reason: str) -> int:
|
||||||
|
name = _domain(domain)
|
||||||
|
row = db.site(connection, name)
|
||||||
|
if row is not None:
|
||||||
|
db.remove(connection, row["id"], reason)
|
||||||
|
db.block(connection, name, reason)
|
||||||
|
_build(connection)
|
||||||
|
print(f"removed {name}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def _unblock(connection, domain: str) -> int:
|
||||||
|
name = _domain(domain)
|
||||||
|
db.unblock(connection, name)
|
||||||
|
print(f"{name} can be submitted again")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def _show(connection, domain: str) -> int:
|
||||||
|
name = _domain(domain)
|
||||||
|
row = db.site(connection, name)
|
||||||
|
if row is None:
|
||||||
|
print(f"{name} isn't in the directory", file=sys.stderr)
|
||||||
|
return 1
|
||||||
|
print(f"{row['domain']} — {row['state']}")
|
||||||
|
print(f" page {row['url']}")
|
||||||
|
print(f" title {row['title']}")
|
||||||
|
print(f" listed {row['listed_at']}")
|
||||||
|
print(f" checked {row['last_check_at']} (last passed {row['last_ok_at']})")
|
||||||
|
print(f" next {row['next_check_at']}")
|
||||||
|
if row["reason"]:
|
||||||
|
print(f" dropped {row['reason']}")
|
||||||
|
findings = db.last_findings(connection, row["id"])
|
||||||
|
for finding in findings:
|
||||||
|
print(
|
||||||
|
f" {finding['level']:6} section {finding['section']} — "
|
||||||
|
f"{finding['message']}"
|
||||||
|
)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
486
mews/db.py
Normal file
486
mews/db.py
Normal file
|
|
@ -0,0 +1,486 @@
|
||||||
|
"""The directory's SQLite store.
|
||||||
|
|
||||||
|
One row per site, keyed by registrable domain, plus the submission and check
|
||||||
|
history the rate limits and the recheck schedule read. Client addresses are
|
||||||
|
never stored: a submission keeps a salted hash, and the salt is replaced every
|
||||||
|
month so old hashes stop being comparable to new ones.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import UTC, datetime, timedelta
|
||||||
|
import hashlib
|
||||||
|
import os
|
||||||
|
import secrets
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
SCHEMA_VERSION = "1"
|
||||||
|
|
||||||
|
RECHECK_DAYS = 7
|
||||||
|
RETRY_HOURS = 24
|
||||||
|
TRANSIENT_LIMIT = 3
|
||||||
|
RETENTION_DAYS = 90
|
||||||
|
|
||||||
|
# Rate limits. Listing is slower than checking because it writes to the
|
||||||
|
# directory; checking only costs a fetch.
|
||||||
|
IP_LISTINGS_PER_HOUR = 5
|
||||||
|
IP_LISTINGS_PER_DAY = 20
|
||||||
|
IP_CHECKS_PER_HOUR = 20
|
||||||
|
DOMAIN_COOLDOWN_MINUTES = 10
|
||||||
|
DOMAIN_REJECTS_BEFORE_SLOWDOWN = 3
|
||||||
|
DOMAIN_SLOW_COOLDOWN_HOURS = 24
|
||||||
|
LISTINGS_PER_HOUR = 60
|
||||||
|
|
||||||
|
SCHEMA = """
|
||||||
|
CREATE TABLE IF NOT EXISTS meta (
|
||||||
|
key TEXT PRIMARY KEY,
|
||||||
|
value TEXT NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS sites (
|
||||||
|
id INTEGER PRIMARY KEY,
|
||||||
|
domain TEXT NOT NULL UNIQUE,
|
||||||
|
url TEXT NOT NULL,
|
||||||
|
title TEXT NOT NULL,
|
||||||
|
description TEXT,
|
||||||
|
language TEXT,
|
||||||
|
state TEXT NOT NULL CHECK (state IN ('listed', 'removed')),
|
||||||
|
reason TEXT,
|
||||||
|
listed_at TEXT NOT NULL,
|
||||||
|
last_check_at TEXT,
|
||||||
|
last_ok_at TEXT,
|
||||||
|
next_check_at TEXT,
|
||||||
|
transient_failures INTEGER NOT NULL DEFAULT 0,
|
||||||
|
etag TEXT,
|
||||||
|
last_modified TEXT
|
||||||
|
);
|
||||||
|
CREATE INDEX IF NOT EXISTS sites_due
|
||||||
|
ON sites(next_check_at) WHERE state = 'listed';
|
||||||
|
CREATE INDEX IF NOT EXISTS sites_state ON sites(state, domain);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS submissions (
|
||||||
|
id INTEGER PRIMARY KEY,
|
||||||
|
created_at TEXT NOT NULL,
|
||||||
|
url TEXT NOT NULL,
|
||||||
|
domain TEXT,
|
||||||
|
ip_hash TEXT NOT NULL,
|
||||||
|
outcome TEXT NOT NULL CHECK (outcome IN (
|
||||||
|
'listed', 'updated', 'rejected', 'unreachable',
|
||||||
|
'rate_limited', 'blocked', 'checked')),
|
||||||
|
site_id INTEGER REFERENCES sites(id) ON DELETE SET NULL
|
||||||
|
);
|
||||||
|
CREATE INDEX IF NOT EXISTS submissions_ip ON submissions(ip_hash, created_at);
|
||||||
|
CREATE INDEX IF NOT EXISTS submissions_domain ON submissions(domain, created_at);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS checks (
|
||||||
|
id INTEGER PRIMARY KEY,
|
||||||
|
site_id INTEGER REFERENCES sites(id) ON DELETE CASCADE,
|
||||||
|
checked_at TEXT NOT NULL,
|
||||||
|
kind TEXT NOT NULL CHECK (kind IN ('submission', 'recheck')),
|
||||||
|
result TEXT NOT NULL CHECK (result IN ('pass', 'fail', 'unreachable')),
|
||||||
|
http_status INTEGER,
|
||||||
|
bytes INTEGER,
|
||||||
|
duration_ms INTEGER
|
||||||
|
);
|
||||||
|
CREATE INDEX IF NOT EXISTS checks_site ON checks(site_id, checked_at DESC);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS findings (
|
||||||
|
check_id INTEGER NOT NULL REFERENCES checks(id) ON DELETE CASCADE,
|
||||||
|
section TEXT NOT NULL,
|
||||||
|
level TEXT NOT NULL CHECK (level IN ('must', 'should')),
|
||||||
|
code TEXT NOT NULL,
|
||||||
|
message TEXT NOT NULL,
|
||||||
|
location TEXT
|
||||||
|
);
|
||||||
|
CREATE INDEX IF NOT EXISTS findings_check ON findings(check_id);
|
||||||
|
|
||||||
|
CREATE TABLE IF NOT EXISTS blocked (
|
||||||
|
domain TEXT PRIMARY KEY,
|
||||||
|
blocked_at TEXT NOT NULL,
|
||||||
|
reason TEXT NOT NULL
|
||||||
|
);
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def now() -> datetime:
|
||||||
|
"""Return the current time, in UTC."""
|
||||||
|
return datetime.now(UTC)
|
||||||
|
|
||||||
|
|
||||||
|
def path() -> str:
|
||||||
|
"""Return where the database lives."""
|
||||||
|
return os.environ.get("MEWS_DB", "mews.db")
|
||||||
|
|
||||||
|
|
||||||
|
def connect(database: str | None = None) -> sqlite3.Connection:
|
||||||
|
"""Open the database with the settings every caller wants."""
|
||||||
|
connection = sqlite3.connect(database or path(), timeout=5.0)
|
||||||
|
connection.row_factory = sqlite3.Row
|
||||||
|
connection.execute("PRAGMA journal_mode = WAL")
|
||||||
|
connection.execute("PRAGMA foreign_keys = ON")
|
||||||
|
connection.execute("PRAGMA busy_timeout = 5000")
|
||||||
|
return connection
|
||||||
|
|
||||||
|
|
||||||
|
def init(connection: sqlite3.Connection) -> None:
|
||||||
|
"""Create the schema if it isn't there yet."""
|
||||||
|
connection.executescript(SCHEMA)
|
||||||
|
connection.execute(
|
||||||
|
"INSERT OR IGNORE INTO meta (key, value) VALUES ('schema_version', ?)",
|
||||||
|
(SCHEMA_VERSION,),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def ip_hash(connection: sqlite3.Connection, address: str) -> str:
|
||||||
|
"""Hash a client address with a salt that is replaced each month."""
|
||||||
|
stamp = now().strftime("%Y-%m")
|
||||||
|
row = connection.execute(
|
||||||
|
"SELECT value FROM meta WHERE key = 'ip_salt_month'"
|
||||||
|
).fetchone()
|
||||||
|
if row is None or row["value"] != stamp:
|
||||||
|
connection.execute(
|
||||||
|
"INSERT INTO meta (key, value) VALUES ('ip_salt', ?) "
|
||||||
|
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
||||||
|
(secrets.token_hex(16),),
|
||||||
|
)
|
||||||
|
connection.execute(
|
||||||
|
"INSERT INTO meta (key, value) VALUES ('ip_salt_month', ?) "
|
||||||
|
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
||||||
|
(stamp,),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
salt = connection.execute(
|
||||||
|
"SELECT value FROM meta WHERE key = 'ip_salt'"
|
||||||
|
).fetchone()["value"]
|
||||||
|
return hashlib.sha256(f"{salt}{address}".encode()).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def _since(hours: float) -> str:
|
||||||
|
return (now() - timedelta(hours=hours)).isoformat()
|
||||||
|
|
||||||
|
|
||||||
|
def _count(connection: sqlite3.Connection, sql: str, *args: object) -> int:
|
||||||
|
return connection.execute(sql, args).fetchone()[0]
|
||||||
|
|
||||||
|
|
||||||
|
def blocked_reason(connection: sqlite3.Connection, domain: str) -> str | None:
|
||||||
|
"""Return why a domain was removed for good, or None if it wasn't."""
|
||||||
|
row = connection.execute(
|
||||||
|
"SELECT reason FROM blocked WHERE domain = ?", (domain,)
|
||||||
|
).fetchone()
|
||||||
|
return row["reason"] if row else None
|
||||||
|
|
||||||
|
|
||||||
|
def rate_limited(
|
||||||
|
connection: sqlite3.Connection, *, client: str, domain: str, listing: bool
|
||||||
|
) -> str | None:
|
||||||
|
"""Return why this request can't go ahead now, as a sentence for the author.
|
||||||
|
|
||||||
|
Checking a page is cheap and gets a looser limit than listing a site.
|
||||||
|
"""
|
||||||
|
if not listing:
|
||||||
|
checks = _count(
|
||||||
|
connection,
|
||||||
|
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ?",
|
||||||
|
client,
|
||||||
|
_since(1),
|
||||||
|
)
|
||||||
|
if checks >= IP_CHECKS_PER_HOUR:
|
||||||
|
return "You've checked several pages already. Try again in an hour."
|
||||||
|
return None
|
||||||
|
|
||||||
|
if (
|
||||||
|
_count(
|
||||||
|
connection,
|
||||||
|
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? "
|
||||||
|
"AND outcome IN ('listed', 'updated', 'rejected')",
|
||||||
|
client,
|
||||||
|
_since(1),
|
||||||
|
)
|
||||||
|
>= IP_LISTINGS_PER_HOUR
|
||||||
|
):
|
||||||
|
return "You've submitted several sites already. Try again in an hour."
|
||||||
|
if (
|
||||||
|
_count(
|
||||||
|
connection,
|
||||||
|
"SELECT COUNT(*) FROM submissions WHERE ip_hash = ? AND created_at > ? "
|
||||||
|
"AND outcome IN ('listed', 'updated', 'rejected')",
|
||||||
|
client,
|
||||||
|
_since(24),
|
||||||
|
)
|
||||||
|
>= IP_LISTINGS_PER_DAY
|
||||||
|
):
|
||||||
|
return "You've submitted several sites already. Try again tomorrow."
|
||||||
|
|
||||||
|
rejects = _count(
|
||||||
|
connection,
|
||||||
|
"SELECT COUNT(*) FROM submissions WHERE domain = ? AND outcome = 'rejected' "
|
||||||
|
"AND created_at > ?",
|
||||||
|
domain,
|
||||||
|
_since(24),
|
||||||
|
)
|
||||||
|
cooldown = (
|
||||||
|
DOMAIN_SLOW_COOLDOWN_HOURS
|
||||||
|
if rejects >= DOMAIN_REJECTS_BEFORE_SLOWDOWN
|
||||||
|
else DOMAIN_COOLDOWN_MINUTES / 60
|
||||||
|
)
|
||||||
|
if _count(
|
||||||
|
connection,
|
||||||
|
"SELECT COUNT(*) FROM submissions WHERE domain = ? AND created_at > ?",
|
||||||
|
domain,
|
||||||
|
_since(cooldown),
|
||||||
|
):
|
||||||
|
if cooldown > 1:
|
||||||
|
return (
|
||||||
|
f"{domain} didn't pass the checks a few times today. Fix the "
|
||||||
|
"page and try again tomorrow."
|
||||||
|
)
|
||||||
|
return f"{domain} was submitted a moment ago. Try again in ten minutes."
|
||||||
|
|
||||||
|
if (
|
||||||
|
_count(
|
||||||
|
connection,
|
||||||
|
"SELECT COUNT(*) FROM submissions WHERE created_at > ? "
|
||||||
|
"AND outcome IN ('listed', 'updated')",
|
||||||
|
_since(1),
|
||||||
|
)
|
||||||
|
>= LISTINGS_PER_HOUR
|
||||||
|
):
|
||||||
|
return "The directory is busy right now. Try again in an hour."
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def record_submission(
|
||||||
|
connection: sqlite3.Connection,
|
||||||
|
*,
|
||||||
|
url: str,
|
||||||
|
domain: str | None,
|
||||||
|
client: str,
|
||||||
|
outcome: str,
|
||||||
|
site_id: int | None = None,
|
||||||
|
) -> int:
|
||||||
|
"""Store one submission and return its id."""
|
||||||
|
cursor = connection.execute(
|
||||||
|
"INSERT INTO submissions (created_at, url, domain, ip_hash, outcome, site_id) "
|
||||||
|
"VALUES (?, ?, ?, ?, ?, ?)",
|
||||||
|
(now().isoformat(), url, domain, client, outcome, site_id),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
return int(cursor.lastrowid or 0)
|
||||||
|
|
||||||
|
|
||||||
|
def record_check(
|
||||||
|
connection: sqlite3.Connection,
|
||||||
|
*,
|
||||||
|
site_id: int | None,
|
||||||
|
kind: str,
|
||||||
|
result: str,
|
||||||
|
findings: list[tuple[str, str, str, str, str]] = [], # noqa: B006
|
||||||
|
http_status: int | None = None,
|
||||||
|
size: int | None = None,
|
||||||
|
duration_ms: int | None = None,
|
||||||
|
) -> int:
|
||||||
|
"""Store one check and its findings, returning the check id."""
|
||||||
|
cursor = connection.execute(
|
||||||
|
"INSERT INTO checks (site_id, checked_at, kind, result, http_status, bytes, "
|
||||||
|
"duration_ms) VALUES (?, ?, ?, ?, ?, ?, ?)",
|
||||||
|
(site_id, now().isoformat(), kind, result, http_status, size, duration_ms),
|
||||||
|
)
|
||||||
|
check_id = int(cursor.lastrowid or 0)
|
||||||
|
connection.executemany(
|
||||||
|
"INSERT INTO findings (check_id, section, level, code, message, location) "
|
||||||
|
"VALUES (?, ?, ?, ?, ?, ?)",
|
||||||
|
[(check_id, *finding) for finding in findings],
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
return check_id
|
||||||
|
|
||||||
|
|
||||||
|
def next_check(moment: datetime | None = None) -> str:
|
||||||
|
"""Return when a site that just passed should be looked at again.
|
||||||
|
|
||||||
|
The interval is jittered so submissions made together do not come back as a
|
||||||
|
burst a week later.
|
||||||
|
"""
|
||||||
|
base = (moment or now()) + timedelta(days=RECHECK_DAYS)
|
||||||
|
offset = secrets.randbelow(24 * 60) - 12 * 60
|
||||||
|
return (base + timedelta(minutes=offset)).isoformat()
|
||||||
|
|
||||||
|
|
||||||
|
def upsert_site(
|
||||||
|
connection: sqlite3.Connection,
|
||||||
|
*,
|
||||||
|
domain: str,
|
||||||
|
url: str,
|
||||||
|
title: str,
|
||||||
|
description: str,
|
||||||
|
language: str,
|
||||||
|
etag: str | None,
|
||||||
|
last_modified: str | None,
|
||||||
|
) -> tuple[int, bool]:
|
||||||
|
"""List a site, or refresh the row of one already listed.
|
||||||
|
|
||||||
|
Returns the site id and whether the row already existed.
|
||||||
|
"""
|
||||||
|
moment = now().isoformat()
|
||||||
|
row = connection.execute(
|
||||||
|
"SELECT id FROM sites WHERE domain = ?", (domain,)
|
||||||
|
).fetchone()
|
||||||
|
if row is None:
|
||||||
|
cursor = connection.execute(
|
||||||
|
"INSERT INTO sites (domain, url, title, description, language, state, "
|
||||||
|
"listed_at, last_check_at, last_ok_at, next_check_at, etag, last_modified) "
|
||||||
|
"VALUES (?, ?, ?, ?, ?, 'listed', ?, ?, ?, ?, ?, ?)",
|
||||||
|
(
|
||||||
|
domain,
|
||||||
|
url,
|
||||||
|
title,
|
||||||
|
description,
|
||||||
|
language,
|
||||||
|
moment,
|
||||||
|
moment,
|
||||||
|
moment,
|
||||||
|
next_check(),
|
||||||
|
etag,
|
||||||
|
last_modified,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
return int(cursor.lastrowid or 0), False
|
||||||
|
|
||||||
|
connection.execute(
|
||||||
|
"UPDATE sites SET url = ?, title = ?, description = ?, language = ?, "
|
||||||
|
"state = 'listed', reason = NULL, last_check_at = ?, last_ok_at = ?, "
|
||||||
|
"next_check_at = ?, transient_failures = 0, etag = ?, last_modified = ? "
|
||||||
|
"WHERE id = ?",
|
||||||
|
(
|
||||||
|
url,
|
||||||
|
title,
|
||||||
|
description,
|
||||||
|
language,
|
||||||
|
moment,
|
||||||
|
moment,
|
||||||
|
next_check(),
|
||||||
|
etag,
|
||||||
|
last_modified,
|
||||||
|
row["id"],
|
||||||
|
),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
return int(row["id"]), True
|
||||||
|
|
||||||
|
|
||||||
|
def listed(connection: sqlite3.Connection) -> list[sqlite3.Row]:
|
||||||
|
"""Return every listed site, in the order the directory page shows them."""
|
||||||
|
return list(
|
||||||
|
connection.execute(
|
||||||
|
"SELECT * FROM sites WHERE state = 'listed' ORDER BY domain"
|
||||||
|
).fetchall()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def due(connection: sqlite3.Connection, limit: int = 50) -> list[sqlite3.Row]:
|
||||||
|
"""Return the listed sites whose next check has come round."""
|
||||||
|
return list(
|
||||||
|
connection.execute(
|
||||||
|
"SELECT * FROM sites WHERE state = 'listed' AND next_check_at <= ? "
|
||||||
|
"ORDER BY next_check_at LIMIT ?",
|
||||||
|
(now().isoformat(), limit),
|
||||||
|
).fetchall()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def site(connection: sqlite3.Connection, domain: str) -> sqlite3.Row | None:
|
||||||
|
"""Return one site by registrable domain."""
|
||||||
|
return connection.execute(
|
||||||
|
"SELECT * FROM sites WHERE domain = ?", (domain,)
|
||||||
|
).fetchone()
|
||||||
|
|
||||||
|
|
||||||
|
def mark_pass(connection: sqlite3.Connection, site_id: int, **columns: object) -> None:
|
||||||
|
"""Record a check a site passed."""
|
||||||
|
moment = now().isoformat()
|
||||||
|
connection.execute(
|
||||||
|
"UPDATE sites SET last_check_at = ?, last_ok_at = ?, next_check_at = ?, "
|
||||||
|
"transient_failures = 0, etag = ?, last_modified = ? WHERE id = ?",
|
||||||
|
(
|
||||||
|
moment,
|
||||||
|
moment,
|
||||||
|
next_check(),
|
||||||
|
columns.get("etag"),
|
||||||
|
columns.get("last_modified"),
|
||||||
|
site_id,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def mark_transient(connection: sqlite3.Connection, site_id: int) -> int:
|
||||||
|
"""Count one failure that wasn't the page's fault, and return the new count.
|
||||||
|
|
||||||
|
A site is only dropped once these pile up: a weekend of downtime should not
|
||||||
|
empty the directory.
|
||||||
|
"""
|
||||||
|
connection.execute(
|
||||||
|
"UPDATE sites SET transient_failures = transient_failures + 1, "
|
||||||
|
"last_check_at = ?, next_check_at = ? WHERE id = ?",
|
||||||
|
(
|
||||||
|
now().isoformat(),
|
||||||
|
(now() + timedelta(hours=RETRY_HOURS)).isoformat(),
|
||||||
|
site_id,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
return int(
|
||||||
|
connection.execute(
|
||||||
|
"SELECT transient_failures FROM sites WHERE id = ?", (site_id,)
|
||||||
|
).fetchone()[0]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def remove(connection: sqlite3.Connection, site_id: int, reason: str) -> None:
|
||||||
|
"""Drop a site from the directory, keeping the row for the record."""
|
||||||
|
connection.execute(
|
||||||
|
"UPDATE sites SET state = 'removed', reason = ?, last_check_at = ?, "
|
||||||
|
"next_check_at = NULL WHERE id = ?",
|
||||||
|
(reason, now().isoformat(), site_id),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def block(connection: sqlite3.Connection, domain: str, reason: str) -> None:
|
||||||
|
"""Refuse future submissions from a domain."""
|
||||||
|
connection.execute(
|
||||||
|
"INSERT INTO blocked (domain, blocked_at, reason) VALUES (?, ?, ?) "
|
||||||
|
"ON CONFLICT(domain) DO UPDATE SET reason = excluded.reason",
|
||||||
|
(domain, now().isoformat(), reason),
|
||||||
|
)
|
||||||
|
connection.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def unblock(connection: sqlite3.Connection, domain: str) -> None:
|
||||||
|
"""Allow submissions from a domain again."""
|
||||||
|
connection.execute("DELETE FROM blocked WHERE domain = ?", (domain,))
|
||||||
|
connection.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def last_findings(connection: sqlite3.Connection, site_id: int) -> list[sqlite3.Row]:
|
||||||
|
"""Return the findings from a site's most recent check."""
|
||||||
|
row = connection.execute(
|
||||||
|
"SELECT id FROM checks WHERE site_id = ? ORDER BY checked_at DESC LIMIT 1",
|
||||||
|
(site_id,),
|
||||||
|
).fetchone()
|
||||||
|
if row is None:
|
||||||
|
return []
|
||||||
|
return list(
|
||||||
|
connection.execute(
|
||||||
|
"SELECT * FROM findings WHERE check_id = ?", (row["id"],)
|
||||||
|
).fetchall()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def prune(connection: sqlite3.Connection, days: int = RETENTION_DAYS) -> None:
|
||||||
|
"""Delete submission and check history past the retention window."""
|
||||||
|
cutoff = (now() - timedelta(days=days)).isoformat()
|
||||||
|
connection.execute("DELETE FROM submissions WHERE created_at < ?", (cutoff,))
|
||||||
|
connection.execute("DELETE FROM checks WHERE checked_at < ?", (cutoff,))
|
||||||
|
connection.commit()
|
||||||
221
mews/pages.py
Normal file
221
mews/pages.py
Normal file
|
|
@ -0,0 +1,221 @@
|
||||||
|
"""Emitting the pages mews.page serves.
|
||||||
|
|
||||||
|
Every page the service produces is itself a Mews page, so this module writes the
|
||||||
|
same prologue, head and body shape as md2mews.py. The emitter is repeated here
|
||||||
|
rather than imported because md2mews.py is a standalone script that pulls in a
|
||||||
|
Markdown parser; tests/test_pages.py fails if the two drift apart.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from datetime import UTC, datetime
|
||||||
|
import fcntl
|
||||||
|
from html import escape
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
from mews import VERSION
|
||||||
|
from mews.lint import validate_bytes
|
||||||
|
|
||||||
|
DOCTYPE = (
|
||||||
|
'<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"\n'
|
||||||
|
' "http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">'
|
||||||
|
)
|
||||||
|
VARIANTS = ("mews-warm", "mews-cool", "mews-green", "mews-mono")
|
||||||
|
STYLESHEET = "mews-0.1.css"
|
||||||
|
SITE_VARIANT = "mews-cool"
|
||||||
|
|
||||||
|
# SPEC.md 4.3 caps a page at 256 KB and asks for 64 KB. The directory refuses to
|
||||||
|
# publish a page over the first and warns past the second.
|
||||||
|
SIZE_MUST = 256 * 1024
|
||||||
|
SIZE_SHOULD = 64 * 1024
|
||||||
|
|
||||||
|
|
||||||
|
def _text(value: str) -> str:
|
||||||
|
"""Escape text content, leaving apostrophes alone so prose reads as prose."""
|
||||||
|
return escape(value, quote=False)
|
||||||
|
|
||||||
|
|
||||||
|
def _attr(value: str) -> str:
|
||||||
|
"""Escape a value going into an attribute."""
|
||||||
|
return escape(value, quote=True)
|
||||||
|
|
||||||
|
|
||||||
|
def page(
|
||||||
|
title: str,
|
||||||
|
body: list[str],
|
||||||
|
language: str = "en",
|
||||||
|
direction: str | None = None,
|
||||||
|
variant: str | None = None,
|
||||||
|
description: str | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Wrap body lines in a complete Mews page. Mirrors md2mews.page()."""
|
||||||
|
attrs = f'xml:lang="{language}" lang="{language}"'
|
||||||
|
if direction:
|
||||||
|
attrs += f' dir="{direction}"'
|
||||||
|
head = [
|
||||||
|
f" <title>{_text(title)}</title>",
|
||||||
|
f' <meta name="mews-profile" content="{VERSION}" />',
|
||||||
|
' <meta name="viewport" content="width=device-width" />',
|
||||||
|
]
|
||||||
|
if description:
|
||||||
|
head.append(f' <meta name="description" content="{_attr(description)}" />')
|
||||||
|
head.append(f' <link rel="stylesheet" type="text/css" href="{STYLESHEET}" />')
|
||||||
|
body_attrs = f' class="{variant}"' if variant else ""
|
||||||
|
return "\n".join(
|
||||||
|
[
|
||||||
|
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||||
|
DOCTYPE,
|
||||||
|
f'<html xmlns="http://www.w3.org/1999/xhtml" {attrs}>',
|
||||||
|
" <head>",
|
||||||
|
*head,
|
||||||
|
" </head>",
|
||||||
|
f" <body{body_attrs}>",
|
||||||
|
*body,
|
||||||
|
" </body>",
|
||||||
|
"</html>",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def site_page(title: str, body: list[str], description: str | None = None) -> str:
|
||||||
|
"""Wrap body lines in a page with the mews.page look."""
|
||||||
|
return page(title, body, variant=SITE_VARIANT, description=description)
|
||||||
|
|
||||||
|
|
||||||
|
def message(title: str, paragraphs: list[str], links: list[tuple[str, str]]) -> str:
|
||||||
|
"""Build a short page that says one thing, with a way onwards."""
|
||||||
|
body = [f" <h1>{_text(title)}</h1>"]
|
||||||
|
body += [f" <p>{_text(line)}</p>" for line in paragraphs]
|
||||||
|
body.append(" <hr />")
|
||||||
|
body += [
|
||||||
|
f' <p><a href="{_attr(href)}">{_text(label)}</a></p>'
|
||||||
|
for href, label in links
|
||||||
|
]
|
||||||
|
return site_page(title, body, description=paragraphs[0] if paragraphs else None)
|
||||||
|
|
||||||
|
|
||||||
|
def report(findings_report, *, listed: bool, command: str) -> str:
|
||||||
|
"""Build the page an author gets back after a check."""
|
||||||
|
conforms = findings_report.conforms
|
||||||
|
if listed:
|
||||||
|
title = "Your site is listed"
|
||||||
|
opening = (
|
||||||
|
"That page follows Mews Profile 0.1, so your site is in the "
|
||||||
|
"directory now. Thank you for adding it."
|
||||||
|
)
|
||||||
|
elif conforms:
|
||||||
|
title = "This page conforms"
|
||||||
|
opening = "That page follows Mews Profile 0.1. There is nothing to fix."
|
||||||
|
else:
|
||||||
|
title = "This page doesn't conform yet"
|
||||||
|
opening = (
|
||||||
|
"Here is what to fix. Each line names the section of the spec that "
|
||||||
|
"explains the rule, so you can go and read why."
|
||||||
|
)
|
||||||
|
|
||||||
|
body = [f" <h1>{_text(title)}</h1>", f" <p>{_text(opening)}</p>"]
|
||||||
|
if findings_report.url:
|
||||||
|
body.append(
|
||||||
|
f' <p>Checked <a href="{_attr(findings_report.url)}">'
|
||||||
|
f"{_text(findings_report.url)}</a>.</p>"
|
||||||
|
)
|
||||||
|
|
||||||
|
if findings_report.failures:
|
||||||
|
body.append(" <h2>To fix</h2>")
|
||||||
|
body.append(" <ul>")
|
||||||
|
body += [f" <li>{_text(str(f))}</li>" for f in findings_report.failures]
|
||||||
|
body.append(" </ul>")
|
||||||
|
if findings_report.warnings:
|
||||||
|
body.append(" <h2>Worth fixing</h2>")
|
||||||
|
body.append(" <ul>")
|
||||||
|
body += [f" <li>{_text(str(f))}</li>" for f in findings_report.warnings]
|
||||||
|
body.append(" </ul>")
|
||||||
|
for note in findings_report.notes:
|
||||||
|
body.append(f" <p>{_text(note)}</p>")
|
||||||
|
|
||||||
|
body.append(" <hr />")
|
||||||
|
body.append(
|
||||||
|
' <p>The rules are all in the <a href="/spec/0.1">spec</a>. Checking '
|
||||||
|
"on your own machine is quicker while you are still fixing things:</p>"
|
||||||
|
)
|
||||||
|
body.append(f" <pre>{_text(command)}</pre>")
|
||||||
|
body.append(
|
||||||
|
' <p><a href="/check" accesskey="1">Check another page</a> '
|
||||||
|
'· <a href="/directory" accesskey="2">The directory</a></p>'
|
||||||
|
)
|
||||||
|
return site_page(title, body)
|
||||||
|
|
||||||
|
|
||||||
|
def directory(rows: list[sqlite3.Row]) -> str:
|
||||||
|
"""Build the directory page: every site whose page passes the checks."""
|
||||||
|
body = [" <h1>Mews directory</h1>"]
|
||||||
|
if rows:
|
||||||
|
count = f"{len(rows)} site" + ("" if len(rows) == 1 else "s")
|
||||||
|
body.append(
|
||||||
|
f" <p>{count} listed. Every one of these pages passed the checks "
|
||||||
|
'against <a href="/spec/0.1">Mews Profile 0.1</a>, and gets looked '
|
||||||
|
"at again every week.</p>"
|
||||||
|
)
|
||||||
|
body.append(" <ul>")
|
||||||
|
for row in rows:
|
||||||
|
label = _text(row["title"] or row["domain"])
|
||||||
|
line = f' <li><a href="{_attr(row["url"])}">{label}</a>'
|
||||||
|
line += f" — {_text(row['domain'])}"
|
||||||
|
if row["description"]:
|
||||||
|
line += f". {_text(row['description'])}"
|
||||||
|
body.append(line + "</li>")
|
||||||
|
body.append(" </ul>")
|
||||||
|
else:
|
||||||
|
body.append(" <p>Nothing here yet. Yours could be the first one in!</p>")
|
||||||
|
body.append(" <hr />")
|
||||||
|
body.append(
|
||||||
|
' <p><a href="/check" accesskey="1">Add your site</a> '
|
||||||
|
'· <a href="/" accesskey="0">Mews</a></p>'
|
||||||
|
)
|
||||||
|
body.append(
|
||||||
|
f" <address>Updated {datetime.now(UTC).strftime('%Y-%m-%d %H:%M')} "
|
||||||
|
"UTC</address>"
|
||||||
|
)
|
||||||
|
return site_page(
|
||||||
|
"Mews directory",
|
||||||
|
body,
|
||||||
|
description="Sites whose pages follow Mews Profile 0.1.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def write_directory(target: str, rows: list[sqlite3.Row]) -> list[str]:
|
||||||
|
"""Write the directory page, replacing it only if it conforms.
|
||||||
|
|
||||||
|
The page is validated before it is published, because the one page that has
|
||||||
|
to be a good example is this one. Returns any warnings about it.
|
||||||
|
"""
|
||||||
|
destination = Path(target)
|
||||||
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
text = directory(rows).encode()
|
||||||
|
if len(text) > SIZE_MUST:
|
||||||
|
raise ValueError(
|
||||||
|
"The directory page would be over 256 KB, which section 4.3 "
|
||||||
|
"forbids. Split it before listing more sites."
|
||||||
|
)
|
||||||
|
|
||||||
|
lock = destination.with_name(destination.name + ".lock")
|
||||||
|
with open(lock, "w") as handle:
|
||||||
|
fcntl.flock(handle, fcntl.LOCK_EX)
|
||||||
|
result = validate_bytes(text)
|
||||||
|
if not result.conforms:
|
||||||
|
raise ValueError(
|
||||||
|
"The directory page doesn't conform: "
|
||||||
|
+ "; ".join(str(f) for f in result.failures)
|
||||||
|
)
|
||||||
|
temporary = destination.with_name("." + destination.name + ".tmp")
|
||||||
|
temporary.write_bytes(text)
|
||||||
|
os.replace(temporary, destination)
|
||||||
|
|
||||||
|
warnings = [str(f) for f in result.warnings]
|
||||||
|
if len(text) > SIZE_SHOULD:
|
||||||
|
warnings.append(
|
||||||
|
f"The directory page is {len(text) // 1024} KB, over the 64 KB "
|
||||||
|
"section 4.3 asks for."
|
||||||
|
)
|
||||||
|
return warnings
|
||||||
Loading…
Add table
Add a link
Reference in a new issue