diff --git a/pelican-mews/README.md b/pelican-mews/README.md index 5c11a88..0088b0e 100644 --- a/pelican-mews/README.md +++ b/pelican-mews/README.md @@ -52,6 +52,7 @@ Three optional settings: | `MEWS_STRICT` | `False` | Fail the build on content outside the subset instead of stripping it. Removals are always logged either way. | | `MEWS_CSS_DIR` | `"css"` | Directory below the output root where `mews-0.1.css` is copied. Themes see the resulting path as the `MEWS_CSS` global. | | `MEWS_OUTPUT_PATH` | unset | Put the finished pages in this folder instead of `OUTPUT_PATH`. | +| `MEWS_SUFFIX` | `".xhtml"` | Which written files are Mews pages. Set it to `".html"` on a site that is Mews throughout, and save your pages under that suffix too: servers send `.xhtml` as `application/xhtml+xml`, which makes a browser parse the page as XML and fail outright on a small markup error, while section 7.1 asks for `text/html` so the page still shows. `index.html` is also the directory index everywhere, which `index.xhtml` is not. Note that `.html` puts every page Pelican writes through the plugin, including the index, tag and archive pages it falls back to when the theme has no template for them, so give those a Mews template or switch them off with `DIRECT_TEMPLATES` and the matching `*_SAVE_AS` settings. | ### Rendering beside an existing blog diff --git a/pelican-mews/pelican/plugins/mews/mews.py b/pelican-mews/pelican/plugins/mews/mews.py index babe147..983be2c 100644 --- a/pelican-mews/pelican/plugins/mews/mews.py +++ b/pelican-mews/pelican/plugins/mews/mews.py @@ -47,7 +47,13 @@ LOGGER = logging.getLogger(__name__) CSS_NAME = "mews-0.1.css" DEFAULT_CSS_DIR = "css" -XHTML_SUFFIX = ".xhtml" +# Which files Pelican writes are Mews pages. The default keeps the plugin off +# everything else, so it can run on a site that also produces ordinary HTML. +# A site that is Mews throughout wants ".html" instead: servers send .xhtml as +# application/xhtml+xml, which makes a browser parse as XML and fail hard on a +# small markup error, and section 7.1 asks for text/html precisely to avoid +# that. Set MEWS_SUFFIX to choose. +DEFAULT_SUFFIX = ".xhtml" ALLOWLIST = parse_allowlist( files(__package__).joinpath("dtd", "mews.dtd").read_text(encoding="utf-8") @@ -67,6 +73,12 @@ def _on_initialized(pelican) -> None: overrides.append(templates) +def _suffix() -> str: + """Return the file extension a Mews page is written under, with its dot.""" + suffix = str(_settings.get("MEWS_SUFFIX", DEFAULT_SUFFIX)).strip() + return suffix if suffix.startswith(".") else f".{suffix}" + + def _on_generator_init(generator) -> None: css_dir = generator.settings.get("MEWS_CSS_DIR", DEFAULT_CSS_DIR) generator.env.globals["MEWS_CSS"] = f"{css_dir}/{CSS_NAME}" @@ -74,7 +86,7 @@ def _on_generator_init(generator) -> None: def _on_content_written(path, **_) -> None: file = Path(path) - if file.suffix != XHTML_SUFFIX: + if file.suffix != _suffix(): return try: tree, doctype = parse_page(file.read_bytes()) diff --git a/pelican-mews/tests/test_mews.py b/pelican-mews/tests/test_mews.py index 7267aa0..3b22e83 100644 --- a/pelican-mews/tests/test_mews.py +++ b/pelican-mews/tests/test_mews.py @@ -28,6 +28,44 @@ def test_written_page_matches_expected(tmp_path, case): assert normalize(page) == normalize(expected) +def test_a_site_can_choose_its_own_suffix(tmp_path): + """A site that is Mews throughout writes .html, so servers send text/html.""" + output = build_case( + "basic", + tmp_path, + MEWS_SUFFIX=".html", + PAGE_SAVE_AS="{slug}.html", + PAGE_URL="{slug}.html", + # With .html the plugin sees every page Pelican writes, including the + # index, tags and archives it falls back to when the theme has no + # template for them. Those fallbacks are HTML, not XML, so a site + # using .html has to give every direct template a Mews template or + # switch the lot off, as here. + DIRECT_TEMPLATES=[], + ) + page = (output / "page.html").read_bytes() + assert normalize(page) == normalize( + (FIXTURES / "basic" / "expected.xhtml").read_bytes() + ) + + +def test_the_default_suffix_leaves_other_html_alone(tmp_path): + """Left at its default the plugin must not touch a plain HTML page.""" + output = build_case( + "strip", + tmp_path, + PAGE_SAVE_AS="{slug}.html", + PAGE_URL="{slug}.html", + INDEX_SAVE_AS="", + ) + # The strip fixture carries a , a class and a comment, all three of + # which the plugin removes. Untouched, every one of them survives. + written = (output / "page.html").read_bytes() + assert b"