# Copyright (c) 2025-2026 Zensical and contributors # SPDX-License-Identifier: MIT # All contributions are certified under the DCO # Permission is hereby granted, free of charge, to any person obtaining a copy # of this software and associated documentation files (the "Software"), to # deal in the Software without restriction, including without limitation the # rights to use, copy, modify, merge, publish, distribute, sublicense, and/or # sell copies of the Software, and to permit persons to whom the Software is # furnished to do so, subject to the following conditions: # The above copyright notice and this permission notice shall be included in # all copies or substantial portions of the Software. # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, # FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT. IN NO EVENT SHALL THE # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING # FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS # IN THE SOFTWARE. """LLM text artifacts from the rendered page pipeline.""" from __future__ import annotations import subprocess import sys import time from textwrap import dedent from typing import TYPE_CHECKING import pytest from bs4 import BeautifulSoup import zensical if TYPE_CHECKING: from pathlib import Path def build(root: Path, options: str, *, strict: bool = False) -> Path: """Build the supplied project while retaining caches between builds.""" config = root / "mkdocs.yml" config.write_text( dedent("""\ site_name: LLM test site_url: https://example.org/docs/ """) + options, encoding="utf-8", ) zensical.build(str(config), {"clean": False, "strict": strict}) return root / "site" @pytest.mark.parametrize("directory_urls", [True, False]) @pytest.mark.parametrize("enabled", [True, False]) def test_exposes_markdown_urls_only_for_exported_pages( tmp_path: Path, directory_urls: bool, enabled: bool ) -> None: # An encoded filename and a different export host # are used under a site subpath. docs = tmp_path / "docs" (docs / "guide").mkdir(parents=True) (docs / "index.md").write_text("# Home\n", encoding="utf-8") (docs / "guide" / "a # café.md").write_text("# Guide\n", encoding="utf-8") (docs / "excluded.md").write_text("# Excluded\n", encoding="utf-8") overrides = tmp_path / "overrides" overrides.mkdir() (overrides / "main.html").write_text( '{% if page.markdown_url %}Markdown{% endif %}', encoding="utf-8", ) config = dedent(f"""\ use_directory_urls: {str(directory_urls).lower()} theme: custom_dir: overrides plugins: - llmstxt: enabled: {str(enabled).lower()} base_url: https://cdn.example.org/exports/ sections: {{Pages: [index.md, "guide/*.md"]}} """) guide = ( "guide/a # café/index.html" if directory_urls else "guide/a # café.html" ) excluded = "excluded/index.html" if directory_urls else "excluded.html" guide_url = ( "guide/a%20%23%20caf%C3%A9/index.md" if directory_urls else "guide/a%20%23%20caf%C3%A9.md" ) guide_href = "./index.md" if directory_urls else "./a%20%23%20caf%C3%A9.md" # The same URLs must be exposed on a fresh build and a cached build. for _ in range(2): site = build(tmp_path, config) for output, url, href in ( ("index.html", "index.md", "./index.md"), (guide, guide_url, guide_href), ): page = BeautifulSoup( (site / output).read_text(encoding="utf-8"), "html.parser" ) link = page.find("a") if enabled: assert link is not None assert link["data-url"] == url assert link["href"] == href assert (site / output).with_suffix(".md").is_file() else: assert link is None assert " None: # The post is published at a date-based route instead of its source path. docs = tmp_path / "docs" posts = docs / "blog" / "posts" posts.mkdir(parents=True) (docs / "blog" / "index.md").write_text("# Blog\n", encoding="utf-8") (posts / "entry.md").write_text( "---\ndate: 2026-09-03\n---\n# Hello\n", encoding="utf-8" ) overrides = tmp_path / "overrides" overrides.mkdir() (overrides / "blog-post.html").write_text( "{{ page.markdown_url }}", encoding="utf-8" ) site = build( tmp_path, dedent("""\ theme: custom_dir: overrides plugins: - material/blog: archive: false categories: false - llmstxt: sections: {Posts: ["blog/posts/*.md"]} """), ) post = site / "blog/2026/09/03/hello/index.html" assert post.read_text(encoding="utf-8") == "blog/2026/09/03/hello/index.md" assert post.with_suffix(".md").read_text(encoding="utf-8") == "# Hello\n" @pytest.mark.parametrize("directory_urls", [True, False]) def test_emits_selected_pages_in_section_order_and_uses_final_titles( tmp_path: Path, directory_urls: bool ) -> None: docs = tmp_path / "docs" (docs / "guide" / "nested").mkdir(parents=True) (docs / "index.md").write_text("# Home\n\nWelcome.\n", encoding="utf-8") (docs / "guide" / "a.md").write_text("# First\n\nOne.\n", encoding="utf-8") (docs / "guide" / "nested" / "b.md").write_text( "# Second\n\nTwo.\n", encoding="utf-8" ) (docs / "excluded.md").write_text("# Excluded\n", encoding="utf-8") config = dedent(f"""\ use_directory_urls: {str(directory_urls).lower()} site_description: A short description. nav: - Navigation title: index.md - guide/a.md - guide/nested/b.md plugins: - llmstxt: base_url: https://cdn.example.org/v2 markdown_description: Long **description**. full_output: ai/full.txt sections: Z first: - index.md: Home page - guide/*.md: Guide page - guide/a.md: Updated description A second: - guide/a.md """) site = build(tmp_path, config) first = "guide/a/index.md" if directory_urls else "guide/a.md" second = ( "guide/nested/b/index.md" if directory_urls else "guide/nested/b.md" ) index = (site / "llms.txt").read_text(encoding="utf-8") assert index == dedent(f"""\ # LLM test > A short description. Long **description**. ## Z first - [Navigation title](): Home page - [First](): Updated description - [Second](): Guide page ## A second - [First]() """) assert (site / first).read_text(encoding="utf-8") == "# First\n\nOne.\n" assert (site / second).read_text(encoding="utf-8") == "# Second\n\nTwo.\n" assert not (site / "excluded" / "index.md").exists() assert not (site / "excluded.md").exists() full = (site / "ai" / "full.txt").read_text(encoding="utf-8") assert full.count("# First\n") == 2 assert full.index("# Z first") < full.index("# A second") # A cached build must still emit the page Markdown and both aggregate files. (site / first).unlink() (site / "llms.txt").unlink() (site / "ai" / "full.txt").unlink() build(tmp_path, config) assert (site / first).read_text(encoding="utf-8") == "# First\n\nOne.\n" assert (site / "llms.txt").read_text(encoding="utf-8") == index assert (site / "ai" / "full.txt").read_text(encoding="utf-8") == full def test_rebuilds_gfm_from_generated_html_with_search_disabled( tmp_path: Path, ) -> None: docs = tmp_path / "docs" docs.mkdir() (docs / "index.md").write_text( dedent("""\ # {{ config.site_name }} !!! note "A note" Generated **content**. ```python value = "```" ``` | Name | Value | | ---- | ----- | | one | two | - Parent - Child - [x] Done - [ ] Todo Old content with a note.[^example] [^example]: A **formatted** footnote. """), encoding="utf-8", ) site = build( tmp_path, dedent("""\ plugins: - search: {enabled: false} - macros - llmstxt: sections: {Pages: [index.md]} """), ) markdown = (site / "index.md").read_text(encoding="utf-8") assert markdown.startswith("# LLM test\n") assert ( dedent("""\ > [!NOTE] > > **A note** > > Generated **content**. """) in markdown ) assert '````python\nvalue = "```"\n````' in markdown assert ( dedent("""\ | Name | Value | | --- | --- | | one | two | """) in markdown ) assert "- Parent\n\n - Child" in markdown assert "- [x] Done\n- [ ] Todo" in markdown assert "~~Old content~~ with a note.[^/docs/#example]\n" in markdown assert "[^/docs/#example]: A **formatted** footnote." in markdown assert "footnote-backref" not in markdown assert "headerlink" not in markdown assert "{{" not in markdown assert "!!!" not in markdown assert not (site / "llms-full.txt").exists() @pytest.mark.parametrize("autoclean", [True, False]) def test_converts_admonitions_details_and_callouts_to_github_alerts( tmp_path: Path, autoclean: bool ) -> None: docs = tmp_path / "docs" docs.mkdir() (docs / "index.md").write_text( dedent("""\ # Alerts !!! warning "Before you start" Save your **work**. ??? tip "Try `fast=True`" Use the faster mode. ???+ danger This deletes data. !!! important Read the instructions. !!! caution "" Handle with care. !!! custom "Project policy" Follow these rules. > [!NOTE] From a callout > The body remains readable. """), encoding="utf-8", ) config = dedent(f"""\ plugins: - callouts - llmstxt: autoclean: {str(autoclean).lower()} sections: Pages: [index.md] """) site = build(tmp_path, config) markdown = (site / "index.md").read_text(encoding="utf-8") assert ( dedent("""\ > [!WARNING] > > **Before you start** > > Save your **work**. > [!TIP] > > **Try `fast=True`** > > Use the faster mode. > [!CAUTION] > > **Danger** > > This deletes data. > [!IMPORTANT] > > Read the instructions. > [!CAUTION] > > Handle with care. > [!NOTE] > > **Project policy** > > Follow these rules. > [!NOTE] > > **From a callout** > > The body remains readable. """) in markdown ) # The HTML site keeps its original admonitions and disclosure controls. html = (site / "index.html").read_text(encoding="utf-8") assert '
' in html assert '
' in html assert '
' in html @pytest.mark.parametrize("autoclean", [True, False]) def test_omits_mkdocstrings_source_and_empty_icons_from_llmstxt( tmp_path: Path, autoclean: bool ) -> None: docs = tmp_path / "docs" docs.mkdir() source = tmp_path / "src" source.mkdir() (source / "sample_api.py").write_text( dedent('''\ def greet(name: str) -> str: """Return a greeting. Examples: ```python greet("reader") ``` """ return f"Hello, {name}!" '''), encoding="utf-8", ) (docs / "index.md").write_text( dedent("""\ # API ::: sample_api.greet ## Example ```python from sample_api import greet print(greet("reader")) ``` """), encoding="utf-8", ) config = dedent(f"""\ plugins: - mkdocstrings: handlers: python: paths: [src] options: docstring_style: google heading_level: 2 show_root_heading: true show_root_full_path: false show_symbol_type_heading: true separate_signature: true show_signature_annotations: true show_source: true - llmstxt: autoclean: {str(autoclean).lower()} full_output: llms-full.txt sections: Pages: [index.md] """) site = build(tmp_path, config) # Both LLM outputs keep the API documentation and examples # without the implementation (source code). for name in ("index.md", "llms-full.txt"): markdown = (site / name).read_text(encoding="utf-8") assert "## greet" in markdown assert markdown.count("Return a greeting.") == 1 assert "greet(name: str) -> str" in markdown assert ( dedent("""\ ```python greet("reader") ``` """) in markdown ) assert ( dedent("""\ ```python from sample_api import greet print(greet("reader")) ``` """) in markdown ) assert "" not in markdown assert "Source code in" not in markdown assert "Hello," not in markdown # The website still includes the source listing and heading icon. html = (site / "index.html").read_text(encoding="utf-8") assert 'class="mkdocstrings-source"' in html assert ( 'class="doc-symbol doc-symbol-heading doc-symbol-function">' in html ) assert "Source code in" in html assert "Hello," in html def test_footnotes_remain_distinct_in_aggregated_output(tmp_path: Path) -> None: docs = tmp_path / "docs" docs.mkdir() for name, title in (("index", "Home"), ("second", "Second")): (docs / f"{name}.md").write_text( dedent(f"""\ # {title} A note.[^same] [^same]: Footnote from {title}. """), encoding="utf-8", ) site = build( tmp_path, dedent("""\ plugins: - llmstxt: full_output: llms-full.txt sections: {Pages: [index.md, second.md]} """), ) full = (site / "llms-full.txt").read_text(encoding="utf-8") # Each reference must still point to its page's definition # after aggregation. for path, title in (("/docs/", "Home"), ("/docs/second/", "Second")): assert f"A note.[^{path}#same]\n" in full assert f"[^{path}#same]: Footnote from {title}." in full @pytest.mark.parametrize("strict", [False, True]) def test_startup_waits_for_selected_pages( tmp_path: Path, capfd: pytest.CaptureFixture[str], strict: bool ) -> None: # Explicit paths and a wildcard are selected before any pages are rendered. docs = tmp_path / "docs" (docs / "guide").mkdir(parents=True) sources = { "index.md": "Home", "guide/quickstart.md": "Quickstart", "guide/formatting.md": "Formatting", } for source, title in sources.items(): (docs / source).write_text(f"# {title}\n", encoding="utf-8") config = dedent("""\ plugins: - llmstxt: full_output: llms-full.txt sections: Start: [index.md, guide/quickstart.md] Guides: ["guide/*.md", guide/formatting.md] """) # No missing-page warnings should be emitted on fresh or cached builds. for _ in range(2): site = build(tmp_path, config, strict=strict) assert "llmstxt:" not in capfd.readouterr().err index = (site / "llms.txt").read_text(encoding="utf-8") full = (site / "llms-full.txt").read_text(encoding="utf-8") for title in sources.values(): assert f"[{title}]" in index assert f"# {title}\n" in full def test_empty_site_emits_indexes_without_warnings( tmp_path: Path, capfd: pytest.CaptureFixture[str] ) -> None: # The initial scan is completed even when no documentation pages are found. (tmp_path / "docs").mkdir() site = build( tmp_path, dedent("""\ plugins: - llmstxt: full_output: llms-full.txt sections: {Pages: ["*.md"]} """), strict=True, ) index = (site / "llms.txt").read_text(encoding="utf-8") full = (site / "llms-full.txt").read_text(encoding="utf-8") assert "llmstxt:" not in capfd.readouterr().err assert index == "# LLM test\n\n## Pages\n" assert full == "# LLM test\n\n# Pages\n" @pytest.mark.parametrize("has_pages", [False, True]) def test_missing_explicit_pages_warn_and_strict_builds_fail( tmp_path: Path, capfd: pytest.CaptureFixture[str], has_pages: bool ) -> None: docs = tmp_path / "docs" docs.mkdir() if has_pages: (docs / "index.md").write_text("# Home\n", encoding="utf-8") paths = "index.md, missing.md" if has_pages else "missing.md" config = dedent(f"""\ plugins: - llmstxt: sections: {{Pages: [{paths}]}} """) site = build(tmp_path, config) warnings = capfd.readouterr().err assert warnings.count("Page URI 'missing.md' not found") == 1 assert "Page URI 'index.md' not found" not in warnings assert (site / "llms.txt").is_file() assert (site / "index.md").exists() == has_pages with pytest.raises(Exception, match="strict"): build(tmp_path, config, strict=True) def test_ignores_preprocess_without_loading_the_script( tmp_path: Path, capfd: pytest.CaptureFixture[str], caplog: pytest.LogCaptureFixture, ) -> None: docs = tmp_path / "docs" docs.mkdir() (docs / "index.md").write_text( "# Home\n\nOriginal content.\n", encoding="utf-8" ) # Importing this script would fail the build, even without calling its hook. (tmp_path / "preprocess.py").write_text( 'raise RuntimeError("Preprocess scripts must not be loaded")\n', encoding="utf-8", ) config = dedent("""\ plugins: - llmstxt: preprocess: preprocess.py full_output: llms-full.txt sections: {Pages: ["*.md"]} """) site = build(tmp_path, config, strict=True) assert (site / "index.md").read_text( encoding="utf-8" ) == "# Home\n\nOriginal content.\n" assert "Original content." in (site / "llms-full.txt").read_text( encoding="utf-8" ) captured = capfd.readouterr() assert "preprocess" not in (captured.out + captured.err).lower() assert "preprocess" not in caplog.text.lower() def test_disabled_plugin_emits_no_artifacts(tmp_path: Path) -> None: docs = tmp_path / "docs" docs.mkdir() (docs / "index.md").write_text("# Home\n", encoding="utf-8") site = build( tmp_path, dedent("""\ plugins: - llmstxt: {enabled: false, sections: {Pages: [index.md]}} """), ) assert not (site / "llms.txt").exists() assert not (site / "index.md").exists() def test_fnmatch_patterns_cross_directories_but_keep_their_prefix( tmp_path: Path, ) -> None: docs = tmp_path / "docs" sources = { "guide/direct.md": "Direct", "guide/nested/deep.md": "Deep", "other/guide/wrong.md": "Wrong prefix", "guides/wrong.md": "Wrong directory", } for source, title in sources.items(): file = docs / source file.parent.mkdir(parents=True, exist_ok=True) file.write_text(f"# {title}\n", encoding="utf-8") site = build( tmp_path, dedent("""\ plugins: - llmstxt: sections: All guides: [guide/*.md] Nested guides: [guide/**/*.md] """), ) index = (site / "llms.txt").read_text(encoding="utf-8") assert index.count("[Direct]") == 1 assert index.count("[Deep]") == 2 assert "Wrong" not in index @pytest.mark.parametrize("directory_urls", [True, False]) def test_toml_configuration_and_readable_output_urls( tmp_path: Path, directory_urls: bool ) -> None: docs = tmp_path / "docs" docs.mkdir() (docs / "index.md").write_text( dedent("""\ # Home [Café notes](caf%C3%A9%20notes.md) [Hash in filename](a%20%23%20caf%C3%A9.md) """), encoding="utf-8", ) (docs / "café notes.md").write_text("# Café notes\n", encoding="utf-8") (docs / "a # café.md").write_text("# Heading #\n", encoding="utf-8") config = tmp_path / "zensical.toml" config.write_text( dedent(f"""\ [project] site_name = "LLM test" site_url = "https://example.org/docs/" use_directory_urls = {str(directory_urls).lower()} [project.plugins.llmstxt.sections] Pages = ["index.md", "café notes.md", "a # café.md"] """), encoding="utf-8", ) zensical.build(str(config), {"clean": False, "strict": False}) site = tmp_path / "site" suffix = "/index.md" if directory_urls else ".md" index = (site / "llms.txt").read_text(encoding="utf-8") page = (site / "index.md").read_text(encoding="utf-8") # Keep spaces and Unicode readable in both the index and converted links. page_suffix = "/index.md" if directory_urls else ".html" for markdown, link_suffix in ((index, suffix), (page, page_suffix)): assert ( f"()" in markdown ) # A literal # in the filename must not become a URL fragment. assert ( f"()" in markdown ) assert (site / f"café notes{suffix}").is_file() assert (site / f"a # café{suffix}").is_file() def test_serve_updates_adds_renames_and_removes_markdown_outputs( tmp_path: Path, ) -> None: docs = tmp_path / "docs" docs.mkdir() source = docs / "post.md" source.write_text("# Original\n", encoding="utf-8") config = tmp_path / "mkdocs.yml" config.write_text( dedent("""\ site_name: Live LLM text site_url: https://example.org/ dev_addr: 127.0.0.1:0 plugins: - llmstxt: full_output: llms-full.txt sections: {Pages: [post.md, '*.md']} """), encoding="utf-8", ) site = tmp_path / "site" log_path = tmp_path / "serve.log" log = log_path.open("w+", encoding="utf-8") process = subprocess.Popen( # noqa: S603 # Run this checkout's CLI on the test project. [ sys.executable, "-m", "zensical", "serve", "--config-file", str(config), ], cwd=tmp_path, stdout=log, stderr=subprocess.STDOUT, ) def wait_for(pages: dict[str, str], absent: tuple[str, ...] = ()) -> None: deadline = time.monotonic() + 15 while time.monotonic() < deadline: try: index = (site / "llms.txt").read_text(encoding="utf-8") full = (site / "llms-full.txt").read_text(encoding="utf-8") present = all( f"[{title}]" in index and f"# {title}\n" in full and (site / name / "index.md").read_text(encoding="utf-8") == f"# {title}\n" for name, title in pages.items() ) removed = all( not (site / name / "index.md").exists() and f"/{name}/index.md" not in index for name in absent ) if present and removed and index.count("- [") == len(pages): return except OSError: pass if process.poll() is not None: break time.sleep(0.02) log.flush() log.seek(0) raise AssertionError(f"LLM text outputs did not settle: {log.read()}") try: wait_for({"post": "Original"}) # The explicit source must be found before startup warnings are checked. startup_log = log_path.read_text(encoding="utf-8") assert "llmstxt:" not in startup_log # Body changes update the per-page output and both aggregate files. source.write_text("# Revised\n", encoding="utf-8") wait_for({"post": "Revised"}) added = docs / "added.md" added.write_text("# Added\n", encoding="utf-8") wait_for({"post": "Revised", "added": "Added"}) # A renamed source retracts its previous Markdown destination. renamed = docs / "renamed.md" source.rename(renamed) wait_for({"renamed": "Revised", "added": "Added"}, ("post",)) # The original explicit path is now absent and must be reported. rename_log = log_path.read_text(encoding="utf-8") assert "Page URI 'post.md' not found" in rename_log renamed.unlink() added.unlink() wait_for({}, ("post", "renamed", "added")) finally: if process.poll() is None: process.terminate() try: process.wait(timeout=5) except subprocess.TimeoutExpired: process.kill() process.wait(timeout=5) log.close()