Files
zensical/python/tests/unit/extensions/test_autorefs.py
T
2026-09-24 15:24:39 +00:00

426 lines
14 KiB
Python
Vendored

# Copyright (c) 2025-2026 Zensical and contributors
# SPDX-License-Identifier: MIT
# All contributions are certified under the DCO
# Permission is hereby granted, free of charge, to any person obtaining a copy
# of this software and associated documentation files (the "Software"), to
# deal in the Software without restriction, including without limitation the
# rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
# sell copies of the Software, and to permit persons to whom the Software is
# furnished to do so, subject to the following conditions:
# The above copyright notice and this permission notice shall be included in
# all copies or substantial portions of the Software.
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
# FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT. IN NO EVENT SHALL THE
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
# FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
# IN THE SOFTWARE.
from __future__ import annotations
from typing import TYPE_CHECKING, Any
import pytest
from tests.unit.extensions.conftest import soup
from zensical.extensions.autorefs import (
get_autorefs_inventory_data,
get_autorefs_page_data,
get_autorefs_store,
reset,
)
from zensical.extensions.context import Page
if TYPE_CHECKING:
from collections.abc import Generator
from markdown import Markdown
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _autorefs(exts: dict[str, dict[str, Any]] | None = None) -> dict[str, Any]:
"""Return md fixture params with AutorefsExtension configured."""
return {
"config": {
"markdown_extensions": {
**(exts or {}),
"zensical.extensions.autorefs": {},
}
}
}
def _autorefs_backlinks() -> dict[str, Any]:
"""Return md fixture params with backlink recording enabled."""
param = _autorefs({"attr_list": {}, "toc": {}})
markdown_extensions = param["config"]["markdown_extensions"]
markdown_extensions["zensical.extensions.autorefs"] = {
"record_backlinks": True
}
return param
# ---------------------------------------------------------------------------
# Fixtures
# ---------------------------------------------------------------------------
@pytest.fixture(autouse=True)
def _reset_autorefs_store() -> Generator[None, None, None]:
"""Reset the global AutorefsStore around each test."""
reset()
yield
reset()
# ---------------------------------------------------------------------------
# Store
# ---------------------------------------------------------------------------
class TestStore:
"""Tests for page-local fact extraction from the transient store."""
def test_page_data_is_taken_without_consuming_inventory(self) -> None:
"""Page registrations leave the global inventory available."""
store = get_autorefs_store()
page = Page(url="guide/", path="guide.md", meta={})
store.set_page(page)
store.register_anchor(page, "target", title="Target")
store.register_anchor(page, "alias", anchor="target", primary=False)
store.register_url("external", "https://example.com/external")
assert get_autorefs_page_data("guide/") == {
"primary": {"target": ["guide/#target"]},
"secondary": {"alias": ["guide/#target"]},
"titles": {"guide/#target": "Target"},
}
assert get_autorefs_page_data("guide/") == {
"primary": {},
"secondary": {},
"titles": {},
}
assert get_autorefs_inventory_data() == {
"external": "https://example.com/external"
}
def test_urls_are_ordered_and_unique(self) -> None:
"""Candidate URLs preserve registration order without duplicates."""
store = get_autorefs_store()
page = Page("page", "page.html")
store.register_anchor(page, "identifier", "first")
store.register_anchor(page, "identifier", "second")
store.register_anchor(page, "identifier", "first")
assert get_autorefs_page_data("page")["primary"] == {
"identifier": ["page#first", "page#second"]
}
def test_page_registrations_remove_urls(self) -> None:
"""Reprocessing a page removes its previously registered URLs."""
store = get_autorefs_store()
first_page = Page("first", "first.html")
second_page = Page("second", "second.html")
store.register_anchor(first_page, "identifier", "anchor")
store.register_anchor(second_page, "identifier", "anchor")
store.set_page(first_page)
assert get_autorefs_page_data("first")["primary"] == {}
assert get_autorefs_page_data("second")["primary"] == {
"identifier": ["second#anchor"]
}
# ---------------------------------------------------------------------------
# Inline processor
# ---------------------------------------------------------------------------
class TestInlineProcessor:
"""Tests for AutorefsInlineProcessor.
The extension converts unresolved Markdown reference-style links into
`<autoref identifier="...">` elements that the Rust side resolves later.
"""
@pytest.mark.parametrize(
"md",
[pytest.param(_autorefs(), id="default")],
indirect=["md"],
)
def test_implicit_reference(self, md: Markdown) -> None:
"""`[Foo][]` produces an `<autoref>` element with identifier="Foo"."""
html = soup(md.convert("[Foo][]"))
autoref = html.find("autoref")
assert autoref is not None
assert autoref["identifier"] == "Foo"
@pytest.mark.parametrize(
"md",
[pytest.param(_autorefs(), id="default")],
indirect=["md"],
)
def test_explicit_reference_with_formatted_text(self, md: Markdown) -> None:
"""`[**Foo**][Foo]` wraps the bold text inside the autoref element."""
html = soup(md.convert("[**Foo**][Foo]"))
autoref = html.find("autoref")
assert autoref is not None
assert autoref["identifier"] == "Foo"
strong = autoref.find("strong")
assert strong is not None
assert strong.get_text() == "Foo"
@pytest.mark.parametrize(
"md",
[pytest.param(_autorefs(), id="default")],
indirect=["md"],
)
def test_implicit_backtick_reference_is_exact(self, md: Markdown) -> None:
"""``[`Foo`][]`` uses the code content as exact identifier (no slug)."""
html = soup(md.convert("[`Foo`][]"))
autoref = html.find("autoref")
assert autoref is not None
assert autoref["identifier"] == "Foo"
code = autoref.find("code")
assert code is not None
assert code.get_text() == "Foo"
assert "slug" not in autoref.attrs
@pytest.mark.parametrize(
"md",
[
pytest.param(
_autorefs(
{"pymdownx.highlight": {}, "pymdownx.inlinehilite": {}}
),
id="with_inlinehilite",
)
],
indirect=["md"],
)
def test_implicit_code_inlinehilite_plain_is_exact(
self, md: Markdown
) -> None:
"""``[`pathlib.Path`][]`` with inlinehilite keep exact identifier."""
html = soup(md.convert("[`pathlib.Path`][]"))
autoref = html.find("autoref")
assert autoref is not None
assert autoref["identifier"] == "pathlib.Path"
assert "slug" not in autoref.attrs
@pytest.mark.parametrize(
"md",
[
pytest.param(
_autorefs(
{
"pymdownx.highlight": {},
"pymdownx.inlinehilite": {"style_plain_text": "python"},
}
),
id="with_inlinehilite_python_styled",
)
],
indirect=["md"],
)
def test_implicit_code_inlinehilite_styled_is_exact(
self, md: Markdown
) -> None:
"""``[`pathlib.Path`][]`` with inlinehilite is still exact."""
html = soup(md.convert("[`pathlib.Path`][]"))
autoref = html.find("autoref")
assert autoref is not None
assert autoref["identifier"] == "pathlib.Path"
assert "slug" not in autoref.attrs
@pytest.mark.parametrize(
"md",
[pytest.param(_autorefs(), id="default")],
indirect=["md"],
)
def test_reference_inside_code_not_converted(self, md: Markdown) -> None:
"""`` `[Foo][]` `` is not converted into an autoref."""
html = soup(md.convert("`[Foo][]`"))
assert html.find("autoref") is None
code = html.find("code")
assert code is not None
assert code.get_text() == "[Foo][]"
@pytest.mark.parametrize(
"md",
[pytest.param(_autorefs(), id="default")],
indirect=["md"],
)
def test_multiline_reference_uses_explicit_identifier(
self, md: Markdown
) -> None:
"""References spanning two lines use explicit identifiers (no slug)."""
html = soup(md.convert("[Foo\nbar][foo-bar]"))
autoref = html.find("autoref")
assert autoref is not None
assert autoref["identifier"] == "foo-bar"
assert "slug" not in autoref.attrs
@pytest.mark.parametrize(
"md",
[pytest.param(_autorefs(), id="default")],
indirect=["md"],
)
def test_implicit_reference_with_space_is_slugified(
self, md: Markdown
) -> None:
"""`[Foo bar][]` uses the text as identifier and adds a slug."""
html = soup(md.convert("[Foo bar][]"))
autoref = html.find("autoref")
assert autoref is not None
assert autoref["identifier"] == "Foo bar"
assert autoref["slug"] == "foo-bar"
@pytest.mark.parametrize(
("md", "markdown_ref", "exact_expected"),
[
pytest.param(_autorefs(), "[Foo][]", False, id="bare_implicit"),
pytest.param(
_autorefs(), "[\\`Foo][]", False, id="escaped_backtick_start"
),
pytest.param(
_autorefs(),
"[\\`\\`Foo][]",
False,
id="two_escaped_backticks_start",
),
pytest.param(
_autorefs(),
"[\\`\\`Foo\\`][]",
False,
id="mixed_escaped_backticks",
),
pytest.param(
_autorefs(), "[Foo\\`][]", False, id="escaped_backtick_end"
),
pytest.param(
_autorefs(),
"[Foo\\`\\`][]",
False,
id="two_escaped_backticks_end",
),
pytest.param(
_autorefs(),
"[\\`Foo\\`\\`][]",
False,
id="outer_escaped_backticks",
),
pytest.param(
_autorefs(),
"[`Foo` `Bar`][]",
False,
id="two_separate_code_spans",
),
pytest.param(
_autorefs(), "[Foo][Foo]", True, id="explicit_identifier"
),
pytest.param(_autorefs(), "[`Foo`][]", True, id="single_code_span"),
pytest.param(
_autorefs(),
"[`Foo``Bar`][]",
True,
id="code_span_two_backticks",
),
pytest.param(
_autorefs(),
"[`Foo```Bar`][]",
True,
id="code_span_three_backticks",
),
pytest.param(
_autorefs(),
"[``Foo```Bar``][]",
True,
id="double_backtick_three",
),
pytest.param(
_autorefs(),
"[``Foo`Bar``][]",
True,
id="double_backtick_single",
),
pytest.param(
_autorefs(),
"[```Foo``Bar```][]",
True,
id="triple_backtick_double",
),
],
indirect=["md"],
)
def test_mark_identifiers_as_exact(
self, md: Markdown, markdown_ref: str, exact_expected: bool
) -> None:
"""Code/explicit identifiers have no slug; bare-text identifiers do."""
html = soup(md.convert(markdown_ref))
autoref = html.find("autoref")
assert autoref is not None
if exact_expected:
assert "slug" not in autoref.attrs
else:
assert "slug" in autoref.attrs
# ---------------------------------------------------------------------------
# Rust HTML visitor boundary
# ---------------------------------------------------------------------------
class TestTreeprocessorBoundary:
"""Autorefs leaves post-Markdown HTML processing to Rust."""
@pytest.mark.parametrize(
"md",
[
pytest.param(
_autorefs({"attr_list": {}, "toc": {}}),
id="with_heading_ids",
)
],
indirect=["md"],
)
def test_does_not_register_treeprocessors(self, md: Markdown) -> None:
"""The Python extension only retains its inline processor."""
assert "autorefs-anchors" not in md.treeprocessors
assert "autorefs-headings" not in md.treeprocessors
assert "mkdocs-autorefs-backlinks" not in md.treeprocessors
assert "data-zensical-autoref" not in md.convert("[Foo][foo]")
@pytest.mark.parametrize(
"md",
[pytest.param(_autorefs_backlinks(), id="with_backlinks")],
indirect=["md"],
)
def test_retains_constant_time_mkdocstrings_context_bridge(
self, md: Markdown
) -> None:
"""Nested Markdown exposes its object ID without scanning in Python."""
processor = md.treeprocessors["mkdocs-autorefs-backlinks"]
processor.initial_id = "object-id"
html = md.convert("[Foo][foo]")
assert (
"<!--zensical:autoref-context:start:6f626a6563742d6964-->" in html
)
assert "<!--zensical:autoref-context:end-->" in html
assert "data-zensical-autoref" in html
assert "backlink-type" not in html
assert "backlink-anchor" not in html