From cf68f6d84dfc03645245a7b25ed9cac8b28d1903 Mon Sep 17 00:00:00 2001 From: Martin Donath Date: Thu, 24 Sep 2026 17:44:41 +0200 Subject: [PATCH] fix: resolve source links to published post URLs (#966) Signed-off-by: squidfunk --- crates/zensical/src/compat/mkdocs/html.rs | 45 ++++++++++- .../zensical/src/compat/mkdocs/plugin/blog.rs | 5 ++ crates/zensical/src/workflow.rs | 80 ++++++++++++++++++- python/tests/integration/test_blog.py | 68 ++++++++++++++++ 4 files changed, 196 insertions(+), 2 deletions(-) diff --git a/crates/zensical/src/compat/mkdocs/html.rs b/crates/zensical/src/compat/mkdocs/html.rs index 0931656..bf6f7dd 100644 --- a/crates/zensical/src/compat/mkdocs/html.rs +++ b/crates/zensical/src/compat/mkdocs/html.rs @@ -27,7 +27,7 @@ use html5gum::emitters::callback::{CallbackEmitter, CallbackEvent}; use html5gum::{Span, Tokenizer}; -use std::collections::HashMap; +use std::collections::{HashMap, HashSet}; use std::convert::Infallible; use std::ops::Range; @@ -84,6 +84,16 @@ struct RewriteUrls<'a> { attribute: bool, } +/// Collects local URLs addressed by rendered page content. +struct LocalTargets<'a> { + /// Route against which relative URLs are resolved. + base: &'a str, + /// Whether the tokenizer is reading a local link or media target. + attribute: bool, + /// Resolved URLs without query parameters or fragments. + targets: HashSet, +} + /// One replacement in the original HTML input. #[derive(Debug, PartialEq, Eq)] struct Edit { @@ -339,6 +349,28 @@ impl Visitor for RewriteUrls<'_> { } } +impl Visitor for LocalTargets<'_> { + fn visit( + &mut self, event: &CallbackEvent<'_>, _span: Span, + _editor: &mut Editor<'_>, + ) { + match event { + CallbackEvent::OpenStartTag { .. } => self.attribute = false, + CallbackEvent::AttributeName { name } => { + self.attribute = matches!(*name, b"href" | b"src"); + } + CallbackEvent::AttributeValue { value } if self.attribute => { + let value = String::from_utf8_lossy(value); + if let Some(target) = url::resolve(self.base, &value) { + let end = target.find(['?', '#']).unwrap_or(target.len()); + self.targets.insert(target[..end].to_owned()); + } + } + _ => {} + } + } +} + // ---------------------------------------------------------------------------- // Functions // ---------------------------------------------------------------------------- @@ -395,6 +427,17 @@ pub fn rewrite_urls( scan(input, &mut [&mut visitor]) } +/// Returns the local URL paths referenced by a page's rendered HTML. +pub fn local_targets(input: &str, base: &str) -> HashSet { + let mut visitor = LocalTargets { + base, + attribute: false, + targets: HashSet::new(), + }; + scan(input, &mut [&mut visitor]); + visitor.targets +} + /// Returns whether a byte is HTML whitespace. fn is_whitespace(byte: u8) -> bool { matches!(byte, b'\t' | b'\n' | 0x0c | b'\r' | b' ') diff --git a/crates/zensical/src/compat/mkdocs/plugin/blog.rs b/crates/zensical/src/compat/mkdocs/plugin/blog.rs index c00415b..d0b72fc 100644 --- a/crates/zensical/src/compat/mkdocs/plugin/blog.rs +++ b/crates/zensical/src/compat/mkdocs/plugin/blog.rs @@ -243,6 +243,11 @@ impl Blog { } } + /// Returns whether no native blog instances are enabled. + pub fn is_empty(&self) -> bool { + self.instances.is_empty() + } + /// Routes posts before Markdown rendering and preserves ordinary pages. pub fn setup(&self, dependencies: Dependencies<'_>) -> Output { let blog = self.clone(); diff --git a/crates/zensical/src/workflow.rs b/crates/zensical/src/workflow.rs index b630314..aba9feb 100644 --- a/crates/zensical/src/workflow.rs +++ b/crates/zensical/src/workflow.rs @@ -25,9 +25,10 @@ //! Workflow definitions. +use percent_encoding::percent_decode_str; use regex::Regex; use serde::{Deserialize, Serialize}; -use std::collections::BTreeMap; +use std::collections::{BTreeMap, HashMap}; use std::fs; use std::hash::{DefaultHasher, Hash, Hasher}; use std::ops::Deref; @@ -226,6 +227,19 @@ struct RenderedPage { impl Value for RenderedPage {} +/// A source page's ordinary route and its configured published route. +#[derive(Clone, Debug, PartialEq, Eq)] +struct RoutedLink { + /// Route produced by ordinary Markdown source link resolution. + source: String, + /// Decoded spelling used by unescaped Markdown source links. + decoded: String, + /// Route where the page is actually published. + published: String, +} + +impl Value for RoutedLink {} + // ---------------------------------------------------------------------------- // ---------------------------------------------------------------------------- @@ -314,6 +328,11 @@ impl Main { // once the complete navigation is available. let rendered_page = apply_navigation_titles(&provisional, &resolution); let rendered_page = apply_blog(&rendered_page, &blog_patches); + let rendered_page = if blogs.is_empty() { + rendered_page + } else { + apply_routed_links(&self.config, &rendered_page, &markdown) + }; let rendered_page = apply_tags(&plugins.tags, &rendered_page); let autorefs = if let Some(autorefs) = autorefs { autorefs @@ -408,6 +427,65 @@ fn apply_blog( ) } +/// Resolve source-file links after all pages have their published routes. +fn apply_routed_links( + config: &Config, pages: &Stream, + routed: &Stream, +) -> Stream { + let config = config.clone(); + let relocations = routed.filter_map(move |descriptor: &PageDescriptor| { + if !matches!(&descriptor.origin, PageOrigin::Source(_)) { + return Ok(None); + } + let source = PageRoute::from_source( + &config, + descriptor.document.source.clone(), + )? + .url; + let published = descriptor.route.url.clone(); + let decoded = + percent_decode_str(&source).decode_utf8_lossy().into_owned(); + Ok::<_, PathError>((source != published).then_some(RoutedLink { + source, + decoded, + published, + })) + }); + let selected = relocations.select(pages, |rendered| { + let targets = + html::local_targets(&rendered.page.content, &rendered.page.url); + move |route: &RoutedLink| { + targets.contains(&route.source) || targets.contains(&route.decoded) + } + }); + (pages.clone(), selected).join().map( + |(rendered, routes): &(RenderedPage, Vec<(Key, RoutedLink)>)| { + let mut rendered = rendered.clone(); + let mut mappings = HashMap::new(); + for (_, route) in routes { + mappings + .entry(route.decoded.clone()) + .or_insert_with(|| route.published.clone()); + } + for (_, route) in routes { + mappings.insert(route.source.clone(), route.published.clone()); + } + if let Some(content) = html::rewrite_urls( + &rendered.page.content, + &rendered.page.url, + &mappings, + ) { + rendered.page.apply_derived( + Some(content), + None, + BTreeMap::new(), + ); + } + rendered + }, + ) +} + fn resolve_navigation( config: &Config, strict: bool, blogs: &blog::Blog, sources: &Stream, navigation_pages: &Stream, diff --git a/python/tests/integration/test_blog.py b/python/tests/integration/test_blog.py index b12c84c..33f21ba 100644 --- a/python/tests/integration/test_blog.py +++ b/python/tests/integration/test_blog.py @@ -139,6 +139,65 @@ def test_posts_are_routed_from_dates_and_native_unicode_slugs( assert not (tmp_path / "site" / "blog" / "posts" / "hello").exists() +@pytest.mark.parametrize( + ("source_name", "link_name"), + [ + ("first.md", "first.md"), + ("café.md", "café.md"), + ("café.md", "caf%C3%A9.md"), + ], +) +def test_source_links_follow_published_post_routes( + tmp_path: Path, source_name: str, link_name: str +) -> None: + config = _project(tmp_path) + (tmp_path / "docs" / "index.md").write_text( + f"[First](blog/posts/{link_name}#details)\n", encoding="utf-8" + ) + _post( + tmp_path, + source_name, + "First post", + "2024-09-01", + body="## Details\n\nBody.", + ) + _post( + tmp_path, + "second.md", + "Second post", + "2024-09-02", + body=f"[First]({link_name}#details)", + ) + + zensical.build(str(config), {"clean": False, "strict": True}) + + site = tmp_path / "site" + home = (site / "index.html").read_text("utf-8") + second = site.joinpath( + "blog", "2024", "09", "02", "second-post", "index.html" + ).read_text("utf-8") + assert 'href="blog/2024/09/01/first-post/#details"' in home + assert 'href="../../01/first-post/#details"' in second + assert "blog/posts/first/" not in home + second + + _post( + tmp_path, + source_name, + "Renamed post", + "2024-08-31", + body="## Details\n\nBody.", + ) + zensical.build(str(config), {"clean": False, "strict": True}) + + home = (site / "index.html").read_text("utf-8") + second = site.joinpath( + "blog", "2024", "09", "02", "second-post", "index.html" + ).read_text("utf-8") + assert 'href="blog/2024/08/31/renamed-post/#details"' in home + assert 'href="../../../08/31/renamed-post/#details"' in second + assert "first-post/" not in home + second + + def test_post_anchor_aliases_survive_routing(tmp_path: Path) -> None: """An empty Markdown link still aliases the next heading after routing.""" config = _project(tmp_path) @@ -243,6 +302,9 @@ def test_single_page_keeps_empty_pagination_context(tmp_path: Path) -> None: def test_serve_reconciles_routes_views_and_pagination(tmp_path: Path) -> None: """Retained blog revisions retract every superseded output.""" config = _project(tmp_path, per_page=1, archive=True, categories=True) + (tmp_path / "docs" / "index.md").write_text( + "[One](blog/posts/one.md)\n", encoding="utf-8" + ) with config.open("a", encoding="utf-8") as stream: stream.write("dev_addr: 127.0.0.1:0\n") _post( @@ -258,6 +320,7 @@ def test_serve_reconciles_routes_views_and_pagination(tmp_path: Path) -> None: "Two", "2026-09-02", categories=["Alpha"], + body="[One](one.md)", ) log = (tmp_path / "serve.log").open("w+", encoding="utf-8") process = subprocess.Popen( # noqa: S603 @@ -274,6 +337,7 @@ def test_serve_reconciles_routes_views_and_pagination(tmp_path: Path) -> None: stderr=subprocess.STDOUT, ) site = tmp_path / "site" / "blog" + home = tmp_path / "site" / "index.html" index = site / "index.html" second = site / "page" / "2" / "index.html" one = site / "2026" / "09" / "01" / "one" / "index.html" @@ -314,6 +378,8 @@ def test_serve_reconciles_routes_views_and_pagination(tmp_path: Path) -> None: and one.is_file() and contains(index, "Two:") and contains(second, "One:") + and contains(home, 'href="blog/2026/09/01/one/"') + and contains(two, 'href="../../01/one/"') ) ) _post( @@ -330,6 +396,8 @@ def test_serve_reconciles_routes_views_and_pagination(tmp_path: Path) -> None: and beta.is_file() and contains(index, "Moved:") and contains(second, "Two:") + and contains(home, 'href="blog/2026/09/03/moved/"') + and contains(two, 'href="../../03/moved/"') ) ) (tmp_path / "docs" / "blog" / "posts" / "two.md").unlink()