diff --git a/docs/localization.md b/docs/localization.md index 8523a49..8f45b91 100644 --- a/docs/localization.md +++ b/docs/localization.md @@ -88,6 +88,10 @@ Region tags normalize to their base subtag (`fi-FI` → `fi`). ### Rendering - The translated Markdown goes through the same `markdown.render` pipeline. +- Section anchors (`#hash` ids on h1/h2 headings) stay in the original + language: render(anchors_from=...) pins the translated render's heading + ids to the original text's slugs, matched by heading position, so links + to sections don't break across languages. - Navigation/sidebar titles come from the translation's title map, with per-node fallback to the original title (a partially translated tree must still render). diff --git a/pagerite/api.py b/pagerite/api.py index 30c38ab..96865de 100644 --- a/pagerite/api.py +++ b/pagerite/api.py @@ -503,6 +503,14 @@ async def editor_ws(ws: WebSocket) -> None: # The title is injected as h1 when the markdown has # none; the editor's title field edits live-preview. title=msg.get("title") or (node.title if node else ""), + # Pin section anchors to the original language so the + # preview of a translation matches the served page + # (no-op when the previewed markdown is the original). + anchors_from=( + (node_markdown(data, node) or "", node.title) + if node + else None + ), ) await ws.send_json( { diff --git a/pagerite/markdown.py b/pagerite/markdown.py index b331630..af56d6e 100644 --- a/pagerite/markdown.py +++ b/pagerite/markdown.py @@ -402,7 +402,9 @@ def _heading_ids(state) -> None: its self-link is ``href=""`` (back to the top of the page). An author-set `{#id}` always wins; auto ids slugify the heading text (python-slugify, mirroring the editor's slugify.js) and dedupe with - -2/-3 suffixes per render. Headings that already contain a link are + -2/-3 suffixes per render — unless env["anchor_ids"] presets them, as + render(anchors_from=...) does for translated pages so section URLs + stay in the original language. Headings that already contain a link are ``data-line`` records the heading's markdown source line (0-based, after undoing the render(title=...) injection offset via ``env``) — the page editor uses it for section pens and piecewise-linear scroll sync. @@ -443,15 +445,25 @@ def _heading_ids(state) -> None: if len(heads) < ANCHOR_MIN_HEADINGS: return seen: set[str] = set() - for i, token in heads: + preset = state.env.get("anchor_ids") + for k, (i, token) in enumerate(heads): inline = tokens[i + 1] hid = token.attrGet("id") if not isinstance(hid, str) or not hid: - # Slug the visible text, not the raw markdown (`## [a](url)`). - text = "".join( - c.content for c in inline.children if c.type in ("text", "code_inline") - ) - base = slugify(text) or "section" + if preset is not None and k < len(preset): + # Translated render: the original language's slug, matched + # by heading position (a translation never adds, removes or + # reorders headings; a patched one that does falls back to + # slugging its own text past the end of the list). + base = preset[k] + else: + # Slug the visible text, not the raw markdown (`## [a](url)`). + text = "".join( + c.content + for c in inline.children + if c.type in ("text", "code_inline") + ) + base = slugify(text) or "section" hid, n = base, 2 while hid in seen: hid = f"{base}-{n}" @@ -463,6 +475,36 @@ def _heading_ids(state) -> None: wrap(i, token, f"#{hid}") +def anchor_ids(text: str, title: str | None = None) -> list[str]: + """The section anchor ids of text, in heading order. + + render(anchors_from=...) feeds these to _heading_ids via + env["anchor_ids"], pinning a translated render's anchors to the + original language's slugs. The selection mirrors _heading_ids exactly + (the same md instance assigns the ids during this parse, author-set + {#id} included as-is); the in-body title h1 is excluded. + """ + if title and not has_h1(text): + text = f"# {title}\n\n{text}" + tokens = md.parse(text, {"page_path": ""}) + first_h1 = next( + ( + i + for i, t in enumerate(tokens) + if t.type == "heading_open" and t.tag == "h1" and t.level == 0 + ), + None, + ) + return [ + t.attrGet("id") + for i, t in enumerate(tokens) + if t.type == "heading_open" + and t.tag in ("h1", "h2") + and t.level == 0 + and i != first_h1 + ] + + def make_md(*, verbatim: bool = False) -> MarkdownIt: """A fully configured parser. The module-level ``md`` (below) is the render instance; ``verbatim=True`` builds the segmentation instance for @@ -618,12 +660,16 @@ def render( created: datetime | None = None, modified: datetime | None = None, title: str | None = None, + anchors_from: tuple[str, str] | None = None, ) -> Rendered: """Render Markdown text to the article body's HTML and layout flags. ``title`` injects a ``# {title}`` line at the top when the markdown has no h1 of its own, so the implicit page title goes through the exact same pipeline as an explicit one (first-h1 anchor treatment included). + ``anchors_from`` is the (markdown, title) of the ORIGINAL language when + rendering a translation: section anchors are pinned to its slugs so + localized pages keep the original #hash URLs. The top-level blocks are grouped into column segments: boundary blocks (h1/h2 headings, .wide — see _is_boundary) are rendered bare, the runs @@ -641,6 +687,8 @@ def render( right after the article's h1. """ env = {"page_path": page_path, "line_offset": 0} + if anchors_from is not None: + env["anchor_ids"] = anchor_ids(*anchors_from) if title and not has_h1(text): text = f"# {title}\n\n{text}" # The injected title shifts source lines by two; _heading_ids diff --git a/pagerite/views.py b/pagerite/views.py index 059b2b9..52b44a3 100644 --- a/pagerite/views.py +++ b/pagerite/views.py @@ -783,7 +783,11 @@ def page_content( node = resolve(menu, path)[-1] content = node_markdown(data, node) or "" title = node.title + # The original text pins the section anchors: on a translated page the + # heading slugs (and thus #hash URLs) stay in the original language. + anchors_from = None if translation: + anchors_from = (content, title) if translation.markdown is not None: content = translation.markdown title = ( @@ -793,7 +797,9 @@ def page_content( ) # The title is injected into the markdown (as # title when it has no # h1 of its own), so title and content render as one article. - rendered = render(content, path, node.created, node.modified, title=title) + rendered = render( + content, path, node.created, node.modified, title=title, anchors_from=anchors_from + ) # Long articles get .multicol: the article column cap lifts (see the # #content grid in pagerite.css) and the .cols segments lay out in at # most two columns. The html is already segmented by render() — the