Files
pagerite/pagerite/markdown.py
T
2026-09-21 14:43:37 +00:00

877 lines
35 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Markdown rendering.
Raw HTML (including inline scripts) is passed through unfiltered: the
single author is trusted. Extensions: tables and strikethrough (from the
"default" preset), footnotes, definition lists, task lists,
brace-attributes (`{.class width=300}` on any element, images in
particular), admonitions (``!!! note Title`` with an indented body —
note/tip/warning/etc., the title optional) and GitHub-style alerts
(``> [!NOTE]`` / TIP / IMPORTANT / WARNING / CAUTION, rendered in the
same callout styling). ``::: name`` opens a generic container rendered
as ``<div class="name">`` and closed by a matching ``:::`` (nest by
giving the outer container more colons, e.g. `::::`); the name may be
followed by brace attributes (``::: aside {.right}``), or omitted for a
pandoc-style nameless div (``::: {.aside}``). ``::: aside``
floats as a muted side box, floating in the side zone at the article's
left on all but phone widths — the same margin float ``{.margin}`` (or
``::: margin``) gives any block — and ``::: nocols`` opts its section out
of the column layout. A brace-attribute
line as a block's last line (no blank line between) applies to the whole
block, e.g. a paragraph ending with ``{.wide}`` breaks out of the column
layout as a full-width element; written after a block (code fence,
heading, container, ...) it applies to that preceding block. Bare URLs autolink (GFM), with
the ``https://`` scheme hidden in the link text (``http://`` and other
schemes stay visible; manually labelled links are untouched), and
``H~2~O`` / ``x^2^`` give sub/superscripts.
render() also builds the layout structure: the top-level blocks are
segmented for the column layout — h1/h2 headings and ``.wide`` blocks
stand on their own, the runs between them are wrapped in
``<div class="colseg">`` (tagged
``.cols`` when the segment holds enough text — COLS_TEXT — in at least
COLS_PARAS paragraphs or one paragraph long enough to turn .breakable,
unless a ``::: nocols`` container opts it out;
in column segments, paragraphs past BREAKABLE_TEXT are marked
``.breakable`` so they may split across columns). Margin-breakout boxes
(``.margin``, ``::: aside``) stay inside the segment at their anchor
point; pagerite.css takes them out of flow (absolute, off the article's
left border, into the side zone), so the columns flow through as if the
box wasn't there. The result carries
``multicol`` when the whole body justifies columns (views.py puts the
class on the article); how many columns (never more than two), whether
the margin breakout applies and every other viewport adaptation is then
pagerite.css's call. The thresholds measure visible text, code blocks
excluded.
markdown-it's typographer is enabled, so body text gets SmartyPants-style
replacements: straight quotes become curly, ``--`` / ``---`` become en / em
dashes, ``...`` becomes an ellipsis, ``(c)`` becomes ©, and so on. Single
line breaks inside paragraphs become ``<br>`` (``breaks: True``). Code
spans/blocks and raw HTML are left untouched.
Images get special treatment: a relative `src` is resolved against the
page's own path (so `![alt](photo.avif)` in `/docs/design` is served from
`/docs/design/photo.avif`), and an image standing alone in its paragraph
becomes a block `<figure>` — with `<figcaption>` when it has a title.
Images inline with other content stay plain inline `<img>`, as does raw
`<img>` HTML written by the author. Positioning is done with attribute
classes, e.g. `![alt](photo.avif "Caption"){.right}`.
A lone `{name}` or `{name: args}` line is a block directive, expanded by
the caller through render(directives=...) — `{dates}` (built in) expands
to the article's dateline, `{cards}` / `{cards: path ...}` to card rows
of other pages (views.py). Unresolved tags render as the literal source.
"""
import re
from collections.abc import Callable
from datetime import datetime, timedelta
from typing import NamedTuple
from markdown_it import MarkdownIt
from markdown_it.common.utils import escapeHtml
from markdown_it.renderer import RendererHTML
from markdown_it.token import Token
from mdit_py_plugins.admon import admon_plugin
from mdit_py_plugins.attrs import attrs_plugin
from mdit_py_plugins.attrs.parse import ParseError
from mdit_py_plugins.attrs.parse import parse as parse_attrs
from mdit_py_plugins.container import container_plugin
from mdit_py_plugins.deflist import deflist_plugin
from mdit_py_plugins.footnote import footnote_plugin
from mdit_py_plugins.gfm_autolink import gfm_autolink_plugin
from mdit_py_plugins.subscript import sub_plugin
from mdit_py_plugins.superscript import superscript_plugin
from mdit_py_plugins.tasklists import tasklists_plugin
from pygments import highlight
from pygments.formatters import HtmlFormatter
from pygments.lexers import get_lexer_by_name
from pygments.util import ClassNotFound
from slugify import slugify
# Styles in /_assets/pygments-*.css match this formatter (regenerate:
# HtmlFormatter(style="github-dark").get_style_defs("pre code"))
_formatter = HtmlFormatter(style="github-dark", nowrap=True)
_TASK_MARKER_RE = re.compile(r"^(\s*(?:>\s*)*(?:[-*+]|\d+\.)\s+)\[( |x|X)\](\s+|$)")
def _highlight(text: str, lang: str, _attrs: str) -> str:
"""Syntax-highlight a fenced code block with Pygments.
Returns bare spans (nowrap): markdown-it adds the <pre><code> wrapper,
and the stylesheet is scoped to "pre code" to match.
"""
try:
lexer = get_lexer_by_name(lang)
except ClassNotFound:
return "" # fall back to default <pre><code>
return highlight(text, lexer, _formatter)
def _fence_rule(
self: RendererHTML,
tokens,
idx: int,
options,
env: dict,
) -> str:
"""Render a fenced code block.
Like the default fence rule, but block attributes go on the <pre> —
the block element — instead of the <code>, which keeps only the
language class. Attributes are accepted both pandoc-style on the
info line (```{.python .wide #id key=val} — the first class is the
language when no bare language word precedes the braces) and as a
trailing `{...}` line applied by _block_attrs. This is what makes
e.g. `{.wide}` or `{style="..."}` style the block itself.
"""
token = tokens[idx]
info = token.info.strip() if token.info else ""
lang, _, brace = info.partition("{")
lang = lang.split(maxsplit=1)[0] if lang.strip() else ""
if brace:
try:
_, attrs = parse_attrs("{" + brace)
except ParseError:
attrs = {}
classes = attrs.pop("class", "").split()
if not lang and classes:
lang = classes.pop(0)
if classes:
_apply_attrs(token, {"class": " ".join(classes)})
_apply_attrs(token, attrs)
highlighted = _highlight(token.content, lang, "") or escapeHtml(token.content)
code_class = f' class="{options.langPrefix}{lang}"' if lang else ""
return (
f"<pre{self.renderAttrs(token)}><code{code_class}>{highlighted}</code></pre>\n"
)
def _image_rule(
self: RendererHTML,
tokens,
idx: int,
options,
env: dict,
) -> str:
"""Render images, resolving relative srcs against the page path."""
token = tokens[idx]
src = token.attrs["src"]
if not src.startswith(("/", "http://", "https://", "data:")):
page = env.get("page_path", "")
token.attrs["src"] = f"/{page}/{src}" if page else f"/{src}"
token.attrs["alt"] = self.renderInlineAsText(token.children, options, env)
if len(tokens) == 1:
# The only inline content of its paragraph: render as a block
# figure, captioned when titled. (The <p> wrapper is dropped by
# _unwrap_lone_figures below.) {.margin} positions the whole
# figure, so it moves from the img onto the figure wrapper — left
# on the img, the margin-breakout CSS would pull the image out of
# the figure (and mostly off-screen), leaving the caption behind.
classes = (token.attrs.get("class") or "").split()
figure_class = ""
if "margin" in classes:
classes.remove("margin")
if classes:
token.attrs["class"] = " ".join(classes)
else:
del token.attrs["class"]
figure_class = ' class="margin"'
img = self.renderToken(tokens, idx, options, env)
title = token.attrs.get("title")
caption = f"<figcaption>{escapeHtml(title)}</figcaption>" if title else ""
return f"<figure{figure_class}>{img}{caption}</figure>"
img = self.renderToken(tokens, idx, options, env)
# Inline with other content: a plain inline image.
return img
def _unwrap_lone_figures(state) -> None:
"""Drop the <p> wrapper around a lone image.
markdown-it wraps inline content in a paragraph, but our image rule
turns lone images into <figure> — a block element that is invalid
inside <p>. Browsers hoist it out, leaving an empty paragraph whose
margins disturb the layout.
"""
tokens = state.tokens
for i, token in enumerate(tokens):
if token.type != "inline" or not token.children:
continue
# Attrs consumed out of the text (e.g. {style=...} space-separated
# on the image's own line) leave empty text tokens behind — strip
# them so the lone-image check is not thrown off by user styling.
children = [c for c in token.children if c.type != "text" or c.content]
if children:
token.children = children
[child] = children if len(children) == 1 else [None]
if (
child
and child.type == "image"
and tokens[i - 1].type == "paragraph_open"
and tokens[i + 1].type == "paragraph_close"
):
# A lone image becomes a <figure> (see _image_rule); block
# attrs on the paragraph (e.g. a trailing {.wide} line) move
# onto the image so they survive the unwrap.
_apply_attrs(child, tokens[i - 1].attrs or {})
tokens[i - 1].hidden = True
tokens[i + 1].hidden = True
def _tag_task_checkboxes(state) -> None:
"""Tag rendered task-list checkboxes with a stable index.
The public page and the editor preview use the index to identify which
`[ ]`/`[x]` marker in the Markdown source to toggle when a visitor
clicks the checkbox.
"""
index = 0
for token in state.tokens:
if token.type != "inline" or not token.children:
continue
for child in token.children:
if child.type == "html_inline" and 'type="checkbox"' in child.content:
child.content = child.content.replace(
'type="checkbox"',
f'data-task-index="{index}" type="checkbox"',
1,
)
index += 1
def _shorten_autolinks(state) -> None:
"""Hide the https:// scheme in the text of bare autolinked URLs.
GFM linkify sets the link text to the URL itself; only those links
(markup "autolink") are shortened. http:// and other schemes stay
visible, and manually labelled links keep whatever label was written.
"""
for token in state.tokens:
if token.type != "inline" or not token.children:
continue
for i, child in enumerate(token.children):
if child.type == "link_open" and child.markup == "autolink":
text = token.children[i + 1]
if text.type == "text" and text.content.startswith("https://"):
text.content = text.content.removeprefix("https://")
_CONTAINER_NAME_RE = re.compile(r"[a-zA-Z][\w-]*")
def _apply_attrs(token, attrs: dict) -> None:
"""Join/set parsed brace attributes (`{.class key=value}`) on a token."""
for key, value in attrs.items():
if key == "class":
token.attrJoin("class", value)
else:
token.attrSet(key, value)
def _container_validate(params: str, _markup: str) -> bool:
"""`::: name`, optionally followed by brace attrs (`::: aside {.right}`).
Pandoc-style nameless divs (`::: {.aside}`) are accepted too — the
attrs alone give the container its classes.
"""
name, _, rest = params.strip().partition(" ")
if name.startswith("{"):
name, rest = "", params.strip()
elif not _CONTAINER_NAME_RE.fullmatch(name):
return False
rest = rest.strip()
if not rest:
return bool(name) # a nameless container needs the attrs
try:
pos, _ = parse_attrs(rest)
except ParseError:
return False
# parse() stops at (returns the index of) the closing brace.
return pos == len(rest) - 1
def _container_attrs(state) -> None:
"""Apply `::: name {attrs}` classes to container tokens at parse time.
The container plugin's default render is a plain renderToken, so the
name and brace attributes must live on the token itself — and being a
core rule (rather than a render rule) lets the segmentation in
render() see the classes (the ::: nocols opt-out, {.wide}
containers).
"""
for token in state.tokens:
if token.type != "container_block_open":
continue
info = token.info.strip()
name, _, rest = info.partition(" ")
if name.startswith("{"):
name, rest = "", info
if name:
token.attrJoin("class", name)
if rest.strip():
_, attrs = parse_attrs(rest.strip())
_apply_attrs(token, attrs)
def _block_attrs(state) -> None:
"""Apply `{.class key=value}` on a block's last line to the block.
The inline attrs plugin only covers attributes right after an image,
code span or link; this extends the same brace syntax to whole blocks.
A paragraph takes them at the end of its last line, either directly
(a trailing `{.wide}` line, no blank line between) or space-separated
at the end of the text (`some text {.small}`) — a space means the
braces belong to the block, not to an image or link before them.
A lone `{...}` paragraph applies to the previous block instead (this
is how headings take attributes, since a heading's next line always
starts a new paragraph). Runs before the typographer so quotes inside
attributes stay straight.
"""
tokens = state.tokens
for i, token in enumerate(tokens):
if token.type != "inline" or not token.children:
continue
text = token.children[-1]
if text.type != "text":
continue
m = re.search(r"(\{[^{}]*\})\s*$", text.content)
if not m:
continue
start = m.start(1)
if start and not text.content[start - 1].isspace():
continue # glued to the text — literal, or inline attrs
try:
_, attrs = parse_attrs(m.group(1))
except ParseError:
continue
standalone = len(token.children) == 1
if not standalone and start == 0 and token.children[-2].type != "softbreak":
continue
# The target: the enclosing block for a trailing attrs line, or the
# previous same-level block for a standalone attrs paragraph —
# including self-contained blocks like code fences and <hr>. Never
# a hidden token (tight-list paragraphs render no tag to hold the
# attributes) — in that case leave the text untouched instead of
# silently swallowing it.
own = i - 1 # standalone: the attrs paragraph's own opening token
j = i - 1
while j >= 0:
target = tokens[j]
if target.hidden:
pass
elif standalone:
if (
j != own
and target.level == tokens[own].level
and (
target.nesting == 1
or target.type in ("fence", "code_block", "hr")
)
):
break
elif target.nesting == 1:
break
j -= 1
if j < 0:
continue
_apply_attrs(tokens[j], attrs)
if standalone:
tokens[own].hidden = True
token.children = []
tokens[i + 1].hidden = True
elif start == 0:
del token.children[-2:]
else:
# Braces space-separated at the end of a text line: strip them
# (a whitespace-only remainder means they were on a line of
# their own after all — drop the softbreak too).
text.content = text.content[:start].rstrip()
if not text.content and token.children[-2].type == "softbreak":
del token.children[-2:]
#: Minimum number of in-body h1/h2 headings for section anchors to be
#: useful — shorter articles get no ids/self-links at all.
ANCHOR_MIN_HEADINGS = 3
def _heading_ids(state) -> None:
"""Anchor the in-body h1/h2 headings of long-enough articles.
The markdown body's own h1 and h2 headings get a slug id and their
text is wrapped in a self-link (``<a class="anchor" href="#id">``) so
section links are copyable by click or right-click — but only when the
body has at least ANCHOR_MIN_HEADINGS of them; shorter articles stay
anchor-free. The FIRST h1 is the article title: like the implicit
page-title h1 it gets no id, does not count toward the threshold, and
its self-link is ``href=""`` (back to the top of the page). An
author-set `{#id}` always wins; auto ids slugify the heading text
(python-slugify, mirroring the editor's slugify.js) and dedupe with
-2/-3 suffixes per render — unless env["anchor_ids"] presets them, as
render(anchors_from=...) does for translated pages so section URLs
stay in the original language. Headings that already contain a link are
``data-line`` records the heading's markdown source line (0-based, after
undoing the render(title=...) injection offset via ``env``) — the page
editor uses it for section pens and piecewise-linear scroll sync.
"""
tokens = state.tokens
line_offset = state.env.get("line_offset", 0)
def wrap(i: int, token, href: str) -> None:
inline = tokens[i + 1]
if not inline.children or any(c.type == "link_open" for c in inline.children):
return
anchor = Token("link_open", "a", 1)
anchor.attrs = {"href": href, "class": "anchor"}
inline.children = [anchor, *inline.children, Token("link_close", "a", -1)]
# The first in-body h1 is the title: href="" self-link, never an id.
# Only TOP-LEVEL headings participate — h1/h2 nested in ::: containers
# or asides (level > 0) get no anchors, data-lines or pens.
first_h1 = next(
(
i
for i, t in enumerate(tokens)
if t.type == "heading_open" and t.tag == "h1" and t.level == 0
),
None,
)
if first_h1 is not None:
wrap(first_h1, tokens[first_h1], "")
heads = [
(i, token)
for i, token in enumerate(tokens)
if token.type == "heading_open"
and token.tag in ("h1", "h2")
and token.level == 0
and i != first_h1
]
if len(heads) < ANCHOR_MIN_HEADINGS:
return
seen: set[str] = set()
preset = state.env.get("anchor_ids")
for k, (i, token) in enumerate(heads):
inline = tokens[i + 1]
hid = token.attrGet("id")
if not isinstance(hid, str) or not hid:
if preset is not None and k < len(preset):
# Translated render: the original language's slug, matched
# by heading position (a translation never adds, removes or
# reorders headings; a patched one that does falls back to
# slugging its own text past the end of the list).
base = preset[k]
else:
# Slug the visible text, not the raw markdown (`## [a](url)`).
text = "".join(
c.content
for c in inline.children
if c.type in ("text", "code_inline")
)
base = slugify(text) or "section"
hid, n = base, 2
while hid in seen:
hid = f"{base}-{n}"
n += 1
token.attrSet("id", hid)
seen.add(hid)
if token.map:
token.attrSet("data-line", str(max(0, token.map[0] - line_offset)))
wrap(i, token, f"#{hid}")
def anchor_ids(text: str, title: str | None = None) -> list[str]:
"""The section anchor ids of text, in heading order.
render(anchors_from=...) feeds these to _heading_ids via
env["anchor_ids"], pinning a translated render's anchors to the
original language's slugs. The selection mirrors _heading_ids exactly
(the same md instance assigns the ids during this parse, author-set
{#id} included as-is); the in-body title h1 is excluded.
"""
if title and not has_h1(text):
text = f"# {title}\n\n{text}"
tokens = md.parse(text, {"page_path": ""})
first_h1 = next(
(
i
for i, t in enumerate(tokens)
if t.type == "heading_open" and t.tag == "h1" and t.level == 0
),
None,
)
return [
t.attrGet("id")
for i, t in enumerate(tokens)
if t.type == "heading_open"
and t.tag in ("h1", "h2")
and t.level == 0
and i != first_h1
]
#: A lone {...} paragraph: a block directive like {dates} or
#: {cards: docs/* news} — name, then optional ":"-separated argument text.
_DIRECTIVE_RE = re.compile(r"\{([a-z][a-z0-9_-]*)(?::([^{}\n]*))?\}")
def _directives(state) -> None:
"""Turn lone ``{name}`` / ``{name: args}`` paragraphs into directive tokens.
The expansion is not markdown.py's business: _directive_rule delegates
to the resolvers render() put in env["directives"], falling back to the
literal source when the tag is unknown in the context (e.g. the editor
preview of a page that does not exist yet). The ``cards`` directive gets .wide so it
stands alone as a full-width block outside the column segments (the
card markup never flows in columns). Runs on the render instance only —
the verbatim parser keeps the plain paragraph so segments/chunks see
the placeholder source.
"""
tokens = state.tokens
out = []
i = 0
while i < len(tokens):
if (
i + 2 < len(tokens)
and tokens[i].type == "paragraph_open"
and tokens[i + 1].type == "inline"
and tokens[i + 2].type == "paragraph_close"
):
inline = tokens[i + 1]
children = inline.children or []
if len(children) == 1 and children[0].type == "text":
m = _DIRECTIVE_RE.fullmatch(children[0].content.strip())
if m:
token = Token("directive", "", 0)
token.level = tokens[i].level
token.map = tokens[i].map
token.content = m.group(0)
token.meta = {
"name": m.group(1),
"args": (m.group(2) or "").strip(),
}
if m.group(1) == "cards":
token.attrSet("class", "wide")
out.append(token)
i += 3
continue
out.append(tokens[i])
i += 1
state.tokens = out
def _directive_rule(self: RendererHTML, tokens, idx: int, options, env: dict) -> str:
"""Render a directive token via env["directives"][name](args, env);
unresolved tags render as the literal source paragraph."""
token = tokens[idx]
resolver = (env.get("directives") or {}).get(token.meta["name"])
html = resolver(token.meta["args"], env) if resolver else None
if html is None:
return f"<p>{escapeHtml(token.content)}</p>\n"
return html + "\n"
def make_md(*, verbatim: bool = False) -> MarkdownIt:
"""A fully configured parser. The module-level ``md`` (below) is the
render instance; ``verbatim=True`` builds the segmentation instance for
segments.py, where token text must stay byte-identical to the source so
prose spans can be spliced back by offset: no typographer (quotes and
dashes stay straight), no tasklist label wrapping (the item text stays
a plain text token), and soft line breaks (wrapped prose merges into
one segment instead of splitting at hardbreaks)."""
parser = (
MarkdownIt(
"default",
{
"html": True,
"highlight": _highlight,
"typographer": not verbatim,
"breaks": not verbatim,
},
)
.use(attrs_plugin)
.use(admon_plugin)
.use(container_plugin, "block", validate=_container_validate)
.use(footnote_plugin)
.use(deflist_plugin)
# label wrapping (render) puts the item text inside the checkbox
# <label> html_inline; without it the text stays a plain token.
.use(
tasklists_plugin, enabled=True, label=not verbatim, label_after=not verbatim
)
.use(gfm_autolink_plugin)
.use(sub_plugin)
.use(superscript_plugin)
)
parser.add_render_rule("image", _image_rule)
parser.add_render_rule("fence", _fence_rule)
parser.add_render_rule("directive", _directive_rule)
# GFM alerts (`> [!NOTE]` etc.), built into markdown-it-py's blockquote rule.
parser.options["alerts"] = True
# Block attrs must be stripped before the typographer curlifies their quotes.
parser.core.ruler.before("replacements", "block_attrs", _block_attrs)
parser.core.ruler.push("container_attrs", _container_attrs)
parser.core.ruler.push("unwrap_lone_figures", _unwrap_lone_figures)
parser.core.ruler.push("tag_task_checkboxes", _tag_task_checkboxes)
parser.core.ruler.push("shorten_autolinks", _shorten_autolinks)
parser.core.ruler.push("heading_ids", _heading_ids)
if not verbatim:
parser.core.ruler.push("directives", _directives)
return parser
md = make_md()
# Text-length thresholds (visible characters, code blocks excluded) for the
# column layout: the article goes .multicol past MULTICOL_TEXT, and a column
# segment gets .cols past COLS_TEXT — provided it also has at least
# COLS_PARAS paragraphs or a paragraph long enough to turn .breakable: a
# lone unbreakable paragraph would fill a column on its own and strand the
# rest (e.g. a floated figure) in the other, leaving a mostly empty column.
MULTICOL_TEXT = 1800
COLS_TEXT = 600
COLS_PARAS = 2
#: Paragraphs past this visible length are marked .breakable, letting them
#: split across columns (shorter ones stay unbreakable so a paragraph never
#: straddles the column gap).
BREAKABLE_TEXT = 800
_PRE_BLOCK_RE = re.compile(r"<pre\b.*?</pre>", re.DOTALL)
_TAG_RE = re.compile(r"<[^>]+>")
_PARA_OPEN_RE = re.compile(r"<p[\s>]")
_PARA_RE = re.compile(r"<p((?:\s[^>]*)?)>(.*?)</p>", re.DOTALL)
# Classes that take their block out of the column flow: .wide is a
# full-width separator that splits the column segments. Margin-breakout
# boxes (.margin/.aside) are NOT boundaries: they stay inside the segment
# at their anchor point, and CSS positions them absolutely out of the
# article's left border (the zone rules anchor off the article), so the
# column flow is unaffected.
_WIDE = "wide"
class Rendered(NamedTuple):
"""render() result: the segmented body HTML, and whether the article
should carry .multicol (enough visible text to justify columns)."""
html: str
multicol: bool
def _classes(token) -> set[str]:
return set((token.attrGet("class") or "").split())
def _text_len(html: str) -> int:
"""Visible-text length of rendered HTML, code blocks excluded."""
return len(_TAG_RE.sub("", _PRE_BLOCK_RE.sub("", html)).strip())
def _breakable_paras(html: str) -> str:
"""Mark column-filling paragraphs .breakable so they may split.
Columns keep paragraphs whole (break-inside: avoid-column), but a
paragraph long enough to fill a column would strand everything after
it in a column of its own — these get .breakable, and pagerite.css
lets them split across the column gap. Only applied to .cols segments.
"""
def repl(m: re.Match[str]) -> str:
attrs, body = m.group(1), m.group(2)
if _text_len(body) <= BREAKABLE_TEXT:
return m.group(0)
if 'class="' in attrs:
attrs = attrs.replace('class="', 'class="breakable ', 1)
else:
attrs = f'{attrs} class="breakable"'
return f"<p{attrs}>{body}</p>"
return _PARA_RE.sub(repl, html)
def _top_level_blocks(tokens: list) -> list[list]:
"""Split the token stream into its top-level blocks.
A new block starts at each level-0 opening/self-contained token;
closing and nested tokens (inline children, sub-containers) belong to
the current block, so every slice is balanced and renders on its own.
"""
blocks = []
for token in tokens:
if token.level == 0 and token.nesting >= 0:
blocks.append([token])
elif blocks:
blocks[-1].append(token)
return blocks
def _is_boundary(block: list) -> bool:
"""True for blocks that never go inside a column segment (see the
_WIDE comment above): h1/h2 headings and anything carrying .wide."""
first = block[0]
if first.type == "heading_open" and first.tag in ("h1", "h2"):
return True
for token in block:
if _WIDE in _classes(token):
return True
if token.type == "inline":
children = token.children or []
if any(_WIDE in _classes(c) for c in children):
return True
return False
def render(
text: str,
page_path: str = "",
created: datetime | None = None,
modified: datetime | None = None,
title: str | None = None,
anchors_from: tuple[str, str] | None = None,
directives: dict[str, Callable[[str, dict], str | None]] | None = None,
) -> Rendered:
"""Render Markdown text to the article body's HTML and layout flags.
``title`` injects a ``# {title}`` line at the top when the markdown has
no h1 of its own, so the implicit page title goes through the exact
same pipeline as an explicit one (first-h1 anchor treatment included).
``anchors_from`` is the (markdown, title) of the ORIGINAL language when
rendering a translation: section anchors are pinned to its slugs so
localized pages keep the original #hash URLs.
The top-level blocks are grouped into column segments: boundary blocks
(h1/h2 headings, .wide — see _is_boundary) are rendered bare, the runs
between them wrapped in <div class="colseg">.
A segment is tagged .cols when it holds enough text (COLS_TEXT) in at
least two paragraphs (COLS_PARAS) or one breakable-length paragraph,
and no ::: nocols container; its long paragraphs are marked .breakable;
the article is .multicol when the whole body exceeds MULTICOL_TEXT.
pagerite.css keys all column and margin-breakout layout off these
classes.
A ``{dates}`` line expands to the article's published/updated dateline
(needs ``created``/``modified``). Block directives in general — a lone
``{name}`` or ``{name: args}`` line — are expanded by the resolvers
passed as ``directives`` (name → (args, env) → HTML or None), with
``dates`` built in when ``created`` is given; unresolved tags render as
the literal source (e.g. in the editor preview of a not-yet-created
page).
Position is the author's choice — the dateline typically goes
right after the article's h1.
"""
directives = dict(directives or {})
if created is not None:
directives.setdefault("dates", lambda _args, _env: _dateline(created, modified))
env = {"page_path": page_path, "line_offset": 0, "directives": directives}
if anchors_from is not None:
env["anchor_ids"] = anchor_ids(*anchors_from)
if title and not has_h1(text):
text = f"# {title}\n\n{text}"
# The injected title shifts source lines by two; _heading_ids
# subtracts this from its data-line attributes.
env["line_offset"] = 2
blocks = _top_level_blocks(md.parse(text, env))
# Group consecutive non-boundary blocks into segments (is_segment,
# flat tokens); boundary blocks stand on their own between them.
groups: list[tuple[bool, list]] = []
for block in blocks:
if _is_boundary(block):
groups.append((False, block))
elif groups and groups[-1][0]:
groups[-1][1].extend(block)
else:
groups.append((True, list(block)))
parts = []
total = 0
for is_segment, group in groups:
html = md.renderer.render(group, md.options, env)
if not html.strip():
continue # e.g. a consumed standalone-attrs paragraph
text_len = _text_len(html)
total += text_len
if not is_segment:
parts.append(html)
continue
nocols = any(
"nocols" in _classes(t) for t in group if t.type == "container_block_open"
)
marked = _breakable_paras(html)
cols = (
" cols"
if text_len > COLS_TEXT
and not nocols
and (len(_PARA_OPEN_RE.findall(html)) >= COLS_PARAS or marked != html)
else ""
)
if cols:
html = marked
parts.append(f'<div class="colseg{cols}">{html}</div>')
html = "".join(parts)
return Rendered(html, total > MULTICOL_TEXT)
def _dateline(created: datetime, modified: datetime | None) -> str:
"""Dateline for the ``{dates}`` tag: "1 Jan 2026", plus
" edited 3 Jan 2026" when the last edit came >= 48h after
publishing (quick fixes right after posting stay unmentioned)."""
out = f'<time datetime="{created.isoformat()}">{created.day} {created:%b %Y}</time>'
if modified is not None and modified - created >= timedelta(hours=48):
out += f' edited <time datetime="{modified.isoformat()}">{modified.day} {modified:%b %Y}</time>'
return f'<p class="dateline">{out}</p>'
def has_h1(text: str) -> bool:
"""True if the Markdown source itself contains an h1 heading.
When it does, the article owns its heading and render(title=...) does
not inject the page title as an h1 (the title is still used for the
document <title> and navigation labels).
"""
return any(t.type == "heading_open" and t.tag == "h1" for t in md.parse(text))
def toggle_task(text: str, index: int) -> str | None:
"""Toggle the Nth task-list checkbox marker in ``text``.
Returns the modified Markdown source, or ``None`` if the index is out
of range or the marker could not be found.
"""
tokens = md.parse(text, {"page_path": ""})
checkbox_lines: list[int | None] = []
for token in tokens:
if token.type == "inline" and token.children:
for child in token.children:
if child.type == "html_inline" and 'type="checkbox"' in child.content:
checkbox_lines.append(token.map[0] if token.map else None)
break
if not (0 <= index < len(checkbox_lines)):
return None
line_idx = checkbox_lines[index]
if line_idx is None or line_idx < 0:
return None
lines = text.splitlines(keepends=True)
if line_idx >= len(lines):
return None
line = lines[line_idx]
def repl(m: re.Match[str]) -> str:
prefix = m.group(1)
marker = m.group(2)
new_marker = "x" if marker.strip() == "" else " "
return f"{prefix}[{new_marker}]{m.group(3)}"
new_line = _TASK_MARKER_RE.sub(repl, line, count=1)
if new_line == line:
return None
lines[line_idx] = new_line
return "".join(lines)