diff --git a/pagerite/chunks.py b/pagerite/chunks.py new file mode 100644 index 0000000..9d118e6 --- /dev/null +++ b/pagerite/chunks.py @@ -0,0 +1,150 @@ +"""Block-level Markdown chunking for content-addressed storage. + +A page's Markdown is split into deterministic block-level chunks, each +stored once under its content hash in ``Data.chunks`` (docs/migrate.md). +Shared by the render/save pipeline (app.py, views.py, i18n.py) and the +schema migration (migrations.py), so a chunk's key is stable no matter +where the split happens. +""" + +import re + +import blake3 + +#: Fenced code block opener/closer: up to 3 spaces indent, then 3+ +#: backticks or tildes (CommonMark). +_FENCE_OPEN = re.compile(r"^ {0,3}(`{3,}|~{3,})") + +#: HTML block openers that may span blank lines (CommonMark types 1-5: +#: script/pre/style/textarea, comments, processing instructions, +#: declarations, CDATA) with their closing condition. Other HTML blocks +#: end at the first blank line, which the generic blank-line split +#: already does. +_HTML_ATOMIC = ( + (re.compile(r"^ {0,3}<(?:script|pre|style|textarea)(?:\s|>|$)", re.I), + re.compile(r"", re.I)), + (re.compile(r"^ {0,3}")), + (re.compile(r"^ {0,3}<\?"), re.compile(r"\?>")), + (re.compile(r"^ {0,3}")), + (re.compile(r"^ {0,3}")), +) + +#: First line of a generic HTML block (a block-level tag). +_HTML_TAG = re.compile(r"^ {0,3}]*>") + + +def _fence_close(line: str, opener: str) -> bool: + """True when ``line`` closes a code fence opened by ``opener``: the + same marker char, at least as many, and nothing else on the line.""" + stripped = line.strip() + return ( + len(stripped) >= len(opener) + and stripped[0] == opener[0] + and set(stripped) == {opener[0]} + ) + + +def chunk_markdown(markdown: str) -> list[str]: + """Split Markdown into block-level chunks, deterministically. + + Blocks are separated by blank lines; fenced code blocks and the + multi-line HTML blocks (comments, script/pre/style, CDATA...) are + kept atomic, even across blank lines, and end at their closing + condition. Chunks carry no surrounding blank lines and no trailing + newline; rejoining with ``join_chunks`` reproduces the source modulo + blank-line normalization. + """ + chunks: list[str] = [] + buf: list[str] = [] + fence = "" # opener marker of the code fence we are in ("" = outside) + html_end: re.Pattern | None = None # closes the atomic HTML block we are in + + def flush() -> None: + text = "\n".join(buf).strip("\n") + if text.strip(): + chunks.append(text) + buf.clear() + + for line in markdown.split("\n"): + if fence: + buf.append(line) + if _fence_close(line, fence): + fence = "" + flush() + continue + if html_end is not None: + buf.append(line) + if html_end.search(line): + html_end = None + flush() + continue + if not line.strip(): + flush() + continue + if m := _FENCE_OPEN.match(line): + # Fences interrupt paragraphs (CommonMark): start a new block. + flush() + fence = m.group(1) + buf.append(line) + continue + if not buf: + for open_re, close_re in _HTML_ATOMIC: + if open_re.match(line): + buf.append(line) + if close_re.search(line): # opens and closes on one line + flush() + else: + html_end = close_re + break + else: + buf.append(line) + continue + buf.append(line) + flush() # an unterminated fence/HTML block runs to EOF, kept as code/HTML + return chunks + + +def _normalize(text: str) -> str: + """Whitespace-insensitive chunk identity: strip trailing whitespace + per line and collapse surrounding blank lines, so whitespace-only + source edits don't invalidate translations.""" + return "\n".join(line.rstrip() for line in text.split("\n")).strip("\n") + + +def chunk_key(text: str) -> str: + """Content key of a chunk: blake3 hex (16-byte digest, 32 hex chars) + of the normalized text — the same hasher app.py's file store uses.""" + return blake3.blake3(_normalize(text).encode()).hexdigest(16) + + +def needs_translation(chunk: str) -> bool: + """False for chunks without prose: pure code fences and HTML blocks. + + These are inherently no-translate (docs/migrate.md): derived from the + chunk text itself, nothing is stored. + """ + if _FENCE_OPEN.match(chunk): + return False + first = chunk.split("\n", 1)[0] + if any(open_re.match(first) for open_re, _ in _HTML_ATOMIC): + return False + return not _HTML_TAG.match(first) + + +def join_chunks(chunks: list[str]) -> str: + """The stored page form of chunks: blocks joined by a blank line, + with a trailing newline ("" for no chunks).""" + return "\n\n".join(chunks) + "\n" if chunks else "" + + +def store_chunks(store: dict[str, str], markdown: str) -> list[str]: + """Chunk ``markdown`` into ``store`` (hash -> text); return the ordered + hashes. Unchanged chunks keep their hashes, so only genuinely new text + lands in the kanta change diff. First writer wins: variants sharing a + key differ only in insignificant whitespace (see chunk_key).""" + hashes = [] + for chunk in chunk_markdown(markdown): + key = chunk_key(chunk) + store.setdefault(key, chunk) + hashes.append(key) + return hashes