From cc750236bc01c6d873fddc9dcea02c4b5ba0c252 Mon Sep 17 00:00:00 2001 From: Leo Vasanko Date: Wed, 2 Sep 2026 02:09:27 +0000 Subject: [PATCH] Chunk keys as 9-byte bytes digests; trans nested by hash -> lang --- docs/localization.md | 17 +++++++++-------- docs/migrate.md | 30 ++++++++++++++++++------------ pagerite/chunks.py | 15 ++++++++++----- pagerite/data.py | 23 +++++++++++++---------- pagerite/i18n.py | 4 ++-- pagerite/migrations.py | 4 +++- 6 files changed, 55 insertions(+), 38 deletions(-) diff --git a/docs/localization.md b/docs/localization.md index 6da55c2..7f2cf1f 100644 --- a/docs/localization.md +++ b/docs/localization.md @@ -100,7 +100,7 @@ list blocks, tables, HTML blocks. A chunk's identity is its **source text**, gettext-msgid style: ```python -chunk_key = blake3(normalize(chunk_text), digest_size=16).hex() +chunk_key = blake3(normalize(chunk_text)).digest(9) # bytes; base64 at the JSON level ``` (`normalize`: strip trailing whitespace per line, collapse surrounding blank @@ -153,10 +153,11 @@ Full storage design and the `migrate_v3` restructuring live in `docs/migrate.md`. The short version, as it concerns this document: - Originals **and** translations are content-addressed text chunks in flat - stores: `Data.chunks: dict[hash, str]` and - `Data.trans: dict[f"{chunk_hash}:{lang}", str]` — path-independent, so - repeated paragraphs and menu titles are translated once and article moves - touch nothing. `Node.chunks: list[hash]` gives each article its order. + stores: `Data.chunks: dict[bytes, str]` and + `Data.trans: dict[bytes, dict[str, str]]` (chunk hash → lang → text) — + path-independent, so repeated paragraphs and menu titles are translated + once and article moves touch nothing. `Node.chunks: list[bytes]` gives + each article its order. - `Node` gains **`language: str = ""`**, inherited down the tree like `banner` (empty = nearest ancestor, front page last, site default `en` final). `select_language` and `` use the resolved value instead @@ -177,7 +178,7 @@ def get_translation(path, lang, data) -> Translation | None: if lang not in node.langs: return None hybrid = "\n\n".join( - chunks[h] if h in node.no_trans else trans.get(f"{h}:{lang}", chunks[h]) + chunks[h] if h in node.no_trans else trans.get(h, {}).get(lang, chunks[h]) for h in node.chunks ) for patch in data.patches.get(f"{path}:{lang}", []): @@ -191,8 +192,8 @@ def get_translation(path, lang, data) -> Translation | None: hreflang never probe the `trans` store chunk by chunk. A stale key is benign (the "translation" just renders as the original). - `titles` for nav/sidebar/cards: each node's translated title is - `trans.get(f"{hash(node.title)}:{lang}")` with per-node fallback — one dict - lookup per nav item at render time. + `trans.get(hash(node.title), {}).get(lang)` with per-node fallback — one + dict lookup per nav item at render time. - Cache invalidation: writes to `chunks` / `trans` / `patches` (translator, editor saves) call `_invalidate_pages()`, same as content writes. diff --git a/docs/migrate.md b/docs/migrate.md index 1bf0380..d5889c1 100644 --- a/docs/migrate.md +++ b/docs/migrate.md @@ -36,13 +36,13 @@ class Node(msgspec.Struct, omit_defaults=True): ... #: Replaces `content: str | None`. None = pure category label; #: a list (possibly empty) = a page, as ordered chunk hashes. - chunks: list[str] | None = None + chunks: list[bytes] | None = None #: Primary language of the article (BCP-47 base tag). "" = inherit #: (nearest ancestor, front page last, site default "en" final). language: str = "" #: Chunk hashes the editor marked "do not translate" (always served #: from the original). Presence-keys, value always True. - no_trans: dict[str, True] = {} + no_trans: dict[bytes, True] = {} #: Languages this article is available in (besides its primary #: language). Presence-keys, value always True — rendering, language #: selection and hreflang alternates read this set instead of probing @@ -52,12 +52,14 @@ class Node(msgspec.Struct, omit_defaults=True): class Data(msgspec.Struct): ... - #: All original-language text, content-addressed: blake3(normalized, - #: digest 16) hex -> Markdown chunk. Shared by every article. - chunks: dict[str, str] = {} - #: Machine translations: f"{chunk_hash}:{lang}" -> translated Markdown. + #: All original-language text, content-addressed: blake3(normalized) + #: digest[:9] -> Markdown chunk. Shared by every article. Keys are + #: bytes; kanta/msgspec base64-encode them at the JSON level. + chunks: dict[bytes, str] = {} + #: Machine translations: chunk hash -> lang -> translated Markdown + #: (nested, not tuple keys: msgspec's JSON serializer rejects them). #: Also used for node titles (hash of the title text). - trans: dict[str, str] = {} + trans: dict[bytes, dict[str, str]] = {} #: User override patches per article and language: #: f"{path}:{lang}" -> ordered patches (see localization.md). patches: dict[str, list[Patch]] = {} @@ -70,7 +72,7 @@ Notes: building hrefs. `migrate_v3` audits existing stored paths (translation keys, analytics references, any path-valued fields) and normalizes them. - **Titles are chunks too**, by hash only: the nav renderer looks up - `trans.get(f"{hash(node.title)}:{lang}")`. No separate title storage; + `trans.get(hash(node.title), {}).get(lang)`. No separate title storage; editing a title invalidates its translations automatically. - **Per-hunk options** live in two places: *inherent* options are derived at chunking time (code fences and HTML blocks are marked no-translate without @@ -88,7 +90,7 @@ that article rendering, `select_language`'s availability check, and hreflang alternate links never enumerate chunks. It is written by whoever writes translation data, in the same transaction: -- **Translator job:** after writing `trans[f"{h}:{lang}"]` entries for an +- **Translator job:** after writing `trans[h][lang]` entries for an article's chunks (or its title), set `node.langs[lang] = True`. - **Translated-view save:** appending the first patch for `f"{path}:{lang}"` sets `node.langs[lang] = True` (patches alone make the version exist). @@ -101,7 +103,7 @@ translation data, in the same transaction: - **Render:** `text = "\n\n".join(chunks[h] for h in node.chunks)` for the original; for language `L` (only ever attempted when `L in node.langs`), - per chunk `trans[f"{h}:{L}"]` unless missing or `h in node.no_trans`, + per chunk `trans.get(h, {}).get(L)` unless missing or `h in node.no_trans`, falling back to `chunks[h]`; then apply `patches.get(f"{path}:{L}", [])` in order (per-hunk, best effort); then `markdown.render` as today. All of this assembles the `Translation` the phase-1 plumbing already consumes. @@ -135,10 +137,14 @@ Chunking must be deterministic and shared with render/save, so ## Implementation notes (deviations from the plan above) - Chunking lives in `pagerite/chunks.py`; hashing uses the `blake3` package - (already a dependency) with a 16-byte digest (`hexdigest(16)`). + (already a dependency), truncated to a 9-byte `bytes` digest (kanta's + JSON persistence base64-encodes bytes keys to 12-char strings). +- `trans` is keyed `hash -> lang -> text` (nested dict), not by + `f"{hash}:{lang}"` tuples: msgspec's JSON serializer only supports + str-like/number-like dict keys, and kanta persists as JSON lines. - `Translation.titles` stayed keyed by node path (phase-1 shape, views untouched): `get_translation` builds it by walking the menu with the same - per-title `trans[f"{chunk_key(node.title)}:{lang}"]` lookups. + per-title `trans.get(chunk_key(node.title), {}).get(lang)` lookups. - Insert hunks anchor on the whole preceding block (not just its tail) — a stronger, simpler search context. - `make_patch` diffs with `SequenceMatcher(autojunk=False)` so patches are diff --git a/pagerite/chunks.py b/pagerite/chunks.py index 9d118e6..604bf8a 100644 --- a/pagerite/chunks.py +++ b/pagerite/chunks.py @@ -111,10 +111,15 @@ def _normalize(text: str) -> str: return "\n".join(line.rstrip() for line in text.split("\n")).strip("\n") -def chunk_key(text: str) -> str: - """Content key of a chunk: blake3 hex (16-byte digest, 32 hex chars) - of the normalized text — the same hasher app.py's file store uses.""" - return blake3.blake3(_normalize(text).encode()).hexdigest(16) +def chunk_key(text: str) -> bytes: + """Content key of a chunk: the first 9 bytes of the blake3 digest of + the normalized text (72 bits — a site's chunk count stays far below + the birthday bound), using the same hasher as app.py's file store. + + Keys are bytes: kanta/msgspec base64-encode them at the JSON + persistence level, so the raw database dicts carry 12-char strings. + """ + return blake3.blake3(_normalize(text).encode()).digest(9) def needs_translation(chunk: str) -> bool: @@ -137,7 +142,7 @@ def join_chunks(chunks: list[str]) -> str: return "\n\n".join(chunks) + "\n" if chunks else "" -def store_chunks(store: dict[str, str], markdown: str) -> list[str]: +def store_chunks(store: dict[bytes, str], markdown: str) -> list[bytes]: """Chunk ``markdown`` into ``store`` (hash -> text); return the ordered hashes. Unchanged chunks keep their hashes, so only genuinely new text lands in the kanta change diff. First writer wins: variants sharing a diff --git a/pagerite/data.py b/pagerite/data.py index 056b4ad..01c72bb 100644 --- a/pagerite/data.py +++ b/pagerite/data.py @@ -39,16 +39,16 @@ class Node(msgspec.Struct, omit_defaults=True): title: str = "" order: float = 0 - #: Ordered chunk hashes into ``Data.chunks``; None = pure category - #: label (its URL renders a placeholder page), a list (possibly - #: empty) = a page. - chunks: list[str] | None = None + #: Ordered chunk hashes (9-byte keys into ``Data.chunks``); None = + #: pure category label (its URL renders a placeholder page), a list + #: (possibly empty) = a page. + chunks: list[bytes] | None = None #: Primary language of the article (BCP-47 base tag). "" = inherit #: (nearest ancestor, front page last, site default "en" final). language: str = "" #: Chunk hashes the editor marked "do not translate" (always served #: from the original). Presence-keys, value always True. - no_trans: dict[str, bool] = {} + no_trans: dict[bytes, bool] = {} #: Languages this article is available in (besides its primary #: language). Presence-keys, value always True — the availability #: index for rendering, language selection and hreflang alternates; @@ -102,11 +102,14 @@ class Data(msgspec.Struct): #: /favicon.ico. favicon: str = "" #: All original-language page text, content-addressed: - #: chunk_key -> Markdown chunk. Shared by every article. - chunks: dict[str, str] = {} - #: Machine translations: f"{chunk_hash}:{lang}" -> translated - #: Markdown. Also used for node titles (hash of the title text). - trans: dict[str, str] = {} + #: chunk_key (9 bytes; base64 at the JSON level) -> Markdown chunk. + #: Shared by every article. + chunks: dict[bytes, str] = {} + #: Machine translations: chunk hash -> lang -> translated Markdown + #: (a nested dict rather than tuple keys, which msgspec's JSON + #: serializer does not support). Also used for node titles (hash of + #: the title text). + trans: dict[bytes, dict[str, str]] = {} #: User override patches per article and language: #: f"{path}:{lang}" -> ordered patches (paths without leading slash). patches: dict[str, list[Patch]] = {} diff --git a/pagerite/i18n.py b/pagerite/i18n.py index 6ae7802..d239bb0 100644 --- a/pagerite/i18n.py +++ b/pagerite/i18n.py @@ -133,7 +133,7 @@ def hybrid_markdown(data: Data, node: Node, path: str, lang: str) -> str: hybrid = join_chunks([ data.chunks.get(h, "") if h in node.no_trans - else data.trans.get(f"{h}:{lang}") or data.chunks.get(h, "") + else data.trans.get(h, {}).get(lang) or data.chunks.get(h, "") for h in node.chunks or [] ]) for patch in data.patches.get(f"{path}:{lang}", []): @@ -154,7 +154,7 @@ def title_map(data: Data, lang: str) -> dict[str, str]: for slug, node in nodes.items(): path = f"{prefix}/{slug}" if prefix else slug if node.title: - t = data.trans.get(f"{chunk_key(node.title)}:{lang}") + t = data.trans.get(chunk_key(node.title), {}).get(lang) if t: titles[path] = t walk(node.children, path) diff --git a/pagerite/migrations.py b/pagerite/migrations.py index ed61ad2..e3ba86c 100644 --- a/pagerite/migrations.py +++ b/pagerite/migrations.py @@ -141,6 +141,8 @@ def migrate_v3(d: dict) -> None: list as ``chunks`` (an absent content stays absent, i.e. None = a pure category label; "" chunks to an empty list = an empty page). + Chunk keys are 9-byte blake3 digests; at this raw JSON level they are + base64 strings (decoding into the structs restores ``bytes`` keys). ``trans``/``patches`` start empty; the translator job fills them and maintains the ``langs`` index as translations land. ``language``, ``no_trans`` and ``langs`` need nothing — struct defaults cover them. @@ -155,7 +157,7 @@ def migrate_v3(d: dict) -> None: if isinstance(content, str): hashes = [] for chunk in chunk_markdown(content): - key = chunk_key(chunk) + key = base64.b64encode(chunk_key(chunk)).decode() store.setdefault(key, chunk) hashes.append(key) node["chunks"] = hashes