From 967e1343db8b59882bf0ca4a36f4af0e05302eda Mon Sep 17 00:00:00 2001 From: Alessandro Afloarei Date: Mon, 31 Aug 2026 14:46:16 +0000 Subject: [PATCH 1/3] feat(family-2): parser-safe preamble hoisting + list neutralization Two measured loss modes in the pinned CodeWikiBench parser (probed directly against 5e728fb4, 2026-08-31): 1. Preamble text under a heading with child headings has no dict key to hang on and is dropped whole. Fix: hoistPreambles packages each preamble as a synthetic first child heading ("Overview", with collision fallbacks), the same key shape CodeWiki's own example output uses. 2. Within a section span, everything after the first list dies: a trailing paragraph, a second list, and (the corpus staple) list items directly above a Sources line. Fix: neutralizeNonFinalLists escapes the markers of any list that is not the span's trailing list (1. -> 1\.), which renders identically but parses as paragraphs, which always survive. No words are added, removed, or reordered by either transform; this is packaging for their parser, not editing. Measured on the committed hono corpora (dry-run, all pages parsed both before and after): hono-2026-07-v2: retained-word fraction 0.7339 -> 0.9516 hono-2026-07: 0.73x -> 0.9508 Word-multiset diff shows the residual gap is ~2/3 markdown syntax tokens (heading markers, blockquote prefixes, fence info-strings, alert tags), so content-level retention is ~99%. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01DfgE3Yucn5LUDZPpqJzhz2 --- runners/codewikibench.test.ts | 125 ++++++++++++++++++++++++++ runners/codewikibench.ts | 161 +++++++++++++++++++++++++++++++++- 2 files changed, 285 insertions(+), 1 deletion(-) diff --git a/runners/codewikibench.test.ts b/runners/codewikibench.test.ts index 9c8889c..05c90ac 100644 --- a/runners/codewikibench.test.ts +++ b/runners/codewikibench.test.ts @@ -1,5 +1,7 @@ import { describe, it, expect } from "vitest"; import { + hoistPreambles, + neutralizeNonFinalLists, planEvaluatorInput, stripLeadingDetailsBlock, transpileMdx, @@ -283,3 +285,126 @@ describe("planEvaluatorInput — merged with user_guide", () => { expect(new Set(slugs).size).toBe(slugs.length); }); }); + +// --------------------------------------------------------------------------- +// hoistPreambles +// --------------------------------------------------------------------------- + +describe("hoistPreambles", () => { + it("hoists an intro paragraph under an H1 with H2 children into ## Overview", () => { + const out = hoistPreambles("# Page\n\nIntro words.\n\n## First Section\n\nBody.\n"); + expect(out).toContain("# Page\n\n## Overview\n\nIntro words."); + expect(out.indexOf("## Overview")).toBeLessThan(out.indexOf("## First Section")); + }); + + it("hoists a preamble under an H2 with H3 children into ### Overview", () => { + const out = hoistPreambles("# P\n\n## Section\n\nPreamble.\n\n### Child\n\nLeaf.\n"); + expect(out).toContain("## Section\n\n### Overview\n\nPreamble."); + }); + + it("leaves leaf sections untouched", () => { + const md = "# P\n\n## Leaf\n\nJust a paragraph.\n\n1. item\n"; + expect(hoistPreambles(md)).toBe(md); + }); + + it("leaves a childful section with no preamble untouched", () => { + const md = "# P\n\n## Section\n\n### Child\n\nBody.\n"; + expect(hoistPreambles(md)).toBe(md); + }); + + it("ignores heading-looking lines inside fences, as content and as boundary", () => { + const md = "# P\n\nIntro.\n\n```sh\n# not a heading\n## also not\n```\n\n## Real Child\n\nBody.\n"; + const out = hoistPreambles(md); + expect(out).toContain("# P\n\n## Overview\n\nIntro."); + // The fence stays verbatim inside the hoisted preamble. + expect(out).toContain("```sh\n# not a heading\n## also not\n```"); + expect(out.match(/## Overview/g)).toHaveLength(1); + }); + + it("falls back to Introduction when a sibling child is already titled Overview", () => { + const out = hoistPreambles("# P\n\nIntro.\n\n## Overview\n\nReal overview.\n"); + expect(out).toContain("# P\n\n## Introduction\n\nIntro."); + }); + + it("hoists every childful section independently", () => { + const out = hoistPreambles( + "# P\n\nTop intro.\n\n## A\n\nA preamble.\n\n### A1\n\nLeaf.\n\n## B\n\nB leaf only.\n", + ); + expect(out).toContain("# P\n\n## Overview\n\nTop intro."); + expect(out).toContain("## A\n\n### Overview\n\nA preamble."); + // B is a leaf: untouched. + expect(out).toContain("## B\n\nB leaf only."); + }); +}); + +describe("neutralizeNonFinalLists", () => { + it("escapes a list followed by a paragraph, keeps the words verbatim", () => { + const out = neutralizeNonFinalLists("## S\n\n1. First step.\n2. Second step.\n\nAfter-list paragraph.\n"); + expect(out).toContain("1\\. First step."); + expect(out).toContain("2\\. Second step."); + expect(out).toContain("After-list paragraph."); + }); + + it("keeps a section-final list as a real list", () => { + const md = "## S\n\nLead-in.\n\n1. Only list.\n2. Still fine.\n"; + expect(neutralizeNonFinalLists(md)).toBe(md); + }); + + it("escapes the first of two lists, keeps the second", () => { + const out = neutralizeNonFinalLists("## S\n\n1. A.\n\nBetween.\n\n- B.\n"); + expect(out).toContain("1\\. A."); + expect(out).toContain("\nBetween."); + expect(out).toContain("\n- B."); + expect(out).not.toContain("\\- B."); + }); + + it("escapes nested markers of a non-final list block", () => { + const out = neutralizeNonFinalLists("## S\n\n- Parent.\n 1. Child.\n\nTail paragraph.\n"); + expect(out).toContain("\\- Parent."); + expect(out).toContain(" 1\\. Child."); + }); + + it("treats each heading span independently", () => { + const out = neutralizeNonFinalLists("## A\n\n1. NonFinal.\n\nTail.\n\n## B\n\n1. Final.\n"); + expect(out).toContain("1\\. NonFinal."); + expect(out).toContain("\n1. Final."); + }); + + it("never touches list-looking lines inside fences", () => { + const md = "## S\n\n```txt\n1. not a list\n- neither\n```\n\nTail.\n"; + expect(neutralizeNonFinalLists(md)).toBe(md); + }); +}); + +describe("neutralizeNonFinalLists — tight lists", () => { + it("escapes a tight list (lead-in directly above the markers) when content follows", () => { + const out = neutralizeNonFinalLists( + "## S\n\nThe check sequence is:\n1. First check.\n2. Second check.\n\nSources: [x.ts:1-2](y)\n", + ); + expect(out).toContain("The check sequence is:"); + expect(out).toContain("1\\. First check."); + expect(out).toContain("2\\. Second check."); + expect(out).toContain("Sources: [x.ts:1-2](y)"); + }); + + it("keeps a tight section-final block untouched", () => { + const md = "## S\n\nMechanism:\n1. Detect.\n2. Rewrite.\n"; + expect(neutralizeNonFinalLists(md)).toBe(md); + }); +}); + +describe("neutralizeNonFinalLists — trailing Sources line", () => { + it("escapes a span-final list whose block ends with a Sources paragraph", () => { + const out = neutralizeNonFinalLists( + "## S\n\nOnion model:\n\n1. Receive context.\n2. Dispatch next.\nSources: [x.ts:1-2](y)\n", + ); + expect(out).toContain("1\\. Receive context."); + expect(out).toContain("2\\. Dispatch next."); + expect(out).toContain("Sources: [x.ts:1-2](y)"); + }); + + it("keeps a trailing list whose continuations are indented", () => { + const md = "## S\n\nLead.\n\n- Parent.\n 1. Child one.\n 2. Child two.\n"; + expect(neutralizeNonFinalLists(md)).toBe(md); + }); +}); diff --git a/runners/codewikibench.ts b/runners/codewikibench.ts index a1de667..407a7f7 100644 --- a/runners/codewikibench.ts +++ b/runners/codewikibench.ts @@ -313,6 +313,160 @@ export function stripLeadingDetailsBlock(md: string): string { return md.replace(/^(\s*(?:# [^\n]*\n)?\s*)
[\s\S]*?<\/details>\s*/, "$1"); } +// --------------------------------------------------------------------------- +// hoistPreambles — parser-safe preamble hoisting +// --------------------------------------------------------------------------- + +/** + * The evaluator's parser (`markdown_to_json.jsonify`) maps each heading to a + * dict entry; a section that has CHILD headings becomes a dict of those + * children, and any loose content between the parent heading and its first + * child heading has nowhere to hang — it is silently dropped. Measured + * directly against the pinned parser (2026-08-31 probe, see the dry-run + * results): this preamble drop is the ONLY construct lost. Paragraphs, + * ordered and unordered lists, tables, blockquotes, and fences under LEAF + * headings all survive verbatim. + * + * The fix packages each preamble as a synthetic first child heading one + * level deeper (default title "Overview" — the same key shape CodeWiki's own + * example output uses, e.g. "Purpose"), so the words their judge scores are + * the words doc0 wrote. Content is never altered, reordered, or summarized: + * this is packaging for their parser, not editing. + * + * Scope notes: + * - ATX headings only (`#`…`######`) — doc0-generated GFM never emits setext + * headings, and the corpus exporter preserves that. + * - Fences and inline code are masked first (same `maskCode` as + * `transpileMdx`), so a `#` line inside a code sample neither hoists nor + * terminates a section. + * - A synthetic title colliding with a real sibling child would merge two + * dict keys in their parser, so the title falls back Overview → + * Introduction → Preamble → "Overview N". + * - Content before the file's FIRST heading is out of scope: exported corpus + * pages always open with their H1 (and `stripLeadingDetailsBlock` runs + * earlier in the chain). + */ +export function hoistPreambles(md: string): string { + const { working, fences, inlines } = maskCode(md); + const lines = working.split("\n"); + + interface HeadingLine { + idx: number; + level: number; + text: string; + } + const headings: HeadingLine[] = []; + lines.forEach((line, idx) => { + const m = /^(#{1,6})\s+(.*)$/.exec(line); + if (m) headings.push({ idx, level: m[1].length, text: (m[2] ?? "").trim() }); + }); + + const insertions: { at: number; heading: string }[] = []; + for (let i = 0; i < headings.length; i += 1) { + const parent = headings[i]; + const next = headings[i + 1]; + // Leaf section (no following heading, or the next heading is a sibling or + // an ancestor): its content survives their parser untouched. + if (!parent || !next || next.level <= parent.level) continue; + + const preamble = lines.slice(parent.idx + 1, next.idx); + if (!preamble.some((l) => l.trim().length > 0)) continue; + + const childLevel = parent.level + 1; + const sectionEndIdx = + headings.slice(i + 1).find((h) => h.level <= parent.level)?.idx ?? lines.length; + const siblingTitles = new Set( + headings + .filter((h) => h.idx > parent.idx && h.idx < sectionEndIdx && h.level === childLevel) + .map((h) => h.text.toLowerCase()), + ); + let title = "Overview"; + if (siblingTitles.has(title.toLowerCase())) title = "Introduction"; + if (siblingTitles.has(title.toLowerCase())) title = "Preamble"; + for (let n = 2; siblingTitles.has(title.toLowerCase()); n += 1) title = `Overview ${n}`; + + insertions.push({ at: parent.idx + 1, heading: `${"#".repeat(childLevel)} ${title}` }); + } + + for (const ins of insertions.reverse()) { + lines.splice(ins.at, 0, "", ins.heading); + } + return unmaskCode(lines.join("\n"), fences, inlines); +} + +// --------------------------------------------------------------------------- +// neutralizeNonFinalLists — parser-safe list flattening +// --------------------------------------------------------------------------- + +/** + * Second measured loss mode in the pinned parser (2026-08-31 probe): within a + * section's content span, `markdown_to_json` keeps everything up to and + * including the FIRST list, then drops whatever follows it — a paragraph + * after a list, and any second list, vanish. A list that closes its section + * survives whole. + * + * Content-preserving fix: any list block that is NOT the final block of its + * span gets its markers escaped (`1.` → `1\.`, `-` → `\-`), which renders as + * the identical visible text but parses as ordinary paragraphs — and + * paragraphs always survive. The section-final list keeps real list syntax. + * No words are added, removed, or reordered. + * + * Mechanics: spans are the line ranges between ATX headings (fences masked + * first, so fence content is never touched and never splits a block). Blocks + * are blank-line-separated; a block whose first line carries a list marker is + * a list block. A loose list that blank-splits into several blocks keeps its + * final physical block as a real list and has the earlier items escaped — + * visually unchanged, and every word survives either way. + */ +export function neutralizeNonFinalLists(md: string): string { + const { working, fences, inlines } = maskCode(md); + const lines = working.split("\n"); + + const isHeading = (l: string): boolean => /^#{1,6}\s+/.test(l); + const isMarker = (l: string): boolean => /^\s*(?:[-*+]|\d+[.)])\s+/.test(l); + const isIndented = (l: string): boolean => /^\s+\S/.test(l); + const isBlank = (l: string): boolean => l.trim() === ""; + + const escapeMarker = (i: number): void => { + const line = lines[i] ?? ""; + lines[i] = line + .replace(/^(\s*)(\d+)([.)])(\s)/, "$1$2\\$3$4") + .replace(/^(\s*)([-*+])(\s)/, "$1\\$2$3"); + }; + + // Line-wise rule per inter-heading span: a list-marker line SURVIVES only + // when every non-blank line after it (to the span's end) is itself a list + // marker or an indented continuation — i.e. only the span's trailing list + // keeps real list syntax. This covers every measured death shape: a list + // followed by a paragraph, a second list, a tight lead-in block, and the + // corpus staple of a "Sources:" line directly under the final item (which + // kills the items but survives itself). + const processSpan = (start: number, end: number): void => { + // suffixClean[j]: every non-blank line in [j, end) is marker/indented. + const suffixClean: boolean[] = new Array(end - start + 1); + suffixClean[end - start] = true; + for (let j = end - 1; j >= start; j -= 1) { + const line = lines[j] ?? ""; + const rest = suffixClean[j - start + 1] ?? true; + suffixClean[j - start] = isBlank(line) ? rest : (isMarker(line) || isIndented(line)) && rest; + } + for (let j = start; j < end; j += 1) { + const line = lines[j] ?? ""; + if (isMarker(line) && !(suffixClean[j - start + 1] ?? true)) escapeMarker(j); + } + }; + + let spanStart = 0; + for (let i = 0; i <= lines.length; i += 1) { + if (i === lines.length || isHeading(lines[i] ?? "")) { + processSpan(spanStart, i); + spanStart = i + 1; + } + } + + return unmaskCode(lines.join("\n"), fences, inlines); +} + // --------------------------------------------------------------------------- // module_tree.json — hierarchy + user_guide merge // --------------------------------------------------------------------------- @@ -420,7 +574,12 @@ async function loadCorpusPages(dir: string): Promise { return Promise.all( entries.map(async (f) => ({ slug: f.slice(0, -3), - content: transpileMdx(stripLeadingDetailsBlock(await readFile(join(dir, f), "utf-8"))), + // Order matters: components demote first (Step/Accordion introduce real + // headings), preambles hoist next (this reshapes the heading spans), and + // list neutralization runs last against the final span layout. + content: neutralizeNonFinalLists( + hoistPreambles(transpileMdx(stripLeadingDetailsBlock(await readFile(join(dir, f), "utf-8")))), + ), })), ); } From e8b85692199045a4a5e641c6c8e29580707f5169 Mon Sep 17 00:00:00 2001 From: Alessandro Afloarei Date: Mon, 31 Aug 2026 14:51:07 +0000 Subject: [PATCH 2/3] docs: external-benchmarks table carries all four published CodeWikiBench systems CodeWiki (Sonnet-4) is named as the benchmark authors' own generator, and the deepwiki-open and OpenDeepWiki rows join from the paper's Table 1. Notes the Family-2 comparability contract: their published per-repo rubrics + their exact judge panel, so doc0's row compares per-repo against every published row without re-running any peer. The 7-vs-21-repo scope discrepancy is cited as-published per docs/codewikibench-pinned.md (f). Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01DfgE3Yucn5LUDZPpqJzhz2 --- docs/comparison-systems.md | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/docs/comparison-systems.md b/docs/comparison-systems.md index 663fb1f..5e52402 100644 --- a/docs/comparison-systems.md +++ b/docs/comparison-systems.md @@ -60,13 +60,17 @@ Computed directly against the corpus and freshly-fetched peer content with the s ## External benchmarks (not yet run by us) -CodeWikiBench (ACL Findings 2026) publishes its own evaluator and competitor scores. doc0 has not been run through it yet — Family 2 of this benchmark suite (see `docs/DESIGN.md`) targets exactly that. Listed here for reference only; **do not compare these numbers directly to the claim-support numbers above** — different corpus (21 repos, 7 languages), different judge, different metric definition entirely. +CodeWikiBench (ACL Findings 2026) publishes its own evaluator and competitor scores, including for [CodeWiki](https://github.com/FSoft-AI4Code/CodeWiki), the paper authors' own generator. doc0 has not been run through it yet — Family 2 of this benchmark suite (see `docs/DESIGN.md`) targets exactly that. Listed here for reference only; **do not compare these numbers directly to the claim-support numbers above** — different corpus, different judge, different metric definition entirely. -| System | CodeWikiBench score | Source | +The published averages below are the paper's Table 1 "Average" row (4 systems, 7 repositories with per-repo detail; the project page describes the same 68.79/64.06 headline as a 21-repo result — a scope discrepancy we cite as-published without resolving, see `docs/codewikibench-pinned.md` §(f)). The Family-2 run reuses their published per-repo rubrics and their exact judge panel (Gemini 2.5 Flash, GPT-OSS-120B, Kimi K2 Instruct, averaged), so doc0's row will be comparable per-repo against every published row, CodeWiki's included, without re-running any peer. + +| System | CodeWikiBench score (avg) | Source | |---|---|---| -| CodeWiki-Sonnet-4 | 68.8% | published, external | -| DeepWiki | 64.1% | published, external | -| doc0 | — | not yet run | +| CodeWiki (Sonnet-4) — the benchmark authors' generator | 68.79% | published, Table 1 | +| DeepWiki | 64.06% | published, Table 1 | +| deepwiki-open | 50.05% | published, Table 1 | +| OpenDeepWiki | 47.13% | published, Table 1 | +| doc0 | — | not yet run (Family 2) | ## Change log From fcd281c1df0cb5e992575b5e74c9c124ad510c19 Mon Sep 17 00:00:00 2001 From: Alessandro Afloarei Date: Wed, 2 Sep 2026 15:18:12 +0000 Subject: [PATCH 3/3] fix(codewikibench): normalize skipped heading levels before hoisting preambles The pinned evaluator parser (markdown_to_json 2.1.2) splits a section on headings at exactly one level below its own and discards every block before the first one it finds. A first child that skips a level (H1 straight to H3, the shape of corpora/redis-2026-07/cluster-routing.md) is never a key: the synthetic "## Overview" swallowed the H3, became childful, and the preamble was dropped again. Placing the synthetic heading at the child's level does not help either, since the parser then skips over both of them together (probed directly against the pinned parser). normalizeHeadingLevels re-levels every heading to exactly one deeper than its nearest shallower predecessor (stack walk, first heading keeps its level), and hoistPreambles runs it first so its one-level-deeper invariant holds. Heading text and order are untouched. On the redis page, parser retention goes from 0.834 to 0.997; the transform is a byte-identical no-op on the other 542 corpus pages. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01HS8GHxjM3NfU2nPk1hwVJx --- runners/codewikibench.test.ts | 58 +++++++++++++++++++++++ runners/codewikibench.ts | 87 +++++++++++++++++++++++++++++------ 2 files changed, 131 insertions(+), 14 deletions(-) diff --git a/runners/codewikibench.test.ts b/runners/codewikibench.test.ts index 05c90ac..80d7bb6 100644 --- a/runners/codewikibench.test.ts +++ b/runners/codewikibench.test.ts @@ -2,6 +2,7 @@ import { describe, it, expect } from "vitest"; import { hoistPreambles, neutralizeNonFinalLists, + normalizeHeadingLevels, planEvaluatorInput, stripLeadingDetailsBlock, transpileMdx, @@ -335,6 +336,63 @@ describe("hoistPreambles", () => { // B is a leaf: untouched. expect(out).toContain("## B\n\nB leaf only."); }); + + // The pinned parser splits each section on headings at EXACTLY parent+1 and + // discards everything before the first one it finds. A first child that + // skips a level (H1 straight to H3, the shape of the committed + // redis cluster-routing page) is never a key, so a synthetic ## Overview + // would swallow it and lose the preamble again, while a synthetic ### would + // be skipped over together with the child. The skip is normalized away + // first, so the preamble lands under a leaf key beside its real sibling. + it("hoists a preamble whose first child skips a level into a leaf sibling", () => { + const out = hoistPreambles("# P\n\nIntro.\n\n### Child\n\nBody.\n\n## Sec\n\nS.\n"); + expect(out).toBe("# P\n\n## Overview\n\nIntro.\n\n## Child\n\nBody.\n\n## Sec\n\nS.\n"); + }); + + it("checks the synthetic title against the promoted sibling", () => { + const out = hoistPreambles("# P\n\nIntro.\n\n### Overview\n\nReal.\n"); + expect(out).toBe("# P\n\n## Introduction\n\nIntro.\n\n## Overview\n\nReal.\n"); + }); +}); + +// --------------------------------------------------------------------------- +// normalizeHeadingLevels +// --------------------------------------------------------------------------- + +describe("normalizeHeadingLevels", () => { + it("promotes a child that skips a level to exactly parent+1", () => { + expect(normalizeHeadingLevels("# P\n\n### Child\n\nBody.\n\n## Sec\n\nS.\n")).toBe( + "# P\n\n## Child\n\nBody.\n\n## Sec\n\nS.\n", + ); + }); + + it("re-levels descendants relative to their promoted parent", () => { + expect(normalizeHeadingLevels("# P\n\n### A\n\n##### A1\n\n## B\n")).toBe( + "# P\n\n## A\n\n### A1\n\n## B\n", + ); + }); + + it("makes a shallower heading after a skip a sibling of the skipped one, never its parent", () => { + // H4 under H2 skips H3; the H3 that follows closes the H4 and is H2's + // child too, so both come out at level 3. + expect(normalizeHeadingLevels("# P\n\n## S\n\n#### Deep\n\nD.\n\n### Next\n\nN.\n")).toBe( + "# P\n\n## S\n\n### Deep\n\nD.\n\n### Next\n\nN.\n", + ); + }); + + it("leaves a well-formed ladder byte-identical", () => { + const md = "# P\n\nIntro.\n\n## A\n\n### A1\n\nLeaf.\n\n## B\n\nB.\n"; + expect(normalizeHeadingLevels(md)).toBe(md); + }); + + it("starts the ladder at the page's first heading level", () => { + expect(normalizeHeadingLevels("## Top\n\n#### Child\n")).toBe("## Top\n\n### Child\n"); + }); + + it("never touches heading-looking lines inside fences", () => { + const md = "# P\n\n```sh\n#### not a heading\n```\n\n### Child\n"; + expect(normalizeHeadingLevels(md)).toBe("# P\n\n```sh\n#### not a heading\n```\n\n## Child\n"); + }); }); describe("neutralizeNonFinalLists", () => { diff --git a/runners/codewikibench.ts b/runners/codewikibench.ts index 407a7f7..02e8bcd 100644 --- a/runners/codewikibench.ts +++ b/runners/codewikibench.ts @@ -313,6 +313,68 @@ export function stripLeadingDetailsBlock(md: string): string { return md.replace(/^(\s*(?:# [^\n]*\n)?\s*)
[\s\S]*?<\/details>\s*/, "$1"); } +// --------------------------------------------------------------------------- +// normalizeHeadingLevels — parser-safe heading ladder +// --------------------------------------------------------------------------- + +const ATX_HEADING_RE = /^(#{1,6})\s+(.*)$/; + +interface HeadingLine { + idx: number; + level: number; + text: string; +} + +/** + * The evaluator's parser (`markdown_to_json`, `_dictify_blocks`) splits a + * section on headings at EXACTLY one level below its own and, when it finds + * any, discards every block before the first one (`dictify_list_by`). A + * heading that skips a level is therefore never a key. Measured on the pinned + * parser (2026-09-02 probe): an H3 directly under an H1 followed by an H2 + * loses the H1's preamble AND the whole H3 section, and H4s under an H2 with + * no H3 flatten into the H2's text. Its own docstring says as much ("if you + * jump ... the high-numbered headings won't be treated as keys"). + * + * The fix re-levels every heading to exactly one deeper than its nearest + * shallower predecessor (a stack walk; the first heading keeps its level, the + * parser roots at the page minimum anyway), so the ladder never skips: + * H1→H3→H2 becomes H1→H2→H2, H2→H4→H3 becomes H2→H3→H3. Heading text and + * order are untouched, and the nesting is exactly what a stack-based reader + * already infers — packaging for their parser, not editing. + * + * `hoistPreambles` runs this first so its "synthetic child one level deeper" + * invariant holds; the exported form exists for direct pinning. + */ +export function normalizeHeadingLevels(md: string): string { + const { working, fences, inlines } = maskCode(md); + const lines = working.split("\n"); + normalizeHeadingLines(lines); + return unmaskCode(lines.join("\n"), fences, inlines); +} + +/** + * Re-levels the ATX headings of a masked line array in place and returns + * them with their normalized levels. + */ +function normalizeHeadingLines(lines: string[]): HeadingLine[] { + const headings: HeadingLine[] = []; + // Ancestors still open: the raw level that opened them, and the level they + // were assigned. + const open: { raw: number; level: number }[] = []; + lines.forEach((line, idx) => { + const m = ATX_HEADING_RE.exec(line); + if (!m) return; + const raw = m[1].length; + while (open.length > 0 && (open[open.length - 1]?.raw ?? 0) >= raw) open.pop(); + const parent = open[open.length - 1]; + const level = parent ? parent.level + 1 : raw; + open.push({ raw, level }); + if (level !== raw) lines[idx] = `${"#".repeat(level)} ${m[2] ?? ""}`; + headings.push({ idx, level, text: (m[2] ?? "").trim() }); + }); + return headings; +} + // --------------------------------------------------------------------------- // hoistPreambles — parser-safe preamble hoisting // --------------------------------------------------------------------------- @@ -323,9 +385,11 @@ export function stripLeadingDetailsBlock(md: string): string { * children, and any loose content between the parent heading and its first * child heading has nowhere to hang — it is silently dropped. Measured * directly against the pinned parser (2026-08-31 probe, see the dry-run - * results): this preamble drop is the ONLY construct lost. Paragraphs, - * ordered and unordered lists, tables, blockquotes, and fences under LEAF - * headings all survive verbatim. + * results): under a well-formed heading ladder this preamble drop is the + * ONLY construct lost. Paragraphs, ordered and unordered lists, tables, + * blockquotes, and fences under LEAF headings all survive verbatim. (A + * skipped heading level is the other loss mode; `normalizeHeadingLevels` + * above removes it first.) * * The fix packages each preamble as a synthetic first child heading one * level deeper (default title "Overview" — the same key shape CodeWiki's own @@ -339,6 +403,11 @@ export function stripLeadingDetailsBlock(md: string): string { * - Fences and inline code are masked first (same `maskCode` as * `transpileMdx`), so a `#` line inside a code sample neither hoists nor * terminates a section. + * - Heading levels are normalized before anything is hoisted: a first child + * that skips a level would otherwise nest under the synthetic heading + * (making it childful, so the parser drops the preamble again), and a + * synthetic heading placed at the child's skipped level would be skipped + * over with it. The sibling-title check runs on the normalized ladder. * - A synthetic title colliding with a real sibling child would merge two * dict keys in their parser, so the title falls back Overview → * Introduction → Preamble → "Overview N". @@ -349,17 +418,7 @@ export function stripLeadingDetailsBlock(md: string): string { export function hoistPreambles(md: string): string { const { working, fences, inlines } = maskCode(md); const lines = working.split("\n"); - - interface HeadingLine { - idx: number; - level: number; - text: string; - } - const headings: HeadingLine[] = []; - lines.forEach((line, idx) => { - const m = /^(#{1,6})\s+(.*)$/.exec(line); - if (m) headings.push({ idx, level: m[1].length, text: (m[2] ?? "").trim() }); - }); + const headings = normalizeHeadingLines(lines); const insertions: { at: number; heading: string }[] = []; for (let i = 0; i < headings.length; i += 1) {