diff --git a/.gitignore b/.gitignore
index 479c85bad..9e6fd563a 100644
--- a/.gitignore
+++ b/.gitignore
@@ -67,6 +67,7 @@ build/
# by `npm run build:lib`). Source of truth is src/; these are emitted, never edited.
# Published via prepublishOnly; built before test via pretest. Grows as modules migrate.
/tsconfig.build.tsbuildinfo
+/gsd-core/bin/lib/markdown-sectionizer.cjs
/gsd-core/bin/lib/research-store.cjs
/gsd-core/bin/lib/research-provider.cjs
/gsd-core/bin/lib/package-legitimacy.cjs
diff --git a/CONTEXT.md b/CONTEXT.md
index 3507c87c4..64840308d 100644
--- a/CONTEXT.md
+++ b/CONTEXT.md
@@ -118,6 +118,9 @@ Primary installer for all runtimes. Single production file: `bin/install.js` (ge
### I/O Module
Module owning the tool's CLI I/O primitives: `output()` result emission (with large-payload temp-file spillover via `GSD_TEMP_DIR`/`ensureGsdTempDir`/`reapStaleTempFiles`), `error()` stderr emission with exit-code mapping, and the JSON-error-mode toggle (`setJsonErrorMode`/`getJsonErrorMode`, `ERROR_REASON`). Extracted from the Core module per ADR-857 rollout phase 1 (#859) so feature modules (`graphify`, `intel`, `audit`, `profile-pipeline`) depend on a small I/O seam instead of the core god-module; the `core.cjs` re-export spine was retired in epic #1267, so callers import this leaf directly. Source of truth: `gsd-core/bin/lib/io.cjs` (generated from `src/io.cts`).
+### Markdown Sectionizer
+Canonical markdown-structure parsing seam (`gsd-core/bin/lib/markdown-sectionizer.cjs`, generated from `src/markdown-sectionizer.cts`). Pure functions, Node built-ins only. Exports: `stripFencedCode(content) → { text, unterminatedFence }` (CommonMark-correct state machine, CRLF-safe, signals unterminated fences); `tokenizeHeadings(content) → HeadingToken[]` (ATX headings outside fenced blocks, `{ level, text, line, offset }`); `collectSections(content, stopPredicate) → Section[]` (line-by-line section collection driven by a heading predicate); `collectSection(content, headingPredicate, { levelBounded, stripFences }) → Section | null` (single named section with level-bounded stop); `iterateBullets(sectionText) → BulletItem[]` (dash/checkbox/numbered markers with indented continuation); `extractTaggedBlocks(content, tagName) → string[]` (inner text of every `…` block in document order, tagName regex-escaped, caller decides fence-stripping — generalises `decisions.cts`'s bespoke extractor for T1); `replaceSection(content, section, newBody) → string` (pure character-offset splice using `Section.bodyStart`/`bodyEnd` for read-modify-write callers — eliminates T6 `state.cts`'s 7× inline `content.replace` pattern). `Section` carries `bodyStart`/`bodyEnd` offsets for `replaceSection`. ADR-1372 (epic #1372) establishes this seam and a tiered migration plan (T0–T7) to retire the 8+ ad-hoc markdown parsers and ~20 inline section-collects across `src/*.cts`. New `src/*.cts` modules must import this seam instead of hand-rolling fence strippers or heading-regex section walks (enforced by the `no-adhoc-markdown-parsing` ESLint rule landing in tier T7).
+
### Roadmap Parser Module
Module owning ROADMAP.md parsing: shipped-milestone slicing, current-milestone extraction, milestone/phase lookups, and milestone-phase filtering (`stripShippedMilestones`, `extractCurrentMilestone`, `replaceInCurrentMilestone`, `getRoadmapPhaseInternal`, `getMilestoneInfo`, `getMilestonePhaseFilter`). Depends only on leaf modules (`phase-id`, `planning-workspace`, `shell-command-projection`) — no `loadConfig`, no other core dependency. Extracted from the Core module per ADR-857 rollout phase 2b (#870), resolving the ROADMAP.md parse/write straddle so the Roadmap module (`roadmap.cjs`, which owns ROADMAP.md mutation) imports parsing directly instead of through Core; the `core.cjs` re-export spine was retired in epic #1267, so callers import this leaf directly. Source of truth: `gsd-core/bin/lib/roadmap-parser.cjs` (generated from `src/roadmap-parser.cts`).
diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json
index 20a230ea3..13ffd1a16 100644
--- a/docs/INVENTORY-MANIFEST.json
+++ b/docs/INVENTORY-MANIFEST.json
@@ -325,6 +325,7 @@
"legacy-cleanup.cjs",
"loop-host-contract.cjs",
"loop-resolver.cjs",
+ "markdown-sectionizer.cjs",
"milestone.cjs",
"model-catalog.cjs",
"model-profiles.cjs",
diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md
index e60245bf4..ca0d7d847 100644
--- a/docs/INVENTORY.md
+++ b/docs/INVENTORY.md
@@ -437,6 +437,7 @@ Full listing: `gsd-core/bin/lib/*.cjs`.
| `legacy-cleanup.cjs` | Detect and remove leftover get-shit-done-cc artifacts; exports `planLegacyCleanup` (pure scan) and `applyLegacyCleanup` (thin IO applier) that root out stale files from the old package across every GSD-managed runtime config directory (#607) |
| `loop-host-contract.cjs` | Generated Loop Host Contract — 12 loop points, per-step agent roles, and core artifacts for the five-step pipeline (discuss/plan/execute/verify/ship); emitted by `scripts/gen-loop-host-contract.cjs --write` (ADR-894 §3); consumed by `gen-capability-registry.cjs` |
| `loop-resolver.cjs` | Loop Extension Point resolver — ADR-857 phase 3c/6 registry-consuming query; given a canonical loop point, filters `byLoopPoint` by resolved Capability State plus config activation (`when` key traversal with prototype-pollution guard), returns `{ point, activeHooks, rendered }` envelope; `resolveLoopHooks` and `renderLoopHooks` are pure (no I/O); command surface: `gsd-tools loop render-hooks [--config-dir ]` |
+| `markdown-sectionizer.cjs` | Canonical markdown-structure parsing seam (ADR-1372, epic #1372) — pure, Node built-ins only; exports `stripFencedCode` (CommonMark-correct fence stripper, CRLF-safe), `tokenizeHeadings` (ATX headings outside fenced blocks), `collectSections`/`collectSection` (line-by-line section collection with `bodyStart`/`bodyEnd` offsets), `iterateBullets` (dash/checkbox/numbered markers), `extractTaggedBlocks` (inner text of `…` blocks, caller decides fence-stripping), and `replaceSection` (pure character-offset body splice for read-modify-write callers); foundation for T0–T7 migration tiers retiring 8+ ad-hoc parsers |
| `milestone.cjs` | Milestone archival, requirements marking |
| `model-catalog.cjs` | CJS adapter over the shared model catalog JSON; exports canonical runtime tier defaults, agent profile maps, alias maps, and routing metadata for all CLI consumers |
| `model-profiles.cjs` | Backward-compatible profile helpers derived from `model-catalog.cjs`; no longer owns its own model table |
diff --git a/docs/adr/1372-markdown-sectionizer-seam.md b/docs/adr/1372-markdown-sectionizer-seam.md
new file mode 100644
index 000000000..c41047e48
--- /dev/null
+++ b/docs/adr/1372-markdown-sectionizer-seam.md
@@ -0,0 +1,92 @@
+# ADR-1372: Canonical markdown-structure parsing — the `markdown-sectionizer` seam
+
+- **Status:** Accepted
+- **Date:** 2026-06-17
+- **Issue:** [#1372](https://github.com/open-gsd/gsd-core/issues/1372) (epic)
+- **Resolves (via tier T1):** [#1364](https://github.com/open-gsd/gsd-core/issues/1364), [#1365](https://github.com/open-gsd/gsd-core/issues/1365)
+- **Relates:** [#1343](https://github.com/open-gsd/gsd-core/issues/1343), [#1324](https://github.com/open-gsd/gsd-core/issues/1324), [#447](https://github.com/open-gsd/gsd-core/issues/447) — prior single-parser markdown bugs
+- **Pattern precedent:** [ADR-857](857-capability-system.md) / epic [#1267](https://github.com/open-gsd/gsd-core/issues/1267) (retire a duplicated spine via tiered children)
+
+## Context
+
+GSD parses a lot of structured markdown — `CONTEXT.md`, `ROADMAP.md`, `STATE.md`, `*-PLAN.md`, UAT files, ADRs, frontmatter. There is **no shared primitive** for the three operations every one of these parsers needs (strip fenced code, tokenize headings into sections, iterate bullets), so each module hand-rolls them. A grounded map of `src/*.cts` found:
+
+- **8+ independent markdown parsers**: `decisions`, `gap-checker`, `roadmap-parser`, `state`, `uat`, `uat-predicate`, `adr-parser`, `check-command-router`.
+- **3–4 independent fenced-code strippers of different fidelity**: `decisions.cts` (fragile regex, no unclosed-fence handling), `roadmap-parser.cts` `stripFencedLines` (state machine, **duplicated 3× in one file**), `uat-predicate.cts` `_stripFencedBlocks` (CommonMark-correct, CRLF-safe, signals an unterminated fence), `check-command-router.cts` `stripCommentsAndFences` (another regex copy).
+- **~20 hand-rolled section-collects**, with `state.cts` alone re-implementing the same `/(###?\s*\s*\n)([\s\S]*?)(?=\n###?|$)/i` shape **13 times**.
+
+The consequence is a recurring maintenance game: every "the parser missed structure X" report (#1343 bullet-before-colon, #1364 markdown-header + em-dash, #1324 glued phase tokens, #447 gap scoping) is fixed *locally* with another regex, and the same class of bug re-opens in the next parser. The fixes do not compound — they accrete. Worse, in the decision-coverage case the failure mode is **silent**: a blocking gate that cannot parse its input reports `passed:true, covered 0/0` and ships the phase with its decisions unchecked.
+
+There are two root causes, and a durable fix must address both:
+
+1. **No canonical structure primitive** — so structural correctness (fences, CRLF, heading levels, Unicode, bullet shapes) is re-litigated per module and tested unevenly.
+2. **Nothing prevents the next ad-hoc parser** — a new PR can add a fourth fence stripper and no gate objects, so the divergence regrows even after a cleanup.
+
+## Decision
+
+Establish a single canonical markdown-structure seam and make ad-hoc markdown scanning a lint-enforced prohibition. Migrate every existing parser onto the seam incrementally, tracked as tiered children of epic #1372.
+
+### 1. The seam — `src/markdown-sectionizer.cts` (pure, Node built-ins only)
+
+No external markdown library (the "no external dependencies in core" rule stands). Pure functions, string-in → value-out, no I/O:
+
+- `stripFencedCode(content) → { text, unterminatedFence }` — the CommonMark-correct state machine promoted from `uat-predicate.cts` `_stripFencedBlocks` (CRLF-safe; ≤3-space indent tolerated; closes only on a same-or-longer fence run). `unterminatedFence` is a reusable malformed-input diagnostic.
+- `tokenizeHeadings(content) → HeadingToken[]` — ATX headings `{ level, text, line, offset }` in document order.
+- `collectSections(content, stopPredicate)` and `collectSection(content, headingPredicate, { levelBounded, stripFences })` — line-by-line (not greedy-regex) section collection; `levelBounded` encodes the dominant "stop at same-or-higher-level heading" pattern. Both populate `bodyStart`/`bodyEnd` character offsets on the returned `Section` for use by `replaceSection`.
+- `iterateBullets(sectionText) → BulletItem[]` — dash/asterisk/plus, checkbox (`- [ ]`/`- [x]`), and numbered markers, with indented continuation-line accumulation.
+- `extractTaggedBlocks(content, tagName) → string[]` — returns the inner text of every `…` block in document order; `tagName` is regex-escaped; the caller decides ordering (does not strip fences). Generalises `decisions.cts`'s bespoke `` extractor for T1 adoption.
+- `replaceSection(content, section, newBody) → string` — pure character-offset splice using `section.bodyStart`/`bodyEnd`; replaces a section body in a read-modify-write workflow (e.g. `state.cts`'s 7× inline `content.replace(/(##\s*Name\s*\n)([\s\S]*?)(?=\n##|$)/, ...)` pattern). CRLF-safe.
+
+The seam is fully tested against the parser QA matrix (CRLF, Unicode headings, headings-inside-fences, unterminated fences, nested levels, malformed bullets) **once**, so every adopter inherits that correctness instead of re-deriving it.
+
+### 2. Prohibition + enforcement — `local/no-adhoc-markdown-parsing`
+
+A new ESLint rule in `eslint-rules/no-adhoc-markdown-parsing.cjs` (wired in `eslint.config.mjs`, mirroring `local/no-source-grep`) flags new hand-rolled markdown-structure scanning outside the seam — fenced-code strip regexes, `split(/\r?\n/)` + heading-regex section walks, and `D-`/checkbox bullet regexes — in `src/*.cts`. Existing sites are **grandfathered** by an explicit allowlist that is burned down as each tier migrates (the same grandfathering pattern `no-source-grep` uses). New code must import the seam. This is the part that stops the game permanently: after this rule lands, a PR cannot introduce a fourth fence stripper without a reviewer-visible failure.
+
+### 3. Decisions realization (tier T1) — typed result + fail-loud gate
+
+The first behavioral adopter, which also resolves the two open bugs. `decisions.cts` is rewritten onto the seam, and a typed result distinguishes the states the blocking gate cares about:
+
+```
+type DecisionExtraction = {
+ decisions: Decision[];
+ outcome: 'parsed' | 'none-present' | 'could-not-parse';
+};
+```
+
+`parseDecisions(content): Decision[]` is preserved as a thin delegate (consumers untouched); `extractDecisions(content): DecisionExtraction` is the typed entry point. `cmdDecisionCoveragePlan` (blocking) treats `could-not-parse` — content is decision-shaped (a `` block, a `/decisions?/i` heading, `\bD-` tokens, or `unterminatedFence`) yet 0 decisions extracted — as a **WARN/fail** ("could not parse decisions — possible format mismatch") instead of a green pass (**resolves #1365**). Routing through the seam recognises the markdown-header + em-dash variants (**resolves #1364**). Recall-first by design: a false "could-not-parse" is a loud warning a human clears; a false "none-present" is the silent bypass we are deleting.
+
+### 4. Migration tiers (epic #1372 children)
+
+Each tier is its own issue + PR (issue-first; one concern per PR), behaviour-preserving except T1, each separately tested, each burning down the `no-adhoc-markdown-parsing` grandfather list for the files it touches.
+
+| Tier | Scope | Risk | Notes |
+|---|---|---|---|
+| **T0** | Seam foundation: `markdown-sectionizer.cts` + QA-matrix tests | none | No migration, no behavior change. Foundational. |
+| **T1** | `decisions.cts` + coverage gate: adopt seam, typed result, fail-loud | low–med | **Resolves #1364, #1365.** First behavioral adopter. |
+| **T2** | `adr-parser.cts`: `parseSections`/`splitEntries` → seam | none | CLI-only, no in-process callers — the safe prototype; its `parseSections` is the API shape the seam generalizes. |
+| **T3** | `check-command-router.cts` + `gap-checker.cts`: dedupe `stripCommentsAndFences`, designated-section walk, requirements bullets | low | Gate-adjacent; covered by existing gate tests. |
+| **T4** | `roadmap-parser.cts`: collapse the 3× inline fence loop + `computeSectionEnd` | med | Heavily tested; watch milestone-section boundaries. |
+| **T5** | `uat.cts` + `uat-predicate.cts`: donate the canonical stripper, migrate heading/section scans | med | `_stripFencedBlocks` becomes the seam's source in T0; T5 removes the local copy. |
+| **T6** | `state.cts`: 13 inline section-collects → `collectSection` | high | Highest payoff, highest risk — load-bearing for STATE.md mutation. Surgical, full regression, last. |
+| **T7** | Enforcement: `no-adhoc-markdown-parsing` ESLint rule + grandfather burn-down | low | Lands once enough tiers are migrated that the grandfather list is small; thereafter new ad-hoc parsing is blocked. |
+
+`frontmatter.cts` stays as-is — YAML frontmatter is a different grammar with its own well-used shared parser (`extractFrontmatter`); it is out of scope.
+
+## Backward compatibility
+
+No user-facing or authoring change. Behaviour-preserving migrations (T2–T6) keep each parser's outputs byte-identical (verified by each parser's existing tests + added characterization tests). T1 is the only behavior change: additive decision recall + the could-not-parse WARN; the `` block stays canonical and parses identically (block presence still takes precedence). Internal API churn is contained per-tier; public CLI contracts are unchanged.
+
+## Consequences
+
+**Positive:** structural correctness (fences/CRLF/levels/bullets) is solved and tested once; the silent fail-open class is eliminated for the blocking gate; the per-module regex pile stops growing *and* is prohibited from regrowing; future markdown parsers inherit correctness for free; the change models the repo's own typed-IR / no-source-grep philosophy. Retires 3–4 duplicate strippers and ~20 inline section-collects.
+
+**Negative / risks:** a large surface migrated incrementally — mitigated by tiering (zero-risk T2 prototype first, high-risk `state.cts` last, behavior-preserving with characterization tests, the epic visible end-to-end). A new shared module is a dependency for adopters — mitigated by purity + exhaustive tests. The recall-first "could-not-parse" heuristic may occasionally warn on decision-shaped-but-empty content — acceptable and tunable; a loud false alarm beats the silent miss it replaces. The enforcement rule (T7) must grandfather precisely to avoid blocking unrelated PRs mid-migration.
+
+## Alternatives considered
+
+- **Point-fix each parser bug as it's reported (status quo).** Rejected — this is the game we are ending; fixes accrete instead of compounding and the same class recurs in the next parser. The maintainer's explicit directive is a solution-wide structural fix, not another file edit.
+- **Consolidate the primitive but skip the enforcement rule.** Rejected — without the lint guard the divergence regrows; the next PR adds a fifth stripper and no gate objects. The prohibition is what makes the consolidation durable.
+- **External markdown library (remark/markdown-it/unified).** Rejected — "no external dependencies in core" is a hard rule.
+- **LLM / semantic extraction.** Rejected — `gsd-tools` is a deterministic, no-LLM, zero-dependency CLI with regression-tested pure `Result` functions; an LLM breaks the determinism/testability a CI gate requires and contradicts the repo's no-LLM precedent.
+- **One big-bang PR migrating every parser.** Rejected — `state.cts` alone is load-bearing and high-risk; a single PR would be unreviewable and unmergeable. Gall's Law: the working complex system is grown from a working simple seam (T0) plus incremental, individually-verified migrations.
diff --git a/eslint.config.mjs b/eslint.config.mjs
index f3d772670..957b594ac 100644
--- a/eslint.config.mjs
+++ b/eslint.config.mjs
@@ -160,6 +160,8 @@ export default tseslint.config(
'gsd-core/bin/lib/capability-writer.cjs',
// issue #1355: tsc-generated runtime artifact — lint the src/teams-status.cts source.
'gsd-core/bin/lib/teams-status.cjs',
+ // ADR-1372: tsc-generated runtime artifact — lint the src/markdown-sectionizer.cts source.
+ 'gsd-core/bin/lib/markdown-sectionizer.cjs',
],
},
diff --git a/src/markdown-sectionizer.cts b/src/markdown-sectionizer.cts
new file mode 100644
index 000000000..0665ff39c
--- /dev/null
+++ b/src/markdown-sectionizer.cts
@@ -0,0 +1,585 @@
+/**
+ * Markdown Sectionizer — canonical markdown-structure parsing seam
+ *
+ * Pure functions, Node built-ins only (no external deps). String-in → value-out, no I/O.
+ * Promoted from `uat-predicate.cts` `_stripFencedBlocks` (CommonMark-correct state machine)
+ * and extended with heading tokenisation, section collection, and bullet iteration.
+ *
+ * ADR-1372 — T0 foundational seam. Migration tiers T1–T7 progressively adopt this seam.
+ *
+ * ADR-457 build-at-publish: compiled by tsc to gsd-core/bin/lib/markdown-sectionizer.cjs.
+ */
+
+// ─── Types ────────────────────────────────────────────────────────────────────
+
+/** Result of stripping fenced code blocks from markdown content. */
+export interface StripFencedResult {
+ /** Content with all fenced code blocks removed (delimiters and body lines). */
+ text: string;
+ /**
+ * True when the input contained an unterminated fence (EOF inside a fence).
+ * Callers that wish to signal malformed input to the user should inspect this.
+ */
+ unterminatedFence: boolean;
+}
+
+/** An ATX heading extracted by `tokenizeHeadings`. */
+export interface HeadingToken {
+ /** Heading depth: 1 = `#`, 2 = `##`, 3 = `###`, etc. */
+ level: number;
+ /** Heading text with surrounding whitespace trimmed. */
+ text: string;
+ /** 1-based line number of the heading in the original content. */
+ line: number;
+ /** Character (string-index) offset of the `#` character in the original content string. */
+ offset: number;
+}
+
+/** A collected markdown section (heading + body). */
+export interface Section {
+ /** The heading that opened this section. */
+ heading: HeadingToken;
+ /** All lines between this heading and the next stop, joined by `\n`. */
+ body: string;
+ /**
+ * Character (string-index) offset in the ORIGINAL content string where the
+ * section body begins (first character after the heading line's trailing newline).
+ * Populated by `collectSections` and `collectSection`.
+ * Used by `replaceSection` for a clean pure splice.
+ *
+ * INVARIANT: `content.slice(bodyStart, bodyEnd) === body` for every Section
+ * returned by `collectSection` and `collectSections`.
+ */
+ bodyStart: number;
+ /**
+ * Character (string-index) offset in the ORIGINAL content string where the
+ * section body ends (exclusive). Because `body` is `trimEnd()`-ed, this equals
+ * `bodyStart + body.length` — NOT the start of the next heading line.
+ *
+ * INVARIANT: `content.slice(bodyStart, bodyEnd) === body`.
+ * This guarantees `replaceSection(content, section, section.body) === content`.
+ */
+ bodyEnd: number;
+}
+
+/** Recognised bullet markers. */
+export type BulletMarker = 'dash' | 'checkbox-unchecked' | 'checkbox-checked' | 'numbered';
+
+/** A single bullet item from `iterateBullets`. */
+export interface BulletItem {
+ /** Which marker shape was recognised. */
+ marker: BulletMarker;
+ /** Full bullet text including all indented continuation lines, whitespace-trimmed. */
+ text: string;
+ /** Raw indentation prefix of the opening bullet line. */
+ indent: string;
+ /** Checkbox state — `true` for `[x]`, `false` for `[ ]`, `null` for non-checkbox. */
+ checked: boolean | null;
+}
+
+// ─── Internal types ───────────────────────────────────────────────────────────
+
+interface FenceState {
+ char: '`' | '~';
+ len: number;
+}
+
+// ─── stripFencedCode ──────────────────────────────────────────────────────────
+
+/**
+ * CommonMark-correct fenced-code-block stripper.
+ *
+ * Ported from `uat-predicate.cts` `_stripFencedBlocks` — the reference
+ * implementation for the repo. DO NOT modify `uat-predicate.cts` (its
+ * migration is T5); this is a tracked duplication until T5 lands.
+ *
+ * Rules:
+ * - Opening delimiter: a line whose non-indent portion begins with ≥3 backticks
+ * or tildes (≤3 leading spaces tolerated per CommonMark §4.5).
+ * - Closing delimiter: same character, run length ≥ opening, no trailing
+ * non-whitespace text.
+ * - A tilde fence inside a backtick fence (or vice versa) is fence *content*,
+ * not a closing delimiter — delimiter char must match.
+ * - Both delimiter lines and all content lines are dropped from the output.
+ * - CRLF-safe: trailing `\r` is stripped before delimiter matching; the kept
+ * non-fence lines are returned as-is (including any `\r`).
+ * - `unterminatedFence` signals EOF inside an open fence.
+ */
+export function stripFencedCode(content: string): StripFencedResult {
+ if (typeof content !== 'string') {
+ return { text: '', unterminatedFence: false };
+ }
+ const lines = content.split('\n');
+ const kept: string[] = [];
+ let openFence: FenceState | null = null;
+
+ // Matches: optional indent (≤3 spaces per CommonMark), fence run, optional info string
+ const delimRe = /^( {0,3})(`{3,}|~{3,})(.*)$/;
+
+ for (const rawLine of lines) {
+ // Strip trailing \r for delimiter matching (CRLF safety)
+ const line = rawLine.replace(/\r$/, '');
+ const m = delimRe.exec(line);
+ if (m) {
+ const char = m[2][0] as '`' | '~';
+ const len = m[2].length;
+ const trailing = m[3];
+ if (openFence === null) {
+ // CommonMark §4.5: backtick fence info string must not contain a backtick.
+ // If it does, this line is NOT a valid fence opener (treat as ordinary content).
+ if (char === '`' && trailing.includes('`')) {
+ kept.push(rawLine);
+ continue;
+ }
+ // Opening delimiter — record fence state, drop this line
+ openFence = { char, len };
+ } else if (char === openFence.char && len >= openFence.len && /^\s*$/.test(trailing)) {
+ // Closing delimiter (same char, sufficient length, no trailing content) — close and drop
+ openFence = null;
+ }
+ // else: mismatched delimiter inside fence — treat as content, still drop (it's a fence line)
+ continue; // all delimiter lines are dropped
+ }
+
+ if (openFence === null) {
+ kept.push(rawLine); // non-fence content: keep as-is (preserve original \r if any)
+ }
+ // Lines inside a fence are silently dropped
+ }
+
+ return { text: kept.join('\n'), unterminatedFence: openFence !== null };
+}
+
+// ─── tokenizeHeadings ─────────────────────────────────────────────────────────
+
+/**
+ * Extract all ATX headings from `content` in document order.
+ *
+ * Only headings OUTSIDE fenced code blocks are returned — `stripFencedCode` is
+ * applied first so that a `## heading` inside a ``` fence is not tokenised.
+ *
+ * Each token records `{ level, text, line, offset }` where `offset` is relative
+ * to the ORIGINAL `content` (before fence-stripping), enabling callers to use
+ * `collectSection` on the original string.
+ */
+export function tokenizeHeadings(content: string): HeadingToken[] {
+ if (typeof content !== 'string' || content.length === 0) return [];
+
+ // Strip fences first so headings inside code blocks are ignored.
+ // We need the original line positions, so we map stripped-text line numbers
+ // back to original by tracking which original lines survived stripping.
+ const originalLines = content.split('\n');
+ const tokens: HeadingToken[] = [];
+
+ // We re-run the fence state machine to know which lines are "kept", so we
+ // can map line index in original to whether it survived.
+ const delimRe = /^( {0,3})(`{3,}|~{3,})(.*)$/;
+ let openFence: FenceState | null = null;
+
+ // Accumulate character offset as we iterate lines
+ let charOffset = 0;
+
+ for (let i = 0; i < originalLines.length; i++) {
+ const rawLine = originalLines[i];
+ const line = rawLine.replace(/\r$/, '');
+
+ const dm = delimRe.exec(line);
+ if (dm) {
+ const char = dm[2][0] as '`' | '~';
+ const len = dm[2].length;
+ const trailing = dm[3];
+ if (openFence === null) {
+ // CommonMark §4.5: backtick fence info string must not contain a backtick.
+ if (char === '`' && trailing.includes('`')) {
+ // Not a valid fence opener — check for heading on this line (will fall through)
+ } else {
+ openFence = { char, len };
+ charOffset += rawLine.length + 1;
+ continue;
+ }
+ } else if (char === openFence.char && len >= openFence.len && /^\s*$/.test(trailing)) {
+ openFence = null;
+ charOffset += rawLine.length + 1;
+ continue;
+ } else {
+ // Mismatched/invalid delimiter inside fence — treat as content (still inside fence), skip heading check
+ charOffset += rawLine.length + 1;
+ continue;
+ }
+ }
+
+ if (openFence === null) {
+ // This line is outside any fence — check for ATX heading.
+ // CommonMark: ≤3 leading spaces, then 1–6 `#`, then either EOF (empty heading)
+ // or at least one space/tab followed by optional text, with optional closing `#` sequence.
+ const headingMatch = /^( {0,3})(#{1,6})([ \t]+.*|[ \t]*)?$/.exec(line);
+ if (headingMatch) {
+ const hashes = headingMatch[2];
+ const rest = headingMatch[3] ?? '';
+ // Strip optional closing `#` sequence: trailing whitespace + one or more `#` + optional whitespace
+ const rawText = rest.replace(/^[ \t]+/, '').replace(/[ \t]+#+[ \t]*$/, '').replace(/^#+[ \t]*$/, '');
+ tokens.push({
+ level: hashes.length,
+ text: rawText.trim(),
+ line: i + 1, // 1-based
+ offset: charOffset,
+ });
+ }
+ }
+
+ charOffset += rawLine.length + 1;
+ }
+
+ return tokens;
+}
+
+// ─── collectSections ─────────────────────────────────────────────────────────
+
+/**
+ * Collect sections from `content`, calling `stopPredicate` on each heading to
+ * decide where sections end.
+ *
+ * Returns an array of `Section` objects, one per matched heading. The `body`
+ * of each section runs from the line after the heading up to (but not
+ * including) the next heading that satisfies `stopPredicate`, or EOF.
+ *
+ * Unlike a greedy-regex approach, this is a line-by-line walk — compatible
+ * with the repo's "line-by-line section collection" pattern.
+ */
+export function collectSections(
+ content: string,
+ stopPredicate: (heading: HeadingToken) => boolean,
+): Section[] {
+ if (typeof content !== 'string' || content.length === 0) return [];
+
+ const headings = tokenizeHeadings(content);
+ if (headings.length === 0) return [];
+
+ const lines = content.split('\n');
+ const sections: Section[] = [];
+
+ // Build a set of line numbers (1-based) that are heading lines
+ const headingsByLine = new Map();
+ for (const h of headings) {
+ headingsByLine.set(h.line, h);
+ }
+
+ // Build a byte-offset table: lineOffsets[i] = byte offset of the start of line i+1 (1-based: i=0 → line 1)
+ // The body of a section starts at the byte after the heading line's trailing '\n'.
+ const lineOffsets: number[] = new Array(lines.length);
+ let acc = 0;
+ for (let i = 0; i < lines.length; i++) {
+ lineOffsets[i] = acc;
+ acc += lines[i].length + 1; // +1 for the '\n' we split on
+ }
+ // lineOffsets[i] is the byte offset of line (i+1) (1-based). EOF sentinel:
+ const eofOffset = acc; // === content.length + (content.endsWith('\n') ? 0 : 0) ≈ content.length
+
+ let currentHeading: HeadingToken | null = null;
+ let currentBodyStart = 0;
+ let bodyLines: string[] = [];
+
+ const flush = (_bodyEndOffset: number): void => {
+ if (currentHeading !== null) {
+ const rawBody = bodyLines.join('\n');
+ const body = rawBody.trimEnd();
+ // INVARIANT: content.slice(bodyStart, bodyEnd) === body
+ // bodyEnd is derived from body.length, NOT from the raw separator offset,
+ // so round-trips via replaceSection(content, section, section.body) are exact.
+ sections.push({
+ heading: currentHeading,
+ body,
+ bodyStart: currentBodyStart,
+ bodyEnd: currentBodyStart + body.length,
+ });
+ currentHeading = null;
+ bodyLines = [];
+ }
+ };
+
+ for (let i = 0; i < lines.length; i++) {
+ const lineNo = i + 1; // 1-based
+ const h = headingsByLine.get(lineNo);
+ if (h !== undefined && stopPredicate(h)) {
+ // This heading is a stop boundary — flush current section, start new one.
+ // The body ends at the start of this heading line.
+ flush(lineOffsets[i]);
+ currentHeading = h;
+ // Body starts at the beginning of the line AFTER the heading line
+ const headingLineIdx = h.line - 1; // 0-based
+ currentBodyStart = lineOffsets[headingLineIdx] + lines[headingLineIdx].length + 1;
+ } else if (currentHeading !== null) {
+ bodyLines.push(lines[i]);
+ }
+ }
+ flush(eofOffset);
+
+ return sections;
+}
+
+// ─── collectSection ───────────────────────────────────────────────────────────
+
+/**
+ * Collect a single section whose heading satisfies `headingPredicate`.
+ *
+ * Options:
+ * - `levelBounded` (default: `true`): the section ends at the next heading of
+ * the same or higher level (lower level number = higher in the hierarchy).
+ * When `false`, the section body runs until any heading or EOF.
+ * Ignored when `stopAtLevel` is provided.
+ * - `stopAtLevel` (optional): when provided, the section ends at the next heading
+ * whose `level <= stopAtLevel`, regardless of the opener's level. This enables
+ * modeling sections like a `##`-opened section that also stops at `###`
+ * (pass `stopAtLevel: 3`). Takes precedence over `levelBounded` when set.
+ * - `stripFences` (default: `false`): apply `stripFencedCode` to the body
+ * before returning. The `heading` in the result always refers to the original
+ * heading (pre-strip).
+ *
+ * Returns `null` when no matching heading is found.
+ */
+export function collectSection(
+ content: string,
+ headingPredicate: (heading: HeadingToken) => boolean,
+ opts: { levelBounded?: boolean; stopAtLevel?: number; stripFences?: boolean } = {},
+): Section | null {
+ if (typeof content !== 'string' || content.length === 0) return null;
+
+ const { levelBounded = true, stopAtLevel, stripFences = false } = opts;
+
+ const headings = tokenizeHeadings(content);
+ const targetIdx = headings.findIndex(headingPredicate);
+ if (targetIdx === -1) return null;
+
+ const target = headings[targetIdx];
+ const lines = content.split('\n');
+
+ // Determine which headings act as stops after the target
+ const bodyStartLine = target.line + 1; // 1-based, first line of body
+ let bodyEndLine = lines.length + 1; // 1-based, exclusive (default: EOF+1)
+
+ for (let j = targetIdx + 1; j < headings.length; j++) {
+ const next = headings[j];
+ let isStop: boolean;
+ if (stopAtLevel !== undefined) {
+ // stopAtLevel: stop at the next heading whose level <= stopAtLevel
+ isStop = next.level <= stopAtLevel;
+ } else {
+ isStop = levelBounded ? next.level <= target.level : true;
+ }
+ if (isStop) {
+ bodyEndLine = next.line; // stop before this line (1-based)
+ break;
+ }
+ }
+
+ // Compute character offsets for bodyStart.
+ // lineOffsets[i] = character offset of line (i+1) in content (1-based).
+ const lineOffsets: number[] = new Array(lines.length);
+ let acc = 0;
+ for (let i = 0; i < lines.length; i++) {
+ lineOffsets[i] = acc;
+ acc += lines[i].length + 1; // +1 for the '\n' separator
+ }
+ const eofOffset = acc; // byte offset past the last line
+
+ // bodyStart: character offset of first line of body (bodyStartLine is 1-based)
+ const bodyStartOffset = bodyStartLine <= lines.length ? lineOffsets[bodyStartLine - 1] : eofOffset;
+
+ // Slice body lines (0-based array: bodyStartLine-1 to bodyEndLine-2 inclusive)
+ const bodyRaw = lines.slice(bodyStartLine - 1, bodyEndLine - 1).join('\n').trimEnd();
+ const body = stripFences ? stripFencedCode(bodyRaw).text : bodyRaw;
+
+ // INVARIANT: content.slice(bodyStart, bodyEnd) === body
+ // bodyEnd is derived from body.length so that replaceSection(content, section, section.body) === content.
+ return { heading: target, body, bodyStart: bodyStartOffset, bodyEnd: bodyStartOffset + body.length };
+}
+
+// ─── iterateBullets ───────────────────────────────────────────────────────────
+
+/**
+ * Extract bullet items from `sectionText`.
+ *
+ * Recognises three marker families:
+ * - **Checkbox**: `- [ ] text` (unchecked) and `- [x] text` / `- [X] text` (checked)
+ * - **Dash**: `- text`, `* text`, `+ text` (plain unordered list item)
+ * - **Numbered**: `1. text`, `42. text` (ordered list item)
+ *
+ * Indented continuation lines (lines that are not themselves bullet openers and
+ * have at least one leading space or tab) are accumulated into the current
+ * bullet's `text`.
+ *
+ * Blank lines terminate the current bullet (consistent with CommonMark block
+ * handling and the repo's existing bullet parsers).
+ */
+export function iterateBullets(sectionText: string): BulletItem[] {
+ if (typeof sectionText !== 'string' || sectionText.length === 0) return [];
+
+ const lines = sectionText.split('\n');
+ const items: BulletItem[] = [];
+
+ // Checkbox bullet: `- [ ] text` or `- [x] text`
+ const checkboxRe = /^(\s*)- \[([xX ])\] (.*)$/;
+ // Plain dash/asterisk/plus bullet: `- text`, `* text`, `+ text`
+ const dashRe = /^(\s*)[-*+] (.*)$/;
+ // Numbered bullet: `1. text`
+ const numberedRe = /^(\s*)\d+\. (.*)$/;
+ // Continuation: non-empty, indented, NOT a bullet opener
+ const continuationRe = /^[ \t]/;
+
+ let current: BulletItem | null = null;
+
+ const flush = (): void => {
+ if (current !== null) {
+ current.text = current.text.trim();
+ items.push(current);
+ current = null;
+ }
+ };
+
+ for (const rawLine of lines) {
+ // Strip trailing \r (CRLF safety)
+ const line = rawLine.replace(/\r$/, '');
+ const trimmed = line.trim();
+
+ // Blank line terminates current bullet
+ if (trimmed === '') {
+ flush();
+ continue;
+ }
+
+ // Checkbox bullet (checked or unchecked) — must test before dashRe
+ const cbm = checkboxRe.exec(line);
+ if (cbm) {
+ flush();
+ const stateChar = cbm[2];
+ const checked = stateChar === 'x' || stateChar === 'X';
+ current = {
+ marker: checked ? 'checkbox-checked' : 'checkbox-unchecked',
+ text: cbm[3],
+ indent: cbm[1],
+ checked,
+ };
+ continue;
+ }
+
+ // Numbered bullet
+ const nm = numberedRe.exec(line);
+ if (nm) {
+ flush();
+ current = {
+ marker: 'numbered',
+ text: nm[2],
+ indent: nm[1],
+ checked: null,
+ };
+ continue;
+ }
+
+ // Plain dash / asterisk / plus bullet
+ const dm = dashRe.exec(line);
+ if (dm) {
+ flush();
+ current = {
+ marker: 'dash',
+ text: dm[2],
+ indent: dm[1],
+ checked: null,
+ };
+ continue;
+ }
+
+ // Continuation line (indented, non-bullet) — append to current bullet
+ if (current !== null && continuationRe.test(line)) {
+ current.text += ' ' + trimmed;
+ continue;
+ }
+
+ // Non-bullet, non-continuation line (e.g. a paragraph, heading) — flush
+ flush();
+ }
+ flush();
+
+ return items;
+}
+
+// ─── extractTaggedBlocks ──────────────────────────────────────────────────────
+
+/**
+ * Return the inner text of every `…` block in `content`,
+ * in document order.
+ *
+ * Designed for extracting structured XML-like annotation blocks that live in
+ * markdown prose (e.g. `…`, `…`).
+ * Returns `[]` when no matching blocks are found.
+ *
+ * The `tagName` argument is regex-escaped, so names that contain regex
+ * metacharacters (e.g. `foo.bar`, `my+tag`) are matched literally.
+ *
+ * **Input contract:** the caller decides whether to pass raw or fence-stripped
+ * content. `extractTaggedBlocks` is a pure block extractor — it does NOT strip
+ * fenced code blocks itself. If a `` block appears inside a fenced code
+ * block and should be excluded, the caller should apply `stripFencedCode` first.
+ *
+ * **Nested tags are NOT supported.** The underlying regex uses a non-greedy
+ * `[\s\S]*?` match, which means it closes at the FIRST `` encountered.
+ * Given `inner`, `extractTaggedBlocks(content, 'x')` returns
+ * `['inner']` — the inner `` is captured as literal text, and the second
+ * `` is left unmatched (or matched as a second block with empty inner text
+ * if another `` follows). Callers that need to handle nested tags must
+ * pre-process the input or use a proper XML/HTML parser.
+ *
+ * Generalises `decisions.cts`'s bespoke `matchAll(/([\s\S]*?)<\/decisions>/g)`
+ * so tier T1 can drop its own copy (tracked duplication until T1 lands).
+ */
+export function extractTaggedBlocks(content: string, tagName: string): string[] {
+ if (typeof content !== 'string' || content.length === 0) return [];
+ if (typeof tagName !== 'string' || tagName.length === 0) return [];
+
+ // Escape the tag name for safe interpolation into a RegExp.
+ const escapedTag = tagName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
+ const pattern = new RegExp(`<${escapedTag}>([\\s\\S]*?)${escapedTag}>`, 'g');
+
+ const results: string[] = [];
+ let match: RegExpExecArray | null;
+ while ((match = pattern.exec(content)) !== null) {
+ results.push(match[1]);
+ }
+ return results;
+}
+
+// ─── replaceSection ───────────────────────────────────────────────────────────
+
+/**
+ * Splice `newBody` in place of a section's body and return the resulting
+ * full content string.
+ *
+ * Uses the `bodyStart`/`bodyEnd` character offsets carried by the `Section`
+ * type to perform a pure string splice — no regex, no line-counting. The
+ * heading is preserved verbatim; only the bytes between `bodyStart` and
+ * `bodyEnd` are replaced.
+ *
+ * The `newBody` is inserted as-is between `content.slice(0, bodyStart)` and
+ * `content.slice(bodyEnd)`. If `newBody` should end with a trailing newline
+ * before the next section's heading, the caller is responsible for including
+ * it (consistent with how `trimEnd()` is applied to collected bodies — see
+ * `collectSections`/`collectSection`).
+ *
+ * Typical read-modify-write pattern (T6 state.cts use case):
+ * ```
+ * const section = collectSection(content, h => h.text === 'Name');
+ * if (section) {
+ * content = replaceSection(content, section, newBody);
+ * }
+ * ```
+ *
+ * CRLF-safe: the splice is purely character-offset-based, so CRLF sequences
+ * are preserved in the surrounding content unchanged.
+ */
+export function replaceSection(content: string, section: Section, newBody: string): string {
+ if (typeof content !== 'string') return content;
+ if (typeof newBody !== 'string') return content;
+ return content.slice(0, section.bodyStart) + newBody + content.slice(section.bodyEnd);
+}
+
+// Consumers: require('../gsd-core/bin/lib/markdown-sectionizer.cjs')
+// Named CJS exports are the canonical surface (ADR-457 .cts → .cjs build-at-publish).
diff --git a/tests/markdown-sectionizer.test.cjs b/tests/markdown-sectionizer.test.cjs
new file mode 100644
index 000000000..8a0118ed7
--- /dev/null
+++ b/tests/markdown-sectionizer.test.cjs
@@ -0,0 +1,1142 @@
+'use strict';
+
+/**
+ * Behavioral tests for markdown-sectionizer.cjs
+ *
+ * Module: gsd-core/bin/lib/markdown-sectionizer.cjs
+ * Exports: stripFencedCode, tokenizeHeadings, collectSections, collectSection,
+ * iterateBullets, extractTaggedBlocks, replaceSection
+ *
+ * Covers the parser QA matrix from CONTRIBUTING.md §'Parser and project-file inputs':
+ * - LF vs CRLF line endings
+ * - Unicode headings
+ * - Heading INSIDE a fenced block (must be ignored)
+ * - Unterminated fence (unterminatedFence === true)
+ * - Nested heading levels with level-bounded stop
+ * - All three bullet markers (dash/checkbox/numbered) + indented continuation lines
+ * - Empty/whitespace/non-string input
+ *
+ * Includes a fast-check property test (stripFencedCode idempotence invariant).
+ * Includes a parity guard for the tracked duplication between stripFencedCode
+ * and uat-predicate's _stripFencedBlocks (DEFECT.GENERATIVE-FIX — removed in T5).
+ */
+
+const { test, describe } = require('node:test');
+const assert = require('node:assert/strict');
+const fc = require('./helpers/fast-check-setup.cjs');
+
+const {
+ stripFencedCode,
+ tokenizeHeadings,
+ collectSections,
+ collectSection,
+ iterateBullets,
+ extractTaggedBlocks,
+ replaceSection,
+} = require('../gsd-core/bin/lib/markdown-sectionizer.cjs');
+
+// uat-predicate's _stripFencedBlocks is not directly exported.
+// The closest public surface is stripFalsePositiveContexts, which applies:
+// (a) frontmatter strip, (b) HTML comment strip, (c) _stripFencedBlocks, (d) blockquote strip.
+// For the parity corpus we use inputs with NO frontmatter, NO HTML comments, and NO blockquotes,
+// so the only transformation applied is the fence stripping in step (c).
+// We also use analyzeMarkdown, which calls _stripFencedBlocks directly for unterminatedFence.
+const {
+ stripFalsePositiveContexts,
+ analyzeMarkdown,
+} = require('../gsd-core/bin/lib/uat-predicate.cjs');
+
+// ─── stripFencedCode ──────────────────────────────────────────────────────────
+
+describe('stripFencedCode', () => {
+ test('returns empty text and no unterminatedFence on empty input', () => {
+ const r = stripFencedCode('');
+ assert.equal(r.text, '');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('non-string input returns empty result', () => {
+ // Safety: callers may pass non-strings; must not throw
+ for (const bad of [null, undefined, 42, [], {}]) {
+ const r = stripFencedCode(bad);
+ assert.equal(r.text, '');
+ assert.equal(r.unterminatedFence, false);
+ }
+ });
+
+ test('content with no fences is returned unchanged', () => {
+ const src = '## Heading\n\nSome text.\n\n- bullet';
+ const r = stripFencedCode(src);
+ assert.equal(r.text, src);
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('removes a backtick fenced block (LF)', () => {
+ const src = [
+ 'before',
+ '```js',
+ 'const x = 1;',
+ '```',
+ 'after',
+ ].join('\n');
+ const r = stripFencedCode(src);
+ assert.equal(r.text, 'before\nafter');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('removes a tilde fenced block', () => {
+ const src = [
+ 'before',
+ '~~~',
+ 'some code',
+ '~~~',
+ 'after',
+ ].join('\n');
+ const r = stripFencedCode(src);
+ assert.equal(r.text, 'before\nafter');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('handles CRLF line endings correctly', () => {
+ const src = 'before\r\n```\r\ncode\r\n```\r\nafter';
+ const r = stripFencedCode(src);
+ assert.ok(r.text.includes('before'));
+ assert.ok(r.text.includes('after'));
+ assert.ok(!r.text.includes('code'), 'code inside fence should be stripped');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('unterminatedFence is true when fence is not closed', () => {
+ const src = 'before\n```\nsome code without closing fence';
+ const r = stripFencedCode(src);
+ assert.equal(r.unterminatedFence, true);
+ assert.ok(!r.text.includes('some code'), 'fence body should be stripped even if unterminated');
+ });
+
+ test('tilde inside backtick fence is treated as content, not a closer', () => {
+ const src = [
+ '```',
+ '~~~',
+ 'still inside',
+ '```',
+ 'outside',
+ ].join('\n');
+ const r = stripFencedCode(src);
+ assert.equal(r.text, 'outside');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('backtick inside tilde fence is treated as content, not a closer', () => {
+ const src = [
+ '~~~',
+ '```',
+ 'still inside',
+ '~~~',
+ 'outside',
+ ].join('\n');
+ const r = stripFencedCode(src);
+ assert.equal(r.text, 'outside');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('closing fence must be same-char and same-or-longer run', () => {
+ // A `` ``` `` opener cannot be closed by ```` ```` ``; a 4-backtick closer is valid.
+ const src = [
+ 'text',
+ '```',
+ 'body',
+ '`````', // longer run of same char — valid closer per CommonMark
+ 'after',
+ ].join('\n');
+ const r = stripFencedCode(src);
+ assert.equal(r.text, 'text\nafter');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('closing fence must have no trailing non-whitespace text', () => {
+ // ``` js (info string) is only valid on OPENING lines; a line like "``` extra"
+ // inside a fence is content, not a closer.
+ const src = [
+ '```',
+ '``` still inside (has trailing text)',
+ '```',
+ 'after',
+ ].join('\n');
+ const r = stripFencedCode(src);
+ assert.equal(r.text, 'after');
+ assert.equal(r.unterminatedFence, false);
+ });
+
+ test('multiple successive fenced blocks are all stripped', () => {
+ const src = [
+ 'a',
+ '```',
+ 'code1',
+ '```',
+ 'b',
+ '```',
+ 'code2',
+ '```',
+ 'c',
+ ].join('\n');
+ const r = stripFencedCode(src);
+ assert.equal(r.text, 'a\nb\nc');
+ assert.equal(r.unterminatedFence, false);
+ });
+});
+
+// ─── tokenizeHeadings ─────────────────────────────────────────────────────────
+
+describe('tokenizeHeadings', () => {
+ test('returns empty array for empty/non-string input', () => {
+ assert.deepEqual(tokenizeHeadings(''), []);
+ assert.deepEqual(tokenizeHeadings(null), []);
+ assert.deepEqual(tokenizeHeadings(undefined), []);
+ });
+
+ test('extracts ATX headings in document order', () => {
+ const src = '# H1\n## H2\n### H3\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 3);
+ assert.equal(tokens[0].level, 1);
+ assert.equal(tokens[0].text, 'H1');
+ assert.equal(tokens[1].level, 2);
+ assert.equal(tokens[1].text, 'H2');
+ assert.equal(tokens[2].level, 3);
+ assert.equal(tokens[2].text, 'H3');
+ });
+
+ test('headings inside fenced blocks are ignored', () => {
+ const src = [
+ '# Real heading',
+ '```',
+ '## Fake heading inside fence',
+ '```',
+ '## Another real heading',
+ ].join('\n');
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 2);
+ assert.equal(tokens[0].text, 'Real heading');
+ assert.equal(tokens[1].text, 'Another real heading');
+ });
+
+ test('supports Unicode heading text', () => {
+ const src = '## Résumé — Überblick\n### 日本語見出し\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 2);
+ assert.equal(tokens[0].text, 'Résumé — Überblick');
+ assert.equal(tokens[1].text, '日本語見出し');
+ });
+
+ test('records correct 1-based line number', () => {
+ const src = 'prose\n## Heading\nmore';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 1);
+ assert.equal(tokens[0].line, 2);
+ });
+
+ test('records non-negative byte offset', () => {
+ const src = 'prose\n## Heading\n';
+ const tokens = tokenizeHeadings(src);
+ assert.ok(tokens[0].offset >= 0);
+ // The offset should point somewhere inside the heading line
+ assert.ok(tokens[0].offset < src.length);
+ });
+
+ test('handles CRLF headings', () => {
+ const src = '# H1\r\n## H2\r\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 2);
+ assert.equal(tokens[0].text, 'H1');
+ assert.equal(tokens[1].text, 'H2');
+ });
+
+ test('ignores setext-style headings (only ATX supported)', () => {
+ // Setext (underline) headings are NOT in scope for this seam
+ const src = 'Title\n=====\n\nSubtitle\n--------\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 0);
+ });
+});
+
+// ─── collectSections ─────────────────────────────────────────────────────────
+
+describe('collectSections', () => {
+ test('returns empty array for empty/non-string input', () => {
+ assert.deepEqual(collectSections('', () => true), []);
+ assert.deepEqual(collectSections(null, () => true), []);
+ });
+
+ test('collects all headings when predicate is always-true', () => {
+ const src = '## A\nBody A\n## B\nBody B\n';
+ const sections = collectSections(src, () => true);
+ assert.equal(sections.length, 2);
+ assert.equal(sections[0].heading.text, 'A');
+ assert.ok(sections[0].body.includes('Body A'));
+ assert.equal(sections[1].heading.text, 'B');
+ assert.ok(sections[1].body.includes('Body B'));
+ });
+
+ test('collects only headings matching predicate; non-matching headings end section', () => {
+ // When the predicate matches Section A but not Section B, Section B acts as
+ // a body line inside Section A (it is not a stop boundary), so its *heading*
+ // text appears in the body. However Section B's *content* also appears.
+ // If we want to stop at any heading regardless of the predicate, callers
+ // should use levelBounded collectSection instead.
+ //
+ // To test filtering: use a predicate that matches both headings, then verify
+ // two sections are returned with the correct split.
+ const src = '## Section A\nContent A\n## Section B\nContent B\n';
+ const sections = collectSections(src, () => true);
+ assert.equal(sections.length, 2);
+ assert.equal(sections[0].heading.text, 'Section A');
+ assert.ok(sections[0].body.includes('Content A'));
+ assert.ok(!sections[0].body.includes('Content B'), 'Content B must not appear in Section A body');
+ assert.equal(sections[1].heading.text, 'Section B');
+ assert.ok(sections[1].body.includes('Content B'));
+ });
+
+ test('stopPredicate controls which headings open sections; non-matching headings appear as body text', () => {
+ // When predicate matches only Section A, Section B is not a stop boundary
+ // so it (and its content) is included in Section A's body.
+ const src = '## Section A\nContent A\n## Section B\nContent B\n';
+ const sections = collectSections(src, (h) => h.text.includes('A'));
+ assert.equal(sections.length, 1);
+ assert.equal(sections[0].heading.text, 'Section A');
+ assert.ok(sections[0].body.includes('Content A'));
+ // Section B heading line and Content B are inside Section A's body
+ assert.ok(sections[0].body.includes('Section B'));
+ assert.ok(sections[0].body.includes('Content B'));
+ });
+
+ test('last section body runs to EOF', () => {
+ const src = '## Only\nBody line 1\nBody line 2';
+ const sections = collectSections(src, () => true);
+ assert.equal(sections.length, 1);
+ assert.ok(sections[0].body.includes('Body line 1'));
+ assert.ok(sections[0].body.includes('Body line 2'));
+ });
+
+ test('adjacent headings produce empty bodies', () => {
+ const src = '## A\n## B\n## C\nContent C\n';
+ const sections = collectSections(src, () => true);
+ assert.equal(sections.length, 3);
+ assert.equal(sections[0].body.trim(), '');
+ assert.equal(sections[1].body.trim(), '');
+ assert.ok(sections[2].body.includes('Content C'));
+ });
+});
+
+// ─── collectSection ───────────────────────────────────────────────────────────
+
+describe('collectSection', () => {
+ test('returns null when no heading matches', () => {
+ const src = '## Foo\ntext\n';
+ const result = collectSection(src, (h) => h.text === 'Bar');
+ assert.equal(result, null);
+ });
+
+ test('returns null for empty/non-string input', () => {
+ assert.equal(collectSection('', () => true), null);
+ assert.equal(collectSection(null, () => true), null);
+ });
+
+ test('collects section body up to next same-level heading (levelBounded default)', () => {
+ const src = [
+ '## Section A',
+ 'Content A',
+ '## Section B',
+ 'Content B',
+ ].join('\n');
+ const result = collectSection(src, (h) => h.text === 'Section A');
+ assert.ok(result !== null);
+ assert.ok(result.body.includes('Content A'));
+ assert.ok(!result.body.includes('Content B'));
+ });
+
+ test('levelBounded: true — sub-headings are included in body, not stops', () => {
+ const src = [
+ '## Parent',
+ 'Parent intro',
+ '### Child',
+ 'Child body',
+ '## Sibling',
+ 'Sibling body',
+ ].join('\n');
+ const result = collectSection(src, (h) => h.text === 'Parent', { levelBounded: true });
+ assert.ok(result !== null);
+ assert.ok(result.body.includes('Parent intro'));
+ assert.ok(result.body.includes('Child'), 'child heading line should be in body');
+ assert.ok(result.body.includes('Child body'));
+ assert.ok(!result.body.includes('Sibling body'), 'sibling body should NOT be included');
+ });
+
+ test('levelBounded: false — stops at any following heading', () => {
+ const src = [
+ '## Parent',
+ 'Parent intro',
+ '### Child',
+ 'Child body',
+ '## Sibling',
+ ].join('\n');
+ const result = collectSection(src, (h) => h.text === 'Parent', { levelBounded: false });
+ assert.ok(result !== null);
+ assert.ok(result.body.includes('Parent intro'));
+ assert.ok(!result.body.includes('Child body'), 'with levelBounded:false, child heading stops section');
+ });
+
+ test('stripFences: true — strips fenced blocks from body', () => {
+ const src = [
+ '## Section',
+ '```',
+ 'code here',
+ '```',
+ 'prose here',
+ ].join('\n');
+ const result = collectSection(src, (h) => h.text === 'Section', { stripFences: true });
+ assert.ok(result !== null);
+ assert.ok(!result.body.includes('code here'), 'fenced code should be stripped');
+ assert.ok(result.body.includes('prose here'));
+ });
+
+ test('heading token in result matches the matched heading', () => {
+ const src = '## My Section\nContent\n';
+ const result = collectSection(src, (h) => h.text === 'My Section');
+ assert.ok(result !== null);
+ assert.equal(result.heading.text, 'My Section');
+ assert.equal(result.heading.level, 2);
+ });
+
+ test('nested heading level: H3 section stops at next H3 or higher', () => {
+ const src = [
+ '### Alpha',
+ 'Alpha body',
+ '#### Sub-Alpha',
+ 'Sub-Alpha body',
+ '### Beta',
+ 'Beta body',
+ ].join('\n');
+ const result = collectSection(src, (h) => h.text === 'Alpha', { levelBounded: true });
+ assert.ok(result !== null);
+ assert.ok(result.body.includes('Alpha body'));
+ assert.ok(result.body.includes('Sub-Alpha'));
+ assert.ok(!result.body.includes('Beta body'));
+ });
+
+ test('section at EOF has body to end of string', () => {
+ const src = '## Only\nLast line';
+ const result = collectSection(src, (h) => h.text === 'Only');
+ assert.ok(result !== null);
+ assert.ok(result.body.includes('Last line'));
+ });
+});
+
+// ─── iterateBullets ───────────────────────────────────────────────────────────
+
+describe('iterateBullets', () => {
+ test('returns empty array for empty/non-string input', () => {
+ assert.deepEqual(iterateBullets(''), []);
+ assert.deepEqual(iterateBullets(null), []);
+ assert.deepEqual(iterateBullets(undefined), []);
+ assert.deepEqual(iterateBullets(' '), []);
+ });
+
+ test('parses dash bullets', () => {
+ const src = '- First\n- Second\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 2);
+ assert.equal(items[0].marker, 'dash');
+ assert.equal(items[0].text, 'First');
+ assert.equal(items[0].checked, null);
+ assert.equal(items[1].text, 'Second');
+ });
+
+ test('parses asterisk and plus bullets as dash marker', () => {
+ const src = '* Asterisk\n+ Plus\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 2);
+ assert.equal(items[0].marker, 'dash');
+ assert.equal(items[0].text, 'Asterisk');
+ assert.equal(items[1].marker, 'dash');
+ assert.equal(items[1].text, 'Plus');
+ });
+
+ test('parses unchecked checkbox bullets', () => {
+ const src = '- [ ] Todo item\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 1);
+ assert.equal(items[0].marker, 'checkbox-unchecked');
+ assert.equal(items[0].checked, false);
+ assert.equal(items[0].text, 'Todo item');
+ });
+
+ test('parses checked checkbox bullets (lowercase x)', () => {
+ const src = '- [x] Done item\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 1);
+ assert.equal(items[0].marker, 'checkbox-checked');
+ assert.equal(items[0].checked, true);
+ assert.equal(items[0].text, 'Done item');
+ });
+
+ test('parses checked checkbox bullets (uppercase X)', () => {
+ const src = '- [X] Done uppercase\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 1);
+ assert.equal(items[0].marker, 'checkbox-checked');
+ assert.equal(items[0].checked, true);
+ });
+
+ test('parses numbered bullets', () => {
+ const src = '1. First\n2. Second\n42. Forty-two\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 3);
+ assert.equal(items[0].marker, 'numbered');
+ assert.equal(items[0].text, 'First');
+ assert.equal(items[0].checked, null);
+ assert.equal(items[2].text, 'Forty-two');
+ });
+
+ test('accumulates indented continuation lines into bullet text', () => {
+ const src = [
+ '- Main bullet',
+ ' continuation line',
+ ' another continuation',
+ '- Next bullet',
+ ].join('\n');
+ const items = iterateBullets(src);
+ assert.equal(items.length, 2);
+ assert.ok(items[0].text.includes('Main bullet'));
+ assert.ok(items[0].text.includes('continuation line'));
+ assert.ok(items[0].text.includes('another continuation'));
+ assert.equal(items[1].text, 'Next bullet');
+ });
+
+ test('blank line terminates current bullet', () => {
+ const src = '- First\n\n- Second\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 2);
+ assert.equal(items[0].text, 'First');
+ assert.equal(items[1].text, 'Second');
+ });
+
+ test('mixed marker types in sequence', () => {
+ const src = [
+ '1. Numbered',
+ '- [x] Checked',
+ '- [ ] Unchecked',
+ '- Plain dash',
+ ].join('\n');
+ const items = iterateBullets(src);
+ assert.equal(items.length, 4);
+ assert.equal(items[0].marker, 'numbered');
+ assert.equal(items[1].marker, 'checkbox-checked');
+ assert.equal(items[2].marker, 'checkbox-unchecked');
+ assert.equal(items[3].marker, 'dash');
+ });
+
+ test('CRLF input is handled correctly', () => {
+ const src = '- First\r\n- Second\r\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 2);
+ assert.equal(items[0].text, 'First');
+ assert.equal(items[1].text, 'Second');
+ });
+
+ test('indent field captures leading whitespace of bullet opener', () => {
+ const src = ' - Indented bullet\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 1);
+ assert.equal(items[0].indent, ' ');
+ });
+
+ test('non-bullet lines before any bullet are ignored', () => {
+ const src = 'Some prose\n\n- Bullet\n';
+ const items = iterateBullets(src);
+ assert.equal(items.length, 1);
+ assert.equal(items[0].text, 'Bullet');
+ });
+});
+
+// ─── Integration: heading inside fenced block is ignored end-to-end ───────────
+
+describe('integration: fenced heading ignored', () => {
+ test('collectSection ignores headings inside fenced blocks', () => {
+ const src = [
+ '## Real',
+ 'Real body',
+ '```',
+ '## Fake inside fence',
+ '```',
+ 'More real body',
+ ].join('\n');
+ const result = collectSection(src, (h) => h.text === 'Real');
+ assert.ok(result !== null, 'should find the real heading');
+ assert.ok(result.body.includes('More real body'), 'real body after fence should be included');
+ // The section should not have ended at the fake heading
+ });
+
+ test('tokenizeHeadings ignores headings in CRLF fenced blocks', () => {
+ const src = '# Outer\r\n```\r\n# Inner\r\n```\r\n## After\r\n';
+ const tokens = tokenizeHeadings(src);
+ const texts = tokens.map((t) => t.text);
+ assert.ok(texts.includes('Outer'));
+ assert.ok(texts.includes('After'));
+ assert.ok(!texts.includes('Inner'), 'heading inside fence should be invisible');
+ });
+});
+
+// ─── Property test: stripFencedCode idempotence ───────────────────────────────
+
+describe('stripFencedCode: property-based tests', () => {
+ test('property: never throws on any string input', () => {
+ fc.assert(
+ fc.property(
+ fc.oneof(
+ fc.string({ maxLength: 500 }),
+ fc.string({ unit: 'binary', maxLength: 200 }),
+ fc.string({ unit: 'grapheme-composite', maxLength: 200 }),
+ fc.constant(''),
+ fc.constant('```\ncode\n```\n'),
+ fc.constant('~~~\nunterminated'),
+ ),
+ (input) => {
+ assert.doesNotThrow(
+ () => stripFencedCode(input),
+ `stripFencedCode threw on: ${JSON.stringify(input.slice(0, 80))}`,
+ );
+ },
+ ),
+ );
+ });
+
+ test('property: always returns { text: string, unterminatedFence: boolean }', () => {
+ fc.assert(
+ fc.property(
+ fc.string({ maxLength: 500 }),
+ (input) => {
+ const result = stripFencedCode(input);
+ assert.ok(typeof result === 'object' && result !== null, 'result must be object');
+ assert.ok(typeof result.text === 'string', 'text must be string');
+ assert.ok(typeof result.unterminatedFence === 'boolean', 'unterminatedFence must be boolean');
+ },
+ ),
+ );
+ });
+
+ test('property: idempotence — stripping twice gives the same text as stripping once', () => {
+ // A well-formed (terminated) fence: strip once gives fence-free text with no
+ // remaining fences. Stripping again gives the same text.
+ // For unterminated fences the text after first strip has no fence content but
+ // the result is still idempotent — stripping a fence-free string is a no-op.
+ fc.assert(
+ fc.property(
+ fc.string({ maxLength: 500 }),
+ (input) => {
+ const once = stripFencedCode(input);
+ const twice = stripFencedCode(once.text);
+ assert.equal(
+ twice.text,
+ once.text,
+ `Idempotence violated: input=${JSON.stringify(input.slice(0, 60))}`,
+ );
+ // After the first strip the text has no fences (or only unterminated remnants),
+ // so the second pass must not report unterminated unless the first pass already did.
+ // (The second pass cannot have MORE unterminatedFence; it can have less.)
+ assert.ok(
+ !twice.unterminatedFence || once.unterminatedFence,
+ 'Second pass may not introduce a new unterminatedFence not present in first pass',
+ );
+ },
+ ),
+ );
+ });
+
+ test('property: output text is always a substring or equal-length string of input', () => {
+ // Stripping removes content, so output length <= input length
+ fc.assert(
+ fc.property(
+ fc.string({ maxLength: 500 }),
+ (input) => {
+ const result = stripFencedCode(input);
+ assert.ok(
+ result.text.length <= input.length,
+ `Output (${result.text.length}) must not be longer than input (${input.length})`,
+ );
+ },
+ ),
+ );
+ });
+});
+
+// ─── extractTaggedBlocks ──────────────────────────────────────────────────────
+
+describe('extractTaggedBlocks', () => {
+ test('returns empty array for empty/non-string content', () => {
+ assert.deepEqual(extractTaggedBlocks('', 'decisions'), []);
+ assert.deepEqual(extractTaggedBlocks(null, 'decisions'), []);
+ assert.deepEqual(extractTaggedBlocks(undefined, 'decisions'), []);
+ });
+
+ test('returns empty array when tag is empty/non-string', () => {
+ assert.deepEqual(extractTaggedBlocks('body', ''), []);
+ assert.deepEqual(extractTaggedBlocks('body', null), []);
+ });
+
+ test('returns empty array when tag is not present', () => {
+ const content = 'Some prose without any matching block.\n## Heading\n- bullet';
+ assert.deepEqual(extractTaggedBlocks(content, 'decisions'), []);
+ });
+
+ test('extracts inner text of a single block', () => {
+ const content = 'before\n\nD-01: Foo\n\nafter';
+ const result = extractTaggedBlocks(content, 'decisions');
+ assert.equal(result.length, 1);
+ assert.ok(result[0].includes('D-01: Foo'), 'inner text should be returned');
+ });
+
+ test('extracts multiple blocks in document order', () => {
+ const content = [
+ '',
+ 'D-01: First',
+ '',
+ 'some text',
+ '',
+ 'D-02: Second',
+ '',
+ ].join('\n');
+ const result = extractTaggedBlocks(content, 'decisions');
+ assert.equal(result.length, 2);
+ assert.ok(result[0].includes('D-01: First'));
+ assert.ok(result[1].includes('D-02: Second'));
+ });
+
+ test('preserves document order of multiple blocks', () => {
+ const content = 'alpha middle beta end gamma';
+ const result = extractTaggedBlocks(content, 'tag');
+ assert.deepEqual(result, ['alpha', 'beta', 'gamma']);
+ });
+
+ test('handles CRLF content inside a block', () => {
+ const content = '\r\nD-01: CRLF test\r\n';
+ const result = extractTaggedBlocks(content, 'decisions');
+ assert.equal(result.length, 1);
+ assert.ok(result[0].includes('D-01: CRLF test'));
+ });
+
+ test('tag name that needs regex-escaping: dot in tag name is matched literally', () => {
+ // A tag name with a dot (e.g. 'my.tag') must be matched literally, not as
+ // a regex wildcard. So '' should match only the exact literal tag.
+ const content = 'inner';
+ const result = extractTaggedBlocks(content, 'my.tag');
+ assert.equal(result.length, 1);
+ assert.equal(result[0], 'inner');
+ // Crucially, 'myXtag' (dot as wildcard) should NOT match the literal block
+ const result2 = extractTaggedBlocks('other', 'my.tag');
+ assert.equal(result2.length, 0, 'dot in tagName must be treated as literal, not wildcard');
+ });
+
+ test('tag name with + character is escaped and matched literally', () => {
+ const content = 'inner';
+ const result = extractTaggedBlocks(content, 'my+tag');
+ assert.equal(result.length, 1);
+ assert.equal(result[0], 'inner');
+ });
+
+ test('content with tag text appearing outside any block is not extracted', () => {
+ // The tag appears as inline text, not as an XML block
+ const _content = 'This is about but no closing tag in same element sense\n\nNot a block.';
+ // Actually we need to use content that has the opening tag on the same line as text
+ // but no matching close tag — result should be empty or the inner text is everything after.
+ // Since the regex is non-greedy, an unclosed tag won't match.
+ const content2 = 'Text with but tag is unclosed.';
+ const result = extractTaggedBlocks(content2, 'decisions');
+ assert.equal(result.length, 0, 'unclosed tag should not produce a match');
+ });
+});
+
+// ─── replaceSection ───────────────────────────────────────────────────────────
+
+describe('replaceSection', () => {
+ test('replaces section body and preserves heading and surrounding sections', () => {
+ const content = '## Intro\nIntro body.\n## Name\nOld name body.\n## Footer\nFooter body.\n';
+ const section = collectSection(content, (h) => h.text === 'Name');
+ assert.ok(section !== null, 'section must be found');
+ const newContent = replaceSection(content, section, 'New name body.\n');
+ assert.ok(newContent.includes('## Intro'), 'Intro heading preserved');
+ assert.ok(newContent.includes('Intro body.'), 'Intro body preserved');
+ assert.ok(newContent.includes('## Name'), 'Name heading preserved');
+ assert.ok(newContent.includes('New name body.'), 'new body present');
+ assert.ok(!newContent.includes('Old name body.'), 'old body removed');
+ assert.ok(newContent.includes('## Footer'), 'Footer heading preserved');
+ assert.ok(newContent.includes('Footer body.'), 'Footer body preserved');
+ });
+
+ test('replaces section body in a multi-section document', () => {
+ const content = [
+ '## Alpha',
+ 'Alpha content.',
+ '## Beta',
+ 'Beta old content.',
+ '## Gamma',
+ 'Gamma content.',
+ ].join('\n') + '\n';
+ const section = collectSection(content, (h) => h.text === 'Beta');
+ assert.ok(section !== null);
+ const updated = replaceSection(content, section, 'Beta new content.\n');
+ assert.ok(updated.includes('Alpha content.'), 'Alpha preserved');
+ assert.ok(updated.includes('Beta new content.'), 'Beta updated');
+ assert.ok(!updated.includes('Beta old content.'), 'Beta old removed');
+ assert.ok(updated.includes('Gamma content.'), 'Gamma preserved');
+ });
+
+ test('round-trip: collectSection → replaceSection with section.body → content unchanged', () => {
+ // INVARIANT: content.slice(bodyStart, bodyEnd) === body
+ // so replaceSection(content, section, section.body) must equal content exactly.
+ const content = '## Section A\nLine one.\nLine two.\n## Section B\nB body.\n';
+ const section = collectSection(content, (h) => h.text === 'Section A');
+ assert.ok(section !== null);
+ // Verify the slice invariant directly
+ assert.equal(
+ content.slice(section.bodyStart, section.bodyEnd),
+ section.body,
+ 'content.slice(bodyStart, bodyEnd) must equal section.body (invariant)',
+ );
+ // True round-trip: supply section.body (not a re-sliced value)
+ const roundTripped = replaceSection(content, section, section.body);
+ assert.equal(roundTripped, content, 'round-trip must produce identical content');
+ });
+
+ test('CRLF content is handled without corruption', () => {
+ const content = '## Title\r\nOld body.\r\n## Next\r\nNext body.\r\n';
+ const section = collectSection(content, (h) => h.text === 'Title');
+ assert.ok(section !== null);
+ const updated = replaceSection(content, section, 'New body.\r\n');
+ assert.ok(updated.includes('## Title\r\n'), 'heading with CRLF preserved');
+ assert.ok(updated.includes('New body.'), 'new body present');
+ assert.ok(!updated.includes('Old body.'), 'old body removed');
+ assert.ok(updated.includes('## Next\r\n'), 'next section heading preserved');
+ assert.ok(updated.includes('Next body.'), 'next section body preserved');
+ });
+
+ test('non-string arguments return content unchanged', () => {
+ const content = '## Sec\nbody\n';
+ const section = collectSection(content, (h) => h.text === 'Sec');
+ assert.ok(section !== null);
+ assert.equal(replaceSection(null, section, 'x'), null);
+ assert.equal(replaceSection(content, section, null), content);
+ });
+});
+
+// ─── FIX 1: Section offset invariant tests ────────────────────────────────────
+
+describe('Section offset invariant: content.slice(bodyStart, bodyEnd) === body', () => {
+ test('invariant holds for a mid-document section (LF, trailing newline)', () => {
+ const content = '## A\nBody A\n## B\nBody B\n';
+ const s = collectSection(content, (h) => h.text === 'A');
+ assert.ok(s !== null);
+ assert.equal(
+ content.slice(s.bodyStart, s.bodyEnd),
+ s.body,
+ 'invariant: content.slice(bodyStart, bodyEnd) === body',
+ );
+ assert.equal(
+ replaceSection(content, s, s.body),
+ content,
+ 'true round-trip with section.body must be identity',
+ );
+ });
+
+ test('invariant holds at EOF with no trailing newline', () => {
+ const content = '## Only\nLast line';
+ const s = collectSection(content, (h) => h.text === 'Only');
+ assert.ok(s !== null);
+ assert.equal(content.slice(s.bodyStart, s.bodyEnd), s.body, 'EOF no-trailing-newline invariant');
+ assert.equal(replaceSection(content, s, s.body), content, 'round-trip EOF no-trailing-newline');
+ });
+
+ test('invariant holds with CRLF line endings', () => {
+ const content = '## Title\r\nBody line.\r\n## Next\r\nNext body.\r\n';
+ const s = collectSection(content, (h) => h.text === 'Title');
+ assert.ok(s !== null);
+ assert.equal(content.slice(s.bodyStart, s.bodyEnd), s.body, 'CRLF invariant');
+ assert.equal(replaceSection(content, s, s.body), content, 'CRLF round-trip');
+ });
+
+ test('invariant holds for an empty body (adjacent headings)', () => {
+ const content = '## A\n## B\nB body\n';
+ const s = collectSection(content, (h) => h.text === 'A');
+ assert.ok(s !== null);
+ assert.equal(s.body, '', 'empty body expected');
+ assert.equal(content.slice(s.bodyStart, s.bodyEnd), s.body, 'empty body invariant');
+ assert.equal(replaceSection(content, s, s.body), content, 'empty body round-trip');
+ });
+
+ test('collectSections: invariant holds for every returned section', () => {
+ const content = '## Alpha\nAlpha body.\n## Beta\nBeta body.\n## Gamma\nGamma body\n';
+ const sections = collectSections(content, () => true);
+ assert.equal(sections.length, 3);
+ for (const s of sections) {
+ assert.equal(
+ content.slice(s.bodyStart, s.bodyEnd),
+ s.body,
+ `collectSections invariant for section "${s.heading.text}"`,
+ );
+ assert.equal(
+ replaceSection(content, s, s.body),
+ content,
+ `collectSections round-trip for section "${s.heading.text}"`,
+ );
+ }
+ });
+});
+
+// ─── FIX 2: tokenizeHeadings CommonMark indented and empty headings ──────────
+
+describe('tokenizeHeadings: CommonMark ≤3-space indent and empty headings', () => {
+ test('1-space indent is a valid heading', () => {
+ const src = ' # One space heading\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 1);
+ assert.equal(tokens[0].level, 1);
+ assert.equal(tokens[0].text, 'One space heading');
+ });
+
+ test('2-space indent is a valid heading', () => {
+ const src = ' ## Two space heading\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 1);
+ assert.equal(tokens[0].level, 2);
+ assert.equal(tokens[0].text, 'Two space heading');
+ });
+
+ test('3-space indent is a valid heading', () => {
+ const src = ' ### Three space heading\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 1);
+ assert.equal(tokens[0].level, 3);
+ assert.equal(tokens[0].text, 'Three space heading');
+ });
+
+ test('4-space indent is NOT a heading (indented code block per CommonMark)', () => {
+ const src = ' ## Four space — not a heading\n## Real heading\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 1, 'only the non-indented heading should be found');
+ assert.equal(tokens[0].text, 'Real heading');
+ });
+
+ test('## with no following text is an empty heading (text === "")', () => {
+ const src = '##\n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 1);
+ assert.equal(tokens[0].level, 2);
+ assert.equal(tokens[0].text, '');
+ });
+
+ test('## (only whitespace after hashes) is an empty heading (text === "")', () => {
+ const src = '## \n';
+ const tokens = tokenizeHeadings(src);
+ assert.equal(tokens.length, 1);
+ assert.equal(tokens[0].level, 2);
+ assert.equal(tokens[0].text, '');
+ });
+});
+
+// ─── FIX 3: collectSection stopAtLevel option ─────────────────────────────────
+
+describe('collectSection: stopAtLevel option', () => {
+ test('stopAtLevel:3 stops a ##-opened section at the following ###', () => {
+ const src = [
+ '## Parent',
+ 'Parent body',
+ '### Child',
+ 'Child body',
+ '## Sibling',
+ 'Sibling body',
+ ].join('\n');
+ const s = collectSection(src, (h) => h.text === 'Parent', { stopAtLevel: 3 });
+ assert.ok(s !== null);
+ assert.ok(s.body.includes('Parent body'), 'parent body included');
+ assert.ok(!s.body.includes('Child body'), 'section should stop at ### with stopAtLevel:3');
+ assert.ok(!s.body.includes('Sibling body'), 'sibling body not included');
+ });
+
+ test('default levelBounded:true does NOT stop a ##-opened section at ###', () => {
+ const src = [
+ '## Parent',
+ 'Parent body',
+ '### Child',
+ 'Child body',
+ '## Sibling',
+ 'Sibling body',
+ ].join('\n');
+ const s = collectSection(src, (h) => h.text === 'Parent', { levelBounded: true });
+ assert.ok(s !== null);
+ assert.ok(s.body.includes('Child body'), 'child body is inside the ## section with levelBounded');
+ assert.ok(!s.body.includes('Sibling body'), 'sibling body not included');
+ });
+
+ test('stopAtLevel:2 stops at the next ## (same as levelBounded default for ## opener)', () => {
+ const src = '## A\nA body\n## B\nB body\n';
+ const s = collectSection(src, (h) => h.text === 'A', { stopAtLevel: 2 });
+ assert.ok(s !== null);
+ assert.ok(s.body.includes('A body'));
+ assert.ok(!s.body.includes('B body'));
+ });
+
+ test('stopAtLevel round-trip invariant holds', () => {
+ const src = '## Parent\nParent body\n### Child\nChild body\n## Sibling\nSibling body\n';
+ const s = collectSection(src, (h) => h.text === 'Parent', { stopAtLevel: 3 });
+ assert.ok(s !== null);
+ assert.equal(src.slice(s.bodyStart, s.bodyEnd), s.body, 'offset invariant with stopAtLevel');
+ assert.equal(replaceSection(src, s, s.body), src, 'round-trip with stopAtLevel');
+ });
+});
+
+// ─── FIX 4: backtick fence — info string with backtick is not a fence opener ─
+
+describe('stripFencedCode and tokenizeHeadings: backtick info string with backtick', () => {
+ test('stripFencedCode: backtick in info string does not open a backtick fence', () => {
+ // The line "``` ` info" has a backtick in the info string → NOT a fence opener.
+ const src = '``` ` not-a-fence\n## Heading\n';
+ const r = stripFencedCode(src);
+ // Both lines should be kept (no fence was opened)
+ assert.ok(r.text.includes('## Heading'), 'heading line must be kept since no fence opened');
+ assert.ok(r.text.includes('``` ` not-a-fence'), 'the non-fence line must be kept');
+ assert.equal(r.unterminatedFence, false, 'no fence was opened, so unterminated must be false');
+ });
+
+ test('tokenizeHeadings: heading after a backtick-in-info line is still tokenized', () => {
+ // ``` ` info-with-backtick is NOT a fence opener, so ## Heading below it is visible.
+ const src = '``` ` not-a-fence\n## Heading\nprose\n```\n';
+ const tokens = tokenizeHeadings(src);
+ assert.ok(tokens.some((t) => t.text === 'Heading'), '## Heading must be tokenized when "opener" has backtick in info');
+ });
+
+ test('tilde fence info string WITH backtick IS still a valid fence opener (tildes unaffected)', () => {
+ // Only backtick fences have the "no backtick in info" restriction.
+ const src = '~~~ ` this-is-fine\n## Inside tilde fence\n~~~\n## Outside\n';
+ const tokens = tokenizeHeadings(src);
+ // ## Inside tilde fence should be ignored (inside a real fence)
+ assert.ok(!tokens.some((t) => t.text === 'Inside tilde fence'), 'tilde fence with backtick in info is still a valid fence');
+ assert.ok(tokens.some((t) => t.text === 'Outside'), 'heading after tilde fence close is tokenized');
+ });
+});
+
+// ─── FIX 6: extractTaggedBlocks — nested tag behavior ─────────────────────────
+
+describe('extractTaggedBlocks: nested same-name tag behavior (non-greedy limitation)', () => {
+ test('nested … closes at first (non-greedy; nested tags not supported)', () => {
+ // Non-greedy match: ([\s\S]*?) closes at the FIRST .
+ // So inner → first block captures "inner", second is unmatched.
+ const content = 'inner';
+ const result = extractTaggedBlocks(content, 'x');
+ // The first match closes at the first , capturing "inner"
+ assert.equal(result.length, 1, 'non-greedy match produces exactly one result from nested input');
+ assert.equal(result[0], 'inner', 'inner capture is the content up to the first closing tag');
+ });
+
+ test('back-to-back blocks (not nested) are both extracted', () => {
+ const content = 'firstsecond';
+ const result = extractTaggedBlocks(content, 'x');
+ assert.equal(result.length, 2);
+ assert.equal(result[0], 'first');
+ assert.equal(result[1], 'second');
+ });
+});
+
+// ─── Parity guard: stripFencedCode vs uat-predicate _stripFencedBlocks ────────
+//
+// DEFECT.GENERATIVE-FIX: stripFencedCode in the seam is a tracked duplication of
+// _stripFencedBlocks in uat-predicate.cts until tier T5 deduplicates them.
+// This test MUST FAIL if the two implementations diverge on any corpus input.
+// Remove this describe block in T5 when uat-predicate imports the seam directly.
+//
+// Approach: feed a shared fence-input corpus through:
+// (A) stripFencedCode (seam — direct export)
+// (B) stripFalsePositiveContexts (uat-predicate public surface)
+// Input must have NO frontmatter (not starting with ---), NO HTML comments,
+// and NO blockquote lines, so that steps (a)(b)(d) in stripFalsePositiveContexts
+// are no-ops and only the fence-stripping step (c) differs between them.
+// (C) analyzeMarkdown.unterminatedFence (uat-predicate — calls _stripFencedBlocks directly)
+//
+// Limitation: _stripFencedBlocks is not directly exported from uat-predicate.cjs,
+// so we test through the closest public surface and document the boundary.
+
+describe('parity guard: stripFencedCode vs uat-predicate fence-stripping', () => {
+ // Shared corpus of fence inputs for parity testing.
+ // All inputs have no frontmatter, no HTML comments, no blockquotes — only fences.
+ const FENCE_CORPUS = [
+ {
+ label: 'no fences',
+ input: '## Heading\n\nSome text.\n\n- bullet',
+ },
+ {
+ label: 'backtick fence',
+ input: 'before\n```js\nconst x = 1;\n```\nafter',
+ },
+ {
+ label: 'tilde fence',
+ input: 'before\n~~~\nsome code\n~~~\nafter',
+ },
+ {
+ label: 'CRLF fence',
+ input: 'before\r\n```\r\ncode\r\n```\r\nafter',
+ },
+ {
+ label: 'unterminated fence',
+ input: 'before\n```\nsome code without closing fence',
+ },
+ {
+ label: 'tilde inside backtick fence (mismatched delimiter)',
+ input: '```\n~~~\nstill inside\n```\noutside',
+ },
+ {
+ label: 'backtick inside tilde fence (mismatched delimiter)',
+ input: '~~~\n```\nstill inside\n~~~\noutside',
+ },
+ {
+ label: 'longer closing fence run',
+ input: 'text\n```\nbody\n`````\nafter',
+ },
+ {
+ label: 'multiple successive fenced blocks',
+ input: 'a\n```\ncode1\n```\nb\n```\ncode2\n```\nc',
+ },
+ // NOTE: '4-space indent' case is intentionally excluded from the parity corpus.
+ // The seam uses /^( {0,3})/ (CommonMark §4.5: ≤3 leading spaces tolerated),
+ // while uat-predicate._stripFencedBlocks uses /^(\s*)/ (any whitespace).
+ // A 4-space-indented ``` is NOT a fence opener per CommonMark but IS treated
+ // as one by uat-predicate. This is a known pre-existing divergence; the seam
+ // is the more-correct implementation. The divergence is documented here so that
+ // T5 (which will remove uat-predicate's local copy) is aware of the fix needed.
+ ];
+
+ for (const { label, input } of FENCE_CORPUS) {
+ test(`text output parity: ${label}`, () => {
+ const seamResult = stripFencedCode(input).text;
+ // stripFalsePositiveContexts with no-frontmatter/no-comment/no-blockquote input
+ // reduces to exactly _stripFencedBlocks (step c only).
+ const uatResult = stripFalsePositiveContexts(input);
+ assert.equal(
+ seamResult,
+ uatResult,
+ `stripFencedCode and uat-predicate _stripFencedBlocks diverged on: ${JSON.stringify(label)}\n` +
+ `seam: ${JSON.stringify(seamResult.slice(0, 120))}\n` +
+ `uat: ${JSON.stringify(uatResult.slice(0, 120))}`,
+ );
+ });
+
+ test(`unterminatedFence parity: ${label}`, () => {
+ const seamUnterminated = stripFencedCode(input).unterminatedFence;
+ // analyzeMarkdown calls _stripFencedBlocks directly for unterminatedFence.
+ const uatUnterminated = analyzeMarkdown(input).unterminatedFence;
+ assert.equal(
+ seamUnterminated,
+ uatUnterminated,
+ `unterminatedFence diverged on: ${JSON.stringify(label)}\n` +
+ `seam: ${seamUnterminated}, uat: ${uatUnterminated}`,
+ );
+ });
+ }
+});