Final convergence review found the shared <tag> seam introduced 3 behavior regressions; fixed all + locked with tests: - #557 REGRESSION: stripTaggedBlocks's attribute-tolerance stripped `<details open>` (the ACTIVE-milestone marker) that the old `<details>`-only regex preserved. The seam now takes `allowAttributes` (default false) — details/decisions strip is attr-INTOLERANT (preserves `<details open>`); only `<task type="…">` opts in. Regression test added to roadmap-parser + markdown-sectionizer suites. - verify.cts actionZones (negative-grep-echo security scan): reverted to a bounded to-first-close scan `<action>([\s\S]{0,20000}?)</action>` so a grep-echo trick can't hide behind an unterminated inner <action> (the seam's stop-at-next-open would drop it). ReDoS-safe via the cap. - check-command-router HTML-comment strip: `(?:-->|$)` fallback wiped to EOF (fail-closed spurious gate block) — replaced with stop-at-next-open so an unclosed `<!--` leaves downstream tags intact. - Updated the extractTaggedBlocks nested-tag tests to the new (stop-at-next-open) behavior: `<x><x>inner</x></x>` -> ['inner']. All vectors still linear; every fix verified in-process. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -520,22 +520,25 @@ export function iterateBullets(sectionText: string): BulletItem[] {
|
||||
* fenced code blocks itself. If a `<tagName>` block appears inside a fenced code
|
||||
* block and should be excluded, the caller should apply `stripFencedCode` first.
|
||||
*
|
||||
* **Nested tags are NOT supported.** The underlying regex uses a non-greedy
|
||||
* `[\s\S]*?` match, which means it closes at the FIRST `</tagName>` encountered.
|
||||
* Given `<x><x>inner</x></x>`, `extractTaggedBlocks(content, 'x')` returns
|
||||
* `['<x>inner']` — the inner `<x>` is captured as literal text, and the second
|
||||
* `</x>` is left unmatched (or matched as a second block with empty inner text
|
||||
* if another `<x>` follows). Callers that need to handle nested tags must
|
||||
* pre-process the input or use a proper XML/HTML parser.
|
||||
* **Nested tags are NOT supported.** The body scan terminates at the NEXT
|
||||
* opening of the same tag (the ReDoS-safe boundary, #2128). Given
|
||||
* `<x><x>inner</x></x>`, `extractTaggedBlocks(content, 'x')` returns `['inner']`
|
||||
* — the well-formed inner block; the unterminated outer `<x>` is skipped.
|
||||
* Callers that need true nesting must use a proper XML/HTML parser.
|
||||
*
|
||||
* `allowAttributes` (default `false`): when `true`, the opening tag may carry
|
||||
* bounded attributes (`<tag foo="x">`) — needed for `<task type="…">` blocks.
|
||||
* Leave `false` for tags that must match exactly (e.g. `<decisions>`), and never
|
||||
* enable it for a tag where an attributed form is semantically distinct.
|
||||
*
|
||||
* Generalises `decisions.cts`'s bespoke `matchAll(/<decisions>([\s\S]*?)<\/decisions>/g)`
|
||||
* so tier T1 can drop its own copy (tracked duplication until T1 lands).
|
||||
*/
|
||||
export function extractTaggedBlocks(content: string, tagName: string): string[] {
|
||||
export function extractTaggedBlocks(content: string, tagName: string, allowAttributes = false): string[] {
|
||||
if (typeof content !== 'string' || content.length === 0) return [];
|
||||
if (typeof tagName !== 'string' || tagName.length === 0) return [];
|
||||
|
||||
const pattern = taggedBlockPattern(tagName, 'g');
|
||||
const pattern = taggedBlockPattern(tagName, 'g', allowAttributes);
|
||||
const results: string[] = [];
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = pattern.exec(content)) !== null) {
|
||||
@@ -548,30 +551,37 @@ export function extractTaggedBlocks(content: string, tagName: string): string[]
|
||||
* Build the single, ReDoS-safe `<tag>…</tag>` block regex shared by
|
||||
* `extractTaggedBlocks` (extract bodies) and `stripTaggedBlocks` (remove blocks).
|
||||
*
|
||||
* Safety: the body uses a `(?:(?!<tag[\s>])[\s\S])*?` negative-lookahead scan
|
||||
* that terminates at the NEXT opening `<tag>` instead of lazily rescanning the
|
||||
* whole remaining document for a `</tag>` that may never appear — so a large
|
||||
* document full of unclosed `<tag>` openings stays LINEAR, not quadratic
|
||||
* (#2128). The opening tag tolerates optional attributes (`<tag foo="bar">`),
|
||||
* bounded to 1000 chars so the attribute scan cannot itself ReDoS.
|
||||
* Group 1 is the block body.
|
||||
* Safety: the body terminates at the NEXT opening of this tag (stop-at-next-open)
|
||||
* instead of lazily rescanning the whole remaining document for a `</tag>` that
|
||||
* may never appear — so a document full of unclosed `<tag>` openings scans
|
||||
* LINEARLY, not quadratically (#2128). Group 1 is the block body.
|
||||
*
|
||||
* `allowAttributes`: when `true`, the opener accepts bounded attributes
|
||||
* (`<tag foo="x">`) and the body boundary is `<tag` followed by a space or `>`.
|
||||
* When `false`, the opener is the EXACT `<tag>` and the boundary is exact `<tag>`,
|
||||
* so an attributed `<tag foo>` is neither an opener nor a boundary — it is body
|
||||
* content. That exact form is load-bearing for `<details>` stripping: `<details
|
||||
* open>` marks the ACTIVE milestone and must be preserved, not stripped (#557).
|
||||
*/
|
||||
function taggedBlockPattern(tagName: string, flags: string): RegExp {
|
||||
function taggedBlockPattern(tagName: string, flags: string, allowAttributes: boolean): RegExp {
|
||||
const esc = tagName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
return new RegExp(`<${esc}(?:\\s[^>]{0,1000})?>((?:(?!<${esc}[\\s>])[\\s\\S])*?)</${esc}>`, flags);
|
||||
const open = allowAttributes ? `<${esc}(?:\\s[^>]{0,1000})?>` : `<${esc}>`;
|
||||
const boundary = allowAttributes ? `<${esc}[\\s>]` : `<${esc}>`;
|
||||
return new RegExp(`${open}((?:(?!${boundary})[\\s\\S])*?)</${esc}>`, flags);
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove every `<tagName>…</tagName>` block (opening tag, body, and closing tag)
|
||||
* from `content`. The ReDoS-safe counterpart to `extractTaggedBlocks` — same
|
||||
* hardened pattern, `.replace(…, '')` instead of body extraction. Case-insensitive
|
||||
* by default (matching the `<details>` strip call sites); pass `caseSensitive`
|
||||
* to force exact-case matching.
|
||||
* hardened pattern, `.replace(…, '')` instead of body extraction. `allowAttributes`
|
||||
* defaults to `false` so `<details open>` (active milestone) is preserved (#557);
|
||||
* case-insensitive by default (matching the `<details>` strip call sites), pass
|
||||
* `caseSensitive` to force exact-case matching.
|
||||
*/
|
||||
export function stripTaggedBlocks(content: string, tagName: string, caseSensitive = false): string {
|
||||
export function stripTaggedBlocks(content: string, tagName: string, allowAttributes = false, caseSensitive = false): string {
|
||||
if (typeof content !== 'string' || content.length === 0) return '';
|
||||
if (typeof tagName !== 'string' || tagName.length === 0) return content;
|
||||
return content.replace(taggedBlockPattern(tagName, caseSensitive ? 'g' : 'gi'), '');
|
||||
return content.replace(taggedBlockPattern(tagName, caseSensitive ? 'g' : 'gi', allowAttributes), '');
|
||||
}
|
||||
|
||||
// ─── replaceSection ───────────────────────────────────────────────────────────
|
||||
|
||||
Reference in New Issue
Block a user