From d2d2f7c088bed7d588749a05bc89d1012f870c48 Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Fri, 31 Jul 2026 10:56:11 -0400 Subject: [PATCH] fix(#2848): non-Latin titles no longer produce empty slugs (Cyrillic transliteration) (#2934) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * test(#2848): add failing-first regression for non-Latin slug transliteration generateSlugInternal and slugify both strip non-ASCII chars with no transliteration step, so an all-Cyrillic title reduces to an empty slug. 12-row matrix: Cyrillic regression (both impls), Latin negative control, multi-letter mappings, soft/hard sign drops, Ukrainian extras, null contract, CJK unaffected, mixed scripts, truncation parity, slugify's distinct no-truncate contract. * fix(#2848): transliterate Cyrillic titles to ASCII before slug strip Both generateSlugInternal (src/core-utils.cts) and slugify (src/gsd2-import.cts) stripped non-ASCII with no transliteration, so an all-Cyrillic title reduced to an empty slug, producing unnamed phase directories (01-) and empty milestone_slug init JSON. Add a shared transliterateForSlug primitive (core-utils) covering Russian + the reported Ukrainian/Belarusian extras (і ї є ґ ў), with multi-letter mappings (ж→zh ч→ch ш→sh щ→sch ю→yu я→ya) and dropped soft/hard signs (ъ ь). It runs BEFORE the existing ASCII filter, so Latin-script text hits zero map entries and is byte-for-byte unchanged (negative control). slugify consumes the shared primitive, preserving its distinct single hyphen-strip + no-truncation contract. CJK/unmapped scripts keep the existing strip-to-ASCII behavior. Also corrects two test assertions to match the chosen й→y mapping and the б→b (not bie) transliteration. * changeset(#2848): Fixed — non-Latin slug transliteration * changeset(#2848): backfill PR number 2934 --------- Co-authored-by: sim --- .changeset/zesty-dogs-swim.md | 5 +++ src/core-utils.cts | 52 ++++++++++++++++++++++++- src/gsd2-import.cts | 9 ++++- tests/core-utils.test.cjs | 72 +++++++++++++++++++++++++++++++++++ tests/gsd2-import.test.cjs | 27 +++++++++++++ 5 files changed, 163 insertions(+), 2 deletions(-) create mode 100644 .changeset/zesty-dogs-swim.md diff --git a/.changeset/zesty-dogs-swim.md b/.changeset/zesty-dogs-swim.md new file mode 100644 index 000000000..7a2788fb5 --- /dev/null +++ b/.changeset/zesty-dogs-swim.md @@ -0,0 +1,5 @@ +--- +type: Fixed +pr: 2934 +--- +**Non-Latin phase and milestone titles no longer produce empty slugs** — a Cyrillic title used to reduce to an empty slug, creating unnamed phase directories (bare numeric prefix like `01-`) and empty `milestone_slug` fields. Titles are now transliterated to ASCII before the slug filter, so a non-Latin title yields a usable slug. Latin-script output is unchanged. (#2848) diff --git a/src/core-utils.cts b/src/core-utils.cts index cc7f77291..b3ec0690c 100644 --- a/src/core-utils.cts +++ b/src/core-utils.cts @@ -95,7 +95,56 @@ function pathExistsInternal(cwd: string, targetPath: string): boolean { function generateSlugInternal(text: string | null | undefined): string | null { if (!text) return null; - return text.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').substring(0, 60); + return transliterateForSlug(text).replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').substring(0, 60); +} + +// ─── Transliteration (#2848) ───────────────────────────────────────────────── +// +// Non-Latin titles used to reduce to an empty slug: the `[^a-z0-9]+` strip +// removed every character of an all-Cyrillic title and the hyphen cleanup left +// "". Callers then created unnamed phase directories (`01-`) and empty +// `milestone_slug` init JSON. The fix transliterates Cyrillic to ASCII BEFORE +// the existing ASCII filter, so a non-Latin title yields a usable ASCII slug +// while Latin-script text (which hits zero map entries) is byte-for-byte +// unchanged — the negative control is satisfied by construction. +// +// Multi-letter mappings (ж→zh, ч→ch, ш→sh, щ→sch, ю→yu, я→ya) are applied as a +// single pass; soft/hard signs (ъ, ь) drop to nothing rather than a hyphen. +// Scope is Cyrillic (Russian + the reported Ukrainian/Belarusian extras +// і ї є ґ ў) per the issue's confirmed-working patch. CJK and other +// non-transliterated scripts keep the existing strip-to-ASCII behavior. +const CYRILLIC_TRANSLITERATION: Readonly> = { + // multi-letter first (longest-match-safe within a single pass via ordered keys) + а: 'a', б: 'b', в: 'v', г: 'g', д: 'd', е: 'e', ё: 'e', ж: 'zh', + з: 'z', и: 'i', й: 'y', к: 'k', л: 'l', м: 'm', н: 'n', о: 'o', + п: 'p', р: 'r', с: 's', т: 't', у: 'u', ф: 'f', х: 'h', ц: 'ts', + ч: 'ch', ш: 'sh', щ: 'sch', ъ: '', ы: 'y', ь: '', э: 'e', ю: 'yu', + я: 'ya', + // Ukrainian / Belarusian extras reported in #2848 + є: 'ye', і: 'i', ї: 'yi', ґ: 'g', ў: 'u', +}; + +const CYRILLIC_TRANSLITERATION_KEYS = Object.keys(CYRILLIC_TRANSLITERATION); + +/** + * Lowercase + transliterate Cyrillic characters to ASCII. The output still + * contains non-ASCII for scripts outside the map (CJK, etc.) — the caller's + * existing `[^a-z0-9]+` filter handles those. Latin-script input is returned + * lowercased with no other change. + * + * Shared by `generateSlugInternal` (core-utils) and `slugify` (gsd2-import) so + * the transliteration step is not duplicated across the two slug helpers (#2848 + * explicitly requires both be fixed). + */ +function transliterateForSlug(text: string): string { + const lowered = text.toLowerCase(); + let out = ''; + for (const ch of lowered) { + out += CYRILLIC_TRANSLITERATION_KEYS.includes(ch) + ? CYRILLIC_TRANSLITERATION[ch] + : ch; + } + return out; } // ─── Phase file helpers ────────────────────────────────────────────────────── @@ -248,6 +297,7 @@ export = { extractOneLinerFromBody, pathExistsInternal, generateSlugInternal, + transliterateForSlug, filterPlanFiles, filterSummaryFiles, getPhaseFileStats, diff --git a/src/gsd2-import.cts b/src/gsd2-import.cts index acb245bd3..54cf0a396 100644 --- a/src/gsd2-import.cts +++ b/src/gsd2-import.cts @@ -24,9 +24,12 @@ import path from 'node:path'; import { platformWriteSync } from './shell-command-projection.cjs'; import { formatGsdSlash, resolveRuntime } from './runtime-slash.cjs'; import { realClock } from './clock.cjs'; +// eslint-disable-next-line @typescript-eslint/no-require-imports -- core-utils.cjs is an export= CommonJS module +import coreUtilsMod = require('./core-utils.cjs'); // eslint-disable-next-line @typescript-eslint/no-require-imports import ioMod = require('./io.cjs'); const { output } = ioMod; +const { transliterateForSlug } = coreUtilsMod; // ─── Types ─────────────────────────────────────────────────────────────────── @@ -88,7 +91,11 @@ function zeroPad(n: number, width = 2): string { } function slugify(title: string): string { - return title.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); + // #2848: transliterate Cyrillic to ASCII before the filter so a non-Latin + // title does not collapse to an empty slug. The shared primitive keeps this + // in sync with generateSlugInternal. slugify's DISTINCT contract is preserved: + // single leading/trailing hyphen strip (/^-|-$/), and NO 60-char truncation. + return transliterateForSlug(title).replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, ''); } // ─── GSD-2 Parser ─────────────────────────────────────────────────────────── diff --git a/tests/core-utils.test.cjs b/tests/core-utils.test.cjs index e2f7dbe40..3d416d1b5 100644 --- a/tests/core-utils.test.cjs +++ b/tests/core-utils.test.cjs @@ -212,6 +212,78 @@ describe('generateSlugInternal', () => { test('preserves numbers in slug', () => { assert.strictEqual(coreUtils.generateSlugInternal('Phase 42 Done'), 'phase-42-done'); }); + + // ─── #2858 — wait, #2848: non-Latin (Cyrillic) titles must not produce an + // empty slug. A transliteration map is applied before the ASCII filter so the + // title's meaning is preserved as ASCII. Latin-script output is byte-for-byte + // unchanged (negative control below). + + test('#2848 row 1 — Cyrillic title produces a non-empty transliterated slug', () => { + // Russian "Проверка гипотезы" → "proverka gipotezy" → slug. + const result = coreUtils.generateSlugInternal('Проверка гипотезы'); + assert.ok(typeof result === 'string' && result.length > 0, `Cyrillic title must not produce an empty slug; got: ${JSON.stringify(result)}`); + assert.ok(/^[a-z0-9]+(-[a-z0-9]+)*$/.test(result), `slug must be ASCII-only and well-formed; got: ${result}`); + assert.strictEqual(result, 'proverka-gipotezy'); + }); + + test('#2848 row 3 — Latin-script output is byte-for-byte unchanged (negative control)', () => { + // These must remain identical to the pre-fix outputs. + assert.strictEqual(coreUtils.generateSlugInternal('Hello World!'), 'hello-world'); + assert.strictEqual(coreUtils.generateSlugInternal('Setup environment'), 'setup-environment'); + assert.strictEqual(coreUtils.generateSlugInternal(' Hello '), 'hello'); + assert.strictEqual(coreUtils.generateSlugInternal('Phase 42 Done'), 'phase-42-done'); + }); + + test('#2848 row 4 — multi-letter Cyrillic mappings transliterate correctly', () => { + // ж→zh ч→ch ш→sh щ→sch ю→yu я→ya. 'Яша Щучин' → ya-sh-a + sch-u-ch-i-n. + const result = coreUtils.generateSlugInternal('Яша Щучин'); + assert.ok(result, `expected non-empty slug; got: ${result}`); + assert.ok(result.includes('yasha'), `я→ya + ш→sh + а→a = yasha expected; got: ${result}`); + assert.ok(result.includes('schuchin'), `щ→sch + у→u + ч→ch expected; got: ${result}`); + assert.strictEqual(result, 'yasha-schuchin'); + // Spot-check each multi-letter mapping in isolation. + assert.strictEqual(coreUtils.generateSlugInternal('ж'), 'zh'); + assert.strictEqual(coreUtils.generateSlugInternal('ч'), 'ch'); + assert.strictEqual(coreUtils.generateSlugInternal('ш'), 'sh'); + assert.strictEqual(coreUtils.generateSlugInternal('щ'), 'sch'); + assert.strictEqual(coreUtils.generateSlugInternal('ю'), 'yu'); + assert.strictEqual(coreUtils.generateSlugInternal('я'), 'ya'); + }); + + test('#2848 row 5 — soft/hard signs (ъ ь) drop cleanly without hyphen runs', () => { + // "Объект день" — ъ and ь should disappear, NOT produce consecutive hyphens. + const result = coreUtils.generateSlugInternal('Объект день'); + assert.ok(result, `expected non-empty slug; got: ${result}`); + assert.ok(!result.includes('--'), `no double hyphens from dropped signs; got: ${result}`); + assert.strictEqual(result, 'obekt-den'); + }); + + test('#2848 row 6 — Ukrainian/Belarusian Cyrillic extras transliterate', () => { + // і ї є ґ ў — non-Russian Cyrillic letters in the reported scope. + assert.strictEqual(coreUtils.generateSlugInternal('і'), 'i'); + assert.strictEqual(coreUtils.generateSlugInternal('ї'), 'yi'); + assert.strictEqual(coreUtils.generateSlugInternal('є'), 'ye'); + assert.strictEqual(coreUtils.generateSlugInternal('ґ'), 'g'); + assert.strictEqual(coreUtils.generateSlugInternal('ў'), 'u'); + }); + + test('#2848 row 7 — null/undefined/empty still return null (contract preserved)', () => { + assert.strictEqual(coreUtils.generateSlugInternal(null), null); + assert.strictEqual(coreUtils.generateSlugInternal(undefined), null); + assert.strictEqual(coreUtils.generateSlugInternal(''), null); + }); + + test('#2848 row 9 — mixed Latin+Cyrillic title transliterates correctly', () => { + assert.strictEqual(coreUtils.generateSlugInternal('Phase Фаза 42'), 'phase-faza-42'); + }); + + test('#2848 row 10 — truncation still applies after transliteration (≤60 chars)', () => { + // A long Cyrillic title transliterates to a longer ASCII string; the 60-char + // cap must still bind the result. + const long = 'Проверка'.repeat(20); + const result = coreUtils.generateSlugInternal(long); + assert.ok(result !== null && result.length <= 60, `truncation must still apply; got len ${result && result.length}`); + }); }); // ─── filterPlanFiles ────────────────────────────────────────────────────────── diff --git a/tests/gsd2-import.test.cjs b/tests/gsd2-import.test.cjs index 16a26a34d..e7d33b55b 100644 --- a/tests/gsd2-import.test.cjs +++ b/tests/gsd2-import.test.cjs @@ -262,6 +262,33 @@ describe('slugify', () => { test('strips leading/trailing hyphens', () => { assert.strictEqual(slugify(' spaces '), 'spaces'); }); + + // ─── #2848: non-Latin (Cyrillic) titles must not produce an empty slug. The + // transliteration map (shared with generateSlugInternal) is applied before the + // ASCII filter. slugify keeps its DISTINCT contract: single leading/trailing + // hyphen strip, NO truncation. + + test('#2848 row 2 — Cyrillic title produces a non-empty transliterated slug', () => { + const result = slugify('Настройка окружения'); + assert.ok(typeof result === 'string' && result.length > 0, `Cyrillic title must not produce an empty slug; got: ${JSON.stringify(result)}`); + assert.ok(/^[a-z0-9]+(-[a-z0-9]+)*$/.test(result), `slug must be ASCII-only and well-formed; got: ${result}`); + assert.strictEqual(result, 'nastroyka-okruzheniya'); + }); + + test('#2848 row 3 — Latin-script output is byte-for-byte unchanged (negative control)', () => { + assert.strictEqual(slugify('Auth System'), 'auth-system'); + assert.strictEqual(slugify('My Feature (v2)'), 'my-feature-v2'); + assert.strictEqual(slugify(' spaces '), 'spaces'); + }); + + test('#2848 row 11 — slugify does NOT truncate (distinct from generateSlugInternal contract)', () => { + // generateSlugInternal truncates at 60; slugify must NOT — that difference is + // existing, documented behavior, not the bug. A long Cyrillic title produces a + // long ASCII slug that exceeds 60 chars. + const long = 'Настройка'.repeat(20); + const result = slugify(long); + assert.ok(result.length > 60, `slugify must not truncate; got len ${result.length}`); + }); }); describe('zeroPad', () => {