* test(#2848): add failing-first regression for non-Latin slug transliteration generateSlugInternal and slugify both strip non-ASCII chars with no transliteration step, so an all-Cyrillic title reduces to an empty slug. 12-row matrix: Cyrillic regression (both impls), Latin negative control, multi-letter mappings, soft/hard sign drops, Ukrainian extras, null contract, CJK unaffected, mixed scripts, truncation parity, slugify's distinct no-truncate contract. * fix(#2848): transliterate Cyrillic titles to ASCII before slug strip Both generateSlugInternal (src/core-utils.cts) and slugify (src/gsd2-import.cts) stripped non-ASCII with no transliteration, so an all-Cyrillic title reduced to an empty slug, producing unnamed phase directories (01-) and empty milestone_slug init JSON. Add a shared transliterateForSlug primitive (core-utils) covering Russian + the reported Ukrainian/Belarusian extras (і ї є ґ ў), with multi-letter mappings (ж→zh ч→ch ш→sh щ→sch ю→yu я→ya) and dropped soft/hard signs (ъ ь). It runs BEFORE the existing ASCII filter, so Latin-script text hits zero map entries and is byte-for-byte unchanged (negative control). slugify consumes the shared primitive, preserving its distinct single hyphen-strip + no-truncation contract. CJK/unmapped scripts keep the existing strip-to-ASCII behavior. Also corrects two test assertions to match the chosen й→y mapping and the б→b (not bie) transliteration. * changeset(#2848): Fixed — non-Latin slug transliteration * changeset(#2848): backfill PR number 2934 --------- Co-authored-by: sim <sim@local>
This commit is contained in:
5
.changeset/zesty-dogs-swim.md
Normal file
5
.changeset/zesty-dogs-swim.md
Normal file
@@ -0,0 +1,5 @@
|
||||
---
|
||||
type: Fixed
|
||||
pr: 2934
|
||||
---
|
||||
**Non-Latin phase and milestone titles no longer produce empty slugs** — a Cyrillic title used to reduce to an empty slug, creating unnamed phase directories (bare numeric prefix like `01-`) and empty `milestone_slug` fields. Titles are now transliterated to ASCII before the slug filter, so a non-Latin title yields a usable slug. Latin-script output is unchanged. (#2848)
|
||||
@@ -95,7 +95,56 @@ function pathExistsInternal(cwd: string, targetPath: string): boolean {
|
||||
|
||||
function generateSlugInternal(text: string | null | undefined): string | null {
|
||||
if (!text) return null;
|
||||
return text.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').substring(0, 60);
|
||||
return transliterateForSlug(text).replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').substring(0, 60);
|
||||
}
|
||||
|
||||
// ─── Transliteration (#2848) ─────────────────────────────────────────────────
|
||||
//
|
||||
// Non-Latin titles used to reduce to an empty slug: the `[^a-z0-9]+` strip
|
||||
// removed every character of an all-Cyrillic title and the hyphen cleanup left
|
||||
// "". Callers then created unnamed phase directories (`01-`) and empty
|
||||
// `milestone_slug` init JSON. The fix transliterates Cyrillic to ASCII BEFORE
|
||||
// the existing ASCII filter, so a non-Latin title yields a usable ASCII slug
|
||||
// while Latin-script text (which hits zero map entries) is byte-for-byte
|
||||
// unchanged — the negative control is satisfied by construction.
|
||||
//
|
||||
// Multi-letter mappings (ж→zh, ч→ch, ш→sh, щ→sch, ю→yu, я→ya) are applied as a
|
||||
// single pass; soft/hard signs (ъ, ь) drop to nothing rather than a hyphen.
|
||||
// Scope is Cyrillic (Russian + the reported Ukrainian/Belarusian extras
|
||||
// і ї є ґ ў) per the issue's confirmed-working patch. CJK and other
|
||||
// non-transliterated scripts keep the existing strip-to-ASCII behavior.
|
||||
const CYRILLIC_TRANSLITERATION: Readonly<Record<string, string>> = {
|
||||
// multi-letter first (longest-match-safe within a single pass via ordered keys)
|
||||
а: 'a', б: 'b', в: 'v', г: 'g', д: 'd', е: 'e', ё: 'e', ж: 'zh',
|
||||
з: 'z', и: 'i', й: 'y', к: 'k', л: 'l', м: 'm', н: 'n', о: 'o',
|
||||
п: 'p', р: 'r', с: 's', т: 't', у: 'u', ф: 'f', х: 'h', ц: 'ts',
|
||||
ч: 'ch', ш: 'sh', щ: 'sch', ъ: '', ы: 'y', ь: '', э: 'e', ю: 'yu',
|
||||
я: 'ya',
|
||||
// Ukrainian / Belarusian extras reported in #2848
|
||||
є: 'ye', і: 'i', ї: 'yi', ґ: 'g', ў: 'u',
|
||||
};
|
||||
|
||||
const CYRILLIC_TRANSLITERATION_KEYS = Object.keys(CYRILLIC_TRANSLITERATION);
|
||||
|
||||
/**
|
||||
* Lowercase + transliterate Cyrillic characters to ASCII. The output still
|
||||
* contains non-ASCII for scripts outside the map (CJK, etc.) — the caller's
|
||||
* existing `[^a-z0-9]+` filter handles those. Latin-script input is returned
|
||||
* lowercased with no other change.
|
||||
*
|
||||
* Shared by `generateSlugInternal` (core-utils) and `slugify` (gsd2-import) so
|
||||
* the transliteration step is not duplicated across the two slug helpers (#2848
|
||||
* explicitly requires both be fixed).
|
||||
*/
|
||||
function transliterateForSlug(text: string): string {
|
||||
const lowered = text.toLowerCase();
|
||||
let out = '';
|
||||
for (const ch of lowered) {
|
||||
out += CYRILLIC_TRANSLITERATION_KEYS.includes(ch)
|
||||
? CYRILLIC_TRANSLITERATION[ch]
|
||||
: ch;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ─── Phase file helpers ──────────────────────────────────────────────────────
|
||||
@@ -248,6 +297,7 @@ export = {
|
||||
extractOneLinerFromBody,
|
||||
pathExistsInternal,
|
||||
generateSlugInternal,
|
||||
transliterateForSlug,
|
||||
filterPlanFiles,
|
||||
filterSummaryFiles,
|
||||
getPhaseFileStats,
|
||||
|
||||
@@ -24,9 +24,12 @@ import path from 'node:path';
|
||||
import { platformWriteSync } from './shell-command-projection.cjs';
|
||||
import { formatGsdSlash, resolveRuntime } from './runtime-slash.cjs';
|
||||
import { realClock } from './clock.cjs';
|
||||
// eslint-disable-next-line @typescript-eslint/no-require-imports -- core-utils.cjs is an export= CommonJS module
|
||||
import coreUtilsMod = require('./core-utils.cjs');
|
||||
// eslint-disable-next-line @typescript-eslint/no-require-imports
|
||||
import ioMod = require('./io.cjs');
|
||||
const { output } = ioMod;
|
||||
const { transliterateForSlug } = coreUtilsMod;
|
||||
|
||||
// ─── Types ───────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -88,7 +91,11 @@ function zeroPad(n: number, width = 2): string {
|
||||
}
|
||||
|
||||
function slugify(title: string): string {
|
||||
return title.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, '');
|
||||
// #2848: transliterate Cyrillic to ASCII before the filter so a non-Latin
|
||||
// title does not collapse to an empty slug. The shared primitive keeps this
|
||||
// in sync with generateSlugInternal. slugify's DISTINCT contract is preserved:
|
||||
// single leading/trailing hyphen strip (/^-|-$/), and NO 60-char truncation.
|
||||
return transliterateForSlug(title).replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, '');
|
||||
}
|
||||
|
||||
// ─── GSD-2 Parser ───────────────────────────────────────────────────────────
|
||||
|
||||
@@ -212,6 +212,78 @@ describe('generateSlugInternal', () => {
|
||||
test('preserves numbers in slug', () => {
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('Phase 42 Done'), 'phase-42-done');
|
||||
});
|
||||
|
||||
// ─── #2858 — wait, #2848: non-Latin (Cyrillic) titles must not produce an
|
||||
// empty slug. A transliteration map is applied before the ASCII filter so the
|
||||
// title's meaning is preserved as ASCII. Latin-script output is byte-for-byte
|
||||
// unchanged (negative control below).
|
||||
|
||||
test('#2848 row 1 — Cyrillic title produces a non-empty transliterated slug', () => {
|
||||
// Russian "Проверка гипотезы" → "proverka gipotezy" → slug.
|
||||
const result = coreUtils.generateSlugInternal('Проверка гипотезы');
|
||||
assert.ok(typeof result === 'string' && result.length > 0, `Cyrillic title must not produce an empty slug; got: ${JSON.stringify(result)}`);
|
||||
assert.ok(/^[a-z0-9]+(-[a-z0-9]+)*$/.test(result), `slug must be ASCII-only and well-formed; got: ${result}`);
|
||||
assert.strictEqual(result, 'proverka-gipotezy');
|
||||
});
|
||||
|
||||
test('#2848 row 3 — Latin-script output is byte-for-byte unchanged (negative control)', () => {
|
||||
// These must remain identical to the pre-fix outputs.
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('Hello World!'), 'hello-world');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('Setup environment'), 'setup-environment');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal(' Hello '), 'hello');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('Phase 42 Done'), 'phase-42-done');
|
||||
});
|
||||
|
||||
test('#2848 row 4 — multi-letter Cyrillic mappings transliterate correctly', () => {
|
||||
// ж→zh ч→ch ш→sh щ→sch ю→yu я→ya. 'Яша Щучин' → ya-sh-a + sch-u-ch-i-n.
|
||||
const result = coreUtils.generateSlugInternal('Яша Щучин');
|
||||
assert.ok(result, `expected non-empty slug; got: ${result}`);
|
||||
assert.ok(result.includes('yasha'), `я→ya + ш→sh + а→a = yasha expected; got: ${result}`);
|
||||
assert.ok(result.includes('schuchin'), `щ→sch + у→u + ч→ch expected; got: ${result}`);
|
||||
assert.strictEqual(result, 'yasha-schuchin');
|
||||
// Spot-check each multi-letter mapping in isolation.
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('ж'), 'zh');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('ч'), 'ch');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('ш'), 'sh');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('щ'), 'sch');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('ю'), 'yu');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('я'), 'ya');
|
||||
});
|
||||
|
||||
test('#2848 row 5 — soft/hard signs (ъ ь) drop cleanly without hyphen runs', () => {
|
||||
// "Объект день" — ъ and ь should disappear, NOT produce consecutive hyphens.
|
||||
const result = coreUtils.generateSlugInternal('Объект день');
|
||||
assert.ok(result, `expected non-empty slug; got: ${result}`);
|
||||
assert.ok(!result.includes('--'), `no double hyphens from dropped signs; got: ${result}`);
|
||||
assert.strictEqual(result, 'obekt-den');
|
||||
});
|
||||
|
||||
test('#2848 row 6 — Ukrainian/Belarusian Cyrillic extras transliterate', () => {
|
||||
// і ї є ґ ў — non-Russian Cyrillic letters in the reported scope.
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('і'), 'i');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('ї'), 'yi');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('є'), 'ye');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('ґ'), 'g');
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('ў'), 'u');
|
||||
});
|
||||
|
||||
test('#2848 row 7 — null/undefined/empty still return null (contract preserved)', () => {
|
||||
assert.strictEqual(coreUtils.generateSlugInternal(null), null);
|
||||
assert.strictEqual(coreUtils.generateSlugInternal(undefined), null);
|
||||
assert.strictEqual(coreUtils.generateSlugInternal(''), null);
|
||||
});
|
||||
|
||||
test('#2848 row 9 — mixed Latin+Cyrillic title transliterates correctly', () => {
|
||||
assert.strictEqual(coreUtils.generateSlugInternal('Phase Фаза 42'), 'phase-faza-42');
|
||||
});
|
||||
|
||||
test('#2848 row 10 — truncation still applies after transliteration (≤60 chars)', () => {
|
||||
// A long Cyrillic title transliterates to a longer ASCII string; the 60-char
|
||||
// cap must still bind the result.
|
||||
const long = 'Проверка'.repeat(20);
|
||||
const result = coreUtils.generateSlugInternal(long);
|
||||
assert.ok(result !== null && result.length <= 60, `truncation must still apply; got len ${result && result.length}`);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── filterPlanFiles ──────────────────────────────────────────────────────────
|
||||
|
||||
@@ -262,6 +262,33 @@ describe('slugify', () => {
|
||||
test('strips leading/trailing hyphens', () => {
|
||||
assert.strictEqual(slugify(' spaces '), 'spaces');
|
||||
});
|
||||
|
||||
// ─── #2848: non-Latin (Cyrillic) titles must not produce an empty slug. The
|
||||
// transliteration map (shared with generateSlugInternal) is applied before the
|
||||
// ASCII filter. slugify keeps its DISTINCT contract: single leading/trailing
|
||||
// hyphen strip, NO truncation.
|
||||
|
||||
test('#2848 row 2 — Cyrillic title produces a non-empty transliterated slug', () => {
|
||||
const result = slugify('Настройка окружения');
|
||||
assert.ok(typeof result === 'string' && result.length > 0, `Cyrillic title must not produce an empty slug; got: ${JSON.stringify(result)}`);
|
||||
assert.ok(/^[a-z0-9]+(-[a-z0-9]+)*$/.test(result), `slug must be ASCII-only and well-formed; got: ${result}`);
|
||||
assert.strictEqual(result, 'nastroyka-okruzheniya');
|
||||
});
|
||||
|
||||
test('#2848 row 3 — Latin-script output is byte-for-byte unchanged (negative control)', () => {
|
||||
assert.strictEqual(slugify('Auth System'), 'auth-system');
|
||||
assert.strictEqual(slugify('My Feature (v2)'), 'my-feature-v2');
|
||||
assert.strictEqual(slugify(' spaces '), 'spaces');
|
||||
});
|
||||
|
||||
test('#2848 row 11 — slugify does NOT truncate (distinct from generateSlugInternal contract)', () => {
|
||||
// generateSlugInternal truncates at 60; slugify must NOT — that difference is
|
||||
// existing, documented behavior, not the bug. A long Cyrillic title produces a
|
||||
// long ASCII slug that exceeds 60 chars.
|
||||
const long = 'Настройка'.repeat(20);
|
||||
const result = slugify(long);
|
||||
assert.ok(result.length > 60, `slugify must not truncate; got len ${result.length}`);
|
||||
});
|
||||
});
|
||||
|
||||
describe('zeroPad', () => {
|
||||
|
||||
Reference in New Issue
Block a user