From caa4827360ab2dee332c0e8e81af4841b54123fc Mon Sep 17 00:00:00 2001 From: Schalli Date: Mon, 7 Sep 2026 13:06:25 +0200 Subject: [PATCH] test(i18n): Waechter gegen neue Umlaut-Ersatzschreibweisen MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Legt umlaut-guard.spec.ts an, das ausschliesslich das geparste JSON von de.json/en.json liest (nie repo-weit ueber Quelldateien greift, sonst schluege es am Woerterbuch selbst an) und drei Dinge prueft: kein Token aus UMLAUT_REPLACEMENTS mehr in de.json, jedes verbliebene ae/oe/ue/ss-Token steht auf UMLAUT_ALLOWLIST, und de.json/en.json haben denselben Schluesselsatz. Manuell mit einem probeweise eingefuegten "fuer" verifiziert: Meldung nennt "für" und den Schluesselpfad, danach zurueckgenommen. Beim Aufbau des Waechters kamen zwei Luecken der Task-1-Wortliste ans Licht: "Bestaetigen" (Grossschreibung, common.confirm) fehlte in UMLAUT_REPLACEMENTS und blieb faelschlich falsch geschrieben; die Ersetzungen zu Passwoerter/vertrauenswuerdigen/ausschliessen/ entschluesselt ergeben nach der Korrektur korrektes Deutsch, das aber weiterhin ae/oe/ue/ss enthaelt und deshalb auf UMLAUT_ALLOWLIST ergaenzt werden musste. Beide Luecken behoben (Rule 1 - Bug). pnpm --filter @tessera/web test: 38 Testdateien, 225 Tests gruen. pnpm --filter @tessera/web run type-check: sauber. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01K5jtbGzC5Sf9npJ3JCjKhq --- apps/web/src/messages/de.json | 2 +- apps/web/src/messages/umlaut-dictionary.ts | 5 ++ apps/web/src/messages/umlaut-guard.spec.ts | 98 ++++++++++++++++++++++ 3 files changed, 104 insertions(+), 1 deletion(-) create mode 100644 apps/web/src/messages/umlaut-guard.spec.ts diff --git a/apps/web/src/messages/de.json b/apps/web/src/messages/de.json index 0e47de6..0205fa8 100644 --- a/apps/web/src/messages/de.json +++ b/apps/web/src/messages/de.json @@ -5,7 +5,7 @@ "save": "Speichern", "cancel": "Abbrechen", "menu": "Menu", - "confirm": "Bestaetigen", + "confirm": "Bestätigen", "delete": "Löschen", "edit": "Bearbeiten", "create": "Erstellen", diff --git a/apps/web/src/messages/umlaut-dictionary.ts b/apps/web/src/messages/umlaut-dictionary.ts index c0e41c8..b6ccd32 100644 --- a/apps/web/src/messages/umlaut-dictionary.ts +++ b/apps/web/src/messages/umlaut-dictionary.ts @@ -59,6 +59,7 @@ export const UMLAUT_REPLACEMENTS: Record = { Zusammenfuehren: 'Zusammenführen', zusammenfuehren: 'zusammenführen', bestaetigen: 'bestätigen', + Bestaetigen: 'Bestätigen', Passwoerter: 'Passwörter', spaeter: 'später', ueberein: 'überein', @@ -157,4 +158,8 @@ export const UMLAUT_ALLOWLIST: readonly string[] = [ 'Fasst', 'Verschlüsselung', 'empfaenger', + 'Passwörter', + 'vertrauenswürdigen', + 'ausschließen', + 'entschlüsselt', ]; diff --git a/apps/web/src/messages/umlaut-guard.spec.ts b/apps/web/src/messages/umlaut-guard.spec.ts new file mode 100644 index 0000000..e12fb4b --- /dev/null +++ b/apps/web/src/messages/umlaut-guard.spec.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from 'vitest'; +import de from './de.json'; +import en from './en.json'; +import { UMLAUT_ALLOWLIST, UMLAUT_REPLACEMENTS } from './umlaut-dictionary'; + +/** + * umlaut-guard — regression guard for the de.json umlaut correction. + * + * This spec reads exclusively the parsed JSON content of `de.json` / + * `en.json` and never greps repo-wide over source files. Grepping + * source files would trigger on `umlaut-dictionary.ts` itself, since it + * necessarily contains the wrong substitute spellings as its map keys + * (e.g. `'fuer'`, `'loeschen'`). + * + * The second test below is the actual regression guard: it fails on any + * NEWLY introduced substitute spelling (`ae`/`oe`/`ue`/`ss` token not on + * the allowlist) without pinning today's wording — it pins vocabulary, + * not phrasing. Fixing a failure is a one-line addition to either + * `UMLAUT_REPLACEMENTS` (if it is a substitute spelling) or + * `UMLAUT_ALLOWLIST` (if it is correct German), as instructed by the + * assertion message itself. + */ + +const UMLAUT_TOKEN_RE = /[A-Za-zÄÖÜäöüß]+/g; +const SUSPECT_RE = /(ae|oe|ue|ss)/; + +/** Recursively flattens a nested message object into dot-joined leaf key paths. */ +function flattenKeys(obj: unknown, prefix = ''): string[] { + if (obj === null || typeof obj !== 'object') { + return [prefix]; + } + return Object.entries(obj as Record).flatMap(([key, value]) => + flattenKeys(value, prefix ? `${prefix}.${key}` : key), + ); +} + +/** + * Walks every leaf string value in `obj`, skipping values that contain + * `@` (e.g. the `testToPlaceholder` example email address), and invokes + * `visit` with each whole-word token found plus its dot-joined key path. + */ +function walkTokens(obj: unknown, visit: (token: string, path: string) => void, prefix = ''): void { + if (obj === null || typeof obj !== 'object') { + return; + } + for (const [key, value] of Object.entries(obj as Record)) { + const path = prefix ? `${prefix}.${key}` : key; + if (typeof value === 'string') { + if (value.includes('@')) { + continue; + } + for (const token of value.match(UMLAUT_TOKEN_RE) ?? []) { + visit(token, path); + } + } else if (value && typeof value === 'object') { + walkTokens(value, visit, path); + } + } +} + +describe('de.json umlaut regression guard', () => { + it('contains no token from UMLAUT_REPLACEMENTS (known substitute spelling)', () => { + const hits: string[] = []; + walkTokens(de, (token, path) => { + if (Object.prototype.hasOwnProperty.call(UMLAUT_REPLACEMENTS, token)) { + hits.push(`${path}: "${token}" should be "${UMLAUT_REPLACEMENTS[token]}"`); + } + }); + expect(hits, `Substitute spellings found in de.json:\n${hits.join('\n')}`).toEqual([]); + }); + + it('flags any new ae/oe/ue/ss token not on UMLAUT_ALLOWLIST', () => { + const allowlist = new Set(UMLAUT_ALLOWLIST); + const hits: string[] = []; + walkTokens(de, (token, path) => { + if (SUSPECT_RE.test(token) && !allowlist.has(token)) { + hits.push( + `${path}: "${token}" is a new word not on UMLAUT_ALLOWLIST. ` + + `Add it to UMLAUT_REPLACEMENTS in umlaut-dictionary.ts if it is a substitute ` + + `spelling, or to UMLAUT_ALLOWLIST if it is already correct German.`, + ); + } + }); + expect(hits, `New ae/oe/ue/ss words found in de.json:\n${hits.join('\n')}`).toEqual([]); + }); + + it('has an identical (recursively-flattened) key set in de and en', () => { + const deKeys = flattenKeys(de).sort(); + const enKeys = flattenKeys(en).sort(); + + const missingInEn = deKeys.filter((key) => !enKeys.includes(key)); + const missingInDe = enKeys.filter((key) => !deKeys.includes(key)); + + expect(missingInEn, `Keys present in de.json but missing in en.json: ${missingInEn.join(', ')}`).toEqual([]); + expect(missingInDe, `Keys present in en.json but missing in de.json: ${missingInDe.join(', ')}`).toEqual([]); + expect(deKeys).toEqual(enKeys); + }); +});