test(i18n): Waechter gegen neue Umlaut-Ersatzschreibweisen
Legt umlaut-guard.spec.ts an, das ausschliesslich das geparste JSON von de.json/en.json liest (nie repo-weit ueber Quelldateien greift, sonst schluege es am Woerterbuch selbst an) und drei Dinge prueft: kein Token aus UMLAUT_REPLACEMENTS mehr in de.json, jedes verbliebene ae/oe/ue/ss-Token steht auf UMLAUT_ALLOWLIST, und de.json/en.json haben denselben Schluesselsatz. Manuell mit einem probeweise eingefuegten "fuer" verifiziert: Meldung nennt "für" und den Schluesselpfad, danach zurueckgenommen. Beim Aufbau des Waechters kamen zwei Luecken der Task-1-Wortliste ans Licht: "Bestaetigen" (Grossschreibung, common.confirm) fehlte in UMLAUT_REPLACEMENTS und blieb faelschlich falsch geschrieben; die Ersetzungen zu Passwoerter/vertrauenswuerdigen/ausschliessen/ entschluesselt ergeben nach der Korrektur korrektes Deutsch, das aber weiterhin ae/oe/ue/ss enthaelt und deshalb auf UMLAUT_ALLOWLIST ergaenzt werden musste. Beide Luecken behoben (Rule 1 - Bug). pnpm --filter @tessera/web test: 38 Testdateien, 225 Tests gruen. pnpm --filter @tessera/web run type-check: sauber. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K5jtbGzC5Sf9npJ3JCjKhq
This commit is contained in:
@@ -5,7 +5,7 @@
|
||||
"save": "Speichern",
|
||||
"cancel": "Abbrechen",
|
||||
"menu": "Menu",
|
||||
"confirm": "Bestaetigen",
|
||||
"confirm": "Bestätigen",
|
||||
"delete": "Löschen",
|
||||
"edit": "Bearbeiten",
|
||||
"create": "Erstellen",
|
||||
|
||||
@@ -59,6 +59,7 @@ export const UMLAUT_REPLACEMENTS: Record<string, string> = {
|
||||
Zusammenfuehren: 'Zusammenführen',
|
||||
zusammenfuehren: 'zusammenführen',
|
||||
bestaetigen: 'bestätigen',
|
||||
Bestaetigen: 'Bestätigen',
|
||||
Passwoerter: 'Passwörter',
|
||||
spaeter: 'später',
|
||||
ueberein: 'überein',
|
||||
@@ -157,4 +158,8 @@ export const UMLAUT_ALLOWLIST: readonly string[] = [
|
||||
'Fasst',
|
||||
'Verschlüsselung',
|
||||
'empfaenger',
|
||||
'Passwörter',
|
||||
'vertrauenswürdigen',
|
||||
'ausschließen',
|
||||
'entschlüsselt',
|
||||
];
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import de from './de.json';
|
||||
import en from './en.json';
|
||||
import { UMLAUT_ALLOWLIST, UMLAUT_REPLACEMENTS } from './umlaut-dictionary';
|
||||
|
||||
/**
|
||||
* umlaut-guard — regression guard for the de.json umlaut correction.
|
||||
*
|
||||
* This spec reads exclusively the parsed JSON content of `de.json` /
|
||||
* `en.json` and never greps repo-wide over source files. Grepping
|
||||
* source files would trigger on `umlaut-dictionary.ts` itself, since it
|
||||
* necessarily contains the wrong substitute spellings as its map keys
|
||||
* (e.g. `'fuer'`, `'loeschen'`).
|
||||
*
|
||||
* The second test below is the actual regression guard: it fails on any
|
||||
* NEWLY introduced substitute spelling (`ae`/`oe`/`ue`/`ss` token not on
|
||||
* the allowlist) without pinning today's wording — it pins vocabulary,
|
||||
* not phrasing. Fixing a failure is a one-line addition to either
|
||||
* `UMLAUT_REPLACEMENTS` (if it is a substitute spelling) or
|
||||
* `UMLAUT_ALLOWLIST` (if it is correct German), as instructed by the
|
||||
* assertion message itself.
|
||||
*/
|
||||
|
||||
const UMLAUT_TOKEN_RE = /[A-Za-zÄÖÜäöüß]+/g;
|
||||
const SUSPECT_RE = /(ae|oe|ue|ss)/;
|
||||
|
||||
/** Recursively flattens a nested message object into dot-joined leaf key paths. */
|
||||
function flattenKeys(obj: unknown, prefix = ''): string[] {
|
||||
if (obj === null || typeof obj !== 'object') {
|
||||
return [prefix];
|
||||
}
|
||||
return Object.entries(obj as Record<string, unknown>).flatMap(([key, value]) =>
|
||||
flattenKeys(value, prefix ? `${prefix}.${key}` : key),
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Walks every leaf string value in `obj`, skipping values that contain
|
||||
* `@` (e.g. the `testToPlaceholder` example email address), and invokes
|
||||
* `visit` with each whole-word token found plus its dot-joined key path.
|
||||
*/
|
||||
function walkTokens(obj: unknown, visit: (token: string, path: string) => void, prefix = ''): void {
|
||||
if (obj === null || typeof obj !== 'object') {
|
||||
return;
|
||||
}
|
||||
for (const [key, value] of Object.entries(obj as Record<string, unknown>)) {
|
||||
const path = prefix ? `${prefix}.${key}` : key;
|
||||
if (typeof value === 'string') {
|
||||
if (value.includes('@')) {
|
||||
continue;
|
||||
}
|
||||
for (const token of value.match(UMLAUT_TOKEN_RE) ?? []) {
|
||||
visit(token, path);
|
||||
}
|
||||
} else if (value && typeof value === 'object') {
|
||||
walkTokens(value, visit, path);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
describe('de.json umlaut regression guard', () => {
|
||||
it('contains no token from UMLAUT_REPLACEMENTS (known substitute spelling)', () => {
|
||||
const hits: string[] = [];
|
||||
walkTokens(de, (token, path) => {
|
||||
if (Object.prototype.hasOwnProperty.call(UMLAUT_REPLACEMENTS, token)) {
|
||||
hits.push(`${path}: "${token}" should be "${UMLAUT_REPLACEMENTS[token]}"`);
|
||||
}
|
||||
});
|
||||
expect(hits, `Substitute spellings found in de.json:\n${hits.join('\n')}`).toEqual([]);
|
||||
});
|
||||
|
||||
it('flags any new ae/oe/ue/ss token not on UMLAUT_ALLOWLIST', () => {
|
||||
const allowlist = new Set(UMLAUT_ALLOWLIST);
|
||||
const hits: string[] = [];
|
||||
walkTokens(de, (token, path) => {
|
||||
if (SUSPECT_RE.test(token) && !allowlist.has(token)) {
|
||||
hits.push(
|
||||
`${path}: "${token}" is a new word not on UMLAUT_ALLOWLIST. ` +
|
||||
`Add it to UMLAUT_REPLACEMENTS in umlaut-dictionary.ts if it is a substitute ` +
|
||||
`spelling, or to UMLAUT_ALLOWLIST if it is already correct German.`,
|
||||
);
|
||||
}
|
||||
});
|
||||
expect(hits, `New ae/oe/ue/ss words found in de.json:\n${hits.join('\n')}`).toEqual([]);
|
||||
});
|
||||
|
||||
it('has an identical (recursively-flattened) key set in de and en', () => {
|
||||
const deKeys = flattenKeys(de).sort();
|
||||
const enKeys = flattenKeys(en).sort();
|
||||
|
||||
const missingInEn = deKeys.filter((key) => !enKeys.includes(key));
|
||||
const missingInDe = enKeys.filter((key) => !deKeys.includes(key));
|
||||
|
||||
expect(missingInEn, `Keys present in de.json but missing in en.json: ${missingInEn.join(', ')}`).toEqual([]);
|
||||
expect(missingInDe, `Keys present in en.json but missing in de.json: ${missingInDe.join(', ')}`).toEqual([]);
|
||||
expect(deKeys).toEqual(enKeys);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user