test(i18n): Waechter gegen neue Umlaut-Ersatzschreibweisen
Legt umlaut-guard.spec.ts an, das ausschliesslich das geparste JSON von de.json/en.json liest (nie repo-weit ueber Quelldateien greift, sonst schluege es am Woerterbuch selbst an) und drei Dinge prueft: kein Token aus UMLAUT_REPLACEMENTS mehr in de.json, jedes verbliebene ae/oe/ue/ss-Token steht auf UMLAUT_ALLOWLIST, und de.json/en.json haben denselben Schluesselsatz. Manuell mit einem probeweise eingefuegten "fuer" verifiziert: Meldung nennt "für" und den Schluesselpfad, danach zurueckgenommen. Beim Aufbau des Waechters kamen zwei Luecken der Task-1-Wortliste ans Licht: "Bestaetigen" (Grossschreibung, common.confirm) fehlte in UMLAUT_REPLACEMENTS und blieb faelschlich falsch geschrieben; die Ersetzungen zu Passwoerter/vertrauenswuerdigen/ausschliessen/ entschluesselt ergeben nach der Korrektur korrektes Deutsch, das aber weiterhin ae/oe/ue/ss enthaelt und deshalb auf UMLAUT_ALLOWLIST ergaenzt werden musste. Beide Luecken behoben (Rule 1 - Bug). pnpm --filter @tessera/web test: 38 Testdateien, 225 Tests gruen. pnpm --filter @tessera/web run type-check: sauber. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01K5jtbGzC5Sf9npJ3JCjKhq
This commit is contained in:
@@ -5,7 +5,7 @@
|
|||||||
"save": "Speichern",
|
"save": "Speichern",
|
||||||
"cancel": "Abbrechen",
|
"cancel": "Abbrechen",
|
||||||
"menu": "Menu",
|
"menu": "Menu",
|
||||||
"confirm": "Bestaetigen",
|
"confirm": "Bestätigen",
|
||||||
"delete": "Löschen",
|
"delete": "Löschen",
|
||||||
"edit": "Bearbeiten",
|
"edit": "Bearbeiten",
|
||||||
"create": "Erstellen",
|
"create": "Erstellen",
|
||||||
|
|||||||
@@ -59,6 +59,7 @@ export const UMLAUT_REPLACEMENTS: Record<string, string> = {
|
|||||||
Zusammenfuehren: 'Zusammenführen',
|
Zusammenfuehren: 'Zusammenführen',
|
||||||
zusammenfuehren: 'zusammenführen',
|
zusammenfuehren: 'zusammenführen',
|
||||||
bestaetigen: 'bestätigen',
|
bestaetigen: 'bestätigen',
|
||||||
|
Bestaetigen: 'Bestätigen',
|
||||||
Passwoerter: 'Passwörter',
|
Passwoerter: 'Passwörter',
|
||||||
spaeter: 'später',
|
spaeter: 'später',
|
||||||
ueberein: 'überein',
|
ueberein: 'überein',
|
||||||
@@ -157,4 +158,8 @@ export const UMLAUT_ALLOWLIST: readonly string[] = [
|
|||||||
'Fasst',
|
'Fasst',
|
||||||
'Verschlüsselung',
|
'Verschlüsselung',
|
||||||
'empfaenger',
|
'empfaenger',
|
||||||
|
'Passwörter',
|
||||||
|
'vertrauenswürdigen',
|
||||||
|
'ausschließen',
|
||||||
|
'entschlüsselt',
|
||||||
];
|
];
|
||||||
|
|||||||
@@ -0,0 +1,98 @@
|
|||||||
|
import { describe, expect, it } from 'vitest';
|
||||||
|
import de from './de.json';
|
||||||
|
import en from './en.json';
|
||||||
|
import { UMLAUT_ALLOWLIST, UMLAUT_REPLACEMENTS } from './umlaut-dictionary';
|
||||||
|
|
||||||
|
/**
|
||||||
|
* umlaut-guard — regression guard for the de.json umlaut correction.
|
||||||
|
*
|
||||||
|
* This spec reads exclusively the parsed JSON content of `de.json` /
|
||||||
|
* `en.json` and never greps repo-wide over source files. Grepping
|
||||||
|
* source files would trigger on `umlaut-dictionary.ts` itself, since it
|
||||||
|
* necessarily contains the wrong substitute spellings as its map keys
|
||||||
|
* (e.g. `'fuer'`, `'loeschen'`).
|
||||||
|
*
|
||||||
|
* The second test below is the actual regression guard: it fails on any
|
||||||
|
* NEWLY introduced substitute spelling (`ae`/`oe`/`ue`/`ss` token not on
|
||||||
|
* the allowlist) without pinning today's wording — it pins vocabulary,
|
||||||
|
* not phrasing. Fixing a failure is a one-line addition to either
|
||||||
|
* `UMLAUT_REPLACEMENTS` (if it is a substitute spelling) or
|
||||||
|
* `UMLAUT_ALLOWLIST` (if it is correct German), as instructed by the
|
||||||
|
* assertion message itself.
|
||||||
|
*/
|
||||||
|
|
||||||
|
const UMLAUT_TOKEN_RE = /[A-Za-zÄÖÜäöüß]+/g;
|
||||||
|
const SUSPECT_RE = /(ae|oe|ue|ss)/;
|
||||||
|
|
||||||
|
/** Recursively flattens a nested message object into dot-joined leaf key paths. */
|
||||||
|
function flattenKeys(obj: unknown, prefix = ''): string[] {
|
||||||
|
if (obj === null || typeof obj !== 'object') {
|
||||||
|
return [prefix];
|
||||||
|
}
|
||||||
|
return Object.entries(obj as Record<string, unknown>).flatMap(([key, value]) =>
|
||||||
|
flattenKeys(value, prefix ? `${prefix}.${key}` : key),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Walks every leaf string value in `obj`, skipping values that contain
|
||||||
|
* `@` (e.g. the `testToPlaceholder` example email address), and invokes
|
||||||
|
* `visit` with each whole-word token found plus its dot-joined key path.
|
||||||
|
*/
|
||||||
|
function walkTokens(obj: unknown, visit: (token: string, path: string) => void, prefix = ''): void {
|
||||||
|
if (obj === null || typeof obj !== 'object') {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
for (const [key, value] of Object.entries(obj as Record<string, unknown>)) {
|
||||||
|
const path = prefix ? `${prefix}.${key}` : key;
|
||||||
|
if (typeof value === 'string') {
|
||||||
|
if (value.includes('@')) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
for (const token of value.match(UMLAUT_TOKEN_RE) ?? []) {
|
||||||
|
visit(token, path);
|
||||||
|
}
|
||||||
|
} else if (value && typeof value === 'object') {
|
||||||
|
walkTokens(value, visit, path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
describe('de.json umlaut regression guard', () => {
|
||||||
|
it('contains no token from UMLAUT_REPLACEMENTS (known substitute spelling)', () => {
|
||||||
|
const hits: string[] = [];
|
||||||
|
walkTokens(de, (token, path) => {
|
||||||
|
if (Object.prototype.hasOwnProperty.call(UMLAUT_REPLACEMENTS, token)) {
|
||||||
|
hits.push(`${path}: "${token}" should be "${UMLAUT_REPLACEMENTS[token]}"`);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
expect(hits, `Substitute spellings found in de.json:\n${hits.join('\n')}`).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('flags any new ae/oe/ue/ss token not on UMLAUT_ALLOWLIST', () => {
|
||||||
|
const allowlist = new Set(UMLAUT_ALLOWLIST);
|
||||||
|
const hits: string[] = [];
|
||||||
|
walkTokens(de, (token, path) => {
|
||||||
|
if (SUSPECT_RE.test(token) && !allowlist.has(token)) {
|
||||||
|
hits.push(
|
||||||
|
`${path}: "${token}" is a new word not on UMLAUT_ALLOWLIST. ` +
|
||||||
|
`Add it to UMLAUT_REPLACEMENTS in umlaut-dictionary.ts if it is a substitute ` +
|
||||||
|
`spelling, or to UMLAUT_ALLOWLIST if it is already correct German.`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
expect(hits, `New ae/oe/ue/ss words found in de.json:\n${hits.join('\n')}`).toEqual([]);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('has an identical (recursively-flattened) key set in de and en', () => {
|
||||||
|
const deKeys = flattenKeys(de).sort();
|
||||||
|
const enKeys = flattenKeys(en).sort();
|
||||||
|
|
||||||
|
const missingInEn = deKeys.filter((key) => !enKeys.includes(key));
|
||||||
|
const missingInDe = enKeys.filter((key) => !deKeys.includes(key));
|
||||||
|
|
||||||
|
expect(missingInEn, `Keys present in de.json but missing in en.json: ${missingInEn.join(', ')}`).toEqual([]);
|
||||||
|
expect(missingInDe, `Keys present in en.json but missing in de.json: ${missingInDe.join(', ')}`).toEqual([]);
|
||||||
|
expect(deKeys).toEqual(enKeys);
|
||||||
|
});
|
||||||
|
});
|
||||||
Reference in New Issue
Block a user