Files
tessera-ctl/apps/web/src/messages/umlaut-guard.spec.ts
T
schalli 85d2d772b6 fix(i18n): drei uebersehene Umlaute und die Luecke, durch die sie schluepften
Die Gegenprobe im Browser hat drei Stellen gefunden, die der erste Durchgang nicht
erwischt hat — darunter zwei gut sichtbare Schaltflaechen:

  Aenderungen speichern  ->  Änderungen speichern
  Oeffnen                ->  Öffnen
  Eine Aenderung ...     ->  Eine Änderung ...

Die Ursache ist dieselbe fuer alle drei und steckte im Waechter selbst: sein
Verdachtsmuster /(ae|oe|ue|ss)/ war case-sensitiv. "Aenderungen" beginnt mit "Ae",
nicht mit "ae", und ist deshalb durchgerutscht — der Waechter konnte gar nicht
anschlagen. Muster jetzt case-insensitiv; damit erfasst es auch die
grossgeschriebenen Formen.

Durch die schaerfere Pruefung melden sich neu die Abkuerzungen RSS, RSSGenerator und
SSL. Sie tragen ein doppeltes S ohne Umlaut-Bezug und stehen jetzt auf der
Positivliste.

Ausserdem zwei Meldungen des LDAP-Abgleichs korrigiert, die dem Administrator in der
Oberflaeche angezeigt werden (result.errors landet in der Fehlerliste der
LDAP-Seite): "ungueltiger ldapObjectGuid-Wert" und "Base-DN-Konfiguration pruefen.
Nicht geloescht." Drei Tests pinnen diese Texte bewusst und wurden mitgezogen.

Bewusst NICHT angefasst: die Warnung in crypto.service.ts. Sie geht ueber
logger.warn ins Protokoll und nicht an einen Nutzer.

Unabhaengig gegengeprueft: von allen Tokens in de.json, die ae/oe/ue tragen, ist
keines mehr eine Ersatzschreibung. 642 API-Tests und 225 Web-Tests gruen.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01K5jtbGzC5Sf9npJ3JCjKhq
2026-09-07 13:16:57 +02:00

103 lines
4.2 KiB
TypeScript

import { describe, expect, it } from 'vitest';
import de from './de.json';
import en from './en.json';
import { UMLAUT_ALLOWLIST, UMLAUT_REPLACEMENTS } from './umlaut-dictionary';
/**
* umlaut-guard — regression guard for the de.json umlaut correction.
*
* This spec reads exclusively the parsed JSON content of `de.json` /
* `en.json` and never greps repo-wide over source files. Grepping
* source files would trigger on `umlaut-dictionary.ts` itself, since it
* necessarily contains the wrong substitute spellings as its map keys
* (e.g. `'fuer'`, `'loeschen'`).
*
* The second test below is the actual regression guard: it fails on any
* NEWLY introduced substitute spelling (`ae`/`oe`/`ue`/`ss` token not on
* the allowlist) without pinning today's wording — it pins vocabulary,
* not phrasing. Fixing a failure is a one-line addition to either
* `UMLAUT_REPLACEMENTS` (if it is a substitute spelling) or
* `UMLAUT_ALLOWLIST` (if it is correct German), as instructed by the
* assertion message itself.
*/
const UMLAUT_TOKEN_RE = /[A-Za-zÄÖÜäöüß]+/g;
// Case-insensitive, sonst rutscht jede grossgeschriebene Ersatzschreibung durch:
// "Aenderungen" beginnt mit "Ae", nicht mit "ae". Genau daran sind beim ersten
// Durchgang "Aenderung", "Aenderungen" und "Oeffnen" vorbeigekommen — darunter
// zwei gut sichtbare Schaltflaechen.
const SUSPECT_RE = /(ae|oe|ue|ss)/i;
/** Recursively flattens a nested message object into dot-joined leaf key paths. */
function flattenKeys(obj: unknown, prefix = ''): string[] {
if (obj === null || typeof obj !== 'object') {
return [prefix];
}
return Object.entries(obj as Record<string, unknown>).flatMap(([key, value]) =>
flattenKeys(value, prefix ? `${prefix}.${key}` : key),
);
}
/**
* Walks every leaf string value in `obj`, skipping values that contain
* `@` (e.g. the `testToPlaceholder` example email address), and invokes
* `visit` with each whole-word token found plus its dot-joined key path.
*/
function walkTokens(obj: unknown, visit: (token: string, path: string) => void, prefix = ''): void {
if (obj === null || typeof obj !== 'object') {
return;
}
for (const [key, value] of Object.entries(obj as Record<string, unknown>)) {
const path = prefix ? `${prefix}.${key}` : key;
if (typeof value === 'string') {
if (value.includes('@')) {
continue;
}
for (const token of value.match(UMLAUT_TOKEN_RE) ?? []) {
visit(token, path);
}
} else if (value && typeof value === 'object') {
walkTokens(value, visit, path);
}
}
}
describe('de.json umlaut regression guard', () => {
it('contains no token from UMLAUT_REPLACEMENTS (known substitute spelling)', () => {
const hits: string[] = [];
walkTokens(de, (token, path) => {
if (Object.prototype.hasOwnProperty.call(UMLAUT_REPLACEMENTS, token)) {
hits.push(`${path}: "${token}" should be "${UMLAUT_REPLACEMENTS[token]}"`);
}
});
expect(hits, `Substitute spellings found in de.json:\n${hits.join('\n')}`).toEqual([]);
});
it('flags any new ae/oe/ue/ss token not on UMLAUT_ALLOWLIST', () => {
const allowlist = new Set(UMLAUT_ALLOWLIST);
const hits: string[] = [];
walkTokens(de, (token, path) => {
if (SUSPECT_RE.test(token) && !allowlist.has(token)) {
hits.push(
`${path}: "${token}" is a new word not on UMLAUT_ALLOWLIST. ` +
`Add it to UMLAUT_REPLACEMENTS in umlaut-dictionary.ts if it is a substitute ` +
`spelling, or to UMLAUT_ALLOWLIST if it is already correct German.`,
);
}
});
expect(hits, `New ae/oe/ue/ss words found in de.json:\n${hits.join('\n')}`).toEqual([]);
});
it('has an identical (recursively-flattened) key set in de and en', () => {
const deKeys = flattenKeys(de).sort();
const enKeys = flattenKeys(en).sort();
const missingInEn = deKeys.filter((key) => !enKeys.includes(key));
const missingInDe = enKeys.filter((key) => !deKeys.includes(key));
expect(missingInEn, `Keys present in de.json but missing in en.json: ${missingInEn.join(', ')}`).toEqual([]);
expect(missingInDe, `Keys present in en.json but missing in de.json: ${missingInDe.join(', ')}`).toEqual([]);
expect(deKeys).toEqual(enKeys);
});
});