fix(dkv): extract invoice number/date from PDF, rename export files to RG-DKV format
Tessera CI/CD / Lint & Type Check (push) Successful in 41s
Tessera CI/CD / Tests (push) Successful in 38s
Tessera CI/CD / Build & Publish Images (push) Successful in 23s

- Parser now extracts Rechnungsnummer (DD/DDDDDDDDD/DDD) and Rechnungsdatum
  from PDF text, so filename doesn't rely on email subject
- Export filename changed from DKV_YYYY-MM_... to RG-DKV-{nr}-{YYMMDD}.xlsx
  e.g. RG-DKV-26-650869002-002-260331.xlsx
- Subject fallback now also matches slash-separated invoice numbers (26/NNN/NNN)
- writeAndPrune simplified to accept baseName instead of separate fields
- Validation regex and prune prefix updated to match new RG-DKV- pattern

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-06-30 09:39:51 +02:00
parent 5c1aa03270
commit 66ffad149e
3 changed files with 64 additions and 24 deletions
+29 -2
View File
@@ -31,7 +31,11 @@ export class DkvParserService {
* @returns Array of vehicle blocks with transactions
* @throws Error with generic message if no vehicle blocks are found
*/
async parsePdf(buffer: Buffer): Promise<DkvVehicleBlock[]> {
async parsePdf(buffer: Buffer): Promise<{
vehicles: DkvVehicleBlock[];
rechnungsnummer: string | null;
rechnungsdatum: string | null;
}> {
let text: string;
try {
text = await this.extractText(buffer);
@@ -52,7 +56,30 @@ export class DkvParserService {
`DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`,
);
return vehicles;
return {
vehicles,
rechnungsnummer: this.extractRechnungsnummer(text),
rechnungsdatum: this.extractRechnungsdatum(text),
};
}
/** Extract first DKV invoice number (DD/DDDDDDDDD/DDD) from PDF text. */
private extractRechnungsnummer(text: string): string | null {
const m = text.match(/\b(\d{2}\/\d{9}\/\d{3})\b/);
return m ? m[1] : null;
}
/** Extract invoice date ("DD.MM.YYYY") appearing near "Rechnungsdatum" in PDF text. */
private extractRechnungsdatum(text: string): string | null {
// "Rechnungsdatum" is followed (within ~200 chars) by a date string
const m = text.match(/Rechnungsdatum[:\s\t]*(\d{2}\.\d{2}\.\d{4})/);
if (m) return m[1];
// Fallback: second line after "Rechnungsdatum:" marker
const idx = text.indexOf('Rechnungsdatum');
if (idx === -1) return null;
const after = text.slice(idx, idx + 200);
const dm = after.match(/(\d{2}\.\d{2}\.\d{4})/);
return dm ? dm[1] : null;
}
// ─── Private: PDF text extraction ──────────────────────────────────────────