fix(dkv): extract invoice number/date from PDF, rename export files to RG-DKV format
Tessera CI/CD / Lint & Type Check (push) Successful in 41s
Tessera CI/CD / Tests (push) Successful in 38s
Tessera CI/CD / Build & Publish Images (push) Successful in 23s

- Parser now extracts Rechnungsnummer (DD/DDDDDDDDD/DDD) and Rechnungsdatum
  from PDF text, so filename doesn't rely on email subject
- Export filename changed from DKV_YYYY-MM_... to RG-DKV-{nr}-{YYMMDD}.xlsx
  e.g. RG-DKV-26-650869002-002-260331.xlsx
- Subject fallback now also matches slash-separated invoice numbers (26/NNN/NNN)
- writeAndPrune simplified to accept baseName instead of separate fields
- Validation regex and prune prefix updated to match new RG-DKV- pattern

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-06-30 09:39:51 +02:00
parent 5c1aa03270
commit 66ffad149e
3 changed files with 64 additions and 24 deletions
+3 -3
View File
@@ -13,7 +13,7 @@ const MAX_EXPORT_FILES = 10;
/**
* Glob pattern for DKV export files — used for prune selection.
*/
const DKV_FILE_PREFIX = 'DKV_';
const DKV_FILE_PREFIX = 'RG-DKV-';
const DKV_FILE_SUFFIX = '.xlsx';
/**
@@ -128,7 +128,7 @@ export class DkvExportService {
* @param invoiceMonth - Month string "YYYY-MM", e.g. "2026-04"
* @returns The filename that was written (relative to user-files/)
*/
writeAndPrune(buffer: Buffer, rechnungsnummer: string, invoiceMonth: string): string {
writeAndPrune(buffer: Buffer, baseName: string): string {
// Ensure the directory exists
if (!fs.existsSync(this.userFilesDir)) {
fs.mkdirSync(this.userFilesDir, { recursive: true });
@@ -136,7 +136,7 @@ export class DkvExportService {
}
// Build filename server-side — NEVER from request input (T-07-09 path-traversal mitigation)
const filename = `DKV_${invoiceMonth}_${rechnungsnummer}.xlsx`;
const filename = `${baseName}.xlsx`;
const filePath = path.join(this.userFilesDir, filename);
// Write the file
+29 -2
View File
@@ -31,7 +31,11 @@ export class DkvParserService {
* @returns Array of vehicle blocks with transactions
* @throws Error with generic message if no vehicle blocks are found
*/
async parsePdf(buffer: Buffer): Promise<DkvVehicleBlock[]> {
async parsePdf(buffer: Buffer): Promise<{
vehicles: DkvVehicleBlock[];
rechnungsnummer: string | null;
rechnungsdatum: string | null;
}> {
let text: string;
try {
text = await this.extractText(buffer);
@@ -52,7 +56,30 @@ export class DkvParserService {
`DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`,
);
return vehicles;
return {
vehicles,
rechnungsnummer: this.extractRechnungsnummer(text),
rechnungsdatum: this.extractRechnungsdatum(text),
};
}
/** Extract first DKV invoice number (DD/DDDDDDDDD/DDD) from PDF text. */
private extractRechnungsnummer(text: string): string | null {
const m = text.match(/\b(\d{2}\/\d{9}\/\d{3})\b/);
return m ? m[1] : null;
}
/** Extract invoice date ("DD.MM.YYYY") appearing near "Rechnungsdatum" in PDF text. */
private extractRechnungsdatum(text: string): string | null {
// "Rechnungsdatum" is followed (within ~200 chars) by a date string
const m = text.match(/Rechnungsdatum[:\s\t]*(\d{2}\.\d{2}\.\d{4})/);
if (m) return m[1];
// Fallback: second line after "Rechnungsdatum:" marker
const idx = text.indexOf('Rechnungsdatum');
if (idx === -1) return null;
const after = text.slice(idx, idx + 200);
const dm = after.match(/(\d{2}\.\d{2}\.\d{4})/);
return dm ? dm[1] : null;
}
// ─── Private: PDF text extraction ──────────────────────────────────────────
+32 -19
View File
@@ -342,12 +342,12 @@ export class DkvService {
vehicleFormatString: string,
): Promise<void> {
// D-10: Up to 3 parse retries
let vehicles: DkvVehicleBlock[] | null = null;
let parseResult: Awaited<ReturnType<typeof this.parser.parsePdf>> | null = null;
let parseError: string | null = null;
for (let attempt = 1; attempt <= 3; attempt++) {
try {
vehicles = await this.parser.parsePdf(pdfBuffer);
parseResult = await this.parser.parsePdf(pdfBuffer);
parseError = null;
break;
} catch (err) {
@@ -358,12 +358,12 @@ export class DkvService {
}
}
if (!vehicles) {
if (!parseResult) {
// Record parse failure in history (D-10)
await this.prisma.dkvInvoiceHistory.create({
data: {
tenantId,
rechnungsnummer: this._extractInvoiceNumber(email.subject, email.uid),
rechnungsnummer: this._buildRechnungsnummer(null, email.subject, email.uid),
anzahlFahrzeuge: 0,
anzahlTransaktionen: 0,
status: 'Fehler',
@@ -373,9 +373,13 @@ export class DkvService {
return;
}
// Extract invoice metadata for filename + history record
const rechnungsnummer = this._extractInvoiceNumber(email.subject, email.uid);
const invoiceMonth = this._extractInvoiceMonth(vehicles);
const { vehicles, rechnungsnummer: rgNr, rechnungsdatum: rgDat } = parseResult;
// Build filename: RG-DKV-{nr}-{date}.xlsx
const rechnungsnummer = this._buildRechnungsnummer(rgNr, email.subject, email.uid);
const datePart = rgDat ? _formatDateYYMMDD(rgDat) : this._extractInvoiceMonth(vehicles).replace('-', '');
const exportBaseName = `RG-DKV-${rechnungsnummer}-${datePart}`;
const anzahlTransaktionen = vehicles.reduce((s, v) => s + v.transactions.length, 0);
// Build export rows — look up drivers from vehicle master
@@ -383,7 +387,7 @@ export class DkvService {
// Generate xlsx buffer and write to user-files/ (DkvExportService)
const xlsxBuffer = this.exporter.buildExcelBuffer(exportRows);
const exportFilename = this.exporter.writeAndPrune(xlsxBuffer, rechnungsnummer, invoiceMonth);
const exportFilename = this.exporter.writeAndPrune(xlsxBuffer, exportBaseName);
// D-16: SMTP send with 3-retry exponential backoff
let smtpStatus: 'Verarbeitet' | 'Versand fehlgeschlagen' = 'Verarbeitet';
@@ -552,7 +556,7 @@ export class DkvService {
filename.includes('/') ||
filename.includes('\\') ||
filename.includes('..') ||
!/^DKV_[\w\-]+\.xlsx$/.test(filename)
!/^(RG-DKV-|DKV_)[\w\-]+\.xlsx$/.test(filename)
) {
throw new BadRequestException('Invalid export filename');
}
@@ -624,17 +628,19 @@ export class DkvService {
}
/**
* Extract a DKV invoice number from the email subject line.
* Pattern: DD-DDDDDDDDD-DDD (e.g. "26-651566449-001")
* Falls back to "email-{uid}" when no match is found.
* Build a safe rechnungsnummer string for use in filenames.
* Priority: PDF-extracted number → email subject → uid fallback.
* All slashes and non-safe chars are replaced so the value is path-safe.
*/
private _extractInvoiceNumber(subject: string, uid: number | string): string {
const match = subject?.match(/(\d{2}-\d{9}-\d{3})/);
if (match?.[1]) return match[1];
// Fallback when invoice number cannot be parsed from subject.
// Exchange UniqueIds are base64 and can contain '+', '/', '=' which would
// introduce path separators into the generated filename (WR-04).
// Sanitise to [a-zA-Z0-9-] before the value reaches the filesystem.
private _buildRechnungsnummer(
fromPdf: string | null,
subject: string,
uid: number | string,
): string {
if (fromPdf) return fromPdf.replace(/\//g, '-');
// DKV email subjects carry the invoice number with slashes: "26/650869002/002"
const m = subject?.match(/(\d{2}[\/\-]\d{9}[\/\-]\d{3})/);
if (m?.[1]) return m[1].replace(/\//g, '-');
return `email-${String(uid).replace(/[^a-zA-Z0-9\-]/g, '_')}`;
}
@@ -694,6 +700,13 @@ function _delay(ms: number): Promise<void> {
return new Promise((resolve) => setTimeout(resolve, ms));
}
/** Format "DD.MM.YYYY" → "YYMMDD" for compact filename date segment. */
function _formatDateYYMMDD(date: string): string {
const [d, m, y] = date.split('.');
if (!d || !m || !y) return date.replace(/\./g, '');
return `${y.slice(2)}${m}${d}`;
}
/**
* Normalize a Kennzeichen for fuzzy vehicle lookup.
* DKV PDF extraction may omit hyphens or use spaces as separators.