import { Injectable, Logger } from '@nestjs/common'; import { PDFParse } from 'pdf-parse'; import type { DkvVehicleBlock, DkvTransaction } from './dkv.types.js'; /** * DkvParserService * * Extracts structured vehicle/transaction data from DKV E-Rechnung PDF buffers. * * Security notes: * - Parser operates only on buffers fetched by the inbox provider (T-07-01) * - Calls destroy() to free PDF parser memory after extraction (T-07-01) * - Generic error messages only on parse failure — no PDF content in logs (T-07-02) * * Wave 0 validation: the parsing logic here was empirically derived by running * dkv-parser.validate.ts against user-files/invoice.pdf (April 2026, 27 vehicles). * Result: 27 vehicle blocks, 66 transactions total. * * Two PDF extraction formats are handled: * 1. Single-transaction (tab-separated): one row per transaction in the PDF * 2. Multi-transaction (columnar): all values per column stacked vertically in the extracted text */ @Injectable() export class DkvParserService { private readonly logger = new Logger(DkvParserService.name); /** * Parse a DKV E-Rechnung PDF buffer into structured vehicle blocks. * * @param buffer - Raw PDF bytes (from email attachment) * @returns Array of vehicle blocks with transactions * @throws Error with generic message if no vehicle blocks are found */ async parsePdf(buffer: Buffer): Promise<{ vehicles: DkvVehicleBlock[]; rechnungsnummer: string | null; rechnungsdatum: string | null; }> { let text: string; try { text = await this.extractText(buffer); } catch (err) { this.logger.error('DKV PDF text extraction failed'); throw new Error('DKV invoice parsing failed — text extraction error'); } const vehicles = this.parseDkvText(text); if (vehicles.length === 0) { this.logger.warn('DKV PDF parsed but yielded zero vehicle blocks'); throw new Error('DKV invoice parsing failed — no vehicle data found'); } const totalTx = vehicles.reduce((s, v) => s + v.transactions.length, 0); this.logger.log( `DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`, ); return { vehicles, rechnungsnummer: this.extractRechnungsnummer(text), rechnungsdatum: this.extractRechnungsdatum(text), }; } /** Extract first DKV invoice number (DD/DDDDDDDDD/DDD) from PDF text. */ private extractRechnungsnummer(text: string): string | null { const m = text.match(/\b(\d{2}\/\d{9}\/\d{3})\b/); return m ? m[1] : null; } /** Extract invoice date ("DD.MM.YYYY") from PDF text. * DKV PDFs always print the date on the line immediately after the invoice * number (DD/DDDDDDDDD/DDD), so we anchor on the number → date pairing. * The old "Rechnungsdatum[:\s]*date" approach picked up the payment-due date * ("10 Tage nach Rechnungsdatum. Die Abbuchung erfolgt am: 10.04.2026"). */ private extractRechnungsdatum(text: string): string | null { const m = text.match(/\d{2}\/\d{9}\/\d{3}\s+(\d{2}\.\d{2}\.\d{4})/); return m ? m[1] : null; } // ─── Private: PDF text extraction ────────────────────────────────────────── private async extractText(buffer: Buffer): Promise { // pdf-parse v2 class-based API — do NOT use v1 pdfParse(buffer) function call const parser = new PDFParse({ data: buffer }); const result = await parser.getText(); await parser.destroy(); // always free memory (T-07-01) return result.text; } // ─── Private: Vehicle block parser ───────────────────────────────────────── private parseDkvText(text: string): DkvVehicleBlock[] { const vehicles: DkvVehicleBlock[] = []; // Anchor on VEHICLE: marker — each block extends until next VEHICLE: or end // Kennzeichen format: "GP-JL 728E", "GP ML 720", etc. const vehicleBlockPattern = /VEHICLE:\s+([A-Z0-9 ._\-]+?)\s+CARD NO\.:\s+(\S+)([\s\S]*?)(?=VEHICLE:|$)/g; let match: RegExpExecArray | null; while ((match = vehicleBlockPattern.exec(text)) !== null) { const kennzeichen = match[1].trim(); const cardNumber = match[2].trim(); const blockText = match[3]; const transactions = this.parseTransactionRows(blockText); vehicles.push({ kennzeichen, cardNumber, transactions }); } return vehicles; } // ─── Private: Transaction row dispatcher ─────────────────────────────────── private parseTransactionRows(blockText: string): DkvTransaction[] { const lines = blockText .split('\n') .map(l => l.trim()) .filter(l => l && !l.startsWith('»')); if (lines.length === 0) return []; const datePattern = /^\d{2}\.\d{2}\.\d{4}/; // If any date-starting line also contains a tab → single-tx tab format const hasTabbedDateLine = lines.some(l => datePattern.test(l) && l.includes('\t')); if (hasTabbedDateLine) { return lines .filter(l => datePattern.test(l) && l.includes('\t')) .map(line => this.parseSingleTxLine(line)) .filter((tx): tx is DkvTransaction => tx !== null); } // Multi-transaction columnar format return this.parseMultiTxColumnar(lines); } // ─── Private: Single-transaction tab-separated line ──────────────────────── // // DKV produces two distinct single-tx tab formats: // // FUEL format: // [0] Lieferdatum DD.MM.YYYY // [1] Station Name e.g. "ESSO" // [2] Ort e.g. "DONZDORF" // [3] Servicest.-Nr e.g. "0071132" // [4] Transaktions-Nr e.g. "189945" ← purely numeric // [5] "KM PRODUKT" e.g. "68040 DIESEL" (km and product merged) // or just PRODUKT when driver did not enter km // [6] "CODE UNIT" e.g. "0009 LTR" // [7] Menge e.g. "45,910" // [10] Bezugswert brutto // [11] Bezugswert netto // // EV CHARGING format — detected by "DDDD KWH" unit field: // kwhIdx=5 (standard EV, station and ort separate): // [1] Station Name, [2] Ort, [3] Charge Point ID, // [4] Produkt, [5] "0964 KWH", [6] Menge, [9] netto, [12] brutto // kwhIdx=4 (compact EV, station+ort merged into [1]): // [1] "Name ORT" merged, [2] Charge Point ID, // [3] Produkt, [4] "0963 KWH", [5] Menge, [8] netto, [11] brutto // // SERVICE ROWS (e.g. "DKV Analytics Premiu") appear inside a VEHICLE: block but // are not vehicle transactions — detected by non-numeric fields[4] → skip. private parseSingleTxLine(line: string): DkvTransaction | null { const fields = line.split('\t').map(f => f.trim()); if (fields.length < 8) return null; const datePattern = /^\d{2}\.\d{2}\.\d{4}$/; if (!datePattern.test(fields[0])) return null; // Detect EV charging: look for a unit field matching "DDDD KWH" or "DDDD MIN" const kwhIdx = fields.findIndex(f => /^\d{4}\s+(KWH|MIN)$/i.test(f)); if (kwhIdx !== -1) { // kwhIdx=5: ort is separate in fields[2]; kwhIdx=4: station+ort merged in fields[1] const ort = kwhIdx >= 5 ? (fields[2] ?? '') : (fields[1] ?? ''); const mengeIdx = kwhIdx + 1; const nettoIdx = kwhIdx + 3; const bruttoIdx = kwhIdx + 6; return { lieferdatum: fields[0], ort, kilometerstand: 0, produkt: fields[kwhIdx - 1] ?? '', menge: parseDE(fields[mengeIdx] || '0'), einheit: fields[kwhIdx]?.split(/\s+/)[1] ?? 'KWH', netto: parseDE(fields[nettoIdx] || '0'), brutto: parseDE(fields[bruttoIdx] || '0'), }; } // Standard fuel rows: fields[4] must be a numeric transaction number. // Non-numeric field[4] = service charge row (e.g. DKV Analytics fee) → skip. if (!/^\d+$/.test(fields[4] ?? '')) return null; // fields[5] = "KM PRODUKT" merged, or just PRODUKT when driver skipped km entry const kmAndProdukt = fields[5].split(/\s+/); const km = kmAndProdukt[0] ?? '0'; const produkt = kmAndProdukt.slice(1).join(' '); // fields[6] = "UNITCODE EINHEIT" — Einheit is everything after the first token const unitParts = fields[6].split(/\s+/); const einheit = unitParts.slice(1).join(' ') || unitParts[0]; return { lieferdatum: fields[0], ort: fields[2] ?? '', kilometerstand: parseDE(km), produkt, menge: parseDE(fields[7] || '0'), einheit, netto: parseDE(fields[11] || '0'), brutto: parseDE(fields[10] || '0'), }; } // ─── Private: Multi-transaction columnar format ───────────────────────────── // // When a vehicle has multiple transactions, pdf-parse extracts the PDF table // in columnar order (all dates, then all names, then all orts, etc.). // // Column groups (each group has n items, where n = transaction count): // group 0: dates // group 1: station names // group 2: orts // group 3: station numbers // group 4: transaction numbers // group 5: km values // group 6: products // group 7: unit codes (e.g. "0036") // group 8: units (e.g. "LTR") // group 9: quantities (Menge) // group 10: unit price brutto // group 11: unit price netto // group 12: total brutto (Bezugswert) // group 13: total netto (Bezugswert) // ... (discounts, taxes, final totals — not extracted) private parseMultiTxColumnar(lines: string[]): DkvTransaction[] { const datePattern = /^\d{2}\.\d{2}\.\d{4}$/; // Count consecutive date lines at the start to determine n let n = 0; for (const line of lines) { if (datePattern.test(line)) n++; else break; } if (n === 0) return []; const g = (groupIdx: number): string[] => lines.slice(groupIdx * n, (groupIdx + 1) * n); const dates = g(0); const orts = g(2); const kms = g(5); const products = g(6); const units = g(8); const quantities = g(9); const totals_brutto = g(12); const totals_netto = g(13); return dates.map((date, i) => ({ lieferdatum: date, ort: orts[i] ?? '', // German number format: "19.234,56" → 19234.56 (strip dots, replace comma). // Use || '0' (not ??) so empty strings also fall back to '0' (avoids NaN). kilometerstand: parseDE(kms[i] || '0'), produkt: products[i] ?? '', menge: parseDE(quantities[i] || '0'), einheit: units[i] ?? '', netto: parseDE(totals_netto[i] || '0'), brutto: parseDE(totals_brutto[i] || '0'), })); } } // ─── German number parser (module-level — used in private methods above) ────── function parseDE(raw: string): number { return parseFloat(raw.replace(/\./g, '').replace(',', '.')); }