a44e40f101
Date fix: previous regex matched payment-due date ("10 Tage nach Rechnungsdatum...
10.04.2026") instead of actual Rechnungsdatum. New approach anchors on the
invoice number line (DD/DDDDDDDDD/DDD) and takes the date on the next line,
which is always the actual Rechnungsdatum in DKV PDFs.
Exchange dedup: FindItem now filters IsRead=false (combined with sender filter
via <t:And>), so already-processed emails are skipped automatically.
After downloading attachments, UpdateItem marks the message as read
(using ItemId + ChangeKey from GetItem response), mirroring IMAP \Seen behavior.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
285 lines
11 KiB
TypeScript
285 lines
11 KiB
TypeScript
import { Injectable, Logger } from '@nestjs/common';
|
|
import { PDFParse } from 'pdf-parse';
|
|
import type { DkvVehicleBlock, DkvTransaction } from './dkv.types.js';
|
|
|
|
/**
|
|
* DkvParserService
|
|
*
|
|
* Extracts structured vehicle/transaction data from DKV E-Rechnung PDF buffers.
|
|
*
|
|
* Security notes:
|
|
* - Parser operates only on buffers fetched by the inbox provider (T-07-01)
|
|
* - Calls destroy() to free PDF parser memory after extraction (T-07-01)
|
|
* - Generic error messages only on parse failure — no PDF content in logs (T-07-02)
|
|
*
|
|
* Wave 0 validation: the parsing logic here was empirically derived by running
|
|
* dkv-parser.validate.ts against user-files/invoice.pdf (April 2026, 27 vehicles).
|
|
* Result: 27 vehicle blocks, 66 transactions total.
|
|
*
|
|
* Two PDF extraction formats are handled:
|
|
* 1. Single-transaction (tab-separated): one row per transaction in the PDF
|
|
* 2. Multi-transaction (columnar): all values per column stacked vertically in the extracted text
|
|
*/
|
|
@Injectable()
|
|
export class DkvParserService {
|
|
private readonly logger = new Logger(DkvParserService.name);
|
|
|
|
/**
|
|
* Parse a DKV E-Rechnung PDF buffer into structured vehicle blocks.
|
|
*
|
|
* @param buffer - Raw PDF bytes (from email attachment)
|
|
* @returns Array of vehicle blocks with transactions
|
|
* @throws Error with generic message if no vehicle blocks are found
|
|
*/
|
|
async parsePdf(buffer: Buffer): Promise<{
|
|
vehicles: DkvVehicleBlock[];
|
|
rechnungsnummer: string | null;
|
|
rechnungsdatum: string | null;
|
|
}> {
|
|
let text: string;
|
|
try {
|
|
text = await this.extractText(buffer);
|
|
} catch (err) {
|
|
this.logger.error('DKV PDF text extraction failed');
|
|
throw new Error('DKV invoice parsing failed — text extraction error');
|
|
}
|
|
|
|
const vehicles = this.parseDkvText(text);
|
|
|
|
if (vehicles.length === 0) {
|
|
this.logger.warn('DKV PDF parsed but yielded zero vehicle blocks');
|
|
throw new Error('DKV invoice parsing failed — no vehicle data found');
|
|
}
|
|
|
|
const totalTx = vehicles.reduce((s, v) => s + v.transactions.length, 0);
|
|
this.logger.log(
|
|
`DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`,
|
|
);
|
|
|
|
return {
|
|
vehicles,
|
|
rechnungsnummer: this.extractRechnungsnummer(text),
|
|
rechnungsdatum: this.extractRechnungsdatum(text),
|
|
};
|
|
}
|
|
|
|
/** Extract first DKV invoice number (DD/DDDDDDDDD/DDD) from PDF text. */
|
|
private extractRechnungsnummer(text: string): string | null {
|
|
const m = text.match(/\b(\d{2}\/\d{9}\/\d{3})\b/);
|
|
return m ? m[1] : null;
|
|
}
|
|
|
|
/** Extract invoice date ("DD.MM.YYYY") from PDF text.
|
|
* DKV PDFs always print the date on the line immediately after the invoice
|
|
* number (DD/DDDDDDDDD/DDD), so we anchor on the number → date pairing.
|
|
* The old "Rechnungsdatum[:\s]*date" approach picked up the payment-due date
|
|
* ("10 Tage nach Rechnungsdatum. Die Abbuchung erfolgt am: 10.04.2026").
|
|
*/
|
|
private extractRechnungsdatum(text: string): string | null {
|
|
const m = text.match(/\d{2}\/\d{9}\/\d{3}\s+(\d{2}\.\d{2}\.\d{4})/);
|
|
return m ? m[1] : null;
|
|
}
|
|
|
|
// ─── Private: PDF text extraction ──────────────────────────────────────────
|
|
|
|
private async extractText(buffer: Buffer): Promise<string> {
|
|
// pdf-parse v2 class-based API — do NOT use v1 pdfParse(buffer) function call
|
|
const parser = new PDFParse({ data: buffer });
|
|
const result = await parser.getText();
|
|
await parser.destroy(); // always free memory (T-07-01)
|
|
return result.text;
|
|
}
|
|
|
|
// ─── Private: Vehicle block parser ─────────────────────────────────────────
|
|
|
|
private parseDkvText(text: string): DkvVehicleBlock[] {
|
|
const vehicles: DkvVehicleBlock[] = [];
|
|
|
|
// Anchor on VEHICLE: marker — each block extends until next VEHICLE: or end
|
|
// Kennzeichen format: "GP-JL 728E", "GP ML 720", etc.
|
|
const vehicleBlockPattern =
|
|
/VEHICLE:\s+([A-Z0-9 ._\-]+?)\s+CARD NO\.:\s+(\S+)([\s\S]*?)(?=VEHICLE:|$)/g;
|
|
|
|
let match: RegExpExecArray | null;
|
|
while ((match = vehicleBlockPattern.exec(text)) !== null) {
|
|
const kennzeichen = match[1].trim();
|
|
const cardNumber = match[2].trim();
|
|
const blockText = match[3];
|
|
const transactions = this.parseTransactionRows(blockText);
|
|
vehicles.push({ kennzeichen, cardNumber, transactions });
|
|
}
|
|
|
|
return vehicles;
|
|
}
|
|
|
|
// ─── Private: Transaction row dispatcher ───────────────────────────────────
|
|
|
|
private parseTransactionRows(blockText: string): DkvTransaction[] {
|
|
const lines = blockText
|
|
.split('\n')
|
|
.map(l => l.trim())
|
|
.filter(l => l && !l.startsWith('»'));
|
|
|
|
if (lines.length === 0) return [];
|
|
|
|
const datePattern = /^\d{2}\.\d{2}\.\d{4}/;
|
|
|
|
// If any date-starting line also contains a tab → single-tx tab format
|
|
const hasTabbedDateLine = lines.some(l => datePattern.test(l) && l.includes('\t'));
|
|
|
|
if (hasTabbedDateLine) {
|
|
return lines
|
|
.filter(l => datePattern.test(l) && l.includes('\t'))
|
|
.map(line => this.parseSingleTxLine(line))
|
|
.filter((tx): tx is DkvTransaction => tx !== null);
|
|
}
|
|
|
|
// Multi-transaction columnar format
|
|
return this.parseMultiTxColumnar(lines);
|
|
}
|
|
|
|
// ─── Private: Single-transaction tab-separated line ────────────────────────
|
|
//
|
|
// DKV produces two distinct single-tx tab formats:
|
|
//
|
|
// FUEL format:
|
|
// [0] Lieferdatum DD.MM.YYYY
|
|
// [1] Station Name e.g. "ESSO"
|
|
// [2] Ort e.g. "DONZDORF"
|
|
// [3] Servicest.-Nr e.g. "0071132"
|
|
// [4] Transaktions-Nr e.g. "189945" ← purely numeric
|
|
// [5] "KM PRODUKT" e.g. "68040 DIESEL" (km and product merged)
|
|
// or just PRODUKT when driver did not enter km
|
|
// [6] "CODE UNIT" e.g. "0009 LTR"
|
|
// [7] Menge e.g. "45,910"
|
|
// [10] Bezugswert brutto
|
|
// [11] Bezugswert netto
|
|
//
|
|
// EV CHARGING format — detected by "DDDD KWH" unit field:
|
|
// kwhIdx=5 (standard EV, station and ort separate):
|
|
// [1] Station Name, [2] Ort, [3] Charge Point ID,
|
|
// [4] Produkt, [5] "0964 KWH", [6] Menge, [9] netto, [12] brutto
|
|
// kwhIdx=4 (compact EV, station+ort merged into [1]):
|
|
// [1] "Name ORT" merged, [2] Charge Point ID,
|
|
// [3] Produkt, [4] "0963 KWH", [5] Menge, [8] netto, [11] brutto
|
|
//
|
|
// SERVICE ROWS (e.g. "DKV Analytics Premiu") appear inside a VEHICLE: block but
|
|
// are not vehicle transactions — detected by non-numeric fields[4] → skip.
|
|
|
|
private parseSingleTxLine(line: string): DkvTransaction | null {
|
|
const fields = line.split('\t').map(f => f.trim());
|
|
if (fields.length < 8) return null;
|
|
|
|
const datePattern = /^\d{2}\.\d{2}\.\d{4}$/;
|
|
if (!datePattern.test(fields[0])) return null;
|
|
|
|
// Detect EV charging: look for a unit field matching "DDDD KWH" or "DDDD MIN"
|
|
const kwhIdx = fields.findIndex(f => /^\d{4}\s+(KWH|MIN)$/i.test(f));
|
|
if (kwhIdx !== -1) {
|
|
// kwhIdx=5: ort is separate in fields[2]; kwhIdx=4: station+ort merged in fields[1]
|
|
const ort = kwhIdx >= 5 ? (fields[2] ?? '') : (fields[1] ?? '');
|
|
const mengeIdx = kwhIdx + 1;
|
|
const nettoIdx = kwhIdx + 3;
|
|
const bruttoIdx = kwhIdx + 6;
|
|
return {
|
|
lieferdatum: fields[0],
|
|
ort,
|
|
kilometerstand: 0,
|
|
produkt: fields[kwhIdx - 1] ?? '',
|
|
menge: parseDE(fields[mengeIdx] || '0'),
|
|
einheit: fields[kwhIdx]?.split(/\s+/)[1] ?? 'KWH',
|
|
netto: parseDE(fields[nettoIdx] || '0'),
|
|
brutto: parseDE(fields[bruttoIdx] || '0'),
|
|
};
|
|
}
|
|
|
|
// Standard fuel rows: fields[4] must be a numeric transaction number.
|
|
// Non-numeric field[4] = service charge row (e.g. DKV Analytics fee) → skip.
|
|
if (!/^\d+$/.test(fields[4] ?? '')) return null;
|
|
|
|
// fields[5] = "KM PRODUKT" merged, or just PRODUKT when driver skipped km entry
|
|
const kmAndProdukt = fields[5].split(/\s+/);
|
|
const km = kmAndProdukt[0] ?? '0';
|
|
const produkt = kmAndProdukt.slice(1).join(' ');
|
|
|
|
// fields[6] = "UNITCODE EINHEIT" — Einheit is everything after the first token
|
|
const unitParts = fields[6].split(/\s+/);
|
|
const einheit = unitParts.slice(1).join(' ') || unitParts[0];
|
|
|
|
return {
|
|
lieferdatum: fields[0],
|
|
ort: fields[2] ?? '',
|
|
kilometerstand: parseDE(km),
|
|
produkt,
|
|
menge: parseDE(fields[7] || '0'),
|
|
einheit,
|
|
netto: parseDE(fields[11] || '0'),
|
|
brutto: parseDE(fields[10] || '0'),
|
|
};
|
|
}
|
|
|
|
// ─── Private: Multi-transaction columnar format ─────────────────────────────
|
|
//
|
|
// When a vehicle has multiple transactions, pdf-parse extracts the PDF table
|
|
// in columnar order (all dates, then all names, then all orts, etc.).
|
|
//
|
|
// Column groups (each group has n items, where n = transaction count):
|
|
// group 0: dates
|
|
// group 1: station names
|
|
// group 2: orts
|
|
// group 3: station numbers
|
|
// group 4: transaction numbers
|
|
// group 5: km values
|
|
// group 6: products
|
|
// group 7: unit codes (e.g. "0036")
|
|
// group 8: units (e.g. "LTR")
|
|
// group 9: quantities (Menge)
|
|
// group 10: unit price brutto
|
|
// group 11: unit price netto
|
|
// group 12: total brutto (Bezugswert)
|
|
// group 13: total netto (Bezugswert)
|
|
// ... (discounts, taxes, final totals — not extracted)
|
|
|
|
private parseMultiTxColumnar(lines: string[]): DkvTransaction[] {
|
|
const datePattern = /^\d{2}\.\d{2}\.\d{4}$/;
|
|
|
|
// Count consecutive date lines at the start to determine n
|
|
let n = 0;
|
|
for (const line of lines) {
|
|
if (datePattern.test(line)) n++;
|
|
else break;
|
|
}
|
|
if (n === 0) return [];
|
|
|
|
const g = (groupIdx: number): string[] =>
|
|
lines.slice(groupIdx * n, (groupIdx + 1) * n);
|
|
|
|
const dates = g(0);
|
|
const orts = g(2);
|
|
const kms = g(5);
|
|
const products = g(6);
|
|
const units = g(8);
|
|
const quantities = g(9);
|
|
const totals_brutto = g(12);
|
|
const totals_netto = g(13);
|
|
|
|
return dates.map((date, i) => ({
|
|
lieferdatum: date,
|
|
ort: orts[i] ?? '',
|
|
// German number format: "19.234,56" → 19234.56 (strip dots, replace comma).
|
|
// Use || '0' (not ??) so empty strings also fall back to '0' (avoids NaN).
|
|
kilometerstand: parseDE(kms[i] || '0'),
|
|
produkt: products[i] ?? '',
|
|
menge: parseDE(quantities[i] || '0'),
|
|
einheit: units[i] ?? '',
|
|
netto: parseDE(totals_netto[i] || '0'),
|
|
brutto: parseDE(totals_brutto[i] || '0'),
|
|
}));
|
|
}
|
|
}
|
|
|
|
// ─── German number parser (module-level — used in private methods above) ──────
|
|
function parseDE(raw: string): number {
|
|
return parseFloat(raw.replace(/\./g, '').replace(',', '.'));
|
|
}
|