Files
tessera-ctl/apps/api/src/dkv/dkv-parser.service.ts
T
schalli a44e40f101
Tessera CI/CD / Lint & Type Check (push) Successful in 44s
Tessera CI/CD / Tests (push) Successful in 44s
Tessera CI/CD / Build & Publish Images (push) Successful in 21s
fix(dkv): correct invoice date extraction; add Exchange IsRead filter + mark-as-read
Date fix: previous regex matched payment-due date ("10 Tage nach Rechnungsdatum...
10.04.2026") instead of actual Rechnungsdatum. New approach anchors on the
invoice number line (DD/DDDDDDDDD/DDD) and takes the date on the next line,
which is always the actual Rechnungsdatum in DKV PDFs.

Exchange dedup: FindItem now filters IsRead=false (combined with sender filter
via <t:And>), so already-processed emails are skipped automatically.
After downloading attachments, UpdateItem marks the message as read
(using ItemId + ChangeKey from GetItem response), mirroring IMAP \Seen behavior.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-30 09:47:17 +02:00

285 lines
11 KiB
TypeScript

import { Injectable, Logger } from '@nestjs/common';
import { PDFParse } from 'pdf-parse';
import type { DkvVehicleBlock, DkvTransaction } from './dkv.types.js';
/**
* DkvParserService
*
* Extracts structured vehicle/transaction data from DKV E-Rechnung PDF buffers.
*
* Security notes:
* - Parser operates only on buffers fetched by the inbox provider (T-07-01)
* - Calls destroy() to free PDF parser memory after extraction (T-07-01)
* - Generic error messages only on parse failure — no PDF content in logs (T-07-02)
*
* Wave 0 validation: the parsing logic here was empirically derived by running
* dkv-parser.validate.ts against user-files/invoice.pdf (April 2026, 27 vehicles).
* Result: 27 vehicle blocks, 66 transactions total.
*
* Two PDF extraction formats are handled:
* 1. Single-transaction (tab-separated): one row per transaction in the PDF
* 2. Multi-transaction (columnar): all values per column stacked vertically in the extracted text
*/
@Injectable()
export class DkvParserService {
private readonly logger = new Logger(DkvParserService.name);
/**
* Parse a DKV E-Rechnung PDF buffer into structured vehicle blocks.
*
* @param buffer - Raw PDF bytes (from email attachment)
* @returns Array of vehicle blocks with transactions
* @throws Error with generic message if no vehicle blocks are found
*/
async parsePdf(buffer: Buffer): Promise<{
vehicles: DkvVehicleBlock[];
rechnungsnummer: string | null;
rechnungsdatum: string | null;
}> {
let text: string;
try {
text = await this.extractText(buffer);
} catch (err) {
this.logger.error('DKV PDF text extraction failed');
throw new Error('DKV invoice parsing failed — text extraction error');
}
const vehicles = this.parseDkvText(text);
if (vehicles.length === 0) {
this.logger.warn('DKV PDF parsed but yielded zero vehicle blocks');
throw new Error('DKV invoice parsing failed — no vehicle data found');
}
const totalTx = vehicles.reduce((s, v) => s + v.transactions.length, 0);
this.logger.log(
`DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`,
);
return {
vehicles,
rechnungsnummer: this.extractRechnungsnummer(text),
rechnungsdatum: this.extractRechnungsdatum(text),
};
}
/** Extract first DKV invoice number (DD/DDDDDDDDD/DDD) from PDF text. */
private extractRechnungsnummer(text: string): string | null {
const m = text.match(/\b(\d{2}\/\d{9}\/\d{3})\b/);
return m ? m[1] : null;
}
/** Extract invoice date ("DD.MM.YYYY") from PDF text.
* DKV PDFs always print the date on the line immediately after the invoice
* number (DD/DDDDDDDDD/DDD), so we anchor on the number → date pairing.
* The old "Rechnungsdatum[:\s]*date" approach picked up the payment-due date
* ("10 Tage nach Rechnungsdatum. Die Abbuchung erfolgt am: 10.04.2026").
*/
private extractRechnungsdatum(text: string): string | null {
const m = text.match(/\d{2}\/\d{9}\/\d{3}\s+(\d{2}\.\d{2}\.\d{4})/);
return m ? m[1] : null;
}
// ─── Private: PDF text extraction ──────────────────────────────────────────
private async extractText(buffer: Buffer): Promise<string> {
// pdf-parse v2 class-based API — do NOT use v1 pdfParse(buffer) function call
const parser = new PDFParse({ data: buffer });
const result = await parser.getText();
await parser.destroy(); // always free memory (T-07-01)
return result.text;
}
// ─── Private: Vehicle block parser ─────────────────────────────────────────
private parseDkvText(text: string): DkvVehicleBlock[] {
const vehicles: DkvVehicleBlock[] = [];
// Anchor on VEHICLE: marker — each block extends until next VEHICLE: or end
// Kennzeichen format: "GP-JL 728E", "GP ML 720", etc.
const vehicleBlockPattern =
/VEHICLE:\s+([A-Z0-9 ._\-]+?)\s+CARD NO\.:\s+(\S+)([\s\S]*?)(?=VEHICLE:|$)/g;
let match: RegExpExecArray | null;
while ((match = vehicleBlockPattern.exec(text)) !== null) {
const kennzeichen = match[1].trim();
const cardNumber = match[2].trim();
const blockText = match[3];
const transactions = this.parseTransactionRows(blockText);
vehicles.push({ kennzeichen, cardNumber, transactions });
}
return vehicles;
}
// ─── Private: Transaction row dispatcher ───────────────────────────────────
private parseTransactionRows(blockText: string): DkvTransaction[] {
const lines = blockText
.split('\n')
.map(l => l.trim())
.filter(l => l && !l.startsWith('»'));
if (lines.length === 0) return [];
const datePattern = /^\d{2}\.\d{2}\.\d{4}/;
// If any date-starting line also contains a tab → single-tx tab format
const hasTabbedDateLine = lines.some(l => datePattern.test(l) && l.includes('\t'));
if (hasTabbedDateLine) {
return lines
.filter(l => datePattern.test(l) && l.includes('\t'))
.map(line => this.parseSingleTxLine(line))
.filter((tx): tx is DkvTransaction => tx !== null);
}
// Multi-transaction columnar format
return this.parseMultiTxColumnar(lines);
}
// ─── Private: Single-transaction tab-separated line ────────────────────────
//
// DKV produces two distinct single-tx tab formats:
//
// FUEL format:
// [0] Lieferdatum DD.MM.YYYY
// [1] Station Name e.g. "ESSO"
// [2] Ort e.g. "DONZDORF"
// [3] Servicest.-Nr e.g. "0071132"
// [4] Transaktions-Nr e.g. "189945" ← purely numeric
// [5] "KM PRODUKT" e.g. "68040 DIESEL" (km and product merged)
// or just PRODUKT when driver did not enter km
// [6] "CODE UNIT" e.g. "0009 LTR"
// [7] Menge e.g. "45,910"
// [10] Bezugswert brutto
// [11] Bezugswert netto
//
// EV CHARGING format — detected by "DDDD KWH" unit field:
// kwhIdx=5 (standard EV, station and ort separate):
// [1] Station Name, [2] Ort, [3] Charge Point ID,
// [4] Produkt, [5] "0964 KWH", [6] Menge, [9] netto, [12] brutto
// kwhIdx=4 (compact EV, station+ort merged into [1]):
// [1] "Name ORT" merged, [2] Charge Point ID,
// [3] Produkt, [4] "0963 KWH", [5] Menge, [8] netto, [11] brutto
//
// SERVICE ROWS (e.g. "DKV Analytics Premiu") appear inside a VEHICLE: block but
// are not vehicle transactions — detected by non-numeric fields[4] → skip.
private parseSingleTxLine(line: string): DkvTransaction | null {
const fields = line.split('\t').map(f => f.trim());
if (fields.length < 8) return null;
const datePattern = /^\d{2}\.\d{2}\.\d{4}$/;
if (!datePattern.test(fields[0])) return null;
// Detect EV charging: look for a unit field matching "DDDD KWH" or "DDDD MIN"
const kwhIdx = fields.findIndex(f => /^\d{4}\s+(KWH|MIN)$/i.test(f));
if (kwhIdx !== -1) {
// kwhIdx=5: ort is separate in fields[2]; kwhIdx=4: station+ort merged in fields[1]
const ort = kwhIdx >= 5 ? (fields[2] ?? '') : (fields[1] ?? '');
const mengeIdx = kwhIdx + 1;
const nettoIdx = kwhIdx + 3;
const bruttoIdx = kwhIdx + 6;
return {
lieferdatum: fields[0],
ort,
kilometerstand: 0,
produkt: fields[kwhIdx - 1] ?? '',
menge: parseDE(fields[mengeIdx] || '0'),
einheit: fields[kwhIdx]?.split(/\s+/)[1] ?? 'KWH',
netto: parseDE(fields[nettoIdx] || '0'),
brutto: parseDE(fields[bruttoIdx] || '0'),
};
}
// Standard fuel rows: fields[4] must be a numeric transaction number.
// Non-numeric field[4] = service charge row (e.g. DKV Analytics fee) → skip.
if (!/^\d+$/.test(fields[4] ?? '')) return null;
// fields[5] = "KM PRODUKT" merged, or just PRODUKT when driver skipped km entry
const kmAndProdukt = fields[5].split(/\s+/);
const km = kmAndProdukt[0] ?? '0';
const produkt = kmAndProdukt.slice(1).join(' ');
// fields[6] = "UNITCODE EINHEIT" — Einheit is everything after the first token
const unitParts = fields[6].split(/\s+/);
const einheit = unitParts.slice(1).join(' ') || unitParts[0];
return {
lieferdatum: fields[0],
ort: fields[2] ?? '',
kilometerstand: parseDE(km),
produkt,
menge: parseDE(fields[7] || '0'),
einheit,
netto: parseDE(fields[11] || '0'),
brutto: parseDE(fields[10] || '0'),
};
}
// ─── Private: Multi-transaction columnar format ─────────────────────────────
//
// When a vehicle has multiple transactions, pdf-parse extracts the PDF table
// in columnar order (all dates, then all names, then all orts, etc.).
//
// Column groups (each group has n items, where n = transaction count):
// group 0: dates
// group 1: station names
// group 2: orts
// group 3: station numbers
// group 4: transaction numbers
// group 5: km values
// group 6: products
// group 7: unit codes (e.g. "0036")
// group 8: units (e.g. "LTR")
// group 9: quantities (Menge)
// group 10: unit price brutto
// group 11: unit price netto
// group 12: total brutto (Bezugswert)
// group 13: total netto (Bezugswert)
// ... (discounts, taxes, final totals — not extracted)
private parseMultiTxColumnar(lines: string[]): DkvTransaction[] {
const datePattern = /^\d{2}\.\d{2}\.\d{4}$/;
// Count consecutive date lines at the start to determine n
let n = 0;
for (const line of lines) {
if (datePattern.test(line)) n++;
else break;
}
if (n === 0) return [];
const g = (groupIdx: number): string[] =>
lines.slice(groupIdx * n, (groupIdx + 1) * n);
const dates = g(0);
const orts = g(2);
const kms = g(5);
const products = g(6);
const units = g(8);
const quantities = g(9);
const totals_brutto = g(12);
const totals_netto = g(13);
return dates.map((date, i) => ({
lieferdatum: date,
ort: orts[i] ?? '',
// German number format: "19.234,56" → 19234.56 (strip dots, replace comma).
// Use || '0' (not ??) so empty strings also fall back to '0' (avoids NaN).
kilometerstand: parseDE(kms[i] || '0'),
produkt: products[i] ?? '',
menge: parseDE(quantities[i] || '0'),
einheit: units[i] ?? '',
netto: parseDE(totals_netto[i] || '0'),
brutto: parseDE(totals_brutto[i] || '0'),
}));
}
}
// ─── German number parser (module-level — used in private methods above) ──────
function parseDE(raw: string): number {
return parseFloat(raw.replace(/\./g, '').replace(',', '.'));
}