fix(dkv): extract invoice number/date from PDF, rename export files to RG-DKV format
- Parser now extracts Rechnungsnummer (DD/DDDDDDDDD/DDD) and Rechnungsdatum
from PDF text, so filename doesn't rely on email subject
- Export filename changed from DKV_YYYY-MM_... to RG-DKV-{nr}-{YYMMDD}.xlsx
e.g. RG-DKV-26-650869002-002-260331.xlsx
- Subject fallback now also matches slash-separated invoice numbers (26/NNN/NNN)
- writeAndPrune simplified to accept baseName instead of separate fields
- Validation regex and prune prefix updated to match new RG-DKV- pattern
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -13,7 +13,7 @@ const MAX_EXPORT_FILES = 10;
|
|||||||
/**
|
/**
|
||||||
* Glob pattern for DKV export files — used for prune selection.
|
* Glob pattern for DKV export files — used for prune selection.
|
||||||
*/
|
*/
|
||||||
const DKV_FILE_PREFIX = 'DKV_';
|
const DKV_FILE_PREFIX = 'RG-DKV-';
|
||||||
const DKV_FILE_SUFFIX = '.xlsx';
|
const DKV_FILE_SUFFIX = '.xlsx';
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -128,7 +128,7 @@ export class DkvExportService {
|
|||||||
* @param invoiceMonth - Month string "YYYY-MM", e.g. "2026-04"
|
* @param invoiceMonth - Month string "YYYY-MM", e.g. "2026-04"
|
||||||
* @returns The filename that was written (relative to user-files/)
|
* @returns The filename that was written (relative to user-files/)
|
||||||
*/
|
*/
|
||||||
writeAndPrune(buffer: Buffer, rechnungsnummer: string, invoiceMonth: string): string {
|
writeAndPrune(buffer: Buffer, baseName: string): string {
|
||||||
// Ensure the directory exists
|
// Ensure the directory exists
|
||||||
if (!fs.existsSync(this.userFilesDir)) {
|
if (!fs.existsSync(this.userFilesDir)) {
|
||||||
fs.mkdirSync(this.userFilesDir, { recursive: true });
|
fs.mkdirSync(this.userFilesDir, { recursive: true });
|
||||||
@@ -136,7 +136,7 @@ export class DkvExportService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Build filename server-side — NEVER from request input (T-07-09 path-traversal mitigation)
|
// Build filename server-side — NEVER from request input (T-07-09 path-traversal mitigation)
|
||||||
const filename = `DKV_${invoiceMonth}_${rechnungsnummer}.xlsx`;
|
const filename = `${baseName}.xlsx`;
|
||||||
const filePath = path.join(this.userFilesDir, filename);
|
const filePath = path.join(this.userFilesDir, filename);
|
||||||
|
|
||||||
// Write the file
|
// Write the file
|
||||||
|
|||||||
@@ -31,7 +31,11 @@ export class DkvParserService {
|
|||||||
* @returns Array of vehicle blocks with transactions
|
* @returns Array of vehicle blocks with transactions
|
||||||
* @throws Error with generic message if no vehicle blocks are found
|
* @throws Error with generic message if no vehicle blocks are found
|
||||||
*/
|
*/
|
||||||
async parsePdf(buffer: Buffer): Promise<DkvVehicleBlock[]> {
|
async parsePdf(buffer: Buffer): Promise<{
|
||||||
|
vehicles: DkvVehicleBlock[];
|
||||||
|
rechnungsnummer: string | null;
|
||||||
|
rechnungsdatum: string | null;
|
||||||
|
}> {
|
||||||
let text: string;
|
let text: string;
|
||||||
try {
|
try {
|
||||||
text = await this.extractText(buffer);
|
text = await this.extractText(buffer);
|
||||||
@@ -52,7 +56,30 @@ export class DkvParserService {
|
|||||||
`DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`,
|
`DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`,
|
||||||
);
|
);
|
||||||
|
|
||||||
return vehicles;
|
return {
|
||||||
|
vehicles,
|
||||||
|
rechnungsnummer: this.extractRechnungsnummer(text),
|
||||||
|
rechnungsdatum: this.extractRechnungsdatum(text),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Extract first DKV invoice number (DD/DDDDDDDDD/DDD) from PDF text. */
|
||||||
|
private extractRechnungsnummer(text: string): string | null {
|
||||||
|
const m = text.match(/\b(\d{2}\/\d{9}\/\d{3})\b/);
|
||||||
|
return m ? m[1] : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Extract invoice date ("DD.MM.YYYY") appearing near "Rechnungsdatum" in PDF text. */
|
||||||
|
private extractRechnungsdatum(text: string): string | null {
|
||||||
|
// "Rechnungsdatum" is followed (within ~200 chars) by a date string
|
||||||
|
const m = text.match(/Rechnungsdatum[:\s\t]*(\d{2}\.\d{2}\.\d{4})/);
|
||||||
|
if (m) return m[1];
|
||||||
|
// Fallback: second line after "Rechnungsdatum:" marker
|
||||||
|
const idx = text.indexOf('Rechnungsdatum');
|
||||||
|
if (idx === -1) return null;
|
||||||
|
const after = text.slice(idx, idx + 200);
|
||||||
|
const dm = after.match(/(\d{2}\.\d{2}\.\d{4})/);
|
||||||
|
return dm ? dm[1] : null;
|
||||||
}
|
}
|
||||||
|
|
||||||
// ─── Private: PDF text extraction ──────────────────────────────────────────
|
// ─── Private: PDF text extraction ──────────────────────────────────────────
|
||||||
|
|||||||
@@ -342,12 +342,12 @@ export class DkvService {
|
|||||||
vehicleFormatString: string,
|
vehicleFormatString: string,
|
||||||
): Promise<void> {
|
): Promise<void> {
|
||||||
// D-10: Up to 3 parse retries
|
// D-10: Up to 3 parse retries
|
||||||
let vehicles: DkvVehicleBlock[] | null = null;
|
let parseResult: Awaited<ReturnType<typeof this.parser.parsePdf>> | null = null;
|
||||||
let parseError: string | null = null;
|
let parseError: string | null = null;
|
||||||
|
|
||||||
for (let attempt = 1; attempt <= 3; attempt++) {
|
for (let attempt = 1; attempt <= 3; attempt++) {
|
||||||
try {
|
try {
|
||||||
vehicles = await this.parser.parsePdf(pdfBuffer);
|
parseResult = await this.parser.parsePdf(pdfBuffer);
|
||||||
parseError = null;
|
parseError = null;
|
||||||
break;
|
break;
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
@@ -358,12 +358,12 @@ export class DkvService {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!vehicles) {
|
if (!parseResult) {
|
||||||
// Record parse failure in history (D-10)
|
// Record parse failure in history (D-10)
|
||||||
await this.prisma.dkvInvoiceHistory.create({
|
await this.prisma.dkvInvoiceHistory.create({
|
||||||
data: {
|
data: {
|
||||||
tenantId,
|
tenantId,
|
||||||
rechnungsnummer: this._extractInvoiceNumber(email.subject, email.uid),
|
rechnungsnummer: this._buildRechnungsnummer(null, email.subject, email.uid),
|
||||||
anzahlFahrzeuge: 0,
|
anzahlFahrzeuge: 0,
|
||||||
anzahlTransaktionen: 0,
|
anzahlTransaktionen: 0,
|
||||||
status: 'Fehler',
|
status: 'Fehler',
|
||||||
@@ -373,9 +373,13 @@ export class DkvService {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Extract invoice metadata for filename + history record
|
const { vehicles, rechnungsnummer: rgNr, rechnungsdatum: rgDat } = parseResult;
|
||||||
const rechnungsnummer = this._extractInvoiceNumber(email.subject, email.uid);
|
|
||||||
const invoiceMonth = this._extractInvoiceMonth(vehicles);
|
// Build filename: RG-DKV-{nr}-{date}.xlsx
|
||||||
|
const rechnungsnummer = this._buildRechnungsnummer(rgNr, email.subject, email.uid);
|
||||||
|
const datePart = rgDat ? _formatDateYYMMDD(rgDat) : this._extractInvoiceMonth(vehicles).replace('-', '');
|
||||||
|
const exportBaseName = `RG-DKV-${rechnungsnummer}-${datePart}`;
|
||||||
|
|
||||||
const anzahlTransaktionen = vehicles.reduce((s, v) => s + v.transactions.length, 0);
|
const anzahlTransaktionen = vehicles.reduce((s, v) => s + v.transactions.length, 0);
|
||||||
|
|
||||||
// Build export rows — look up drivers from vehicle master
|
// Build export rows — look up drivers from vehicle master
|
||||||
@@ -383,7 +387,7 @@ export class DkvService {
|
|||||||
|
|
||||||
// Generate xlsx buffer and write to user-files/ (DkvExportService)
|
// Generate xlsx buffer and write to user-files/ (DkvExportService)
|
||||||
const xlsxBuffer = this.exporter.buildExcelBuffer(exportRows);
|
const xlsxBuffer = this.exporter.buildExcelBuffer(exportRows);
|
||||||
const exportFilename = this.exporter.writeAndPrune(xlsxBuffer, rechnungsnummer, invoiceMonth);
|
const exportFilename = this.exporter.writeAndPrune(xlsxBuffer, exportBaseName);
|
||||||
|
|
||||||
// D-16: SMTP send with 3-retry exponential backoff
|
// D-16: SMTP send with 3-retry exponential backoff
|
||||||
let smtpStatus: 'Verarbeitet' | 'Versand fehlgeschlagen' = 'Verarbeitet';
|
let smtpStatus: 'Verarbeitet' | 'Versand fehlgeschlagen' = 'Verarbeitet';
|
||||||
@@ -552,7 +556,7 @@ export class DkvService {
|
|||||||
filename.includes('/') ||
|
filename.includes('/') ||
|
||||||
filename.includes('\\') ||
|
filename.includes('\\') ||
|
||||||
filename.includes('..') ||
|
filename.includes('..') ||
|
||||||
!/^DKV_[\w\-]+\.xlsx$/.test(filename)
|
!/^(RG-DKV-|DKV_)[\w\-]+\.xlsx$/.test(filename)
|
||||||
) {
|
) {
|
||||||
throw new BadRequestException('Invalid export filename');
|
throw new BadRequestException('Invalid export filename');
|
||||||
}
|
}
|
||||||
@@ -624,17 +628,19 @@ export class DkvService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Extract a DKV invoice number from the email subject line.
|
* Build a safe rechnungsnummer string for use in filenames.
|
||||||
* Pattern: DD-DDDDDDDDD-DDD (e.g. "26-651566449-001")
|
* Priority: PDF-extracted number → email subject → uid fallback.
|
||||||
* Falls back to "email-{uid}" when no match is found.
|
* All slashes and non-safe chars are replaced so the value is path-safe.
|
||||||
*/
|
*/
|
||||||
private _extractInvoiceNumber(subject: string, uid: number | string): string {
|
private _buildRechnungsnummer(
|
||||||
const match = subject?.match(/(\d{2}-\d{9}-\d{3})/);
|
fromPdf: string | null,
|
||||||
if (match?.[1]) return match[1];
|
subject: string,
|
||||||
// Fallback when invoice number cannot be parsed from subject.
|
uid: number | string,
|
||||||
// Exchange UniqueIds are base64 and can contain '+', '/', '=' which would
|
): string {
|
||||||
// introduce path separators into the generated filename (WR-04).
|
if (fromPdf) return fromPdf.replace(/\//g, '-');
|
||||||
// Sanitise to [a-zA-Z0-9-] before the value reaches the filesystem.
|
// DKV email subjects carry the invoice number with slashes: "26/650869002/002"
|
||||||
|
const m = subject?.match(/(\d{2}[\/\-]\d{9}[\/\-]\d{3})/);
|
||||||
|
if (m?.[1]) return m[1].replace(/\//g, '-');
|
||||||
return `email-${String(uid).replace(/[^a-zA-Z0-9\-]/g, '_')}`;
|
return `email-${String(uid).replace(/[^a-zA-Z0-9\-]/g, '_')}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -694,6 +700,13 @@ function _delay(ms: number): Promise<void> {
|
|||||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Format "DD.MM.YYYY" → "YYMMDD" for compact filename date segment. */
|
||||||
|
function _formatDateYYMMDD(date: string): string {
|
||||||
|
const [d, m, y] = date.split('.');
|
||||||
|
if (!d || !m || !y) return date.replace(/\./g, '');
|
||||||
|
return `${y.slice(2)}${m}${d}`;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Normalize a Kennzeichen for fuzzy vehicle lookup.
|
* Normalize a Kennzeichen for fuzzy vehicle lookup.
|
||||||
* DKV PDF extraction may omit hyphens or use spaces as separators.
|
* DKV PDF extraction may omit hyphens or use spaces as separators.
|
||||||
|
|||||||
Reference in New Issue
Block a user