fix(dkv): extract invoice number/date from PDF, rename export files to RG-DKV format
- Parser now extracts Rechnungsnummer (DD/DDDDDDDDD/DDD) and Rechnungsdatum
from PDF text, so filename doesn't rely on email subject
- Export filename changed from DKV_YYYY-MM_... to RG-DKV-{nr}-{YYMMDD}.xlsx
e.g. RG-DKV-26-650869002-002-260331.xlsx
- Subject fallback now also matches slash-separated invoice numbers (26/NNN/NNN)
- writeAndPrune simplified to accept baseName instead of separate fields
- Validation regex and prune prefix updated to match new RG-DKV- pattern
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -31,7 +31,11 @@ export class DkvParserService {
|
||||
* @returns Array of vehicle blocks with transactions
|
||||
* @throws Error with generic message if no vehicle blocks are found
|
||||
*/
|
||||
async parsePdf(buffer: Buffer): Promise<DkvVehicleBlock[]> {
|
||||
async parsePdf(buffer: Buffer): Promise<{
|
||||
vehicles: DkvVehicleBlock[];
|
||||
rechnungsnummer: string | null;
|
||||
rechnungsdatum: string | null;
|
||||
}> {
|
||||
let text: string;
|
||||
try {
|
||||
text = await this.extractText(buffer);
|
||||
@@ -52,7 +56,30 @@ export class DkvParserService {
|
||||
`DKV PDF parsed: ${vehicles.length} vehicle(s), ${totalTx} transaction(s)`,
|
||||
);
|
||||
|
||||
return vehicles;
|
||||
return {
|
||||
vehicles,
|
||||
rechnungsnummer: this.extractRechnungsnummer(text),
|
||||
rechnungsdatum: this.extractRechnungsdatum(text),
|
||||
};
|
||||
}
|
||||
|
||||
/** Extract first DKV invoice number (DD/DDDDDDDDD/DDD) from PDF text. */
|
||||
private extractRechnungsnummer(text: string): string | null {
|
||||
const m = text.match(/\b(\d{2}\/\d{9}\/\d{3})\b/);
|
||||
return m ? m[1] : null;
|
||||
}
|
||||
|
||||
/** Extract invoice date ("DD.MM.YYYY") appearing near "Rechnungsdatum" in PDF text. */
|
||||
private extractRechnungsdatum(text: string): string | null {
|
||||
// "Rechnungsdatum" is followed (within ~200 chars) by a date string
|
||||
const m = text.match(/Rechnungsdatum[:\s\t]*(\d{2}\.\d{2}\.\d{4})/);
|
||||
if (m) return m[1];
|
||||
// Fallback: second line after "Rechnungsdatum:" marker
|
||||
const idx = text.indexOf('Rechnungsdatum');
|
||||
if (idx === -1) return null;
|
||||
const after = text.slice(idx, idx + 200);
|
||||
const dm = after.match(/(\d{2}\.\d{2}\.\d{4})/);
|
||||
return dm ? dm[1] : null;
|
||||
}
|
||||
|
||||
// ─── Private: PDF text extraction ──────────────────────────────────────────
|
||||
|
||||
Reference in New Issue
Block a user