import { Injectable, Logger } from '@nestjs/common'; import { createHash } from 'crypto'; import { XMLParser } from 'fast-xml-parser'; import { PrismaService } from '../../prisma/prisma.service'; import type { RawTenderRecord, SourceType } from '../tender.types'; import type { TenderSourceAdapter } from './tender-source-adapter.interface'; /** * RssAdapter — INGEST-04. Parses admin-managed RSS feed URLs into * RawTenderRecord[] via the same `fast-xml-parser` configuration * doe-opendata.adapter.ts already uses for eForms-DE XML * (`removeNSPrefix`/`ignoreAttributes`/`attributeNamePrefix` — * 14-RESEARCH.md "RSS parsing"). * * `portals: ['rss']` is a SYMBOLIC placeholder, not a real hostname list * (14-RESEARCH.md Pitfall 3): the actual feed hostnames are admin-supplied * RUNTIME data (`TenderRssFeedSource` rows, wired in Plan 14-02 Task 2), * added long after `SourceRegistry.register()`'s DI-boot-time denylist * check runs. The `DENYLISTED_PORTALS` gate therefore provides ZERO * protection for RSS feed URLs — a SEPARATE, save-time hostname/SSRF guard * is enforced in `tender-rss-feed.service.ts` (D-14, Plan 14-02 Task 2, see * threat T-14-02-01). Do not remove that guard under the assumption this * adapter's `portals` array already covers it. * * Live-verified shapes (14-RESEARCH.md, captured 2026-07-23): * - service.bund.de: `` present; `` uses raw numeric * HTML character references (e.g. `Übermittlung...`), NOT CDATA — * decoded explicitly below (Rule 1: fast-xml-parser only decodes the 5 * predefined XML entities by default, not numeric character refs). * - subreport-elvis: `<item><pubDate>` is ABSENT entirely (`publishedAt` * is always null for this source); `<title>`/`<description>` are * CDATA-wrapped (auto-merged to plain strings by fast-xml-parser, no * decoding needed). * * Baseline field mapping only (title/link/guid) — `buyerName`/ * `procedureType`/`deadlineAt` stay null. The optional service.bund.de * `<description>` label-extraction enrichment (Erfüllungsort/ * Vergabestelle/Angebotsfrist) is explicitly out of scope for this task * (RESEARCH.md Pattern 3 "optional enhancement"); `<description>` is never * read by this adapter, so no raw HTML fragment is ever carried into a * RawTenderRecord (V5 — no stored-XSS surface, same text-only discipline * as NetServerAdapter/CosinexAdapter). * * `fetchTenders()` (Plan 14-02 Task 2) is an internal-fan-out adapter * (14-RESEARCH.md Pattern 1, same shape as `NetServerAdapter`'s * multi-portal loop): reads every `isActive` `TenderRssFeedSource` row and * fetches+parses each feed independently, catch-per-feed (D-01 discipline) * — one broken/unreachable feed never blocks the others in the same tick. * `_dayCursor` is accepted for interface conformance but NOT used as a * filter — RSS has no day-batch concept; it is polled every scheduler * tick via `pollGranularity: 'tick'` (D-15, wired in * `tender-ingestion.service.ts`), not gated by the day-cursor at all. */ const RSS_FETCH_TIMEOUT_MS = 15_000; /** RSS feeds are normally small; an admin-supplied URL is less trusted than a hardcoded portal (RESEARCH.md Security Domain, DoS mitigation). */ const RSS_RESPONSE_SIZE_CEILING_BYTES = 10 * 1024 * 1024; @Injectable() export class RssAdapter implements TenderSourceAdapter { readonly sourceType: SourceType = 'rss'; /** Symbolic placeholder — see class docstring re: Pitfall 3. */ readonly portals = ['rss'] as const; private readonly logger = new Logger(RssAdapter.name); private readonly xmlParser = new XMLParser({ removeNSPrefix: true, ignoreAttributes: false, attributeNamePrefix: '@_', }); constructor(private readonly prisma: PrismaService) {} /** * Internal fan-out over every active admin-managed feed (D-14/D-08, * global — no tenantId). Catch-per-feed (D-01): a single feed that fails * to fetch/parse is skipped (logger.warn), never aborting the rest. */ async fetchTenders(_dayCursor: string): Promise<RawTenderRecord[]> { const feeds = await this.prisma.tenderRssFeedSource.findMany({ where: { isActive: true }, }); const records: RawTenderRecord[] = []; for (const feed of feeds) { try { const xml = await this.fetchFeedXml(feed.url); records.push(...this.parseFeed(xml, feed.label)); } catch (error) { this.logger.warn( `RSS feed '${feed.label}' (${feed.url}) fetch failed, skipping this feed for this tick: ${(error as Error).message}`, ); } } return records; } /** * Native fetch + AbortController 15s timeout (project-wide convention, * DoeOpenDataAdapter/NetServerAdapter/CosinexAdapter idiom) plus a * response-size ceiling (T-14-02-02, DoS mitigation) — checked both via * the `Content-Length` header (fast-path, may be absent/wrong) and the * actually-received byte count (authoritative). */ private async fetchFeedXml(url: string): Promise<string> { const controller = new AbortController(); const timeout = setTimeout( () => controller.abort(), RSS_FETCH_TIMEOUT_MS, ); try { const res = await fetch(url, { signal: controller.signal, redirect: 'follow', }); if (!res.ok) { throw new Error(`RSS feed fetch failed: HTTP ${res.status}`); } const contentLength = res.headers.get('content-length'); if ( contentLength && Number(contentLength) > RSS_RESPONSE_SIZE_CEILING_BYTES ) { throw new Error( `RSS feed exceeds size ceiling (Content-Length: ${contentLength} bytes)`, ); } const buffer = await res.arrayBuffer(); if (buffer.byteLength > RSS_RESPONSE_SIZE_CEILING_BYTES) { throw new Error( `RSS feed exceeds size ceiling (${buffer.byteLength} bytes received)`, ); } return new TextDecoder('utf-8').decode(buffer); } finally { clearTimeout(timeout); } } /** * Pure XML -> RawTenderRecord[] mapping. Never throws (Pitfall 2 * discipline, mirrors NetServer/cosinex): malformed/empty XML, or a feed * with no `<channel>`/`<item>` at all, resolves to `[]`. A single item * missing `<link>` is skipped (logger.warn) without aborting the rest. */ parseFeed(xml: string, feedLabel: string): RawTenderRecord[] { const fetchedAt = new Date(); let items: unknown[]; try { const parsed = this.xmlParser.parse(xml) as { rss?: { channel?: { item?: unknown | unknown[] } }; }; const channelItem = parsed.rss?.channel?.item; items = Array.isArray(channelItem) ? channelItem : channelItem ? [channelItem] : []; } catch (error) { this.logger.warn( `RSS feed '${feedLabel}' totally unparsable, returning [] (mirrors NetServer/cosinex Pitfall 2): ${(error as Error).message}`, ); return []; } const records: RawTenderRecord[] = []; for (const item of items) { try { const node = item as Record<string, unknown>; const link = extractTagText(node.link); if (!link) { this.logger.warn( `Skipping RSS item without <link> (feed '${feedLabel}')`, ); continue; // no stable URL -> unusable item, skip (Pitfall 2 discipline) } const rawTitle = extractTagText(node.title); const title = rawTitle ? decodeNumericEntities(rawTitle).trim() : 'Unbenannte Ausschreibung'; const guidText = extractTagText(node.guid); const sourceNoticeId = guidText || createHash('sha256').update(link).digest('hex').slice(0, 40); const pubDateRaw = extractTagText(node.pubDate); const publishedAt = parseRssDate(pubDateRaw); records.push({ sourceType: this.sourceType, sourcePortal: feedLabel, sourceNoticeId, sourceUrl: link, fetchedAt, publishedAt, eformsPayload: null, // RSS has no eForms/OCDS structure; the extracted baseline fields // are carried through this generic bag — same convention as // NetServerAdapter/CosinexAdapter (13-04/13-05) — so // TenderNormalizerService.normalizeBag() maps them without any // RSS-specific normalizer code (D-04/D-05). ocdsPayload: { title, buyerName: null, procedureType: null, legalFramework: null, deadlineAt: null, }, }); } catch (error) { this.logger.warn( `Skipping unparsable RSS item (feed '${feedLabel}'): ${(error as Error).message}`, ); } } return records; } } /** * fast-xml-parser represents a plain tag as a raw string/number, and a tag * with attributes (e.g. subreport-elvis's `<guid isPermaLink="false">`) as * `{ '@_attr': ..., '#text': value }` — mirrors * tender-normalizer.service.ts's `textValue()` helper (kept local/ * duplicated rather than imported: this is adapter-layer XML-shape * handling, not normalizer-layer field mapping). */ function extractTagText(node: unknown): string | null { if (node === null || node === undefined) return null; if (typeof node === 'string') return node || null; if (typeof node === 'number') return String(node); if ( typeof node === 'object' && '#text' in (node as Record<string, unknown>) ) { const t = (node as Record<string, unknown>)['#text']; if (t === null || t === undefined) return null; return String(t); } return null; } /** * Rule 1 fix: fast-xml-parser only decodes the 5 predefined XML entities * (`&` `<` `>` `"` `'`) — numeric character references * (`Ü`, `ß`) pass through UNDECODED (confirmed live against * fast-xml-parser v5.10.1, see class docstring). service.bund.de titles use * numeric refs directly (not CDATA-wrapped, unlike subreport-elvis) — * without this decode step, titles would literally show "Übermittlung * ..." to admins/users instead of "Übermittlung...". */ function decodeNumericEntities(text: string): string { return text .replace(/&#(\d+);/g, (_match, dec: string) => String.fromCodePoint(Number(dec)), ) .replace(/&#x([0-9a-fA-F]+);/g, (_match, hex: string) => String.fromCodePoint(Number.parseInt(hex, 16)), ); } /** * RSS `pubDate` is RFC-822 (`Thu, 23 Jul 2026 11:15:00 +0200`), parseable * directly by `Date`. Absent/malformed -> null (subreport-elvis has no * per-item `pubDate` at all — confirmed live, see class docstring). */ function parseRssDate(raw: string | null): Date | null { if (!raw) return null; const parsed = new Date(raw); return Number.isNaN(parsed.getTime()) ? null : parsed; }