Files
tessera-ctl/apps/api/src/tenders/adapters/rss.adapter.ts
T
schalli e812738c3a feat(14-02): add TenderRssFeedSource CRUD, tick poll gate, and wire RssAdapter
Global admin-managed RSS feed list (TenderRssFeedSource, D-08/D-14) with
a save-time hostname/SSRF guard (TenderRssFeedSourceService) — RSS feed
URLs are runtime admin input, so the code-level SourceRegistry denylist
gate does not cover them; a separate check rejects DENYLISTED_PORTALS
hostnames, non-http(s) schemes, and private/loopback hosts.

Adds TenderSourcePollConfig.pollGranularity ('day' | 'tick', D-15):
pollDueSources() branches per source — 'day' sources keep the existing
lastIngestedDay gate byte-unchanged, 'tick' sources (rss) fetch on every
active scheduler tick regardless of lastIngestedDay, since the day-cursor
gate was built for a genuine daily batch-export API and would otherwise
silently cap RSS to one fetch per calendar day.

Wires RssAdapter.fetchTenders() to fan out over active feed rows (native
fetch + AbortController 15s + response-size ceiling, catch-per-feed),
registers it in tenders.module.ts, and seeds the 'rss' poll config
active with pollGranularity='tick' plus a default-active service.bund.de
feed row (subreport-elvis has no single canonical URL — zero rows seeded,
admin adds relevant municipality feeds).

Migration applied locally per project convention (host -> container IP).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-23 13:24:36 +02:00

282 lines
11 KiB
TypeScript

import { Injectable, Logger } from '@nestjs/common';
import { createHash } from 'crypto';
import { XMLParser } from 'fast-xml-parser';
import { PrismaService } from '../../prisma/prisma.service';
import type { RawTenderRecord, SourceType } from '../tender.types';
import type { TenderSourceAdapter } from './tender-source-adapter.interface';
/**
* RssAdapter — INGEST-04. Parses admin-managed RSS feed URLs into
* RawTenderRecord[] via the same `fast-xml-parser` configuration
* doe-opendata.adapter.ts already uses for eForms-DE XML
* (`removeNSPrefix`/`ignoreAttributes`/`attributeNamePrefix` —
* 14-RESEARCH.md "RSS parsing").
*
* `portals: ['rss']` is a SYMBOLIC placeholder, not a real hostname list
* (14-RESEARCH.md Pitfall 3): the actual feed hostnames are admin-supplied
* RUNTIME data (`TenderRssFeedSource` rows, wired in Plan 14-02 Task 2),
* added long after `SourceRegistry.register()`'s DI-boot-time denylist
* check runs. The `DENYLISTED_PORTALS` gate therefore provides ZERO
* protection for RSS feed URLs — a SEPARATE, save-time hostname/SSRF guard
* is enforced in `tender-rss-feed.service.ts` (D-14, Plan 14-02 Task 2, see
* threat T-14-02-01). Do not remove that guard under the assumption this
* adapter's `portals` array already covers it.
*
* Live-verified shapes (14-RESEARCH.md, captured 2026-07-23):
* - service.bund.de: `<item><pubDate>` present; `<title>` uses raw numeric
* HTML character references (e.g. `&#220;bermittlung...`), NOT CDATA —
* decoded explicitly below (Rule 1: fast-xml-parser only decodes the 5
* predefined XML entities by default, not numeric character refs).
* - subreport-elvis: `<item><pubDate>` is ABSENT entirely (`publishedAt`
* is always null for this source); `<title>`/`<description>` are
* CDATA-wrapped (auto-merged to plain strings by fast-xml-parser, no
* decoding needed).
*
* Baseline field mapping only (title/link/guid) — `buyerName`/
* `procedureType`/`deadlineAt` stay null. The optional service.bund.de
* `<description>` label-extraction enrichment (Erfüllungsort/
* Vergabestelle/Angebotsfrist) is explicitly out of scope for this task
* (RESEARCH.md Pattern 3 "optional enhancement"); `<description>` is never
* read by this adapter, so no raw HTML fragment is ever carried into a
* RawTenderRecord (V5 — no stored-XSS surface, same text-only discipline
* as NetServerAdapter/CosinexAdapter).
*
* `fetchTenders()` (Plan 14-02 Task 2) is an internal-fan-out adapter
* (14-RESEARCH.md Pattern 1, same shape as `NetServerAdapter`'s
* multi-portal loop): reads every `isActive` `TenderRssFeedSource` row and
* fetches+parses each feed independently, catch-per-feed (D-01 discipline)
* — one broken/unreachable feed never blocks the others in the same tick.
* `_dayCursor` is accepted for interface conformance but NOT used as a
* filter — RSS has no day-batch concept; it is polled every scheduler
* tick via `pollGranularity: 'tick'` (D-15, wired in
* `tender-ingestion.service.ts`), not gated by the day-cursor at all.
*/
const RSS_FETCH_TIMEOUT_MS = 15_000;
/** RSS feeds are normally small; an admin-supplied URL is less trusted than a hardcoded portal (RESEARCH.md Security Domain, DoS mitigation). */
const RSS_RESPONSE_SIZE_CEILING_BYTES = 10 * 1024 * 1024;
@Injectable()
export class RssAdapter implements TenderSourceAdapter {
readonly sourceType: SourceType = 'rss';
/** Symbolic placeholder — see class docstring re: Pitfall 3. */
readonly portals = ['rss'] as const;
private readonly logger = new Logger(RssAdapter.name);
private readonly xmlParser = new XMLParser({
removeNSPrefix: true,
ignoreAttributes: false,
attributeNamePrefix: '@_',
});
constructor(private readonly prisma: PrismaService) {}
/**
* Internal fan-out over every active admin-managed feed (D-14/D-08,
* global — no tenantId). Catch-per-feed (D-01): a single feed that fails
* to fetch/parse is skipped (logger.warn), never aborting the rest.
*/
async fetchTenders(_dayCursor: string): Promise<RawTenderRecord[]> {
const feeds = await this.prisma.tenderRssFeedSource.findMany({
where: { isActive: true },
});
const records: RawTenderRecord[] = [];
for (const feed of feeds) {
try {
const xml = await this.fetchFeedXml(feed.url);
records.push(...this.parseFeed(xml, feed.label));
} catch (error) {
this.logger.warn(
`RSS feed '${feed.label}' (${feed.url}) fetch failed, skipping this feed for this tick: ${(error as Error).message}`,
);
}
}
return records;
}
/**
* Native fetch + AbortController 15s timeout (project-wide convention,
* DoeOpenDataAdapter/NetServerAdapter/CosinexAdapter idiom) plus a
* response-size ceiling (T-14-02-02, DoS mitigation) — checked both via
* the `Content-Length` header (fast-path, may be absent/wrong) and the
* actually-received byte count (authoritative).
*/
private async fetchFeedXml(url: string): Promise<string> {
const controller = new AbortController();
const timeout = setTimeout(
() => controller.abort(),
RSS_FETCH_TIMEOUT_MS,
);
try {
const res = await fetch(url, {
signal: controller.signal,
redirect: 'follow',
});
if (!res.ok) {
throw new Error(`RSS feed fetch failed: HTTP ${res.status}`);
}
const contentLength = res.headers.get('content-length');
if (
contentLength &&
Number(contentLength) > RSS_RESPONSE_SIZE_CEILING_BYTES
) {
throw new Error(
`RSS feed exceeds size ceiling (Content-Length: ${contentLength} bytes)`,
);
}
const buffer = await res.arrayBuffer();
if (buffer.byteLength > RSS_RESPONSE_SIZE_CEILING_BYTES) {
throw new Error(
`RSS feed exceeds size ceiling (${buffer.byteLength} bytes received)`,
);
}
return new TextDecoder('utf-8').decode(buffer);
} finally {
clearTimeout(timeout);
}
}
/**
* Pure XML -> RawTenderRecord[] mapping. Never throws (Pitfall 2
* discipline, mirrors NetServer/cosinex): malformed/empty XML, or a feed
* with no `<channel>`/`<item>` at all, resolves to `[]`. A single item
* missing `<link>` is skipped (logger.warn) without aborting the rest.
*/
parseFeed(xml: string, feedLabel: string): RawTenderRecord[] {
const fetchedAt = new Date();
let items: unknown[];
try {
const parsed = this.xmlParser.parse(xml) as {
rss?: { channel?: { item?: unknown | unknown[] } };
};
const channelItem = parsed.rss?.channel?.item;
items = Array.isArray(channelItem)
? channelItem
: channelItem
? [channelItem]
: [];
} catch (error) {
this.logger.warn(
`RSS feed '${feedLabel}' totally unparsable, returning [] (mirrors NetServer/cosinex Pitfall 2): ${(error as Error).message}`,
);
return [];
}
const records: RawTenderRecord[] = [];
for (const item of items) {
try {
const node = item as Record<string, unknown>;
const link = extractTagText(node.link);
if (!link) {
this.logger.warn(
`Skipping RSS item without <link> (feed '${feedLabel}')`,
);
continue; // no stable URL -> unusable item, skip (Pitfall 2 discipline)
}
const rawTitle = extractTagText(node.title);
const title = rawTitle
? decodeNumericEntities(rawTitle).trim()
: 'Unbenannte Ausschreibung';
const guidText = extractTagText(node.guid);
const sourceNoticeId =
guidText ||
createHash('sha256').update(link).digest('hex').slice(0, 40);
const pubDateRaw = extractTagText(node.pubDate);
const publishedAt = parseRssDate(pubDateRaw);
records.push({
sourceType: this.sourceType,
sourcePortal: feedLabel,
sourceNoticeId,
sourceUrl: link,
fetchedAt,
publishedAt,
eformsPayload: null,
// RSS has no eForms/OCDS structure; the extracted baseline fields
// are carried through this generic bag — same convention as
// NetServerAdapter/CosinexAdapter (13-04/13-05) — so
// TenderNormalizerService.normalizeBag() maps them without any
// RSS-specific normalizer code (D-04/D-05).
ocdsPayload: {
title,
buyerName: null,
procedureType: null,
legalFramework: null,
deadlineAt: null,
},
});
} catch (error) {
this.logger.warn(
`Skipping unparsable RSS item (feed '${feedLabel}'): ${(error as Error).message}`,
);
}
}
return records;
}
}
/**
* fast-xml-parser represents a plain tag as a raw string/number, and a tag
* with attributes (e.g. subreport-elvis's `<guid isPermaLink="false">`) as
* `{ '@_attr': ..., '#text': value }` — mirrors
* tender-normalizer.service.ts's `textValue()` helper (kept local/
* duplicated rather than imported: this is adapter-layer XML-shape
* handling, not normalizer-layer field mapping).
*/
function extractTagText(node: unknown): string | null {
if (node === null || node === undefined) return null;
if (typeof node === 'string') return node || null;
if (typeof node === 'number') return String(node);
if (
typeof node === 'object' &&
'#text' in (node as Record<string, unknown>)
) {
const t = (node as Record<string, unknown>)['#text'];
if (t === null || t === undefined) return null;
return String(t);
}
return null;
}
/**
* Rule 1 fix: fast-xml-parser only decodes the 5 predefined XML entities
* (`&amp;` `&lt;` `&gt;` `&quot;` `&apos;`) — numeric character references
* (`&#220;`, `&#x00DF;`) pass through UNDECODED (confirmed live against
* fast-xml-parser v5.10.1, see class docstring). service.bund.de titles use
* numeric refs directly (not CDATA-wrapped, unlike subreport-elvis) —
* without this decode step, titles would literally show "&#220;bermittlung
* ..." to admins/users instead of "Übermittlung...".
*/
function decodeNumericEntities(text: string): string {
return text
.replace(/&#(\d+);/g, (_match, dec: string) =>
String.fromCodePoint(Number(dec)),
)
.replace(/&#x([0-9a-fA-F]+);/g, (_match, hex: string) =>
String.fromCodePoint(Number.parseInt(hex, 16)),
);
}
/**
* RSS `pubDate` is RFC-822 (`Thu, 23 Jul 2026 11:15:00 +0200`), parseable
* directly by `Date`. Absent/malformed -> null (subreport-elvis has no
* per-item `pubDate` at all — confirmed live, see class docstring).
*/
function parseRssDate(raw: string | null): Date | null {
if (!raw) return null;
const parsed = new Date(raw);
return Number.isNaN(parsed.getTime()) ? null : parsed;
}