Files
tessera-ctl/apps/api/src/tenders/adapters/rss.adapter.spec.ts
T
schalli e812738c3a feat(14-02): add TenderRssFeedSource CRUD, tick poll gate, and wire RssAdapter
Global admin-managed RSS feed list (TenderRssFeedSource, D-08/D-14) with
a save-time hostname/SSRF guard (TenderRssFeedSourceService) — RSS feed
URLs are runtime admin input, so the code-level SourceRegistry denylist
gate does not cover them; a separate check rejects DENYLISTED_PORTALS
hostnames, non-http(s) schemes, and private/loopback hosts.

Adds TenderSourcePollConfig.pollGranularity ('day' | 'tick', D-15):
pollDueSources() branches per source — 'day' sources keep the existing
lastIngestedDay gate byte-unchanged, 'tick' sources (rss) fetch on every
active scheduler tick regardless of lastIngestedDay, since the day-cursor
gate was built for a genuine daily batch-export API and would otherwise
silently cap RSS to one fetch per calendar day.

Wires RssAdapter.fetchTenders() to fan out over active feed rows (native
fetch + AbortController 15s + response-size ceiling, catch-per-feed),
registers it in tenders.module.ts, and seeds the 'rss' poll config
active with pollGranularity='tick' plus a default-active service.bund.de
feed row (subreport-elvis has no single canonical URL — zero rows seeded,
admin adds relevant municipality feeds).

Migration applied locally per project convention (host -> container IP).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-23 13:24:36 +02:00

316 lines
12 KiB
TypeScript

import { readFileSync } from 'fs';
import { join } from 'path';
import { afterEach, describe, expect, it, vi } from 'vitest';
import { RssAdapter } from './rss.adapter';
/**
* Real, live-captured RSS fixtures (both fetched directly against the real
* endpoints, 2026-07-23) — mirrors cosinex.adapter.spec.ts's fixture/mock
* spec style. `service-bund-feed.xml` proves the pubDate-present,
* numeric-HTML-entity-title shape; `subreport-elvis-feed.xml` proves the
* pubDate-absent, CDATA-title shape (14-RESEARCH.md live captures).
*/
const FIXTURES_DIR = join(__dirname, '..', '__fixtures__');
const SERVICE_BUND_FIXTURE = readFileSync(
join(FIXTURES_DIR, 'service-bund-feed.xml'),
'utf8',
);
const SUBREPORT_ELVIS_FIXTURE = readFileSync(
join(FIXTURES_DIR, 'subreport-elvis-feed.xml'),
'utf8',
);
/** Minimal prisma-shaped fake — only `tenderRssFeedSource.findMany` is used by RssAdapter. */
function makeFakePrisma(feeds: any[] = []) {
return {
tenderRssFeedSource: {
findMany: vi.fn(async () => feeds),
},
} as any;
}
/** Stubs global `fetch` to resolve with `xml` for every call (fan-out tests). */
function stubFetchResolvingXml(xmlByUrl: Record<string, string>): void {
vi.stubGlobal(
'fetch',
vi.fn(async (url: string) => {
const xml = xmlByUrl[url];
if (xml === undefined) {
return { ok: false, status: 404 } as Response;
}
const bytes = Buffer.from(xml, 'utf8');
return {
ok: true,
status: 200,
headers: new Headers({ 'content-length': String(bytes.byteLength) }),
arrayBuffer: async () =>
bytes.buffer.slice(bytes.byteOffset, bytes.byteOffset + bytes.byteLength),
} as unknown as Response;
}),
);
}
describe('RssAdapter', () => {
it('declares sourceType rss and the symbolic rss portal', () => {
const adapter = new RssAdapter(makeFakePrisma());
expect(adapter.sourceType).toBe('rss');
expect(adapter.portals).toEqual(['rss']);
});
describe('parseFeed (pure, fixture-driven)', () => {
it('parses service.bund.de items: sourceType/sourcePortal set, pubDate present, numeric HTML entities decoded in title', () => {
const adapter = new RssAdapter(makeFakePrisma());
const records = adapter.parseFeed(SERVICE_BUND_FIXTURE, 'service-bund');
expect(records.length).toBeGreaterThanOrEqual(1);
for (const record of records) {
expect(record.sourceType).toBe('rss');
expect(record.sourcePortal).toBe('service-bund');
expect(record.sourceUrl).toMatch(/^https:\/\/www\.service\.bund\.de\//);
expect(record.eformsPayload).toBeNull();
}
// First fixture item's title is numeric-HTML-entity-encoded
// (`&#220;bermittlung...`), NOT CDATA — must decode to a real "Ü".
const first = records[0];
expect(first).toBeDefined();
const payload = first!.ocdsPayload as { title: string };
expect(payload.title).toMatch(/^Übermittlung von Bohrungsdaten/);
expect(payload.title).not.toContain('&#');
// Every service.bund.de fixture item carries a <pubDate>.
expect(records.every((r) => r.publishedAt instanceof Date)).toBe(true);
});
it('uses the item guid as sourceNoticeId for service.bund.de (plain-text guid, no isPermaLink attribute)', () => {
const adapter = new RssAdapter(makeFakePrisma());
const records = adapter.parseFeed(SERVICE_BUND_FIXTURE, 'service-bund');
const ids = records.map((r) => r.sourceNoticeId);
expect(ids.every((id) => id.startsWith('https://'))).toBe(true);
expect(new Set(ids).size).toBe(ids.length);
});
it('parses subreport-elvis items: publishedAt null (no item pubDate), title from CDATA, guid isPermaLink=false extracted as plain text', () => {
const adapter = new RssAdapter(makeFakePrisma());
const records = adapter.parseFeed(
SUBREPORT_ELVIS_FIXTURE,
'subreport-neuss',
);
expect(records.length).toBeGreaterThanOrEqual(1);
for (const record of records) {
expect(record.sourceType).toBe('rss');
expect(record.sourcePortal).toBe('subreport-neuss');
expect(record.publishedAt).toBeNull();
expect(record.sourceUrl).toMatch(
/^https:\/\/www\.subreport-elvis\.de\//,
);
}
const first = records[0];
expect(first).toBeDefined();
// guid is `<guid isPermaLink="false">E73433797-558464</guid>` —
// fast-xml-parser represents this as { '#text': ..., '@_isPermaLink': ... },
// extractTagText must pull out the plain '#text' value.
expect(first!.sourceNoticeId).toBe('E73433797-558464');
const payload = first!.ocdsPayload as { title: string };
expect(payload.title).toMatch(/^E73433797:/);
});
it('never carries description HTML into the record (V5 — no stored-XSS surface)', () => {
const adapter = new RssAdapter(makeFakePrisma());
const records = adapter.parseFeed(
SUBREPORT_ELVIS_FIXTURE,
'subreport-neuss',
);
for (const record of records) {
const serialized = JSON.stringify(record.ocdsPayload);
expect(serialized).not.toContain('<table>');
expect(serialized).not.toContain('<a href');
}
});
it('bare-minimum bag: buyerName/procedureType/deadlineAt always null (baseline mapping only, D-04/D-05)', () => {
const adapter = new RssAdapter(makeFakePrisma());
const records = adapter.parseFeed(SERVICE_BUND_FIXTURE, 'service-bund');
for (const record of records) {
const payload = record.ocdsPayload as {
buyerName: string | null;
procedureType: string | null;
legalFramework: string | null;
deadlineAt: string | null;
};
expect(payload.buyerName).toBeNull();
expect(payload.procedureType).toBeNull();
expect(payload.legalFramework).toBeNull();
expect(payload.deadlineAt).toBeNull();
}
});
it('returns [] for empty XML instead of throwing', () => {
const adapter = new RssAdapter(makeFakePrisma());
expect(adapter.parseFeed('', 'empty-feed')).toEqual([]);
});
it('returns [] for malformed XML instead of throwing', () => {
const adapter = new RssAdapter(makeFakePrisma());
expect(
adapter.parseFeed('<rss><channel><item><title>', 'broken-feed'),
).toEqual([]);
});
it('returns [] for well-formed XML with no <item> at all', () => {
const adapter = new RssAdapter(makeFakePrisma());
const xml =
'<?xml version="1.0"?><rss version="2.0"><channel><title>Empty</title></channel></rss>';
expect(adapter.parseFeed(xml, 'no-items-feed')).toEqual([]);
});
it('skips a single item missing <link> without aborting the rest', () => {
const adapter = new RssAdapter(makeFakePrisma());
const xml = `<?xml version="1.0"?>
<rss version="2.0"><channel>
<item><title>Ohne Link</title><guid>no-link-guid</guid></item>
<item><title>Mit Link</title><link>https://example.invalid/a</link><guid>with-link-guid</guid></item>
</channel></rss>`;
const records = adapter.parseFeed(xml, 'mixed-feed');
expect(records).toHaveLength(1);
expect(records[0]?.sourceNoticeId).toBe('with-link-guid');
});
it('falls back to sha256(link) for sourceNoticeId when guid is absent', () => {
const adapter = new RssAdapter(makeFakePrisma());
const xml = `<?xml version="1.0"?>
<rss version="2.0"><channel>
<item><title>No Guid</title><link>https://example.invalid/no-guid</link></item>
</channel></rss>`;
const records = adapter.parseFeed(xml, 'no-guid-feed');
expect(records).toHaveLength(1);
expect(records[0]?.sourceNoticeId).toMatch(/^[a-f0-9]{40}$/);
});
});
describe('normalize() dispatch', () => {
it("'rss' sourceType routes through TenderNormalizerService.normalizeBag() (proven end-to-end via tender-normalizer.service.spec.ts)", () => {
// See tender-normalizer.service.spec.ts's "generic ocdsPayload bag"
// describe block for the actual normalize() assertions — this spec
// only proves RssAdapter's OWN output shape (parseFeed), not the
// normalizer dispatch itself (kept in the normalizer's own spec file
// to avoid duplicating TenderNormalizerService test infrastructure).
const adapter = new RssAdapter(makeFakePrisma());
expect(adapter.sourceType).toBe('rss');
});
});
describe('fetchTenders (internal fan-out over active TenderRssFeedSource rows, Plan 14-02 Task 2)', () => {
afterEach(() => {
vi.unstubAllGlobals();
});
it('fetches and parses every active feed, fanning out over findMany({isActive:true})', async () => {
const feeds = [
{ id: 'f1', url: 'https://service-bund.invalid/rss.xml', label: 'service-bund', isActive: true },
{ id: 'f2', url: 'https://subreport.invalid/rss.xml', label: 'subreport-neuss', isActive: true },
];
const prisma = makeFakePrisma(feeds);
stubFetchResolvingXml({
'https://service-bund.invalid/rss.xml': SERVICE_BUND_FIXTURE,
'https://subreport.invalid/rss.xml': SUBREPORT_ELVIS_FIXTURE,
});
const adapter = new RssAdapter(prisma);
const records = await adapter.fetchTenders('2026-07-23');
expect(prisma.tenderRssFeedSource.findMany).toHaveBeenCalledWith({
where: { isActive: true },
});
const portals = new Set(records.map((r) => r.sourcePortal));
expect(portals).toEqual(new Set(['service-bund', 'subreport-neuss']));
expect(records.length).toBeGreaterThan(1);
});
it('catch-per-feed: a feed that fails to fetch is skipped, the other feed still resolves (D-01)', async () => {
const feeds = [
{ id: 'f1', url: 'https://broken.invalid/rss.xml', label: 'broken', isActive: true },
{ id: 'f2', url: 'https://ok.invalid/rss.xml', label: 'ok-feed', isActive: true },
];
const prisma = makeFakePrisma(feeds);
stubFetchResolvingXml({
'https://ok.invalid/rss.xml': SERVICE_BUND_FIXTURE,
// 'broken.invalid' deliberately absent -> stub resolves 404 -> throws
});
const adapter = new RssAdapter(prisma);
const records = await adapter.fetchTenders('2026-07-23');
expect(records.every((r) => r.sourcePortal === 'ok-feed')).toBe(true);
expect(records.length).toBeGreaterThanOrEqual(1);
});
it('returns [] when there are no active feeds', async () => {
const prisma = makeFakePrisma([]);
const adapter = new RssAdapter(prisma);
const records = await adapter.fetchTenders('2026-07-23');
expect(records).toEqual([]);
});
it('returns [] without throwing when a feed fetch throws (network error)', async () => {
const feeds = [
{ id: 'f1', url: 'https://unreachable.invalid/rss.xml', label: 'unreachable', isActive: true },
];
const prisma = makeFakePrisma(feeds);
vi.stubGlobal(
'fetch',
vi.fn(async () => {
throw new Error('network unreachable');
}),
);
const adapter = new RssAdapter(prisma);
const records = await adapter.fetchTenders('2026-07-23');
expect(records).toEqual([]);
});
it('rejects a feed response exceeding the size ceiling (DoS mitigation, T-14-02-02)', async () => {
const feeds = [
{ id: 'f1', url: 'https://huge.invalid/rss.xml', label: 'huge', isActive: true },
];
const prisma = makeFakePrisma(feeds);
vi.stubGlobal(
'fetch',
vi.fn(async () => {
return {
ok: true,
status: 200,
headers: new Headers({ 'content-length': String(11 * 1024 * 1024) }),
arrayBuffer: async () => new ArrayBuffer(0),
} as unknown as Response;
}),
);
const adapter = new RssAdapter(prisma);
const records = await adapter.fetchTenders('2026-07-23');
expect(records).toEqual([]); // oversized feed skipped, catch-per-feed (D-01)
});
});
it('never imports or uses axios (native fetch is the sole HTTP client convention)', () => {
const source = readFileSync(join(__dirname, 'rss.adapter.ts'), 'utf8');
expect(source).not.toMatch(/from ['"]axios['"]/);
});
});