feat(14-02): add RssAdapter.parseFeed + 'rss' normalizer dispatch

Fixture-first RSS parsing (INGEST-04): parses live-captured
service.bund.de (pubDate present, numeric-HTML-entity titles) and
subreport-elvis (pubDate absent, CDATA titles) feed shapes into
RawTenderRecord[] via fast-xml-parser, mirroring the DoeOpenDataAdapter
config. SourceType extended with 'rss'; normalize() dispatches 'rss'
through the existing normalizeBag() path unchanged (D-04/D-05).

Rule 1 fix: fast-xml-parser only decodes the 5 predefined XML entities,
not numeric character references — added an explicit decode step so
service.bund.de titles ("Übermittlung...") render correctly
instead of leaking raw entity syntax.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-07-23 13:17:02 +02:00
parent a47c0c57ee
commit 3a96cbbbe6
7 changed files with 597 additions and 6 deletions
@@ -0,0 +1,189 @@
import { readFileSync } from 'fs';
import { join } from 'path';
import { describe, expect, it } from 'vitest';
import { RssAdapter } from './rss.adapter';
/**
* Real, live-captured RSS fixtures (both fetched directly against the real
* endpoints, 2026-07-23) — mirrors cosinex.adapter.spec.ts's fixture/mock
* spec style. `service-bund-feed.xml` proves the pubDate-present,
* numeric-HTML-entity-title shape; `subreport-elvis-feed.xml` proves the
* pubDate-absent, CDATA-title shape (14-RESEARCH.md live captures).
*/
const FIXTURES_DIR = join(__dirname, '..', '__fixtures__');
const SERVICE_BUND_FIXTURE = readFileSync(
join(FIXTURES_DIR, 'service-bund-feed.xml'),
'utf8',
);
const SUBREPORT_ELVIS_FIXTURE = readFileSync(
join(FIXTURES_DIR, 'subreport-elvis-feed.xml'),
'utf8',
);
describe('RssAdapter', () => {
it('declares sourceType rss and the symbolic rss portal', () => {
const adapter = new RssAdapter();
expect(adapter.sourceType).toBe('rss');
expect(adapter.portals).toEqual(['rss']);
});
describe('parseFeed (pure, fixture-driven)', () => {
it('parses service.bund.de items: sourceType/sourcePortal set, pubDate present, numeric HTML entities decoded in title', () => {
const adapter = new RssAdapter();
const records = adapter.parseFeed(SERVICE_BUND_FIXTURE, 'service-bund');
expect(records.length).toBeGreaterThanOrEqual(1);
for (const record of records) {
expect(record.sourceType).toBe('rss');
expect(record.sourcePortal).toBe('service-bund');
expect(record.sourceUrl).toMatch(/^https:\/\/www\.service\.bund\.de\//);
expect(record.eformsPayload).toBeNull();
}
// First fixture item's title is numeric-HTML-entity-encoded
// (`&#220;bermittlung...`), NOT CDATA — must decode to a real "Ü".
const first = records[0];
expect(first).toBeDefined();
const payload = first!.ocdsPayload as { title: string };
expect(payload.title).toMatch(/^Übermittlung von Bohrungsdaten/);
expect(payload.title).not.toContain('&#');
// Every service.bund.de fixture item carries a <pubDate>.
expect(records.every((r) => r.publishedAt instanceof Date)).toBe(true);
});
it('uses the item guid as sourceNoticeId for service.bund.de (plain-text guid, no isPermaLink attribute)', () => {
const adapter = new RssAdapter();
const records = adapter.parseFeed(SERVICE_BUND_FIXTURE, 'service-bund');
const ids = records.map((r) => r.sourceNoticeId);
expect(ids.every((id) => id.startsWith('https://'))).toBe(true);
expect(new Set(ids).size).toBe(ids.length);
});
it('parses subreport-elvis items: publishedAt null (no item pubDate), title from CDATA, guid isPermaLink=false extracted as plain text', () => {
const adapter = new RssAdapter();
const records = adapter.parseFeed(
SUBREPORT_ELVIS_FIXTURE,
'subreport-neuss',
);
expect(records.length).toBeGreaterThanOrEqual(1);
for (const record of records) {
expect(record.sourceType).toBe('rss');
expect(record.sourcePortal).toBe('subreport-neuss');
expect(record.publishedAt).toBeNull();
expect(record.sourceUrl).toMatch(
/^https:\/\/www\.subreport-elvis\.de\//,
);
}
const first = records[0];
expect(first).toBeDefined();
// guid is `<guid isPermaLink="false">E73433797-558464</guid>` —
// fast-xml-parser represents this as { '#text': ..., '@_isPermaLink': ... },
// extractTagText must pull out the plain '#text' value.
expect(first!.sourceNoticeId).toBe('E73433797-558464');
const payload = first!.ocdsPayload as { title: string };
expect(payload.title).toMatch(/^E73433797:/);
});
it('never carries description HTML into the record (V5 — no stored-XSS surface)', () => {
const adapter = new RssAdapter();
const records = adapter.parseFeed(
SUBREPORT_ELVIS_FIXTURE,
'subreport-neuss',
);
for (const record of records) {
const serialized = JSON.stringify(record.ocdsPayload);
expect(serialized).not.toContain('<table>');
expect(serialized).not.toContain('<a href');
}
});
it('bare-minimum bag: buyerName/procedureType/deadlineAt always null (baseline mapping only, D-04/D-05)', () => {
const adapter = new RssAdapter();
const records = adapter.parseFeed(SERVICE_BUND_FIXTURE, 'service-bund');
for (const record of records) {
const payload = record.ocdsPayload as {
buyerName: string | null;
procedureType: string | null;
legalFramework: string | null;
deadlineAt: string | null;
};
expect(payload.buyerName).toBeNull();
expect(payload.procedureType).toBeNull();
expect(payload.legalFramework).toBeNull();
expect(payload.deadlineAt).toBeNull();
}
});
it('returns [] for empty XML instead of throwing', () => {
const adapter = new RssAdapter();
expect(adapter.parseFeed('', 'empty-feed')).toEqual([]);
});
it('returns [] for malformed XML instead of throwing', () => {
const adapter = new RssAdapter();
expect(
adapter.parseFeed('<rss><channel><item><title>', 'broken-feed'),
).toEqual([]);
});
it('returns [] for well-formed XML with no <item> at all', () => {
const adapter = new RssAdapter();
const xml =
'<?xml version="1.0"?><rss version="2.0"><channel><title>Empty</title></channel></rss>';
expect(adapter.parseFeed(xml, 'no-items-feed')).toEqual([]);
});
it('skips a single item missing <link> without aborting the rest', () => {
const adapter = new RssAdapter();
const xml = `<?xml version="1.0"?>
<rss version="2.0"><channel>
<item><title>Ohne Link</title><guid>no-link-guid</guid></item>
<item><title>Mit Link</title><link>https://example.invalid/a</link><guid>with-link-guid</guid></item>
</channel></rss>`;
const records = adapter.parseFeed(xml, 'mixed-feed');
expect(records).toHaveLength(1);
expect(records[0]?.sourceNoticeId).toBe('with-link-guid');
});
it('falls back to sha256(link) for sourceNoticeId when guid is absent', () => {
const adapter = new RssAdapter();
const xml = `<?xml version="1.0"?>
<rss version="2.0"><channel>
<item><title>No Guid</title><link>https://example.invalid/no-guid</link></item>
</channel></rss>`;
const records = adapter.parseFeed(xml, 'no-guid-feed');
expect(records).toHaveLength(1);
expect(records[0]?.sourceNoticeId).toMatch(/^[a-f0-9]{40}$/);
});
});
describe('normalize() dispatch', () => {
it("'rss' sourceType routes through TenderNormalizerService.normalizeBag() (proven end-to-end via tender-normalizer.service.spec.ts)", () => {
// See tender-normalizer.service.spec.ts's "generic ocdsPayload bag"
// describe block for the actual normalize() assertions — this spec
// only proves RssAdapter's OWN output shape (parseFeed), not the
// normalizer dispatch itself (kept in the normalizer's own spec file
// to avoid duplicating TenderNormalizerService test infrastructure).
const adapter = new RssAdapter();
expect(adapter.sourceType).toBe('rss');
});
});
it('never imports or uses axios (native fetch is the sole HTTP client convention)', () => {
const source = readFileSync(join(__dirname, 'rss.adapter.ts'), 'utf8');
expect(source).not.toMatch(/from ['"]axios['"]/);
});
});