import { lookup } from 'node:dns/promises';
import { isIP } from 'node:net';
import { Injectable } from '@nestjs/common';
import { Agent, type Response as UndiciResponse, fetch as undiciFetch } from 'undici';
/**
* Server-side favicon / icon discovery with SSRF protection (T-08-05).
*
* Ported from personal-dashboard/src/lib/favorite-icons.ts.
* Security guards:
* - DNS resolves every URL (including redirects) and checks for private IP ranges
* - redirect: 'manual' — follows redirects manually so each hop is re-validated
* - 4000 ms AbortController timeout per request
* - 200 000 character HTML cap to prevent memory exhaustion (T-08-09)
* - Blocked hostnames: localhost, .local, 0.0.0.0
* - 260917-jdd: Zertifikatsfehler des Zielhosts werden toleriert (siehe
* LENIENT_TLS_AGENT unten) — DNS-Pruefung, Redirect-Limit, Timeout und
* Groessendeckel bleiben davon unberuehrt.
*/
const FALLBACK_ICON_PATH = '/favicon.ico';
const HTML_FETCH_TIMEOUT_MS = 4000;
const ICON_FETCH_TIMEOUT_MS = 4000;
const MAX_REDIRECTS = 2;
const MAX_HTML_CHARS = 200000;
const MAX_ICON_BYTES = 1_000_000;
/**
* quick-261001-hbi — oeffentlicher Symbol-Dienst als letzter Rueckfall. Manche
* Seiten setzen ihr Symbol erst per JavaScript (hosteurope.de: im HTML nur
* ``, `/favicon.ico` liefert eine
* HTML-Seite) — ohne Browser findet die Suche dort nichts. DuckDuckGo kennt
* das gerenderte Symbol und antwortet fuer Unbekanntes mit 404 (dann bleibt
* der Buchstabe). Gefragt wird NUR fuer oeffentlich erreichbare Adressen,
* damit interne Hostnamen (docuvita.ctl.local, private IPs) das Haus nie
* verlassen; der Dienst erfaehrt nur den Hostnamen.
*/
const PUBLIC_ICON_SERVICE = 'https://icons.duckduckgo.com/ip3/';
/**
* 260917-jdd — Ziel ist ein Bildchen, kein Geheimnis: selbstsignierte,
* abgelaufene oder falsch benannte Zertifikate sollen das Symbol eines
* Favoriten nicht verhindern. Dieser Dispatcher gilt AUSSCHLIESSLICH fuer
* die beiden Aufrufe in dieser Datei (Dispatcher pro Aufruf, keine
* prozessweite Abschaltung der Zertifikatspruefung — insbesondere NICHT
* ueber die Node-Umgebungsvariable, die mit NODE_TLS_ beginnt).
*
* Der Dispatcher wirkt nur zusammen mit undicis EIGENEM `fetch` — Nodes
* globales `fetch` ignoriert einen Agent aus dem npm-Paket (andere Klasse,
* Node 24 buendelt intern undici 7.25.0). Gemessen 2026-09-17 gegen
* self-signed.badssl.com: `undiciFetch(url, { dispatcher: new Agent(...) })`
* -> Status 200; `globalThis.fetch` derselben URL -> DEPTH_ZERO_SELF_SIGNED_CERT.
* Deshalb der Modulimport oben statt des globalen `fetch`.
*
* DNS-Pruefung (isPublicHttpUrl), Redirect-Limit (MAX_REDIRECTS), Timeout
* und Groessendeckel (MAX_ICON_BYTES/MAX_HTML_CHARS) bleiben davon
* unberuehrt (T-JDD-01).
*/
const LENIENT_TLS_AGENT = new Agent({ connect: { rejectUnauthorized: false } });
type FetchHtmlResult = {
html: string;
finalUrl: string;
/** false = die Seite antwortete mit einem Fehlerstatus (z. B. 400/404), lieferte aber HTML (260929-lh3). */
ok: boolean;
};
function isPrivateIpv4(address: string): boolean {
const parts = address.split('.').map((part) => Number.parseInt(part, 10));
if (
parts.length !== 4 ||
parts.some((part) => !Number.isInteger(part) || part < 0 || part > 255)
) {
return true;
}
const [a, b] = parts;
return (
a === 0 ||
a === 10 ||
a === 127 ||
(a === 100 && b !== undefined && b >= 64 && b <= 127) ||
(a === 169 && b === 254) ||
(a === 172 && b !== undefined && b >= 16 && b <= 31) ||
(a === 192 && b === 168) ||
(a === 192 && b === 0) ||
(a === 198 && (b === 18 || b === 19)) ||
a >= 224
);
}
function isPrivateIpv6(address: string): boolean {
const lower = address.toLowerCase();
if (
lower === '::' ||
lower === '::1' ||
lower.startsWith('fc') ||
lower.startsWith('fd') ||
lower.startsWith('fe80:') ||
lower.startsWith('ff')
) {
return true;
}
// IPv4-mapped IPv6 (::ffff:) — delegate to isPrivateIpv4 to cover all
// RFC 1918 ranges (10.x, 172.16-31.x, 192.168.x) and 169.254.x link-local
const v4MappedMatch = lower.match(/^::ffff:(\d+\.\d+\.\d+\.\d+)$/);
if (v4MappedMatch) {
return isPrivateIpv4(v4MappedMatch[1]);
}
return false;
}
function isPrivateIpAddress(address: string): boolean {
const version = isIP(address);
if (version === 4) return isPrivateIpv4(address);
if (version === 6) return isPrivateIpv6(address);
return true; // Unknown format → block by default
}
function isBlockedHostname(hostname: string): boolean {
const h = hostname.trim().toLowerCase();
return h === 'localhost' || h.endsWith('.localhost') || h.endsWith('.local') || h === '0.0.0.0';
}
export async function isPublicHttpUrl(url: URL): Promise {
if (url.protocol !== 'http:' && url.protocol !== 'https:') {
return false;
}
if (isBlockedHostname(url.hostname)) {
return false;
}
const directVersion = isIP(url.hostname);
if (directVersion !== 0) {
return !isPrivateIpAddress(url.hostname);
}
try {
const addresses = await lookup(url.hostname, { all: true });
if (addresses.length === 0) return false;
return addresses.every((a) => !isPrivateIpAddress(a.address));
} catch {
return false;
}
}
/**
* Normalize a user-entered site URL by prepending https:// when no scheme is
* present, so "ctl.de" becomes "https://ctl.de". Without this, `new URL()`
* throws on a bare host and icon discovery silently falls back to a broken
* relative "/favicon.ico" (which then 502s through the icon proxy and the
* widget shows the first-letter placeholder instead of the real favicon).
*/
export function normalizeUrl(raw: string): string {
const trimmed = raw.trim();
if (!trimmed) return trimmed;
// Already has a scheme (http://, https://, ftp://, ...) — leave untouched.
if (/^[a-z][a-z0-9+.-]*:\/\//i.test(trimmed)) return trimmed;
return `https://${trimmed}`;
}
function getOriginFaviconUrl(pageUrl: string): string {
try {
const url = new URL(pageUrl);
return new URL(FALLBACK_ICON_PATH, url.origin).toString();
} catch {
return FALLBACK_ICON_PATH;
}
}
function parseAttributes(tag: string): Record {
const attrs: Record = {};
const re = /([a-zA-Z_:.-]+)\s*=\s*("([^"]*)"|'([^']*)'|([^\s"'>]+))/g;
for (let m = re.exec(tag); m !== null; m = re.exec(tag)) {
const key = m[1].toLowerCase();
const value = m[3] ?? m[4] ?? m[5] ?? '';
attrs[key] = value;
}
return attrs;
}
function toAbsoluteUrl(value: string | undefined, base: string): string | null {
if (!value) return null;
try {
const url = new URL(value, base);
if (url.protocol !== 'http:' && url.protocol !== 'https:') return null;
return url.toString();
} catch {
return null;
}
}
function extractIconFromHtml(html: string, baseUrl: string, linkTagsOnly = false): string | null {
const linkTags = html.match(/]*>/gi) ?? [];
const metaTags = html.match(/]*>/gi) ?? [];
const linkCandidates = linkTags
.map((tag) => parseAttributes(tag))
.map((a) => ({
rel: (a.rel ?? '').toLowerCase(),
href: toAbsoluteUrl(a.href, baseUrl),
}))
.filter((c) => c.href);
const appleTouchIcon = linkCandidates.find((c) => c.rel.includes('apple-touch-icon'))?.href;
if (appleTouchIcon) return appleTouchIcon;
const icon = linkCandidates.find((c) => c.rel.split(/\s+/).includes('icon'))?.href;
if (icon) return icon;
const shortcutIcon = linkCandidates.find((c) => c.rel.includes('shortcut icon'))?.href;
if (shortcutIcon) return shortcutIcon;
const imageSrc = linkCandidates.find((c) => c.rel.includes('image_src'))?.href;
if (imageSrc) return imageSrc;
// 260929-lh3: eine Fehlerseite (Status != 2xx) traegt kein Vorschaubild der
// Seite — nur die ausdruecklichen Symbol-Verweise () zaehlen.
if (linkTagsOnly) return null;
const metaImage = metaTags
.map((tag) => parseAttributes(tag))
.map((a) => ({
property: (a.property ?? a.name ?? '').toLowerCase(),
content: toAbsoluteUrl(a.content, baseUrl),
}))
.find(
(c) =>
c.content &&
(c.property === 'og:image' || c.property === 'og:logo' || c.property === 'twitter:image'),
)?.content;
return metaImage ?? null;
}
/**
* Shared SSRF-guarded fetch used by every outbound request this service makes.
* Follows redirects manually (up to MAX_REDIRECTS) so each hop is re-validated
* against isPublicHttpUrl before being requested — a redirect target must not
* be able to bypass the private-IP/blocked-hostname guard.
*
* Returns the final non-redirect Response, or null if the guard blocks any
* hop, the request errors, or the redirect budget is exhausted.
*/
async function fetchWithRedirectGuard(
pageUrl: URL,
options: {
accept: string;
timeoutMs: number;
userAgent?: string;
/** 260929-lh3: auch eine 4xx/5xx-Antwort zurueckgeben (nur fuer die HTML-Suche). */
allowErrorStatus?: boolean;
},
): Promise<{ response: UndiciResponse; finalUrl: URL } | null> {
let currentUrl = pageUrl;
for (let redirectCount = 0; redirectCount <= MAX_REDIRECTS; redirectCount++) {
const isPublic = await isPublicHttpUrl(currentUrl);
if (!isPublic) return null;
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), options.timeoutMs);
try {
const response = await undiciFetch(currentUrl.toString(), {
dispatcher: LENIENT_TLS_AGENT, // 260917-jdd: siehe Kommentar an der Konstante
redirect: 'manual', // SSRF: follow manually so each hop is re-validated
signal: controller.signal,
headers: {
Accept: options.accept,
'User-Agent': options.userAgent ?? 'tessera/1.0',
},
});
if (response.status >= 300 && response.status < 400) {
const location = response.headers.get('location');
if (!location) return null;
currentUrl = new URL(location, currentUrl);
continue;
}
if (!response.ok && !options.allowErrorStatus) return null;
return { response, finalUrl: currentUrl };
} catch {
return null;
} finally {
clearTimeout(timeout);
}
}
return null;
}
/** Minimaler Ausschnitt einer Antwort, den die beiden Helfer brauchen. */
type BodyResponse = Pick;
/**
* Verwirft den Body einer nicht gebrauchten Antwort. Fehler (bereits
* gelesen/abgebrochen) sind egal.
*/
export function discardBody(response: Pick): void {
try {
response.body?.cancel().catch(() => {});
} catch {
// Body gesperrt oder schon verbraucht — nichts zu tun.
}
}
/**
* Liest den Antworttext hoechstens bis `maxChars` Zeichen und bricht den
* Stream danach ab (T-08-09). Vorher wurde der komplette Body gelesen und
* erst danach abgeschnitten — eine riesige Seite landete ganz im Speicher.
* Dekodiert wird UTF-8 wie bei `Response.text()`; da jedes Zeichen aus
* mindestens einem Byte entsteht, bleibt der Speicher bei ~maxChars plus
* einem Chunk. Ohne Stream (`body === null`) wie bisher ueber `text()`.
* `timeoutMs` begrenzt zusaetzlich die Lesedauer: die Zeitgrenze von
* `fetchWithRedirectGuard` endet mit den Kopfzeilen, ein Server, der den
* Body tropfenweise liefert, hielte die Anfrage sonst beliebig lange auf.
* Nach Ablauf zaehlt, was bis dahin gelesen ist.
*/
export async function readTextCapped(
response: BodyResponse,
maxChars: number,
timeoutMs = HTML_FETCH_TIMEOUT_MS,
): Promise {
if (!response.body) {
return (await response.text()).slice(0, maxChars);
}
const reader = response.body.getReader();
const decoder = new TextDecoder();
let text = '';
// cancel() beendet ein haengendes read() mit done: true.
const deadline = setTimeout(() => void reader.cancel().catch(() => {}), timeoutMs);
try {
while (true) {
const { done, value } = await reader.read();
if (done) {
text += decoder.decode();
break;
}
text += decoder.decode(value, { stream: true });
if (text.length >= maxChars) {
await reader.cancel().catch(() => {});
break;
}
}
} finally {
clearTimeout(deadline);
}
return text.slice(0, maxChars);
}
async function fetchHtml(pageUrl: URL): Promise {
const result = await fetchWithRedirectGuard(pageUrl, {
accept: 'text/html,application/xhtml+xml,*/*',
timeoutMs: HTML_FETCH_TIMEOUT_MS,
// 260929-lh3: Server wie docuvita antworten dem Server mit 400, tragen im
// HTML aber trotzdem den — den Verweis wollen wir haben.
allowErrorStatus: true,
});
if (!result) return null;
const contentType = result.response.headers.get('content-type') ?? '';
if (!contentType.toLowerCase().includes('text/html')) {
// Kein HTML (auch bei Fehlerstatus dank allowErrorStatus hier moeglich):
// Body verwerfen, sonst haelt undici die Verbindung bis zum Timeout offen.
discardBody(result.response);
return null;
}
// T-08-09: HTML cap — schon beim Lesen, nicht erst nach dem kompletten Body.
const html = await readTextCapped(result.response, MAX_HTML_CHARS);
return {
html,
finalUrl: result.finalUrl.toString(),
ok: result.response.ok,
};
}
@Injectable()
export class IconDiscoveryService {
/**
* Discover the best icon URL for a given web page URL.
* Falls back to /favicon.ico when discovery fails or URL is private.
*
* SSRF protection: every URL and redirect target is validated against
* private IP ranges, blocked hostnames, and forced-proxy vectors (T-08-05).
*/
async discoverFavoriteIconUrl(pageUrl: string): Promise {
const normalized = normalizeUrl(pageUrl);
const fallback = getOriginFaviconUrl(normalized);
try {
const url = new URL(normalized);
const htmlResult = await fetchHtml(url);
if (!htmlResult) return fallback;
return extractIconFromHtml(htmlResult.html, htmlResult.finalUrl, !htmlResult.ok) ?? fallback;
} catch {
return fallback;
}
}
/**
* quick-261001-hbi: Symbol fuer die Seite `pageUrl` beim oeffentlichen
* Symbol-Dienst holen (siehe PUBLIC_ICON_SERVICE). Wirft, wenn die Seite
* nicht oeffentlich erreichbar ist (dann wird der Dienst NICHT gefragt) oder
* der Dienst kein Symbol kennt (404) — wie `fetchIconBytes`.
*/
async fetchPublicServiceIconBytes(
pageUrl: string,
): Promise<{ contentType: string; body: Buffer }> {
const page = new URL(normalizeUrl(pageUrl));
if (!(await isPublicHttpUrl(page))) {
throw new Error('Page is not public, icon service not asked');
}
return this.fetchIconBytes(`${PUBLIC_ICON_SERVICE}${page.hostname}.ico`);
}
/**
* Fetch the raw bytes of a stored icon URL, SSRF-guarded, for streaming
* back to the browser from Tessera's own origin (avoids Cross-Origin-
* Resource-Policy blocks on hotlinked cross-origin loads).
*
* Throws on any failure — blocked target, timeout, non-image content-type,
* or an oversized body. Callers must not return a placeholder image; let
* the caller map the failure to an HTTP error status instead.
*/
async fetchIconBytes(iconUrl: string): Promise<{ contentType: string; body: Buffer }> {
const url = new URL(iconUrl);
const result = await fetchWithRedirectGuard(url, {
// A browser-realistic Accept header and User-Agent avoid tripping
// bot-mitigation WAFs that block non-browser clients (observed
// reproducibly: chatgpt.com/favicon.ico returns 403 for our default
// "tessera/1.0" UA and 200 for a real Chrome UA string, confirmed by
// repeated direct comparison). The HTML-discovery path is unaffected —
// this User-Agent override only applies to this byte-fetch call.
accept: 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8',
userAgent:
'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
timeoutMs: ICON_FETCH_TIMEOUT_MS,
});
if (!result) {
throw new Error('Icon fetch blocked or failed');
}
const contentType = result.response.headers.get('content-type') ?? '';
if (!contentType.toLowerCase().startsWith('image/')) {
throw new Error(`Icon response is not an image (${contentType})`);
}
const arrayBuffer = await result.response.arrayBuffer();
if (arrayBuffer.byteLength > MAX_ICON_BYTES) {
throw new Error('Icon response exceeds size limit');
}
return { contentType, body: Buffer.from(arrayBuffer) };
}
}