2dd11b439d
hosteurope.de liefert im HTML nur einen leeren data:-Platzhalter, das echte Symbol setzt erst JavaScript; /favicon.ico antwortet mit HTML. Die Kachel zeigte deshalb nur den Buchstaben. Scheitert das gespeicherte Symbol, fragt getIconBytes jetzt einmal den DuckDuckGo-Symboldienst - nur fuer oeffentlich erreichbare Seiten, interne Hostnamen verlassen das Haus nicht; kennt der Dienst nichts (404), bleibt es beim Buchstaben. Wirkt auch fuer bestehende Favoriten. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
501 lines
16 KiB
TypeScript
501 lines
16 KiB
TypeScript
import { lookup } from 'node:dns/promises';
|
|
import { isIP } from 'node:net';
|
|
import { Injectable } from '@nestjs/common';
|
|
import { Agent, type Response as UndiciResponse, fetch as undiciFetch } from 'undici';
|
|
|
|
/**
|
|
* Server-side favicon / icon discovery with SSRF protection (T-08-05).
|
|
*
|
|
* Ported from personal-dashboard/src/lib/favorite-icons.ts.
|
|
* Security guards:
|
|
* - DNS resolves every URL (including redirects) and checks for private IP ranges
|
|
* - redirect: 'manual' — follows redirects manually so each hop is re-validated
|
|
* - 4000 ms AbortController timeout per request
|
|
* - 200 000 character HTML cap to prevent memory exhaustion (T-08-09)
|
|
* - Blocked hostnames: localhost, .local, 0.0.0.0
|
|
* - 260917-jdd: Zertifikatsfehler des Zielhosts werden toleriert (siehe
|
|
* LENIENT_TLS_AGENT unten) — DNS-Pruefung, Redirect-Limit, Timeout und
|
|
* Groessendeckel bleiben davon unberuehrt.
|
|
*/
|
|
|
|
const FALLBACK_ICON_PATH = '/favicon.ico';
|
|
const HTML_FETCH_TIMEOUT_MS = 4000;
|
|
const ICON_FETCH_TIMEOUT_MS = 4000;
|
|
const MAX_REDIRECTS = 2;
|
|
const MAX_HTML_CHARS = 200000;
|
|
const MAX_ICON_BYTES = 1_000_000;
|
|
|
|
/**
|
|
* quick-261001-hbi — oeffentlicher Symbol-Dienst als letzter Rueckfall. Manche
|
|
* Seiten setzen ihr Symbol erst per JavaScript (hosteurope.de: im HTML nur
|
|
* `<link rel="icon" href="data:;base64,=">`, `/favicon.ico` liefert eine
|
|
* HTML-Seite) — ohne Browser findet die Suche dort nichts. DuckDuckGo kennt
|
|
* das gerenderte Symbol und antwortet fuer Unbekanntes mit 404 (dann bleibt
|
|
* der Buchstabe). Gefragt wird NUR fuer oeffentlich erreichbare Adressen,
|
|
* damit interne Hostnamen (docuvita.ctl.local, private IPs) das Haus nie
|
|
* verlassen; der Dienst erfaehrt nur den Hostnamen.
|
|
*/
|
|
const PUBLIC_ICON_SERVICE = 'https://icons.duckduckgo.com/ip3/';
|
|
|
|
/**
|
|
* 260917-jdd — Ziel ist ein Bildchen, kein Geheimnis: selbstsignierte,
|
|
* abgelaufene oder falsch benannte Zertifikate sollen das Symbol eines
|
|
* Favoriten nicht verhindern. Dieser Dispatcher gilt AUSSCHLIESSLICH fuer
|
|
* die beiden Aufrufe in dieser Datei (Dispatcher pro Aufruf, keine
|
|
* prozessweite Abschaltung der Zertifikatspruefung — insbesondere NICHT
|
|
* ueber die Node-Umgebungsvariable, die mit NODE_TLS_ beginnt).
|
|
*
|
|
* Der Dispatcher wirkt nur zusammen mit undicis EIGENEM `fetch` — Nodes
|
|
* globales `fetch` ignoriert einen Agent aus dem npm-Paket (andere Klasse,
|
|
* Node 24 buendelt intern undici 7.25.0). Gemessen 2026-09-17 gegen
|
|
* self-signed.badssl.com: `undiciFetch(url, { dispatcher: new Agent(...) })`
|
|
* -> Status 200; `globalThis.fetch` derselben URL -> DEPTH_ZERO_SELF_SIGNED_CERT.
|
|
* Deshalb der Modulimport oben statt des globalen `fetch`.
|
|
*
|
|
* DNS-Pruefung (isPublicHttpUrl), Redirect-Limit (MAX_REDIRECTS), Timeout
|
|
* und Groessendeckel (MAX_ICON_BYTES/MAX_HTML_CHARS) bleiben davon
|
|
* unberuehrt (T-JDD-01).
|
|
*/
|
|
const LENIENT_TLS_AGENT = new Agent({ connect: { rejectUnauthorized: false } });
|
|
|
|
type FetchHtmlResult = {
|
|
html: string;
|
|
finalUrl: string;
|
|
/** false = die Seite antwortete mit einem Fehlerstatus (z. B. 400/404), lieferte aber HTML (260929-lh3). */
|
|
ok: boolean;
|
|
};
|
|
|
|
function isPrivateIpv4(address: string): boolean {
|
|
const parts = address.split('.').map((part) => Number.parseInt(part, 10));
|
|
|
|
if (
|
|
parts.length !== 4 ||
|
|
parts.some((part) => !Number.isInteger(part) || part < 0 || part > 255)
|
|
) {
|
|
return true;
|
|
}
|
|
|
|
const [a, b] = parts;
|
|
|
|
return (
|
|
a === 0 ||
|
|
a === 10 ||
|
|
a === 127 ||
|
|
(a === 100 && b !== undefined && b >= 64 && b <= 127) ||
|
|
(a === 169 && b === 254) ||
|
|
(a === 172 && b !== undefined && b >= 16 && b <= 31) ||
|
|
(a === 192 && b === 168) ||
|
|
(a === 192 && b === 0) ||
|
|
(a === 198 && (b === 18 || b === 19)) ||
|
|
a >= 224
|
|
);
|
|
}
|
|
|
|
function isPrivateIpv6(address: string): boolean {
|
|
const lower = address.toLowerCase();
|
|
|
|
if (
|
|
lower === '::' ||
|
|
lower === '::1' ||
|
|
lower.startsWith('fc') ||
|
|
lower.startsWith('fd') ||
|
|
lower.startsWith('fe80:') ||
|
|
lower.startsWith('ff')
|
|
) {
|
|
return true;
|
|
}
|
|
|
|
// IPv4-mapped IPv6 (::ffff:<ipv4>) — delegate to isPrivateIpv4 to cover all
|
|
// RFC 1918 ranges (10.x, 172.16-31.x, 192.168.x) and 169.254.x link-local
|
|
const v4MappedMatch = lower.match(/^::ffff:(\d+\.\d+\.\d+\.\d+)$/);
|
|
if (v4MappedMatch) {
|
|
return isPrivateIpv4(v4MappedMatch[1]);
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
function isPrivateIpAddress(address: string): boolean {
|
|
const version = isIP(address);
|
|
|
|
if (version === 4) return isPrivateIpv4(address);
|
|
if (version === 6) return isPrivateIpv6(address);
|
|
|
|
return true; // Unknown format → block by default
|
|
}
|
|
|
|
function isBlockedHostname(hostname: string): boolean {
|
|
const h = hostname.trim().toLowerCase();
|
|
|
|
return h === 'localhost' || h.endsWith('.localhost') || h.endsWith('.local') || h === '0.0.0.0';
|
|
}
|
|
|
|
export async function isPublicHttpUrl(url: URL): Promise<boolean> {
|
|
if (url.protocol !== 'http:' && url.protocol !== 'https:') {
|
|
return false;
|
|
}
|
|
|
|
if (isBlockedHostname(url.hostname)) {
|
|
return false;
|
|
}
|
|
|
|
const directVersion = isIP(url.hostname);
|
|
|
|
if (directVersion !== 0) {
|
|
return !isPrivateIpAddress(url.hostname);
|
|
}
|
|
|
|
try {
|
|
const addresses = await lookup(url.hostname, { all: true });
|
|
|
|
if (addresses.length === 0) return false;
|
|
|
|
return addresses.every((a) => !isPrivateIpAddress(a.address));
|
|
} catch {
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Normalize a user-entered site URL by prepending https:// when no scheme is
|
|
* present, so "ctl.de" becomes "https://ctl.de". Without this, `new URL()`
|
|
* throws on a bare host and icon discovery silently falls back to a broken
|
|
* relative "/favicon.ico" (which then 502s through the icon proxy and the
|
|
* widget shows the first-letter placeholder instead of the real favicon).
|
|
*/
|
|
export function normalizeUrl(raw: string): string {
|
|
const trimmed = raw.trim();
|
|
if (!trimmed) return trimmed;
|
|
// Already has a scheme (http://, https://, ftp://, ...) — leave untouched.
|
|
if (/^[a-z][a-z0-9+.-]*:\/\//i.test(trimmed)) return trimmed;
|
|
return `https://${trimmed}`;
|
|
}
|
|
|
|
function getOriginFaviconUrl(pageUrl: string): string {
|
|
try {
|
|
const url = new URL(pageUrl);
|
|
|
|
return new URL(FALLBACK_ICON_PATH, url.origin).toString();
|
|
} catch {
|
|
return FALLBACK_ICON_PATH;
|
|
}
|
|
}
|
|
|
|
function parseAttributes(tag: string): Record<string, string> {
|
|
const attrs: Record<string, string> = {};
|
|
const re = /([a-zA-Z_:.-]+)\s*=\s*("([^"]*)"|'([^']*)'|([^\s"'>]+))/g;
|
|
|
|
for (let m = re.exec(tag); m !== null; m = re.exec(tag)) {
|
|
const key = m[1].toLowerCase();
|
|
const value = m[3] ?? m[4] ?? m[5] ?? '';
|
|
|
|
attrs[key] = value;
|
|
}
|
|
|
|
return attrs;
|
|
}
|
|
|
|
function toAbsoluteUrl(value: string | undefined, base: string): string | null {
|
|
if (!value) return null;
|
|
|
|
try {
|
|
const url = new URL(value, base);
|
|
|
|
if (url.protocol !== 'http:' && url.protocol !== 'https:') return null;
|
|
|
|
return url.toString();
|
|
} catch {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
function extractIconFromHtml(html: string, baseUrl: string, linkTagsOnly = false): string | null {
|
|
const linkTags = html.match(/<link\b[^>]*>/gi) ?? [];
|
|
const metaTags = html.match(/<meta\b[^>]*>/gi) ?? [];
|
|
|
|
const linkCandidates = linkTags
|
|
.map((tag) => parseAttributes(tag))
|
|
.map((a) => ({
|
|
rel: (a.rel ?? '').toLowerCase(),
|
|
href: toAbsoluteUrl(a.href, baseUrl),
|
|
}))
|
|
.filter((c) => c.href);
|
|
|
|
const appleTouchIcon = linkCandidates.find((c) => c.rel.includes('apple-touch-icon'))?.href;
|
|
|
|
if (appleTouchIcon) return appleTouchIcon;
|
|
|
|
const icon = linkCandidates.find((c) => c.rel.split(/\s+/).includes('icon'))?.href;
|
|
|
|
if (icon) return icon;
|
|
|
|
const shortcutIcon = linkCandidates.find((c) => c.rel.includes('shortcut icon'))?.href;
|
|
|
|
if (shortcutIcon) return shortcutIcon;
|
|
|
|
const imageSrc = linkCandidates.find((c) => c.rel.includes('image_src'))?.href;
|
|
|
|
if (imageSrc) return imageSrc;
|
|
|
|
// 260929-lh3: eine Fehlerseite (Status != 2xx) traegt kein Vorschaubild der
|
|
// Seite — nur die ausdruecklichen Symbol-Verweise (<link rel=...icon>) zaehlen.
|
|
if (linkTagsOnly) return null;
|
|
|
|
const metaImage = metaTags
|
|
.map((tag) => parseAttributes(tag))
|
|
.map((a) => ({
|
|
property: (a.property ?? a.name ?? '').toLowerCase(),
|
|
content: toAbsoluteUrl(a.content, baseUrl),
|
|
}))
|
|
.find(
|
|
(c) =>
|
|
c.content &&
|
|
(c.property === 'og:image' || c.property === 'og:logo' || c.property === 'twitter:image'),
|
|
)?.content;
|
|
|
|
return metaImage ?? null;
|
|
}
|
|
|
|
/**
|
|
* Shared SSRF-guarded fetch used by every outbound request this service makes.
|
|
* Follows redirects manually (up to MAX_REDIRECTS) so each hop is re-validated
|
|
* against isPublicHttpUrl before being requested — a redirect target must not
|
|
* be able to bypass the private-IP/blocked-hostname guard.
|
|
*
|
|
* Returns the final non-redirect Response, or null if the guard blocks any
|
|
* hop, the request errors, or the redirect budget is exhausted.
|
|
*/
|
|
async function fetchWithRedirectGuard(
|
|
pageUrl: URL,
|
|
options: {
|
|
accept: string;
|
|
timeoutMs: number;
|
|
userAgent?: string;
|
|
/** 260929-lh3: auch eine 4xx/5xx-Antwort zurueckgeben (nur fuer die HTML-Suche). */
|
|
allowErrorStatus?: boolean;
|
|
},
|
|
): Promise<{ response: UndiciResponse; finalUrl: URL } | null> {
|
|
let currentUrl = pageUrl;
|
|
|
|
for (let redirectCount = 0; redirectCount <= MAX_REDIRECTS; redirectCount++) {
|
|
const isPublic = await isPublicHttpUrl(currentUrl);
|
|
|
|
if (!isPublic) return null;
|
|
|
|
const controller = new AbortController();
|
|
const timeout = setTimeout(() => controller.abort(), options.timeoutMs);
|
|
|
|
try {
|
|
const response = await undiciFetch(currentUrl.toString(), {
|
|
dispatcher: LENIENT_TLS_AGENT, // 260917-jdd: siehe Kommentar an der Konstante
|
|
redirect: 'manual', // SSRF: follow manually so each hop is re-validated
|
|
signal: controller.signal,
|
|
headers: {
|
|
Accept: options.accept,
|
|
'User-Agent': options.userAgent ?? 'tessera/1.0',
|
|
},
|
|
});
|
|
|
|
if (response.status >= 300 && response.status < 400) {
|
|
const location = response.headers.get('location');
|
|
|
|
if (!location) return null;
|
|
|
|
currentUrl = new URL(location, currentUrl);
|
|
continue;
|
|
}
|
|
|
|
if (!response.ok && !options.allowErrorStatus) return null;
|
|
|
|
return { response, finalUrl: currentUrl };
|
|
} catch {
|
|
return null;
|
|
} finally {
|
|
clearTimeout(timeout);
|
|
}
|
|
}
|
|
|
|
return null;
|
|
}
|
|
|
|
/** Minimaler Ausschnitt einer Antwort, den die beiden Helfer brauchen. */
|
|
type BodyResponse = Pick<UndiciResponse, 'body' | 'text'>;
|
|
|
|
/**
|
|
* Verwirft den Body einer nicht gebrauchten Antwort. Fehler (bereits
|
|
* gelesen/abgebrochen) sind egal.
|
|
*/
|
|
export function discardBody(response: Pick<UndiciResponse, 'body'>): void {
|
|
try {
|
|
response.body?.cancel().catch(() => {});
|
|
} catch {
|
|
// Body gesperrt oder schon verbraucht — nichts zu tun.
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Liest den Antworttext hoechstens bis `maxChars` Zeichen und bricht den
|
|
* Stream danach ab (T-08-09). Vorher wurde der komplette Body gelesen und
|
|
* erst danach abgeschnitten — eine riesige Seite landete ganz im Speicher.
|
|
* Dekodiert wird UTF-8 wie bei `Response.text()`; da jedes Zeichen aus
|
|
* mindestens einem Byte entsteht, bleibt der Speicher bei ~maxChars plus
|
|
* einem Chunk. Ohne Stream (`body === null`) wie bisher ueber `text()`.
|
|
* `timeoutMs` begrenzt zusaetzlich die Lesedauer: die Zeitgrenze von
|
|
* `fetchWithRedirectGuard` endet mit den Kopfzeilen, ein Server, der den
|
|
* Body tropfenweise liefert, hielte die Anfrage sonst beliebig lange auf.
|
|
* Nach Ablauf zaehlt, was bis dahin gelesen ist.
|
|
*/
|
|
export async function readTextCapped(
|
|
response: BodyResponse,
|
|
maxChars: number,
|
|
timeoutMs = HTML_FETCH_TIMEOUT_MS,
|
|
): Promise<string> {
|
|
if (!response.body) {
|
|
return (await response.text()).slice(0, maxChars);
|
|
}
|
|
|
|
const reader = response.body.getReader();
|
|
const decoder = new TextDecoder();
|
|
let text = '';
|
|
// cancel() beendet ein haengendes read() mit done: true.
|
|
const deadline = setTimeout(() => void reader.cancel().catch(() => {}), timeoutMs);
|
|
|
|
try {
|
|
while (true) {
|
|
const { done, value } = await reader.read();
|
|
|
|
if (done) {
|
|
text += decoder.decode();
|
|
break;
|
|
}
|
|
|
|
text += decoder.decode(value, { stream: true });
|
|
|
|
if (text.length >= maxChars) {
|
|
await reader.cancel().catch(() => {});
|
|
break;
|
|
}
|
|
}
|
|
} finally {
|
|
clearTimeout(deadline);
|
|
}
|
|
|
|
return text.slice(0, maxChars);
|
|
}
|
|
|
|
async function fetchHtml(pageUrl: URL): Promise<FetchHtmlResult | null> {
|
|
const result = await fetchWithRedirectGuard(pageUrl, {
|
|
accept: 'text/html,application/xhtml+xml,*/*',
|
|
timeoutMs: HTML_FETCH_TIMEOUT_MS,
|
|
// 260929-lh3: Server wie docuvita antworten dem Server mit 400, tragen im
|
|
// HTML aber trotzdem den <link rel=icon> — den Verweis wollen wir haben.
|
|
allowErrorStatus: true,
|
|
});
|
|
|
|
if (!result) return null;
|
|
|
|
const contentType = result.response.headers.get('content-type') ?? '';
|
|
|
|
if (!contentType.toLowerCase().includes('text/html')) {
|
|
// Kein HTML (auch bei Fehlerstatus dank allowErrorStatus hier moeglich):
|
|
// Body verwerfen, sonst haelt undici die Verbindung bis zum Timeout offen.
|
|
discardBody(result.response);
|
|
return null;
|
|
}
|
|
|
|
// T-08-09: HTML cap — schon beim Lesen, nicht erst nach dem kompletten Body.
|
|
const html = await readTextCapped(result.response, MAX_HTML_CHARS);
|
|
|
|
return {
|
|
html,
|
|
finalUrl: result.finalUrl.toString(),
|
|
ok: result.response.ok,
|
|
};
|
|
}
|
|
|
|
@Injectable()
|
|
export class IconDiscoveryService {
|
|
/**
|
|
* Discover the best icon URL for a given web page URL.
|
|
* Falls back to <origin>/favicon.ico when discovery fails or URL is private.
|
|
*
|
|
* SSRF protection: every URL and redirect target is validated against
|
|
* private IP ranges, blocked hostnames, and forced-proxy vectors (T-08-05).
|
|
*/
|
|
async discoverFavoriteIconUrl(pageUrl: string): Promise<string> {
|
|
const normalized = normalizeUrl(pageUrl);
|
|
const fallback = getOriginFaviconUrl(normalized);
|
|
|
|
try {
|
|
const url = new URL(normalized);
|
|
const htmlResult = await fetchHtml(url);
|
|
|
|
if (!htmlResult) return fallback;
|
|
|
|
return extractIconFromHtml(htmlResult.html, htmlResult.finalUrl, !htmlResult.ok) ?? fallback;
|
|
} catch {
|
|
return fallback;
|
|
}
|
|
}
|
|
|
|
/**
|
|
* quick-261001-hbi: Symbol fuer die Seite `pageUrl` beim oeffentlichen
|
|
* Symbol-Dienst holen (siehe PUBLIC_ICON_SERVICE). Wirft, wenn die Seite
|
|
* nicht oeffentlich erreichbar ist (dann wird der Dienst NICHT gefragt) oder
|
|
* der Dienst kein Symbol kennt (404) — wie `fetchIconBytes`.
|
|
*/
|
|
async fetchPublicServiceIconBytes(
|
|
pageUrl: string,
|
|
): Promise<{ contentType: string; body: Buffer }> {
|
|
const page = new URL(normalizeUrl(pageUrl));
|
|
if (!(await isPublicHttpUrl(page))) {
|
|
throw new Error('Page is not public, icon service not asked');
|
|
}
|
|
return this.fetchIconBytes(`${PUBLIC_ICON_SERVICE}${page.hostname}.ico`);
|
|
}
|
|
|
|
/**
|
|
* Fetch the raw bytes of a stored icon URL, SSRF-guarded, for streaming
|
|
* back to the browser from Tessera's own origin (avoids Cross-Origin-
|
|
* Resource-Policy blocks on hotlinked cross-origin <img> loads).
|
|
*
|
|
* Throws on any failure — blocked target, timeout, non-image content-type,
|
|
* or an oversized body. Callers must not return a placeholder image; let
|
|
* the caller map the failure to an HTTP error status instead.
|
|
*/
|
|
async fetchIconBytes(iconUrl: string): Promise<{ contentType: string; body: Buffer }> {
|
|
const url = new URL(iconUrl);
|
|
|
|
const result = await fetchWithRedirectGuard(url, {
|
|
// A browser-realistic Accept header and User-Agent avoid tripping
|
|
// bot-mitigation WAFs that block non-browser clients (observed
|
|
// reproducibly: chatgpt.com/favicon.ico returns 403 for our default
|
|
// "tessera/1.0" UA and 200 for a real Chrome UA string, confirmed by
|
|
// repeated direct comparison). The HTML-discovery path is unaffected —
|
|
// this User-Agent override only applies to this byte-fetch call.
|
|
accept: 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8',
|
|
userAgent:
|
|
'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
|
timeoutMs: ICON_FETCH_TIMEOUT_MS,
|
|
});
|
|
|
|
if (!result) {
|
|
throw new Error('Icon fetch blocked or failed');
|
|
}
|
|
|
|
const contentType = result.response.headers.get('content-type') ?? '';
|
|
|
|
if (!contentType.toLowerCase().startsWith('image/')) {
|
|
throw new Error(`Icon response is not an image (${contentType})`);
|
|
}
|
|
|
|
const arrayBuffer = await result.response.arrayBuffer();
|
|
|
|
if (arrayBuffer.byteLength > MAX_ICON_BYTES) {
|
|
throw new Error('Icon response exceeds size limit');
|
|
}
|
|
|
|
return { contentType, body: Buffer.from(arrayBuffer) };
|
|
}
|
|
}
|