30d6e0a5df
A bare "image/*" Accept paired with the tessera/1.0 User-Agent tripped Cloudflare bot mitigation on some sites -- caught live testing against chatgpt.com/favicon.ico, which returned 403 with this combo but 200 with a realistic browser-style image Accept list. Isolated via direct fetch comparison inside the API container before landing the fix. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
360 lines
9.8 KiB
TypeScript
360 lines
9.8 KiB
TypeScript
import { Injectable } from '@nestjs/common';
|
|
import { lookup } from 'dns/promises';
|
|
import { isIP } from 'net';
|
|
|
|
/**
|
|
* Server-side favicon / icon discovery with SSRF protection (T-08-05).
|
|
*
|
|
* Ported from personal-dashboard/src/lib/favorite-icons.ts.
|
|
* Security guards:
|
|
* - DNS resolves every URL (including redirects) and checks for private IP ranges
|
|
* - redirect: 'manual' — follows redirects manually so each hop is re-validated
|
|
* - 4000 ms AbortController timeout per request
|
|
* - 200 000 character HTML cap to prevent memory exhaustion (T-08-09)
|
|
* - Blocked hostnames: localhost, .local, 0.0.0.0
|
|
*/
|
|
|
|
const FALLBACK_ICON_PATH = '/favicon.ico';
|
|
const HTML_FETCH_TIMEOUT_MS = 4000;
|
|
const ICON_FETCH_TIMEOUT_MS = 4000;
|
|
const MAX_REDIRECTS = 2;
|
|
const MAX_HTML_CHARS = 200000;
|
|
const MAX_ICON_BYTES = 1_000_000;
|
|
|
|
type FetchHtmlResult = {
|
|
html: string;
|
|
finalUrl: string;
|
|
};
|
|
|
|
function isPrivateIpv4(address: string): boolean {
|
|
const parts = address.split('.').map((part) => Number.parseInt(part, 10));
|
|
|
|
if (
|
|
parts.length !== 4 ||
|
|
parts.some(
|
|
(part) => !Number.isInteger(part) || part < 0 || part > 255,
|
|
)
|
|
) {
|
|
return true;
|
|
}
|
|
|
|
const [a, b] = parts;
|
|
|
|
return (
|
|
a === 0 ||
|
|
a === 10 ||
|
|
a === 127 ||
|
|
(a === 100 && b !== undefined && b >= 64 && b <= 127) ||
|
|
(a === 169 && b === 254) ||
|
|
(a === 172 && b !== undefined && b >= 16 && b <= 31) ||
|
|
(a === 192 && b === 168) ||
|
|
(a === 192 && b === 0) ||
|
|
(a === 198 && (b === 18 || b === 19)) ||
|
|
a >= 224
|
|
);
|
|
}
|
|
|
|
function isPrivateIpv6(address: string): boolean {
|
|
const lower = address.toLowerCase();
|
|
|
|
if (
|
|
lower === '::' ||
|
|
lower === '::1' ||
|
|
lower.startsWith('fc') ||
|
|
lower.startsWith('fd') ||
|
|
lower.startsWith('fe80:') ||
|
|
lower.startsWith('ff')
|
|
) {
|
|
return true;
|
|
}
|
|
|
|
// IPv4-mapped IPv6 (::ffff:<ipv4>) — delegate to isPrivateIpv4 to cover all
|
|
// RFC 1918 ranges (10.x, 172.16-31.x, 192.168.x) and 169.254.x link-local
|
|
const v4MappedMatch = lower.match(/^::ffff:(\d+\.\d+\.\d+\.\d+)$/);
|
|
if (v4MappedMatch) {
|
|
return isPrivateIpv4(v4MappedMatch[1]);
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
function isPrivateIpAddress(address: string): boolean {
|
|
const version = isIP(address);
|
|
|
|
if (version === 4) return isPrivateIpv4(address);
|
|
if (version === 6) return isPrivateIpv6(address);
|
|
|
|
return true; // Unknown format → block by default
|
|
}
|
|
|
|
function isBlockedHostname(hostname: string): boolean {
|
|
const h = hostname.trim().toLowerCase();
|
|
|
|
return (
|
|
h === 'localhost' ||
|
|
h.endsWith('.localhost') ||
|
|
h.endsWith('.local') ||
|
|
h === '0.0.0.0'
|
|
);
|
|
}
|
|
|
|
export async function isPublicHttpUrl(url: URL): Promise<boolean> {
|
|
if (url.protocol !== 'http:' && url.protocol !== 'https:') {
|
|
return false;
|
|
}
|
|
|
|
if (isBlockedHostname(url.hostname)) {
|
|
return false;
|
|
}
|
|
|
|
const directVersion = isIP(url.hostname);
|
|
|
|
if (directVersion !== 0) {
|
|
return !isPrivateIpAddress(url.hostname);
|
|
}
|
|
|
|
try {
|
|
const addresses = await lookup(url.hostname, { all: true });
|
|
|
|
if (addresses.length === 0) return false;
|
|
|
|
return addresses.every((a) => !isPrivateIpAddress(a.address));
|
|
} catch {
|
|
return false;
|
|
}
|
|
}
|
|
|
|
function getOriginFaviconUrl(pageUrl: string): string {
|
|
try {
|
|
const url = new URL(pageUrl);
|
|
|
|
return new URL(FALLBACK_ICON_PATH, url.origin).toString();
|
|
} catch {
|
|
return FALLBACK_ICON_PATH;
|
|
}
|
|
}
|
|
|
|
function parseAttributes(tag: string): Record<string, string> {
|
|
const attrs: Record<string, string> = {};
|
|
const re = /([a-zA-Z_:.-]+)\s*=\s*("([^"]*)"|'([^']*)'|([^\s"'>]+))/g;
|
|
let m: RegExpExecArray | null;
|
|
|
|
while ((m = re.exec(tag)) !== null) {
|
|
const key = m[1].toLowerCase();
|
|
const value = m[3] ?? m[4] ?? m[5] ?? '';
|
|
|
|
attrs[key] = value;
|
|
}
|
|
|
|
return attrs;
|
|
}
|
|
|
|
function toAbsoluteUrl(value: string | undefined, base: string): string | null {
|
|
if (!value) return null;
|
|
|
|
try {
|
|
const url = new URL(value, base);
|
|
|
|
if (url.protocol !== 'http:' && url.protocol !== 'https:') return null;
|
|
|
|
return url.toString();
|
|
} catch {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
function extractIconFromHtml(html: string, baseUrl: string): string | null {
|
|
const linkTags = html.match(/<link\b[^>]*>/gi) ?? [];
|
|
const metaTags = html.match(/<meta\b[^>]*>/gi) ?? [];
|
|
|
|
const linkCandidates = linkTags
|
|
.map((tag) => parseAttributes(tag))
|
|
.map((a) => ({
|
|
rel: (a.rel ?? '').toLowerCase(),
|
|
href: toAbsoluteUrl(a.href, baseUrl),
|
|
}))
|
|
.filter((c) => c.href);
|
|
|
|
const appleTouchIcon = linkCandidates.find((c) =>
|
|
c.rel.includes('apple-touch-icon'),
|
|
)?.href;
|
|
|
|
if (appleTouchIcon) return appleTouchIcon;
|
|
|
|
const icon = linkCandidates.find((c) =>
|
|
c.rel.split(/\s+/).includes('icon'),
|
|
)?.href;
|
|
|
|
if (icon) return icon;
|
|
|
|
const shortcutIcon = linkCandidates.find((c) =>
|
|
c.rel.includes('shortcut icon'),
|
|
)?.href;
|
|
|
|
if (shortcutIcon) return shortcutIcon;
|
|
|
|
const imageSrc = linkCandidates.find((c) =>
|
|
c.rel.includes('image_src'),
|
|
)?.href;
|
|
|
|
if (imageSrc) return imageSrc;
|
|
|
|
const metaImage = metaTags
|
|
.map((tag) => parseAttributes(tag))
|
|
.map((a) => ({
|
|
property: (a.property ?? a.name ?? '').toLowerCase(),
|
|
content: toAbsoluteUrl(a.content, baseUrl),
|
|
}))
|
|
.find(
|
|
(c) =>
|
|
c.content &&
|
|
(c.property === 'og:image' ||
|
|
c.property === 'og:logo' ||
|
|
c.property === 'twitter:image'),
|
|
)?.content;
|
|
|
|
return metaImage ?? null;
|
|
}
|
|
|
|
/**
|
|
* Shared SSRF-guarded fetch used by every outbound request this service makes.
|
|
* Follows redirects manually (up to MAX_REDIRECTS) so each hop is re-validated
|
|
* against isPublicHttpUrl before being requested — a redirect target must not
|
|
* be able to bypass the private-IP/blocked-hostname guard.
|
|
*
|
|
* Returns the final non-redirect Response, or null if the guard blocks any
|
|
* hop, the request errors, or the redirect budget is exhausted.
|
|
*/
|
|
async function fetchWithRedirectGuard(
|
|
pageUrl: URL,
|
|
options: { accept: string; timeoutMs: number },
|
|
): Promise<{ response: Response; finalUrl: URL } | null> {
|
|
let currentUrl = pageUrl;
|
|
|
|
for (let redirectCount = 0; redirectCount <= MAX_REDIRECTS; redirectCount++) {
|
|
const isPublic = await isPublicHttpUrl(currentUrl);
|
|
|
|
if (!isPublic) return null;
|
|
|
|
const controller = new AbortController();
|
|
const timeout = setTimeout(() => controller.abort(), options.timeoutMs);
|
|
|
|
try {
|
|
const response = await fetch(currentUrl.toString(), {
|
|
redirect: 'manual', // SSRF: follow manually so each hop is re-validated
|
|
signal: controller.signal,
|
|
headers: {
|
|
Accept: options.accept,
|
|
'User-Agent': 'tessera/1.0',
|
|
},
|
|
});
|
|
|
|
if (response.status >= 300 && response.status < 400) {
|
|
const location = response.headers.get('location');
|
|
|
|
if (!location) return null;
|
|
|
|
currentUrl = new URL(location, currentUrl);
|
|
continue;
|
|
}
|
|
|
|
if (!response.ok) return null;
|
|
|
|
return { response, finalUrl: currentUrl };
|
|
} catch {
|
|
return null;
|
|
} finally {
|
|
clearTimeout(timeout);
|
|
}
|
|
}
|
|
|
|
return null;
|
|
}
|
|
|
|
async function fetchHtml(pageUrl: URL): Promise<FetchHtmlResult | null> {
|
|
const result = await fetchWithRedirectGuard(pageUrl, {
|
|
accept: 'text/html,application/xhtml+xml,*/*',
|
|
timeoutMs: HTML_FETCH_TIMEOUT_MS,
|
|
});
|
|
|
|
if (!result) return null;
|
|
|
|
const contentType = result.response.headers.get('content-type') ?? '';
|
|
|
|
if (!contentType.toLowerCase().includes('text/html')) return null;
|
|
|
|
const html = await result.response.text();
|
|
|
|
return {
|
|
html: html.slice(0, MAX_HTML_CHARS), // T-08-09: HTML cap
|
|
finalUrl: result.finalUrl.toString(),
|
|
};
|
|
}
|
|
|
|
@Injectable()
|
|
export class IconDiscoveryService {
|
|
/**
|
|
* Discover the best icon URL for a given web page URL.
|
|
* Falls back to <origin>/favicon.ico when discovery fails or URL is private.
|
|
*
|
|
* SSRF protection: every URL and redirect target is validated against
|
|
* private IP ranges, blocked hostnames, and forced-proxy vectors (T-08-05).
|
|
*/
|
|
async discoverFavoriteIconUrl(pageUrl: string): Promise<string> {
|
|
const fallback = getOriginFaviconUrl(pageUrl);
|
|
|
|
try {
|
|
const url = new URL(pageUrl);
|
|
const htmlResult = await fetchHtml(url);
|
|
|
|
if (!htmlResult) return fallback;
|
|
|
|
return extractIconFromHtml(htmlResult.html, htmlResult.finalUrl) ?? fallback;
|
|
} catch {
|
|
return fallback;
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Fetch the raw bytes of a stored icon URL, SSRF-guarded, for streaming
|
|
* back to the browser from Tessera's own origin (avoids Cross-Origin-
|
|
* Resource-Policy blocks on hotlinked cross-origin <img> loads).
|
|
*
|
|
* Throws on any failure — blocked target, timeout, non-image content-type,
|
|
* or an oversized body. Callers must not return a placeholder image; let
|
|
* the caller map the failure to an HTTP error status instead.
|
|
*/
|
|
async fetchIconBytes(
|
|
iconUrl: string,
|
|
): Promise<{ contentType: string; body: Buffer }> {
|
|
const url = new URL(iconUrl);
|
|
|
|
const result = await fetchWithRedirectGuard(url, {
|
|
// A bare "image/*" Accept header (paired with our non-browser
|
|
// User-Agent) trips bot-mitigation WAFs on some sites (observed:
|
|
// chatgpt.com/favicon.ico returns 403 with this combo) — a realistic
|
|
// browser-style image Accept list avoids that false positive.
|
|
accept: 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8',
|
|
timeoutMs: ICON_FETCH_TIMEOUT_MS,
|
|
});
|
|
|
|
if (!result) {
|
|
throw new Error('Icon fetch blocked or failed');
|
|
}
|
|
|
|
const contentType = result.response.headers.get('content-type') ?? '';
|
|
|
|
if (!contentType.toLowerCase().startsWith('image/')) {
|
|
throw new Error(`Icon response is not an image (${contentType})`);
|
|
}
|
|
|
|
const arrayBuffer = await result.response.arrayBuffer();
|
|
|
|
if (arrayBuffer.byteLength > MAX_ICON_BYTES) {
|
|
throw new Error('Icon response exceeds size limit');
|
|
}
|
|
|
|
return { contentType, body: Buffer.from(arrayBuffer) };
|
|
}
|
|
}
|