diff --git a/apps/api/src/favorites/icon-discovery.service.ts b/apps/api/src/favorites/icon-discovery.service.ts index 83ecff3..b73f6e8 100644 --- a/apps/api/src/favorites/icon-discovery.service.ts +++ b/apps/api/src/favorites/icon-discovery.service.ts @@ -227,7 +227,7 @@ function extractIconFromHtml(html: string, baseUrl: string): string | null { */ async function fetchWithRedirectGuard( pageUrl: URL, - options: { accept: string; timeoutMs: number }, + options: { accept: string; timeoutMs: number; userAgent?: string }, ): Promise<{ response: Response; finalUrl: URL } | null> { let currentUrl = pageUrl; @@ -245,7 +245,7 @@ async function fetchWithRedirectGuard( signal: controller.signal, headers: { Accept: options.accept, - 'User-Agent': 'tessera/1.0', + 'User-Agent': options.userAgent ?? 'tessera/1.0', }, }); @@ -330,11 +330,15 @@ export class IconDiscoveryService { const url = new URL(iconUrl); const result = await fetchWithRedirectGuard(url, { - // A bare "image/*" Accept header (paired with our non-browser - // User-Agent) trips bot-mitigation WAFs on some sites (observed: - // chatgpt.com/favicon.ico returns 403 with this combo) — a realistic - // browser-style image Accept list avoids that false positive. + // A browser-realistic Accept header and User-Agent avoid tripping + // bot-mitigation WAFs that block non-browser clients (observed + // reproducibly: chatgpt.com/favicon.ico returns 403 for our default + // "tessera/1.0" UA and 200 for a real Chrome UA string, confirmed by + // repeated direct comparison). The HTML-discovery path is unaffected — + // this User-Agent override only applies to this byte-fetch call. accept: 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8', + userAgent: + 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36', timeoutMs: ICON_FETCH_TIMEOUT_MS, });