// Lightweight, dependency-free site reading shared by the chat agents // (onboarding + SAM): discover URLs from the sitemap (falling back to the // homepage) and extract readable text from each page via plain fetch. This is // enough to let the model infer what a site does. JS-heavy sites degrade // gracefully (less text); a Browser Rendering upgrade can slot in behind this // same interface later. import { normalizeAndValidateStartUrl } from "@/server/lib/audit/url-policy"; export const MAX_PAGES = 5; const PER_PAGE_CHAR_LIMIT = 4000; const FETCH_TIMEOUT_MS = 10_000; const MAX_RESPONSE_BYTES = 2_000_000; const USER_AGENT = "OpenSEO-Onboarding/1.0 (+https://openseo.so)"; type ScrapedPage = { url: string; title: string | null; text: string; }; type SiteReadResult = { pages: ScrapedPage[]; /** True when we couldn't read any page (blocked, offline, etc.). */ blocked: boolean; }; // Bounded read: accumulate up to MAX_RESPONSE_BYTES regardless of whether // content-length is present (chunked / CDN responses often omit it). async function readBoundedText(response: Response): Promise { const reader = response.body?.getReader(); if (!reader) return null; const decoder = new TextDecoder(); let result = ""; let bytesRead = 0; while (true) { const { done, value } = await reader.read(); if (done) break; bytesRead += value.byteLength; if (bytesRead > MAX_RESPONSE_BYTES) { await reader.cancel(); return null; } result += decoder.decode(value, { stream: true }); } result += decoder.decode(); return result; } async function fetchText(url: string): Promise { try { const response = await fetch(url, { headers: { "user-agent": USER_AGENT, accept: "text/html,*/*" }, // Manual redirects so we can re-validate each hop against the SSRF guard; // redirect:"follow" would let a 30x to an internal host bypass it. redirect: "manual", signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), }); if (response.status >= 300 && response.status < 400) { const location = response.headers.get("location"); if (!location) return null; let redirectUrl: string; try { // Re-validates host, blocks private IPs, and does DoH DNS resolution. redirectUrl = await normalizeAndValidateStartUrl( new URL(location, url).toString(), ); } catch { return null; // blocked or invalid redirect destination } // One hop only; fetch the validated destination without following further. const redirected = await fetch(redirectUrl, { headers: { "user-agent": USER_AGENT, accept: "text/html,*/*" }, redirect: "manual", signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), }); if (!redirected.ok) return null; return await readBoundedText(redirected); } if (!response.ok) return null; return await readBoundedText(response); } catch { return null; } } /** * Pulls page URLs from a sitemap body. Resolves relative/protocol-relative * entries against the origin and keeps same-origin HTML pages only. Nested * sitemap files (a sitemap index) are skipped rather than fetched as pages — * good enough for v1; we fall back to the homepage if nothing usable is found. */ function parseSitemapUrls(xml: string, origin: string): string[] { const urls: string[] = []; const regex = /\s*([^<\s]+)\s*<\/loc>/gi; let match: RegExpExecArray | null; while ((match = regex.exec(xml)) !== null) { let resolved: string; try { resolved = new URL(match[1], origin).toString(); } catch { continue; } if (resolved.startsWith(origin) && !resolved.endsWith(".xml")) { urls.push(resolved); } } return urls; } function extractTitle(html: string): string | null { const match = /]*>([^<]*)<\/title>/i.exec(html); return match ? decodeEntities(match[1].trim()) : null; } /** Strips scripts/styles/tags and collapses whitespace into readable text. */ function htmlToText(html: string): string { const withoutBlocks = html .replace(/]*>[\s\S]*?<\/script>/gi, " ") .replace(/]*>[\s\S]*?<\/style>/gi, " ") .replace(/]*>[\s\S]*?<\/noscript>/gi, " ") .replace(//g, " "); const text = withoutBlocks .replace(/<[^>]+>/g, " ") .replace(/\s+/g, " ") .trim(); return decodeEntities(text); } function decodeEntities(value: string): string { return value .replace(/&/g, "&") .replace(/</g, "<") .replace(/>/g, ">") .replace(/"/g, '"') .replace(/'/g, "'") .replace(/ /g, " "); } /** Fetches one (already-validated) URL and shapes it as a page, or null if it * couldn't be read or yielded no text. */ async function scrapePage(url: string): Promise { const html = await fetchText(url); if (!html) { return null; } const text = htmlToText(html).slice(0, PER_PAGE_CHAR_LIMIT); if (text.length === 0) { return null; } return { url, title: extractTitle(html), text }; } /** * Reads a specific list of page URLs as plain text — used when the caller names * exact pages (the user's own or a competitor's) rather than asking us to * discover a site. Each URL is independently run through the SSRF guard, so a * blocked or unreachable URL is skipped rather than failing the batch. */ export async function readPages( urls: string[], maxPages: number = MAX_PAGES, ): Promise { const pages: ScrapedPage[] = []; for (const rawUrl of urls.slice(0, maxPages)) { let url: string; try { // Re-validates host, blocks private/metadata IPs, does DoH DNS resolution. url = await normalizeAndValidateStartUrl(rawUrl); } catch { continue; // blocked or unparseable URL } const page = await scrapePage(url); if (page) { pages.push(page); } } return { pages, blocked: pages.length === 0 }; } /** * Lists a site's page URLs without reading them: the homepage plus what the * sitemap declares, capped at `limit`. Lets an agent see what a site has and * choose which pages to read (readPages) instead of blindly taking the first N. */ export async function discoverSiteUrls( domain: string, limit: number, ): Promise<{ urls: string[]; blocked: boolean }> { let rootUrl: string; try { rootUrl = await normalizeAndValidateStartUrl(domain); } catch { // Blocked (private/metadata host) or unparseable domain — nothing to list. return { urls: [], blocked: true }; } const origin = new URL(rootUrl).origin; const sitemap = await fetchText(`${origin}/sitemap.xml`); const discovered = sitemap ? parseSitemapUrls(sitemap, origin) : []; const urls = [rootUrl, ...discovered.filter((url) => url !== rootUrl)]; return { urls: urls.slice(0, limit), blocked: false }; } /** * Discovers a site's representative URLs (homepage + sitemap) and reads them. * Just URL discovery on top of readPages, which does the validated fetching. */ export async function readSite( domain: string, maxPages: number = MAX_PAGES, ): Promise { const discovered = await discoverSiteUrls(domain, maxPages); if (discovered.blocked) { return { pages: [], blocked: true }; } return readPages(discovered.urls, maxPages); }