import type { WorkflowStep } from "cloudflare:workers"; import type { BillingCustomerContext } from "@/server/billing/subscription"; import { discoverUrls, parseRobotsTxt } from "@/server/lib/audit/discovery"; import { fetchLighthouseResult, selectLighthouseSample, storeLighthouseResult, } from "@/server/lib/audit/lighthouse"; import { getOrigin, isSameOrigin, normalizeUrl, } from "@/server/lib/audit/url-utils"; import { isCrawlableUrl } from "@/server/lib/audit/url-policy"; import { AuditRepository } from "@/server/features/audit/repositories/AuditRepository"; import { getAuditScratchpad } from "@/server/features/audit/AuditScratchpad"; import { AuditProgressKV } from "@/server/lib/audit/progress-kv"; import { runMultipageChecks } from "@/server/lib/audit/issues/multipage"; import type { DetectedIssue } from "@/server/lib/audit/issues/page-reporters"; import type { AuditConfig } from "@/server/lib/audit/types"; import { captureServerEvent } from "@/server/lib/posthog"; import { runCrawlPhase, type CrawlPhaseResult, } from "@/server/workflows/siteAuditWorkflowCrawl"; import { pgStep } from "@/server/workflows/pgStep"; import { DB_STEP, DISCOVERY_STEP, LIGHTHOUSE_FETCH_STEP, LIGHTHOUSE_PERSIST_STEP, MULTIPAGE_CHECKS_STEP, } from "@/server/workflows/auditStepConfigs"; const LEGACY_LIGHTHOUSE_URL_BATCH_SIZE = 10; /** Frontier seeds per scratchpad RPC call. */ const SEED_RPC_BATCH = 2_000; type LighthouseBatchBoundary = | { schema: "retry-safe-v2" } | { completed: number; failed: number }; type AuditPhasesParams = { auditId: string; workflowInstanceId: string; billingCustomer: BillingCustomerContext; projectId: string; startUrl: string; config: AuditConfig; }; export async function runAuditPhases( step: WorkflowStep, params: AuditPhasesParams, ) { const { auditId, workflowInstanceId, billingCustomer, projectId, startUrl, config, } = params; const origin = getOrigin(startUrl); const maxPages = config.maxPages; const discovery = await runDiscoveryPhase(step, { auditId, workflowInstanceId, origin, startUrl, maxPages, }); // Parsed outside the step from checkpointed text, so replays see the exact // robots rules the original run used (a live re-fetch could differ and // desync the frontier from already-persisted crawl batches). const robots = parseRobotsTxt(origin, discovery.robotsText); const crawl = await runCrawlPhase(step, { auditId, workflowInstanceId, origin, maxPages, robots, seededCount: discovery.seededCount, }); await runLighthousePhase(step, { auditId, workflowInstanceId, billingCustomer, projectId, startUrl, config, }); await finalizeAudit({ step, auditId, workflowInstanceId, billingCustomer, projectId, startUrl, config, crawl, }); } async function runDiscoveryPhase( step: WorkflowStep, input: { auditId: string; workflowInstanceId: string; origin: string; startUrl: string; maxPages: number; }, ) { const { auditId, workflowInstanceId, origin, startUrl, maxPages } = input; // "-v2": the checkpoint shape changed (seeds now live in the scratchpad DO // instead of the step return). A pre-refactor instance replayed under this // code must re-run discovery — resuming from the old cached {sitemapUrls} // shape would leave the scratchpad empty and finalize a zero-page audit. return pgStep(step, "discover-urls-v2", DISCOVERY_STEP, async () => { const result = await discoverUrls(origin, maxPages); const robots = parseRobotsTxt(origin, result.robotsText); const scratchpad = getAuditScratchpad(auditId); // Seeds go straight into the scratchpad frontier — nothing large is // returned as step state (an uncapped seed list used to blow the ~1MiB // step-output limit on big sitemaps). let seededCount = 0; const normalizedStart = normalizeUrl(startUrl) ?? startUrl; if ( robots.isAllowed(normalizedStart) && isSameOrigin(normalizedStart, origin) ) { await scratchpad.seedStart(normalizedStart); seededCount += 1; } // The start URL is deliberately not excluded here: seedSitemapUrls // upserts, so a start URL that also appears in the sitemap keeps its // link-queue position but gains the in-sitemap flag. const seen = new Set(); const seeds: string[] = []; for (const url of result.urls) { const normalized = normalizeUrl(url); if (!normalized || seen.has(normalized)) continue; seen.add(normalized); if (!isSameOrigin(normalized, origin)) continue; if (!isCrawlableUrl(normalized)) continue; if (!robots.isAllowed(normalized)) continue; seeds.push(normalized); } for (let i = 0; i < seeds.length; i += SEED_RPC_BATCH) { await scratchpad.seedSitemapUrls(seeds.slice(i, i + SEED_RPC_BATCH)); } seededCount += seeds.filter((seed) => seed !== normalizedStart).length; await AuditRepository.updateAuditProgress(auditId, workflowInstanceId, { pagesTotal: Math.min(seededCount, maxPages), currentPhase: "crawling", }); return { robotsText: result.robotsText, seededCount }; }); } type LighthousePhaseParams = { auditId: string; workflowInstanceId: string; billingCustomer: BillingCustomerContext; projectId: string; startUrl: string; config: AuditConfig; }; export async function runLighthousePhase( step: WorkflowStep, params: LighthousePhaseParams, ) { const { auditId, workflowInstanceId, billingCustomer, projectId, startUrl, config, } = params; if (config.lighthouseStrategy === "none") return; const lighthouseWork = await selectLighthousePages({ step, auditId, workflowInstanceId, startUrl, strategy: config.lighthouseStrategy, }); let completedChecks = 0; let failedChecks = 0; for ( let batchStart = 0; batchStart < lighthouseWork.length; batchStart += LEGACY_LIGHTHOUSE_URL_BATCH_SIZE ) { const batchIndex = Math.floor( batchStart / LEGACY_LIGHTHOUSE_URL_BATCH_SIZE, ); const boundary = await pgStep( step, `lighthouse-batch-${batchIndex + 1}`, DB_STEP, async (): Promise => ({ schema: "retry-safe-v2", }), ); // Older deployments used this checkpoint name for a complete paid batch. // If that cached shape replays under current code, all results and progress // are already persisted; skip the batch instead of buying it again. if ("completed" in boundary) { completedChecks += boundary.completed; failedChecks += boundary.failed; continue; } const batch = lighthouseWork.slice( batchStart, batchStart + LEGACY_LIGHTHOUSE_URL_BATCH_SIZE, ); for (const [batchOffset, { url, pageId }] of batch.entries()) { const index = batchStart + batchOffset; // The paid calls are checkpointed separately from all storage. With // Workflow retries disabled, a later R2/DB/progress failure cannot replay // DataForSEO. One URL groups its mobile + desktop checks into one compact // checkpoint rather than returning a whole Lighthouse batch. const fetched = await pgStep( step, `lighthouse-fetch-${index + 1}`, LIGHTHOUSE_FETCH_STEP, () => Promise.all([ fetchLighthouseResult(url, pageId, "mobile", billingCustomer), fetchLighthouseResult(url, pageId, "desktop", billingCustomer), ]), ); const priorCompleted = completedChecks; const priorFailed = failedChecks; const counts = await pgStep( step, `lighthouse-persist-${index + 1}`, LIGHTHOUSE_PERSIST_STEP, async () => { const results = await Promise.all( fetched.map((result) => storeLighthouseResult({ projectId, auditId, fetched: result, }), ), ); await AuditRepository.insertLighthouseResults(auditId, results); const failed = results.filter((result) => result.errorMessage).length; const completed = results.length - failed; await AuditRepository.updateAuditProgress( auditId, workflowInstanceId, { lighthouseCompleted: priorCompleted + completed, lighthouseFailed: priorFailed + failed, }, ); return { completed, failed }; }, ); completedChecks += counts.completed; failedChecks += counts.failed; } } } async function selectLighthousePages(params: { step: WorkflowStep; auditId: string; workflowInstanceId: string; startUrl: string; strategy: AuditConfig["lighthouseStrategy"]; }) { const { step, auditId, workflowInstanceId, startUrl, strategy } = params; return pgStep(step, "select-lighthouse-sample", DB_STEP, async () => { // Crawled pages come from the DB — the crawl phase no longer holds a // whole-crawl page list in memory. const crawledPages = await AuditRepository.getPagesForAudit(auditId); const sample = selectLighthouseSample( crawledPages.map((page) => ({ url: page.url, statusCode: page.statusCode ?? 0, })), startUrl, strategy, ); const selectedUrls = new Set(sample); await AuditRepository.updateAuditProgress(auditId, workflowInstanceId, { currentPhase: "lighthouse", lighthouseTotal: sample.length * 2, lighthouseCompleted: 0, lighthouseFailed: 0, }); return crawledPages.flatMap((page) => selectedUrls.has(page.url) ? [{ url: page.url, pageId: page.id }] : [], ); }); } async function finalizeAudit(args: { step: WorkflowStep; auditId: string; workflowInstanceId: string; billingCustomer: BillingCustomerContext; projectId: string; startUrl: string; config: AuditConfig; crawl: CrawlPhaseResult; }) { const { step, auditId, workflowInstanceId, billingCustomer, projectId, startUrl, config, crawl, } = args; await pgStep(step, "multipage-checks", MULTIPAGE_CHECKS_STEP, async () => { await AuditRepository.updateAuditProgress(auditId, workflowInstanceId, { currentPhase: "finalizing", }); // Integrity guard: pages are persisted inside crawl-chunk steps. If the // crawl claims pages but the DB has none, fail loudly instead of // completing with an empty audit. if ( crawl.pagesCrawled > 0 && !(await AuditRepository.hasPagesForAudit(auditId)) ) { throw new Error( `Audit ${auditId}: crawl reported ${crawl.pagesCrawled} pages but none were persisted`, ); } const issues = await runMultipageChecks({ auditId }); issues.push(...(await runScratchpadLinkChecks(auditId, startUrl, crawl))); await AuditRepository.insertIssues(auditId, issues); return { issueCount: issues.length }; }); await pgStep(step, "finalize", DB_STEP, async () => { const blockedPages = await AuditRepository.countBlockedPages(auditId); await AuditRepository.completeAudit(auditId, workflowInstanceId, { pagesCrawled: crawl.pagesCrawled, pagesTotal: crawl.pagesCrawled, }); await captureServerEvent({ distinctId: billingCustomer.userId, event: "site_audit:complete", organizationId: billingCustomer.organizationId, properties: { project_id: projectId, status: "completed", pages_crawled: crawl.pagesCrawled, pages_total: crawl.pagesCrawled, crawl_completed: crawl.completed, pages_blocked: blockedPages, run_lighthouse: config.lighthouseStrategy !== "none", }, }); await AuditProgressKV.clear(auditId); // Crawl scratch state (frontier, links, mirror) is no longer needed. await getAuditScratchpad(auditId).destroy(); }); } /** * The two finalize checks that need link edges run as SQL inside the * audit's scratchpad DO; map their rows onto DetectedIssue. */ async function runScratchpadLinkChecks( auditId: string, startUrl: string, crawl: CrawlPhaseResult, ): Promise { const scratchpad = getAuditScratchpad(auditId); const { brokenLinks, orphanPages } = await scratchpad.runFinalizeChecks({ // Page rows store normalized URLs; normalize the start URL the same way // so the orphan exclusion matches. startUrl: normalizeUrl(startUrl) ?? startUrl, // Orphan detection only makes sense when the crawl wasn't truncated. crawlCompleted: crawl.completed, }); return [ ...brokenLinks.map((row) => ({ issueType: "broken-internal-link" as const, pageId: row.sourcePageId, pageUrl: row.sourceUrl, dedupeKey: row.targetUrl, details: { targetUrl: row.targetUrl, targetStatus: row.targetStatus }, })), ...orphanPages.map((row) => ({ issueType: "orphan-page" as const, pageId: row.pageId, pageUrl: row.url, })), ]; }