metatroncubeswdev 17dfca406f
Some checks failed
CI / ci (push) Has been cancelled
CI / docker-build (push) Has been cancelled
Publish Docker image / docker (push) Has been cancelled
Upload sourcemaps / upload (push) Has been cancelled
Audit crawler: send a browser navigation header set (tier-1 anti-bot)
The crawler identified as `OpenSEO-Audit/1.0` with almost no headers, which
naive bot filters and security plugins block outright.

- New crawl-request.ts: AUDIT_USER_AGENT (a current Chrome string),
  buildAuditHeaders() (Accept, Accept-Language, Sec-Fetch-*, Sec-Ch-Ua,
  Upgrade-Insecure-Requests), and fetchForAudit() — fetch + those headers +
  one retry on a transient 429/503.
- Wired into the page crawl (site-audit-workflow-helpers), robots.txt +
  sitemap discovery, and start-URL redirect probing.

Gets past the naive tier; still reported as "blocked" for JS/TLS challenges
(Cloudflare Managed Challenge, DataDome) — those need a real browser. Doc note
points operators at WAF IP/UA allowlisting for their own sites.

No env dependency (keeps the audit lib importable without a cloudflare:workers
mock). tsc / oxlint / knip clean; new crawl-request.test.ts (5); suite
otherwise unchanged.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-10 16:35:50 -04:00

287 lines
7.4 KiB
TypeScript

import { AppError } from "@/server/lib/errors";
import { buildAuditHeaders } from "./crawl-request";
const BLOCKED_HOSTS = new Set([
"localhost",
"metadata.google.internal",
"metadata",
"169.254.169.254",
"100.100.100.200",
]);
const BLOCKED_HOST_SUFFIXES = [
".localhost",
".local",
".localdomain",
".internal",
".home.arpa",
];
const DOH_ENDPOINT = "https://cloudflare-dns.com/dns-query";
function normalizeHost(hostname: string): string {
let host = hostname.toLowerCase().trim();
if (host.startsWith("[") && host.endsWith("]")) {
host = host.slice(1, -1);
}
if (host.includes("%")) {
host = host.split("%", 1)[0];
}
if (host.endsWith(".")) {
host = host.slice(0, -1);
}
return host;
}
function isPrivateIpv4(host: string): boolean {
const parts = normalizeHost(host)
.split(".")
.map((x) => Number(x));
if (
parts.length !== 4 ||
parts.some((x) => !Number.isInteger(x) || x < 0 || x > 255)
) {
return false;
}
const [a, b] = parts;
if (a === 10) return true;
if (a === 127) return true;
if (a === 0) return true;
if (a === 169 && b === 254) return true;
if (a === 172 && b >= 16 && b <= 31) return true;
if (a === 192 && b === 168) return true;
if (a === 100 && b >= 64 && b <= 127) return true;
if (a === 198 && (b === 18 || b === 19)) return true;
if (a >= 224) return true;
return false;
}
function parseMappedIpv4FromIpv6(host: string): string | null {
const normalized = normalizeHost(host);
if (!normalized.startsWith("::ffff:")) return null;
const mapped = normalized.slice("::ffff:".length);
if (/^\d+\.\d+\.\d+\.\d+$/.test(mapped)) {
return mapped;
}
const segments = mapped.split(":").filter(Boolean);
if (segments.length !== 2) return null;
const high = Number.parseInt(segments[0], 16);
const low = Number.parseInt(segments[1], 16);
if (
!Number.isFinite(high) ||
!Number.isFinite(low) ||
high < 0 ||
high > 0xffff ||
low < 0 ||
low > 0xffff
) {
return null;
}
const a = (high >> 8) & 0xff;
const b = high & 0xff;
const c = (low >> 8) & 0xff;
const d = low & 0xff;
return `${a}.${b}.${c}.${d}`;
}
function isPrivateIpv6(host: string): boolean {
const value = normalizeHost(host);
if (value === "::1" || value === "::") return true;
if (value.startsWith("fc") || value.startsWith("fd")) return true;
if (
value.startsWith("fe8") ||
value.startsWith("fe9") ||
value.startsWith("fea") ||
value.startsWith("feb")
) {
return true;
}
const mappedIpv4 = parseMappedIpv4FromIpv6(value);
if (mappedIpv4 && isPrivateIpv4(mappedIpv4)) {
return true;
}
return false;
}
function isIpLiteral(host: string): boolean {
const normalized = normalizeHost(host);
return /^\d+\.\d+\.\d+\.\d+$/.test(normalized) || normalized.includes(":");
}
function isBlockedHost(hostname: string): boolean {
const host = normalizeHost(hostname);
if (!host) return true;
if (BLOCKED_HOSTS.has(host)) return true;
if (BLOCKED_HOST_SUFFIXES.some((suffix) => host.endsWith(suffix))) {
return true;
}
if (isIpLiteral(host)) {
return isPrivateIpv4(host) || isPrivateIpv6(host);
}
return false;
}
type DnsJsonAnswer = {
type?: number;
data?: string;
};
type DnsJsonResponse = {
Status?: number;
Answer?: DnsJsonAnswer[];
};
async function resolveAddressRecords(
hostname: string,
type: "A" | "AAAA",
): Promise<string[]> {
const response = await fetch(
`${DOH_ENDPOINT}?name=${encodeURIComponent(hostname)}&type=${type}`,
{
headers: { Accept: "application/dns-json" },
signal: AbortSignal.timeout(2_500),
},
);
if (!response.ok) return [];
const body: DnsJsonResponse = await response.json();
if (body.Status !== 0 || !Array.isArray(body.Answer)) return [];
const expectedType = type === "A" ? 1 : 28;
return body.Answer.filter(
(answer): answer is Required<Pick<DnsJsonAnswer, "data" | "type">> =>
answer.type === expectedType && typeof answer.data === "string",
).map((answer) => normalizeHost(answer.data));
}
async function hostnameResolvesToBlockedAddress(
hostname: string,
): Promise<boolean> {
const host = normalizeHost(hostname);
if (!host || isIpLiteral(host)) return false;
try {
const [v4, v6] = await Promise.all([
resolveAddressRecords(host, "A"),
resolveAddressRecords(host, "AAAA"),
]);
const addresses = [...v4, ...v6];
if (addresses.length === 0) return false;
return addresses.some(
(address) => isPrivateIpv4(address) || isPrivateIpv6(address),
);
} catch {
return false;
}
}
/**
* Synchronous SSRF check for URLs discovered mid-crawl (links, redirect
* targets, sitemap entries). Blocks non-http(s) schemes, private/loopback IP
* literals, and internal hostnames. DNS resolution is only performed for the
* start URL (see normalizeAndValidateStartUrl); per-link DoH lookups would be
* prohibitively slow.
*/
export function isCrawlableUrl(url: string): boolean {
let parsed: URL;
try {
parsed = new URL(url);
} catch {
return false;
}
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") {
return false;
}
return !isBlockedHost(parsed.hostname);
}
export async function normalizeAndValidateStartUrl(
input: string,
): Promise<string> {
let raw = input.trim();
if (!raw) throw new AppError("VALIDATION_ERROR");
if (!raw.startsWith("http://") && !raw.startsWith("https://")) {
raw = `https://${raw}`;
}
let parsed: URL;
try {
parsed = new URL(raw);
} catch {
throw new AppError("VALIDATION_ERROR");
}
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") {
throw new AppError("VALIDATION_ERROR");
}
if (isBlockedHost(parsed.hostname)) {
throw new AppError("CRAWL_TARGET_BLOCKED");
}
if (await hostnameResolvesToBlockedAddress(parsed.hostname)) {
throw new AppError("CRAWL_TARGET_BLOCKED");
}
parsed.hash = "";
return parsed.toString();
}
const START_URL_REDIRECT_HOPS = 5;
const START_URL_PROBE_TIMEOUT_MS = 10_000;
/**
* Follow redirects on the audit's start URL so the audit anchors to the
* site's real origin. Without this, auditing a domain that 301s elsewhere
* (…net -> …com, apex -> www) dead-ends after one page: the redirect target
* is a different origin, so the same-origin crawl policy can't follow it.
*
* Every hop re-runs the full start-URL validation (SSRF, blocked hosts), so
* a redirect can't smuggle the audit somewhere the user couldn't have
* pointed it directly. Probe failures (timeouts, HEAD rejected) fall back
* to the last validated URL — the crawl records the real fetch result.
*/
export async function resolveStartUrlRedirects(
startUrl: string,
): Promise<string> {
let current = startUrl;
for (let hop = 0; hop < START_URL_REDIRECT_HOPS; hop++) {
let response: Response;
try {
response = await fetch(current, {
method: "HEAD",
redirect: "manual",
headers: buildAuditHeaders(),
signal: AbortSignal.timeout(START_URL_PROBE_TIMEOUT_MS),
});
} catch {
return current;
}
if (response.status < 300 || response.status >= 400) return current;
const location = response.headers.get("location");
if (!location) return current;
let next: URL;
try {
next = new URL(location, current);
} catch {
return current;
}
current = await normalizeAndValidateStartUrl(next.toString());
}
return current;
}