Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 61 additions & 0 deletions lib/audit/links-crawl-child.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
// Child-process entry for the link crawl. Forked by links-engine.ts.
//
// Contract: argv[2] is the target URL, argv[3] an optional per-link timeout in
// ms (tests use a short one to make the abort race deterministic). We print
// exactly one JSON object to stdout and exit 0. Anything else on stdout would
// corrupt the parse, so all diagnostics go to stderr (which the parent forwards
// to the worker log).
//
// This process is expendable by design — see the comment in links-crawl.ts for
// why linkinator can hard-exit the process it runs in. Both exit paths salvage
// whatever the accumulator holds, so a crash 200 pages into a crawl still
// reports those 200 pages instead of losing the audit.

import { crawlLinks, newAccumulator, type CrawlAccumulator } from "./links-crawl";

export type ChildReport = {
acc: CrawlAccumulator;
/** Set when the crawl ended early; the parent turns this into a warn finding. */
crashed: string | null;
};

const acc = newAccumulator();
let emitted = false;

function emit(crashed: string | null): void {
if (emitted) return; // an uncaught error while shutting down must not double-print
emitted = true;
const report: ChildReport = { acc, crashed };
process.stdout.write(JSON.stringify(report), () => process.exit(0));
// linkinator leaves sockets open, so the natural exit may never come; and if
// stdout's drain callback is itself lost, exit anyway rather than hang the
// parent until its kill timer fires.
setTimeout(() => process.exit(0), 2_000).unref();
}

// An unhandled 'error' event on a Readable arrives here as an uncaught
// exception. This handler is the whole point of the child process: catch it,
// keep the partial crawl, and leave the parent worker untouched.
process.on("uncaughtException", (err: unknown) => {
const message = err instanceof Error ? err.message : String(err);
console.error(`[links-crawl] uncaught in crawl child: ${message}`);
emit(message);
});
process.on("unhandledRejection", (err: unknown) => {
const message = err instanceof Error ? err.message : String(err);
console.error(`[links-crawl] unhandled rejection in crawl child: ${message}`);
emit(message);
});

const target = process.argv[2];
if (!target) {
console.error("[links-crawl] no target URL argument");
process.exit(2);
}

const perLinkTimeoutMs = Number(process.argv[3]) || undefined;

crawlLinks(target, acc, { perLinkTimeoutMs }).then(
() => emit(null),
(err: unknown) => emit(err instanceof Error ? err.message : String(err)),
);
121 changes: 121 additions & 0 deletions lib/audit/links-crawl.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,121 @@
// The raw linkinator crawl, split out from links-engine.ts so it can run in a
// disposable child process.
//
// Why a child process: linkinator applies its per-link `timeout` as an
// AbortSignal.timeout on the fetch, converts the response body with
// Readable.fromWeb(), then pipes it to the HTML parser with the 'error' handler
// attached to the *destination* (build/src/links.js:158). pipe() does not
// forward source errors, so when the abort fires mid-body the source Readable
// emits an unhandled 'error' event and Node hard-exits the process. That is not
// catchable from linksAudit()'s try/catch, and because start.sh runs the worker
// and Next.js under `wait -n`, it took the whole container down with every
// in-flight audit. Run it somewhere we can afford to lose.
//
// Crawl bounds live here too: linkinator has no built-in page cap or
// AbortSignal, so we bound the crawl with its `linksToSkip` extension point.
// Once we exceed a page / link / wall-clock budget the predicate returns true
// for every remaining link, which stops both checking and recursion. This keeps
// a runaway crawl well under the worker's 7-minute stuck-audit cutoff.

import { LinkChecker, LinkState } from "linkinator";

const UA = "CrawlProofBot/1.0 (+https://crawlproof.com/bot)";

// Budgets — generous enough for a typical CrawlProof property, hard-capped so
// the worker can't blow past the 7-minute stuck-sweep cutoff.
export const MAX_PAGES = 250; // distinct internal pages we recurse into
export const MAX_LINKS = 5000; // total links checked (internal + external)
export const DEADLINE_MS = 4 * 60 * 1000; // wall-clock crawl budget
export const PER_LINK_TIMEOUT_MS = 10_000;
const CONCURRENCY = 25;

export type Capped = null | "pages" | "links" | "time";

/** A linkinator LinkResult, reduced to what survives JSON round-tripping. */
export type CrawlLink = {
url: string;
status: number;
state: string;
parent: string | null;
};

/**
* Filled in as the crawl runs rather than read off the resolved return value,
* so a crash partway through still leaves usable results behind.
*/
export type CrawlAccumulator = {
pagesCrawled: number;
capped: Capped;
links: CrawlLink[];
};

export function newAccumulator(): CrawlAccumulator {
return { pagesCrawled: 0, capped: null, links: [] };
}

export function rootOf(targetUrl: string): string {
// Crawl from the root domain, not the submitted deep link — the user asked
// the bot to sweep the whole property, not just one page.
const u = new URL(targetUrl);
return `${u.protocol}//${u.host}/`;
}

export async function crawlLinks(
targetUrl: string,
acc: CrawlAccumulator,
opts: { perLinkTimeoutMs?: number } = {},
): Promise<void> {
const started = Date.now();
const root = rootOf(targetUrl);
const checker = new LinkChecker();

checker.on("pagestart", () => {
acc.pagesCrawled++;
});

// Accumulate incrementally — `check()`'s return value is unreachable if the
// crawl dies partway through.
checker.on("link", (link: { url: string; status?: number; state: string; parent?: string }) => {
acc.links.push({
url: link.url,
status: link.status ?? 0,
state: link.state,
parent: link.parent ?? null,
});
});

// Doubles as the crawl's kill-switch: returning true marks a link SKIPPED,
// which also prevents linkinator from recursing into it.
const linksToSkip = async (link: string): Promise<boolean> => {
// linkinator already skips non-http(s) schemes, but guard anyway.
if (!/^https?:\/\//i.test(link)) return true;
if (Date.now() - started > DEADLINE_MS) {
acc.capped ??= "time";
return true;
}
if (acc.pagesCrawled >= MAX_PAGES) {
acc.capped ??= "pages";
return true;
}
if (checkedCount(acc) >= MAX_LINKS) {
acc.capped ??= "links";
return true;
}
return false;
};

await checker.check({
path: root,
recurse: true,
concurrency: CONCURRENCY,
timeout: opts.perLinkTimeoutMs ?? PER_LINK_TIMEOUT_MS,
userAgent: UA,
retry: true,
linksToSkip,
});
}

/** Links we actually hit the network for (SKIPPED ones were never fetched). */
export function checkedCount(acc: CrawlAccumulator): number {
return acc.links.filter((l) => l.state !== LinkState.SKIPPED).length;
}
Loading
Loading