From 7cbb3ba8dc41799db6c8ee4cbd300d6100751e80 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 00:56:43 +0000 Subject: [PATCH] feat(leads): render seed pages in a browser, and follow listings one hop Seeding from a directory only worked when the directory was server-rendered. A growing share of the pages worth seeding ship an empty shell and load their listings over XHR, and fetching one of those returns HTTP 200 with no links -- which reads as "this directory has no businesses on it" rather than "we couldn't see them". Seeds now fall back to Chromium when a plain fetch is refused or comes back with nothing usable. Fetch still runs first because it is an order of magnitude cheaper and most directories don't need more. Rendering is deliberately generic: no per-directory API clients to write and rewrite as each site moves its endpoints. The second problem was structural. extractOutboundProspects drops same-host links as internal navigation, but a platform directory keeps every listing on its own domain -- so the businesses were being filtered out by design, and rendering alone would still have found nothing. Seeds can now take a second hop: open the listing entries and take the outbound site from each. That hop is gated on the first one coming up short. A listicle that already yielded a page of businesses has nothing to gain from opening its own internal links, and opening a dozen pages per seed would dominate a campaign tick. Seeds are user-supplied URLs and a JS-executing browser is a sharper tool than fetch, so seed loading now refuses hosts that resolve into private space -- including the cloud metadata endpoint, which was reachable before. Chromium was already in the production image, so this costs runtime memory rather than image size. Playwright moves to dependencies and is pinned to 1.60.0 to match the image it is launched from; floating it on the "next" tag would have drifted off that pin on any fresh install. It is also marked external so Next doesn't bundle a package that resolves a real binary through its own layout. Rendering does not defeat bot protection. A site behind a Cloudflare managed challenge stays blocked -- headless and headed Chromium both sit on the interstitial from a datacenter IP -- but the failure is now reported as a challenge instead of a bare HTTP 403. Co-Authored-By: Claude Opus 5 (1M context) --- lib/outreach/discover.ts | 191 ++++++++++++++++++++++++++++++++--- lib/outreach/render.ts | 191 +++++++++++++++++++++++++++++++++++ next.config.ts | 4 + package-lock.json | 4 +- package.json | 2 +- tests/seed-discovery.test.ts | 116 +++++++++++++++++++++ 6 files changed, 491 insertions(+), 17 deletions(-) create mode 100644 lib/outreach/render.ts create mode 100644 tests/seed-discovery.test.ts diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index 6c2b968b..86f30cf5 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -108,25 +108,187 @@ export function extractOutboundProspects(input: { return [...out.values()]; } -export async function discoverFromSeed(input: { - seedUrl: string; +/** + * Same-host links that look like an entry for one business rather than site + * furniture. + * + * On a platform directory — an artist marketplace, an agency roster — the + * listing page links to profiles on its own domain, and the business's real + * website only appears on the profile. Those links are invisible to + * `extractOutboundProspects`, which drops same-host hrefs as navigation, so + * they are collected separately for the second hop. + */ +export function extractSameHostLinks(input: { + html: string; + sourceUrl: string; limit?: number; -}): Promise<{ prospects: DiscoveredProspect[]; error?: string }> { +}): string[] { + const $ = cheerio.load(input.html); + const sourceHost = normalizeHost(input.sourceUrl); + const sourcePath = (() => { + try { + return new URL(input.sourceUrl).pathname; + } catch { + return "/"; + } + })(); + const out = new Set(); + + $("a[href]").each((_, el) => { + if (out.size >= (input.limit ?? 20)) return false; + const href = ($(el).attr("href") ?? "").trim(); + if (!href || href.startsWith("#") || href.startsWith("mailto:") || href.startsWith("tel:")) { + return undefined; + } + let url: URL; + try { + url = new URL(href, input.sourceUrl); + } catch { + return undefined; + } + if (url.protocol !== "https:") return undefined; + if (normalizeHost(url.hostname) !== sourceHost) return undefined; + if (ASSET_RE.test(url.pathname)) return undefined; + if (url.pathname === sourcePath || url.pathname === "/") return undefined; + if (NON_DETAIL_PATH_RE.test(url.pathname)) return undefined; + + // Detail pages sit shallow: /username, /agency/acme. Anything deeper is + // usually a sub-tab of a profile rather than another business. + const depth = url.pathname.split("/").filter(Boolean).length; + if (depth < 1 || depth > 2) return undefined; + + // Query strings on a directory are filters and paging, not new entries. + out.add(`${url.origin}${url.pathname}`.replace(/\/$/, "")); + return undefined; + }); + + return [...out]; +} + +/** Same-host paths that are navigation or account plumbing, never a business. */ +const NON_DETAIL_PATH_RE = + /^\/(search|login|signin|signup|register|about|contact|terms|privacy|pricing|blog|jobs|help|support|faq|settings|account|cart|checkout|categories|category|tags?|page|feed|rss|api)(\/|$)/i; + +const SEED_UA = "CrawlProofOutreach/1.0 (+https://crawlproof.com)"; + +/** Statuses that mean "the server refused a bot", not "the page is missing". */ +function looksBlocked(status: number): boolean { + return status === 401 || status === 403 || status === 405 || status === 429 || status === 503; +} + +async function fetchHtml( + url: string, +): Promise<{ ok: true; html: string } | { ok: false; error: string; status?: number }> { try { - const res = await fetch(input.seedUrl, { - headers: { "user-agent": "CrawlProofOutreach/1.0 (+https://crawlproof.com)" }, + const res = await fetch(url, { + headers: { "user-agent": SEED_UA }, signal: AbortSignal.timeout(15_000), redirect: "follow", }); - if (!res.ok) return { prospects: [], error: `seed ${input.seedUrl} returned HTTP ${res.status}` }; - const html = await res.text(); - return { prospects: extractOutboundProspects({ html, sourceUrl: input.seedUrl, limit: input.limit }) }; + if (!res.ok) return { ok: false, status: res.status, error: `HTTP ${res.status}` }; + return { ok: true, html: await res.text() }; } catch (err) { - return { - prospects: [], - error: `seed ${input.seedUrl} failed: ${err instanceof Error ? err.message : "unknown"}`, - }; + return { ok: false, error: err instanceof Error ? err.message : "unknown" }; + } +} + +/** + * Get a seed page's HTML, rendering it in Chromium when a plain fetch won't do. + * + * Fetch runs first because it is an order of magnitude cheaper and most + * directories are still server-rendered. The browser is the fallback for the + * two cases fetch cannot handle: the server refused us, or it returned a page + * whose listings arrive over XHR — which from here looks identical to a + * directory with nothing on it. + */ +async function loadSeedHtml( + url: string, + allowRender: boolean, +): Promise<{ html: string; rendered: boolean } | { error: string }> { + const direct = await fetchHtml(url); + if (direct.ok) { + const hasCandidates = extractOutboundProspects({ html: direct.html, sourceUrl: url, limit: 1 }).length > 0; + if (hasCandidates || !allowRender) return { html: direct.html, rendered: false }; + } else if (!allowRender) { + return { error: direct.error }; + } + + const { renderPage } = await import("./render"); + const rendered = await renderPage(url); + if (rendered.ok) return { html: rendered.html, rendered: true }; + + // Prefer the render error: when both fail it is the more specific of the + // two, and it distinguishes a bot challenge from an ordinary failure. + if (direct.ok) return { html: direct.html, rendered: false }; + return { error: rendered.error }; +} + +export async function discoverFromSeed(input: { + seedUrl: string; + limit?: number; + /** Allow the Chromium fallback. On by default. */ + render?: boolean; + /** + * 1 follows only outbound links on the seed page. 2 additionally opens + * same-host listing entries and takes the outbound link from each, which is + * what platform directories need — their listings all live on their own + * domain, so depth 1 finds nothing there. + */ + depth?: 1 | 2; + /** Cap on second-hop pages opened, since each is a full page load. */ + maxDetailPages?: number; + /** + * Only take the second hop when the first found fewer than this many + * businesses. Keeps ordinary directories at one cheap page load. + */ + detailHopThreshold?: number; +}): Promise<{ prospects: DiscoveredProspect[]; error?: string; notes?: string[] }> { + const limit = input.limit ?? 100; + const allowRender = input.render !== false; + const notes: string[] = []; + + const seed = await loadSeedHtml(input.seedUrl, allowRender); + if ("error" in seed) { + return { prospects: [], error: `seed ${input.seedUrl} failed: ${seed.error}` }; } + if (seed.rendered) notes.push(`rendered ${input.seedUrl} in a browser`); + + const merged = new Map(); + for (const p of extractOutboundProspects({ html: seed.html, sourceUrl: input.seedUrl, limit })) { + if (!merged.has(p.host)) merged.set(p.host, p); + } + + // The second hop is expensive — one page load per listing, each possibly a + // browser render — so it only runs when the first hop came up short. A + // listicle that already yielded a page of businesses has nothing to gain + // from opening its own internal links; a platform directory yields nothing + // at all on the first hop, which is exactly the signal to go deeper. + const firstHopThin = merged.size < (input.detailHopThreshold ?? 3); + if ((input.depth ?? 1) >= 2 && firstHopThin && merged.size < limit) { + const detailPages = extractSameHostLinks({ + html: seed.html, + sourceUrl: input.seedUrl, + limit: input.maxDetailPages ?? 12, + }); + if (detailPages.length === 0) { + notes.push("no listing entries found to open for a second hop"); + } + for (const detailUrl of detailPages) { + if (merged.size >= limit) break; + const detail = await loadSeedHtml(detailUrl, allowRender); + if ("error" in detail) continue; + for (const p of extractOutboundProspects({ + html: detail.html, + sourceUrl: detailUrl, + limit: limit - merged.size, + })) { + if (!merged.has(p.host)) merged.set(p.host, p); + } + } + notes.push(`opened ${detailPages.length} listing entries`); + } + + return { prospects: [...merged.values()], notes: notes.length ? notes : undefined }; } /** Which search backend to use. */ @@ -229,7 +391,10 @@ export async function discoverProspects(input: { for (const seedUrl of (input.seedUrls ?? []).slice(0, 10)) { if (merged.size >= limit) break; - const res = await discoverFromSeed({ seedUrl, limit }); + // Depth 2 by default: a platform directory keeps every listing on its own + // domain, so depth 1 silently returns nothing for exactly the pages users + // most often paste in. + const res = await discoverFromSeed({ seedUrl, limit, depth: 2 }); if (res.error) errors.push(res.error); for (const p of res.prospects) if (!merged.has(p.host)) merged.set(p.host, p); } diff --git a/lib/outreach/render.ts b/lib/outreach/render.ts new file mode 100644 index 00000000..9999991e --- /dev/null +++ b/lib/outreach/render.ts @@ -0,0 +1,191 @@ +// Headless-browser page rendering for seed discovery. +// +// A plain fetch only sees the HTML the server sent. A growing share of the +// pages worth seeding — marketplace category pages, artist and agency +// directories, anything built as an SPA — ship an empty shell and load the +// listings over XHR afterwards. Fetching those returns 200 and zero links, +// which reads as "this directory has no businesses on it" rather than "we +// couldn't see them". +// +// This renders the page in Chromium instead, which is deliberately generic: +// no per-directory API clients to write and re-write as each site changes its +// endpoints. +// +// Chromium is already in the production image (mcr.microsoft.com/playwright), +// so this costs memory at runtime rather than image size. To keep that bounded +// the browser is launched once and shared, images/media/fonts are blocked +// (they cost bandwidth and memory and never contain a link), and the whole +// thing shuts itself down after a spell of inactivity rather than pinning a +// Chromium process for the life of the container. +// +// What this does NOT do is defeat bot protection. A site behind a Cloudflare +// managed challenge stays blocked — headless and headed Chromium both sit on +// the interstitial from a datacenter IP, and the page never renders. Rendering +// solves "the HTML arrives empty", not "the site doesn't want us". + +import type { Browser, BrowserContext } from "playwright"; +import { isPrivateAddress } from "./mailboxDiscovery"; +import dns from "node:dns/promises"; +import net from "node:net"; + +const NAV_TIMEOUT_MS = 25_000; +const SETTLE_MS = 2_500; +const IDLE_SHUTDOWN_MS = 60_000; +const MAX_HTML_BYTES = 3 * 1024 * 1024; + +// Chromium is heavy enough that a per-call launch would dominate the cost of +// a campaign tick, so one instance is shared across renders. +let browserPromise: Promise | null = null; +let idleTimer: NodeJS.Timeout | null = null; +let inFlight = 0; + +const UA = + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/140.0.0.0 Safari/537.36"; + +async function getBrowser(): Promise { + if (!browserPromise) { + // Imported lazily so that importing this module — which the discovery + // path does unconditionally — doesn't pull Chromium bindings into + // processes that never render anything. + browserPromise = import("playwright").then(({ chromium }) => + chromium.launch({ + args: ["--no-sandbox", "--disable-dev-shm-usage", "--disable-gpu"], + }), + ); + } + return browserPromise; +} + +function touchIdleTimer() { + if (idleTimer) clearTimeout(idleTimer); + idleTimer = setTimeout(() => { + if (inFlight > 0) return; + void closeBrowser(); + }, IDLE_SHUTDOWN_MS); + // Don't hold the process open just to keep an idle browser around. + idleTimer.unref?.(); +} + +export async function closeBrowser(): Promise { + const pending = browserPromise; + browserPromise = null; + if (idleTimer) { + clearTimeout(idleTimer); + idleTimer = null; + } + if (!pending) return; + try { + const browser = await pending; + await browser.close(); + } catch { + // Already gone, or never started — nothing to clean up. + } +} + +/** + * Seed URLs come from users, so a renderer will happily point at anything it + * is given. Resolve first and refuse anything in private space. + */ +async function hostIsPublic(hostname: string): Promise { + if (net.isIP(hostname)) return !isPrivateAddress(hostname); + try { + const addrs = await dns.lookup(hostname, { all: true }); + return addrs.length > 0 && addrs.every((a) => !isPrivateAddress(a.address)); + } catch { + return false; + } +} + +export type RenderResult = + | { ok: true; html: string; status: number; finalUrl: string } + | { ok: false; error: string; status?: number; challenged?: boolean }; + +/** Page titles a bot-protection interstitial serves instead of the real page. */ +const CHALLENGE_RE = /just a moment|attention required|verifying you are human|checking your browser/i; + +/** + * Load `url` in Chromium and return the DOM after scripts have run. + * + * Resolves rather than throws: discovery treats a failed render as "this seed + * produced nothing", not as a reason to abort a campaign tick. + */ +export async function renderPage(url: string): Promise { + let parsed: URL; + try { + parsed = new URL(url); + } catch { + return { ok: false, error: "not a valid URL" }; + } + if (parsed.protocol !== "https:" && parsed.protocol !== "http:") { + return { ok: false, error: "only http(s) URLs can be rendered" }; + } + if (!(await hostIsPublic(parsed.hostname))) { + return { ok: false, error: "host does not resolve, or resolves to a private address" }; + } + + let context: BrowserContext | null = null; + inFlight += 1; + try { + const browser = await getBrowser(); + context = await browser.newContext({ + userAgent: UA, + viewport: { width: 1440, height: 900 }, + locale: "en-US", + javaScriptEnabled: true, + }); + + // Images, media and fonts carry no links and dominate page weight. + await context.route("**/*", (route) => { + const type = route.request().resourceType(); + if (type === "image" || type === "media" || type === "font") return route.abort(); + return route.continue(); + }); + + const page = await context.newPage(); + page.setDefaultTimeout(NAV_TIMEOUT_MS); + + const response = await page.goto(url, { + waitUntil: "domcontentloaded", + timeout: NAV_TIMEOUT_MS, + }); + const status = response?.status() ?? 0; + + // Let XHR-driven listings arrive. networkidle is the obvious choice but + // never fires on pages that poll or hold a socket open, so this waits for + // quiet with a hard ceiling instead. + await page + .waitForLoadState("networkidle", { timeout: SETTLE_MS * 2 }) + .catch(() => page.waitForTimeout(SETTLE_MS)); + + const title = await page.title().catch(() => ""); + if (CHALLENGE_RE.test(title)) { + return { + ok: false, + status, + challenged: true, + error: + "the site served a bot-protection challenge instead of the page — rendering can't get past it", + }; + } + if (status >= 400) { + return { ok: false, status, error: `rendered with HTTP ${status}` }; + } + + const html = await page.content(); + return { + ok: true, + status, + finalUrl: page.url(), + html: html.length > MAX_HTML_BYTES ? html.slice(0, MAX_HTML_BYTES) : html, + }; + } catch (error) { + return { + ok: false, + error: error instanceof Error ? error.message.slice(0, 200) : "render failed", + }; + } finally { + inFlight -= 1; + if (context) await context.close().catch(() => {}); + touchIdleTimer(); + } +} diff --git a/next.config.ts b/next.config.ts index 7b84c764..d8caf2b6 100644 --- a/next.config.ts +++ b/next.config.ts @@ -6,6 +6,10 @@ const nextConfig: NextConfig = { // Produce a self-contained server bundle in .next/standalone/ so the // production Dockerfile stays small. Used by the Railway image. output: "standalone", + // Playwright launches a real Chromium binary and resolves it through its own + // package layout, so bundling it into a server chunk breaks that lookup at + // runtime. Keep it external and let Node require it from node_modules. + serverExternalPackages: ["playwright", "playwright-core"], experimental: { serverActions: { bodySizeLimit: "2mb", diff --git a/package-lock.json b/package-lock.json index 3fccedb6..151552ec 100644 --- a/package-lock.json +++ b/package-lock.json @@ -28,6 +28,7 @@ "next": "^16.0.0", "nodemailer": "^8.0.10", "openai": "^6.37.0", + "playwright": "1.60.0", "posthog-js": "^1.381.0", "qrcode.react": "^4.2.0", "react": "^19.0.0", @@ -46,7 +47,6 @@ "@types/node": "^22.9.0", "@types/react": "^19.0.0", "@types/react-dom": "^19.0.0", - "playwright": "next", "postcss": "^8.4.49", "tailwindcss": "^4.0.0", "tsx": "^4.19.0", @@ -5526,7 +5526,6 @@ "version": "1.60.0", "resolved": "https://registry.npmjs.org/playwright/-/playwright-1.60.0.tgz", "integrity": "sha512-hheHdokM8cdqCb0lcE3s+zT4t4W+vvjpGxsZlDnikarzx8tSzMebh3UiFtgqwFwnTnjYQcsyMF8ei2mCO/tpeA==", - "devOptional": true, "license": "Apache-2.0", "dependencies": { "playwright-core": "1.60.0" @@ -5558,7 +5557,6 @@ "version": "1.60.0", "resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.60.0.tgz", "integrity": "sha512-9bW6zvX/m0lEbgTKJ6YppOKx8H3VOPBMOCFh2irXFOT4BbHgrx5hPjwJYLT40Lu+4qtD36qKc/Hn56StUW57IA==", - "devOptional": true, "license": "Apache-2.0", "bin": { "playwright-core": "cli.js" diff --git a/package.json b/package.json index 65800fd3..ae624c8b 100644 --- a/package.json +++ b/package.json @@ -40,6 +40,7 @@ "next": "^16.0.0", "nodemailer": "^8.0.10", "openai": "^6.37.0", + "playwright": "1.60.0", "posthog-js": "^1.381.0", "qrcode.react": "^4.2.0", "react": "^19.0.0", @@ -58,7 +59,6 @@ "@types/node": "^22.9.0", "@types/react": "^19.0.0", "@types/react-dom": "^19.0.0", - "playwright": "next", "postcss": "^8.4.49", "tailwindcss": "^4.0.0", "tsx": "^4.19.0", diff --git a/tests/seed-discovery.test.ts b/tests/seed-discovery.test.ts new file mode 100644 index 00000000..eddc10de --- /dev/null +++ b/tests/seed-discovery.test.ts @@ -0,0 +1,116 @@ +import { describe, it, expect } from "vitest"; +import { extractSameHostLinks, extractOutboundProspects } from "@/lib/outreach/discover"; + +// Shaped like a platform directory: every listing links to a profile on the +// platform's own domain, alongside the usual nav furniture. +const DIRECTORY_HTML = ` + + + + Follow us +`; + +// A profile page: this is where the artist's own site finally appears. +const PROFILE_HTML = ` + +

Jane Doe

+ Profile + janedoe.design + Twitter + ArtStation +`; + +describe("extractSameHostLinks", () => { + const links = extractSameHostLinks({ + html: DIRECTORY_HTML, + sourceUrl: "https://www.platform.test/search?sort_by=followers", + }); + + it("collects same-host listing entries", () => { + expect(links).toContain("https://www.platform.test/janedoe"); + expect(links).toContain("https://www.platform.test/john-smith"); + }); + + it("treats an absolute same-host URL the same as a relative one", () => { + expect(links).toContain("https://www.platform.test/kim_lee"); + }); + + it("allows a two-segment detail path", () => { + expect(links).toContain("https://www.platform.test/studios/acme"); + }); + + it("skips navigation and account plumbing", () => { + for (const path of ["/", "/search", "/login", "/about", "/pricing"]) { + expect(links).not.toContain(`https://www.platform.test${path}`); + } + }); + + it("skips paths deeper than a detail page", () => { + expect(links.some((l) => l.includes("/likes/extra/deep"))).toBe(false); + }); + + it("skips assets", () => { + expect(links.some((l) => l.endsWith(".png"))).toBe(false); + }); + + it("collapses query-string variants onto one entry", () => { + expect(links.filter((l) => l === "https://www.platform.test/janedoe")).toHaveLength(1); + }); + + it("never returns another host", () => { + expect(links.some((l) => l.includes("twitter.com"))).toBe(false); + }); + + it("honours the limit", () => { + expect( + extractSameHostLinks({ + html: DIRECTORY_HTML, + sourceUrl: "https://www.platform.test/search", + limit: 2, + }), + ).toHaveLength(2); + }); +}); + +describe("two-hop shape", () => { + it("finds nothing outbound on the directory page itself", () => { + // The whole reason depth 2 exists: every listing is same-host, so the + // first hop yields no businesses at all. + const first = extractOutboundProspects({ + html: DIRECTORY_HTML, + sourceUrl: "https://www.platform.test/search", + }); + expect(first.map((p) => p.host)).not.toContain("platform.test"); + expect(first.some((p) => p.host.includes("janedoe"))).toBe(false); + }); + + it("finds the business site on the profile page", () => { + const second = extractOutboundProspects({ + html: PROFILE_HTML, + sourceUrl: "https://www.platform.test/janedoe", + }); + expect(second.map((p) => p.host)).toContain("janedoe.design"); + }); + + it("still filters platforms out on the second hop", () => { + const hosts = extractOutboundProspects({ + html: PROFILE_HTML, + sourceUrl: "https://www.platform.test/janedoe", + }).map((p) => p.host); + expect(hosts).not.toContain("twitter.com"); + }); +});