diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index 6c2b968b..86f30cf5 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -108,25 +108,187 @@ export function extractOutboundProspects(input: { return [...out.values()]; } -export async function discoverFromSeed(input: { - seedUrl: string; +/** + * Same-host links that look like an entry for one business rather than site + * furniture. + * + * On a platform directory — an artist marketplace, an agency roster — the + * listing page links to profiles on its own domain, and the business's real + * website only appears on the profile. Those links are invisible to + * `extractOutboundProspects`, which drops same-host hrefs as navigation, so + * they are collected separately for the second hop. + */ +export function extractSameHostLinks(input: { + html: string; + sourceUrl: string; limit?: number; -}): Promise<{ prospects: DiscoveredProspect[]; error?: string }> { +}): string[] { + const $ = cheerio.load(input.html); + const sourceHost = normalizeHost(input.sourceUrl); + const sourcePath = (() => { + try { + return new URL(input.sourceUrl).pathname; + } catch { + return "/"; + } + })(); + const out = new Set(); + + $("a[href]").each((_, el) => { + if (out.size >= (input.limit ?? 20)) return false; + const href = ($(el).attr("href") ?? "").trim(); + if (!href || href.startsWith("#") || href.startsWith("mailto:") || href.startsWith("tel:")) { + return undefined; + } + let url: URL; + try { + url = new URL(href, input.sourceUrl); + } catch { + return undefined; + } + if (url.protocol !== "https:") return undefined; + if (normalizeHost(url.hostname) !== sourceHost) return undefined; + if (ASSET_RE.test(url.pathname)) return undefined; + if (url.pathname === sourcePath || url.pathname === "/") return undefined; + if (NON_DETAIL_PATH_RE.test(url.pathname)) return undefined; + + // Detail pages sit shallow: /username, /agency/acme. Anything deeper is + // usually a sub-tab of a profile rather than another business. + const depth = url.pathname.split("/").filter(Boolean).length; + if (depth < 1 || depth > 2) return undefined; + + // Query strings on a directory are filters and paging, not new entries. + out.add(`${url.origin}${url.pathname}`.replace(/\/$/, "")); + return undefined; + }); + + return [...out]; +} + +/** Same-host paths that are navigation or account plumbing, never a business. */ +const NON_DETAIL_PATH_RE = + /^\/(search|login|signin|signup|register|about|contact|terms|privacy|pricing|blog|jobs|help|support|faq|settings|account|cart|checkout|categories|category|tags?|page|feed|rss|api)(\/|$)/i; + +const SEED_UA = "CrawlProofOutreach/1.0 (+https://crawlproof.com)"; + +/** Statuses that mean "the server refused a bot", not "the page is missing". */ +function looksBlocked(status: number): boolean { + return status === 401 || status === 403 || status === 405 || status === 429 || status === 503; +} + +async function fetchHtml( + url: string, +): Promise<{ ok: true; html: string } | { ok: false; error: string; status?: number }> { try { - const res = await fetch(input.seedUrl, { - headers: { "user-agent": "CrawlProofOutreach/1.0 (+https://crawlproof.com)" }, + const res = await fetch(url, { + headers: { "user-agent": SEED_UA }, signal: AbortSignal.timeout(15_000), redirect: "follow", }); - if (!res.ok) return { prospects: [], error: `seed ${input.seedUrl} returned HTTP ${res.status}` }; - const html = await res.text(); - return { prospects: extractOutboundProspects({ html, sourceUrl: input.seedUrl, limit: input.limit }) }; + if (!res.ok) return { ok: false, status: res.status, error: `HTTP ${res.status}` }; + return { ok: true, html: await res.text() }; } catch (err) { - return { - prospects: [], - error: `seed ${input.seedUrl} failed: ${err instanceof Error ? err.message : "unknown"}`, - }; + return { ok: false, error: err instanceof Error ? err.message : "unknown" }; + } +} + +/** + * Get a seed page's HTML, rendering it in Chromium when a plain fetch won't do. + * + * Fetch runs first because it is an order of magnitude cheaper and most + * directories are still server-rendered. The browser is the fallback for the + * two cases fetch cannot handle: the server refused us, or it returned a page + * whose listings arrive over XHR — which from here looks identical to a + * directory with nothing on it. + */ +async function loadSeedHtml( + url: string, + allowRender: boolean, +): Promise<{ html: string; rendered: boolean } | { error: string }> { + const direct = await fetchHtml(url); + if (direct.ok) { + const hasCandidates = extractOutboundProspects({ html: direct.html, sourceUrl: url, limit: 1 }).length > 0; + if (hasCandidates || !allowRender) return { html: direct.html, rendered: false }; + } else if (!allowRender) { + return { error: direct.error }; + } + + const { renderPage } = await import("./render"); + const rendered = await renderPage(url); + if (rendered.ok) return { html: rendered.html, rendered: true }; + + // Prefer the render error: when both fail it is the more specific of the + // two, and it distinguishes a bot challenge from an ordinary failure. + if (direct.ok) return { html: direct.html, rendered: false }; + return { error: rendered.error }; +} + +export async function discoverFromSeed(input: { + seedUrl: string; + limit?: number; + /** Allow the Chromium fallback. On by default. */ + render?: boolean; + /** + * 1 follows only outbound links on the seed page. 2 additionally opens + * same-host listing entries and takes the outbound link from each, which is + * what platform directories need — their listings all live on their own + * domain, so depth 1 finds nothing there. + */ + depth?: 1 | 2; + /** Cap on second-hop pages opened, since each is a full page load. */ + maxDetailPages?: number; + /** + * Only take the second hop when the first found fewer than this many + * businesses. Keeps ordinary directories at one cheap page load. + */ + detailHopThreshold?: number; +}): Promise<{ prospects: DiscoveredProspect[]; error?: string; notes?: string[] }> { + const limit = input.limit ?? 100; + const allowRender = input.render !== false; + const notes: string[] = []; + + const seed = await loadSeedHtml(input.seedUrl, allowRender); + if ("error" in seed) { + return { prospects: [], error: `seed ${input.seedUrl} failed: ${seed.error}` }; } + if (seed.rendered) notes.push(`rendered ${input.seedUrl} in a browser`); + + const merged = new Map(); + for (const p of extractOutboundProspects({ html: seed.html, sourceUrl: input.seedUrl, limit })) { + if (!merged.has(p.host)) merged.set(p.host, p); + } + + // The second hop is expensive — one page load per listing, each possibly a + // browser render — so it only runs when the first hop came up short. A + // listicle that already yielded a page of businesses has nothing to gain + // from opening its own internal links; a platform directory yields nothing + // at all on the first hop, which is exactly the signal to go deeper. + const firstHopThin = merged.size < (input.detailHopThreshold ?? 3); + if ((input.depth ?? 1) >= 2 && firstHopThin && merged.size < limit) { + const detailPages = extractSameHostLinks({ + html: seed.html, + sourceUrl: input.seedUrl, + limit: input.maxDetailPages ?? 12, + }); + if (detailPages.length === 0) { + notes.push("no listing entries found to open for a second hop"); + } + for (const detailUrl of detailPages) { + if (merged.size >= limit) break; + const detail = await loadSeedHtml(detailUrl, allowRender); + if ("error" in detail) continue; + for (const p of extractOutboundProspects({ + html: detail.html, + sourceUrl: detailUrl, + limit: limit - merged.size, + })) { + if (!merged.has(p.host)) merged.set(p.host, p); + } + } + notes.push(`opened ${detailPages.length} listing entries`); + } + + return { prospects: [...merged.values()], notes: notes.length ? notes : undefined }; } /** Which search backend to use. */ @@ -229,7 +391,10 @@ export async function discoverProspects(input: { for (const seedUrl of (input.seedUrls ?? []).slice(0, 10)) { if (merged.size >= limit) break; - const res = await discoverFromSeed({ seedUrl, limit }); + // Depth 2 by default: a platform directory keeps every listing on its own + // domain, so depth 1 silently returns nothing for exactly the pages users + // most often paste in. + const res = await discoverFromSeed({ seedUrl, limit, depth: 2 }); if (res.error) errors.push(res.error); for (const p of res.prospects) if (!merged.has(p.host)) merged.set(p.host, p); } diff --git a/lib/outreach/render.ts b/lib/outreach/render.ts new file mode 100644 index 00000000..9999991e --- /dev/null +++ b/lib/outreach/render.ts @@ -0,0 +1,191 @@ +// Headless-browser page rendering for seed discovery. +// +// A plain fetch only sees the HTML the server sent. A growing share of the +// pages worth seeding — marketplace category pages, artist and agency +// directories, anything built as an SPA — ship an empty shell and load the +// listings over XHR afterwards. Fetching those returns 200 and zero links, +// which reads as "this directory has no businesses on it" rather than "we +// couldn't see them". +// +// This renders the page in Chromium instead, which is deliberately generic: +// no per-directory API clients to write and re-write as each site changes its +// endpoints. +// +// Chromium is already in the production image (mcr.microsoft.com/playwright), +// so this costs memory at runtime rather than image size. To keep that bounded +// the browser is launched once and shared, images/media/fonts are blocked +// (they cost bandwidth and memory and never contain a link), and the whole +// thing shuts itself down after a spell of inactivity rather than pinning a +// Chromium process for the life of the container. +// +// What this does NOT do is defeat bot protection. A site behind a Cloudflare +// managed challenge stays blocked — headless and headed Chromium both sit on +// the interstitial from a datacenter IP, and the page never renders. Rendering +// solves "the HTML arrives empty", not "the site doesn't want us". + +import type { Browser, BrowserContext } from "playwright"; +import { isPrivateAddress } from "./mailboxDiscovery"; +import dns from "node:dns/promises"; +import net from "node:net"; + +const NAV_TIMEOUT_MS = 25_000; +const SETTLE_MS = 2_500; +const IDLE_SHUTDOWN_MS = 60_000; +const MAX_HTML_BYTES = 3 * 1024 * 1024; + +// Chromium is heavy enough that a per-call launch would dominate the cost of +// a campaign tick, so one instance is shared across renders. +let browserPromise: Promise | null = null; +let idleTimer: NodeJS.Timeout | null = null; +let inFlight = 0; + +const UA = + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/140.0.0.0 Safari/537.36"; + +async function getBrowser(): Promise { + if (!browserPromise) { + // Imported lazily so that importing this module — which the discovery + // path does unconditionally — doesn't pull Chromium bindings into + // processes that never render anything. + browserPromise = import("playwright").then(({ chromium }) => + chromium.launch({ + args: ["--no-sandbox", "--disable-dev-shm-usage", "--disable-gpu"], + }), + ); + } + return browserPromise; +} + +function touchIdleTimer() { + if (idleTimer) clearTimeout(idleTimer); + idleTimer = setTimeout(() => { + if (inFlight > 0) return; + void closeBrowser(); + }, IDLE_SHUTDOWN_MS); + // Don't hold the process open just to keep an idle browser around. + idleTimer.unref?.(); +} + +export async function closeBrowser(): Promise { + const pending = browserPromise; + browserPromise = null; + if (idleTimer) { + clearTimeout(idleTimer); + idleTimer = null; + } + if (!pending) return; + try { + const browser = await pending; + await browser.close(); + } catch { + // Already gone, or never started — nothing to clean up. + } +} + +/** + * Seed URLs come from users, so a renderer will happily point at anything it + * is given. Resolve first and refuse anything in private space. + */ +async function hostIsPublic(hostname: string): Promise { + if (net.isIP(hostname)) return !isPrivateAddress(hostname); + try { + const addrs = await dns.lookup(hostname, { all: true }); + return addrs.length > 0 && addrs.every((a) => !isPrivateAddress(a.address)); + } catch { + return false; + } +} + +export type RenderResult = + | { ok: true; html: string; status: number; finalUrl: string } + | { ok: false; error: string; status?: number; challenged?: boolean }; + +/** Page titles a bot-protection interstitial serves instead of the real page. */ +const CHALLENGE_RE = /just a moment|attention required|verifying you are human|checking your browser/i; + +/** + * Load `url` in Chromium and return the DOM after scripts have run. + * + * Resolves rather than throws: discovery treats a failed render as "this seed + * produced nothing", not as a reason to abort a campaign tick. + */ +export async function renderPage(url: string): Promise { + let parsed: URL; + try { + parsed = new URL(url); + } catch { + return { ok: false, error: "not a valid URL" }; + } + if (parsed.protocol !== "https:" && parsed.protocol !== "http:") { + return { ok: false, error: "only http(s) URLs can be rendered" }; + } + if (!(await hostIsPublic(parsed.hostname))) { + return { ok: false, error: "host does not resolve, or resolves to a private address" }; + } + + let context: BrowserContext | null = null; + inFlight += 1; + try { + const browser = await getBrowser(); + context = await browser.newContext({ + userAgent: UA, + viewport: { width: 1440, height: 900 }, + locale: "en-US", + javaScriptEnabled: true, + }); + + // Images, media and fonts carry no links and dominate page weight. + await context.route("**/*", (route) => { + const type = route.request().resourceType(); + if (type === "image" || type === "media" || type === "font") return route.abort(); + return route.continue(); + }); + + const page = await context.newPage(); + page.setDefaultTimeout(NAV_TIMEOUT_MS); + + const response = await page.goto(url, { + waitUntil: "domcontentloaded", + timeout: NAV_TIMEOUT_MS, + }); + const status = response?.status() ?? 0; + + // Let XHR-driven listings arrive. networkidle is the obvious choice but + // never fires on pages that poll or hold a socket open, so this waits for + // quiet with a hard ceiling instead. + await page + .waitForLoadState("networkidle", { timeout: SETTLE_MS * 2 }) + .catch(() => page.waitForTimeout(SETTLE_MS)); + + const title = await page.title().catch(() => ""); + if (CHALLENGE_RE.test(title)) { + return { + ok: false, + status, + challenged: true, + error: + "the site served a bot-protection challenge instead of the page — rendering can't get past it", + }; + } + if (status >= 400) { + return { ok: false, status, error: `rendered with HTTP ${status}` }; + } + + const html = await page.content(); + return { + ok: true, + status, + finalUrl: page.url(), + html: html.length > MAX_HTML_BYTES ? html.slice(0, MAX_HTML_BYTES) : html, + }; + } catch (error) { + return { + ok: false, + error: error instanceof Error ? error.message.slice(0, 200) : "render failed", + }; + } finally { + inFlight -= 1; + if (context) await context.close().catch(() => {}); + touchIdleTimer(); + } +} diff --git a/next.config.ts b/next.config.ts index 7b84c764..d8caf2b6 100644 --- a/next.config.ts +++ b/next.config.ts @@ -6,6 +6,10 @@ const nextConfig: NextConfig = { // Produce a self-contained server bundle in .next/standalone/ so the // production Dockerfile stays small. Used by the Railway image. output: "standalone", + // Playwright launches a real Chromium binary and resolves it through its own + // package layout, so bundling it into a server chunk breaks that lookup at + // runtime. Keep it external and let Node require it from node_modules. + serverExternalPackages: ["playwright", "playwright-core"], experimental: { serverActions: { bodySizeLimit: "2mb", diff --git a/package-lock.json b/package-lock.json index 3fccedb6..151552ec 100644 --- a/package-lock.json +++ b/package-lock.json @@ -28,6 +28,7 @@ "next": "^16.0.0", "nodemailer": "^8.0.10", "openai": "^6.37.0", + "playwright": "1.60.0", "posthog-js": "^1.381.0", "qrcode.react": "^4.2.0", "react": "^19.0.0", @@ -46,7 +47,6 @@ "@types/node": "^22.9.0", "@types/react": "^19.0.0", "@types/react-dom": "^19.0.0", - "playwright": "next", "postcss": "^8.4.49", "tailwindcss": "^4.0.0", "tsx": "^4.19.0", @@ -5526,7 +5526,6 @@ "version": "1.60.0", "resolved": "https://registry.npmjs.org/playwright/-/playwright-1.60.0.tgz", "integrity": "sha512-hheHdokM8cdqCb0lcE3s+zT4t4W+vvjpGxsZlDnikarzx8tSzMebh3UiFtgqwFwnTnjYQcsyMF8ei2mCO/tpeA==", - "devOptional": true, "license": "Apache-2.0", "dependencies": { "playwright-core": "1.60.0" @@ -5558,7 +5557,6 @@ "version": "1.60.0", "resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.60.0.tgz", "integrity": "sha512-9bW6zvX/m0lEbgTKJ6YppOKx8H3VOPBMOCFh2irXFOT4BbHgrx5hPjwJYLT40Lu+4qtD36qKc/Hn56StUW57IA==", - "devOptional": true, "license": "Apache-2.0", "bin": { "playwright-core": "cli.js" diff --git a/package.json b/package.json index 65800fd3..ae624c8b 100644 --- a/package.json +++ b/package.json @@ -40,6 +40,7 @@ "next": "^16.0.0", "nodemailer": "^8.0.10", "openai": "^6.37.0", + "playwright": "1.60.0", "posthog-js": "^1.381.0", "qrcode.react": "^4.2.0", "react": "^19.0.0", @@ -58,7 +59,6 @@ "@types/node": "^22.9.0", "@types/react": "^19.0.0", "@types/react-dom": "^19.0.0", - "playwright": "next", "postcss": "^8.4.49", "tailwindcss": "^4.0.0", "tsx": "^4.19.0", diff --git a/tests/seed-discovery.test.ts b/tests/seed-discovery.test.ts new file mode 100644 index 00000000..eddc10de --- /dev/null +++ b/tests/seed-discovery.test.ts @@ -0,0 +1,116 @@ +import { describe, it, expect } from "vitest"; +import { extractSameHostLinks, extractOutboundProspects } from "@/lib/outreach/discover"; + +// Shaped like a platform directory: every listing links to a profile on the +// platform's own domain, alongside the usual nav furniture. +const DIRECTORY_HTML = ` + + + + Follow us +`; + +// A profile page: this is where the artist's own site finally appears. +const PROFILE_HTML = ` + +

Jane Doe

+ Profile + janedoe.design + Twitter + ArtStation +`; + +describe("extractSameHostLinks", () => { + const links = extractSameHostLinks({ + html: DIRECTORY_HTML, + sourceUrl: "https://www.platform.test/search?sort_by=followers", + }); + + it("collects same-host listing entries", () => { + expect(links).toContain("https://www.platform.test/janedoe"); + expect(links).toContain("https://www.platform.test/john-smith"); + }); + + it("treats an absolute same-host URL the same as a relative one", () => { + expect(links).toContain("https://www.platform.test/kim_lee"); + }); + + it("allows a two-segment detail path", () => { + expect(links).toContain("https://www.platform.test/studios/acme"); + }); + + it("skips navigation and account plumbing", () => { + for (const path of ["/", "/search", "/login", "/about", "/pricing"]) { + expect(links).not.toContain(`https://www.platform.test${path}`); + } + }); + + it("skips paths deeper than a detail page", () => { + expect(links.some((l) => l.includes("/likes/extra/deep"))).toBe(false); + }); + + it("skips assets", () => { + expect(links.some((l) => l.endsWith(".png"))).toBe(false); + }); + + it("collapses query-string variants onto one entry", () => { + expect(links.filter((l) => l === "https://www.platform.test/janedoe")).toHaveLength(1); + }); + + it("never returns another host", () => { + expect(links.some((l) => l.includes("twitter.com"))).toBe(false); + }); + + it("honours the limit", () => { + expect( + extractSameHostLinks({ + html: DIRECTORY_HTML, + sourceUrl: "https://www.platform.test/search", + limit: 2, + }), + ).toHaveLength(2); + }); +}); + +describe("two-hop shape", () => { + it("finds nothing outbound on the directory page itself", () => { + // The whole reason depth 2 exists: every listing is same-host, so the + // first hop yields no businesses at all. + const first = extractOutboundProspects({ + html: DIRECTORY_HTML, + sourceUrl: "https://www.platform.test/search", + }); + expect(first.map((p) => p.host)).not.toContain("platform.test"); + expect(first.some((p) => p.host.includes("janedoe"))).toBe(false); + }); + + it("finds the business site on the profile page", () => { + const second = extractOutboundProspects({ + html: PROFILE_HTML, + sourceUrl: "https://www.platform.test/janedoe", + }); + expect(second.map((p) => p.host)).toContain("janedoe.design"); + }); + + it("still filters platforms out on the second hop", () => { + const hosts = extractOutboundProspects({ + html: PROFILE_HTML, + sourceUrl: "https://www.platform.test/janedoe", + }).map((p) => p.host); + expect(hosts).not.toContain("twitter.com"); + }); +});