diff --git a/lib/outreach/documents.ts b/lib/outreach/documents.ts new file mode 100644 index 00000000..bc176e1f --- /dev/null +++ b/lib/outreach/documents.ts @@ -0,0 +1,139 @@ +// Contact details that only exist inside a PDF. +// +// Companies put the address you actually want in a capability statement, a +// media kit or a data-sheet, and link to it from a page whose HTML says +// nothing. The crawler reads the HTML, finds no address, and moves on — while +// the answer sits one link away in a file it never opened. +// +// Bounded on purpose. A PDF is a slow, large fetch that yields at most a few +// addresses, so only documents linked from a page already being read are +// considered, only the first few, and only up to a size worth the wait. + +import { extractText, getDocumentProxy } from "unpdf"; +import { discoverContactEmails, normalizeHost, type ContactCandidate } from "./cold"; + +const MAX_PDF_BYTES = 12 * 1024 * 1024; +const MAX_PDF_PAGES = 25; +const FETCH_TIMEOUT_MS = 20_000; + +/** Filenames that suggest a document naming humans, best first. */ +const PROMISING_NAME = + /(capabilit|contact|team|leadership|about|company|profile|brochure|media[-_ ]?kit|overview|prospectus|annual|fact[-_ ]?sheet)/i; + +/** + * PDFs worth opening, most promising first. + * + * Every PDF on a site is not worth a twelve-megabyte download; a "capability + * statement" is, and a terms-and-conditions is not. Ordering by filename is + * a weak signal but it is free, and it decides which few get opened. + */ +export function pdfLinksFrom(html: string, sourceUrl: string, limit = 3): string[] { + const out = new Set(); + for (const m of html.matchAll(/href=["']([^"']+\.pdf(?:\?[^"']*)?)["']/gi)) { + try { + const url = new URL(m[1], sourceUrl); + if (url.protocol !== "https:" && url.protocol !== "http:") continue; + out.add(url.toString()); + } catch { + // Unparseable href. + } + } + return [...out] + .sort((a, b) => Number(PROMISING_NAME.test(b)) - Number(PROMISING_NAME.test(a))) + .slice(0, limit); +} + +/** Text of a PDF, or null. Never throws — a bad document is not an error. */ +export async function pdfText(url: string): Promise { + try { + const res = await fetch(url, { + headers: { "user-agent": "CrawlProofOutreach/1.0 (+https://crawlproof.com)" }, + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), + redirect: "follow", + }); + if (!res.ok) return null; + + const type = res.headers.get("content-type") ?? ""; + // A .pdf href that serves HTML is a login wall or a 404 page dressed as + // one; parsing it wastes the download and finds nothing. + if (type && !/pdf|octet-stream/i.test(type)) return null; + + const buf = await res.arrayBuffer(); + if (buf.byteLength > MAX_PDF_BYTES) return null; + + const doc = await getDocumentProxy(new Uint8Array(buf)); + // Page cap rather than whole-document: contact details live at the front + // or the back, and a 400-page report costs far more to parse than the + // address is worth. + const { text } = await extractText(doc, { mergePages: true }); + return typeof text === "string" ? text.slice(0, 200_000) : null; + } catch { + return null; + } +} + +/** + * Addresses found inside the documents a page links to. + * + * `discoverContactEmails` expects markup, so the extracted text is wrapped + * before being handed over — that keeps one implementation of what counts as + * an address, including the obfuscation handling, rather than a second one + * that drifts. + */ +export async function contactsFromDocuments(input: { + html: string; + sourceUrl: string; + host: string; + limit?: number; +}): Promise<{ candidates: ContactCandidate[]; opened: string[] }> { + const host = normalizeHost(input.host); + const links = pdfLinksFrom(input.html, input.sourceUrl, input.limit ?? 3); + const found = new Map(); + const opened: string[] = []; + + for (const link of links) { + const text = await pdfText(link); + if (!text) continue; + opened.push(link); + for (const c of discoverContactEmails(`
${text}
`, host)) { + if (!found.has(c.email)) found.set(c.email, c); + } + // One document with a same-domain address is enough; the rest are + // downloads that cannot improve on it. + if ([...found.values()].some((c) => c.sameDomain)) break; + } + + return { candidates: [...found.values()], opened }; +} + +/** Paths that name the people at a company rather than describing it. */ +const TEAM_PATH_RE = + /\/(team|our-team|people|our-people|leadership|management|staff|founders|executives|who-we-are|meet-the-team|board)(\/|$|\.html?$)/i; + +/** + * Same-host pages that list the people at a company. + * + * A named person's address outperforms info@ by enough to be worth one extra + * fetch, and a team page is where those names and addresses are published. + */ +export function teamPageLinks(html: string, sourceUrl: string, limit = 2): string[] { + const out = new Set(); + let sourceHost = ""; + try { + sourceHost = normalizeHost(new URL(sourceUrl).hostname); + } catch { + return []; + } + + for (const m of html.matchAll(/]*href=["']([^"']+)["']/gi)) { + try { + const url = new URL(m[1], sourceUrl); + if (normalizeHost(url.hostname) !== sourceHost) continue; + if (!TEAM_PATH_RE.test(url.pathname)) continue; + out.add(`${url.origin}${url.pathname}`.replace(/\/$/, "")); + } catch { + // Unparseable href. + } + } + return [...out].slice(0, limit); +} diff --git a/lib/outreach/pipeline.ts b/lib/outreach/pipeline.ts index 21603028..a8f33c95 100644 --- a/lib/outreach/pipeline.ts +++ b/lib/outreach/pipeline.ts @@ -49,6 +49,7 @@ import { findContactViaSearch } from "./contactFallback"; import { loadProjectMailbox } from "./senderMailbox"; import { loadRecipientContext, recipientContextPrompt } from "./recipientContext"; import { upsertContact } from "./contacts"; +import { contactsFromDocuments, teamPageLinks } from "./documents"; export type ProspectRow = { id: string; @@ -250,6 +251,27 @@ export async function findContact(host: string): Promise { if (html && collect(html)) break; } + // The site's own pages had nothing. Two places remain that are still on + // their domain and still free, and both are tried before the search + // fallback that costs money and long before guessing an address. + if (home && !found.some((c) => c.sameDomain)) { + // A team page names people, and a named person's address beats info@ by + // enough to be worth one more fetch. + for (const link of teamPageLinks(home, `https://${host}/`)) { + const html = await fetchPage(link); + if (html && collect(html)) return dedupe(found); + } + + // And the address is often only inside a linked document — a capability + // statement or a media kit — on a page whose HTML says nothing. + const docs = await contactsFromDocuments({ + html: home, + sourceUrl: `https://${host}/`, + host, + }); + found.push(...docs.candidates); + } + return dedupe(found); } diff --git a/package-lock.json b/package-lock.json index 151552ec..91e64f76 100644 --- a/package-lock.json +++ b/package-lock.json @@ -38,6 +38,7 @@ "resend": "^4.0.0", "socks": "^2.8.9", "stripe": "^17.4.0", + "unpdf": "^1.8.0", "zod": "^3.23.8" }, "devDependencies": { @@ -6689,6 +6690,23 @@ "integrity": "sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==", "license": "MIT" }, + "node_modules/unpdf": { + "version": "1.8.0", + "resolved": "https://registry.npmjs.org/unpdf/-/unpdf-1.8.0.tgz", + "integrity": "sha512-jQlkckbe5nKxRHQdbvo9IKHlU6Cjq00UlxxB8lrCIxvS2o5IbAL1wM+4a1Yc3+BIohg9p5AsoaftwO+G4Aq/qQ==", + "license": "MIT", + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "@napi-rs/canvas": "^0.1.69 || ^1.0.0" + }, + "peerDependenciesMeta": { + "@napi-rs/canvas": { + "optional": true + } + } + }, "node_modules/unpipe": { "version": "1.0.0", "resolved": "https://registry.npmjs.org/unpipe/-/unpipe-1.0.0.tgz", diff --git a/package.json b/package.json index ae624c8b..fc9238cd 100644 --- a/package.json +++ b/package.json @@ -50,6 +50,7 @@ "resend": "^4.0.0", "socks": "^2.8.9", "stripe": "^17.4.0", + "unpdf": "^1.8.0", "zod": "^3.23.8" }, "devDependencies": { diff --git a/supabase/migrations/20260728080000_ai_spend_cron.sql b/supabase/migrations/20260728080000_ai_spend_cron.sql new file mode 100644 index 00000000..1fe1ea8b --- /dev/null +++ b/supabase/migrations/20260728080000_ai_spend_cron.sql @@ -0,0 +1,24 @@ +-- Schedule the daily AI spend warning. +-- +-- Hourly rather than daily, so a day that runs away is noticed while it is +-- still running. The alert de-duplicates on (day, threshold), so the extra +-- runs cost a query and send nothing. +-- +-- The route only ever warns — it does not throttle, pause, or block. A +-- budget alarm that turns the product off is worse than the bill it was +-- meant to prevent. + +select cron.schedule( + 'crawlproof-ai-spend', + '7 * * * *', + $$ + select net.http_post( + url := (select value from public.cron_config where key = 'site_url') || '/api/cron/ai-spend', + headers := jsonb_build_object( + 'content-type', 'application/json', + 'x-cron-secret', (select value from public.cron_config where key = 'cron_secret') + ), + body := '{}'::jsonb + ); + $$ +); diff --git a/tests/documents.test.ts b/tests/documents.test.ts new file mode 100644 index 00000000..ea820bba --- /dev/null +++ b/tests/documents.test.ts @@ -0,0 +1,76 @@ +import { describe, it, expect } from "vitest"; +import { pdfLinksFrom, teamPageLinks } from "@/lib/outreach/documents"; + +describe("pdfLinksFrom", () => { + const html = ` + Terms + Capability statement + Media kit + About`; + + it("puts documents likely to name people first", () => { + // A twelve-megabyte download is worth spending on a capability + // statement and not on a terms-and-conditions, and the filename is the + // only signal available before paying for it. + const links = pdfLinksFrom(html, "https://acme.test/", 3); + expect(links[0]).toMatch(/capability-statement|media-kit/); + expect(links[links.length - 1]).toMatch(/terms-and-conditions/); + }); + + it("resolves relative hrefs and keeps off-host documents", () => { + // A media kit on a CDN is still the company's own document. + const links = pdfLinksFrom(html, "https://acme.test/"); + expect(links).toContain("https://cdn.other.test/media-kit.pdf"); + }); + + it("ignores links that are not documents", () => { + expect(pdfLinksFrom(html, "https://acme.test/").some((l) => l.endsWith("/about"))).toBe(false); + }); + + it("honours the limit, since each link is a real download", () => { + expect(pdfLinksFrom(html, "https://acme.test/", 1)).toHaveLength(1); + }); + + it("handles a query string after the extension", () => { + const links = pdfLinksFrom(`x`, "https://acme.test/"); + expect(links[0]).toContain("brochure.pdf?v=2"); + }); + + it("returns nothing when there are no documents", () => { + expect(pdfLinksFrom(`About`, "https://acme.test/")).toEqual([]); + }); +}); + +describe("teamPageLinks", () => { + const html = ` + Our team + Leadership + Blog post + Someone else's team + About`; + + it("finds the pages that name people", () => { + const links = teamPageLinks(html, "https://acme.test/", 5); + expect(links).toContain("https://acme.test/our-team"); + expect(links).toContain("https://acme.test/leadership"); + }); + + it("stays on the prospect's own host", () => { + // Another company's team page names the wrong people entirely. + expect(teamPageLinks(html, "https://acme.test/", 5)).not.toContain("https://other.test/team"); + }); + + it("does not mistake a blog post for a team page", () => { + const links = teamPageLinks(html, "https://acme.test/", 5); + expect(links.some((l) => l.includes("/blog/"))).toBe(false); + }); + + it("normalises a trailing slash so one page is not fetched twice", () => { + const links = teamPageLinks(`TT`, "https://acme.test/"); + expect(links).toEqual(["https://acme.test/team"]); + }); + + it("returns nothing for an unparseable source", () => { + expect(teamPageLinks(html, "not a url")).toEqual([]); + }); +});