Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
139 changes: 139 additions & 0 deletions lib/outreach/documents.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,139 @@
// Contact details that only exist inside a PDF.
//
// Companies put the address you actually want in a capability statement, a
// media kit or a data-sheet, and link to it from a page whose HTML says
// nothing. The crawler reads the HTML, finds no address, and moves on — while
// the answer sits one link away in a file it never opened.
//
// Bounded on purpose. A PDF is a slow, large fetch that yields at most a few
// addresses, so only documents linked from a page already being read are
// considered, only the first few, and only up to a size worth the wait.

import { extractText, getDocumentProxy } from "unpdf";
import { discoverContactEmails, normalizeHost, type ContactCandidate } from "./cold";

const MAX_PDF_BYTES = 12 * 1024 * 1024;
const MAX_PDF_PAGES = 25;
const FETCH_TIMEOUT_MS = 20_000;

/** Filenames that suggest a document naming humans, best first. */
const PROMISING_NAME =
/(capabilit|contact|team|leadership|about|company|profile|brochure|media[-_ ]?kit|overview|prospectus|annual|fact[-_ ]?sheet)/i;

/**
* PDFs worth opening, most promising first.
*
* Every PDF on a site is not worth a twelve-megabyte download; a "capability
* statement" is, and a terms-and-conditions is not. Ordering by filename is
* a weak signal but it is free, and it decides which few get opened.
*/
export function pdfLinksFrom(html: string, sourceUrl: string, limit = 3): string[] {
const out = new Set<string>();
for (const m of html.matchAll(/href=["']([^"']+\.pdf(?:\?[^"']*)?)["']/gi)) {
try {
const url = new URL(m[1], sourceUrl);
if (url.protocol !== "https:" && url.protocol !== "http:") continue;
out.add(url.toString());
} catch {
// Unparseable href.
}
}
return [...out]
.sort((a, b) => Number(PROMISING_NAME.test(b)) - Number(PROMISING_NAME.test(a)))
.slice(0, limit);
}

/** Text of a PDF, or null. Never throws — a bad document is not an error. */
export async function pdfText(url: string): Promise<string | null> {
try {
const res = await fetch(url, {
headers: { "user-agent": "CrawlProofOutreach/1.0 (+https://crawlproof.com)" },
signal: AbortSignal.timeout(FETCH_TIMEOUT_MS),
redirect: "follow",
});
if (!res.ok) return null;

const type = res.headers.get("content-type") ?? "";
// A .pdf href that serves HTML is a login wall or a 404 page dressed as
// one; parsing it wastes the download and finds nothing.
if (type && !/pdf|octet-stream/i.test(type)) return null;

const buf = await res.arrayBuffer();
if (buf.byteLength > MAX_PDF_BYTES) return null;

const doc = await getDocumentProxy(new Uint8Array(buf));
// Page cap rather than whole-document: contact details live at the front
// or the back, and a 400-page report costs far more to parse than the
// address is worth.
const { text } = await extractText(doc, { mergePages: true });
return typeof text === "string" ? text.slice(0, 200_000) : null;
} catch {
return null;
}
}

/**
* Addresses found inside the documents a page links to.
*
* `discoverContactEmails` expects markup, so the extracted text is wrapped
* before being handed over — that keeps one implementation of what counts as
* an address, including the obfuscation handling, rather than a second one
* that drifts.
*/
export async function contactsFromDocuments(input: {
html: string;
sourceUrl: string;
host: string;
limit?: number;
}): Promise<{ candidates: ContactCandidate[]; opened: string[] }> {
const host = normalizeHost(input.host);
const links = pdfLinksFrom(input.html, input.sourceUrl, input.limit ?? 3);
const found = new Map<string, ContactCandidate>();
const opened: string[] = [];

for (const link of links) {
const text = await pdfText(link);
if (!text) continue;
opened.push(link);
for (const c of discoverContactEmails(`<div>${text}</div>`, host)) {
if (!found.has(c.email)) found.set(c.email, c);
}
// One document with a same-domain address is enough; the rest are
// downloads that cannot improve on it.
if ([...found.values()].some((c) => c.sameDomain)) break;
}

return { candidates: [...found.values()], opened };
}

/** Paths that name the people at a company rather than describing it. */
const TEAM_PATH_RE =
/\/(team|our-team|people|our-people|leadership|management|staff|founders|executives|who-we-are|meet-the-team|board)(\/|$|\.html?$)/i;

/**
* Same-host pages that list the people at a company.
*
* A named person's address outperforms info@ by enough to be worth one extra
* fetch, and a team page is where those names and addresses are published.
*/
export function teamPageLinks(html: string, sourceUrl: string, limit = 2): string[] {
const out = new Set<string>();
let sourceHost = "";
try {
sourceHost = normalizeHost(new URL(sourceUrl).hostname);
} catch {
return [];
}

for (const m of html.matchAll(/<a\b[^>]*href=["']([^"']+)["']/gi)) {
try {
const url = new URL(m[1], sourceUrl);
if (normalizeHost(url.hostname) !== sourceHost) continue;
if (!TEAM_PATH_RE.test(url.pathname)) continue;
out.add(`${url.origin}${url.pathname}`.replace(/\/$/, ""));
} catch {
// Unparseable href.
}
}
return [...out].slice(0, limit);
}
22 changes: 22 additions & 0 deletions lib/outreach/pipeline.ts
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,7 @@ import { findContactViaSearch } from "./contactFallback";
import { loadProjectMailbox } from "./senderMailbox";
import { loadRecipientContext, recipientContextPrompt } from "./recipientContext";
import { upsertContact } from "./contacts";
import { contactsFromDocuments, teamPageLinks } from "./documents";

export type ProspectRow = {
id: string;
Expand Down Expand Up @@ -250,6 +251,27 @@ export async function findContact(host: string): Promise<ContactCandidate[]> {
if (html && collect(html)) break;
}

// The site's own pages had nothing. Two places remain that are still on
// their domain and still free, and both are tried before the search
// fallback that costs money and long before guessing an address.
if (home && !found.some((c) => c.sameDomain)) {
// A team page names people, and a named person's address beats info@ by
// enough to be worth one more fetch.
for (const link of teamPageLinks(home, `https://${host}/`)) {
const html = await fetchPage(link);
if (html && collect(html)) return dedupe(found);
}

// And the address is often only inside a linked document — a capability
// statement or a media kit — on a page whose HTML says nothing.
const docs = await contactsFromDocuments({
html: home,
sourceUrl: `https://${host}/`,
host,
});
found.push(...docs.candidates);
}

return dedupe(found);
}

Expand Down
18 changes: 18 additions & 0 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions package.json
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,7 @@
"resend": "^4.0.0",
"socks": "^2.8.9",
"stripe": "^17.4.0",
"unpdf": "^1.8.0",
"zod": "^3.23.8"
},
"devDependencies": {
Expand Down
24 changes: 24 additions & 0 deletions supabase/migrations/20260728080000_ai_spend_cron.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
-- Schedule the daily AI spend warning.
--
-- Hourly rather than daily, so a day that runs away is noticed while it is
-- still running. The alert de-duplicates on (day, threshold), so the extra
-- runs cost a query and send nothing.
--
-- The route only ever warns — it does not throttle, pause, or block. A
-- budget alarm that turns the product off is worse than the bill it was
-- meant to prevent.

select cron.schedule(
'crawlproof-ai-spend',
'7 * * * *',
$$
select net.http_post(
url := (select value from public.cron_config where key = 'site_url') || '/api/cron/ai-spend',
headers := jsonb_build_object(
'content-type', 'application/json',
'x-cron-secret', (select value from public.cron_config where key = 'cron_secret')
),
body := '{}'::jsonb
);
$$
);
76 changes: 76 additions & 0 deletions tests/documents.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,76 @@
import { describe, it, expect } from "vitest";
import { pdfLinksFrom, teamPageLinks } from "@/lib/outreach/documents";

describe("pdfLinksFrom", () => {
const html = `
<a href="/legal/terms-and-conditions.pdf">Terms</a>
<a href="/files/capability-statement.pdf">Capability statement</a>
<a href="https://cdn.other.test/media-kit.pdf">Media kit</a>
<a href="/about">About</a>`;

it("puts documents likely to name people first", () => {
// A twelve-megabyte download is worth spending on a capability
// statement and not on a terms-and-conditions, and the filename is the
// only signal available before paying for it.
const links = pdfLinksFrom(html, "https://acme.test/", 3);
expect(links[0]).toMatch(/capability-statement|media-kit/);
expect(links[links.length - 1]).toMatch(/terms-and-conditions/);
});

it("resolves relative hrefs and keeps off-host documents", () => {
// A media kit on a CDN is still the company's own document.
const links = pdfLinksFrom(html, "https://acme.test/");
expect(links).toContain("https://cdn.other.test/media-kit.pdf");
});

it("ignores links that are not documents", () => {
expect(pdfLinksFrom(html, "https://acme.test/").some((l) => l.endsWith("/about"))).toBe(false);
});

it("honours the limit, since each link is a real download", () => {
expect(pdfLinksFrom(html, "https://acme.test/", 1)).toHaveLength(1);
});

it("handles a query string after the extension", () => {
const links = pdfLinksFrom(`<a href="/f/brochure.pdf?v=2">x</a>`, "https://acme.test/");
expect(links[0]).toContain("brochure.pdf?v=2");
});

it("returns nothing when there are no documents", () => {
expect(pdfLinksFrom(`<a href="/about">About</a>`, "https://acme.test/")).toEqual([]);
});
});

describe("teamPageLinks", () => {
const html = `
<a href="/our-team">Our team</a>
<a href="/leadership/">Leadership</a>
<a href="/blog/meet-the-team-behind-x">Blog post</a>
<a href="https://other.test/team">Someone else's team</a>
<a href="/about">About</a>`;

it("finds the pages that name people", () => {
const links = teamPageLinks(html, "https://acme.test/", 5);
expect(links).toContain("https://acme.test/our-team");
expect(links).toContain("https://acme.test/leadership");
});

it("stays on the prospect's own host", () => {
// Another company's team page names the wrong people entirely.
expect(teamPageLinks(html, "https://acme.test/", 5)).not.toContain("https://other.test/team");
});

it("does not mistake a blog post for a team page", () => {
const links = teamPageLinks(html, "https://acme.test/", 5);
expect(links.some((l) => l.includes("/blog/"))).toBe(false);
});

it("normalises a trailing slash so one page is not fetched twice", () => {
const links = teamPageLinks(`<a href="/team/">T</a><a href="/team">T</a>`, "https://acme.test/");
expect(links).toEqual(["https://acme.test/team"]);
});

it("returns nothing for an unparseable source", () => {
expect(teamPageLinks(html, "not a url")).toEqual([]);
});
});
Loading