From cdd8bc7788f7aef57cb9c1005f919e80052f1b2d Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Sun, 26 Jul 2026 17:46:42 +0000 Subject: [PATCH] fix(leads): find contact addresses, and stop leads stranding at "new" MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every lead showed "no contact address found". Two separate causes. 1. Nothing came back after the scan finished. Discovery queues a free scan and returns immediately, so a new lead has no findings and no contact yet. A campaign tick revisits its own leads; leads added by hand from the finder had nothing revisiting them, so they sat at "new" beside a completed scan forever. All nine leads in production were in exactly that state. Adds a "Check scans" button that re-researches whatever is waiting, and shows how many that is. 2. Contact discovery was too naive for how sites actually publish addresses. Measured against the nine real agency sites in the pipeline, it found 2/9. Diagnosing those pages showed three causes, none of which were guesses: - not one of them had a mailto: link; they all use contact forms - two hid the address behind Cloudflare's data-cfemail encoding - path guessing missed: /contact 404s where /contact-us works, and one address existed only on /privacy-policy So it now decodes Cloudflare addresses, reads "hello (at) example (dot) com" and HTML-entity forms, and follows the site's own contact-ish links instead of guessing paths — the homepage nav knows where its contact page is and we do not. Links are ranked, not first-six-wins: the naive version spent its whole budget on /about/are-we-fit and /company/block-inc while the address sat on /privacy-policy. Legal pages rank high deliberately, since a site that hides its address everywhere else still has to print it there. Same nine sites: 2/9 -> 6/9. The three misses are directories rather than agencies, so they are the wrong leads regardless. Co-Authored-By: Claude Opus 5 (1M context) --- app/(app)/projects/[id]/leads/page.tsx | 18 +++- app/actions/leads.ts | 62 ++++++++++++ components/leads/refresh-leads.tsx | 45 +++++++++ lib/outreach/cold.ts | 134 +++++++++++++++++++++++-- lib/outreach/pipeline.ts | 60 ++++++++--- tests/cold-outreach.test.ts | 64 ++++++++++++ 6 files changed, 357 insertions(+), 26 deletions(-) create mode 100644 components/leads/refresh-leads.tsx diff --git a/app/(app)/projects/[id]/leads/page.tsx b/app/(app)/projects/[id]/leads/page.tsx index 000ce683..d8ae81b1 100644 --- a/app/(app)/projects/[id]/leads/page.tsx +++ b/app/(app)/projects/[id]/leads/page.tsx @@ -7,6 +7,7 @@ import { LeadFinder } from "@/components/leads/lead-finder"; import { LeadActions } from "@/components/leads/lead-actions"; import { CampaignPanel, type CampaignSummary } from "@/components/leads/campaign-panel"; import { SenderAddress } from "@/components/leads/sender-address"; +import { RefreshLeads } from "@/components/leads/refresh-leads"; import { loadAddressSettings } from "@/lib/outreach/postalAddress"; export const metadata = { title: "Leads" }; @@ -92,6 +93,12 @@ export default async function LeadsPage({ const byStatus = new Map(); for (const p of prospects) byStatus.set(p.status, (byStatus.get(p.status) ?? 0) + 1); + // Leads whose scan may have landed since they were added, plus researched + // ones we never found an address for — what "Check scans" would work on. + const pendingCount = prospects.filter( + (p) => p.channel === "email" && (p.status === "new" || !p.contact_email), + ).length; + const since = Date.now() - 24 * 3600 * 1000; const liveToday = sends.filter((s) => !s.dry_run && new Date(s.sent_at).getTime() >= since).length; @@ -122,10 +129,13 @@ export default async function LeadsPage({

Pipeline

-

- {[...byStatus.entries()].map(([s, n]) => `${n} ${s}`).join(" · ") || "no leads yet"} ·{" "} - {liveToday}/{env.outreachDailyCap} sent today -

+
+

+ {[...byStatus.entries()].map(([s, n]) => `${n} ${s}`).join(" · ") || "no leads yet"} ·{" "} + {liveToday}/{env.outreachDailyCap} sent today +

+ +
{prospects.length === 0 ? ( diff --git a/app/actions/leads.ts b/app/actions/leads.ts index c607cb9e..a44e74d9 100644 --- a/app/actions/leads.ts +++ b/app/actions/leads.ts @@ -138,6 +138,68 @@ export async function researchLeadAction(input: { }; } +/** + * Re-run research on every lead still waiting on a scan. + * + * The gap this fills: discovery queues a free scan and returns immediately, + * so a new lead is created with no findings and no contact. Something has to + * come back once the scan lands. A campaign tick does that for its own + * leads; leads added by hand from the finder had nothing, so they sat at + * "new" forever with "no contact address found" next to a finished scan. + */ +export async function refreshLeadsAction(input: { + projectId: string; + limit?: number; +}): Promise | Err> { + const auth = await requireLeadAccess(input.projectId); + if (!auth.ok) return auth; + + const { data } = await serviceClient() + .from("outreach_prospects") + .select("target_key, site_url, campaign_id, status, contact_email") + .eq("project_id", input.projectId) + .eq("channel", "email") + .in("status", ["new", "researched"]) + .order("created_at", { ascending: true }) + .limit(Math.min(input.limit ?? 25, 50)); + + const rows = (data as Array<{ + target_key: string; + site_url: string | null; + campaign_id: string | null; + status: string; + contact_email: string | null; + }> | null) ?? []; + + // Nothing to do for leads that already have what they need. + const pending = rows.filter((r) => r.status === "new" || !r.contact_email); + if (!pending.length) { + return { ok: true, researched: 0, contacts: 0, note: "Every lead is already researched." }; + } + + let researched = 0; + let contacts = 0; + let stillScanning = 0; + for (const row of pending) { + const res = await researchProspect({ + userId: auth.userId, + projectId: input.projectId, + url: row.site_url ?? `https://${row.target_key}`, + campaignId: row.campaign_id, + }); + if (res.status === "researched") { + researched += 1; + if (res.contact) contacts += 1; + } else if (res.status === "scanning") { + stillScanning += 1; + } + } + + const parts = [`${researched} researched`, `${contacts} with a contact address`]; + if (stillScanning) parts.push(`${stillScanning} still scanning`); + return { ok: true, researched, contacts, note: parts.join(", ") + "." }; +} + export async function draftLeadAction(input: { projectId: string; host: string; diff --git a/components/leads/refresh-leads.tsx b/components/leads/refresh-leads.tsx new file mode 100644 index 00000000..fce79751 --- /dev/null +++ b/components/leads/refresh-leads.tsx @@ -0,0 +1,45 @@ +"use client"; + +import { useState, useTransition } from "react"; +import { useRouter } from "next/navigation"; +import { refreshLeadsAction } from "@/app/actions/leads"; + +/** + * "Check scans" — pulls finished scans into the leads that are waiting on + * them and looks for a contact address. + * + * Discovery queues a scan and returns immediately, so a fresh lead has no + * findings and no contact yet. This is what closes that loop for leads added + * by hand; a campaign does it on its own tick. + */ +export function RefreshLeads({ projectId, pendingCount }: { projectId: string; pendingCount: number }) { + const router = useRouter(); + const [pending, start] = useTransition(); + const [note, setNote] = useState(null); + const [error, setError] = useState(null); + + const run = () => + start(async () => { + setNote(null); + setError(null); + const res = await refreshLeadsAction({ projectId }); + if (res.ok) { + setNote(res.note); + router.refresh(); + } else setError(res.error); + }); + + return ( +
+ {note && {note}} + {error && {error}} + +
+ ); +} diff --git a/lib/outreach/cold.ts b/lib/outreach/cold.ts index 406376e6..14b975d6 100644 --- a/lib/outreach/cold.ts +++ b/lib/outreach/cold.ts @@ -158,23 +158,31 @@ export function explainSuppression(reason: SuppressionReason): string { export function discoverContactEmails(html: string, host: string): ContactCandidate[] { const apex = apexOf(host); const out = new Map(); + const add = (email: string, source: ContactCandidate["source"]) => { + if (!looksLikeEmail(email) || out.has(email)) return; + out.set(email, { email, source, sameDomain: apexOf(domainOf(email)) === apex }); + }; for (const match of html.matchAll(/mailto:([^"'?>\s]+)/gi)) { - const email = normalizeEmail(decodeURIComponent(match[1] ?? "")); - if (!looksLikeEmail(email)) continue; - out.set(email, { email, source: "mailto", sameDomain: apexOf(domainOf(email)) === apex }); + add(normalizeEmail(decodeURIComponent(match[1] ?? "")), "mailto"); + } + + // Cloudflare-protected addresses. Counted as "mailto" because that is what + // they are — an address the site deliberately published, just encoded. On + // a site that hides its address this way it is usually the only one there. + for (const match of html.matchAll(/data-cfemail=["']([0-9a-fA-F]+)["']/g)) { + const decoded = decodeCfEmail(match[1] ?? ""); + if (decoded) add(decoded, "mailto"); + } + for (const match of html.matchAll(/\/cdn-cgi\/l\/email-protection#([0-9a-fA-F]+)/g)) { + const decoded = decodeCfEmail(match[1] ?? ""); + if (decoded) add(decoded, "mailto"); } // Strip tags so we don't harvest addresses out of tracking-script config // blobs, which are mostly vendor addresses and never the owner's. const text = html.replace(//gi, " ").replace(/<[^>]+>/g, " "); - for (const match of text.matchAll(EMAIL_RE)) { - const email = normalizeEmail(match[0]); - if (!looksLikeEmail(email) || out.has(email)) continue; - // Image filenames and asset hashes routinely satisfy the address shape. - if (/\.(png|jpe?g|gif|svg|webp|css|js)$/i.test(email)) continue; - out.set(email, { email, source: "text", sameDomain: apexOf(domainOf(email)) === apex }); - } + for (const email of deobfuscateEmails(text)) add(email, "text"); return rankContacts([...out.values()]); } @@ -329,3 +337,109 @@ export function unsupportedClaims(body: string, facts: ProspectFacts): string[] } return problems; } + +// ------------------------------------------------- obfuscation & link crawl + +/** + * Cloudflare's email obfuscation. The address is hex in data-cfemail (or the + * /cdn-cgi/l/email-protection# fragment); the first byte is an XOR key for + * the rest. Worth decoding: measured on real agency sites, this is often the + * only address published anywhere on the domain. + */ +export function decodeCfEmail(hex: string): string | null { + const clean = hex.trim().toLowerCase(); + if (!/^[0-9a-f]{6,}$/.test(clean) || clean.length % 2 !== 0) return null; + const key = parseInt(clean.slice(0, 2), 16); + let out = ""; + for (let i = 2; i < clean.length; i += 2) { + out += String.fromCharCode(parseInt(clean.slice(i, i + 2), 16) ^ key); + } + return looksLikeEmail(out) ? normalizeEmail(out) : null; +} + +/** + * Undo the human-readable obfuscations. A business that writes "hello (at) + * example (dot) com" did so to stop naive scrapers, but it still wants to be + * emailed — and the address is on its own public contact page. + */ +export function deobfuscateEmails(text: string): string[] { + const normalized = text + .replace(/�?64;|@|@/gi, "@") + .replace(/�?46;|.|./gi, ".") + .replace(/\s*[([{<]\s*(?:at|@)\s*[)\]}>]\s*/gi, "@") + .replace(/\s+at\s+/gi, "@") + .replace(/\s*[([{<]\s*(?:dot|\.)\s*[)\]}>]\s*/gi, ".") + .replace(/\s+dot\s+/gi, "."); + const out = new Set(); + for (const match of normalized.matchAll(EMAIL_RE)) { + const email = normalizeEmail(match[0]); + if (looksLikeEmail(email)) out.add(email); + } + return [...out]; +} + +/** + * Where a contact address might live, best-first. + * + * Ranked rather than first-N-wins: the naive version spent its whole budget + * on /about/are-we-fit and /company/block-inc while the real address sat on + * /privacy-policy. Legal pages score high on purpose — a site that hides its + * address everywhere else still has to print it there. + * + * "company" is deliberately absent: on a directory site it matches every + * listing on the page. + */ +const LINK_PRIORITY: Array<{ re: RegExp; score: number }> = [ + { re: /contact|get-?in-?touch|reach-?us|write-?us/i, score: 10 }, + { re: /impressum|imprint|legal-?notice/i, score: 9 }, + { re: /privacy|terms|legal|gdpr/i, score: 7 }, + { re: /about/i, score: 5 }, + { re: /support|help-?cent|team|staff/i, score: 4 }, +]; + +export function contactLinksFrom(html: string, baseUrl: string): string[] { + const scored = new Map(); + let host = ""; + try { + host = new URL(baseUrl).hostname; + } catch { + return []; + } + + for (const match of html.matchAll(/]*href=["']([^"']+)["'][^>]*>([\s\S]{0,120}?)<\/a>/gi)) { + const href = match[1] ?? ""; + const text = (match[2] ?? "").replace(/<[^>]+>/g, " "); + const haystack = `${href} ${text}`; + + let score = 0; + for (const rule of LINK_PRIORITY) { + if (rule.re.test(haystack)) score = Math.max(score, rule.score); + } + if (!score) continue; + + try { + const url = new URL(href, baseUrl); + // Same site only — a "Privacy" link pointing at a third-party policy + // host is that vendor's address, not the prospect's. + if (url.hostname !== host) continue; + if (url.protocol !== "https:" && url.protocol !== "http:") continue; + url.hash = ""; + const path = url.pathname.replace(/\/$/, ""); + if (!path) continue; // the homepage we are already on + + // Prefer /about over /about/are-we-fit: the shallower page carries the + // contact details, the deeper one is marketing. + const depth = path.split("/").filter(Boolean).length; + const final = score - (depth - 1) * 2; + const absolute = url.toString(); + if ((scored.get(absolute) ?? -Infinity) < final) scored.set(absolute, final); + } catch { + // Unparseable href; skip. + } + } + + return [...scored.entries()] + .sort((a, b) => b[1] - a[1]) + .slice(0, 6) + .map(([url]) => url); +} diff --git a/lib/outreach/pipeline.ts b/lib/outreach/pipeline.ts index d04a2c5e..30ade340 100644 --- a/lib/outreach/pipeline.ts +++ b/lib/outreach/pipeline.ts @@ -27,6 +27,7 @@ import { coldOutreachEmailHtml, sendColdOutreachEmail } from "@/lib/email"; import { CONTACT_PATHS, bestContact, + contactLinksFrom, discoverContactEmails, explainSuppression, looksLikeEmail, @@ -189,29 +190,64 @@ export async function startFreeScan(input: { } /** - * Fetch a few likely pages and read the contact address the business - * publishes. Only the prospect's own site is touched — no data broker, no - * purchased list, which is both the ethical line and the reason the address - * is current. + * Read the contact address the business publishes on its own site. No data + * broker and no purchased list — which is both the ethical line and the + * reason the address is current. + * + * Follows the site's own contact-ish links rather than only guessing paths. + * Measured against nine real agency sites, guessing found 2/9: /contact 404s + * where /contact-us works, and one address only existed on /privacy-policy. + * The homepage's own nav knows where its contact page is; we don't. */ export async function findContact(host: string): Promise { const found: ContactCandidate[] = []; - for (const path of CONTACT_PATHS) { + const visited = new Set(); + + const fetchPage = async (url: string): Promise => { + if (visited.has(url) || visited.size >= 7) return null; + visited.add(url); try { - const res = await fetch(`https://${host}${path}`, { + const res = await fetch(url, { headers: { "user-agent": "CrawlProofOutreach/1.0 (+https://crawlproof.com)" }, signal: AbortSignal.timeout(8_000), redirect: "follow", }); - if (!res.ok) continue; - found.push(...discoverContactEmails(await res.text(), host)); - // Homepage plus one contact page is normally enough; stop rather than - // crawling a stranger's site hunting for addresses. - if (found.some((c) => c.sameDomain)) break; + if (!res.ok) return null; + return await res.text(); } catch { - // A dead path is normal. Try the next one. + return null; + } + }; + + const collect = (html: string) => { + found.push(...discoverContactEmails(html, host)); + return found.some((c) => c.sameDomain); + }; + + // Some hosts only answer on www; without the retry those sites yield + // nothing at all rather than a missing address. + const home = (await fetchPage(`https://${host}/`)) ?? (await fetchPage(`https://www.${host}/`)); + if (home) { + if (collect(home)) return dedupe(found); + + for (const link of contactLinksFrom(home, `https://${host}/`)) { + const html = await fetchPage(link); + if (html && collect(html)) return dedupe(found); } } + + // Backstop for sites whose nav is rendered client-side, so the homepage + // HTML carries no links to follow. + for (const path of CONTACT_PATHS) { + if (path === "/") continue; + const html = await fetchPage(`https://${host}${path}`); + if (html && collect(html)) break; + } + + return dedupe(found); +} + +function dedupe(found: ContactCandidate[]): ContactCandidate[] { const seen = new Set(); return found.filter((c) => !seen.has(c.email) && seen.add(c.email)); } diff --git a/tests/cold-outreach.test.ts b/tests/cold-outreach.test.ts index a8517223..f1ccfaf2 100644 --- a/tests/cold-outreach.test.ts +++ b/tests/cold-outreach.test.ts @@ -11,6 +11,9 @@ import { stepGuidance, suppressionReason, unsupportedClaims, + contactLinksFrom, + decodeCfEmail, + deobfuscateEmails, type ProspectFacts, } from "@/lib/outreach/cold"; import { @@ -364,3 +367,64 @@ describe("postal address precedence", () => { expect(pickPostalAddress({ account: " Me, 3 Home Rd " }).address).toBe("Me, 3 Home Rd"); }); }); + +describe("obfuscated contact discovery", () => { + it("decodes a Cloudflare-protected address", () => { + // Encoded with key 0x27, the form Cloudflare emits in data-cfemail. + const plain = "support@thomasdigital.com"; + const key = 0x27; + const hex = + key.toString(16).padStart(2, "0") + + [...plain].map((c) => (c.charCodeAt(0) ^ key).toString(16).padStart(2, "0")).join(""); + expect(decodeCfEmail(hex)).toBe(plain); + }); + + it("rejects junk rather than emitting a garbage address", () => { + expect(decodeCfEmail("zzzz")).toBeNull(); + expect(decodeCfEmail("27")).toBeNull(); + expect(decodeCfEmail("2712")).toBeNull(); + }); + + it("reads addresses written to defeat scrapers", () => { + expect(deobfuscateEmails("hello (at) example (dot) com")).toContain("hello@example.com"); + expect(deobfuscateEmails("jane at example dot com")).toContain("jane@example.com"); + expect(deobfuscateEmails("team@example.com")).toContain("team@example.com"); + }); + + it("finds nothing in prose that merely contains the word at", () => { + expect(deobfuscateEmails("We are at the office today")).toEqual([]); + }); +}); + +describe("contactLinksFrom", () => { + const base = "https://example.com/"; + + it("ranks a contact page above a deep marketing page", () => { + const html = ` + Are we a fit? + Contact + About`; + expect(contactLinksFrom(html, base)[0]).toBe("https://example.com/contact-us"); + }); + + it("keeps legal pages, where a hidden address usually is", () => { + const html = `PrivacyBlog`; + expect(contactLinksFrom(html, base)).toEqual(["https://example.com/privacy-policy"]); + }); + + it("does not treat directory listings as contact pages", () => { + // "company" matched every listing on a jobs directory and ate the budget. + const html = `BlockBraze`; + expect(contactLinksFrom(html, base)).toEqual([]); + }); + + it("prefers the shallower page when both match", () => { + const html = `LeadershipAbout`; + expect(contactLinksFrom(html, base)[0]).toBe("https://example.com/about"); + }); + + it("ignores off-site links and the page itself", () => { + const html = `PrivacyContact home`; + expect(contactLinksFrom(html, base)).toEqual([]); + }); +});