diff --git a/app/(app)/projects/[id]/leads/page.tsx b/app/(app)/projects/[id]/leads/page.tsx index 000ce683..d8ae81b1 100644 --- a/app/(app)/projects/[id]/leads/page.tsx +++ b/app/(app)/projects/[id]/leads/page.tsx @@ -7,6 +7,7 @@ import { LeadFinder } from "@/components/leads/lead-finder"; import { LeadActions } from "@/components/leads/lead-actions"; import { CampaignPanel, type CampaignSummary } from "@/components/leads/campaign-panel"; import { SenderAddress } from "@/components/leads/sender-address"; +import { RefreshLeads } from "@/components/leads/refresh-leads"; import { loadAddressSettings } from "@/lib/outreach/postalAddress"; export const metadata = { title: "Leads" }; @@ -92,6 +93,12 @@ export default async function LeadsPage({ const byStatus = new Map(); for (const p of prospects) byStatus.set(p.status, (byStatus.get(p.status) ?? 0) + 1); + // Leads whose scan may have landed since they were added, plus researched + // ones we never found an address for — what "Check scans" would work on. + const pendingCount = prospects.filter( + (p) => p.channel === "email" && (p.status === "new" || !p.contact_email), + ).length; + const since = Date.now() - 24 * 3600 * 1000; const liveToday = sends.filter((s) => !s.dry_run && new Date(s.sent_at).getTime() >= since).length; @@ -122,10 +129,13 @@ export default async function LeadsPage({

Pipeline

-

- {[...byStatus.entries()].map(([s, n]) => `${n} ${s}`).join(" · ") || "no leads yet"} ·{" "} - {liveToday}/{env.outreachDailyCap} sent today -

+
+

+ {[...byStatus.entries()].map(([s, n]) => `${n} ${s}`).join(" · ") || "no leads yet"} ·{" "} + {liveToday}/{env.outreachDailyCap} sent today +

+ +
{prospects.length === 0 ? ( diff --git a/app/actions/leads.ts b/app/actions/leads.ts index c607cb9e..a44e74d9 100644 --- a/app/actions/leads.ts +++ b/app/actions/leads.ts @@ -138,6 +138,68 @@ export async function researchLeadAction(input: { }; } +/** + * Re-run research on every lead still waiting on a scan. + * + * The gap this fills: discovery queues a free scan and returns immediately, + * so a new lead is created with no findings and no contact. Something has to + * come back once the scan lands. A campaign tick does that for its own + * leads; leads added by hand from the finder had nothing, so they sat at + * "new" forever with "no contact address found" next to a finished scan. + */ +export async function refreshLeadsAction(input: { + projectId: string; + limit?: number; +}): Promise | Err> { + const auth = await requireLeadAccess(input.projectId); + if (!auth.ok) return auth; + + const { data } = await serviceClient() + .from("outreach_prospects") + .select("target_key, site_url, campaign_id, status, contact_email") + .eq("project_id", input.projectId) + .eq("channel", "email") + .in("status", ["new", "researched"]) + .order("created_at", { ascending: true }) + .limit(Math.min(input.limit ?? 25, 50)); + + const rows = (data as Array<{ + target_key: string; + site_url: string | null; + campaign_id: string | null; + status: string; + contact_email: string | null; + }> | null) ?? []; + + // Nothing to do for leads that already have what they need. + const pending = rows.filter((r) => r.status === "new" || !r.contact_email); + if (!pending.length) { + return { ok: true, researched: 0, contacts: 0, note: "Every lead is already researched." }; + } + + let researched = 0; + let contacts = 0; + let stillScanning = 0; + for (const row of pending) { + const res = await researchProspect({ + userId: auth.userId, + projectId: input.projectId, + url: row.site_url ?? `https://${row.target_key}`, + campaignId: row.campaign_id, + }); + if (res.status === "researched") { + researched += 1; + if (res.contact) contacts += 1; + } else if (res.status === "scanning") { + stillScanning += 1; + } + } + + const parts = [`${researched} researched`, `${contacts} with a contact address`]; + if (stillScanning) parts.push(`${stillScanning} still scanning`); + return { ok: true, researched, contacts, note: parts.join(", ") + "." }; +} + export async function draftLeadAction(input: { projectId: string; host: string; diff --git a/components/leads/refresh-leads.tsx b/components/leads/refresh-leads.tsx new file mode 100644 index 00000000..fce79751 --- /dev/null +++ b/components/leads/refresh-leads.tsx @@ -0,0 +1,45 @@ +"use client"; + +import { useState, useTransition } from "react"; +import { useRouter } from "next/navigation"; +import { refreshLeadsAction } from "@/app/actions/leads"; + +/** + * "Check scans" — pulls finished scans into the leads that are waiting on + * them and looks for a contact address. + * + * Discovery queues a scan and returns immediately, so a fresh lead has no + * findings and no contact yet. This is what closes that loop for leads added + * by hand; a campaign does it on its own tick. + */ +export function RefreshLeads({ projectId, pendingCount }: { projectId: string; pendingCount: number }) { + const router = useRouter(); + const [pending, start] = useTransition(); + const [note, setNote] = useState(null); + const [error, setError] = useState(null); + + const run = () => + start(async () => { + setNote(null); + setError(null); + const res = await refreshLeadsAction({ projectId }); + if (res.ok) { + setNote(res.note); + router.refresh(); + } else setError(res.error); + }); + + return ( +
+ {note && {note}} + {error && {error}} + +
+ ); +} diff --git a/lib/outreach/cold.ts b/lib/outreach/cold.ts index 406376e6..14b975d6 100644 --- a/lib/outreach/cold.ts +++ b/lib/outreach/cold.ts @@ -158,23 +158,31 @@ export function explainSuppression(reason: SuppressionReason): string { export function discoverContactEmails(html: string, host: string): ContactCandidate[] { const apex = apexOf(host); const out = new Map(); + const add = (email: string, source: ContactCandidate["source"]) => { + if (!looksLikeEmail(email) || out.has(email)) return; + out.set(email, { email, source, sameDomain: apexOf(domainOf(email)) === apex }); + }; for (const match of html.matchAll(/mailto:([^"'?>\s]+)/gi)) { - const email = normalizeEmail(decodeURIComponent(match[1] ?? "")); - if (!looksLikeEmail(email)) continue; - out.set(email, { email, source: "mailto", sameDomain: apexOf(domainOf(email)) === apex }); + add(normalizeEmail(decodeURIComponent(match[1] ?? "")), "mailto"); + } + + // Cloudflare-protected addresses. Counted as "mailto" because that is what + // they are — an address the site deliberately published, just encoded. On + // a site that hides its address this way it is usually the only one there. + for (const match of html.matchAll(/data-cfemail=["']([0-9a-fA-F]+)["']/g)) { + const decoded = decodeCfEmail(match[1] ?? ""); + if (decoded) add(decoded, "mailto"); + } + for (const match of html.matchAll(/\/cdn-cgi\/l\/email-protection#([0-9a-fA-F]+)/g)) { + const decoded = decodeCfEmail(match[1] ?? ""); + if (decoded) add(decoded, "mailto"); } // Strip tags so we don't harvest addresses out of tracking-script config // blobs, which are mostly vendor addresses and never the owner's. const text = html.replace(//gi, " ").replace(/<[^>]+>/g, " "); - for (const match of text.matchAll(EMAIL_RE)) { - const email = normalizeEmail(match[0]); - if (!looksLikeEmail(email) || out.has(email)) continue; - // Image filenames and asset hashes routinely satisfy the address shape. - if (/\.(png|jpe?g|gif|svg|webp|css|js)$/i.test(email)) continue; - out.set(email, { email, source: "text", sameDomain: apexOf(domainOf(email)) === apex }); - } + for (const email of deobfuscateEmails(text)) add(email, "text"); return rankContacts([...out.values()]); } @@ -329,3 +337,109 @@ export function unsupportedClaims(body: string, facts: ProspectFacts): string[] } return problems; } + +// ------------------------------------------------- obfuscation & link crawl + +/** + * Cloudflare's email obfuscation. The address is hex in data-cfemail (or the + * /cdn-cgi/l/email-protection# fragment); the first byte is an XOR key for + * the rest. Worth decoding: measured on real agency sites, this is often the + * only address published anywhere on the domain. + */ +export function decodeCfEmail(hex: string): string | null { + const clean = hex.trim().toLowerCase(); + if (!/^[0-9a-f]{6,}$/.test(clean) || clean.length % 2 !== 0) return null; + const key = parseInt(clean.slice(0, 2), 16); + let out = ""; + for (let i = 2; i < clean.length; i += 2) { + out += String.fromCharCode(parseInt(clean.slice(i, i + 2), 16) ^ key); + } + return looksLikeEmail(out) ? normalizeEmail(out) : null; +} + +/** + * Undo the human-readable obfuscations. A business that writes "hello (at) + * example (dot) com" did so to stop naive scrapers, but it still wants to be + * emailed — and the address is on its own public contact page. + */ +export function deobfuscateEmails(text: string): string[] { + const normalized = text + .replace(/�?64;|@|@/gi, "@") + .replace(/�?46;|.|./gi, ".") + .replace(/\s*[([{<]\s*(?:at|@)\s*[)\]}>]\s*/gi, "@") + .replace(/\s+at\s+/gi, "@") + .replace(/\s*[([{<]\s*(?:dot|\.)\s*[)\]}>]\s*/gi, ".") + .replace(/\s+dot\s+/gi, "."); + const out = new Set(); + for (const match of normalized.matchAll(EMAIL_RE)) { + const email = normalizeEmail(match[0]); + if (looksLikeEmail(email)) out.add(email); + } + return [...out]; +} + +/** + * Where a contact address might live, best-first. + * + * Ranked rather than first-N-wins: the naive version spent its whole budget + * on /about/are-we-fit and /company/block-inc while the real address sat on + * /privacy-policy. Legal pages score high on purpose — a site that hides its + * address everywhere else still has to print it there. + * + * "company" is deliberately absent: on a directory site it matches every + * listing on the page. + */ +const LINK_PRIORITY: Array<{ re: RegExp; score: number }> = [ + { re: /contact|get-?in-?touch|reach-?us|write-?us/i, score: 10 }, + { re: /impressum|imprint|legal-?notice/i, score: 9 }, + { re: /privacy|terms|legal|gdpr/i, score: 7 }, + { re: /about/i, score: 5 }, + { re: /support|help-?cent|team|staff/i, score: 4 }, +]; + +export function contactLinksFrom(html: string, baseUrl: string): string[] { + const scored = new Map(); + let host = ""; + try { + host = new URL(baseUrl).hostname; + } catch { + return []; + } + + for (const match of html.matchAll(/]*href=["']([^"']+)["'][^>]*>([\s\S]{0,120}?)<\/a>/gi)) { + const href = match[1] ?? ""; + const text = (match[2] ?? "").replace(/<[^>]+>/g, " "); + const haystack = `${href} ${text}`; + + let score = 0; + for (const rule of LINK_PRIORITY) { + if (rule.re.test(haystack)) score = Math.max(score, rule.score); + } + if (!score) continue; + + try { + const url = new URL(href, baseUrl); + // Same site only — a "Privacy" link pointing at a third-party policy + // host is that vendor's address, not the prospect's. + if (url.hostname !== host) continue; + if (url.protocol !== "https:" && url.protocol !== "http:") continue; + url.hash = ""; + const path = url.pathname.replace(/\/$/, ""); + if (!path) continue; // the homepage we are already on + + // Prefer /about over /about/are-we-fit: the shallower page carries the + // contact details, the deeper one is marketing. + const depth = path.split("/").filter(Boolean).length; + const final = score - (depth - 1) * 2; + const absolute = url.toString(); + if ((scored.get(absolute) ?? -Infinity) < final) scored.set(absolute, final); + } catch { + // Unparseable href; skip. + } + } + + return [...scored.entries()] + .sort((a, b) => b[1] - a[1]) + .slice(0, 6) + .map(([url]) => url); +} diff --git a/lib/outreach/pipeline.ts b/lib/outreach/pipeline.ts index d04a2c5e..30ade340 100644 --- a/lib/outreach/pipeline.ts +++ b/lib/outreach/pipeline.ts @@ -27,6 +27,7 @@ import { coldOutreachEmailHtml, sendColdOutreachEmail } from "@/lib/email"; import { CONTACT_PATHS, bestContact, + contactLinksFrom, discoverContactEmails, explainSuppression, looksLikeEmail, @@ -189,29 +190,64 @@ export async function startFreeScan(input: { } /** - * Fetch a few likely pages and read the contact address the business - * publishes. Only the prospect's own site is touched — no data broker, no - * purchased list, which is both the ethical line and the reason the address - * is current. + * Read the contact address the business publishes on its own site. No data + * broker and no purchased list — which is both the ethical line and the + * reason the address is current. + * + * Follows the site's own contact-ish links rather than only guessing paths. + * Measured against nine real agency sites, guessing found 2/9: /contact 404s + * where /contact-us works, and one address only existed on /privacy-policy. + * The homepage's own nav knows where its contact page is; we don't. */ export async function findContact(host: string): Promise { const found: ContactCandidate[] = []; - for (const path of CONTACT_PATHS) { + const visited = new Set(); + + const fetchPage = async (url: string): Promise => { + if (visited.has(url) || visited.size >= 7) return null; + visited.add(url); try { - const res = await fetch(`https://${host}${path}`, { + const res = await fetch(url, { headers: { "user-agent": "CrawlProofOutreach/1.0 (+https://crawlproof.com)" }, signal: AbortSignal.timeout(8_000), redirect: "follow", }); - if (!res.ok) continue; - found.push(...discoverContactEmails(await res.text(), host)); - // Homepage plus one contact page is normally enough; stop rather than - // crawling a stranger's site hunting for addresses. - if (found.some((c) => c.sameDomain)) break; + if (!res.ok) return null; + return await res.text(); } catch { - // A dead path is normal. Try the next one. + return null; + } + }; + + const collect = (html: string) => { + found.push(...discoverContactEmails(html, host)); + return found.some((c) => c.sameDomain); + }; + + // Some hosts only answer on www; without the retry those sites yield + // nothing at all rather than a missing address. + const home = (await fetchPage(`https://${host}/`)) ?? (await fetchPage(`https://www.${host}/`)); + if (home) { + if (collect(home)) return dedupe(found); + + for (const link of contactLinksFrom(home, `https://${host}/`)) { + const html = await fetchPage(link); + if (html && collect(html)) return dedupe(found); } } + + // Backstop for sites whose nav is rendered client-side, so the homepage + // HTML carries no links to follow. + for (const path of CONTACT_PATHS) { + if (path === "/") continue; + const html = await fetchPage(`https://${host}${path}`); + if (html && collect(html)) break; + } + + return dedupe(found); +} + +function dedupe(found: ContactCandidate[]): ContactCandidate[] { const seen = new Set(); return found.filter((c) => !seen.has(c.email) && seen.add(c.email)); } diff --git a/tests/cold-outreach.test.ts b/tests/cold-outreach.test.ts index a8517223..f1ccfaf2 100644 --- a/tests/cold-outreach.test.ts +++ b/tests/cold-outreach.test.ts @@ -11,6 +11,9 @@ import { stepGuidance, suppressionReason, unsupportedClaims, + contactLinksFrom, + decodeCfEmail, + deobfuscateEmails, type ProspectFacts, } from "@/lib/outreach/cold"; import { @@ -364,3 +367,64 @@ describe("postal address precedence", () => { expect(pickPostalAddress({ account: " Me, 3 Home Rd " }).address).toBe("Me, 3 Home Rd"); }); }); + +describe("obfuscated contact discovery", () => { + it("decodes a Cloudflare-protected address", () => { + // Encoded with key 0x27, the form Cloudflare emits in data-cfemail. + const plain = "support@thomasdigital.com"; + const key = 0x27; + const hex = + key.toString(16).padStart(2, "0") + + [...plain].map((c) => (c.charCodeAt(0) ^ key).toString(16).padStart(2, "0")).join(""); + expect(decodeCfEmail(hex)).toBe(plain); + }); + + it("rejects junk rather than emitting a garbage address", () => { + expect(decodeCfEmail("zzzz")).toBeNull(); + expect(decodeCfEmail("27")).toBeNull(); + expect(decodeCfEmail("2712")).toBeNull(); + }); + + it("reads addresses written to defeat scrapers", () => { + expect(deobfuscateEmails("hello (at) example (dot) com")).toContain("hello@example.com"); + expect(deobfuscateEmails("jane at example dot com")).toContain("jane@example.com"); + expect(deobfuscateEmails("team@example.com")).toContain("team@example.com"); + }); + + it("finds nothing in prose that merely contains the word at", () => { + expect(deobfuscateEmails("We are at the office today")).toEqual([]); + }); +}); + +describe("contactLinksFrom", () => { + const base = "https://example.com/"; + + it("ranks a contact page above a deep marketing page", () => { + const html = ` + Are we a fit? + Contact + About`; + expect(contactLinksFrom(html, base)[0]).toBe("https://example.com/contact-us"); + }); + + it("keeps legal pages, where a hidden address usually is", () => { + const html = `PrivacyBlog`; + expect(contactLinksFrom(html, base)).toEqual(["https://example.com/privacy-policy"]); + }); + + it("does not treat directory listings as contact pages", () => { + // "company" matched every listing on a jobs directory and ate the budget. + const html = `BlockBraze`; + expect(contactLinksFrom(html, base)).toEqual([]); + }); + + it("prefers the shallower page when both match", () => { + const html = `LeadershipAbout`; + expect(contactLinksFrom(html, base)[0]).toBe("https://example.com/about"); + }); + + it("ignores off-site links and the page itself", () => { + const html = `PrivacyContact home`; + expect(contactLinksFrom(html, base)).toEqual([]); + }); +});