Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 14 additions & 4 deletions app/(app)/projects/[id]/leads/page.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@ import { LeadFinder } from "@/components/leads/lead-finder";
import { LeadActions } from "@/components/leads/lead-actions";
import { CampaignPanel, type CampaignSummary } from "@/components/leads/campaign-panel";
import { SenderAddress } from "@/components/leads/sender-address";
import { RefreshLeads } from "@/components/leads/refresh-leads";
import { loadAddressSettings } from "@/lib/outreach/postalAddress";

export const metadata = { title: "Leads" };
Expand Down Expand Up @@ -92,6 +93,12 @@ export default async function LeadsPage({
const byStatus = new Map<string, number>();
for (const p of prospects) byStatus.set(p.status, (byStatus.get(p.status) ?? 0) + 1);

// Leads whose scan may have landed since they were added, plus researched
// ones we never found an address for — what "Check scans" would work on.
const pendingCount = prospects.filter(
(p) => p.channel === "email" && (p.status === "new" || !p.contact_email),
).length;

const since = Date.now() - 24 * 3600 * 1000;
const liveToday = sends.filter((s) => !s.dry_run && new Date(s.sent_at).getTime() >= since).length;

Expand Down Expand Up @@ -122,10 +129,13 @@ export default async function LeadsPage({
<section className="card p-4">
<div className="flex flex-wrap items-baseline justify-between gap-2">
<h2 className="text-lg font-semibold">Pipeline</h2>
<p className="text-sm text-[var(--color-muted)]">
{[...byStatus.entries()].map(([s, n]) => `${n} ${s}`).join(" · ") || "no leads yet"} ·{" "}
{liveToday}/{env.outreachDailyCap} sent today
</p>
<div className="flex flex-wrap items-center gap-3">
<p className="text-sm text-[var(--color-muted)]">
{[...byStatus.entries()].map(([s, n]) => `${n} ${s}`).join(" · ") || "no leads yet"} ·{" "}
{liveToday}/{env.outreachDailyCap} sent today
</p>
<RefreshLeads projectId={projectId} pendingCount={pendingCount} />
</div>
</div>

{prospects.length === 0 ? (
Expand Down
62 changes: 62 additions & 0 deletions app/actions/leads.ts
Original file line number Diff line number Diff line change
Expand Up @@ -138,6 +138,68 @@ export async function researchLeadAction(input: {
};
}

/**
* Re-run research on every lead still waiting on a scan.
*
* The gap this fills: discovery queues a free scan and returns immediately,
* so a new lead is created with no findings and no contact. Something has to
* come back once the scan lands. A campaign tick does that for its own
* leads; leads added by hand from the finder had nothing, so they sat at
* "new" forever with "no contact address found" next to a finished scan.
*/
export async function refreshLeadsAction(input: {
projectId: string;
limit?: number;
}): Promise<Ok<{ note: string; researched: number; contacts: number }> | Err> {
const auth = await requireLeadAccess(input.projectId);
if (!auth.ok) return auth;

const { data } = await serviceClient()
.from("outreach_prospects")
.select("target_key, site_url, campaign_id, status, contact_email")
.eq("project_id", input.projectId)
.eq("channel", "email")
.in("status", ["new", "researched"])
.order("created_at", { ascending: true })
.limit(Math.min(input.limit ?? 25, 50));

const rows = (data as Array<{
target_key: string;
site_url: string | null;
campaign_id: string | null;
status: string;
contact_email: string | null;
}> | null) ?? [];

// Nothing to do for leads that already have what they need.
const pending = rows.filter((r) => r.status === "new" || !r.contact_email);
if (!pending.length) {
return { ok: true, researched: 0, contacts: 0, note: "Every lead is already researched." };
}

let researched = 0;
let contacts = 0;
let stillScanning = 0;
for (const row of pending) {
const res = await researchProspect({
userId: auth.userId,
projectId: input.projectId,
url: row.site_url ?? `https://${row.target_key}`,
campaignId: row.campaign_id,
});
if (res.status === "researched") {
researched += 1;
if (res.contact) contacts += 1;
} else if (res.status === "scanning") {
stillScanning += 1;
}
}

const parts = [`${researched} researched`, `${contacts} with a contact address`];
if (stillScanning) parts.push(`${stillScanning} still scanning`);
return { ok: true, researched, contacts, note: parts.join(", ") + "." };
}

export async function draftLeadAction(input: {
projectId: string;
host: string;
Expand Down
45 changes: 45 additions & 0 deletions components/leads/refresh-leads.tsx
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
"use client";

import { useState, useTransition } from "react";
import { useRouter } from "next/navigation";
import { refreshLeadsAction } from "@/app/actions/leads";

/**
* "Check scans" — pulls finished scans into the leads that are waiting on
* them and looks for a contact address.
*
* Discovery queues a scan and returns immediately, so a fresh lead has no
* findings and no contact yet. This is what closes that loop for leads added
* by hand; a campaign does it on its own tick.
*/
export function RefreshLeads({ projectId, pendingCount }: { projectId: string; pendingCount: number }) {
const router = useRouter();
const [pending, start] = useTransition();
const [note, setNote] = useState<string | null>(null);
const [error, setError] = useState<string | null>(null);

const run = () =>
start(async () => {
setNote(null);
setError(null);
const res = await refreshLeadsAction({ projectId });
if (res.ok) {
setNote(res.note);
router.refresh();
} else setError(res.error);
});

return (
<div className="flex flex-wrap items-center justify-end gap-2">
{note && <span className="text-xs text-[var(--color-muted)]">{note}</span>}
{error && <span className="text-xs text-[var(--color-danger,#f87171)]">{error}</span>}
<button onClick={run} disabled={pending || pendingCount === 0} className="btn text-sm">
{pending
? "Checking…"
: pendingCount > 0
? `Check scans (${pendingCount})`
: "All researched"}
</button>
</div>
);
}
134 changes: 124 additions & 10 deletions lib/outreach/cold.ts
Original file line number Diff line number Diff line change
Expand Up @@ -158,23 +158,31 @@ export function explainSuppression(reason: SuppressionReason): string {
export function discoverContactEmails(html: string, host: string): ContactCandidate[] {
const apex = apexOf(host);
const out = new Map<string, ContactCandidate>();
const add = (email: string, source: ContactCandidate["source"]) => {
if (!looksLikeEmail(email) || out.has(email)) return;
out.set(email, { email, source, sameDomain: apexOf(domainOf(email)) === apex });
};

for (const match of html.matchAll(/mailto:([^"'?>\s]+)/gi)) {
const email = normalizeEmail(decodeURIComponent(match[1] ?? ""));
if (!looksLikeEmail(email)) continue;
out.set(email, { email, source: "mailto", sameDomain: apexOf(domainOf(email)) === apex });
add(normalizeEmail(decodeURIComponent(match[1] ?? "")), "mailto");
}

// Cloudflare-protected addresses. Counted as "mailto" because that is what
// they are — an address the site deliberately published, just encoded. On
// a site that hides its address this way it is usually the only one there.
for (const match of html.matchAll(/data-cfemail=["']([0-9a-fA-F]+)["']/g)) {
const decoded = decodeCfEmail(match[1] ?? "");
if (decoded) add(decoded, "mailto");
}
for (const match of html.matchAll(/\/cdn-cgi\/l\/email-protection#([0-9a-fA-F]+)/g)) {
const decoded = decodeCfEmail(match[1] ?? "");
if (decoded) add(decoded, "mailto");
}

// Strip tags so we don't harvest addresses out of tracking-script config
// blobs, which are mostly vendor addresses and never the owner's.
const text = html.replace(/<script[\s\S]*?<\/script>/gi, " ").replace(/<[^>]+>/g, " ");
for (const match of text.matchAll(EMAIL_RE)) {
const email = normalizeEmail(match[0]);
if (!looksLikeEmail(email) || out.has(email)) continue;
// Image filenames and asset hashes routinely satisfy the address shape.
if (/\.(png|jpe?g|gif|svg|webp|css|js)$/i.test(email)) continue;
out.set(email, { email, source: "text", sameDomain: apexOf(domainOf(email)) === apex });
}
for (const email of deobfuscateEmails(text)) add(email, "text");

return rankContacts([...out.values()]);
}
Expand Down Expand Up @@ -329,3 +337,109 @@ export function unsupportedClaims(body: string, facts: ProspectFacts): string[]
}
return problems;
}

// ------------------------------------------------- obfuscation & link crawl

/**
* Cloudflare's email obfuscation. The address is hex in data-cfemail (or the
* /cdn-cgi/l/email-protection# fragment); the first byte is an XOR key for
* the rest. Worth decoding: measured on real agency sites, this is often the
* only address published anywhere on the domain.
*/
export function decodeCfEmail(hex: string): string | null {
const clean = hex.trim().toLowerCase();
if (!/^[0-9a-f]{6,}$/.test(clean) || clean.length % 2 !== 0) return null;
const key = parseInt(clean.slice(0, 2), 16);
let out = "";
for (let i = 2; i < clean.length; i += 2) {
out += String.fromCharCode(parseInt(clean.slice(i, i + 2), 16) ^ key);
}
return looksLikeEmail(out) ? normalizeEmail(out) : null;
}

/**
* Undo the human-readable obfuscations. A business that writes "hello (at)
* example (dot) com" did so to stop naive scrapers, but it still wants to be
* emailed — and the address is on its own public contact page.
*/
export function deobfuscateEmails(text: string): string[] {
const normalized = text
.replace(/&#0?64;|&commat;|&#x40;/gi, "@")
.replace(/&#0?46;|&period;|&#x2e;/gi, ".")
.replace(/\s*[([{<]\s*(?:at|@)\s*[)\]}>]\s*/gi, "@")
.replace(/\s+at\s+/gi, "@")
.replace(/\s*[([{<]\s*(?:dot|\.)\s*[)\]}>]\s*/gi, ".")
.replace(/\s+dot\s+/gi, ".");
const out = new Set<string>();
for (const match of normalized.matchAll(EMAIL_RE)) {
const email = normalizeEmail(match[0]);
if (looksLikeEmail(email)) out.add(email);
}
return [...out];
}

/**
* Where a contact address might live, best-first.
*
* Ranked rather than first-N-wins: the naive version spent its whole budget
* on /about/are-we-fit and /company/block-inc while the real address sat on
* /privacy-policy. Legal pages score high on purpose — a site that hides its
* address everywhere else still has to print it there.
*
* "company" is deliberately absent: on a directory site it matches every
* listing on the page.
*/
const LINK_PRIORITY: Array<{ re: RegExp; score: number }> = [
{ re: /contact|get-?in-?touch|reach-?us|write-?us/i, score: 10 },
{ re: /impressum|imprint|legal-?notice/i, score: 9 },
{ re: /privacy|terms|legal|gdpr/i, score: 7 },
{ re: /about/i, score: 5 },
{ re: /support|help-?cent|team|staff/i, score: 4 },
];

export function contactLinksFrom(html: string, baseUrl: string): string[] {
const scored = new Map<string, number>();
let host = "";
try {
host = new URL(baseUrl).hostname;
} catch {
return [];
}

for (const match of html.matchAll(/<a\b[^>]*href=["']([^"']+)["'][^>]*>([\s\S]{0,120}?)<\/a>/gi)) {
const href = match[1] ?? "";
const text = (match[2] ?? "").replace(/<[^>]+>/g, " ");
const haystack = `${href} ${text}`;

let score = 0;
for (const rule of LINK_PRIORITY) {
if (rule.re.test(haystack)) score = Math.max(score, rule.score);
}
if (!score) continue;

try {
const url = new URL(href, baseUrl);
// Same site only — a "Privacy" link pointing at a third-party policy
// host is that vendor's address, not the prospect's.
if (url.hostname !== host) continue;
if (url.protocol !== "https:" && url.protocol !== "http:") continue;
url.hash = "";
const path = url.pathname.replace(/\/$/, "");
if (!path) continue; // the homepage we are already on

// Prefer /about over /about/are-we-fit: the shallower page carries the
// contact details, the deeper one is marketing.
const depth = path.split("/").filter(Boolean).length;
const final = score - (depth - 1) * 2;
const absolute = url.toString();
if ((scored.get(absolute) ?? -Infinity) < final) scored.set(absolute, final);
} catch {
// Unparseable href; skip.
}
}

return [...scored.entries()]
.sort((a, b) => b[1] - a[1])
.slice(0, 6)
.map(([url]) => url);
}
60 changes: 48 additions & 12 deletions lib/outreach/pipeline.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ import { coldOutreachEmailHtml, sendColdOutreachEmail } from "@/lib/email";
import {
CONTACT_PATHS,
bestContact,
contactLinksFrom,
discoverContactEmails,
explainSuppression,
looksLikeEmail,
Expand Down Expand Up @@ -189,29 +190,64 @@ export async function startFreeScan(input: {
}

/**
* Fetch a few likely pages and read the contact address the business
* publishes. Only the prospect's own site is touched — no data broker, no
* purchased list, which is both the ethical line and the reason the address
* is current.
* Read the contact address the business publishes on its own site. No data
* broker and no purchased list — which is both the ethical line and the
* reason the address is current.
*
* Follows the site's own contact-ish links rather than only guessing paths.
* Measured against nine real agency sites, guessing found 2/9: /contact 404s
* where /contact-us works, and one address only existed on /privacy-policy.
* The homepage's own nav knows where its contact page is; we don't.
*/
export async function findContact(host: string): Promise<ContactCandidate[]> {
const found: ContactCandidate[] = [];
for (const path of CONTACT_PATHS) {
const visited = new Set<string>();

const fetchPage = async (url: string): Promise<string | null> => {
if (visited.has(url) || visited.size >= 7) return null;
visited.add(url);
try {
const res = await fetch(`https://${host}${path}`, {
const res = await fetch(url, {
headers: { "user-agent": "CrawlProofOutreach/1.0 (+https://crawlproof.com)" },
signal: AbortSignal.timeout(8_000),
redirect: "follow",
});
if (!res.ok) continue;
found.push(...discoverContactEmails(await res.text(), host));
// Homepage plus one contact page is normally enough; stop rather than
// crawling a stranger's site hunting for addresses.
if (found.some((c) => c.sameDomain)) break;
if (!res.ok) return null;
return await res.text();
} catch {
// A dead path is normal. Try the next one.
return null;
}
};

const collect = (html: string) => {
found.push(...discoverContactEmails(html, host));
return found.some((c) => c.sameDomain);
};

// Some hosts only answer on www; without the retry those sites yield
// nothing at all rather than a missing address.
const home = (await fetchPage(`https://${host}/`)) ?? (await fetchPage(`https://www.${host}/`));
if (home) {
if (collect(home)) return dedupe(found);

for (const link of contactLinksFrom(home, `https://${host}/`)) {
const html = await fetchPage(link);
if (html && collect(html)) return dedupe(found);
}
}

// Backstop for sites whose nav is rendered client-side, so the homepage
// HTML carries no links to follow.
for (const path of CONTACT_PATHS) {
if (path === "/") continue;
const html = await fetchPage(`https://${host}${path}`);
if (html && collect(html)) break;
}

return dedupe(found);
}

function dedupe(found: ContactCandidate[]): ContactCandidate[] {
const seen = new Set<string>();
return found.filter((c) => !seen.has(c.email) && seen.add(c.email));
}
Expand Down
Loading
Loading