From 472fba28b6c1a03b3fdef810e8c5d2dbafac34e6 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 13:38:44 +0000 Subject: [PATCH] feat(leads): record contacts on both research paths, and let the list out MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit outreach_contacts held nine rows against seventy-seven addressable leads. The cause was a missing call rather than a broken one: researchProspect has two branches, and only the unscanned one recorded a contact. Almost every lead arrived through the scanning branch, which wrote the address onto the prospect and nowhere else — so the durable, org-scoped record the table exists to be was very nearly empty, while the data itself sat in project-scoped prospect rows that have to be read one project at a time. Both branches record now, and the scanning branch links the prospect back to the contact so the relationship works in both directions. Contacts also carry a niche, which they never did. The campaign that found someone is the only statement about them the pipeline can make without asking a model to guess: someone found by a campaign called "game designers" is a game designer. A list that cannot be segmented is worth very little, and segment size is the thing that decides that. The backfill migration is the history the code fix cannot reach — everything found before it existed. Idempotent, since the unique index already holds every row it would add. Applied: 9 contacts to 80, 63 with a niche, 78 prospects linked, none left unrecorded. And the table had no surface at all, which made it a table rather than an asset. There is now a panel showing segment sizes and a CSV export. Fields that begin with =, + or - are prefixed, because a contact list is exactly the kind of file that gets opened in a spreadsheet without a thought. Anyone who asked not to be contacted is excluded from the counts and the export both, so a file cannot reintroduce them elsewhere. Co-Authored-By: Claude Opus 5 (1M context) --- app/(app)/projects/[id]/leads/page.tsx | 12 +- app/api/projects/[id]/contacts.csv/route.ts | 43 ++++++ components/leads/contacts-panel.tsx | 66 +++++++++ lib/outreach/contactsExport.ts | 133 ++++++++++++++++++ lib/outreach/pipeline.ts | 35 ++++- ...60729040000_backfill_outreach_contacts.sql | 63 +++++++++ tests/contacts-export.test.ts | 91 ++++++++++++ 7 files changed, 441 insertions(+), 2 deletions(-) create mode 100644 app/api/projects/[id]/contacts.csv/route.ts create mode 100644 components/leads/contacts-panel.tsx create mode 100644 lib/outreach/contactsExport.ts create mode 100644 supabase/migrations/20260729040000_backfill_outreach_contacts.sql create mode 100644 tests/contacts-export.test.ts diff --git a/app/(app)/projects/[id]/leads/page.tsx b/app/(app)/projects/[id]/leads/page.tsx index 965e6302..0469689a 100644 --- a/app/(app)/projects/[id]/leads/page.tsx +++ b/app/(app)/projects/[id]/leads/page.tsx @@ -11,6 +11,8 @@ import { MailboxConnect, type ConnectedMailbox } from "@/components/leads/mailbo import { SeedLogins } from "@/components/leads/seed-logins"; import { FunnelPanel } from "@/components/leads/funnel-panel"; import { RepliesPanel, type ReplyRow } from "@/components/leads/replies-panel"; +import { ContactsPanel } from "@/components/leads/contacts-panel"; +import { contactNiches } from "@/lib/outreach/contactsExport"; import { campaignFunnels, projectFunnel } from "@/lib/outreach/funnel"; import { listSeedCredentials, type StoredSeedCredential } from "@/lib/outreach/seedCredentials"; import { RefreshLeads } from "@/components/leads/refresh-leads"; @@ -162,9 +164,10 @@ export default async function LeadsPage({ // Measured outcomes, project-wide and per campaign. Both read the same // tables the pipeline already writes, so this costs two queries rather than // any new bookkeeping. - const [funnel, perCampaignFunnel] = await Promise.all([ + const [funnel, perCampaignFunnel, contacts] = await Promise.all([ projectFunnel(projectId), campaignFunnels(projectId), + contactNiches(projectId), ]); // What came back. Read from the connected mailbox by the reply-scan cron, @@ -240,6 +243,13 @@ export default async function LeadsPage({ + + }, +) { + const { id } = await params; + const access = await requireProjectAccess(id, { allowViewer: true }); + if (!access.ok) { + return NextResponse.json({ ok: false, error: "not found" }, { status: 404 }); + } + + const rows = await contactsForProject(id); + const stamp = new Date().toISOString().slice(0, 10); + + // Leading BOM: without it Excel reads the file as the local codepage and + // mangles every accented name in it. + return new NextResponse(`${toCsv(rows)}`, { + status: 200, + headers: { + "content-type": "text/csv; charset=utf-8", + "content-disposition": `attachment; filename="contacts-${stamp}.csv"`, + "cache-control": "no-store", + }, + }); +} diff --git a/components/leads/contacts-panel.tsx b/components/leads/contacts-panel.tsx new file mode 100644 index 00000000..c8e8cf34 --- /dev/null +++ b/components/leads/contacts-panel.tsx @@ -0,0 +1,66 @@ +import type { NicheCount } from "@/lib/outreach/contactsExport"; + +/** + * The contact list, as an asset rather than a table. + * + * Contacts are org-scoped and outlive any single campaign — the same person + * found by two projects is one record. That is the whole reason the table is + * separate from prospects, and until now it had no surface at all, so the + * durable thing the pipeline was building could not be seen or taken away. + * + * Segment sizes rather than a single total, because how many of a kind you + * have is what decides whether a list is worth anything. + */ +export function ContactsPanel({ + projectId, + total, + withEmail, + niches, +}: { + projectId: string; + total: number; + withEmail: number; + niches: NicheCount[]; +}) { + if (total === 0) return null; + + return ( +
+
+
+

Contacts

+

+ Everyone this organization has found, deduplicated across every campaign and project.{" "} + {withEmail} of {total} have an address. +

+
+ + Export CSV + +
+ +
    + {niches.map((n) => ( +
  • + {n.niche === "unsorted" ? ( + unsorted + ) : ( + n.niche + )}{" "} + + {n.withEmail}/{n.contacts} + +
  • + ))} +
+ +

+ Niche comes from the campaign that found them. Anyone who asked not to be contacted is + excluded from both the counts and the export. +

+
+ ); +} diff --git a/lib/outreach/contactsExport.ts b/lib/outreach/contactsExport.ts new file mode 100644 index 00000000..4faaf3a1 --- /dev/null +++ b/lib/outreach/contactsExport.ts @@ -0,0 +1,133 @@ +// The contacts table, as a file you can actually take away. +// +// The table is org-scoped and durable — it outlives any one campaign, which is +// the whole reason it exists separately from prospects. But a list nobody can +// see or export is not an asset, it is a table, and it had no surface at all +// until this. + +import { serviceClient } from "@/lib/supabase/service"; + +export type ContactExportRow = { + email: string | null; + full_name: string | null; + title: string | null; + company_name: string | null; + company_site: string | null; + niche: string | null; + host: string | null; + country: string | null; + phone: string | null; + linkedin_url: string | null; + source_url: string | null; + first_seen_at: string | null; +}; + +export const EXPORT_COLUMNS: Array = [ + "email", + "full_name", + "title", + "company_name", + "company_site", + "niche", + "host", + "country", + "phone", + "linkedin_url", + "source_url", + "first_seen_at", +]; + +/** + * One CSV field. + * + * Quoted whenever it contains a delimiter, a quote or a newline, with inner + * quotes doubled — RFC 4180. A leading =, +, - or @ is prefixed with a single + * quote as well: spreadsheets treat those as formulas, and a contact list is + * exactly the kind of file that gets opened in one without a thought. + */ +export function csvField(value: unknown): string { + if (value === null || value === undefined) return ""; + let s = String(value); + if (/^[=+\-@\t\r]/.test(s)) s = `'${s}`; + if (/[",\n\r]/.test(s)) return `"${s.replace(/"/g, '""')}"`; + return s; +} + +export function toCsv(rows: ContactExportRow[]): string { + const head = EXPORT_COLUMNS.join(","); + const body = rows.map((r) => EXPORT_COLUMNS.map((c) => csvField(r[c])).join(",")); + // CRLF and a trailing newline, which is what RFC 4180 says and what Excel + // wants. + return [head, ...body].join("\r\n") + "\r\n"; +} + +/** Every contact the organization owning this project knows about. */ +export async function contactsForProject(projectId: string): Promise { + const sb = serviceClient(); + const { data: project } = await sb + .from("projects") + .select("organization_id") + .eq("id", projectId) + .maybeSingle(); + const orgId = (project?.organization_id as string | null) ?? null; + if (!orgId) return []; + + const { data } = await sb + .from("outreach_contacts") + .select(EXPORT_COLUMNS.join(", ")) + .eq("organization_id", orgId) + // Anyone who asked not to be contacted is not part of a list that exists + // to be contacted. Excluding them here rather than at the point of sending + // means the exported file cannot reintroduce them somewhere else. + .eq("do_not_contact", false) + .order("first_seen_at", { ascending: false }) + .limit(50_000); + + return (data as ContactExportRow[] | null) ?? []; +} + +export type NicheCount = { niche: string; contacts: number; withEmail: number }; + +/** + * What the list is made of. + * + * Segment size is what decides whether a list is worth anything, so the + * breakdown is the summary rather than the total. + */ +export async function contactNiches(projectId: string): Promise<{ + total: number; + withEmail: number; + niches: NicheCount[]; +}> { + const sb = serviceClient(); + const { data: project } = await sb + .from("projects") + .select("organization_id") + .eq("id", projectId) + .maybeSingle(); + const orgId = (project?.organization_id as string | null) ?? null; + if (!orgId) return { total: 0, withEmail: 0, niches: [] }; + + const { data } = await sb + .from("outreach_contacts") + .select("niche, email") + .eq("organization_id", orgId) + .eq("do_not_contact", false) + .limit(50_000); + + const rows = (data as { niche: string | null; email: string | null }[] | null) ?? []; + const byNiche = new Map(); + for (const r of rows) { + const key = r.niche ?? "unsorted"; + const entry = byNiche.get(key) ?? { niche: key, contacts: 0, withEmail: 0 }; + entry.contacts += 1; + if (r.email) entry.withEmail += 1; + byNiche.set(key, entry); + } + + return { + total: rows.length, + withEmail: rows.filter((r) => r.email).length, + niches: [...byNiche.values()].sort((a, b) => b.contacts - a.contacts), + }; +} diff --git a/lib/outreach/pipeline.ts b/lib/outreach/pipeline.ts index 8c9c6461..5b8ef02b 100644 --- a/lib/outreach/pipeline.ts +++ b/lib/outreach/pipeline.ts @@ -389,6 +389,19 @@ export async function researchProspect(input: { } } + // The same record the unscanned path writes. Missing it here is why the + // shared contacts table held nine rows against seventy-seven addressable + // leads: nearly all of them arrived through this branch, which wrote the + // address onto the prospect and nowhere else. + const scannedContactId = await recordContact({ + projectId: input.projectId, + host, + email: contact?.email ?? null, + label: input.discoveryLabel, + campaignId: input.campaignId, + source: contact?.source === "manual" ? "manual" : contact?.source === "guess" ? "guess" : "page", + }); + const { data, error } = await serviceClient() .from("outreach_prospects") .upsert( @@ -403,6 +416,7 @@ export async function researchProspect(input: { discovery_label: input.discoveryLabel ?? null, contact_email: contact?.email ?? null, contact_source: contact?.source ?? null, + contact_id: scannedContactId, audit_id: audit.id, report_token: audit.share_token, score: audit.score, @@ -454,9 +468,11 @@ async function recordContact(input: { host: string; email: string | null; label?: string | null; + campaignId?: string | null; source: "page" | "search" | "guess" | "manual"; }): Promise { - const { data: project } = await serviceClient() + const sb = serviceClient(); + const { data: project } = await sb .from("projects") .select("organization_id") .eq("id", input.projectId) @@ -464,6 +480,21 @@ async function recordContact(input: { const organizationId = (project?.organization_id as string | null) ?? null; if (!organizationId) return null; + // The campaign that found them is the best available statement of what + // they are: someone found by "game designers in Austin" is a game + // designer in Austin. It is the only niche the pipeline can know without + // asking a model to guess one, and a list without a niche cannot be + // segmented, which is most of what makes a list worth anything. + let niche: string | null = null; + if (input.campaignId) { + const { data: campaign } = await sb + .from("outreach_campaigns") + .select("name") + .eq("id", input.campaignId) + .maybeSingle(); + niche = (campaign?.name as string | null) ?? null; + } + const res = await upsertContact({ organizationId, source: input.source, @@ -475,6 +506,7 @@ async function recordContact(input: { // one rather than guessed into full_name. companyName: input.label ?? null, companySite: `https://${input.host}`, + niche, }, }); return res?.id ?? null; @@ -540,6 +572,7 @@ async function researchWithoutScan(input: { host, email: contact?.email ?? null, label: input.discoveryLabel, + campaignId: input.campaignId, source: contact?.source === "manual" ? "manual" : contact?.source === "guess" ? "guess" : "page", }); diff --git a/supabase/migrations/20260729040000_backfill_outreach_contacts.sql b/supabase/migrations/20260729040000_backfill_outreach_contacts.sql new file mode 100644 index 00000000..28bef664 --- /dev/null +++ b/supabase/migrations/20260729040000_backfill_outreach_contacts.sql @@ -0,0 +1,63 @@ +-- Backfill the shared contacts table from prospects already found. +-- +-- outreach_contacts held nine rows against seventy-seven addressable leads. +-- Nearly every lead arrived through the scanning branch of researchProspect, +-- which wrote the address onto the prospect and nowhere else — so the table +-- meant to be the durable, cross-project record of who we know had almost +-- nothing in it, and the leads themselves were only reachable by reading +-- project-scoped prospect rows one project at a time. +-- +-- The code path is fixed alongside this. This is the history it could not +-- reach: everything found before the fix existed. +-- +-- Idempotent. Re-running inserts nothing new, because the unique index on +-- (organization_id, identity_key) already holds every row this would add. + +with candidates as ( + select distinct on (pr.organization_id, lower(p.contact_email)) + pr.organization_id, + lower(p.contact_email) as email, + p.target_key as host, + -- What the listing called them. A company name far more often than a + -- person's, which is why it is not written into full_name. + nullif(p.discovery_label, '') as company_name, + 'https://' || p.target_key as company_site, + -- The campaign that found them is the only statement of what they are + -- that the pipeline can make without guessing: someone found by a + -- campaign called "game designers" is a game designer. + c.name as niche, + p.site_url as source_url, + -- Prefer the richest row when the same address appears in several + -- projects: one that names the company beats one that does not. + p.created_at + from public.outreach_prospects p + join public.projects pr on pr.id = p.project_id + left join public.outreach_campaigns c on c.id = p.campaign_id + where p.contact_email is not null + and p.contact_email <> '' + and p.channel = 'email' + and pr.organization_id is not null + order by pr.organization_id, + lower(p.contact_email), + (p.discovery_label is not null) desc, + p.created_at asc +) +insert into public.outreach_contacts + (organization_id, email, host, company_name, company_site, niche, source_url, first_seen_at) +select organization_id, email, host, company_name, company_site, niche, source_url, created_at +from candidates +-- identity_key is assigned by the before-insert trigger, so the partial index +-- it backs is the right conflict target here. +on conflict (organization_id, identity_key) where identity_key is not null +do nothing; + +-- Point the prospects at the contact they belong to, so the link works both +-- ways rather than only for rows created after the fix. +update public.outreach_prospects p +set contact_id = c.id +from public.outreach_contacts c +join public.projects pr on pr.organization_id = c.organization_id +where pr.id = p.project_id + and p.contact_email is not null + and lower(p.contact_email) = lower(c.email) + and p.contact_id is distinct from c.id; diff --git a/tests/contacts-export.test.ts b/tests/contacts-export.test.ts new file mode 100644 index 00000000..df6ff37c --- /dev/null +++ b/tests/contacts-export.test.ts @@ -0,0 +1,91 @@ +import { describe, it, expect } from "vitest"; +import { readFileSync } from "node:fs"; +import { csvField, toCsv, EXPORT_COLUMNS } from "@/lib/outreach/contactsExport"; + +const empty = Object.fromEntries(EXPORT_COLUMNS.map((c) => [c, null])) as Parameters< + typeof toCsv +>[0][number]; + +describe("csvField", () => { + it("leaves ordinary text alone", () => { + expect(csvField("Acme Ltd")).toBe("Acme Ltd"); + }); + + it("quotes a field containing the delimiter", () => { + expect(csvField("Acme, Ltd")).toBe('"Acme, Ltd"'); + }); + + it("doubles inner quotes", () => { + expect(csvField('He said "hi"')).toBe('"He said ""hi"""'); + }); + + it("quotes a field containing a newline", () => { + expect(csvField("line one\nline two")).toBe('"line one\nline two"'); + }); + + it("defuses a spreadsheet formula", () => { + // A contact list is exactly the kind of file somebody opens in Excel + // without thinking, and a leading = there is executable. + expect(csvField("=1+1")).toBe("'=1+1"); + expect(csvField("+44 7700 900000")).toBe("'+44 7700 900000"); + expect(csvField("-5")).toBe("'-5"); + expect(csvField("@handle")).toBe("'@handle"); + }); + + it("writes nothing for a missing value", () => { + expect(csvField(null)).toBe(""); + expect(csvField(undefined)).toBe(""); + }); +}); + +describe("toCsv", () => { + it("leads with a header row", () => { + expect(toCsv([]).split("\r\n")[0]).toBe(EXPORT_COLUMNS.join(",")); + }); + + it("still produces a usable file with no contacts", () => { + // A header-only CSV opens; an empty file looks like a failed download. + expect(toCsv([])).toBe(`${EXPORT_COLUMNS.join(",")}\r\n`); + }); + + it("writes one line per contact, in column order", () => { + const csv = toCsv([{ ...empty, email: "a@b.test", company_name: "Acme", niche: "designers" }]); + const [, row] = csv.trim().split("\r\n"); + const cells = row.split(","); + expect(cells[EXPORT_COLUMNS.indexOf("email")]).toBe("a@b.test"); + expect(cells[EXPORT_COLUMNS.indexOf("company_name")]).toBe("Acme"); + expect(cells[EXPORT_COLUMNS.indexOf("niche")]).toBe("designers"); + }); + + it("keeps a comma in a company name from shifting every later column", () => { + const csv = toCsv([{ ...empty, company_name: "Acme, Ltd", niche: "designers" }]); + expect(csv).toContain('"Acme, Ltd"'); + expect(csv.trim().split("\r\n")).toHaveLength(2); + }); + + it("ends with a newline", () => { + expect(toCsv([{ ...empty, email: "a@b.test" }]).endsWith("\r\n")).toBe(true); + }); +}); + +describe("both research paths record a contact", () => { + // The scanning branch never called recordContact, so the shared table held + // nine rows against seventy-seven addressable leads — every one of them + // written onto a prospect and nowhere else. A unit test of upsertContact + // passed throughout, because the bug was a missing call. + const pipeline = readFileSync(new URL("../lib/outreach/pipeline.ts", import.meta.url), "utf8"); + + it("calls recordContact more than once", () => { + const calls = pipeline.match(/await recordContact\(/g) ?? []; + expect(calls.length).toBeGreaterThanOrEqual(2); + }); + + it("links the prospect to the contact on the scanning path", () => { + expect(pipeline).toContain("contact_id: scannedContactId"); + }); + + it("records the niche, without which the list cannot be segmented", () => { + expect(pipeline).toContain("niche,"); + expect(pipeline).toMatch(/from\("outreach_campaigns"\)[\s\S]{0,120}select\("name"\)/); + }); +});