From ae9ce0a23095f5221d0b9867d73252a7382fecf1 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 08:40:23 +0000 Subject: [PATCH] feat(leads): find the people a directory lists, not just the sites it links to MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Link-following assumes a listing points at each business's own site. A people directory does not: it publishes names, titles and locations, and links only to itself. Pointed at one, discovery returned a single host — the directory's own footer — from a page naming sixteen people. So a person is now read as an entity. Structured data first, because a page carrying schema.org Person markup has already said exactly who it is about and nothing needs inferring from prose. Every JSON-LD node is checked rather than the first: a profile page's first block is normally the site's own Organization, which is the difference between a named CTO and the directory's parent company. People are collected alongside prospects, never instead of them. A directory page frequently offers both — the people it lists and its own contact address — and choosing one discards half of what the page gave. Two loading bugs had to be fixed before any of it worked, both from the same wrong question. loadSeedHtml decides whether to render by asking whether a plain fetch produced any outbound candidate, and a directory's own footer answers yes — so the shell was accepted, the client-rendered listing never loaded, and the profiles were invisible. It now also asks whether the page yielded entries, and an entity page that names nobody is retried with a render. The meta fallback is deliberately hard to satisfy, because it produced four fabricated people on the first live run: "AI Leadership Sprint", "For Fractional CTOs", "For CEOs and Founders", "For Board Members" — all capitalised, all two to four words, none of them human. A fabricated name reaches a real inbox addressed to nobody, so a title is only read as a person when the page itself claims to be a profile, and headline words are refused outright. All four are pinned as tests. LinkedIn URLs are normalised rather than stored as found. Directories copy whatever the member pasted, and members paste the URL from their own logged-in view — a settings page with session tracking that opens nothing for anyone else. That looks like a working link until someone clicks it. Employers embedded in a title ("Owner and CTO at Artechra") are split out, since directories routinely fill jobTitle and leave worksFor empty. Verified against a live directory: twelve people, all from structured data, twelve with titles, eleven with a usable LinkedIn profile, and the one pasted settings URL correctly dropped. One-shot lead finding was capped at ten. A directory can list hundreds, so reporting ten looked like the form was broken. Raised, with an explicit ceiling on contact lookups so a large run cannot become an unbounded bill. Co-Authored-By: Claude Opus 5 (1M context) --- app/actions/leads.ts | 9 +- lib/outreach/discover.ts | 106 ++++++++++++- lib/outreach/person.ts | 299 +++++++++++++++++++++++++++++++++++ lib/outreach/render.ts | 19 +++ tests/person-extract.test.ts | 212 +++++++++++++++++++++++++ 5 files changed, 637 insertions(+), 8 deletions(-) create mode 100644 lib/outreach/person.ts create mode 100644 tests/person-extract.test.ts diff --git a/app/actions/leads.ts b/app/actions/leads.ts index a71f7d9b..36befe05 100644 --- a/app/actions/leads.ts +++ b/app/actions/leads.ts @@ -68,7 +68,13 @@ export async function findLeadsAction(input: { return { ok: false, error: "Enter a search query or a directory URL." }; } - const limit = Math.min(input.limit ?? 10, 25); + // A directory can list hundreds of businesses; capping a one-shot run at + // ten meant the form reported a fraction of a page and looked broken. + const limit = Math.min(input.limit ?? 100, 1000); + // Contact lookup is the part that costs money — up to two SERP calls per + // prospect that publishes no address — so a thousand-lead run gets an + // explicit ceiling rather than an unbounded bill. + const contactSearchBudget = { remaining: Math.min(limit, 100) }; const found = await discoverProspects({ queries: input.query?.trim() ? [input.query.trim()] : [], seedUrls: input.seedUrl?.trim() ? [input.seedUrl.trim()] : [], @@ -87,6 +93,7 @@ export async function findLeadsAction(input: { url: candidate.url, discoveredVia: candidate.via, discoveryLabel: candidate.label, + contactSearchBudget, // Finding leads for a project does not scan them. The scan exists to // supply findings for the CrawlProof audit pitch, and firing one at // every discovered business spends worker time on evidence nobody is diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index acc0124e..8c5a6e25 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -24,8 +24,12 @@ import { normalizeHost } from "./cold"; import { loadSeedCredential, makeSeedCodeWaiter, recordSeedCredentialResult, seedHost } from "./seedCredentials"; import type { CodeWaiter } from "@/lib/sp/verificationChallenge"; import { looksLikeLoginWall } from "./loginWall"; +import { extractPerson, type ExtractedPerson } from "./person"; import type { SeedCredentials } from "./seedLogin"; +/** A person found on a page, with where they were found. */ +export type DiscoveredPerson = ExtractedPerson & { sourceUrl: string }; + export type DiscoveredProspect = { host: string; url: string; @@ -148,7 +152,7 @@ export function extractOutboundProspects(input: { const out = new Map(); $("a[href]").each((_, el) => { - if (out.size >= (input.limit ?? 100)) return false; + if (out.size >= (input.limit ?? 1000)) return false; const href = ($(el).attr("href") ?? "").trim(); if (!href || href.startsWith("#") || href.startsWith("mailto:") || href.startsWith("tel:")) { return undefined; @@ -234,7 +238,22 @@ export function extractSameHostLinks(input: { return undefined; }); - return [...out]; + // Entity pages first. Nav and marketing links appear earlier in the + // document than the listing itself, so document order spends the page + // budget on "For CEOs and Founders" and never reaches a single profile. + return [...out].sort((a, b) => Number(isEntityPath(b)) - Number(isEntityPath(a))); +} + +/** Paths that name one entity rather than a section of the site. */ +const ENTITY_PATH_RE = + /\/(profile|profiles|people|person|member|members|author|u|in|company|listing|business)\/[^/]+$/i; + +function isEntityPath(url: string): boolean { + try { + return ENTITY_PATH_RE.test(new URL(url).pathname); + } catch { + return false; + } } /** Same-host paths that are navigation or account plumbing, never a business. */ @@ -292,6 +311,16 @@ async function loadSeedHtml( allowRender: boolean, credentials?: SeedCredentials | null, codeWaiter?: CodeWaiter | null, + /** + * Treat same-host entity links as content too. + * + * A directory's value is the entries it lists, not the sites it links out + * to. Judging "did the fetch work" purely on outbound links accepts the + * shell of a client-rendered directory — its own footer satisfies the + * test — and the listing, which is the entire point of the page, is never + * loaded. + */ + wantEntityLinks = false, ): Promise<{ html: string; rendered: boolean } | { error: string; loginRequired?: boolean }> { const direct = await fetchHtml(url); // A login wall is settled unless we hold a credential — without one, @@ -300,8 +329,14 @@ async function loadSeedHtml( return { error: direct.error, loginRequired: true }; } if (direct.ok) { - const hasCandidates = extractOutboundProspects({ html: direct.html, sourceUrl: url, limit: 1 }).length > 0; - if (hasCandidates || !allowRender) return { html: direct.html, rendered: false }; + const hasCandidates = + extractOutboundProspects({ html: direct.html, sourceUrl: url, limit: 1 }).length > 0; + const hasEntries = + !wantEntityLinks || + extractSameHostLinks({ html: direct.html, sourceUrl: url, limit: 5 }).some(isEntityPath); + if ((hasCandidates && hasEntries) || !allowRender) { + return { html: direct.html, rendered: false }; + } } else if (!allowRender) { return { error: direct.error }; } @@ -342,19 +377,34 @@ export async function discoverFromSeed(input: { detailHopThreshold?: number; }): Promise<{ prospects: DiscoveredProspect[]; + /** + * People named on the pages that were opened. + * + * Collected alongside prospects rather than instead of them. A directory + * page often yields both — the people it lists and its own contact + * address — and picking one throws away half of what the page offered. + */ + people: DiscoveredPerson[]; error?: string; notes?: string[]; /** Set when the seed was withheld pending a login, so the UI can offer to store one. */ loginRequired?: boolean; }> { - const limit = input.limit ?? 100; + const limit = input.limit ?? 1000; const allowRender = input.render !== false; const notes: string[] = []; - const seed = await loadSeedHtml(input.seedUrl, allowRender, input.credentials, input.codeWaiter); + const seed = await loadSeedHtml( + input.seedUrl, + allowRender, + input.credentials, + input.codeWaiter, + (input.depth ?? 1) >= 2, + ); if ("error" in seed) { return { prospects: [], + people: [], error: `seed ${input.seedUrl} failed: ${seed.error}`, loginRequired: seed.loginRequired, }; @@ -362,9 +412,13 @@ export async function discoverFromSeed(input: { if (seed.rendered) notes.push(`rendered ${input.seedUrl} in a browser`); const merged = new Map(); + const people = new Map(); for (const p of extractOutboundProspects({ html: seed.html, sourceUrl: input.seedUrl, limit })) { if (!merged.has(p.host)) merged.set(p.host, p); } + // A seed page is occasionally itself a profile. + const seedPerson = extractPerson(seed.html, input.seedUrl); + if (seedPerson) people.set(seedPerson.fullName, { ...seedPerson, sourceUrl: input.seedUrl }); // The second hop is expensive — one page load per listing, each possibly a // browser render — so it only runs when the first hop came up short. A @@ -392,11 +446,49 @@ export async function discoverFromSeed(input: { })) { if (!merged.has(p.host)) merged.set(p.host, p); } + // Both, from the same fetch. A people directory links only to itself, + // so the outbound pass above finds nothing there — but the page still + // names someone, and that is the whole reason the page was opened. + let person = extractPerson(detail.html, detailUrl); + + // A profile whose identity is injected client-side needs the render + // that loadSeedHtml declined to do. Its "good enough" test is whether + // the fetch produced any outbound candidate, and a directory's own + // footer satisfies that — so the shell is accepted and the person, who + // only exists after scripts run, is never seen. Rendering is retried + // only when the page looks like it should have named someone and + // didn't, so an ordinary listicle still costs one fetch. + if (!person && !detail.rendered && allowRender && isEntityPath(detailUrl)) { + const { renderPage } = await import("./render"); + const rendered = await renderPage(detailUrl, { + credentials: input.credentials, + codeWaiter: input.codeWaiter, + }); + if (rendered.ok) { + person = extractPerson(rendered.html, detailUrl); + for (const p of extractOutboundProspects({ + html: rendered.html, + sourceUrl: detailUrl, + limit: limit - merged.size, + })) { + if (!merged.has(p.host)) merged.set(p.host, p); + } + } + } + + if (person && !people.has(person.fullName)) { + people.set(person.fullName, { ...person, sourceUrl: detailUrl }); + } } notes.push(`opened ${detailPages.length} listing entries`); } - return { prospects: [...merged.values()], notes: notes.length ? notes : undefined }; + if (people.size) notes.push(`named ${people.size} people`); + return { + prospects: [...merged.values()], + people: [...people.values()], + notes: notes.length ? notes : undefined, + }; } /** Which search backend to use. */ diff --git a/lib/outreach/person.ts b/lib/outreach/person.ts new file mode 100644 index 00000000..2e02fa20 --- /dev/null +++ b/lib/outreach/person.ts @@ -0,0 +1,299 @@ +// Read a person off a profile page. +// +// The link-following discovery elsewhere in this directory assumes a page +// points at a business's own site. A people directory does not: it publishes +// a name, a title and a location, and links only to itself. Following links +// there yields the directory's own footer, which is what it did before this +// existed. +// +// So this extracts the person as an entity instead. Structured data first, +// because a page that publishes schema.org Person markup has already told us +// exactly who it is about and nothing needs to be guessed from prose. +// +// The trap worth naming: a profile page usually carries several JSON-LD +// blocks, and the first is typically the site's own Organization. Taking +// `json[0]` gets the directory rather than the person — on the page this was +// built against, that is the difference between "Marc van Neerven, CTO" and +// "StackUp, a technology consultancy". + +export type ExtractedPerson = { + fullName: string; + jobTitle: string | null; + company: string | null; + companySite: string | null; + description: string | null; + linkedinUrl: string | null; + /** Other profiles found on the page, keyed by network. */ + socials: Record; + location: string | null; + /** How we read it, so a weak guess can be told from structured data. */ + source: "json-ld" | "meta"; +}; + +type Json = Record; + +function asString(v: unknown): string | null { + if (typeof v === "string" && v.trim()) return v.trim(); + return null; +} + +/** Flatten @graph containers and arrays into a list of candidate nodes. */ +function flattenNodes(parsed: unknown): Json[] { + const out: Json[] = []; + const visit = (node: unknown) => { + if (Array.isArray(node)) { + node.forEach(visit); + return; + } + if (!node || typeof node !== "object") return; + const obj = node as Json; + out.push(obj); + if (Array.isArray(obj["@graph"])) visit(obj["@graph"]); + }; + visit(parsed); + return out; +} + +function typeOf(node: Json): string[] { + const t = node["@type"]; + if (typeof t === "string") return [t]; + if (Array.isArray(t)) return t.filter((x): x is string => typeof x === "string"); + return []; +} + +const LINKEDIN_RE = /^https?:\/\/([a-z]{2,3}\.)?linkedin\.com\//i; + +/** + * A clean LinkedIn profile URL, or null. + * + * Directories copy whatever the member pasted, and people paste the URL from + * their own logged-in view — which is a settings page carrying session + * tracking parameters, not a profile anyone else can open. Storing that is + * worse than storing nothing: it looks like a working link right up until + * someone clicks it. + */ +export function normalizeLinkedIn(raw: string): string | null { + let url: URL; + try { + url = new URL(raw); + } catch { + return null; + } + if (!/(^|\.)linkedin\.com$/i.test(url.hostname)) return null; + // Only member and company profiles are addressable by other people. + const m = url.pathname.match(/^\/(in|company|school)\/([^/]+)\/?$/i); + if (!m) return null; + return `https://www.linkedin.com/${m[1].toLowerCase()}/${m[2]}`; +} + +/** Which network a profile URL belongs to, for the socials map. */ +function networkOf(url: string): string | null { + const patterns: [RegExp, string][] = [ + [/linkedin\.com/i, "linkedin"], + [/(^|\/\/)(x|twitter)\.com/i, "x"], + [/github\.com/i, "github"], + [/mastodon|\.social/i, "mastodon"], + [/bsky\.app/i, "bluesky"], + [/youtube\.com/i, "youtube"], + [/instagram\.com/i, "instagram"], + ]; + for (const [re, name] of patterns) if (re.test(url)) return name; + return null; +} + +function personFromJsonLd(html: string): ExtractedPerson | null { + const blocks = [ + ...html.matchAll(/]+application\/ld\+json[^>]*>([\s\S]*?)<\/script>/gi), + ].map((m) => m[1]); + + for (const raw of blocks) { + let parsed: unknown; + try { + parsed = JSON.parse(raw.trim()); + } catch { + continue; + } + for (const node of flattenNodes(parsed)) { + // Every node is checked for Person rather than just the first block: + // the site's own Organization normally comes first. + if (!typeOf(node).includes("Person")) continue; + const fullName = asString(node.name); + if (!fullName) continue; + + const worksFor = node.worksFor; + let company: string | null = null; + let companySite: string | null = null; + if (worksFor && typeof worksFor === "object") { + const org = (Array.isArray(worksFor) ? worksFor[0] : worksFor) as Json; + company = asString(org?.name); + companySite = asString(org?.url); + } else { + company = asString(worksFor); + } + + const sameAs = Array.isArray(node.sameAs) + ? node.sameAs.filter((s): s is string => typeof s === "string") + : typeof node.sameAs === "string" + ? [node.sameAs] + : []; + + const socials: Record = {}; + let linkedinUrl: string | null = null; + for (const url of sameAs) { + if (LINKEDIN_RE.test(url) && !linkedinUrl) linkedinUrl = normalizeLinkedIn(url); + const net = networkOf(url); + if (net && !socials[net]) socials[net] = url; + } + + const address = node.address; + let location: string | null = null; + if (address && typeof address === "object") { + const a = address as Json; + location = + [asString(a.addressLocality), asString(a.addressRegion), asString(a.addressCountry)] + .filter(Boolean) + .join(", ") || null; + } else { + location = asString(address); + } + + // Directories frequently write the whole thing into jobTitle — "Owner + // and CTO at Artechra" — leaving worksFor empty. Splitting on " at " + // recovers the employer that would otherwise be lost inside the role. + const rawTitle = asString(node.jobTitle); + let jobTitle = rawTitle; + if (!company && rawTitle) { + const m = rawTitle.match(/^(.*?)\s+at\s+(.+)$/i); + if (m && m[1].trim() && m[2].trim()) { + jobTitle = m[1].trim(); + company = m[2].trim(); + } + } + + return { + fullName, + jobTitle, + company, + companySite, + description: asString(node.description), + linkedinUrl, + socials, + location, + source: "json-ld", + }; + } + } + return null; +} + + +/** + * Words that never begin a person's name and reliably begin a heading. + * + * The meta fallback previously accepted "For Fractional CTOs" and "AI + * Leadership Sprint" as people, because they are two-to-four capitalised + * words like a name. A fabricated name reaches a real inbox addressed to + * nobody, so the bar here is deliberately high: it is better to miss a + * person than to invent one. + */ +const NOT_A_NAME_START = + /^(for|the|a|an|our|your|my|why|how|what|when|top|best|about|meet|join|find|hire|get|introducing|welcome)$/i; + +const NOT_A_NAME_WORD = + /^(ctos?|ceos?|cfos?|founders?|members?|board|sprint|leadership|directory|guide|list|jobs?|careers?|services?|pricing|blog|news|team|home)$/i; + +function looksLikePersonName(value: string): boolean { + const words = value.split(/\s+/).filter(Boolean); + if (words.length < 2 || words.length > 4) return false; + if (NOT_A_NAME_START.test(words[0])) return false; + if (words.some((w) => NOT_A_NAME_WORD.test(w.replace(/[^A-Za-z]/g, "")))) return false; + // Every word starts with a capital, allowing lowercase particles that real + // names carry: van, de, der, bin, al. + const PARTICLE = /^(van|von|de|del|della|der|den|di|da|du|la|le|bin|al|ibn|mac|mc|o')$/i; + return words.every((w) => PARTICLE.test(w) || /^[A-Z]/.test(w)); +} + +/** + * Does the page claim to be about a person at all? + * + * Without this, any well-formed two-word heading on a marketing page becomes + * a contact. A profile URL or an og:type of profile is the page saying so + * itself, which is a far better signal than the shape of its title. + */ +function hasProfileSignal(html: string, url: string): boolean { + if (/\/(profile|people|person|member|members|team|u|author)\//i.test(url)) return true; + const ogType = metaContent(html, "og:type"); + return ogType === "profile"; +} + +function metaContent(html: string, key: string): string | null { + const re = new RegExp( + `]+(?:property|name)=["']${key}["'][^>]*content=["']([^"']*)["']`, + "i", + ); + const alt = new RegExp( + `]+content=["']([^"']*)["'][^>]*(?:property|name)=["']${key}["']`, + "i", + ); + return asString(html.match(re)?.[1]) ?? asString(html.match(alt)?.[1]); +} + +/** + * Fallback for pages with no Person markup. + * + * og:title on a profile page is conventionally "Name - Title | Site". The + * site segment after the final pipe is dropped, because it names the + * directory rather than anyone's employer — treating it as a company is how + * you end up with a thousand people who all work at "Fractional CTO + * Directory". + */ +function personFromMeta(html: string, url: string): ExtractedPerson | null { + const title = metaContent(html, "og:title") ?? asString(html.match(/]*>([^<]*)/i)?.[1]); + if (!title) return null; + + const withoutSite = title.split("|")[0].trim(); + const [namePart, ...rest] = withoutSite.split(/\s+[-–—]\s+/); + const fullName = asString(namePart); + if (!fullName) return null; + if (!looksLikePersonName(fullName)) return null; + if (!hasProfileSignal(html, url)) return null; + + return { + fullName, + jobTitle: rest.length ? rest.join(" - ").trim() : null, + company: null, + companySite: null, + description: metaContent(html, "og:description") ?? metaContent(html, "description"), + linkedinUrl: null, + socials: {}, + location: null, + source: "meta", + }; +} + +/** + * Extract the person a profile page is about, or null when it isn't about one. + * + * Returning null is the common and correct outcome — most pages are not + * profiles, and inventing a person from a headline would put a fabricated + * name into an email. + */ +export function extractPerson(html: string, url = ""): ExtractedPerson | null { + return personFromJsonLd(html) ?? personFromMeta(html, url); +} + +/** + * The search query that stands the best chance of finding this person's + * contact details. + * + * Name alone is ambiguous for anyone without an unusual one, so the employer + * or job title is included as a discriminator. Quoting the name keeps the + * engine from returning everyone who shares a surname. + */ +export function personSearchQuery(person: ExtractedPerson): string { + const parts = [`"${person.fullName}"`]; + if (person.company) parts.push(`"${person.company}"`); + else if (person.jobTitle) parts.push(person.jobTitle); + parts.push("(email OR contact)"); + return parts.join(" "); +} diff --git a/lib/outreach/render.ts b/lib/outreach/render.ts index 3565cdb3..98089100 100644 --- a/lib/outreach/render.ts +++ b/lib/outreach/render.ts @@ -35,6 +35,9 @@ const NAV_TIMEOUT_MS = 25_000; const SETTLE_MS = 2_500; const IDLE_SHUTDOWN_MS = 60_000; const MAX_HTML_BYTES = 3 * 1024 * 1024; +/** Bounded so an infinite feed cannot hold a render open forever. */ +const MAX_SCROLLS = 12; +const SCROLL_SETTLE_MS = 1_500; // Chromium is heavy enough that a per-call launch would dominate the cost of // a campaign tick, so one instance is shared across renders. @@ -175,6 +178,22 @@ export async function renderPage( .waitForLoadState("networkidle", { timeout: SETTLE_MS * 2 }) .catch(() => page.waitForTimeout(SETTLE_MS)); + // Directories commonly load a first page of entries and fetch the rest on + // scroll. Rendering without scrolling therefore sees a third of a list and + // reports it as the whole thing, which is indistinguishable from a short + // directory. Scrolling stops as soon as the page stops growing, so a + // static page costs one measurement rather than the full budget. + let previousHeight = 0; + for (let i = 0; i < MAX_SCROLLS; i++) { + const height = await page.evaluate(() => document.body.scrollHeight).catch(() => 0); + if (!height || height === previousHeight) break; + previousHeight = height; + await page.evaluate(() => window.scrollTo(0, document.body.scrollHeight)).catch(() => {}); + await page + .waitForLoadState("networkidle", { timeout: SCROLL_SETTLE_MS }) + .catch(() => page.waitForTimeout(SCROLL_SETTLE_MS)); + } + const title = await page.title().catch(() => ""); if (CHALLENGE_RE.test(title)) { return { diff --git a/tests/person-extract.test.ts b/tests/person-extract.test.ts new file mode 100644 index 00000000..e8e9bf00 --- /dev/null +++ b/tests/person-extract.test.ts @@ -0,0 +1,212 @@ +import { describe, it, expect } from "vitest"; +import { extractPerson, normalizeLinkedIn, personSearchQuery } from "@/lib/outreach/person"; + +// Taken from a real ctodirectory.com profile. The ordering is the point: the +// site's own Organization block comes first, so an extractor that reads +// json[0] returns the directory instead of the person. +const REAL_PROFILE = ` + + + + + +Marc van Neerven - Chief Technology Officer | Fractional CTO Directory +

Marc van Neerven

`; + +describe("extractPerson on a real directory profile", () => { + const person = extractPerson(REAL_PROFILE); + + it("picks the Person, not the site's own Organization", () => { + // json[0] is StackUp. Reading it would attribute every profile on the + // site to the directory's parent company. + expect(person?.fullName).toBe("Marc van Neerven"); + expect(person?.fullName).not.toBe("StackUp"); + }); + + it("reads the job title", () => { + expect(person?.jobTitle).toBe("Chief Technology Officer"); + }); + + it("prefers structured data over the title tag", () => { + expect(person?.source).toBe("json-ld"); + }); + + it("does not invent an employer the page never stated", () => { + // "Fractional CTO Directory" is the site, not where he works. + expect(person?.company).toBeNull(); + }); +}); + +describe("richer structured data", () => { + const html = ``; + const p = extractPerson(html); + + it("reads employer and their site", () => { + expect(p?.company).toBe("Acme Robotics"); + expect(p?.companySite).toBe("https://acme.test"); + }); + + it("reads location from a structured address", () => { + expect(p?.location).toBe("Bristol, UK"); + }); + + it("pulls LinkedIn out of sameAs", () => { + expect(p?.linkedinUrl).toBe("https://www.linkedin.com/in/janedoe"); + expect(p?.socials.github).toBe("https://github.com/janedoe"); + }); + + it("handles an @graph wrapper", () => { + const graph = ``; + expect(extractPerson(graph)?.fullName).toBe("Sam Patel"); + }); +}); + +describe("meta fallback", () => { + const TITLE = ``; + const PROFILE_URL = "https://ctodirectory.com/profile/chris-sprucefield-m8phr9"; + + it("parses the Name - Title | Site convention on a profile URL", () => { + const p = extractPerson(TITLE, PROFILE_URL); + expect(p?.fullName).toBe("Chris Sprucefield"); + expect(p?.jobTitle).toBe("Fractional CTO"); + expect(p?.source).toBe("meta"); + }); + + it("drops the site segment rather than treating it as an employer", () => { + expect(extractPerson(TITLE, PROFILE_URL)?.company).toBeNull(); + }); + + it("accepts og:type=profile in place of a profile URL", () => { + const html = `${TITLE}`; + expect(extractPerson(html, "https://example.test/x")?.fullName).toBe("Chris Sprucefield"); + }); + + it("refuses the same title on a page that never claims to be a profile", () => { + // Without a profile signal the shape of a title is not evidence about + // whether the page is about a person. + expect(extractPerson(TITLE, "https://ctodirectory.com/about")).toBeNull(); + }); +}); + +describe("headings that a live run mistook for people", () => { + // All four were extracted as contacts from ctodirectory.com marketing + // pages: capitalised, two to four words, and not human beings. A + // fabricated name reaches a real inbox addressed to nobody. + const notPeople = [ + "AI Leadership Sprint", + "For Fractional CTOs", + "For CEOs and Founders", + "For Board Members", + ]; + + for (const heading of notPeople) { + it(`rejects "${heading}"`, () => { + const html = `${heading}`; + expect(extractPerson(html, "https://ctodirectory.com/profile/x")).toBeNull(); + }); + } + + it("still accepts a real name with a lowercase particle", () => { + const html = `Marc van Neerven - Chief Technology Officer | Directory`; + expect(extractPerson(html, "https://ctodirectory.com/profile/marc-van-neerven-7lisfl")?.fullName) + .toBe("Marc van Neerven"); + }); +}); + +describe("refusing to invent a person", () => { + // Putting a fabricated name into a cold email is worse than extracting + // nothing, so anything that doesn't look like a person returns null. + it("returns null for a page with no person at all", () => { + expect(extractPerson("

Just a page

")).toBeNull(); + }); + + it("returns null for a headline rather than a name", () => { + const html = `The Top 500 CTOs To Watch In America This Year`; + expect(extractPerson(html)).toBeNull(); + }); + + it("returns null for a one-word title", () => { + expect(extractPerson(`Directory`)).toBeNull(); + }); + + it("ignores an Organization-only page", () => { + const html = ``; + expect(extractPerson(html)).toBeNull(); + }); +}); + +describe("personSearchQuery", () => { + it("quotes the name and adds the employer as a discriminator", () => { + const q = personSearchQuery({ + fullName: "Jane Doe", + jobTitle: "VP Engineering", + company: "Acme Robotics", + companySite: null, + description: null, + linkedinUrl: null, + socials: {}, + location: null, + source: "json-ld", + }); + expect(q).toContain('"Jane Doe"'); + expect(q).toContain('"Acme Robotics"'); + }); + + it("falls back to job title when there is no employer", () => { + const q = personSearchQuery({ + fullName: "Marc van Neerven", + jobTitle: "Chief Technology Officer", + company: null, + companySite: null, + description: null, + linkedinUrl: null, + socials: {}, + location: null, + source: "json-ld", + }); + expect(q).toContain('"Marc van Neerven"'); + expect(q).toContain("Chief Technology Officer"); + }); +}); + +describe("normalizeLinkedIn", () => { + it("rejects the logged-in settings URL people actually paste", () => { + // Copied verbatim from a live ctodirectory profile. It renders as a + // LinkedIn link and opens nothing for anyone else. + const pasted = + "https://www.linkedin.com/public-profile/settings/?trk=d_flagship3_profile_self_view_public_profile&lipi=urn%3Ali%3Apage%3Ad_flagship3"; + expect(normalizeLinkedIn(pasted)).toBeNull(); + }); + + it("keeps a real profile and strips tracking parameters", () => { + expect(normalizeLinkedIn("https://www.linkedin.com/in/mvneerven/?trk=abc")).toBe( + "https://www.linkedin.com/in/mvneerven", + ); + }); + + it("accepts company pages", () => { + expect(normalizeLinkedIn("https://linkedin.com/company/acme")).toBe( + "https://www.linkedin.com/company/acme", + ); + }); + + it("rejects a non-LinkedIn host that merely contains the word", () => { + expect(normalizeLinkedIn("https://notlinkedin.com.evil.test/in/someone")).toBeNull(); + }); +});