diff --git a/app/actions/leads.ts b/app/actions/leads.ts index a71f7d9b..36befe05 100644 --- a/app/actions/leads.ts +++ b/app/actions/leads.ts @@ -68,7 +68,13 @@ export async function findLeadsAction(input: { return { ok: false, error: "Enter a search query or a directory URL." }; } - const limit = Math.min(input.limit ?? 10, 25); + // A directory can list hundreds of businesses; capping a one-shot run at + // ten meant the form reported a fraction of a page and looked broken. + const limit = Math.min(input.limit ?? 100, 1000); + // Contact lookup is the part that costs money — up to two SERP calls per + // prospect that publishes no address — so a thousand-lead run gets an + // explicit ceiling rather than an unbounded bill. + const contactSearchBudget = { remaining: Math.min(limit, 100) }; const found = await discoverProspects({ queries: input.query?.trim() ? [input.query.trim()] : [], seedUrls: input.seedUrl?.trim() ? [input.seedUrl.trim()] : [], @@ -87,6 +93,7 @@ export async function findLeadsAction(input: { url: candidate.url, discoveredVia: candidate.via, discoveryLabel: candidate.label, + contactSearchBudget, // Finding leads for a project does not scan them. The scan exists to // supply findings for the CrawlProof audit pitch, and firing one at // every discovered business spends worker time on evidence nobody is diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index acc0124e..8c5a6e25 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -24,8 +24,12 @@ import { normalizeHost } from "./cold"; import { loadSeedCredential, makeSeedCodeWaiter, recordSeedCredentialResult, seedHost } from "./seedCredentials"; import type { CodeWaiter } from "@/lib/sp/verificationChallenge"; import { looksLikeLoginWall } from "./loginWall"; +import { extractPerson, type ExtractedPerson } from "./person"; import type { SeedCredentials } from "./seedLogin"; +/** A person found on a page, with where they were found. */ +export type DiscoveredPerson = ExtractedPerson & { sourceUrl: string }; + export type DiscoveredProspect = { host: string; url: string; @@ -148,7 +152,7 @@ export function extractOutboundProspects(input: { const out = new Map(); $("a[href]").each((_, el) => { - if (out.size >= (input.limit ?? 100)) return false; + if (out.size >= (input.limit ?? 1000)) return false; const href = ($(el).attr("href") ?? "").trim(); if (!href || href.startsWith("#") || href.startsWith("mailto:") || href.startsWith("tel:")) { return undefined; @@ -234,7 +238,22 @@ export function extractSameHostLinks(input: { return undefined; }); - return [...out]; + // Entity pages first. Nav and marketing links appear earlier in the + // document than the listing itself, so document order spends the page + // budget on "For CEOs and Founders" and never reaches a single profile. + return [...out].sort((a, b) => Number(isEntityPath(b)) - Number(isEntityPath(a))); +} + +/** Paths that name one entity rather than a section of the site. */ +const ENTITY_PATH_RE = + /\/(profile|profiles|people|person|member|members|author|u|in|company|listing|business)\/[^/]+$/i; + +function isEntityPath(url: string): boolean { + try { + return ENTITY_PATH_RE.test(new URL(url).pathname); + } catch { + return false; + } } /** Same-host paths that are navigation or account plumbing, never a business. */ @@ -292,6 +311,16 @@ async function loadSeedHtml( allowRender: boolean, credentials?: SeedCredentials | null, codeWaiter?: CodeWaiter | null, + /** + * Treat same-host entity links as content too. + * + * A directory's value is the entries it lists, not the sites it links out + * to. Judging "did the fetch work" purely on outbound links accepts the + * shell of a client-rendered directory — its own footer satisfies the + * test — and the listing, which is the entire point of the page, is never + * loaded. + */ + wantEntityLinks = false, ): Promise<{ html: string; rendered: boolean } | { error: string; loginRequired?: boolean }> { const direct = await fetchHtml(url); // A login wall is settled unless we hold a credential — without one, @@ -300,8 +329,14 @@ async function loadSeedHtml( return { error: direct.error, loginRequired: true }; } if (direct.ok) { - const hasCandidates = extractOutboundProspects({ html: direct.html, sourceUrl: url, limit: 1 }).length > 0; - if (hasCandidates || !allowRender) return { html: direct.html, rendered: false }; + const hasCandidates = + extractOutboundProspects({ html: direct.html, sourceUrl: url, limit: 1 }).length > 0; + const hasEntries = + !wantEntityLinks || + extractSameHostLinks({ html: direct.html, sourceUrl: url, limit: 5 }).some(isEntityPath); + if ((hasCandidates && hasEntries) || !allowRender) { + return { html: direct.html, rendered: false }; + } } else if (!allowRender) { return { error: direct.error }; } @@ -342,19 +377,34 @@ export async function discoverFromSeed(input: { detailHopThreshold?: number; }): Promise<{ prospects: DiscoveredProspect[]; + /** + * People named on the pages that were opened. + * + * Collected alongside prospects rather than instead of them. A directory + * page often yields both — the people it lists and its own contact + * address — and picking one throws away half of what the page offered. + */ + people: DiscoveredPerson[]; error?: string; notes?: string[]; /** Set when the seed was withheld pending a login, so the UI can offer to store one. */ loginRequired?: boolean; }> { - const limit = input.limit ?? 100; + const limit = input.limit ?? 1000; const allowRender = input.render !== false; const notes: string[] = []; - const seed = await loadSeedHtml(input.seedUrl, allowRender, input.credentials, input.codeWaiter); + const seed = await loadSeedHtml( + input.seedUrl, + allowRender, + input.credentials, + input.codeWaiter, + (input.depth ?? 1) >= 2, + ); if ("error" in seed) { return { prospects: [], + people: [], error: `seed ${input.seedUrl} failed: ${seed.error}`, loginRequired: seed.loginRequired, }; @@ -362,9 +412,13 @@ export async function discoverFromSeed(input: { if (seed.rendered) notes.push(`rendered ${input.seedUrl} in a browser`); const merged = new Map(); + const people = new Map(); for (const p of extractOutboundProspects({ html: seed.html, sourceUrl: input.seedUrl, limit })) { if (!merged.has(p.host)) merged.set(p.host, p); } + // A seed page is occasionally itself a profile. + const seedPerson = extractPerson(seed.html, input.seedUrl); + if (seedPerson) people.set(seedPerson.fullName, { ...seedPerson, sourceUrl: input.seedUrl }); // The second hop is expensive — one page load per listing, each possibly a // browser render — so it only runs when the first hop came up short. A @@ -392,11 +446,49 @@ export async function discoverFromSeed(input: { })) { if (!merged.has(p.host)) merged.set(p.host, p); } + // Both, from the same fetch. A people directory links only to itself, + // so the outbound pass above finds nothing there — but the page still + // names someone, and that is the whole reason the page was opened. + let person = extractPerson(detail.html, detailUrl); + + // A profile whose identity is injected client-side needs the render + // that loadSeedHtml declined to do. Its "good enough" test is whether + // the fetch produced any outbound candidate, and a directory's own + // footer satisfies that — so the shell is accepted and the person, who + // only exists after scripts run, is never seen. Rendering is retried + // only when the page looks like it should have named someone and + // didn't, so an ordinary listicle still costs one fetch. + if (!person && !detail.rendered && allowRender && isEntityPath(detailUrl)) { + const { renderPage } = await import("./render"); + const rendered = await renderPage(detailUrl, { + credentials: input.credentials, + codeWaiter: input.codeWaiter, + }); + if (rendered.ok) { + person = extractPerson(rendered.html, detailUrl); + for (const p of extractOutboundProspects({ + html: rendered.html, + sourceUrl: detailUrl, + limit: limit - merged.size, + })) { + if (!merged.has(p.host)) merged.set(p.host, p); + } + } + } + + if (person && !people.has(person.fullName)) { + people.set(person.fullName, { ...person, sourceUrl: detailUrl }); + } } notes.push(`opened ${detailPages.length} listing entries`); } - return { prospects: [...merged.values()], notes: notes.length ? notes : undefined }; + if (people.size) notes.push(`named ${people.size} people`); + return { + prospects: [...merged.values()], + people: [...people.values()], + notes: notes.length ? notes : undefined, + }; } /** Which search backend to use. */ diff --git a/lib/outreach/person.ts b/lib/outreach/person.ts new file mode 100644 index 00000000..2e02fa20 --- /dev/null +++ b/lib/outreach/person.ts @@ -0,0 +1,299 @@ +// Read a person off a profile page. +// +// The link-following discovery elsewhere in this directory assumes a page +// points at a business's own site. A people directory does not: it publishes +// a name, a title and a location, and links only to itself. Following links +// there yields the directory's own footer, which is what it did before this +// existed. +// +// So this extracts the person as an entity instead. Structured data first, +// because a page that publishes schema.org Person markup has already told us +// exactly who it is about and nothing needs to be guessed from prose. +// +// The trap worth naming: a profile page usually carries several JSON-LD +// blocks, and the first is typically the site's own Organization. Taking +// `json[0]` gets the directory rather than the person — on the page this was +// built against, that is the difference between "Marc van Neerven, CTO" and +// "StackUp, a technology consultancy". + +export type ExtractedPerson = { + fullName: string; + jobTitle: string | null; + company: string | null; + companySite: string | null; + description: string | null; + linkedinUrl: string | null; + /** Other profiles found on the page, keyed by network. */ + socials: Record; + location: string | null; + /** How we read it, so a weak guess can be told from structured data. */ + source: "json-ld" | "meta"; +}; + +type Json = Record; + +function asString(v: unknown): string | null { + if (typeof v === "string" && v.trim()) return v.trim(); + return null; +} + +/** Flatten @graph containers and arrays into a list of candidate nodes. */ +function flattenNodes(parsed: unknown): Json[] { + const out: Json[] = []; + const visit = (node: unknown) => { + if (Array.isArray(node)) { + node.forEach(visit); + return; + } + if (!node || typeof node !== "object") return; + const obj = node as Json; + out.push(obj); + if (Array.isArray(obj["@graph"])) visit(obj["@graph"]); + }; + visit(parsed); + return out; +} + +function typeOf(node: Json): string[] { + const t = node["@type"]; + if (typeof t === "string") return [t]; + if (Array.isArray(t)) return t.filter((x): x is string => typeof x === "string"); + return []; +} + +const LINKEDIN_RE = /^https?:\/\/([a-z]{2,3}\.)?linkedin\.com\//i; + +/** + * A clean LinkedIn profile URL, or null. + * + * Directories copy whatever the member pasted, and people paste the URL from + * their own logged-in view — which is a settings page carrying session + * tracking parameters, not a profile anyone else can open. Storing that is + * worse than storing nothing: it looks like a working link right up until + * someone clicks it. + */ +export function normalizeLinkedIn(raw: string): string | null { + let url: URL; + try { + url = new URL(raw); + } catch { + return null; + } + if (!/(^|\.)linkedin\.com$/i.test(url.hostname)) return null; + // Only member and company profiles are addressable by other people. + const m = url.pathname.match(/^\/(in|company|school)\/([^/]+)\/?$/i); + if (!m) return null; + return `https://www.linkedin.com/${m[1].toLowerCase()}/${m[2]}`; +} + +/** Which network a profile URL belongs to, for the socials map. */ +function networkOf(url: string): string | null { + const patterns: [RegExp, string][] = [ + [/linkedin\.com/i, "linkedin"], + [/(^|\/\/)(x|twitter)\.com/i, "x"], + [/github\.com/i, "github"], + [/mastodon|\.social/i, "mastodon"], + [/bsky\.app/i, "bluesky"], + [/youtube\.com/i, "youtube"], + [/instagram\.com/i, "instagram"], + ]; + for (const [re, name] of patterns) if (re.test(url)) return name; + return null; +} + +function personFromJsonLd(html: string): ExtractedPerson | null { + const blocks = [ + ...html.matchAll(/]+application\/ld\+json[^>]*>([\s\S]*?)<\/script>/gi), + ].map((m) => m[1]); + + for (const raw of blocks) { + let parsed: unknown; + try { + parsed = JSON.parse(raw.trim()); + } catch { + continue; + } + for (const node of flattenNodes(parsed)) { + // Every node is checked for Person rather than just the first block: + // the site's own Organization normally comes first. + if (!typeOf(node).includes("Person")) continue; + const fullName = asString(node.name); + if (!fullName) continue; + + const worksFor = node.worksFor; + let company: string | null = null; + let companySite: string | null = null; + if (worksFor && typeof worksFor === "object") { + const org = (Array.isArray(worksFor) ? worksFor[0] : worksFor) as Json; + company = asString(org?.name); + companySite = asString(org?.url); + } else { + company = asString(worksFor); + } + + const sameAs = Array.isArray(node.sameAs) + ? node.sameAs.filter((s): s is string => typeof s === "string") + : typeof node.sameAs === "string" + ? [node.sameAs] + : []; + + const socials: Record = {}; + let linkedinUrl: string | null = null; + for (const url of sameAs) { + if (LINKEDIN_RE.test(url) && !linkedinUrl) linkedinUrl = normalizeLinkedIn(url); + const net = networkOf(url); + if (net && !socials[net]) socials[net] = url; + } + + const address = node.address; + let location: string | null = null; + if (address && typeof address === "object") { + const a = address as Json; + location = + [asString(a.addressLocality), asString(a.addressRegion), asString(a.addressCountry)] + .filter(Boolean) + .join(", ") || null; + } else { + location = asString(address); + } + + // Directories frequently write the whole thing into jobTitle — "Owner + // and CTO at Artechra" — leaving worksFor empty. Splitting on " at " + // recovers the employer that would otherwise be lost inside the role. + const rawTitle = asString(node.jobTitle); + let jobTitle = rawTitle; + if (!company && rawTitle) { + const m = rawTitle.match(/^(.*?)\s+at\s+(.+)$/i); + if (m && m[1].trim() && m[2].trim()) { + jobTitle = m[1].trim(); + company = m[2].trim(); + } + } + + return { + fullName, + jobTitle, + company, + companySite, + description: asString(node.description), + linkedinUrl, + socials, + location, + source: "json-ld", + }; + } + } + return null; +} + + +/** + * Words that never begin a person's name and reliably begin a heading. + * + * The meta fallback previously accepted "For Fractional CTOs" and "AI + * Leadership Sprint" as people, because they are two-to-four capitalised + * words like a name. A fabricated name reaches a real inbox addressed to + * nobody, so the bar here is deliberately high: it is better to miss a + * person than to invent one. + */ +const NOT_A_NAME_START = + /^(for|the|a|an|our|your|my|why|how|what|when|top|best|about|meet|join|find|hire|get|introducing|welcome)$/i; + +const NOT_A_NAME_WORD = + /^(ctos?|ceos?|cfos?|founders?|members?|board|sprint|leadership|directory|guide|list|jobs?|careers?|services?|pricing|blog|news|team|home)$/i; + +function looksLikePersonName(value: string): boolean { + const words = value.split(/\s+/).filter(Boolean); + if (words.length < 2 || words.length > 4) return false; + if (NOT_A_NAME_START.test(words[0])) return false; + if (words.some((w) => NOT_A_NAME_WORD.test(w.replace(/[^A-Za-z]/g, "")))) return false; + // Every word starts with a capital, allowing lowercase particles that real + // names carry: van, de, der, bin, al. + const PARTICLE = /^(van|von|de|del|della|der|den|di|da|du|la|le|bin|al|ibn|mac|mc|o')$/i; + return words.every((w) => PARTICLE.test(w) || /^[A-Z]/.test(w)); +} + +/** + * Does the page claim to be about a person at all? + * + * Without this, any well-formed two-word heading on a marketing page becomes + * a contact. A profile URL or an og:type of profile is the page saying so + * itself, which is a far better signal than the shape of its title. + */ +function hasProfileSignal(html: string, url: string): boolean { + if (/\/(profile|people|person|member|members|team|u|author)\//i.test(url)) return true; + const ogType = metaContent(html, "og:type"); + return ogType === "profile"; +} + +function metaContent(html: string, key: string): string | null { + const re = new RegExp( + `]+(?:property|name)=["']${key}["'][^>]*content=["']([^"']*)["']`, + "i", + ); + const alt = new RegExp( + `]+content=["']([^"']*)["'][^>]*(?:property|name)=["']${key}["']`, + "i", + ); + return asString(html.match(re)?.[1]) ?? asString(html.match(alt)?.[1]); +} + +/** + * Fallback for pages with no Person markup. + * + * og:title on a profile page is conventionally "Name - Title | Site". The + * site segment after the final pipe is dropped, because it names the + * directory rather than anyone's employer — treating it as a company is how + * you end up with a thousand people who all work at "Fractional CTO + * Directory". + */ +function personFromMeta(html: string, url: string): ExtractedPerson | null { + const title = metaContent(html, "og:title") ?? asString(html.match(/]*>([^<]*)/i)?.[1]); + if (!title) return null; + + const withoutSite = title.split("|")[0].trim(); + const [namePart, ...rest] = withoutSite.split(/\s+[-–—]\s+/); + const fullName = asString(namePart); + if (!fullName) return null; + if (!looksLikePersonName(fullName)) return null; + if (!hasProfileSignal(html, url)) return null; + + return { + fullName, + jobTitle: rest.length ? rest.join(" - ").trim() : null, + company: null, + companySite: null, + description: metaContent(html, "og:description") ?? metaContent(html, "description"), + linkedinUrl: null, + socials: {}, + location: null, + source: "meta", + }; +} + +/** + * Extract the person a profile page is about, or null when it isn't about one. + * + * Returning null is the common and correct outcome — most pages are not + * profiles, and inventing a person from a headline would put a fabricated + * name into an email. + */ +export function extractPerson(html: string, url = ""): ExtractedPerson | null { + return personFromJsonLd(html) ?? personFromMeta(html, url); +} + +/** + * The search query that stands the best chance of finding this person's + * contact details. + * + * Name alone is ambiguous for anyone without an unusual one, so the employer + * or job title is included as a discriminator. Quoting the name keeps the + * engine from returning everyone who shares a surname. + */ +export function personSearchQuery(person: ExtractedPerson): string { + const parts = [`"${person.fullName}"`]; + if (person.company) parts.push(`"${person.company}"`); + else if (person.jobTitle) parts.push(person.jobTitle); + parts.push("(email OR contact)"); + return parts.join(" "); +} diff --git a/lib/outreach/render.ts b/lib/outreach/render.ts index 3565cdb3..98089100 100644 --- a/lib/outreach/render.ts +++ b/lib/outreach/render.ts @@ -35,6 +35,9 @@ const NAV_TIMEOUT_MS = 25_000; const SETTLE_MS = 2_500; const IDLE_SHUTDOWN_MS = 60_000; const MAX_HTML_BYTES = 3 * 1024 * 1024; +/** Bounded so an infinite feed cannot hold a render open forever. */ +const MAX_SCROLLS = 12; +const SCROLL_SETTLE_MS = 1_500; // Chromium is heavy enough that a per-call launch would dominate the cost of // a campaign tick, so one instance is shared across renders. @@ -175,6 +178,22 @@ export async function renderPage( .waitForLoadState("networkidle", { timeout: SETTLE_MS * 2 }) .catch(() => page.waitForTimeout(SETTLE_MS)); + // Directories commonly load a first page of entries and fetch the rest on + // scroll. Rendering without scrolling therefore sees a third of a list and + // reports it as the whole thing, which is indistinguishable from a short + // directory. Scrolling stops as soon as the page stops growing, so a + // static page costs one measurement rather than the full budget. + let previousHeight = 0; + for (let i = 0; i < MAX_SCROLLS; i++) { + const height = await page.evaluate(() => document.body.scrollHeight).catch(() => 0); + if (!height || height === previousHeight) break; + previousHeight = height; + await page.evaluate(() => window.scrollTo(0, document.body.scrollHeight)).catch(() => {}); + await page + .waitForLoadState("networkidle", { timeout: SCROLL_SETTLE_MS }) + .catch(() => page.waitForTimeout(SCROLL_SETTLE_MS)); + } + const title = await page.title().catch(() => ""); if (CHALLENGE_RE.test(title)) { return { diff --git a/tests/person-extract.test.ts b/tests/person-extract.test.ts new file mode 100644 index 00000000..e8e9bf00 --- /dev/null +++ b/tests/person-extract.test.ts @@ -0,0 +1,212 @@ +import { describe, it, expect } from "vitest"; +import { extractPerson, normalizeLinkedIn, personSearchQuery } from "@/lib/outreach/person"; + +// Taken from a real ctodirectory.com profile. The ordering is the point: the +// site's own Organization block comes first, so an extractor that reads +// json[0] returns the directory instead of the person. +const REAL_PROFILE = ` + + + + + +Marc van Neerven - Chief Technology Officer | Fractional CTO Directory +

Marc van Neerven

`; + +describe("extractPerson on a real directory profile", () => { + const person = extractPerson(REAL_PROFILE); + + it("picks the Person, not the site's own Organization", () => { + // json[0] is StackUp. Reading it would attribute every profile on the + // site to the directory's parent company. + expect(person?.fullName).toBe("Marc van Neerven"); + expect(person?.fullName).not.toBe("StackUp"); + }); + + it("reads the job title", () => { + expect(person?.jobTitle).toBe("Chief Technology Officer"); + }); + + it("prefers structured data over the title tag", () => { + expect(person?.source).toBe("json-ld"); + }); + + it("does not invent an employer the page never stated", () => { + // "Fractional CTO Directory" is the site, not where he works. + expect(person?.company).toBeNull(); + }); +}); + +describe("richer structured data", () => { + const html = ``; + const p = extractPerson(html); + + it("reads employer and their site", () => { + expect(p?.company).toBe("Acme Robotics"); + expect(p?.companySite).toBe("https://acme.test"); + }); + + it("reads location from a structured address", () => { + expect(p?.location).toBe("Bristol, UK"); + }); + + it("pulls LinkedIn out of sameAs", () => { + expect(p?.linkedinUrl).toBe("https://www.linkedin.com/in/janedoe"); + expect(p?.socials.github).toBe("https://github.com/janedoe"); + }); + + it("handles an @graph wrapper", () => { + const graph = ``; + expect(extractPerson(graph)?.fullName).toBe("Sam Patel"); + }); +}); + +describe("meta fallback", () => { + const TITLE = ``; + const PROFILE_URL = "https://ctodirectory.com/profile/chris-sprucefield-m8phr9"; + + it("parses the Name - Title | Site convention on a profile URL", () => { + const p = extractPerson(TITLE, PROFILE_URL); + expect(p?.fullName).toBe("Chris Sprucefield"); + expect(p?.jobTitle).toBe("Fractional CTO"); + expect(p?.source).toBe("meta"); + }); + + it("drops the site segment rather than treating it as an employer", () => { + expect(extractPerson(TITLE, PROFILE_URL)?.company).toBeNull(); + }); + + it("accepts og:type=profile in place of a profile URL", () => { + const html = `${TITLE}`; + expect(extractPerson(html, "https://example.test/x")?.fullName).toBe("Chris Sprucefield"); + }); + + it("refuses the same title on a page that never claims to be a profile", () => { + // Without a profile signal the shape of a title is not evidence about + // whether the page is about a person. + expect(extractPerson(TITLE, "https://ctodirectory.com/about")).toBeNull(); + }); +}); + +describe("headings that a live run mistook for people", () => { + // All four were extracted as contacts from ctodirectory.com marketing + // pages: capitalised, two to four words, and not human beings. A + // fabricated name reaches a real inbox addressed to nobody. + const notPeople = [ + "AI Leadership Sprint", + "For Fractional CTOs", + "For CEOs and Founders", + "For Board Members", + ]; + + for (const heading of notPeople) { + it(`rejects "${heading}"`, () => { + const html = `${heading}`; + expect(extractPerson(html, "https://ctodirectory.com/profile/x")).toBeNull(); + }); + } + + it("still accepts a real name with a lowercase particle", () => { + const html = `Marc van Neerven - Chief Technology Officer | Directory`; + expect(extractPerson(html, "https://ctodirectory.com/profile/marc-van-neerven-7lisfl")?.fullName) + .toBe("Marc van Neerven"); + }); +}); + +describe("refusing to invent a person", () => { + // Putting a fabricated name into a cold email is worse than extracting + // nothing, so anything that doesn't look like a person returns null. + it("returns null for a page with no person at all", () => { + expect(extractPerson("

Just a page

")).toBeNull(); + }); + + it("returns null for a headline rather than a name", () => { + const html = `The Top 500 CTOs To Watch In America This Year`; + expect(extractPerson(html)).toBeNull(); + }); + + it("returns null for a one-word title", () => { + expect(extractPerson(`Directory`)).toBeNull(); + }); + + it("ignores an Organization-only page", () => { + const html = ``; + expect(extractPerson(html)).toBeNull(); + }); +}); + +describe("personSearchQuery", () => { + it("quotes the name and adds the employer as a discriminator", () => { + const q = personSearchQuery({ + fullName: "Jane Doe", + jobTitle: "VP Engineering", + company: "Acme Robotics", + companySite: null, + description: null, + linkedinUrl: null, + socials: {}, + location: null, + source: "json-ld", + }); + expect(q).toContain('"Jane Doe"'); + expect(q).toContain('"Acme Robotics"'); + }); + + it("falls back to job title when there is no employer", () => { + const q = personSearchQuery({ + fullName: "Marc van Neerven", + jobTitle: "Chief Technology Officer", + company: null, + companySite: null, + description: null, + linkedinUrl: null, + socials: {}, + location: null, + source: "json-ld", + }); + expect(q).toContain('"Marc van Neerven"'); + expect(q).toContain("Chief Technology Officer"); + }); +}); + +describe("normalizeLinkedIn", () => { + it("rejects the logged-in settings URL people actually paste", () => { + // Copied verbatim from a live ctodirectory profile. It renders as a + // LinkedIn link and opens nothing for anyone else. + const pasted = + "https://www.linkedin.com/public-profile/settings/?trk=d_flagship3_profile_self_view_public_profile&lipi=urn%3Ali%3Apage%3Ad_flagship3"; + expect(normalizeLinkedIn(pasted)).toBeNull(); + }); + + it("keeps a real profile and strips tracking parameters", () => { + expect(normalizeLinkedIn("https://www.linkedin.com/in/mvneerven/?trk=abc")).toBe( + "https://www.linkedin.com/in/mvneerven", + ); + }); + + it("accepts company pages", () => { + expect(normalizeLinkedIn("https://linkedin.com/company/acme")).toBe( + "https://www.linkedin.com/company/acme", + ); + }); + + it("rejects a non-LinkedIn host that merely contains the word", () => { + expect(normalizeLinkedIn("https://notlinkedin.com.evil.test/in/someone")).toBeNull(); + }); +});