diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index b69a4fa8..acc0124e 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -50,22 +50,11 @@ const NON_PROSPECT_HOSTS = [ "eventbrite.com", "meetup.com", "substack.com", "medium.com", "blogspot.com", "godaddy.com", "cloudflare.com", "gstatic.com", "googleapis.com", "w3.org", "schema.org", "archive.org", "youtu.be", "bit.ly", "goo.gl", "t.co", - // Creative platforms and marketplaces. A profile on one of these is not a - // site the artist owns, and the outreach pipeline needs a domain they do. - "artstation.com", "behance.net", "adobe.com", "dribbble.com", "deviantart.com", - "sketchfab.com", "cgtrader.com", "turbosquid.com", "cults3d.com", "gumroad.com", + // Freelance marketplaces and link-in-bio hosts: a profile on one is not a + // domain anyone owns, and these are cross-niche rather than specific to any + // one campaign's industry. "upwork.com", "fiverr.com", "freelancer.com", "peopleperhour.com", "toptal.com", - "polywork.com", "contra.com", "patreon.com", "ko-fi.com", "buymeacoffee.com", - // Communities, showcases and trade press that dominate these searches. - "blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co", - "80.lv", "cgchannel.com", "gamedeveloper.com", "gamasutra.com", "wingfox.com", - "unrealengine.com", "unity.com", "blender.org", "autodesk.com", "maxon.net", - "itch.io", "gamejolt.com", "steampowered.com", - // Art and games schools whose domains give no hint of what they are. The - // patterns above catch anything with "academy" or ".edu" in it; these have - // to be named, and anything similar that turns up will need adding too. - "vanarts.com", "cgspectrum.com", "thinktankonline.com", "animationmentor.com", - "fxphd.com", "syn-studio.com", "lostboys-studios.com", + "patreon.com", "ko-fi.com", "buymeacoffee.com", "linktr.ee", "gumroad.com", ]; /** @@ -83,19 +72,16 @@ const NON_PROSPECT_HOSTS = [ * cheaper than the alternative. */ const NON_PROSPECT_PATTERNS: RegExp[] = [ - // Education. .edu and .ac.* are decisive; the words are strong signals. + // Education. These are decisive rather than a guess about any industry. /(^|\.)edu(\.[a-z]{2})?$/i, /(^|\.)ac\.[a-z]{2}$/i, - /(^|\.|-)(academy|acad|school|schule|institute|university|college|campus|bootcamp)(\.|-|$)/i, - // Learning and tutorials. - /(^|\.|-)(courses?|tutorials?|learn|training|masterclass)(\.|-|$)/i, - // Community and discussion. - /(^|\.|-)(forums?|community|wiki|discuss|board)(\.|-|$)/i, - // Publishing about the industry rather than working in it. - /(^|\.|-)(magazine|news|blog|press|podcast)(\.|-|$)/i, - // Hiring marketplaces and job boards: the artists there are reachable - // through the platform, not at a site they own. - /(^|\.|-)(jobs?|careers?|hiring|recruit)(\.|-|$)/i, + // Structural subdomains that describe what a page is, in any niche: a + // forum is a forum whether the subject is 3D art or dentistry. + /(^|\.)(forums?|community|wiki|discuss|board)\./i, + /(^|\.)(jobs?|careers?|support|help|docs?|status)\./i, + // A company's blog or newsroom is the same business as its apex domain, so + // treating it as its own prospect would pitch one business twice. + /(^|\.)(blog|news|press)\./i, ]; /** @@ -111,27 +97,25 @@ const NON_PROSPECT_PATTERNS: RegExp[] = [ * to profiles on their own domain, not to sites anyone owns. */ const MINEABLE_SOURCE_PATTERNS: RegExp[] = [ - /(^|\.|-)(forums?|community|discuss|board|wiki)(\.|-|$)/i, - /(^|\.|-)(magazine|news|blog|press)(\.|-|$)/i, - /(^|\.|-)(academy|school|institute|university|college|campus)(\.|-|$)/i, + /(^|\.)(forums?|community|discuss|board|wiki)\./i, + /(^|\.)(blog|news|press)\./i, /(^|\.)edu(\.[a-z]{2})?$/i, /(^|\.)ac\.[a-z]{2}$/i, ]; -const MINEABLE_HOSTS = [ - "blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co", - "80.lv", "cgchannel.com", "gamedeveloper.com", "reddit.com", "vanarts.com", -]; - /** * Is this host worth opening for the links it carries, even though nobody * there is a prospect? + * + * Decided from the shape of the hostname alone. Naming the actual sites would + * mean maintaining a list per industry — the community hubs for 3D artists + * have nothing in common with those for dentists or accountants — and a list + * like that is out of date the first time someone points a campaign at a + * niche nobody anticipated. */ export function isMineableSource(host: string): boolean { const h = normalizeHost(host); if (!h || !h.includes(".")) return false; - const apex = h.split(".").slice(-2).join("."); - if (MINEABLE_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`))) return true; return MINEABLE_SOURCE_PATTERNS.some((re) => re.test(h)); } diff --git a/lib/outreach/pipeline.ts b/lib/outreach/pipeline.ts index a98e8013..d04628aa 100644 --- a/lib/outreach/pipeline.ts +++ b/lib/outreach/pipeline.ts @@ -372,7 +372,9 @@ export async function researchProspect(input: { score_kind: audit.engine === "slop" ? "slop" : "aeo", top_issues: topIssues, quote_usd: quote.cappedForScoping ? null : Math.round(quote.amountUsd), - status: contact ? "researched" : "new", + // Same reasoning as the unscanned path: the scan landed, the search + // ran, and no address exists to send anything to. + status: contact ? "researched" : "skipped", ...(input.notes ? { notes: input.notes } : {}), }, { onConflict: "project_id,channel,target_key" }, @@ -452,8 +454,14 @@ async function researchWithoutScan(input: { discovery_label: input.discoveryLabel ?? null, contact_email: contact?.email ?? null, contact_source: contact?.source ?? null, - status: contact ? "researched" : "new", - ...(input.notes ? { notes: input.notes } : fallbackNote ? { notes: fallbackNote } : {}), + // Crawled the site, then searched for the business, and still found + // no address. There is nothing further to try, so it leaves the + // funnel rather than sitting at "new" and being re-researched on + // every tick for the life of the campaign. + status: contact ? "researched" : "skipped", + ...(input.notes + ? { notes: input.notes } + : { notes: contact ? null : (fallbackNote ?? "no contact address found") }), }, { onConflict: "project_id,channel,target_key" }, ) diff --git a/tests/no-niche-blacklist.test.ts b/tests/no-niche-blacklist.test.ts new file mode 100644 index 00000000..8d2bc086 --- /dev/null +++ b/tests/no-niche-blacklist.test.ts @@ -0,0 +1,66 @@ +import { describe, it, expect } from "vitest"; +import { readFileSync } from "node:fs"; +import path from "node:path"; + +// Discovery must not accumulate a per-industry blacklist. +// +// It nearly did: a run aimed at 3D artists produced art schools, Blender +// forums and CG trade press, and the obvious fix was to name them. That works +// for exactly one campaign. The community hubs for dentists, accountants or +// plumbers have nothing in common with those, so a named list is both endless +// and stale the moment someone points a campaign at a niche nobody +// anticipated. +// +// Filtering therefore has to be decided from the shape of a hostname, or from +// whether a contact can actually be found — never from knowing an industry. +// This test is the thing that notices when that erodes. + +const DISCOVER = path.join(process.cwd(), "lib/outreach/discover.ts"); + +/** Hosts from one real campaign's results. None belongs in the source. */ +const NICHE_HOSTS = [ + "blenderartists.org", + "polycount.com", + "cgsociety.org", + "therookies.co", + "80.lv", + "cgchannel.com", + "gamedeveloper.com", + "wingfox.com", + "vanarts.com", + "cgspectrum.com", + "thinktankonline.com", + "animationmentor.com", + "artstation.com", + "behance.net", + "sketchfab.com", + "cgtrader.com", + "turbosquid.com", + "unrealengine.com", + "blender.org", + "autodesk.com", +]; + +describe("discovery carries no industry-specific hostnames", () => { + const source = readFileSync(DISCOVER, "utf8"); + + for (const host of NICHE_HOSTS) { + it(`does not name ${host}`, () => { + expect( + source.includes(host), + `${host} is hardcoded in discover.ts. Filtering has to work for any niche — decide it from the hostname's shape, or from whether a contact can be found.`, + ).toBe(false); + }); + } + + it("also keeps industry words out of the patterns", () => { + // "academy" and "portfolio" are one industry's vocabulary; "forum" and + // "wiki" describe what a page is in any of them. + for (const word of ["academy", "portfolio", "artist", "3d", "gaming"]) { + expect( + source.toLowerCase().includes(`|${word}`), + `"${word}" reads like industry vocabulary in a filter pattern`, + ).toBe(false); + } + }); +}); diff --git a/tests/prospect-filter.test.ts b/tests/prospect-filter.test.ts index 3a849e07..0f387b90 100644 --- a/tests/prospect-filter.test.ts +++ b/tests/prospect-filter.test.ts @@ -1,43 +1,38 @@ import { describe, it, expect } from "vitest"; import { isMineableSource, isNonProspectHost } from "@/lib/outreach/discover"; -// Every host below actually came out of a live discovery run for -// "3d artist portfolio"-shaped queries. Searching for artists returns the -// industry around artists, and all of it has a contact address, so without -// filtering a campaign ends up cold-emailing an art school about a job. -describe("hosts that a portfolio search keeps returning", () => { +// The filter has to work for a campaign in any niche. Naming the community +// hubs for 3D artists would do nothing for one aimed at dentists or +// accountants, and a list like that is stale the first time someone points a +// campaign somewhere nobody anticipated. So everything below is decided from +// the shape of a hostname, never from knowing an industry. + +describe("hosts that are never a prospect, in any niche", () => { const rejected = [ - // Platforms and marketplaces — a profile there is not a site they own. - "artstation.com", - "brainchild.artstation.com", - "behance.net", - "adobe.com", - "portfolio.adobe.com", - "sketchfab.com", - "cgtrader.com", - "upwork.com", - "fiverr.com", - // Education. - "vanarts.com", + // Education, decisively. "gnomon.edu", "someschool.ac.uk", - "cg-academy.com", - "3d-bootcamp.io", - // Community, showcase and trade press. - "blenderartists.org", - "polycount.com", - "therookies.co", - "80.lv", - "gamedeveloper.com", - "blog.wingfox.com", + "mit.edu", + // Structural subdomains: a forum is a forum whatever the subject. "forums.example.com", + "forum.example.com", + "community.example.com", "wiki.example.com", - // Tooling vendors. - "unrealengine.com", - "blender.org", - // Job boards. "jobs.example.com", "careers.example.com", + "support.example.com", + "docs.example.com", + // Freelance marketplaces and link-in-bio: a profile is not a domain + // anyone owns, and these are cross-niche. + "upwork.com", + "fiverr.com", + "toptal.com", + "linktr.ee", + // Covered by the pre-existing cross-niche list in lib/leadCampaign. + "linkedin.com", + "instagram.com", + "reddit.com", + "x.com", ]; for (const host of rejected) { @@ -47,17 +42,17 @@ describe("hosts that a portfolio search keeps returning", () => { } }); -describe("hosts that are the artists themselves", () => { +describe("hosts that are the business itself", () => { const accepted = [ "jonathancaridia.com", "bengtsondesigns.com", "janedoe.design", "studio-nine.co.uk", "hardsurface.art", - "m-kowalski.dev", - // A real trap: contains "art" and "station" separately but is not the - // platform, and must not be caught by a sloppy substring match. - "artstationary.com", + // Niche-neutral: the filter must not have opinions about industries. + "smile-dental.com", + "bright-accounting.co.uk", + "acme-plumbing.net", ]; for (const host of accepted) { @@ -65,66 +60,67 @@ describe("hosts that are the artists themselves", () => { expect(isNonProspectHost(host)).toBe(false); }); } -}); -describe("filter shape", () => { - it("rejects junk that is not a host at all", () => { - for (const junk of ["", "localhost", "not a host"]) { - expect(isNonProspectHost(junk)).toBe(true); + it("does not reject a domain for containing a keyword mid-word", () => { + // The patterns are anchored on separators, so an ordinary word inside a + // domain does not cost a real prospect. + for (const host of ["schoonerdesign.com", "newsomstudio.com", "boardmanlaw.com"]) { + expect(isNonProspectHost(host), host).toBe(false); } }); - - it("does not reject a domain merely for containing a keyword mid-word", () => { - // "schooner" contains "school"; the pattern is anchored on separators so - // an ordinary word does not cost us a real prospect. - expect(isNonProspectHost("schoonerdesign.com")).toBe(false); - expect(isNonProspectHost("newsomstudio.com")).toBe(false); - }); }); describe("isMineableSource — not a prospect, still worth reading", () => { - // A forum thread or alumni page is never someone to email, but it is where - // artists' own sites actually appear. Discarding those results throws away - // the best source of personal domains in the pipeline. + // A forum thread is where people's own sites actually appear: a personal + // domain rarely out-ranks the community discussing the work. Discarding + // those results throws away the best source of real domains in the pipeline. const mineable = [ - "blenderartists.org", - "polycount.com", - "cgsociety.org", - "therookies.co", - "80.lv", - "gamedeveloper.com", - "reddit.com", - "vanarts.com", + "forums.example.com", + "community.example.com", + "wiki.example.com", + "blog.example.com", + "news.example.com", "gnomon.edu", "someschool.ac.uk", - "forums.example.com", - "blog.wingfox.com", ]; for (const host of mineable) { it(`mines ${host}`, () => { expect(isMineableSource(host)).toBe(true); - // Still never a prospect: mined for links, never emailed. + // Mined for links, never emailed. expect(isNonProspectHost(host)).toBe(true); }); } it("does not mine marketplaces, whose links stay on their own domain", () => { - for (const host of ["artstation.com", "behance.net", "upwork.com", "fiverr.com", "sketchfab.com"]) { + for (const host of ["upwork.com", "fiverr.com", "toptal.com", "linktr.ee"]) { expect(isMineableSource(host), host).toBe(false); } }); - it("does not mine tooling vendors", () => { - for (const host of ["unrealengine.com", "blender.org", "autodesk.com"]) { + it("does not mine a business's own site — it is a prospect, not a source", () => { + for (const host of ["jonathancaridia.com", "smile-dental.com"]) { expect(isMineableSource(host), host).toBe(false); + expect(isNonProspectHost(host), host).toBe(false); } }); +}); - it("does not mine an artist's own site — it is a prospect, not a source", () => { - for (const host of ["jonathancaridia.com", "janedoe.design"]) { - expect(isMineableSource(host), host).toBe(false); - expect(isNonProspectHost(host), host).toBe(false); +describe("filter shape", () => { + it("rejects junk that is not a host at all", () => { + for (const junk of ["", "localhost", "not a host"]) { + expect(isNonProspectHost(junk)).toBe(true); + } + }); + + it("carries no industry-specific hostnames", () => { + // A guard against the filter drifting back into a per-niche blacklist: + // these are real hosts from one campaign's results, and none of them + // should be named in the code. + for (const host of ["blenderartists.org", "vanarts.com", "80.lv", "polycount.com"]) { + // They may still be caught structurally, but must not be hardcoded — + // asserted in tests/no-niche-blacklist.test.ts against the source. + expect(typeof isNonProspectHost(host)).toBe("boolean"); } }); });