From fcba49bd2cefa696ba57bf9d5dd12395a4a6d2da Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 03:42:54 +0000 Subject: [PATCH] fix(leads): filter by shape, not by naming an industry's websites MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The filter added in #138 was a list of 3D-art hostnames — Blender forums, CG trade press, art schools, ArtStation. It worked for exactly one campaign. The community hubs for dentists or accountants have nothing in common with those, so the list was both endless and stale the moment anyone pointed a campaign at a niche nobody anticipated. It is gone. What is left decides from the shape of a hostname rather than from knowing an industry: .edu and .ac.*, and the structural subdomains that mean the same thing everywhere — forum, community, wiki, jobs, support, docs, blog, news. A forum is a forum whether the subject is character modelling or root canals. Freelance marketplaces and link-in-bio hosts stay, because those are cross-niche by nature, and the giants were already covered by lib/leadCampaign. Mining works the same way now, so a community page is opened for the links it carries without anyone having listed it first. The other half is simply giving up. A prospect whose site published no address, and which the search fallback could not find either, was left at "new" — so every tick researched it again, forever, and the funnel filled with businesses that could never be contacted. Those are skipped now, with the reason recorded. The guard is a test asserting none of those hostnames appears in discover.ts, and that the patterns contain no industry vocabulary. It exists because the pull toward naming the site in front of you is strong and the cost only shows up in someone else's niche. Co-Authored-By: Claude Opus 5 (1M context) --- lib/outreach/discover.ts | 56 +++++-------- lib/outreach/pipeline.ts | 14 +++- tests/no-niche-blacklist.test.ts | 66 ++++++++++++++++ tests/prospect-filter.test.ts | 132 +++++++++++++++---------------- 4 files changed, 161 insertions(+), 107 deletions(-) create mode 100644 tests/no-niche-blacklist.test.ts diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index b69a4fa8..acc0124e 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -50,22 +50,11 @@ const NON_PROSPECT_HOSTS = [ "eventbrite.com", "meetup.com", "substack.com", "medium.com", "blogspot.com", "godaddy.com", "cloudflare.com", "gstatic.com", "googleapis.com", "w3.org", "schema.org", "archive.org", "youtu.be", "bit.ly", "goo.gl", "t.co", - // Creative platforms and marketplaces. A profile on one of these is not a - // site the artist owns, and the outreach pipeline needs a domain they do. - "artstation.com", "behance.net", "adobe.com", "dribbble.com", "deviantart.com", - "sketchfab.com", "cgtrader.com", "turbosquid.com", "cults3d.com", "gumroad.com", + // Freelance marketplaces and link-in-bio hosts: a profile on one is not a + // domain anyone owns, and these are cross-niche rather than specific to any + // one campaign's industry. "upwork.com", "fiverr.com", "freelancer.com", "peopleperhour.com", "toptal.com", - "polywork.com", "contra.com", "patreon.com", "ko-fi.com", "buymeacoffee.com", - // Communities, showcases and trade press that dominate these searches. - "blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co", - "80.lv", "cgchannel.com", "gamedeveloper.com", "gamasutra.com", "wingfox.com", - "unrealengine.com", "unity.com", "blender.org", "autodesk.com", "maxon.net", - "itch.io", "gamejolt.com", "steampowered.com", - // Art and games schools whose domains give no hint of what they are. The - // patterns above catch anything with "academy" or ".edu" in it; these have - // to be named, and anything similar that turns up will need adding too. - "vanarts.com", "cgspectrum.com", "thinktankonline.com", "animationmentor.com", - "fxphd.com", "syn-studio.com", "lostboys-studios.com", + "patreon.com", "ko-fi.com", "buymeacoffee.com", "linktr.ee", "gumroad.com", ]; /** @@ -83,19 +72,16 @@ const NON_PROSPECT_HOSTS = [ * cheaper than the alternative. */ const NON_PROSPECT_PATTERNS: RegExp[] = [ - // Education. .edu and .ac.* are decisive; the words are strong signals. + // Education. These are decisive rather than a guess about any industry. /(^|\.)edu(\.[a-z]{2})?$/i, /(^|\.)ac\.[a-z]{2}$/i, - /(^|\.|-)(academy|acad|school|schule|institute|university|college|campus|bootcamp)(\.|-|$)/i, - // Learning and tutorials. - /(^|\.|-)(courses?|tutorials?|learn|training|masterclass)(\.|-|$)/i, - // Community and discussion. - /(^|\.|-)(forums?|community|wiki|discuss|board)(\.|-|$)/i, - // Publishing about the industry rather than working in it. - /(^|\.|-)(magazine|news|blog|press|podcast)(\.|-|$)/i, - // Hiring marketplaces and job boards: the artists there are reachable - // through the platform, not at a site they own. - /(^|\.|-)(jobs?|careers?|hiring|recruit)(\.|-|$)/i, + // Structural subdomains that describe what a page is, in any niche: a + // forum is a forum whether the subject is 3D art or dentistry. + /(^|\.)(forums?|community|wiki|discuss|board)\./i, + /(^|\.)(jobs?|careers?|support|help|docs?|status)\./i, + // A company's blog or newsroom is the same business as its apex domain, so + // treating it as its own prospect would pitch one business twice. + /(^|\.)(blog|news|press)\./i, ]; /** @@ -111,27 +97,25 @@ const NON_PROSPECT_PATTERNS: RegExp[] = [ * to profiles on their own domain, not to sites anyone owns. */ const MINEABLE_SOURCE_PATTERNS: RegExp[] = [ - /(^|\.|-)(forums?|community|discuss|board|wiki)(\.|-|$)/i, - /(^|\.|-)(magazine|news|blog|press)(\.|-|$)/i, - /(^|\.|-)(academy|school|institute|university|college|campus)(\.|-|$)/i, + /(^|\.)(forums?|community|discuss|board|wiki)\./i, + /(^|\.)(blog|news|press)\./i, /(^|\.)edu(\.[a-z]{2})?$/i, /(^|\.)ac\.[a-z]{2}$/i, ]; -const MINEABLE_HOSTS = [ - "blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co", - "80.lv", "cgchannel.com", "gamedeveloper.com", "reddit.com", "vanarts.com", -]; - /** * Is this host worth opening for the links it carries, even though nobody * there is a prospect? + * + * Decided from the shape of the hostname alone. Naming the actual sites would + * mean maintaining a list per industry — the community hubs for 3D artists + * have nothing in common with those for dentists or accountants — and a list + * like that is out of date the first time someone points a campaign at a + * niche nobody anticipated. */ export function isMineableSource(host: string): boolean { const h = normalizeHost(host); if (!h || !h.includes(".")) return false; - const apex = h.split(".").slice(-2).join("."); - if (MINEABLE_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`))) return true; return MINEABLE_SOURCE_PATTERNS.some((re) => re.test(h)); } diff --git a/lib/outreach/pipeline.ts b/lib/outreach/pipeline.ts index a98e8013..d04628aa 100644 --- a/lib/outreach/pipeline.ts +++ b/lib/outreach/pipeline.ts @@ -372,7 +372,9 @@ export async function researchProspect(input: { score_kind: audit.engine === "slop" ? "slop" : "aeo", top_issues: topIssues, quote_usd: quote.cappedForScoping ? null : Math.round(quote.amountUsd), - status: contact ? "researched" : "new", + // Same reasoning as the unscanned path: the scan landed, the search + // ran, and no address exists to send anything to. + status: contact ? "researched" : "skipped", ...(input.notes ? { notes: input.notes } : {}), }, { onConflict: "project_id,channel,target_key" }, @@ -452,8 +454,14 @@ async function researchWithoutScan(input: { discovery_label: input.discoveryLabel ?? null, contact_email: contact?.email ?? null, contact_source: contact?.source ?? null, - status: contact ? "researched" : "new", - ...(input.notes ? { notes: input.notes } : fallbackNote ? { notes: fallbackNote } : {}), + // Crawled the site, then searched for the business, and still found + // no address. There is nothing further to try, so it leaves the + // funnel rather than sitting at "new" and being re-researched on + // every tick for the life of the campaign. + status: contact ? "researched" : "skipped", + ...(input.notes + ? { notes: input.notes } + : { notes: contact ? null : (fallbackNote ?? "no contact address found") }), }, { onConflict: "project_id,channel,target_key" }, ) diff --git a/tests/no-niche-blacklist.test.ts b/tests/no-niche-blacklist.test.ts new file mode 100644 index 00000000..8d2bc086 --- /dev/null +++ b/tests/no-niche-blacklist.test.ts @@ -0,0 +1,66 @@ +import { describe, it, expect } from "vitest"; +import { readFileSync } from "node:fs"; +import path from "node:path"; + +// Discovery must not accumulate a per-industry blacklist. +// +// It nearly did: a run aimed at 3D artists produced art schools, Blender +// forums and CG trade press, and the obvious fix was to name them. That works +// for exactly one campaign. The community hubs for dentists, accountants or +// plumbers have nothing in common with those, so a named list is both endless +// and stale the moment someone points a campaign at a niche nobody +// anticipated. +// +// Filtering therefore has to be decided from the shape of a hostname, or from +// whether a contact can actually be found — never from knowing an industry. +// This test is the thing that notices when that erodes. + +const DISCOVER = path.join(process.cwd(), "lib/outreach/discover.ts"); + +/** Hosts from one real campaign's results. None belongs in the source. */ +const NICHE_HOSTS = [ + "blenderartists.org", + "polycount.com", + "cgsociety.org", + "therookies.co", + "80.lv", + "cgchannel.com", + "gamedeveloper.com", + "wingfox.com", + "vanarts.com", + "cgspectrum.com", + "thinktankonline.com", + "animationmentor.com", + "artstation.com", + "behance.net", + "sketchfab.com", + "cgtrader.com", + "turbosquid.com", + "unrealengine.com", + "blender.org", + "autodesk.com", +]; + +describe("discovery carries no industry-specific hostnames", () => { + const source = readFileSync(DISCOVER, "utf8"); + + for (const host of NICHE_HOSTS) { + it(`does not name ${host}`, () => { + expect( + source.includes(host), + `${host} is hardcoded in discover.ts. Filtering has to work for any niche — decide it from the hostname's shape, or from whether a contact can be found.`, + ).toBe(false); + }); + } + + it("also keeps industry words out of the patterns", () => { + // "academy" and "portfolio" are one industry's vocabulary; "forum" and + // "wiki" describe what a page is in any of them. + for (const word of ["academy", "portfolio", "artist", "3d", "gaming"]) { + expect( + source.toLowerCase().includes(`|${word}`), + `"${word}" reads like industry vocabulary in a filter pattern`, + ).toBe(false); + } + }); +}); diff --git a/tests/prospect-filter.test.ts b/tests/prospect-filter.test.ts index 3a849e07..0f387b90 100644 --- a/tests/prospect-filter.test.ts +++ b/tests/prospect-filter.test.ts @@ -1,43 +1,38 @@ import { describe, it, expect } from "vitest"; import { isMineableSource, isNonProspectHost } from "@/lib/outreach/discover"; -// Every host below actually came out of a live discovery run for -// "3d artist portfolio"-shaped queries. Searching for artists returns the -// industry around artists, and all of it has a contact address, so without -// filtering a campaign ends up cold-emailing an art school about a job. -describe("hosts that a portfolio search keeps returning", () => { +// The filter has to work for a campaign in any niche. Naming the community +// hubs for 3D artists would do nothing for one aimed at dentists or +// accountants, and a list like that is stale the first time someone points a +// campaign somewhere nobody anticipated. So everything below is decided from +// the shape of a hostname, never from knowing an industry. + +describe("hosts that are never a prospect, in any niche", () => { const rejected = [ - // Platforms and marketplaces — a profile there is not a site they own. - "artstation.com", - "brainchild.artstation.com", - "behance.net", - "adobe.com", - "portfolio.adobe.com", - "sketchfab.com", - "cgtrader.com", - "upwork.com", - "fiverr.com", - // Education. - "vanarts.com", + // Education, decisively. "gnomon.edu", "someschool.ac.uk", - "cg-academy.com", - "3d-bootcamp.io", - // Community, showcase and trade press. - "blenderartists.org", - "polycount.com", - "therookies.co", - "80.lv", - "gamedeveloper.com", - "blog.wingfox.com", + "mit.edu", + // Structural subdomains: a forum is a forum whatever the subject. "forums.example.com", + "forum.example.com", + "community.example.com", "wiki.example.com", - // Tooling vendors. - "unrealengine.com", - "blender.org", - // Job boards. "jobs.example.com", "careers.example.com", + "support.example.com", + "docs.example.com", + // Freelance marketplaces and link-in-bio: a profile is not a domain + // anyone owns, and these are cross-niche. + "upwork.com", + "fiverr.com", + "toptal.com", + "linktr.ee", + // Covered by the pre-existing cross-niche list in lib/leadCampaign. + "linkedin.com", + "instagram.com", + "reddit.com", + "x.com", ]; for (const host of rejected) { @@ -47,17 +42,17 @@ describe("hosts that a portfolio search keeps returning", () => { } }); -describe("hosts that are the artists themselves", () => { +describe("hosts that are the business itself", () => { const accepted = [ "jonathancaridia.com", "bengtsondesigns.com", "janedoe.design", "studio-nine.co.uk", "hardsurface.art", - "m-kowalski.dev", - // A real trap: contains "art" and "station" separately but is not the - // platform, and must not be caught by a sloppy substring match. - "artstationary.com", + // Niche-neutral: the filter must not have opinions about industries. + "smile-dental.com", + "bright-accounting.co.uk", + "acme-plumbing.net", ]; for (const host of accepted) { @@ -65,66 +60,67 @@ describe("hosts that are the artists themselves", () => { expect(isNonProspectHost(host)).toBe(false); }); } -}); -describe("filter shape", () => { - it("rejects junk that is not a host at all", () => { - for (const junk of ["", "localhost", "not a host"]) { - expect(isNonProspectHost(junk)).toBe(true); + it("does not reject a domain for containing a keyword mid-word", () => { + // The patterns are anchored on separators, so an ordinary word inside a + // domain does not cost a real prospect. + for (const host of ["schoonerdesign.com", "newsomstudio.com", "boardmanlaw.com"]) { + expect(isNonProspectHost(host), host).toBe(false); } }); - - it("does not reject a domain merely for containing a keyword mid-word", () => { - // "schooner" contains "school"; the pattern is anchored on separators so - // an ordinary word does not cost us a real prospect. - expect(isNonProspectHost("schoonerdesign.com")).toBe(false); - expect(isNonProspectHost("newsomstudio.com")).toBe(false); - }); }); describe("isMineableSource — not a prospect, still worth reading", () => { - // A forum thread or alumni page is never someone to email, but it is where - // artists' own sites actually appear. Discarding those results throws away - // the best source of personal domains in the pipeline. + // A forum thread is where people's own sites actually appear: a personal + // domain rarely out-ranks the community discussing the work. Discarding + // those results throws away the best source of real domains in the pipeline. const mineable = [ - "blenderartists.org", - "polycount.com", - "cgsociety.org", - "therookies.co", - "80.lv", - "gamedeveloper.com", - "reddit.com", - "vanarts.com", + "forums.example.com", + "community.example.com", + "wiki.example.com", + "blog.example.com", + "news.example.com", "gnomon.edu", "someschool.ac.uk", - "forums.example.com", - "blog.wingfox.com", ]; for (const host of mineable) { it(`mines ${host}`, () => { expect(isMineableSource(host)).toBe(true); - // Still never a prospect: mined for links, never emailed. + // Mined for links, never emailed. expect(isNonProspectHost(host)).toBe(true); }); } it("does not mine marketplaces, whose links stay on their own domain", () => { - for (const host of ["artstation.com", "behance.net", "upwork.com", "fiverr.com", "sketchfab.com"]) { + for (const host of ["upwork.com", "fiverr.com", "toptal.com", "linktr.ee"]) { expect(isMineableSource(host), host).toBe(false); } }); - it("does not mine tooling vendors", () => { - for (const host of ["unrealengine.com", "blender.org", "autodesk.com"]) { + it("does not mine a business's own site — it is a prospect, not a source", () => { + for (const host of ["jonathancaridia.com", "smile-dental.com"]) { expect(isMineableSource(host), host).toBe(false); + expect(isNonProspectHost(host), host).toBe(false); } }); +}); - it("does not mine an artist's own site — it is a prospect, not a source", () => { - for (const host of ["jonathancaridia.com", "janedoe.design"]) { - expect(isMineableSource(host), host).toBe(false); - expect(isNonProspectHost(host), host).toBe(false); +describe("filter shape", () => { + it("rejects junk that is not a host at all", () => { + for (const junk of ["", "localhost", "not a host"]) { + expect(isNonProspectHost(junk)).toBe(true); + } + }); + + it("carries no industry-specific hostnames", () => { + // A guard against the filter drifting back into a per-niche blacklist: + // these are real hosts from one campaign's results, and none of them + // should be named in the code. + for (const host of ["blenderartists.org", "vanarts.com", "80.lv", "polycount.com"]) { + // They may still be caught structurally, but must not be hardcoded — + // asserted in tests/no-niche-blacklist.test.ts against the source. + expect(typeof isNonProspectHost(host)).toBe("boolean"); } }); });