From a1012dc578880eed705d51331fc216e0b9b175f0 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 03:36:16 +0000 Subject: [PATCH] feat(leads): filter the industry out of portfolio searches, and mine it instead MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Searching for artists returns the industry around artists. A live run for "3d artist portfolio" produced an art school, a competition site, a Blender forum, a tutorial blog, Adobe, ArtStation and Unreal — every one with a contact address on it, so every one sailed through discovery. Left alone, the campaign would have cold-emailed a university about an equity-only job. Those hosts are no longer prospects. Alongside the explicit list there are patterns for the categories that keep recurring — education, forums, wikis, trade press, job boards — anchored on separators so an ordinary word inside a domain does not cost a real prospect. "schoonerdesign.com" contains "school" and is fine. Patterns cannot catch a school whose domain does not say school, so the well-known ones are named: vanarts.com is the Vancouver Institute of Media Arts, and nothing about the domain says so. Anything similar that turns up will need adding the same way. But a forum thread is not merely noise — it is where artists' own sites actually appear, since a personal domain rarely out-ranks the community discussing the work. So the mineable ones are opened for their outbound links rather than discarded, keeping the exact result URL because the thread carries the links, not the front page. Marketplaces and tooling vendors are deliberately excluded: their pages link to profiles on their own domain, not to sites anyone owns. Mining is capped per tick and only runs while the funnel still has room, so a query returning nothing but forums cannot spend the whole tick on page loads. Co-Authored-By: Claude Opus 5 (1M context) --- lib/outreach/discover.ts | 135 +++++++++++++++++++++++++++++++++- tests/prospect-filter.test.ts | 130 ++++++++++++++++++++++++++++++++ 2 files changed, 261 insertions(+), 4 deletions(-) create mode 100644 tests/prospect-filter.test.ts diff --git a/lib/outreach/discover.ts b/lib/outreach/discover.ts index 26392518..b69a4fa8 100644 --- a/lib/outreach/discover.ts +++ b/lib/outreach/discover.ts @@ -50,8 +50,91 @@ const NON_PROSPECT_HOSTS = [ "eventbrite.com", "meetup.com", "substack.com", "medium.com", "blogspot.com", "godaddy.com", "cloudflare.com", "gstatic.com", "googleapis.com", "w3.org", "schema.org", "archive.org", "youtu.be", "bit.ly", "goo.gl", "t.co", + // Creative platforms and marketplaces. A profile on one of these is not a + // site the artist owns, and the outreach pipeline needs a domain they do. + "artstation.com", "behance.net", "adobe.com", "dribbble.com", "deviantart.com", + "sketchfab.com", "cgtrader.com", "turbosquid.com", "cults3d.com", "gumroad.com", + "upwork.com", "fiverr.com", "freelancer.com", "peopleperhour.com", "toptal.com", + "polywork.com", "contra.com", "patreon.com", "ko-fi.com", "buymeacoffee.com", + // Communities, showcases and trade press that dominate these searches. + "blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co", + "80.lv", "cgchannel.com", "gamedeveloper.com", "gamasutra.com", "wingfox.com", + "unrealengine.com", "unity.com", "blender.org", "autodesk.com", "maxon.net", + "itch.io", "gamejolt.com", "steampowered.com", + // Art and games schools whose domains give no hint of what they are. The + // patterns above catch anything with "academy" or ".edu" in it; these have + // to be named, and anything similar that turns up will need adding too. + "vanarts.com", "cgspectrum.com", "thinktankonline.com", "animationmentor.com", + "fxphd.com", "syn-studio.com", "lostboys-studios.com", ]; +/** + * Categories of site that keep surfacing for portfolio-shaped searches and are + * never the person you were looking for. + * + * Searching for "3d artist portfolio" reliably returns the industry around + * artists rather than artists: the school that teaches them, the forum where + * they post, the marketplace that hosts them, the magazine that covers them. + * Every one of those has a contact address, so without this they sail through + * discovery and a campaign ends up cold-emailing a university. + * + * Matched on the host, so a personal site is only caught if its own domain + * says school or forum or wiki — which is rare enough to accept, and far + * cheaper than the alternative. + */ +const NON_PROSPECT_PATTERNS: RegExp[] = [ + // Education. .edu and .ac.* are decisive; the words are strong signals. + /(^|\.)edu(\.[a-z]{2})?$/i, + /(^|\.)ac\.[a-z]{2}$/i, + /(^|\.|-)(academy|acad|school|schule|institute|university|college|campus|bootcamp)(\.|-|$)/i, + // Learning and tutorials. + /(^|\.|-)(courses?|tutorials?|learn|training|masterclass)(\.|-|$)/i, + // Community and discussion. + /(^|\.|-)(forums?|community|wiki|discuss|board)(\.|-|$)/i, + // Publishing about the industry rather than working in it. + /(^|\.|-)(magazine|news|blog|press|podcast)(\.|-|$)/i, + // Hiring marketplaces and job boards: the artists there are reachable + // through the platform, not at a site they own. + /(^|\.|-)(jobs?|careers?|hiring|recruit)(\.|-|$)/i, +]; + +/** + * Non-prospects that are still worth reading. + * + * A forum thread, a community showcase, a school's alumni page or a + * "best artists of 2026" listicle is never someone to email — but it is full + * of links to the people who are. Dropping those results throws away the best + * source of personal sites in the whole pipeline, so they get mined for + * outbound links instead of discarded. + * + * Marketplaces and tooling vendors are deliberately absent: their pages link + * to profiles on their own domain, not to sites anyone owns. + */ +const MINEABLE_SOURCE_PATTERNS: RegExp[] = [ + /(^|\.|-)(forums?|community|discuss|board|wiki)(\.|-|$)/i, + /(^|\.|-)(magazine|news|blog|press)(\.|-|$)/i, + /(^|\.|-)(academy|school|institute|university|college|campus)(\.|-|$)/i, + /(^|\.)edu(\.[a-z]{2})?$/i, + /(^|\.)ac\.[a-z]{2}$/i, +]; + +const MINEABLE_HOSTS = [ + "blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co", + "80.lv", "cgchannel.com", "gamedeveloper.com", "reddit.com", "vanarts.com", +]; + +/** + * Is this host worth opening for the links it carries, even though nobody + * there is a prospect? + */ +export function isMineableSource(host: string): boolean { + const h = normalizeHost(host); + if (!h || !h.includes(".")) return false; + const apex = h.split(".").slice(-2).join("."); + if (MINEABLE_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`))) return true; + return MINEABLE_SOURCE_PATTERNS.some((re) => re.test(h)); +} + /** File extensions that mean the link is an asset, not a business. */ const ASSET_RE = /\.(pdf|jpe?g|png|gif|svg|webp|mp4|zip|css|js|xml|rss)$/i; @@ -60,7 +143,8 @@ export function isNonProspectHost(host: string): boolean { if (!h || !h.includes(".")) return true; if (isThirdPartyHost(h)) return true; const apex = h.split(".").slice(-2).join("."); - return NON_PROSPECT_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`)); + if (NON_PROSPECT_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`))) return true; + return NON_PROSPECT_PATTERNS.some((re) => re.test(h)); } /** @@ -173,6 +257,9 @@ export function extractSameHostLinks(input: { const NON_DETAIL_PATH_RE = /^\/(search|login|signin|signup|register|about|contact|terms|privacy|pricing|blog|jobs|help|support|faq|settings|account|cart|checkout|categories|category|tags?|page|feed|rss|api)(\/|$)/i; +/** Cap on community pages opened per tick, since each is a full page load. */ +const MAX_MINED_SOURCES = 5; + const SEED_UA = "CrawlProofOutreach/1.0 (+https://crawlproof.com)"; /** Statuses that mean "the server refused a bot", not "the page is missing". */ @@ -350,10 +437,17 @@ export async function discoverFromSearch(input: { source?: SearchSource; /** Org whose stored seed logins apply. Omitted means none are used. */ organizationId?: string | null; -}): Promise<{ prospects: DiscoveredProspect[]; calls: number; error?: string }> { +}): Promise<{ + prospects: DiscoveredProspect[]; + calls: number; + error?: string; + /** Result pages that are not prospects but link to people who are. */ + mineable: string[]; +}> { const limit = Math.min(input.limit ?? 30, 100); const source = input.source ?? "auto"; const out = new Map(); + const mineable = new Set(); const errors: string[] = []; let calls = 0; @@ -366,7 +460,15 @@ export async function discoverFromSearch(input: { if (!res.ok) errors.push(res.error ?? "ValueSERP failed"); for (const r of res.results) { const host = normalizeHost(r.domain || r.url); - if (!host || out.has(host) || isNonProspectHost(host)) continue; + if (!host) continue; + if (isNonProspectHost(host)) { + // Not someone to contact, but a forum thread or alumni page is where + // the personal sites actually are. Keep the exact result URL: the + // thread is what carries the links, not the site's front page. + if (isMineableSource(host) && r.url) mineable.add(r.url); + continue; + } + if (out.has(host)) continue; out.set(host, { host, url: `https://${host}`, @@ -383,7 +485,11 @@ export async function discoverFromSearch(input: { const free = await businessSearch({ query: input.query, limit }); if (free.error) errors.push(free.error); for (const r of free.results) { - if (out.has(r.host) || isNonProspectHost(r.host)) continue; + if (isNonProspectHost(r.host)) { + if (isMineableSource(r.host) && r.url) mineable.add(r.url); + continue; + } + if (out.has(r.host)) continue; out.set(r.host, { host: r.host, url: r.url, @@ -399,6 +505,7 @@ export async function discoverFromSearch(input: { } return { prospects: [...out.values()], + mineable: [...mineable], calls, error: out.size ? undefined : errors.join("; ") || "no results", }; @@ -423,6 +530,7 @@ export async function discoverProspects(input: { const merged = new Map(); const errors: string[] = []; const loginRequiredSeeds: string[] = []; + const mineable = new Set(); let serpCalls = 0; const queries = (input.queries ?? []).slice(0, 5); @@ -434,9 +542,28 @@ export async function discoverProspects(input: { serpCalls += res.calls; if (res.error) errors.push(res.error); for (const p of res.prospects) if (!merged.has(p.host)) merged.set(p.host, p); + for (const url of res.mineable) mineable.add(url); if (merged.size >= limit) break; } + // Search results that were not prospects but link to people who are: forum + // threads, alumni pages, "best artists of" listicles. Mining them is where + // most personal sites come from, since an artist's own domain rarely + // out-ranks the community discussing their work. + // + // Only worth the page loads when the funnel still has room, and capped so a + // query that returns nothing but forums cannot spend the whole tick here. + if (merged.size < limit) { + for (const url of [...mineable].slice(0, MAX_MINED_SOURCES)) { + if (merged.size >= limit) break; + const res = await discoverFromSeed({ seedUrl: url, limit: limit - merged.size, depth: 1 }); + if (res.error) errors.push(res.error); + for (const p of res.prospects) { + if (!merged.has(p.host)) merged.set(p.host, { ...p, via: "seed", source: `mined:${url}` }); + } + } + } + for (const seedUrl of (input.seedUrls ?? []).slice(0, 10)) { if (merged.size >= limit) break; // Depth 2 by default: a platform directory keeps every listing on its own diff --git a/tests/prospect-filter.test.ts b/tests/prospect-filter.test.ts new file mode 100644 index 00000000..3a849e07 --- /dev/null +++ b/tests/prospect-filter.test.ts @@ -0,0 +1,130 @@ +import { describe, it, expect } from "vitest"; +import { isMineableSource, isNonProspectHost } from "@/lib/outreach/discover"; + +// Every host below actually came out of a live discovery run for +// "3d artist portfolio"-shaped queries. Searching for artists returns the +// industry around artists, and all of it has a contact address, so without +// filtering a campaign ends up cold-emailing an art school about a job. +describe("hosts that a portfolio search keeps returning", () => { + const rejected = [ + // Platforms and marketplaces — a profile there is not a site they own. + "artstation.com", + "brainchild.artstation.com", + "behance.net", + "adobe.com", + "portfolio.adobe.com", + "sketchfab.com", + "cgtrader.com", + "upwork.com", + "fiverr.com", + // Education. + "vanarts.com", + "gnomon.edu", + "someschool.ac.uk", + "cg-academy.com", + "3d-bootcamp.io", + // Community, showcase and trade press. + "blenderartists.org", + "polycount.com", + "therookies.co", + "80.lv", + "gamedeveloper.com", + "blog.wingfox.com", + "forums.example.com", + "wiki.example.com", + // Tooling vendors. + "unrealengine.com", + "blender.org", + // Job boards. + "jobs.example.com", + "careers.example.com", + ]; + + for (const host of rejected) { + it(`rejects ${host}`, () => { + expect(isNonProspectHost(host)).toBe(true); + }); + } +}); + +describe("hosts that are the artists themselves", () => { + const accepted = [ + "jonathancaridia.com", + "bengtsondesigns.com", + "janedoe.design", + "studio-nine.co.uk", + "hardsurface.art", + "m-kowalski.dev", + // A real trap: contains "art" and "station" separately but is not the + // platform, and must not be caught by a sloppy substring match. + "artstationary.com", + ]; + + for (const host of accepted) { + it(`accepts ${host}`, () => { + expect(isNonProspectHost(host)).toBe(false); + }); + } +}); + +describe("filter shape", () => { + it("rejects junk that is not a host at all", () => { + for (const junk of ["", "localhost", "not a host"]) { + expect(isNonProspectHost(junk)).toBe(true); + } + }); + + it("does not reject a domain merely for containing a keyword mid-word", () => { + // "schooner" contains "school"; the pattern is anchored on separators so + // an ordinary word does not cost us a real prospect. + expect(isNonProspectHost("schoonerdesign.com")).toBe(false); + expect(isNonProspectHost("newsomstudio.com")).toBe(false); + }); +}); + +describe("isMineableSource — not a prospect, still worth reading", () => { + // A forum thread or alumni page is never someone to email, but it is where + // artists' own sites actually appear. Discarding those results throws away + // the best source of personal domains in the pipeline. + const mineable = [ + "blenderartists.org", + "polycount.com", + "cgsociety.org", + "therookies.co", + "80.lv", + "gamedeveloper.com", + "reddit.com", + "vanarts.com", + "gnomon.edu", + "someschool.ac.uk", + "forums.example.com", + "blog.wingfox.com", + ]; + + for (const host of mineable) { + it(`mines ${host}`, () => { + expect(isMineableSource(host)).toBe(true); + // Still never a prospect: mined for links, never emailed. + expect(isNonProspectHost(host)).toBe(true); + }); + } + + it("does not mine marketplaces, whose links stay on their own domain", () => { + for (const host of ["artstation.com", "behance.net", "upwork.com", "fiverr.com", "sketchfab.com"]) { + expect(isMineableSource(host), host).toBe(false); + } + }); + + it("does not mine tooling vendors", () => { + for (const host of ["unrealengine.com", "blender.org", "autodesk.com"]) { + expect(isMineableSource(host), host).toBe(false); + } + }); + + it("does not mine an artist's own site — it is a prospect, not a source", () => { + for (const host of ["jonathancaridia.com", "janedoe.design"]) { + expect(isMineableSource(host), host).toBe(false); + expect(isNonProspectHost(host), host).toBe(false); + } + }); +});