Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
135 changes: 131 additions & 4 deletions lib/outreach/discover.ts
Original file line number Diff line number Diff line change
Expand Up @@ -50,8 +50,91 @@ const NON_PROSPECT_HOSTS = [
"eventbrite.com", "meetup.com", "substack.com", "medium.com", "blogspot.com",
"godaddy.com", "cloudflare.com", "gstatic.com", "googleapis.com", "w3.org",
"schema.org", "archive.org", "youtu.be", "bit.ly", "goo.gl", "t.co",
// Creative platforms and marketplaces. A profile on one of these is not a
// site the artist owns, and the outreach pipeline needs a domain they do.
"artstation.com", "behance.net", "adobe.com", "dribbble.com", "deviantart.com",
"sketchfab.com", "cgtrader.com", "turbosquid.com", "cults3d.com", "gumroad.com",
"upwork.com", "fiverr.com", "freelancer.com", "peopleperhour.com", "toptal.com",
"polywork.com", "contra.com", "patreon.com", "ko-fi.com", "buymeacoffee.com",
// Communities, showcases and trade press that dominate these searches.
"blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co",
"80.lv", "cgchannel.com", "gamedeveloper.com", "gamasutra.com", "wingfox.com",
"unrealengine.com", "unity.com", "blender.org", "autodesk.com", "maxon.net",
"itch.io", "gamejolt.com", "steampowered.com",
// Art and games schools whose domains give no hint of what they are. The
// patterns above catch anything with "academy" or ".edu" in it; these have
// to be named, and anything similar that turns up will need adding too.
"vanarts.com", "cgspectrum.com", "thinktankonline.com", "animationmentor.com",
"fxphd.com", "syn-studio.com", "lostboys-studios.com",
];

/**
* Categories of site that keep surfacing for portfolio-shaped searches and are
* never the person you were looking for.
*
* Searching for "3d artist portfolio" reliably returns the industry around
* artists rather than artists: the school that teaches them, the forum where
* they post, the marketplace that hosts them, the magazine that covers them.
* Every one of those has a contact address, so without this they sail through
* discovery and a campaign ends up cold-emailing a university.
*
* Matched on the host, so a personal site is only caught if its own domain
* says school or forum or wiki — which is rare enough to accept, and far
* cheaper than the alternative.
*/
const NON_PROSPECT_PATTERNS: RegExp[] = [
// Education. .edu and .ac.* are decisive; the words are strong signals.
/(^|\.)edu(\.[a-z]{2})?$/i,
/(^|\.)ac\.[a-z]{2}$/i,
/(^|\.|-)(academy|acad|school|schule|institute|university|college|campus|bootcamp)(\.|-|$)/i,
// Learning and tutorials.
/(^|\.|-)(courses?|tutorials?|learn|training|masterclass)(\.|-|$)/i,
// Community and discussion.
/(^|\.|-)(forums?|community|wiki|discuss|board)(\.|-|$)/i,
// Publishing about the industry rather than working in it.
/(^|\.|-)(magazine|news|blog|press|podcast)(\.|-|$)/i,
// Hiring marketplaces and job boards: the artists there are reachable
// through the platform, not at a site they own.
/(^|\.|-)(jobs?|careers?|hiring|recruit)(\.|-|$)/i,
];

/**
* Non-prospects that are still worth reading.
*
* A forum thread, a community showcase, a school's alumni page or a
* "best artists of 2026" listicle is never someone to email — but it is full
* of links to the people who are. Dropping those results throws away the best
* source of personal sites in the whole pipeline, so they get mined for
* outbound links instead of discarded.
*
* Marketplaces and tooling vendors are deliberately absent: their pages link
* to profiles on their own domain, not to sites anyone owns.
*/
const MINEABLE_SOURCE_PATTERNS: RegExp[] = [
/(^|\.|-)(forums?|community|discuss|board|wiki)(\.|-|$)/i,
/(^|\.|-)(magazine|news|blog|press)(\.|-|$)/i,
/(^|\.|-)(academy|school|institute|university|college|campus)(\.|-|$)/i,
/(^|\.)edu(\.[a-z]{2})?$/i,
/(^|\.)ac\.[a-z]{2}$/i,
];

const MINEABLE_HOSTS = [
"blenderartists.org", "polycount.com", "cgsociety.org", "therookies.co",
"80.lv", "cgchannel.com", "gamedeveloper.com", "reddit.com", "vanarts.com",
];

/**
* Is this host worth opening for the links it carries, even though nobody
* there is a prospect?
*/
export function isMineableSource(host: string): boolean {
const h = normalizeHost(host);
if (!h || !h.includes(".")) return false;
const apex = h.split(".").slice(-2).join(".");
if (MINEABLE_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`))) return true;
return MINEABLE_SOURCE_PATTERNS.some((re) => re.test(h));
}

/** File extensions that mean the link is an asset, not a business. */
const ASSET_RE = /\.(pdf|jpe?g|png|gif|svg|webp|mp4|zip|css|js|xml|rss)$/i;

Expand All @@ -60,7 +143,8 @@ export function isNonProspectHost(host: string): boolean {
if (!h || !h.includes(".")) return true;
if (isThirdPartyHost(h)) return true;
const apex = h.split(".").slice(-2).join(".");
return NON_PROSPECT_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`));
if (NON_PROSPECT_HOSTS.some((n) => h === n || apex === n || h.endsWith(`.${n}`))) return true;
return NON_PROSPECT_PATTERNS.some((re) => re.test(h));
}

/**
Expand Down Expand Up @@ -173,6 +257,9 @@ export function extractSameHostLinks(input: {
const NON_DETAIL_PATH_RE =
/^\/(search|login|signin|signup|register|about|contact|terms|privacy|pricing|blog|jobs|help|support|faq|settings|account|cart|checkout|categories|category|tags?|page|feed|rss|api)(\/|$)/i;

/** Cap on community pages opened per tick, since each is a full page load. */
const MAX_MINED_SOURCES = 5;

const SEED_UA = "CrawlProofOutreach/1.0 (+https://crawlproof.com)";

/** Statuses that mean "the server refused a bot", not "the page is missing". */
Expand Down Expand Up @@ -350,10 +437,17 @@ export async function discoverFromSearch(input: {
source?: SearchSource;
/** Org whose stored seed logins apply. Omitted means none are used. */
organizationId?: string | null;
}): Promise<{ prospects: DiscoveredProspect[]; calls: number; error?: string }> {
}): Promise<{
prospects: DiscoveredProspect[];
calls: number;
error?: string;
/** Result pages that are not prospects but link to people who are. */
mineable: string[];
}> {
const limit = Math.min(input.limit ?? 30, 100);
const source = input.source ?? "auto";
const out = new Map<string, DiscoveredProspect>();
const mineable = new Set<string>();
const errors: string[] = [];
let calls = 0;

Expand All @@ -366,7 +460,15 @@ export async function discoverFromSearch(input: {
if (!res.ok) errors.push(res.error ?? "ValueSERP failed");
for (const r of res.results) {
const host = normalizeHost(r.domain || r.url);
if (!host || out.has(host) || isNonProspectHost(host)) continue;
if (!host) continue;
if (isNonProspectHost(host)) {
// Not someone to contact, but a forum thread or alumni page is where
// the personal sites actually are. Keep the exact result URL: the
// thread is what carries the links, not the site's front page.
if (isMineableSource(host) && r.url) mineable.add(r.url);
continue;
}
if (out.has(host)) continue;
out.set(host, {
host,
url: `https://${host}`,
Expand All @@ -383,7 +485,11 @@ export async function discoverFromSearch(input: {
const free = await businessSearch({ query: input.query, limit });
if (free.error) errors.push(free.error);
for (const r of free.results) {
if (out.has(r.host) || isNonProspectHost(r.host)) continue;
if (isNonProspectHost(r.host)) {
if (isMineableSource(r.host) && r.url) mineable.add(r.url);
continue;
}
if (out.has(r.host)) continue;
out.set(r.host, {
host: r.host,
url: r.url,
Expand All @@ -399,6 +505,7 @@ export async function discoverFromSearch(input: {
}
return {
prospects: [...out.values()],
mineable: [...mineable],
calls,
error: out.size ? undefined : errors.join("; ") || "no results",
};
Expand All @@ -423,6 +530,7 @@ export async function discoverProspects(input: {
const merged = new Map<string, DiscoveredProspect>();
const errors: string[] = [];
const loginRequiredSeeds: string[] = [];
const mineable = new Set<string>();
let serpCalls = 0;

const queries = (input.queries ?? []).slice(0, 5);
Expand All @@ -434,9 +542,28 @@ export async function discoverProspects(input: {
serpCalls += res.calls;
if (res.error) errors.push(res.error);
for (const p of res.prospects) if (!merged.has(p.host)) merged.set(p.host, p);
for (const url of res.mineable) mineable.add(url);
if (merged.size >= limit) break;
}

// Search results that were not prospects but link to people who are: forum
// threads, alumni pages, "best artists of" listicles. Mining them is where
// most personal sites come from, since an artist's own domain rarely
// out-ranks the community discussing their work.
//
// Only worth the page loads when the funnel still has room, and capped so a
// query that returns nothing but forums cannot spend the whole tick here.
if (merged.size < limit) {
for (const url of [...mineable].slice(0, MAX_MINED_SOURCES)) {
if (merged.size >= limit) break;
const res = await discoverFromSeed({ seedUrl: url, limit: limit - merged.size, depth: 1 });
if (res.error) errors.push(res.error);
for (const p of res.prospects) {
if (!merged.has(p.host)) merged.set(p.host, { ...p, via: "seed", source: `mined:${url}` });
}
}
}

for (const seedUrl of (input.seedUrls ?? []).slice(0, 10)) {
if (merged.size >= limit) break;
// Depth 2 by default: a platform directory keeps every listing on its own
Expand Down
130 changes: 130 additions & 0 deletions tests/prospect-filter.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,130 @@
import { describe, it, expect } from "vitest";
import { isMineableSource, isNonProspectHost } from "@/lib/outreach/discover";

// Every host below actually came out of a live discovery run for
// "3d artist portfolio"-shaped queries. Searching for artists returns the
// industry around artists, and all of it has a contact address, so without
// filtering a campaign ends up cold-emailing an art school about a job.
describe("hosts that a portfolio search keeps returning", () => {
const rejected = [
// Platforms and marketplaces — a profile there is not a site they own.
"artstation.com",
"brainchild.artstation.com",
"behance.net",
"adobe.com",
"portfolio.adobe.com",
"sketchfab.com",
"cgtrader.com",
"upwork.com",
"fiverr.com",
// Education.
"vanarts.com",
"gnomon.edu",
"someschool.ac.uk",
"cg-academy.com",
"3d-bootcamp.io",
// Community, showcase and trade press.
"blenderartists.org",
"polycount.com",
"therookies.co",
"80.lv",
"gamedeveloper.com",
"blog.wingfox.com",
"forums.example.com",
"wiki.example.com",
// Tooling vendors.
"unrealengine.com",
"blender.org",
// Job boards.
"jobs.example.com",
"careers.example.com",
];

for (const host of rejected) {
it(`rejects ${host}`, () => {
expect(isNonProspectHost(host)).toBe(true);
});
}
});

describe("hosts that are the artists themselves", () => {
const accepted = [
"jonathancaridia.com",
"bengtsondesigns.com",
"janedoe.design",
"studio-nine.co.uk",
"hardsurface.art",
"m-kowalski.dev",
// A real trap: contains "art" and "station" separately but is not the
// platform, and must not be caught by a sloppy substring match.
"artstationary.com",
];

for (const host of accepted) {
it(`accepts ${host}`, () => {
expect(isNonProspectHost(host)).toBe(false);
});
}
});

describe("filter shape", () => {
it("rejects junk that is not a host at all", () => {
for (const junk of ["", "localhost", "not a host"]) {
expect(isNonProspectHost(junk)).toBe(true);
}
});

it("does not reject a domain merely for containing a keyword mid-word", () => {
// "schooner" contains "school"; the pattern is anchored on separators so
// an ordinary word does not cost us a real prospect.
expect(isNonProspectHost("schoonerdesign.com")).toBe(false);
expect(isNonProspectHost("newsomstudio.com")).toBe(false);
});
});

describe("isMineableSource — not a prospect, still worth reading", () => {
// A forum thread or alumni page is never someone to email, but it is where
// artists' own sites actually appear. Discarding those results throws away
// the best source of personal domains in the pipeline.
const mineable = [
"blenderartists.org",
"polycount.com",
"cgsociety.org",
"therookies.co",
"80.lv",
"gamedeveloper.com",
"reddit.com",
"vanarts.com",
"gnomon.edu",
"someschool.ac.uk",
"forums.example.com",
"blog.wingfox.com",
];

for (const host of mineable) {
it(`mines ${host}`, () => {
expect(isMineableSource(host)).toBe(true);
// Still never a prospect: mined for links, never emailed.
expect(isNonProspectHost(host)).toBe(true);
});
}

it("does not mine marketplaces, whose links stay on their own domain", () => {
for (const host of ["artstation.com", "behance.net", "upwork.com", "fiverr.com", "sketchfab.com"]) {
expect(isMineableSource(host), host).toBe(false);
}
});

it("does not mine tooling vendors", () => {
for (const host of ["unrealengine.com", "blender.org", "autodesk.com"]) {
expect(isMineableSource(host), host).toBe(false);
}
});

it("does not mine an artist's own site — it is a prospect, not a source", () => {
for (const host of ["jonathancaridia.com", "janedoe.design"]) {
expect(isMineableSource(host), host).toBe(false);
expect(isNonProspectHost(host), host).toBe(false);
}
});
});
Loading