From 7cf54bcd4b5929f98cf8e708472a3c5969786d88 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 13 Aug 2026 03:57:16 +0000 Subject: [PATCH] fix(outreach): stop campaigns burning the ValueSERP plan on dead queries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three active campaigns emptied the 25,000-call monthly ValueSERP plan in six days, after which every search returned HTTP 402 — including the ones a human typed into the lead finder, which surfaced as "No free engine returned results" once the free fallbacks were tried and also failed. The runner was not misbehaving; it was doing what it was told. The funnel gate counts only prospects still in flight, so a campaign that has contacted everyone it found reads as empty forever, re-running its identical five-query list every fifteen minutes and paying full price for hosts it already had. Three campaigns worked out to ~43k searches a month against a 25k plan, at a measured 0-7 new prospects a day. Counting contacted prospects toward the target would have stopped it, but target_pipeline means "leads to keep in the funnel" and defaults to 25 — so that would cap a campaign at 25 leads for its entire life and turn an autopilot into a one-shot. Instead the gate stays as it is and an unproductive pass waits longer before trying again: 30 minutes doubling to a 24h ceiling, reset by a single new prospect, person or intent signal. Nothing is capped, and a dry campaign drops from 96 discovery passes a day to one. Also: - ValueSERP now parks on HTTP 402 for ten minutes rather than letting all ~21 searches in a tick each pay a round-trip to learn the plan is empty. No credit is billed either way; this buys latency and legible errors. - SERP_CALLS_PER_MONTH authorised a single pro account 200k calls out of a shared 25k bucket, so the cost backstop could not actually stop anything. Brought under the vendor plan, which is now named alongside it. The free engines cannot cover for any of this from a server: DuckDuckGo answers its anti-bot challenge (HTTP 202) from every datacentre IP tried, and Mojeek 403s from Railway while serving the same query from a residential host. Co-Authored-By: Claude Opus 5 (1M context) --- lib/alerts/limits.ts | 14 ++- lib/alerts/valueserp.ts | 36 +++++++ lib/outreach/runner.ts | 73 +++++++++++++- ...60813040000_outreach_discovery_backoff.sql | 28 ++++++ tests/discovery-backoff.test.ts | 95 +++++++++++++++++++ tests/lead-billing.test.ts | 9 +- 6 files changed, 250 insertions(+), 5 deletions(-) create mode 100644 supabase/migrations/20260813040000_outreach_discovery_backoff.sql create mode 100644 tests/discovery-backoff.test.ts diff --git a/lib/alerts/limits.ts b/lib/alerts/limits.ts index e278f381..44af8e6e 100644 --- a/lib/alerts/limits.ts +++ b/lib/alerts/limits.ts @@ -19,10 +19,20 @@ export const MAX_ACTIVE_ALERTS: Record = { // Monthly SERP-call budget by plan — the hard cost backstop. Free is set so a // typical user never touches it, while a cap-abuser is bounded to well under // a dollar. Paid budgets assume hourly checks (24× the call volume). +// +// These are per-account and must stay under VALUESERP_MONTHLY_PLAN, which is +// the whole shared bucket: the old pro budget of 200k authorised one account +// to spend eight times everything CrawlProof buys in a month, so the backstop +// could not actually stop anything. A pro account running 250 hourly alerts +// needs 250 × 24 × 30 = 180k to never touch its cap, which the 25k plan cannot +// fund — the cap is the honest number, and the plan is what to raise if a real +// customer starts hitting it. +export const VALUESERP_MONTHLY_PLAN = 25_000; + export const SERP_CALLS_PER_MONTH: Record = { free: 400, - pro: 200_000, - team: 1_000_000, + pro: 8_000, + team: 15_000, }; // Hourly is currently free for everyone (no paywall on frequency yet — the diff --git a/lib/alerts/valueserp.ts b/lib/alerts/valueserp.ts index 82aa2d9b..efa77784 100644 --- a/lib/alerts/valueserp.ts +++ b/lib/alerts/valueserp.ts @@ -50,6 +50,28 @@ export function hasValueSerpKey(): boolean { return Boolean(env.valueSerpApiKey); } +/** + * When the account is out of credits, stop asking until this passes. + * + * ValueSERP answers an exhausted plan with HTTP 402, and the plan is a monthly + * bucket — once it is empty every later call in the cycle gets the same + * answer. Without this, a single campaign tick fires twenty-odd searches that + * are all guaranteed to fail, each paying a network round-trip and filing its + * own error, and the run summary reads as twenty distinct problems rather than + * one. Nothing is billed either way (a 402 consumes no credit), so this buys + * latency and legible errors, not money. + * + * Deliberately short. The reset time is not in the search response, so this + * guesses low rather than parking a working key for hours. + */ +const OUT_OF_CREDITS_COOLDOWN_MS = 10 * 60_000; +let outOfCreditsUntil = 0; + +/** Exposed so tests can reset the module-level cooldown between cases. */ +export function resetSerpCreditCooldown(): void { + outOfCreditsUntil = 0; +} + /** * Run one ValueSERP search. Returns billable `calls` even on an empty result * set so the caller can debit the budget accurately. Retries once on a @@ -63,6 +85,14 @@ export async function searchSerp(input: { if (!env.valueSerpApiKey) { return { ok: false, calls: 0, results: [], error: "VALUESERP_API_KEY not set" }; } + if (Date.now() < outOfCreditsUntil) { + return { + ok: false, + calls: 0, + results: [], + error: "ValueSERP is out of monthly credits (HTTP 402) — not retrying until the cooldown passes", + }; + } const num = Math.min(input.num ?? RESULTS_PER_CHECK, 100); const params = new URLSearchParams({ api_key: env.valueSerpApiKey, @@ -91,6 +121,12 @@ export async function searchSerp(input: { clearTimeout(timer); if (!res.ok) { lastErr = `ValueSERP HTTP ${res.status}`; + // 402 is an empty monthly plan, which no later call in this cycle can + // change. Park every caller rather than letting each one rediscover it. + if (res.status === 402) { + outOfCreditsUntil = Date.now() + OUT_OF_CREDITS_COOLDOWN_MS; + lastErr = "ValueSERP HTTP 402 — monthly credit plan is exhausted"; + } // 4xx (bad query, out of quota) won't fix on retry. if (res.status < 500) return { ok: false, calls: 0, results: [], error: lastErr }; continue; diff --git a/lib/outreach/runner.ts b/lib/outreach/runner.ts index 875d3661..1e34c61a 100644 --- a/lib/outreach/runner.ts +++ b/lib/outreach/runner.ts @@ -47,6 +47,10 @@ export type CampaignRow = { max_score: number; daily_send_limit: number; target_pipeline: number; + /** When discovery may next run. Null means now. See discoveryBackoffMinutes. */ + discovery_backoff_until: string | null; + /** Consecutive discovery passes that produced nothing. Drives the back-off. */ + discovery_dry_streak: number; auto_send: boolean; follow_ups: boolean; angle: string | null; @@ -69,7 +73,7 @@ export type CampaignRow = { }; export const CAMPAIGN_COLUMNS = - "id, project_id, owner_id, name, channel, active, queries, seed_urls, keywords, subreddits, negative_keywords, max_score, daily_send_limit, target_pipeline, auto_send, follow_ups, angle, sender_name, reply_to, last_run_at, pitch_mode, pitch_intro, pitch_ask, pitch_facts, scan_prospects, min_intent, intent_sources, intent_recency, sells_description, alert_email, alerts_enabled"; + "id, project_id, owner_id, name, channel, active, queries, seed_urls, keywords, subreddits, negative_keywords, max_score, daily_send_limit, target_pipeline, discovery_backoff_until, discovery_dry_streak, auto_send, follow_ups, angle, sender_name, reply_to, last_run_at, pitch_mode, pitch_intro, pitch_ask, pitch_facts, scan_prospects, min_intent, intent_sources, intent_recency, sells_description, alert_email, alerts_enabled"; export type TickResult = { campaign: string; @@ -112,6 +116,30 @@ const MAX_CONTACT_SEARCHES_PER_TICK = 10; // same request does not need finding four times an hour. const MAX_INTENT_SEARCHES_PER_TICK = 6; +// Back-off for a campaign whose queries have stopped producing. +// +// The funnel gate below counts only prospects still in flight, so a campaign +// that has contacted everyone it found reads as empty forever and re-runs its +// identical query list every fifteen minutes — paying full search price for +// hosts it already has. Left alone, three campaigns at five queries a tick +// spend ~43k searches a month against a 25k plan, and the whole quota is gone +// in a week. +// +// Counting contacted prospects toward the target would stop it, but it would +// also cap a campaign at target_pipeline leads for its entire life, which +// turns an autopilot into a one-shot. So the gate stays as it is and an +// unproductive pass simply waits longer before trying again: doubling from +// half an hour up to a day. Nothing is capped — a tapped-out query set is +// still retried daily, and one new result resets the campaign to full speed. +const DISCOVERY_BACKOFF_BASE_MIN = 30; +const DISCOVERY_BACKOFF_MAX_MIN = 24 * 60; + +/** How long a campaign should wait after `streak` consecutive empty passes. */ +export function discoveryBackoffMinutes(streak: number): number { + if (streak < 1) return 0; + return Math.min(DISCOVERY_BACKOFF_BASE_MIN * 2 ** (streak - 1), DISCOVERY_BACKOFF_MAX_MIN); +} + export async function runEmailCampaignTick(campaign: CampaignRow): Promise { const sb = serviceClient(); // Shared across every prospect this tick researches, so the ceiling is per @@ -266,9 +294,23 @@ export async function runEmailCampaignTick(campaign: CampaignRow): Promise ["new", "researched", "drafted"].includes(p.status)).length; + // Whether the queries have earned another run yet. Checked before canSpend + // so a backed-off campaign costs nothing at all — not a search, not a + // billing round-trip. + const backoffUntil = campaign.discovery_backoff_until + ? new Date(campaign.discovery_backoff_until) + : null; + const backedOff = backoffUntil !== null && backoffUntil.getTime() > Date.now(); + if (backedOff && backoffUntil) { + result.skipped.push( + `discovery: resting until ${backoffUntil.toISOString()} after ${campaign.discovery_dry_streak} empty pass(es)`, + ); + } // Discovery is the most expensive stage — several search calls before a // single prospect exists — so it is gated like the rest. - if (liveCount < campaign.target_pipeline && (await canSpend())) { + let discoveryRan = false; + if (liveCount < campaign.target_pipeline && !backedOff && (await canSpend())) { + discoveryRan = true; const want = Math.min(campaign.target_pipeline - liveCount, MAX_DISCOVER_PER_TICK); const { data: projectRow } = await sb .from("projects") @@ -399,6 +441,33 @@ export async function runEmailCampaignTick(campaign: CampaignRow): Promise { + it("does not rest a campaign that just produced something", () => { + expect(discoveryBackoffMinutes(0)).toBe(0); + }); + + it("starts at half an hour and doubles", () => { + expect(discoveryBackoffMinutes(1)).toBe(30); + expect(discoveryBackoffMinutes(2)).toBe(60); + expect(discoveryBackoffMinutes(3)).toBe(120); + expect(discoveryBackoffMinutes(4)).toBe(240); + }); + + it("stops doubling at a day, so a tapped-out campaign still retries daily", () => { + expect(discoveryBackoffMinutes(10)).toBe(24 * 60); + expect(discoveryBackoffMinutes(100)).toBe(24 * 60); + }); + + // The whole point of the change. At 5 queries a tick, every 15 minutes, + // three campaigns cost ~43k searches a month against a 25k plan; capped at + // one pass a day they cost ~450. + it("cuts a dry campaign from 96 passes a day to 1", () => { + const passesPerDay = (24 * 60) / discoveryBackoffMinutes(10); + expect(passesPerDay).toBe(1); + const before = (24 * 60) / 15; + expect(before).toBe(96); + }); +}); + +describe("per-account SERP budgets", () => { + // The backstop that could not stop anything: a single pro account was + // authorised for 200k calls a month out of a shared bucket of 25k. + it("keeps every plan under the vendor plan it spends from", () => { + for (const [plan, budget] of Object.entries(SERP_CALLS_PER_MONTH)) { + expect(budget, `${plan} budget exceeds the ValueSERP plan`).toBeLessThanOrEqual( + VALUESERP_MONTHLY_PLAN, + ); + } + }); +}); + +describe("ValueSERP out-of-credit cooldown", () => { + const realFetch = globalThis.fetch; + + beforeEach(() => { + vi.resetModules(); + process.env.VALUESERP_API_KEY = "test-key"; + }); + + afterEach(() => { + globalThis.fetch = realFetch; + vi.restoreAllMocks(); + }); + + it("asks once, then answers from the cooldown instead of re-hitting a dead plan", async () => { + const fetchMock = vi.fn(async () => new Response("", { status: 402 })); + globalThis.fetch = fetchMock as unknown as typeof fetch; + + const { searchSerp, resetSerpCreditCooldown } = await import("@/lib/alerts/valueserp"); + resetSerpCreditCooldown(); + + const first = await searchSerp({ query: "web development agencies", recency: "any" }); + expect(first.ok).toBe(false); + expect(first.error).toContain("402"); + expect(fetchMock).toHaveBeenCalledTimes(1); + + // A campaign tick fires ~21 searches; without the cooldown all of them + // would pay a round-trip to learn the same thing. + for (let i = 0; i < 20; i++) { + const again = await searchSerp({ query: `q${i}`, recency: "any" }); + expect(again.ok).toBe(false); + } + expect(fetchMock).toHaveBeenCalledTimes(1); + }); + + it("never bills a call it did not make", async () => { + globalThis.fetch = vi.fn(async () => new Response("", { status: 402 })) as unknown as typeof fetch; + const { searchSerp, resetSerpCreditCooldown } = await import("@/lib/alerts/valueserp"); + resetSerpCreditCooldown(); + + expect((await searchSerp({ query: "a", recency: "any" })).calls).toBe(0); + expect((await searchSerp({ query: "b", recency: "any" })).calls).toBe(0); + }); +}); diff --git a/tests/lead-billing.test.ts b/tests/lead-billing.test.ts index 90b49257..f0decdc0 100644 --- a/tests/lead-billing.test.ts +++ b/tests/lead-billing.test.ts @@ -130,7 +130,14 @@ describe("the meter is actually connected", () => { it("gates discovery, the most expensive stage, on being able to pay", async () => { const src = await read("lib/outreach/runner.ts"); const discovery = src.slice(src.indexOf("---- 4. Top the funnel up")); - expect(discovery.slice(0, 400)).toMatch(/canSpend\(\)/); + // Read the gate's own condition rather than a fixed byte window. The block + // picked up a discovery back-off check ahead of the gate, which pushed + // canSpend() past the old 400-character slice without weakening anything + // this test exists to protect. + const gateStart = discovery.indexOf("if (liveCount"); + expect(gateStart).toBeGreaterThan(-1); + const gate = discovery.slice(gateStart, discovery.indexOf("{", gateStart)); + expect(gate).toMatch(/canSpend\(\)/); }); it("charges the manual finder", async () => {