From f8fca64bc732604a7802a72c96c9cfb3847ccf1d Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Sat, 25 Jul 2026 09:13:36 +0000 Subject: [PATCH] fix(engines): repair DeepSeek + Perplexity scans, clarify quota failures MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DeepSeek and Perplexity failed on every run. Two distinct causes: - DeepSeek retired the `deepseek-chat` alias; GET /models now serves only deepseek-v4-flash and deepseek-v4-pro, and the old name 400s. Default to flash (matches this engine's "quick, lightweight" billing), overridable via DEEPSEEK_MODEL. The 8K output cap was a deepseek-chat limit — V4 accepts 64K and is a reasoning model whose reasoning_content draws from the same budget, so 8K would truncate the JSON report. - Perplexity's key and model are fine; the shared 90s client timeout was too tight. Sonar does its web retrieval before emitting response headers, and a real audit measures ~221s, so every run died as "Request timed out." Give it 5min x 2 attempts and widen its stuck window to match. Timeouts are now per-engine rather than hardcoded in oa-compat. The SDK clears its timer as soon as fetch() resolves, so on a streaming call it bounds time-to-first-byte only — a provider that opens the stream then stalls was unbounded. Added an idle watchdog that aborts on a gap between chunks, which is what makes raising Perplexity's header timeout safe. OpenAI GPT-5 Mini and Sakana Fugu are not code bugs — both accounts are out of prepaid credit (verified: OpenAI insufficient_quota; Sakana "Prepaid credit balance is exhausted"). Both surfaced as generic rate limits, pointing at a per-minute cap that was never the problem, and Fugu's message hardcoded "after 5 attempts" while maxRetries was 2. They now distinguish out-of-credit from throttling and quote the provider. Verified live against crawlproof.com: DeepSeek 84/100 in 57s, Perplexity 93/100 in 3m41s. Co-Authored-By: Claude Opus 5 (1M context) --- .env.example | 3 + app/llms.txt/route.ts | 2 +- lib/audit/deepseek-engine.ts | 11 ++-- lib/audit/oa-compat-engine.ts | 94 +++++++++++++++++++++++++-- lib/audit/openai-engine.ts | 47 +++++++++++--- lib/audit/perplexity-engine.ts | 12 ++++ lib/audit/timeouts.ts | 6 ++ lib/credits.ts | 2 +- lib/env.ts | 6 ++ tests/contract/audit-timeouts.test.ts | 12 ++++ 10 files changed, 175 insertions(+), 20 deletions(-) diff --git a/.env.example b/.env.example index 09347d7e..031cddd6 100644 --- a/.env.example +++ b/.env.example @@ -63,6 +63,9 @@ GEMINI_API_KEY= DASHSCOPE_API_KEY= MOONSHOT_API_KEY= DEEPSEEK_API_KEY= +# Optional — defaults to deepseek-v4-flash. The only other live model is +# deepseek-v4-pro; the old deepseek-chat alias was retired and now 400s. +DEEPSEEK_MODEL= ZAI_API_KEY= # Optional — defaults to the GLM Coding Plan endpoint. Set to # https://api.z.ai/api/paas/v4 to use pay-as-you-go API billing instead. diff --git a/app/llms.txt/route.ts b/app/llms.txt/route.ts index 5ecb22fe..479e2484 100644 --- a/app/llms.txt/route.ts +++ b/app/llms.txt/route.ts @@ -7,7 +7,7 @@ const body = `# CrawlProof ## Product - Free single-URL AEO audit, 10 per day per IP, no signup required. - Signed-in users get 20 free credits (1 AI-model scan); each AI-model scan costs 20 credits (~$1, volume discounts down to $0.50/scan at the 100-scan pack). No subscription — credits never expire. -- Saved projects, scheduled re-audits, multi-engine LLM scans (Claude Sonnet 4.6, OpenAI GPT-5 Mini, Gemini 2.5 Pro, Perplexity Sonar Pro, Qwen Plus, Kimi v2.6, DeepSeek V3, Z.AI GLM-5.2), consolidated PDF reports, and diff view. +- Saved projects, scheduled re-audits, multi-engine LLM scans (Claude Sonnet 4.6, OpenAI GPT-5 Mini, Gemini 2.5 Pro, Perplexity Sonar Pro, Qwen Plus, Kimi v2.6, DeepSeek V4, Z.AI GLM-5.2), consolidated PDF reports, and diff view. - Identifies as CrawlProofBot/1.0 (+https://crawlproof.com/bot). ## Key pages diff --git a/lib/audit/deepseek-engine.ts b/lib/audit/deepseek-engine.ts index 4e128edf..3b25fb5f 100644 --- a/lib/audit/deepseek-engine.ts +++ b/lib/audit/deepseek-engine.ts @@ -6,10 +6,13 @@ export async function deepseekAudit(targetUrl: string): Promise; + /** + * How long to wait for the provider's *response headers* before aborting + * (default 90s). The SDK clears its timer as soon as `fetch()` resolves, + * so on a streaming call this bounds time-to-first-byte only — mid-stream + * stalls are caught by `idleTimeoutMs` below, not this. + * + * Search-grounded providers do all their retrieval before emitting + * headers, so they need a much larger budget than a plain chat model. + */ + timeoutMs?: number; + /** SDK retry count (default 2 → up to 3 attempts). */ + maxRetries?: number; + /** + * Abort if the stream goes this long between chunks (default 90s). This + * is the guard the SDK doesn't give us: without it a provider that opens + * the stream and then stalls hangs until the worker's stuck-watchdog + * fires, burning the whole audit window on a dead connection. + */ + idleTimeoutMs?: number; }; +/** + * Wrap a stream so a gap between chunks longer than `idleMs` aborts the + * request instead of hanging. Aborting via the SDK's controller releases + * the socket; without it the dead connection lingers for the full run. + */ +async function* withIdleTimeout( + stream: AsyncIterable & { controller?: AbortController }, + idleMs: number, + label: string, +): AsyncGenerator { + const iterator = stream[Symbol.asyncIterator](); + for (;;) { + let timer: ReturnType | undefined; + const idle = new Promise((_, reject) => { + timer = setTimeout(() => { + stream.controller?.abort(); + reject( + new Error( + `${label} opened the stream but sent no data for ${Math.round(idleMs / 1000)}s — treating the connection as stalled.`, + ), + ); + }, idleMs); + }); + try { + const next = await Promise.race([iterator.next(), idle]); + if (next.done) return; + yield next.value; + } finally { + if (timer) clearTimeout(timer); + } + } +} + export async function oaCompatAudit( targetUrl: string, cfg: OACompatConfig, @@ -138,11 +190,16 @@ export async function oaCompatAudit( // homepage scan caused a 10+ min stall against Kimi. Tightened to // 90s per attempt × 3 attempts (~4.5 min worst case) so the user sees // a real failure they can retry instead of an indefinite spinner. + // Engines whose provider searches the web before answering override + // timeoutMs; the idle watchdog below is what keeps that safe. + const timeoutMs = cfg.timeoutMs ?? 90 * 1000; + const maxRetries = cfg.maxRetries ?? 2; + const idleTimeoutMs = cfg.idleTimeoutMs ?? 90 * 1000; const client = new OpenAI({ apiKey: cfg.apiKey, baseURL: cfg.baseURL, - timeout: 90 * 1000, - maxRetries: 2, + timeout: timeoutMs, + maxRetries, }); console.log( @@ -177,8 +234,37 @@ export async function oaCompatAudit( }); } catch (err) { if (err instanceof OpenAI.APIError && err.status === 429) { + // The message used to hardcode "5 attempts" while maxRetries was 2, + // which sent people hunting for a per-minute limit they weren't + // actually hitting. Report the real count, and split "you're out of + // money" from "you're going too fast" — they need different fixes. + const attempts = maxRetries + 1; + const e = err as { code?: string | null; type?: string | null }; + // Real strings seen in prod: OpenAI `insufficient_quota` / "exceeded + // your current quota"; Sakana `usage_limit_reached` / "Prepaid credit + // balance is exhausted"; Z.AI "insufficient balance". + const detail = [e.code, e.type, err.message] + .filter(Boolean) + .join(" ") + .toLowerCase(); + const outOfCredit = [ + "insufficient_quota", + "insufficient balance", + "exceeded your current quota", + "usage_limit_reached", + "balance is exhausted", + "add credits", + "billing", + ].some((needle) => detail.includes(needle)); + throw new Error( + outOfCredit + ? `${cfg.providerLabel} rejected the request (HTTP 429): the account is out of API credit, not rate-limited. Top up billing for this provider — retrying won't help. Provider said: ${err.message}` + : `${cfg.providerLabel} rate-limited the request (HTTP 429) after ${attempts} attempt${attempts === 1 ? "" : "s"}. Check the API key's tier / per-minute limit, then retry.`, + ); + } + if (err instanceof OpenAI.APIConnectionTimeoutError) { throw new Error( - `${cfg.providerLabel} rate-limited the request (HTTP 429) after ${4 + 1} attempts. Provider quota is exhausted — check the API key's tier / per-minute limit.`, + `${cfg.providerLabel} sent no response headers within ${Math.round(timeoutMs / 1000)}s (${maxRetries + 1} attempts). The provider is slow or down — retry, or raise this engine's timeoutMs.`, ); } throw err; @@ -186,7 +272,7 @@ export async function oaCompatAudit( let raw = ""; let finishReason: string | null = null; - for await (const chunk of stream) { + for await (const chunk of withIdleTimeout(stream, idleTimeoutMs, cfg.providerLabel)) { const choice = chunk.choices[0]; const delta = choice?.delta?.content; if (delta) raw += delta; diff --git a/lib/audit/openai-engine.ts b/lib/audit/openai-engine.ts index 3e70119e..ebf68c35 100644 --- a/lib/audit/openai-engine.ts +++ b/lib/audit/openai-engine.ts @@ -64,6 +64,27 @@ For each finding: For score: critical fails dominate. Missing schema, blocking GPTBot, JS-only content → below 50. Clean instrumentation → 80+.`; +/** + * A 429 from this engine is almost always `insufficient_quota` — the shared + * OPENAI_API_KEY ran out of billing credit — not a per-minute limit. The raw + * SDK text ("You exceeded your current quota…") reads like throttling and + * sends people to the Retry button, which can never succeed. + */ +function explainOpenAIError(err: unknown): unknown { + if (err instanceof OpenAI.APIError && err.status === 429) { + const code = String((err as { code?: string }).code ?? ""); + if (code === "insufficient_quota" || /quota|billing/i.test(err.message)) { + return new Error( + "OpenAI rejected the request (HTTP 429): the OPENAI_API_KEY account is out of API credit, not rate-limited. Top up billing at platform.openai.com — retrying won't help.", + ); + } + return new Error( + `OpenAI rate-limited the request (HTTP 429). Wait and retry. Provider said: ${err.message}`, + ); + } + return err; +} + export async function openaiAudit(targetUrl: string): Promise { if (!env.openaiApiKey) { throw new Error("OPENAI_API_KEY is not set — cannot run OpenAI audit."); @@ -86,16 +107,22 @@ export async function openaiAudit(targetUrl: string): Promise // Explicit reasoning.effort=medium so it doesn't default to a // minimal reasoning pass and emit a zero-finding shell (same trap // Claude hit at effort=low). - const response = await client.responses.parse({ - model: "gpt-5-mini", - instructions: SYSTEM_PROMPT, - input: userPrompt, - tools: [{ type: "web_search_preview" }], - reasoning: { effort: "medium" }, - text: { - format: zodTextFormat(ResultSchema, "aeo_audit"), - }, - }); + // .catch() rather than try/catch so the parsed-output generic still + // flows through from zodTextFormat — annotating `let response` erases it. + const response = await client.responses + .parse({ + model: "gpt-5-mini", + instructions: SYSTEM_PROMPT, + input: userPrompt, + tools: [{ type: "web_search_preview" }], + reasoning: { effort: "medium" }, + text: { + format: zodTextFormat(ResultSchema, "aeo_audit"), + }, + }) + .catch((err: unknown) => { + throw explainOpenAIError(err); + }); const parsed = response.output_parsed; if (!parsed) { diff --git a/lib/audit/perplexity-engine.ts b/lib/audit/perplexity-engine.ts index c8229498..237628cf 100644 --- a/lib/audit/perplexity-engine.ts +++ b/lib/audit/perplexity-engine.ts @@ -13,5 +13,17 @@ export async function perplexityAudit(targetUrl: string): Promise> = { claude: CLAUDE_AUDIT_STUCK_AFTER_MS, + perplexity: PERPLEXITY_AUDIT_STUCK_AFTER_MS, }; export function auditStuckAfterMs(engine: string | null | undefined): number { diff --git a/lib/credits.ts b/lib/credits.ts index fb4ff59f..a7f93d46 100644 --- a/lib/credits.ts +++ b/lib/credits.ts @@ -155,7 +155,7 @@ export const ENGINES: Record = { "Google's flagship with live Search grounding. Frames your site the way Google AI Overviews would.", }, deepseek: { - label: "DeepSeek V3", + label: "DeepSeek V4", cost: SCAN_CREDITS, available: true, blurb: diff --git a/lib/env.ts b/lib/env.ts index 96f6a152..6b2211ad 100644 --- a/lib/env.ts +++ b/lib/env.ts @@ -65,6 +65,12 @@ export const env = { dashscopeApiKey: process.env.DASHSCOPE_API_KEY ?? "", // Qwen moonshotApiKey: process.env.MOONSHOT_API_KEY ?? "", // Kimi deepseekApiKey: process.env.DEEPSEEK_API_KEY ?? "", + // DeepSeek retired `deepseek-chat` — GET /models now serves only + // `deepseek-v4-flash` and `deepseek-v4-pro`, and calling the old alias + // 400s ("The supported API model names are deepseek-v4-pro or + // deepseek-v4-flash"). Flash matches this engine's "quick, lightweight + // second opinion" billing; set DEEPSEEK_MODEL=deepseek-v4-pro to upgrade. + deepseekModel: process.env.DEEPSEEK_MODEL ?? "deepseek-v4-flash", zaiApiKey: process.env.ZAI_API_KEY ?? "", // Z.AI / Zhipu GLM // GLM Coding Plan endpoint (monthly subscription). The standard // pay-as-you-go endpoint (.../api/paas/v4) returns 429 "insufficient diff --git a/tests/contract/audit-timeouts.test.ts b/tests/contract/audit-timeouts.test.ts index 29bf4737..b9044416 100644 --- a/tests/contract/audit-timeouts.test.ts +++ b/tests/contract/audit-timeouts.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "vitest"; import { CLAUDE_AUDIT_STUCK_AFTER_MS, DEFAULT_AUDIT_STUCK_AFTER_MS, + PERPLEXITY_AUDIT_STUCK_AFTER_MS, auditStuckAfterMinutes, auditStuckAfterMs, } from "@/lib/audit/timeouts"; @@ -18,6 +19,17 @@ describe("audit stuck timeout policy", () => { expect(auditStuckAfterMinutes("claude")).toBe(15); }); + // Perplexity runs a 5-minute header timeout x 2 attempts (Sonar searches + // the web before it answers). The 7-minute default would reap a healthy + // run mid-way through the second attempt. + it("outlasts Perplexity's worst-case retry budget", () => { + expect(auditStuckAfterMs("perplexity")).toBe(PERPLEXITY_AUDIT_STUCK_AFTER_MS); + expect(auditStuckAfterMinutes("perplexity")).toBe(15); + const perplexityWorstCaseMs = 2 * 5 * 60 * 1000; + expect(DEFAULT_AUDIT_STUCK_AFTER_MS).toBeLessThan(perplexityWorstCaseMs); + expect(auditStuckAfterMs("perplexity")).toBeGreaterThan(perplexityWorstCaseMs); + }); + it("falls back to the default for unknown persisted engine names", () => { expect(auditStuckAfterMs("legacy-engine")).toBe(DEFAULT_AUDIT_STUCK_AFTER_MS); });