diff --git a/.env.example b/.env.example index e124c00a..1cf198b1 100644 --- a/.env.example +++ b/.env.example @@ -65,7 +65,10 @@ CRON_SECRET=change-me-to-a-long-random-string # openai = prefer OPENAI_API_KEY, fall back to Anthropic if needed # anthropic = prefer ANTHROPIC_API_KEY, fall back to OpenAI if needed BACKEND_AI_PROVIDER=openai -BACKEND_AI_OPENAI_MODEL=gpt-5.5 +# Autoblog article models. Defaults are the frontier tier of each provider; +# override to trade quality for price (e.g. gpt-5.6-terra, claude-sonnet-5). +BACKEND_AI_OPENAI_MODEL=gpt-5.6-sol +BACKEND_AI_ANTHROPIC_MODEL=claude-opus-5 ANTHROPIC_API_KEY= OPENAI_API_KEY= GEMINI_API_KEY= diff --git a/lib/ai/spend.ts b/lib/ai/spend.ts index 09fea93d..3882fa5a 100644 --- a/lib/ai/spend.ts +++ b/lib/ai/spend.ts @@ -24,9 +24,16 @@ const RATES: Record = { "claude-sonnet-4-6": { input: 3_000_000, output: 15_000_000 }, "claude-opus-5": { input: 5_000_000, output: 25_000_000 }, "claude-opus-4-8": { input: 5_000_000, output: 25_000_000 }, - // OpenAI + // OpenAI. `gpt-5` is a prefix of every 5.x id, so each newer model needs + // its own row or rateFor() prices it at the 2025 gpt-5 rate. "gpt-5-mini": { input: 250_000, output: 2_000_000 }, "gpt-5": { input: 1_250_000, output: 10_000_000 }, + "gpt-5.5": { input: 5_000_000, output: 30_000_000 }, + // gpt-5.6 Sol is $4/$20 under the promotional pricing published through + // 2026-11-21 (list was $5/$30). Revisit when the promotion ends. + "gpt-5.6-sol": { input: 4_000_000, output: 20_000_000 }, + "gpt-5.6-terra": { input: 2_500_000, output: 15_000_000 }, + "gpt-5.6-luna": { input: 1_000_000, output: 6_000_000 }, }; /** diff --git a/lib/env.ts b/lib/env.ts index a587a426..14f2ee2b 100644 --- a/lib/env.ts +++ b/lib/env.ts @@ -69,8 +69,15 @@ export const env = { // so the only env var per provider is the API key. backendAiProvider: process.env.BACKEND_AI_PROVIDER ?? process.env.AI_TEXT_PROVIDER ?? "openai", + // Autoblog article text. Long-form, structured-output generation on the + // frontier tier of each provider; the model is env-overridable so a price + // cut or a new release is a variable change, not a deploy. Every dated + // Haiku/Sonnet callsite (keyword research, blog detection, pitches) keeps + // its own pin — those are short extraction prompts, not the article. backendAiOpenaiModel: - process.env.BACKEND_AI_OPENAI_MODEL ?? "gpt-5.5", + process.env.BACKEND_AI_OPENAI_MODEL ?? "gpt-5.6-sol", + backendAiAnthropicModel: + process.env.BACKEND_AI_ANTHROPIC_MODEL ?? "claude-opus-5", anthropicApiKey: process.env.ANTHROPIC_API_KEY ?? "", // Daily AI spend that triggers a warning. It warns only — nothing throttles // or pauses on it, because an alarm that turns the product off is worse diff --git a/lib/lx/articleGen.ts b/lib/lx/articleGen.ts index 961470c9..d95bb334 100644 --- a/lib/lx/articleGen.ts +++ b/lib/lx/articleGen.ts @@ -27,6 +27,7 @@ import { type ExchangeCandidate, } from "./exchangeMatcher"; import { generateStructuredOutput } from "./backendAi"; +import { env } from "../env"; import { MAX_PRIOR_BODIES, runQualityGate, @@ -39,7 +40,7 @@ import { } from "../autopilot/entitlements"; const EMBED_MODEL = "text-embedding-3-small"; -const CLAUDE_MODEL = "claude-opus-4-8"; +const CLAUDE_MODEL = env.backendAiAnthropicModel; const IMAGE_MODEL = "gpt-image-2"; const IMAGE_SIZE = "1536x1024"; // gpt-image-2 quality tier: low / medium / high / auto. We pay the premium @@ -1296,8 +1297,10 @@ export async function generateArticle( openai, anthropicModel: CLAUDE_MODEL, // 3,200–4,500 words ≈ ~18k–25k output tokens. JSON escape overhead - // can push that another 30%. 48k gives meaningful headroom. - maxTokens: 48000, + // can push that another 30%, and on Opus 5 / gpt-5.6 the model's + // reasoning tokens come out of the same ceiling. 64k keeps headroom + // for both; the call streams, so the size costs no timeout risk. + maxTokens: 64000, anthropicCacheSystemPrompt: true, }); candidate = normalizeArticleOutput(generated.output); diff --git a/lib/lx/backendAi.ts b/lib/lx/backendAi.ts index 4823bb70..09d48442 100644 --- a/lib/lx/backendAi.ts +++ b/lib/lx/backendAi.ts @@ -63,6 +63,25 @@ export function backendAiTextProviderLabel(): string { return pref === "auto" ? "Anthropic or OpenAI" : pref === "openai" ? "OpenAI" : "Anthropic"; } +/** + * Thinking config for an Anthropic model. + * + * Claude 4.6 and later (Opus 5, Opus 4.8, Sonnet 5, Fable 5.1) run adaptive + * thinking: the model decides per request how much to reason, and `effort` + * bounds it. Sending `disabled` there is the wrong default — on Opus 5 it + * makes the model occasionally write its reasoning into the visible answer, + * and Fable 5.1 rejects it with a 400. Haiku 4.5 predates adaptive thinking + * (it only knows the budget_tokens form), so it stays off there; the Haiku + * callsites are short extraction prompts that pass `anthropicEffort: false`. + */ +export function anthropicThinkingFor( + model: string, +): { type: "adaptive" } | { type: "disabled" } { + return /haiku-4-5|sonnet-4-5|opus-4-5|opus-4-1|claude-3/i.test(model) + ? { type: "disabled" } + : { type: "adaptive" }; +} + export async function generateStructuredOutput( args: StructuredOutputArgs, ): Promise<{ provider: BackendAiProvider; output: T }> { @@ -101,7 +120,7 @@ async function generateWithAnthropic( const stream = args.anthropic.messages.stream({ model: args.anthropicModel, max_tokens: args.maxTokens, - thinking: { type: "disabled" }, + thinking: anthropicThinkingFor(args.anthropicModel), output_config: { ...(args.anthropicEffort === false ? {} diff --git a/lib/lx/guestPostGen.ts b/lib/lx/guestPostGen.ts index aff83203..8d125e71 100644 --- a/lib/lx/guestPostGen.ts +++ b/lib/lx/guestPostGen.ts @@ -42,9 +42,10 @@ import { validateInternalLinks, } from "./articleGen"; import { generateStructuredOutput } from "./backendAi"; +import { env } from "../env"; import { SCAN_CREDITS } from "@/lib/credits"; -const CLAUDE_MODEL = "claude-opus-4-8"; +const CLAUDE_MODEL = env.backendAiAnthropicModel; type SiteCtx = { id: string; @@ -183,7 +184,8 @@ export async function generateGuestPost( anthropic, openai, anthropicModel: CLAUDE_MODEL, - maxTokens: 48000, + // See articleGen: reasoning tokens share this ceiling on Opus 5. + maxTokens: 64000, anthropicCacheSystemPrompt: true, }); article = normalizeArticleOutput(generated.output); diff --git a/tests/ai-spend.test.ts b/tests/ai-spend.test.ts index ffdcb9b9..0fbb27c8 100644 --- a/tests/ai-spend.test.ts +++ b/tests/ai-spend.test.ts @@ -13,6 +13,15 @@ describe("rateFor", () => { expect(rateFor("claude-opus-5")?.output).toBe(25_000_000); }); + it("prices each gpt-5.x release on its own row, not as gpt-5", () => { + // Every 5.x id starts with "gpt-5"; without an exact row the prefix + // match would bill Sol at the 2025 gpt-5 rate. + expect(rateFor("gpt-5.6-sol")?.output).toBe(20_000_000); + expect(rateFor("gpt-5.6-sol-2026-07-09")?.output).toBe(20_000_000); + expect(rateFor("gpt-5.5")?.output).toBe(30_000_000); + expect(rateFor("gpt-5")?.output).toBe(10_000_000); + }); + it("returns null for a model it does not know", () => { expect(rateFor("some-model-we-never-added")).toBeNull(); }); diff --git a/tests/backend-ai-thinking.test.ts b/tests/backend-ai-thinking.test.ts new file mode 100644 index 00000000..f111e60b --- /dev/null +++ b/tests/backend-ai-thinking.test.ts @@ -0,0 +1,17 @@ +import { describe, it, expect } from "vitest"; +import { anthropicThinkingFor } from "@/lib/lx/backendAi"; + +describe("anthropicThinkingFor", () => { + it("runs adaptive thinking on the current frontier models", () => { + // Fable 5.1 rejects `disabled` with a 400; Opus 5 accepts it but leaks + // reasoning into the visible answer. Adaptive is the documented default. + for (const m of ["claude-opus-5", "claude-opus-4-8", "claude-sonnet-5", "claude-fable-5-1"]) { + expect(anthropicThinkingFor(m)).toEqual({ type: "adaptive" }); + } + }); + + it("keeps thinking off on Haiku 4.5, which predates adaptive", () => { + expect(anthropicThinkingFor("claude-haiku-4-5-20251001")).toEqual({ type: "disabled" }); + expect(anthropicThinkingFor("claude-haiku-4-5")).toEqual({ type: "disabled" }); + }); +});