Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -65,7 +65,10 @@ CRON_SECRET=change-me-to-a-long-random-string
# openai = prefer OPENAI_API_KEY, fall back to Anthropic if needed
# anthropic = prefer ANTHROPIC_API_KEY, fall back to OpenAI if needed
BACKEND_AI_PROVIDER=openai
BACKEND_AI_OPENAI_MODEL=gpt-5.5
# Autoblog article models. Defaults are the frontier tier of each provider;
# override to trade quality for price (e.g. gpt-5.6-terra, claude-sonnet-5).
BACKEND_AI_OPENAI_MODEL=gpt-5.6-sol
BACKEND_AI_ANTHROPIC_MODEL=claude-opus-5
ANTHROPIC_API_KEY=
OPENAI_API_KEY=
GEMINI_API_KEY=
Expand Down
9 changes: 8 additions & 1 deletion lib/ai/spend.ts
Original file line number Diff line number Diff line change
Expand Up @@ -24,9 +24,16 @@ const RATES: Record<string, Rate> = {
"claude-sonnet-4-6": { input: 3_000_000, output: 15_000_000 },
"claude-opus-5": { input: 5_000_000, output: 25_000_000 },
"claude-opus-4-8": { input: 5_000_000, output: 25_000_000 },
// OpenAI
// OpenAI. `gpt-5` is a prefix of every 5.x id, so each newer model needs
// its own row or rateFor() prices it at the 2025 gpt-5 rate.
"gpt-5-mini": { input: 250_000, output: 2_000_000 },
"gpt-5": { input: 1_250_000, output: 10_000_000 },
"gpt-5.5": { input: 5_000_000, output: 30_000_000 },
// gpt-5.6 Sol is $4/$20 under the promotional pricing published through
// 2026-11-21 (list was $5/$30). Revisit when the promotion ends.
"gpt-5.6-sol": { input: 4_000_000, output: 20_000_000 },
"gpt-5.6-terra": { input: 2_500_000, output: 15_000_000 },
"gpt-5.6-luna": { input: 1_000_000, output: 6_000_000 },
};

/**
Expand Down
9 changes: 8 additions & 1 deletion lib/env.ts
Original file line number Diff line number Diff line change
Expand Up @@ -69,8 +69,15 @@ export const env = {
// so the only env var per provider is the API key.
backendAiProvider:
process.env.BACKEND_AI_PROVIDER ?? process.env.AI_TEXT_PROVIDER ?? "openai",
// Autoblog article text. Long-form, structured-output generation on the
// frontier tier of each provider; the model is env-overridable so a price
// cut or a new release is a variable change, not a deploy. Every dated
// Haiku/Sonnet callsite (keyword research, blog detection, pitches) keeps
// its own pin — those are short extraction prompts, not the article.
backendAiOpenaiModel:
process.env.BACKEND_AI_OPENAI_MODEL ?? "gpt-5.5",
process.env.BACKEND_AI_OPENAI_MODEL ?? "gpt-5.6-sol",
backendAiAnthropicModel:
process.env.BACKEND_AI_ANTHROPIC_MODEL ?? "claude-opus-5",
anthropicApiKey: process.env.ANTHROPIC_API_KEY ?? "",
// Daily AI spend that triggers a warning. It warns only — nothing throttles
// or pauses on it, because an alarm that turns the product off is worse
Expand Down
9 changes: 6 additions & 3 deletions lib/lx/articleGen.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ import {
type ExchangeCandidate,
} from "./exchangeMatcher";
import { generateStructuredOutput } from "./backendAi";
import { env } from "../env";
import {
MAX_PRIOR_BODIES,
runQualityGate,
Expand All @@ -39,7 +40,7 @@ import {
} from "../autopilot/entitlements";

const EMBED_MODEL = "text-embedding-3-small";
const CLAUDE_MODEL = "claude-opus-4-8";
const CLAUDE_MODEL = env.backendAiAnthropicModel;
const IMAGE_MODEL = "gpt-image-2";
const IMAGE_SIZE = "1536x1024";
// gpt-image-2 quality tier: low / medium / high / auto. We pay the premium
Expand Down Expand Up @@ -1296,8 +1297,10 @@ export async function generateArticle(
openai,
anthropicModel: CLAUDE_MODEL,
// 3,200–4,500 words ≈ ~18k–25k output tokens. JSON escape overhead
// can push that another 30%. 48k gives meaningful headroom.
maxTokens: 48000,
// can push that another 30%, and on Opus 5 / gpt-5.6 the model's
// reasoning tokens come out of the same ceiling. 64k keeps headroom
// for both; the call streams, so the size costs no timeout risk.
maxTokens: 64000,
anthropicCacheSystemPrompt: true,
});
candidate = normalizeArticleOutput(generated.output);
Expand Down
21 changes: 20 additions & 1 deletion lib/lx/backendAi.ts
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,25 @@ export function backendAiTextProviderLabel(): string {
return pref === "auto" ? "Anthropic or OpenAI" : pref === "openai" ? "OpenAI" : "Anthropic";
}

/**
* Thinking config for an Anthropic model.
*
* Claude 4.6 and later (Opus 5, Opus 4.8, Sonnet 5, Fable 5.1) run adaptive
* thinking: the model decides per request how much to reason, and `effort`
* bounds it. Sending `disabled` there is the wrong default — on Opus 5 it
* makes the model occasionally write its reasoning into the visible answer,
* and Fable 5.1 rejects it with a 400. Haiku 4.5 predates adaptive thinking
* (it only knows the budget_tokens form), so it stays off there; the Haiku
* callsites are short extraction prompts that pass `anthropicEffort: false`.
*/
export function anthropicThinkingFor(
model: string,
): { type: "adaptive" } | { type: "disabled" } {
return /haiku-4-5|sonnet-4-5|opus-4-5|opus-4-1|claude-3/i.test(model)
? { type: "disabled" }
: { type: "adaptive" };
}

export async function generateStructuredOutput<T>(
args: StructuredOutputArgs<T>,
): Promise<{ provider: BackendAiProvider; output: T }> {
Expand Down Expand Up @@ -101,7 +120,7 @@ async function generateWithAnthropic<T>(
const stream = args.anthropic.messages.stream({
model: args.anthropicModel,
max_tokens: args.maxTokens,
thinking: { type: "disabled" },
thinking: anthropicThinkingFor(args.anthropicModel),
output_config: {
...(args.anthropicEffort === false
? {}
Expand Down
6 changes: 4 additions & 2 deletions lib/lx/guestPostGen.ts
Original file line number Diff line number Diff line change
Expand Up @@ -42,9 +42,10 @@ import {
validateInternalLinks,
} from "./articleGen";
import { generateStructuredOutput } from "./backendAi";
import { env } from "../env";
import { SCAN_CREDITS } from "@/lib/credits";

const CLAUDE_MODEL = "claude-opus-4-8";
const CLAUDE_MODEL = env.backendAiAnthropicModel;

type SiteCtx = {
id: string;
Expand Down Expand Up @@ -183,7 +184,8 @@ export async function generateGuestPost(
anthropic,
openai,
anthropicModel: CLAUDE_MODEL,
maxTokens: 48000,
// See articleGen: reasoning tokens share this ceiling on Opus 5.
maxTokens: 64000,
anthropicCacheSystemPrompt: true,
});
article = normalizeArticleOutput(generated.output);
Expand Down
9 changes: 9 additions & 0 deletions tests/ai-spend.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,15 @@ describe("rateFor", () => {
expect(rateFor("claude-opus-5")?.output).toBe(25_000_000);
});

it("prices each gpt-5.x release on its own row, not as gpt-5", () => {
// Every 5.x id starts with "gpt-5"; without an exact row the prefix
// match would bill Sol at the 2025 gpt-5 rate.
expect(rateFor("gpt-5.6-sol")?.output).toBe(20_000_000);
expect(rateFor("gpt-5.6-sol-2026-07-09")?.output).toBe(20_000_000);
expect(rateFor("gpt-5.5")?.output).toBe(30_000_000);
expect(rateFor("gpt-5")?.output).toBe(10_000_000);
});

it("returns null for a model it does not know", () => {
expect(rateFor("some-model-we-never-added")).toBeNull();
});
Expand Down
17 changes: 17 additions & 0 deletions tests/backend-ai-thinking.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
import { describe, it, expect } from "vitest";
import { anthropicThinkingFor } from "@/lib/lx/backendAi";

describe("anthropicThinkingFor", () => {
it("runs adaptive thinking on the current frontier models", () => {
// Fable 5.1 rejects `disabled` with a 400; Opus 5 accepts it but leaks
// reasoning into the visible answer. Adaptive is the documented default.
for (const m of ["claude-opus-5", "claude-opus-4-8", "claude-sonnet-5", "claude-fable-5-1"]) {
expect(anthropicThinkingFor(m)).toEqual({ type: "adaptive" });
}
});

it("keeps thinking off on Haiku 4.5, which predates adaptive", () => {
expect(anthropicThinkingFor("claude-haiku-4-5-20251001")).toEqual({ type: "disabled" });
expect(anthropicThinkingFor("claude-haiku-4-5")).toEqual({ type: "disabled" });
});
});
Loading