diff --git a/app/api/cron/ai-spend/route.ts b/app/api/cron/ai-spend/route.ts new file mode 100644 index 00000000..f751b6b4 --- /dev/null +++ b/app/api/cron/ai-spend/route.ts @@ -0,0 +1,88 @@ +import { NextResponse } from "next/server"; +import { serviceClient } from "@/lib/supabase/service"; +import { env } from "@/lib/env"; +import { formatUsd, spendToday, MICROS_PER_DOLLAR } from "@/lib/ai/spend"; +import { sendAiSpendAlertEmail } from "@/lib/email"; + +export const runtime = "nodejs"; + +/** + * Warn when a day's AI spend crosses the threshold. It only warns — nothing + * here throttles, pauses or blocks a request. A budget alarm that silently + * turns the product off is worse than the bill it was meant to prevent. + * + * Run it hourly. The alert de-duplicates on (day, threshold), so a day that + * crosses the line mails once rather than on every run after it. + * + * The balance on the Anthropic account is deliberately not reported: reading + * it needs an Admin key (sk-ant-admin01-...), and a normal API key gets a 401 + * on the cost and usage endpoints. What this can measure exactly is what this + * application spent, which is the number the alert is actually about. + */ +export async function GET(req: Request) { + return POST(req); +} + +export async function POST(req: Request) { + const incoming = + req.headers.get("x-cron-secret") ?? + req.headers.get("authorization")?.replace(/^Bearer\s+/i, ""); + if (incoming !== env.cronSecret) { + return NextResponse.json({ ok: false, error: "unauthorized" }, { status: 401 }); + } + + const thresholdMicros = Math.round(env.aiSpendDailyAlertUsd * MICROS_PER_DOLLAR); + const spend = await spendToday(); + const over = spend.totalMicros >= thresholdMicros; + + if (!over) { + return NextResponse.json({ + ok: true, + day: spend.day, + spend: formatUsd(spend.totalMicros), + threshold: formatUsd(thresholdMicros), + alerted: false, + }); + } + + const sb = serviceClient(); + // Insert first and let the unique index decide. Checking-then-sending would + // mail twice if two runs overlap. + const { error: claimError } = await sb.from("ai_spend_alerts").insert({ + day: spend.day, + threshold_micros: thresholdMicros, + spend_micros: spend.totalMicros, + sent_to: env.aiSpendAlertEmail, + }); + if (claimError) { + // Already alerted for this day and threshold. + return NextResponse.json({ + ok: true, + day: spend.day, + spend: formatUsd(spend.totalMicros), + alerted: false, + note: "already alerted today", + }); + } + + await sendAiSpendAlertEmail({ + to: env.aiSpendAlertEmail, + day: spend.day, + spendLabel: formatUsd(spend.totalMicros), + thresholdLabel: formatUsd(thresholdMicros), + calls: spend.calls, + breakdown: spend.byFeature.map((f) => ({ + feature: f.feature, + spendLabel: formatUsd(f.micros), + calls: f.calls, + })), + }); + + return NextResponse.json({ + ok: true, + day: spend.day, + spend: formatUsd(spend.totalMicros), + threshold: formatUsd(thresholdMicros), + alerted: true, + }); +} diff --git a/lib/ai/spend.ts b/lib/ai/spend.ts new file mode 100644 index 00000000..09fea93d --- /dev/null +++ b/lib/ai/spend.ts @@ -0,0 +1,145 @@ +// What the AI costs, computed at the point of spending it. +// +// The provider will not tell the app: Anthropic's cost and usage reports need +// an Admin key, and a normal API key gets a 401 on them. What every response +// does carry is its token counts, and the per-model rates are published — so +// the cost is calculable here, and here is also the only place that knows +// which feature caused it. +// +// Everything below is in micro-dollars ($1 = 1_000_000). Model rates are per +// million tokens, so one Haiku draft costs a fraction of a cent; in cents, a +// day of real spending rounds to zero. + +import { serviceClient } from "@/lib/supabase/service"; + +const MICROS_PER_DOLLAR = 1_000_000; + +/** Published rates, in micro-dollars per million tokens. */ +type Rate = { input: number; output: number }; + +const RATES: Record = { + // Anthropic + "claude-haiku-4-5": { input: 1_000_000, output: 5_000_000 }, + "claude-sonnet-5": { input: 3_000_000, output: 15_000_000 }, + "claude-sonnet-4-6": { input: 3_000_000, output: 15_000_000 }, + "claude-opus-5": { input: 5_000_000, output: 25_000_000 }, + "claude-opus-4-8": { input: 5_000_000, output: 25_000_000 }, + // OpenAI + "gpt-5-mini": { input: 250_000, output: 2_000_000 }, + "gpt-5": { input: 1_250_000, output: 10_000_000 }, +}; + +/** + * Rate for a model id, tolerating the dated suffixes the APIs return. + * + * A response reports `claude-haiku-4-5-20251001` where the request asked for + * `claude-haiku-4-5`. Matching on the longest known prefix keeps a dated id + * priced instead of silently costing zero. + */ +export function rateFor(model: string): Rate | null { + const id = model.trim().toLowerCase(); + if (RATES[id]) return RATES[id]; + let best: string | null = null; + for (const known of Object.keys(RATES)) { + if (id.startsWith(known) && (!best || known.length > best.length)) best = known; + } + return best ? RATES[best] : null; +} + +export type UsageSample = { + provider: string; + model: string; + feature?: string | null; + inputTokens: number; + outputTokens: number; + cacheReadTokens?: number; + cacheWriteTokens?: number; +}; + +/** Cost of one call, in micro-dollars. Unknown models cost 0 and say so. */ +export function costMicros(sample: UsageSample): { micros: number; rate: Rate | null } { + const rate = rateFor(sample.model); + if (!rate) return { micros: 0, rate: null }; + // Cache reads bill at roughly a tenth of input; cache writes at 1.25x. Both + // are approximations of the published multipliers, and both are far closer + // than ignoring them, which is what counting only fresh input would do. + const inputEquivalent = + sample.inputTokens + + (sample.cacheReadTokens ?? 0) * 0.1 + + (sample.cacheWriteTokens ?? 0) * 1.25; + const micros = + (inputEquivalent * rate.input) / 1_000_000 + (sample.outputTokens * rate.output) / 1_000_000; + return { micros: Math.round(micros), rate }; +} + +/** + * Record one call. + * + * Never throws. This runs after a request the caller already considers + * successful, and failing to write a bookkeeping row is not a reason to fail + * the work that was actually asked for. + */ +export async function recordAiSpend(sample: UsageSample): Promise { + try { + const { micros, rate } = costMicros(sample); + await serviceClient().from("ai_usage").insert({ + provider: sample.provider, + model: sample.model, + feature: sample.feature ?? null, + input_tokens: sample.inputTokens, + output_tokens: sample.outputTokens, + cache_read_tokens: sample.cacheReadTokens ?? 0, + cache_write_tokens: sample.cacheWriteTokens ?? 0, + cost_micros: micros, + rate_input_micros_per_mtok: rate?.input ?? null, + rate_output_micros_per_mtok: rate?.output ?? null, + }); + } catch { + // Bookkeeping is best-effort by design. + } +} + +export type DaySpend = { + day: string; + totalMicros: number; + byFeature: { feature: string; micros: number; calls: number }[]; + calls: number; +}; + +/** Spend since UTC midnight, with the breakdown that makes it actionable. */ +export async function spendToday(): Promise { + const start = new Date(); + start.setUTCHours(0, 0, 0, 0); + + const { data } = await serviceClient() + .from("ai_usage") + .select("feature, cost_micros") + .gte("occurred_at", start.toISOString()); + + const rows = (data as { feature: string | null; cost_micros: number }[] | null) ?? []; + const byFeature = new Map(); + let totalMicros = 0; + for (const r of rows) { + totalMicros += r.cost_micros; + const key = r.feature ?? "(unattributed)"; + const cur = byFeature.get(key) ?? { micros: 0, calls: 0 }; + cur.micros += r.cost_micros; + cur.calls += 1; + byFeature.set(key, cur); + } + + return { + day: start.toISOString().slice(0, 10), + totalMicros, + calls: rows.length, + byFeature: [...byFeature.entries()] + .map(([feature, v]) => ({ feature, ...v })) + .sort((a, b) => b.micros - a.micros), + }; +} + +export function formatUsd(micros: number): string { + return `$${(micros / MICROS_PER_DOLLAR).toFixed(2)}`; +} + +export { MICROS_PER_DOLLAR }; diff --git a/lib/email.ts b/lib/email.ts index 2ee2ce5e..3c540899 100644 --- a/lib/email.ts +++ b/lib/email.ts @@ -996,6 +996,54 @@ export function coldOutreachEmailHtml(input: { }); } +/** + * Daily AI spend warning. + * + * Leads with the number and the breakdown, because "you spent $18" prompts + * the question "on what" and the answer should not require opening a + * dashboard. States plainly that nothing was stopped, so the mail is not read + * as an outage. + */ +export async function sendAiSpendAlertEmail(input: { + to: string; + day: string; + spendLabel: string; + thresholdLabel: string; + calls: number; + breakdown: { feature: string; spendLabel: string; calls: number }[]; +}): Promise<{ sent: boolean; error?: string }> { + const c = client(); + if (!c) return { sent: false, error: "RESEND_API_KEY not set" }; + + const rows = input.breakdown + .map( + (b) => + `${b.feature}` + + `${b.spendLabel}` + + `${b.calls}`, + ) + .join(""); + + const res = await c.send({ + from: env.resendFrom, + to: input.to, + subject: `AI spend ${input.spendLabel} on ${input.day} (over ${input.thresholdLabel})`, + html: [ + `

${input.spendLabel} of AI usage so far on ${input.day}, across ${input.calls} calls.

`, + `

That is over your ${input.thresholdLabel}/day warning line. Nothing has been stopped — this is a heads-up, not an outage.

`, + rows + ? `` + + `` + + `` + + `${rows}
FeatureSpendCalls
` + : "", + `

Computed from token counts on each call at published per-model rates. It measures what this app spent, not the balance on the Anthropic account — reading that needs an Admin API key.

`, + ].join(""), + }); + if (!res.sent) return { sent: false, error: res.error }; + return { sent: true }; +} + export async function sendColdOutreachEmail(input: { to: string; subject: string; diff --git a/lib/env.ts b/lib/env.ts index 3c18d771..89fcf527 100644 --- a/lib/env.ts +++ b/lib/env.ts @@ -60,6 +60,11 @@ export const env = { backendAiOpenaiModel: process.env.BACKEND_AI_OPENAI_MODEL ?? "gpt-5.5", anthropicApiKey: process.env.ANTHROPIC_API_KEY ?? "", + // Daily AI spend that triggers a warning. It warns only — nothing throttles + // or pauses on it, because an alarm that turns the product off is worse + // than the bill it was meant to prevent. + aiSpendDailyAlertUsd: Number(process.env.AI_SPEND_DAILY_ALERT_USD ?? "15"), + aiSpendAlertEmail: process.env.AI_SPEND_ALERT_EMAIL ?? "anthony@profullstack.com", openaiApiKey: process.env.OPENAI_API_KEY ?? "", geminiApiKey: process.env.GEMINI_API_KEY ?? "", dashscopeApiKey: process.env.DASHSCOPE_API_KEY ?? "", // Qwen diff --git a/lib/lx/backendAi.ts b/lib/lx/backendAi.ts index d0da8c43..4823bb70 100644 --- a/lib/lx/backendAi.ts +++ b/lib/lx/backendAi.ts @@ -2,6 +2,7 @@ import Anthropic from "@anthropic-ai/sdk"; import { zodOutputFormat } from "@anthropic-ai/sdk/helpers/zod"; import OpenAI from "openai"; import { z } from "zod/v4"; +import { recordAiSpend } from "@/lib/ai/spend"; import { env } from "../env"; export type BackendAiProvider = "anthropic" | "openai"; @@ -120,6 +121,18 @@ async function generateWithAnthropic( messages: [{ role: "user", content: args.user }], }); const response = await stream.finalMessage(); + // Recorded before the parse check: the tokens were spent whether or not the + // output turns out to be usable, and a failed parse is exactly the kind of + // waste worth seeing in the total. + void recordAiSpend({ + provider: "anthropic", + model: response.model ?? args.anthropicModel, + feature: args.name, + inputTokens: response.usage?.input_tokens ?? 0, + outputTokens: response.usage?.output_tokens ?? 0, + cacheReadTokens: response.usage?.cache_read_input_tokens ?? 0, + cacheWriteTokens: response.usage?.cache_creation_input_tokens ?? 0, + }); const parsed = response.parsed_output as T | null; if (!parsed) { throw new Error( @@ -159,6 +172,16 @@ async function generateWithOpenAI( text: { format }, store: false, }); + // Same reasoning as the Anthropic path: record before any check that can + // throw, because the tokens are already spent by this point. + void recordAiSpend({ + provider: "openai", + model: response.model ?? args.openaiModel ?? env.backendAiOpenaiModel, + feature: args.name, + inputTokens: response.usage?.input_tokens ?? 0, + outputTokens: response.usage?.output_tokens ?? 0, + cacheReadTokens: response.usage?.input_tokens_details?.cached_tokens ?? 0, + }); if (response.error) { throw new Error(response.error.message); } diff --git a/supabase/migrations/20260728060000_ai_spend_ledger.sql b/supabase/migrations/20260728060000_ai_spend_ledger.sql new file mode 100644 index 00000000..f8c4f1f3 --- /dev/null +++ b/supabase/migrations/20260728060000_ai_spend_ledger.sql @@ -0,0 +1,65 @@ +-- What the AI actually costs, recorded per call. +-- +-- Anthropic's cost and usage reports need an Admin key (sk-ant-admin01-...); +-- a normal API key gets 401 on them, so the running total cannot be read back +-- from the provider by the app that is spending the money. Token counts come +-- back on every response though, and the per-model prices are published, so +-- the spend is computable at the point of the call — and that is also the +-- only place that knows which feature caused it. +-- +-- Cost is stored in micro-dollars as an integer. Model prices are quoted per +-- million tokens, so a single Haiku call rounds to zero cents; summing cents +-- would report a day of real spend as $0. + +create table if not exists public.ai_usage ( + id uuid primary key default gen_random_uuid(), + occurred_at timestamptz not null default now(), + + provider text not null, + model text not null, + -- Which part of the product spent it. Without this the total answers "how + -- much" but never "on what", which is the question that follows. + feature text, + + input_tokens int not null default 0, + output_tokens int not null default 0, + cache_read_tokens int not null default 0, + cache_write_tokens int not null default 0, + + -- Millionths of a dollar. $1.00 = 1000000. + cost_micros bigint not null default 0, + -- The rates used, so a later price change doesn't silently rewrite history. + rate_input_micros_per_mtok bigint, + rate_output_micros_per_mtok bigint +); + +create index if not exists ai_usage_occurred_idx on public.ai_usage(occurred_at desc); +create index if not exists ai_usage_feature_idx on public.ai_usage(feature, occurred_at desc); + +-- Recording spend must never be able to fail a request that already +-- succeeded, so writes go through the service client and nothing reads this +-- from a browser. +alter table public.ai_usage enable row level security; + +-- One alert per threshold per day. The check runs on a schedule, so without +-- this a day that crosses the line would mail on every subsequent run. +create table if not exists public.ai_spend_alerts ( + id uuid primary key default gen_random_uuid(), + -- Local day the alert covers. + day date not null, + threshold_micros bigint not null, + spend_micros bigint not null, + sent_to text not null, + sent_at timestamptz not null default now() +); + +create unique index if not exists ai_spend_alerts_day_threshold_idx + on public.ai_spend_alerts(day, threshold_micros); + +alter table public.ai_spend_alerts enable row level security; + +comment on table public.ai_usage is + 'Per-call AI spend, computed from returned token counts and published per-model rates. Anthropic cost reporting needs an Admin key, so this is the only figure the app itself can see.'; + +comment on column public.ai_usage.cost_micros is + 'Millionths of a dollar. Cents would round a whole day of Haiku calls to zero.'; diff --git a/tests/ai-spend.test.ts b/tests/ai-spend.test.ts new file mode 100644 index 00000000..ffdcb9b9 --- /dev/null +++ b/tests/ai-spend.test.ts @@ -0,0 +1,133 @@ +import { describe, it, expect } from "vitest"; +import { costMicros, formatUsd, rateFor } from "@/lib/ai/spend"; + +describe("rateFor", () => { + it("prices a model the API returns with a date suffix", () => { + // The request asks for claude-haiku-4-5; the response says + // claude-haiku-4-5-20251001. Exact-match lookup would price it at zero. + expect(rateFor("claude-haiku-4-5-20251001")).toEqual(rateFor("claude-haiku-4-5")); + }); + + it("prefers the longest matching prefix", () => { + // claude-opus-5 must not be priced as some shorter claude-opus entry. + expect(rateFor("claude-opus-5")?.output).toBe(25_000_000); + }); + + it("returns null for a model it does not know", () => { + expect(rateFor("some-model-we-never-added")).toBeNull(); + }); +}); + +describe("costMicros", () => { + it("prices a Haiku draft at the published rate", () => { + // 600 in @ $1/MTok = $0.0006; 250 out @ $5/MTok = $0.00125. Total + // $0.00185 → 1850 micro-dollars. + const { micros } = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 600, + outputTokens: 250, + }); + expect(micros).toBe(1850); + }); + + it("keeps sub-cent calls from rounding to nothing", () => { + // The reason the ledger is in micro-dollars: in whole cents this is 0, + // and a day of them would report as $0.00. + const { micros } = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 100, + outputTokens: 20, + }); + expect(micros).toBeGreaterThan(0); + expect(micros).toBeLessThan(10_000); // under one cent + }); + + it("bills cache reads far below fresh input", () => { + const fresh = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 10_000, + outputTokens: 0, + }).micros; + const cached = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 0, + cacheReadTokens: 10_000, + outputTokens: 0, + }).micros; + expect(cached).toBeLessThan(fresh); + expect(cached).toBeCloseTo(fresh * 0.1, -1); + }); + + it("bills cache writes above fresh input", () => { + const fresh = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 10_000, + outputTokens: 0, + }).micros; + const written = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 0, + cacheWriteTokens: 10_000, + outputTokens: 0, + }).micros; + expect(written).toBeGreaterThan(fresh); + }); + + it("costs an unknown model zero and says the rate was unknown", () => { + // Reporting zero silently would understate a bill; the null rate is what + // makes that visible in the stored row. + const { micros, rate } = costMicros({ + provider: "anthropic", + model: "brand-new-model", + inputTokens: 1_000_000, + outputTokens: 1_000_000, + }); + expect(micros).toBe(0); + expect(rate).toBeNull(); + }); + + it("prices output above input, as every model does", () => { + const inputOnly = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 1000, + outputTokens: 0, + }).micros; + const outputOnly = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 0, + outputTokens: 1000, + }).micros; + expect(outputOnly).toBeGreaterThan(inputOnly); + }); +}); + +describe("formatUsd", () => { + it("renders micro-dollars as money", () => { + expect(formatUsd(15_000_000)).toBe("$15.00"); + expect(formatUsd(1850)).toBe("$0.00"); + expect(formatUsd(18_432_100)).toBe("$18.43"); + }); +}); + +describe("the threshold in practice", () => { + it("takes a lot of Haiku drafts to reach $15", () => { + // Sanity on the alert being meaningful rather than hair-trigger: at the + // measured ~$0.00185 per draft, $15/day is thousands of drafts. + const perDraft = costMicros({ + provider: "anthropic", + model: "claude-haiku-4-5", + inputTokens: 600, + outputTokens: 250, + }).micros; + const draftsToThreshold = Math.round(15_000_000 / perDraft); + expect(draftsToThreshold).toBeGreaterThan(5_000); + }); +});