Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
88 changes: 88 additions & 0 deletions app/api/cron/ai-spend/route.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
import { NextResponse } from "next/server";
import { serviceClient } from "@/lib/supabase/service";
import { env } from "@/lib/env";
import { formatUsd, spendToday, MICROS_PER_DOLLAR } from "@/lib/ai/spend";
import { sendAiSpendAlertEmail } from "@/lib/email";

export const runtime = "nodejs";

/**
* Warn when a day's AI spend crosses the threshold. It only warns — nothing
* here throttles, pauses or blocks a request. A budget alarm that silently
* turns the product off is worse than the bill it was meant to prevent.
*
* Run it hourly. The alert de-duplicates on (day, threshold), so a day that
* crosses the line mails once rather than on every run after it.
*
* The balance on the Anthropic account is deliberately not reported: reading
* it needs an Admin key (sk-ant-admin01-...), and a normal API key gets a 401
* on the cost and usage endpoints. What this can measure exactly is what this
* application spent, which is the number the alert is actually about.
*/
export async function GET(req: Request) {
return POST(req);
}

export async function POST(req: Request) {
const incoming =
req.headers.get("x-cron-secret") ??
req.headers.get("authorization")?.replace(/^Bearer\s+/i, "");
if (incoming !== env.cronSecret) {
return NextResponse.json({ ok: false, error: "unauthorized" }, { status: 401 });
}

const thresholdMicros = Math.round(env.aiSpendDailyAlertUsd * MICROS_PER_DOLLAR);
const spend = await spendToday();
const over = spend.totalMicros >= thresholdMicros;

if (!over) {
return NextResponse.json({
ok: true,
day: spend.day,
spend: formatUsd(spend.totalMicros),
threshold: formatUsd(thresholdMicros),
alerted: false,
});
}

const sb = serviceClient();
// Insert first and let the unique index decide. Checking-then-sending would
// mail twice if two runs overlap.
const { error: claimError } = await sb.from("ai_spend_alerts").insert({
day: spend.day,
threshold_micros: thresholdMicros,
spend_micros: spend.totalMicros,
sent_to: env.aiSpendAlertEmail,
});
if (claimError) {
// Already alerted for this day and threshold.
return NextResponse.json({
ok: true,
day: spend.day,
spend: formatUsd(spend.totalMicros),
alerted: false,
note: "already alerted today",
});
}

await sendAiSpendAlertEmail({
to: env.aiSpendAlertEmail,
day: spend.day,
spendLabel: formatUsd(spend.totalMicros),
thresholdLabel: formatUsd(thresholdMicros),
calls: spend.calls,
breakdown: spend.byFeature.map((f) => ({
feature: f.feature,
spendLabel: formatUsd(f.micros),
calls: f.calls,
})),
});

return NextResponse.json({
ok: true,
day: spend.day,
spend: formatUsd(spend.totalMicros),
threshold: formatUsd(thresholdMicros),
alerted: true,
});
}
145 changes: 145 additions & 0 deletions lib/ai/spend.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,145 @@
// What the AI costs, computed at the point of spending it.
//
// The provider will not tell the app: Anthropic's cost and usage reports need
// an Admin key, and a normal API key gets a 401 on them. What every response
// does carry is its token counts, and the per-model rates are published — so
// the cost is calculable here, and here is also the only place that knows
// which feature caused it.
//
// Everything below is in micro-dollars ($1 = 1_000_000). Model rates are per
// million tokens, so one Haiku draft costs a fraction of a cent; in cents, a
// day of real spending rounds to zero.

import { serviceClient } from "@/lib/supabase/service";

const MICROS_PER_DOLLAR = 1_000_000;

/** Published rates, in micro-dollars per million tokens. */
type Rate = { input: number; output: number };

const RATES: Record<string, Rate> = {
// Anthropic
"claude-haiku-4-5": { input: 1_000_000, output: 5_000_000 },
"claude-sonnet-5": { input: 3_000_000, output: 15_000_000 },
"claude-sonnet-4-6": { input: 3_000_000, output: 15_000_000 },
"claude-opus-5": { input: 5_000_000, output: 25_000_000 },
"claude-opus-4-8": { input: 5_000_000, output: 25_000_000 },
// OpenAI
"gpt-5-mini": { input: 250_000, output: 2_000_000 },
"gpt-5": { input: 1_250_000, output: 10_000_000 },
};

/**
* Rate for a model id, tolerating the dated suffixes the APIs return.
*
* A response reports `claude-haiku-4-5-20251001` where the request asked for
* `claude-haiku-4-5`. Matching on the longest known prefix keeps a dated id
* priced instead of silently costing zero.
*/
export function rateFor(model: string): Rate | null {
const id = model.trim().toLowerCase();
if (RATES[id]) return RATES[id];
let best: string | null = null;
for (const known of Object.keys(RATES)) {
if (id.startsWith(known) && (!best || known.length > best.length)) best = known;
}
return best ? RATES[best] : null;
}

export type UsageSample = {
provider: string;
model: string;
feature?: string | null;
inputTokens: number;
outputTokens: number;
cacheReadTokens?: number;
cacheWriteTokens?: number;
};

/** Cost of one call, in micro-dollars. Unknown models cost 0 and say so. */
export function costMicros(sample: UsageSample): { micros: number; rate: Rate | null } {
const rate = rateFor(sample.model);
if (!rate) return { micros: 0, rate: null };
// Cache reads bill at roughly a tenth of input; cache writes at 1.25x. Both
// are approximations of the published multipliers, and both are far closer
// than ignoring them, which is what counting only fresh input would do.
const inputEquivalent =
sample.inputTokens +
(sample.cacheReadTokens ?? 0) * 0.1 +
(sample.cacheWriteTokens ?? 0) * 1.25;
const micros =
(inputEquivalent * rate.input) / 1_000_000 + (sample.outputTokens * rate.output) / 1_000_000;
return { micros: Math.round(micros), rate };
}

/**
* Record one call.
*
* Never throws. This runs after a request the caller already considers
* successful, and failing to write a bookkeeping row is not a reason to fail
* the work that was actually asked for.
*/
export async function recordAiSpend(sample: UsageSample): Promise<void> {
try {
const { micros, rate } = costMicros(sample);
await serviceClient().from("ai_usage").insert({
provider: sample.provider,
model: sample.model,
feature: sample.feature ?? null,
input_tokens: sample.inputTokens,
output_tokens: sample.outputTokens,
cache_read_tokens: sample.cacheReadTokens ?? 0,
cache_write_tokens: sample.cacheWriteTokens ?? 0,
cost_micros: micros,
rate_input_micros_per_mtok: rate?.input ?? null,
rate_output_micros_per_mtok: rate?.output ?? null,
});
} catch {
// Bookkeeping is best-effort by design.
}
}

export type DaySpend = {
day: string;
totalMicros: number;
byFeature: { feature: string; micros: number; calls: number }[];
calls: number;
};

/** Spend since UTC midnight, with the breakdown that makes it actionable. */
export async function spendToday(): Promise<DaySpend> {
const start = new Date();
start.setUTCHours(0, 0, 0, 0);

const { data } = await serviceClient()
.from("ai_usage")
.select("feature, cost_micros")
.gte("occurred_at", start.toISOString());

const rows = (data as { feature: string | null; cost_micros: number }[] | null) ?? [];
const byFeature = new Map<string, { micros: number; calls: number }>();
let totalMicros = 0;
for (const r of rows) {
totalMicros += r.cost_micros;
const key = r.feature ?? "(unattributed)";
const cur = byFeature.get(key) ?? { micros: 0, calls: 0 };
cur.micros += r.cost_micros;
cur.calls += 1;
byFeature.set(key, cur);
}

return {
day: start.toISOString().slice(0, 10),
totalMicros,
calls: rows.length,
byFeature: [...byFeature.entries()]
.map(([feature, v]) => ({ feature, ...v }))
.sort((a, b) => b.micros - a.micros),
};
}

export function formatUsd(micros: number): string {
return `$${(micros / MICROS_PER_DOLLAR).toFixed(2)}`;
}

export { MICROS_PER_DOLLAR };
48 changes: 48 additions & 0 deletions lib/email.ts
Original file line number Diff line number Diff line change
Expand Up @@ -996,6 +996,54 @@ export function coldOutreachEmailHtml(input: {
});
}

/**
* Daily AI spend warning.
*
* Leads with the number and the breakdown, because "you spent $18" prompts
* the question "on what" and the answer should not require opening a
* dashboard. States plainly that nothing was stopped, so the mail is not read
* as an outage.
*/
export async function sendAiSpendAlertEmail(input: {
to: string;
day: string;
spendLabel: string;
thresholdLabel: string;
calls: number;
breakdown: { feature: string; spendLabel: string; calls: number }[];
}): Promise<{ sent: boolean; error?: string }> {
const c = client();
if (!c) return { sent: false, error: "RESEND_API_KEY not set" };

const rows = input.breakdown
.map(
(b) =>
`<tr><td style="padding:4px 12px 4px 0">${b.feature}</td>` +
`<td style="padding:4px 12px 4px 0;text-align:right;font-family:monospace">${b.spendLabel}</td>` +
`<td style="padding:4px 0;text-align:right;color:#666">${b.calls}</td></tr>`,
)
.join("");

const res = await c.send({
from: env.resendFrom,
to: input.to,
subject: `AI spend ${input.spendLabel} on ${input.day} (over ${input.thresholdLabel})`,
html: [
`<p><strong>${input.spendLabel}</strong> of AI usage so far on ${input.day}, across ${input.calls} calls.</p>`,
`<p>That is over your ${input.thresholdLabel}/day warning line. <strong>Nothing has been stopped</strong> — this is a heads-up, not an outage.</p>`,
rows
? `<table style="border-collapse:collapse;font-size:14px"><thead><tr>` +
`<th style="text-align:left;padding-right:12px">Feature</th>` +
`<th style="text-align:right;padding-right:12px">Spend</th>` +
`<th style="text-align:right;color:#666">Calls</th></tr></thead><tbody>${rows}</tbody></table>`
: "",
`<p style="color:#666;font-size:13px">Computed from token counts on each call at published per-model rates. It measures what this app spent, not the balance on the Anthropic account — reading that needs an Admin API key.</p>`,
].join(""),
});
if (!res.sent) return { sent: false, error: res.error };
return { sent: true };
}

export async function sendColdOutreachEmail(input: {
to: string;
subject: string;
Expand Down
5 changes: 5 additions & 0 deletions lib/env.ts
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,11 @@ export const env = {
backendAiOpenaiModel:
process.env.BACKEND_AI_OPENAI_MODEL ?? "gpt-5.5",
anthropicApiKey: process.env.ANTHROPIC_API_KEY ?? "",
// Daily AI spend that triggers a warning. It warns only — nothing throttles
// or pauses on it, because an alarm that turns the product off is worse
// than the bill it was meant to prevent.
aiSpendDailyAlertUsd: Number(process.env.AI_SPEND_DAILY_ALERT_USD ?? "15"),
aiSpendAlertEmail: process.env.AI_SPEND_ALERT_EMAIL ?? "anthony@profullstack.com",
openaiApiKey: process.env.OPENAI_API_KEY ?? "",
geminiApiKey: process.env.GEMINI_API_KEY ?? "",
dashscopeApiKey: process.env.DASHSCOPE_API_KEY ?? "", // Qwen
Expand Down
23 changes: 23 additions & 0 deletions lib/lx/backendAi.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@ import Anthropic from "@anthropic-ai/sdk";
import { zodOutputFormat } from "@anthropic-ai/sdk/helpers/zod";
import OpenAI from "openai";
import { z } from "zod/v4";
import { recordAiSpend } from "@/lib/ai/spend";
import { env } from "../env";

export type BackendAiProvider = "anthropic" | "openai";
Expand Down Expand Up @@ -120,6 +121,18 @@ async function generateWithAnthropic<T>(
messages: [{ role: "user", content: args.user }],
});
const response = await stream.finalMessage();
// Recorded before the parse check: the tokens were spent whether or not the
// output turns out to be usable, and a failed parse is exactly the kind of
// waste worth seeing in the total.
void recordAiSpend({
provider: "anthropic",
model: response.model ?? args.anthropicModel,
feature: args.name,
inputTokens: response.usage?.input_tokens ?? 0,
outputTokens: response.usage?.output_tokens ?? 0,
cacheReadTokens: response.usage?.cache_read_input_tokens ?? 0,
cacheWriteTokens: response.usage?.cache_creation_input_tokens ?? 0,
});
const parsed = response.parsed_output as T | null;
if (!parsed) {
throw new Error(
Expand Down Expand Up @@ -159,6 +172,16 @@ async function generateWithOpenAI<T>(
text: { format },
store: false,
});
// Same reasoning as the Anthropic path: record before any check that can
// throw, because the tokens are already spent by this point.
void recordAiSpend({
provider: "openai",
model: response.model ?? args.openaiModel ?? env.backendAiOpenaiModel,
feature: args.name,
inputTokens: response.usage?.input_tokens ?? 0,
outputTokens: response.usage?.output_tokens ?? 0,
cacheReadTokens: response.usage?.input_tokens_details?.cached_tokens ?? 0,
});
if (response.error) {
throw new Error(response.error.message);
}
Expand Down
Loading
Loading