diff --git a/Dockerfile.claude-code b/Dockerfile.claude-code new file mode 100644 index 000000000..b93707c03 --- /dev/null +++ b/Dockerfile.claude-code @@ -0,0 +1,33 @@ +# syntax=docker/dockerfile:1 +# +# Overlay image: the base Instatic app + the Claude Code CLI, so the +# `claude-code` AI provider works inside the container. Kept separate from the +# main Dockerfile so the base image tracks upstream cleanly. +# +# Build: +# docker build -t instatic:local . # base (main Dockerfile) +# docker build -f Dockerfile.claude-code -t instatic-cc:local . # this overlay +# +# Auth is NOT baked in. At run time provide the machine's Claude SUBSCRIPTION +# via CLAUDE_CODE_OAUTH_TOKEN (from `claude setup-token`) — see +# docs/features/claude-code-provider.md and compose.claude-code.yml. + +ARG BASE_IMAGE=instatic:local +FROM ${BASE_IMAGE} + +USER root +# The base `oven/bun` image has no curl; add it just to fetch the installer, +# then drop the apt lists. The installer drops a self-contained binary under +# /root/.local; copy it to a world-readable path and discard the rest. +RUN apt-get update \ + && apt-get install -y --no-install-recommends curl ca-certificates \ + && rm -rf /var/lib/apt/lists/* \ + && curl -fsSL https://claude.ai/install.sh | bash \ + && mkdir -p /opt/claude \ + && cp "$(readlink -f /root/.local/bin/claude)" /opt/claude/claude \ + && chmod 0755 /opt/claude/claude \ + && rm -rf /root/.local \ + && /opt/claude/claude --version + +ENV INSTATIC_CLAUDE_BIN=/opt/claude/claude +USER bun diff --git a/compose.claude-code.yml b/compose.claude-code.yml new file mode 100644 index 000000000..f07bf5060 --- /dev/null +++ b/compose.claude-code.yml @@ -0,0 +1,32 @@ +# compose.claude-code.yml +# Overlay that enables the Claude Code (subscription) AI provider in the +# container. Layer it LAST, on top of the prod (and sqlite) compose files. +# +# Usage: +# docker compose -f compose.prod.yml -f compose.sqlite.yml -f compose.claude-code.yml up -d +# +# Prerequisites (one-time): +# 1. Build the image WITH the Claude Code CLI: +# docker build -t instatic:local . +# docker build -f Dockerfile.claude-code -t instatic-cc:local . +# then set in .env: INSTATIC_IMAGE=instatic-cc:local +# +# 2. Mint a long-lived SUBSCRIPTION token on the host (interactive, one-time): +# claude setup-token +# then put it in .env: CLAUDE_CODE_OAUTH_TOKEN= +# This bills your flat-rate Claude subscription — it is NOT an API key and +# spends no metered API credits. Revoke/rotate it anytime by re-running +# `claude setup-token`. +# +# Note: no ~/.claude mount is needed — the token is the only auth input. + +services: + app: + environment: + # Subscription auth for the spawned `claude` CLI. The driver never strips + # this var (unlike ANTHROPIC_API_KEY), so the CLI authenticates on the + # subscription. Empty → the Claude Code provider will report it can't sign + # in; every other provider keeps working. + CLAUDE_CODE_OAUTH_TOKEN: ${CLAUDE_CODE_OAUTH_TOKEN:-} + # Baked into the overlay image already; set explicitly for clarity. + INSTATIC_CLAUDE_BIN: /opt/claude/claude diff --git a/docs/README.md b/docs/README.md index 833d8cfdb..cc4a9d61e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -152,6 +152,7 @@ Three categories, three voices: | [features/spotlight.md](features/spotlight.md) | Cmd+K command palette | | [features/agent.md](features/agent.md) | AI agent integration and provider-agnostic runtime | | [features/mcp-connectors.md](features/mcp-connectors.md) | Instatic as an MCP server — external AI clients drive the CMS over MCP | +| [features/claude-code-provider.md](features/claude-code-provider.md) | Running the agent on a Claude subscription by spawning the `claude` CLI | | [features/templates.md](features/templates.md) | Entry templates + dynamic bindings + token interpolation | | [features/loops.md](features/loops.md) | `base.loop` + loop entity sources | | [features/cms-native-forms.md](features/cms-native-forms.md) | Visual form primitives and secure public submissions | diff --git a/docs/features/agent.md b/docs/features/agent.md index cc3c90be2..8aa0d798c 100644 --- a/docs/features/agent.md +++ b/docs/features/agent.md @@ -6,7 +6,9 @@ In the Site editor, the agent reads the current page snapshot, plans a sequence In the Content workspace, the agent works against content collections and entries. It reads collection schemas and document state server-side, then mutates the live content editor through a browser bridge so the open draft, Tiptap body editor, and sidebar selection stay authoritative. -The agent runs on a provider-agnostic AI runtime (`server/ai/`) that can drive any supported model (Anthropic Claude, OpenAI, OpenRouter, Ollama, or any OpenAI-compatible endpoint). Every driver talks directly to its provider's REST API over HTTP/SSE — no provider SDKs. All drivers share one multi-turn tool loop (`drivers/http/toolLoop.ts`); each supplies only a small `ProviderAdapter` of pure mapping functions. The plain `@anthropic-ai/sdk` (and any provider SDK) is banned repo-wide. Gated by `ai-driver-isolation.test.ts`. +The agent runs on a provider-agnostic AI runtime (`server/ai/`) that can drive any supported model (Anthropic Claude, OpenAI, OpenRouter, Ollama, or any OpenAI-compatible endpoint). Every HTTP driver talks directly to its provider's REST API over HTTP/SSE — no provider SDKs. Those drivers share one multi-turn tool loop (`drivers/http/toolLoop.ts`); each supplies only a small `ProviderAdapter` of pure mapping functions. The plain `@anthropic-ai/sdk` (and any provider SDK) is banned repo-wide. Gated by `ai-driver-isolation.test.ts`. + +One driver is deliberately not an HTTP driver: [Claude Code](claude-code-provider.md) spawns the local `claude` CLI and runs the agent on the operator's Claude subscription rather than a metered API key. It brings its own tool loop (the CLI's), reaches the CMS over Instatic's own MCP server instead of the shared tool loop, and satisfies the SDK ban by construction — it imports nothing and spawns a binary. --- @@ -72,7 +74,10 @@ server/ai/ │ ├── openai.ts — OpenAI driver: direct POST /v1/responses (no SDK) │ ├── openrouter.ts — OpenRouter driver: direct POST /v1/responses (shared Responses path; live /models; native cost) │ ├── ollama.ts — Ollama driver: POST /v1/chat/completions via shared chatCompletions adapter; live /api/tags catalogue -│ └── openaiCompatible.ts — Custom Provider driver: any /v1/chat/completions endpoint; live GET /v1/models catalogue +│ ├── openaiCompatible.ts — Custom Provider driver: any /v1/chat/completions endpoint; live GET /v1/models catalogue +│ ├── claudeCode.ts — Claude Code driver (NOT an HTTP driver): spawns the `claude` CLI on the machine subscription; argv/env + session mode +│ ├── claudeCodeProcess.ts — one CLI spawn: stderr drain, idle watchdog, session-mode fallback, terminal outcome classification +│ └── claudeCodeEvents.ts — CLI stream-json → AiStreamEvent translation (pure; unit-tested from recorded output) └── runtime/ ├── runner.ts — runChat(): drives a driver, emits stream events ├── persister.ts — ConversationsPersister: messages + usage to DB; writes contextTokens snapshot diff --git a/docs/features/claude-code-provider.md b/docs/features/claude-code-provider.md new file mode 100644 index 000000000..b50ac81a5 --- /dev/null +++ b/docs/features/claude-code-provider.md @@ -0,0 +1,166 @@ +# Claude Code provider (subscription-backed) + +The **Claude Code** provider runs the CMS agent on the machine's logged-in +Claude **subscription** (Pro/Max) instead of a metered Anthropic API key. You +type in the normal AI panel; behind the scenes the driver spawns the headless +`claude` CLI, points it at Instatic's own MCP server, and streams its output +back into the chat — so the agent is real Claude Code (its tool loop, its +harness), editing the live workspace you have open. + +This is the inverse of connecting an *external* Claude Code to Instatic over +[MCP connectors](./mcp-connectors.md): here Instatic drives Claude Code, but the +editing still flows through the same live editor bridge. + +## Why a CLI subprocess (and not the Agent SDK) + +The AI drivers ban every provider **SDK** (`@anthropic-ai/claude-agent-sdk` +included — gated by `ai-driver-isolation.test.ts`) because drivers normally talk +raw provider REST. The Claude Code driver is the deliberate exception: it never +imports an SDK — it **spawns the `claude` binary**, which the gate (import-only) +permits. The subscription cost model only exists in the CLI/subscription auth +path; the Agent SDK bills API credits, so it would defeat the purpose anyway. + +## How it works + +``` +AI panel (Site workspace, open in the browser) + │ POST /admin/api/ai/chat/site + ▼ +server/ai/handlers/chat.ts → resolveDriver('claude-code') + ▼ +server/ai/drivers/claudeCode.ts ── Bun.spawn('claude', …) + │ • ANTHROPIC_API_KEY stripped from the child env → subscription auth + │ • ENABLE_TOOL_SEARCH=false in the child env → present every MCP tool up + │ front. See "Tool visibility" below; without this the turn has no + │ usable tools at all. + │ • --tools "" → strip ALL built-in tools (Bash/Read/Edit/…) + │ • --allowedTools mcp__instatic → pre-approve ONLY this server's tools, + │ so the ~50 mcp__instatic__* tools are directly callable + │ • --system-prompt + │ • --mcp-config --strict-mcp-config + │ • --dangerously-skip-permissions (built-ins stripped; ensures MCP + │ tool calls never block on a permission prompt) + │ • --session-id/--resume keyed on the conversation (deterministic UUID) + ▼ +claude CLI ──MCP (bearer)──▶ /_instatic/mcp + │ │ server/ai/mcp/connectors/internalConnector.ts + │ │ mints a per-user connector granted the + │ │ caller's OWN capabilities (never more) + ▼ ▼ +stream-json stdout executeAiTool / live editor bridge + │ ▼ +translated → AiStreamEvent the OPEN Site/Content workspace (your live edits) +``` + +- **Auth:** default (non-`--bare`) CLI mode reads the OAuth/subscription + credential. `ANTHROPIC_API_KEY`/`ANTHROPIC_AUTH_TOKEN` are removed from the + child env so the CLI can't fall back to metered billing. `--bare` is never + used (it forces API-key auth). +- **Credential-less:** the provider needs no secret. To satisfy the existing + `ai_creds_apikey_shape_check` DB constraint without a migration, the credential + row is stored as `authMode: 'baseUrl'` with the inert sentinel + `claude-code://local` (never dialed). See `CLAUDE_CODE_SENTINEL_BASE_URL`. +- **Tool surface:** the CLI sees only the `mcp__instatic__*` tools (the same + set as `server/ai/mcp/registry.ts`), capability-filtered by the internal + connector. Browser-execution tools route to the owner's open workspace via the + MCP editor bridge; headless reads run in-process. Edits stay drafts until an + explicit `site_publish`. +- **Pricing / context:** billed to the subscription, so `resolveCostUsd` returns + `0` for `claude-code`; usage is reported in the native Anthropic shape, so it + normalises like `anthropic` in `contextTokens.ts`. + +### Tool visibility (why `ENABLE_TOOL_SEARCH=false`) + +`--tools ""` strips every *built-in* tool, which is what keeps Bash/Edit/WebFetch +out of the CMS chat. The CLI's **tool search** (env `ENABLE_TOOL_SEARCH`, default +`auto`) defers a large MCP toolset behind the `ToolSearch` built-in — and that is +a built-in, so `--tools ""` strips it too. With ~50 tools on this server the +default crosses the deferral threshold, and the combination leaves the model with +**zero callable tools**. + +A model with no tools does not error: it narrates a tool call in prose +("`get_context` … calling that now") and ends the turn `subtype: "success"`. That +is why the failure looked like the agent simply stopping a few seconds in. + +Disabling tool search presents all ~50 tools directly, which is both correct and +faster (~5s to the first tool call, versus ~13-19s through search round trips). + +Two independent guards keep this from regressing silently: + +- The driver reads the tool list out of the CLI's `system`/`init` event and + aborts the turn with a clear error if no `mcp__instatic__*` tool is present, + rather than letting the model bluff. `mcp_servers[].status` distinguishes + "tool search hid them" from "the MCP server failed to start". +- Each turn logs `[ai/claude-code] session started — N CMS tool(s)`. `N` is + the number to check first when the agent misbehaves. + +### Other failure modes made visible + +A subprocess can go quiet in ways an HTTP driver cannot, and all of them look +identical from the composer. Each is terminal and named: + +- **Wedged child.** stderr is drained concurrently — an unread pipe blocks the + writer once full (~64KB) — and an idle watchdog + (`INSTATIC_CLAUDE_IDLE_TIMEOUT_MS`, default 180s) kills a CLI that stops + emitting, SIGTERM then SIGKILL. +- **Session mismatch.** `--resume` of a session that isn't on disk, and + `--session-id` of one that already exists, both exit non-zero. Each retries + once as the other mode; safe because the failed attempt emitted nothing. +- **Rate limits.** A `rate_limit_event` that isn't `allowed` is surfaced with its + reset time instead of looking like a stall. +- **Format drift.** Unreadable output lines are counted and logged, so a future + CLI changing its stream-json shape shows up as a warning rather than silence. + +## Setup (bare-metal / dev) + +1. Sign in with `claude` on the server host (subscription, not an API key). +2. Ensure the `claude` binary is on the server's PATH, or set + `INSTATIC_CLAUDE_BIN` to its absolute path. +3. In **AI → Providers**, add the **Claude Code (your subscription)** credential + (no key needed), then pick it as the Site/Content default or per conversation. + +## Setup (Docker / prod) + +The base image has no `claude` binary. Build the overlay image and provide a +subscription token as an env var — no home-directory mount required. + +1. **Build the image with the CLI** (`Dockerfile.claude-code` overlays the base): + ```sh + docker build -t instatic:local . + docker build -f Dockerfile.claude-code -t instatic-cc:local . + ``` + The overlay installs the self-contained `claude` binary to `/opt/claude/claude` + and bakes `INSTATIC_CLAUDE_BIN` pointing at it. + +2. **Mint a subscription token on the host** (interactive, one-time): + ```sh + claude setup-token # requires an active Claude subscription + ``` + This is a long-lived Claude Code OAuth token — it bills the flat-rate + subscription, spends no metered API credits, and is revocable by re-running + the command. + +3. **Wire it in `.env`** next to the compose files: + ```sh + INSTATIC_IMAGE=instatic-cc:local + CLAUDE_CODE_OAUTH_TOKEN= + ``` + +4. **Run with the overlay compose file** (layer it last): + ```sh + docker compose -f compose.prod.yml -f compose.sqlite.yml -f compose.claude-code.yml up -d + ``` + +Inside the container the driver spawns `claude`, which authenticates via +`CLAUDE_CODE_OAUTH_TOKEN` (the driver strips `ANTHROPIC_API_KEY`/ +`ANTHROPIC_AUTH_TOKEN` but preserves this var) and connects to the server's own +MCP endpoint on `127.0.0.1:$PORT` — same-container loopback. Then add the +**Claude Code (your subscription)** credential in **AI → Providers** as above. + +## Limitations + +- Single-operator / personal use. Driving a subscription programmatically to + back an app is a grey area of Anthropic's terms; this is intended for the + operator editing their own site, not multi-tenant serving. +- Subscription rate limits (e.g. Max weekly caps) apply. +- The internal connector token lives in-process and rotates on server restart. diff --git a/server/ai/contextTokens.ts b/server/ai/contextTokens.ts index 0dbfaa9e4..7982759ad 100644 --- a/server/ai/contextTokens.ts +++ b/server/ai/contextTokens.ts @@ -26,7 +26,9 @@ export function normalizeContextTokens( providerId: AiProviderId, usage: ContextUsageTokens, ): number { - if (providerId === 'anthropic') { + // Claude Code reports usage in the native Anthropic shape (input_tokens + // excludes the cache buckets), so it normalises the same way. + if (providerId === 'anthropic' || providerId === 'claude-code') { return usage.promptTokens + (usage.cacheReadTokens ?? 0) + (usage.cacheCreationTokens ?? 0) } return usage.promptTokens diff --git a/server/ai/drivers/claudeCode.ts b/server/ai/drivers/claudeCode.ts new file mode 100644 index 000000000..f48552c68 --- /dev/null +++ b/server/ai/drivers/claudeCode.ts @@ -0,0 +1,312 @@ +/** + * Claude Code driver — subscription-backed, spawns the local `claude` CLI. + * + * Unlike every other driver, this one does NOT talk to a REST API. It spawns + * the installed `claude` binary in headless mode (`claude -p --output-format + * stream-json`) authenticated by the machine's logged-in Claude subscription + * (Pro/Max) — NOT a metered API key. The whole point is to run the CMS agent + * on the flat-rate Claude Code subscription instead of pay-per-token API usage. + * + * How the CLI reaches the CMS tools: the spawned `claude` is pointed at + * Instatic's OWN MCP server (`/_instatic/mcp`) via `--mcp-config`, using an + * internal per-user connector bearer token. So Claude Code drives the exact + * same tool surface the built-in agent panel uses (`server/ai/mcp/registry.ts`), + * and browser-execution tools route to the user's OPEN Site/Content workspace + * through the live editor bridge — i.e. it edits the page you're looking at. + * + * Auth rules (see `claude --help`): + * - Default (non-`--bare`) mode reads the OAuth/keychain subscription creds. + * - `--bare` FORCES ANTHROPIC_API_KEY-only auth — never use it here. + * - If ANTHROPIC_API_KEY is present in the child env the CLI prefers it + * (metered API billing), so the driver STRIPS it from the subprocess env. + * + * The SDK-ban gate (`ai-driver-isolation.test.ts`) is import-based only: + * spawning a binary imports nothing, so this driver is compliant — it never + * imports `@anthropic-ai/claude-agent-sdk` or any provider SDK. + * + * The credential is a sentinel: the Claude Code provider is credential-less + * (auth is the machine subscription), but a conversation must reference a + * credential row. We store an `authMode: 'baseUrl'` row whose `baseUrl` is the + * sentinel `CLAUDE_CODE_SENTINEL_BASE_URL`; the driver ignores it. + * + * Three modules, by responsibility: + * + * claudeCode.ts this file — what to ask the CLI for: the provider + * definition, the argv/env it is invoked with, and the + * session-mode decision. + * claudeCodeProcess.ts how to run it — one spawn, its failure modes, and the + * terminal outcome each produces. + * claudeCodeEvents.ts how to read what came back — stream-json validation + * and translation to `AiStreamEvent`. + * + * A subprocess has many more ways to go quiet than an HTTP driver, and every one + * of them used to look identical from the composer: a few seconds of output, + * then nothing, forever — no error, no `done`. `claudeCodeProcess.ts` exists to + * give each of those silences a name; the one cause that lives HERE is the tool + * surface, because it is decided by the argv/env below: + * + * The CLI defers a large MCP toolset behind its `ToolSearch` built-in, which + * `--tools ''` strips — so the model got ZERO callable tools, bluffed a tool + * call in prose, and ended the turn `success`. `ENABLE_TOOL_SEARCH=false` + * fixes the cause; the `system`/`init` tool count is checked anyway, so a + * future CLI that ignores the env var fails loudly instead of bluffing. + */ + +import { mkdir } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + SYSTEM_PROMPT_DYNAMIC_BOUNDARY, + type AiContentBlock, + type AiMessage, + type AiProviderId, + type AiStreamEvent, +} from '../runtime/types' +import type { + AiProvider, + AiProviderCapabilities, + AiProviderModel, + AiResolvedCredential, + AiStreamRequest, +} from './types' +import { getInternalConnectorToken } from '../mcp/connectors/internalConnector' +import { runClaudeAttempt, type SessionMode } from './claudeCodeProcess' + +/** + * Sentinel `base_url` stored on the credential row so it satisfies the existing + * `ai_creds_apikey_shape_check` DB constraint (the `baseUrl` arm) without a new + * auth mode or migration. The driver never dials it — auth is the machine's + * Claude subscription. + */ +export const CLAUDE_CODE_SENTINEL_BASE_URL = 'claude-code://local' + +const SUPPORTED_AUTH_MODES = ['baseUrl'] as const + +/** + * Model aliases the `claude` CLI resolves to the latest of each family. Using + * aliases (not pinned ids) means new model releases surface with no code change + * — the closest CLI analogue of the other drivers' live `/v1/models` catalogue. + */ +const CLAUDE_CODE_MODELS: AiProviderModel[] = [ + { + id: 'sonnet', + label: 'Claude Sonnet (latest)', + tier: 'balanced', + capabilities: claudeCodeCapabilities(), + catalogueSource: 'live', + }, + { + id: 'opus', + label: 'Claude Opus (latest)', + tier: 'smartest', + capabilities: claudeCodeCapabilities(), + catalogueSource: 'live', + }, + { + id: 'haiku', + label: 'Claude Haiku (latest)', + tier: 'fast', + capabilities: claudeCodeCapabilities(), + catalogueSource: 'live', + }, +] + +function claudeCodeCapabilities(): AiProviderCapabilities { + return { + // Every current Claude model tool-calls and accepts images. Prompt caching + // is handled internally by the CLI (we don't send cache_control), so we + // report it off from the driver's perspective. + toolCalling: true, + visionInput: true, + toolResultImages: true, + promptCache: false, + streaming: true, + } +} + +export const claudeCodeDriver: AiProvider = { + id: 'claude-code' as AiProviderId, + label: 'Claude Code', + supportedAuthModes: [...SUPPORTED_AUTH_MODES], + + capabilities(_modelId: string) { + return claudeCodeCapabilities() + }, + + async listModels(_creds: AiResolvedCredential, _signal?: AbortSignal) { + // Static list — the CLI has no queryable catalogue. Marked `live` so the + // credential `/test` health check (which requires ≥1 live model) passes. + return CLAUDE_CODE_MODELS + }, + + async *stream(req: AiStreamRequest): AsyncIterable { + yield* streamClaudeCode(req) + }, +} + +// --------------------------------------------------------------------------- +// stream() implementation — spawn the CLI, translate its stream-json output +// --------------------------------------------------------------------------- + +/** Absolute path to the `claude` binary; overridable when it isn't on PATH. */ +const CLAUDE_BIN = process.env.INSTATIC_CLAUDE_BIN ?? 'claude' + +/** + * Extra orientation appended to Instatic's site system prompt: the CLI drives + * the CMS through the MCP tools (which it sees under the `mcp__instatic__` + * prefix), and has no filesystem/shell access here. + */ +const MCP_ORIENTATION = + 'You operate this CMS EXCLUSIVELY through the mcp__instatic__* tools shown to ' + + 'you — you have no filesystem, shell, or web access. Call get_context first to ' + + 'orient yourself in the live workspace. All edits are drafts; only call ' + + 'site_publish when the user explicitly asks to publish.' + +async function* streamClaudeCode(req: AiStreamRequest): AsyncIterable { + const { db, userId, capabilities, conversationId } = req.toolContextBase + + // 1) Internal MCP connector token so the spawned CLI can reach /_instatic/mcp + // with exactly this user's capabilities. + let token: string + try { + token = await getInternalConnectorToken(db, userId, capabilities) + } catch (err) { + yield { type: 'error', message: `Claude Code: could not provision MCP access — ${errMsg(err)}` } + return + } + + const port = process.env.PORT ?? '3001' + const mcpConfig = JSON.stringify({ + mcpServers: { + instatic: { + type: 'http', + url: `http://127.0.0.1:${port}/_instatic/mcp`, + headers: { Authorization: `Bearer ${token}` }, + }, + }, + }) + + const userText = latestUserText(req.messages) + if (!userText) { + yield { type: 'error', message: 'Claude Code: no user prompt to send.' } + return + } + + const systemPrompt = `${joinSystemPrompt(req.systemPrompt)}\n\n${MCP_ORIENTATION}` + + // Stable cwd so the CLI's on-disk session (keyed by session id + project) + // resolves across turns and server restarts; never the repo dir (avoids + // CLAUDE.md discovery leaking coding-agent context). + const cwd = join(tmpdir(), 'instatic-claude-code') + await mkdir(cwd, { recursive: true }).catch(() => { /* best-effort */ }) + + // Session continuity: one CLI session per conversation, keyed by a + // deterministic UUID so a restart still resumes. First user turn starts the + // session; later turns resume it (the CLI already holds the prior history, + // so we send only the newest user message). + const sessionUuid = await conversationSessionUuid(conversationId) + const userTurns = req.messages.filter((m) => m.role === 'user').length + const primaryMode: SessionMode = userTurns <= 1 ? 'start' : 'resume' + + const baseCmd = [ + CLAUDE_BIN, + '-p', userText, + '--output-format', 'stream-json', + '--verbose', + '--include-partial-messages', + '--model', req.modelId, + '--system-prompt', systemPrompt, + // Restrict the CLI to Instatic's MCP toolset ONLY. `--tools ''` strips every + // built-in (Bash/Edit/WebFetch/ToolSearch/…) and `--allowedTools` pre-approves + // this server's tools, so the model sees only the ~50 mcp__instatic__* tools + // and calls them directly — no built-in leakage into the CMS chat, no + // permission prompts. + // + // This ONLY holds with tool search disabled (see ENABLE_TOOL_SEARCH in the + // child env below). With it on, the CLI defers a large MCP toolset behind the + // `ToolSearch` built-in — which `--tools ''` strips — leaving the model with + // zero callable tools. It then narrates a tool call in prose and ends the + // turn `subtype: "success"`, so nothing surfaced as an error. + '--tools', '', + '--allowedTools', 'mcp__instatic', + '--mcp-config', mcpConfig, + '--strict-mcp-config', // ignore any other globally-configured MCP servers + '--dangerously-skip-permissions', // built-ins are stripped; guarantees MCP tools never block on a permission prompt + ] + + // Strip API-key auth so the CLI uses the machine's Claude SUBSCRIPTION, not + // metered API billing (the whole point of this provider). + const env = { ...process.env } + delete env.ANTHROPIC_API_KEY + delete env.ANTHROPIC_AUTH_TOKEN + + // Present every MCP tool up front instead of deferring them behind the + // `ToolSearch` built-in. The CLI's default (`auto`) defers once a toolset gets + // large, and this server exposes ~50 tools — comfortably over the threshold. + // Deferral is incompatible with `--tools ''`: the search tool the model would + // need is itself a built-in, so stripping built-ins strands every MCP tool + // behind a door the model cannot open. Disabling it also drops a round trip + // per turn (measured ~5s vs ~13-19s to the first tool call). + env.ENABLE_TOOL_SEARCH = 'false' + + let attempt = yield* runClaudeAttempt(baseCmd, primaryMode, sessionUuid, cwd, env, req.signal) + + // The on-disk session and our expectation can disagree: a first turn whose CLI + // never got to create the session, a `/tmp` wipe, or a re-sent first message. + // Both cases are recoverable, and retrying is safe only because the failed + // attempt emitted nothing. + if (attempt.retryAs) { + console.warn( + `[ai/claude-code] session ${attempt.mode} failed (${attempt.retryReason}); retrying as ${attempt.retryAs}.`, + ) + attempt = yield* runClaudeAttempt(baseCmd, attempt.retryAs, sessionUuid, cwd, env, req.signal) + } + + if (attempt.error) yield { type: 'error', message: attempt.error } + // No `done` here — the runner emits it when this generator returns. +} + +// --------------------------------------------------------------------------- +// Request → CLI input helpers +// --------------------------------------------------------------------------- + +function latestUserText(messages: AiMessage[]): string { + for (let i = messages.length - 1; i >= 0; i -= 1) { + const m = messages[i]! + if (m.role === 'user') return textOfBlocks(m.content) + } + return '' +} + +function textOfBlocks(blocks: AiContentBlock[]): string { + return blocks + .map((b) => (b.kind === 'text' ? b.text : '')) + .filter(Boolean) + .join('\n') + .trim() +} + +/** + * Flatten the canonical `systemPrompt` array (1- or 3-element `[prefix, + * BOUNDARY, suffix]`) into a single string — the CLI takes one `--system-prompt`. + */ +function joinSystemPrompt(parts: string[]): string { + return parts.filter((p) => p !== SYSTEM_PROMPT_DYNAMIC_BOUNDARY).join('\n\n') +} + +/** Deterministic UUIDv4-shaped session id derived from the conversation id. */ +async function conversationSessionUuid(conversationId: string): Promise { + const data = new TextEncoder().encode(`instatic-claude-code:${conversationId}`) + const digest = new Uint8Array(await crypto.subtle.digest('SHA-256', data)).slice(0, 16) + digest[6] = (digest[6]! & 0x0f) | 0x40 // version 4 + digest[8] = (digest[8]! & 0x3f) | 0x80 // variant 10x + const hex = [...digest].map((b) => b.toString(16).padStart(2, '0')).join('') + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}` +} + +function errMsg(err: unknown): string { + return err instanceof Error ? err.message : String(err) +} + +// --------------------------------------------------------------------------- +// stream-json event schema — boundary validation (no `as` on parsed JSON) diff --git a/server/ai/drivers/claudeCodeEvents.ts b/server/ai/drivers/claudeCodeEvents.ts new file mode 100644 index 000000000..1ca3e6ad7 --- /dev/null +++ b/server/ai/drivers/claudeCodeEvents.ts @@ -0,0 +1,277 @@ +/** + * Claude Code stream-json → `AiStreamEvent` translation. + * + * The CLI emits one JSON object per line. This module is the only thing that + * knows that wire format: it validates each line at the boundary and turns the + * handful of event types Instatic cares about into chat events, while + * accumulating the per-turn facts the driver needs to classify how the turn + * ended (see `TranslateState`). + * + * Deliberately pure and synchronous — no subprocess, no IO — so the whole + * translation layer is unit-testable from recorded CLI output. + */ + +import { Type, parseValue, type Static } from '@core/utils/typeboxHelpers' +import type { AiStreamEvent } from '../runtime/types' + +// --------------------------------------------------------------------------- + +export interface TranslateState { + sawText: boolean + /** tool_use_id → display name, so tool_result events can name their call. */ + toolNames: Map + resultSeen: boolean + /** The CLI announced its session (`system`/`init`) — tool counts are known. */ + sawInit: boolean + /** How many `mcp__instatic__*` tools the CLI offered the model at init. */ + mcpToolCount: number + /** The `instatic` server's status at init: 'connected' | 'failed' | 'pending'. */ + mcpServerStatus: string | null + /** Output lines we could not read — a format drift canary, logged at the end. */ + unparseableLines: number + toolCallCount: number + /** Set when the subscription limit blocks the turn, surfaced as the error. */ + rateLimitMessage: string | null + /** A failing `result` event, held for the caller to classify. */ + resultErrorMessage: string | null +} + +/** Fresh per-attempt translation state. Exported for the driver's unit tests. */ +export function createTranslateState(): TranslateState { + return { + sawText: false, + toolNames: new Map(), + resultSeen: false, + sawInit: false, + mcpToolCount: 0, + mcpServerStatus: null, + unparseableLines: 0, + toolCallCount: 0, + rateLimitMessage: null, + resultErrorMessage: null, + } +} + +export function* translateLine(line: string, state: TranslateState): Generator { + let evt: Static + try { + evt = parseValue(ClaudeCodeEventSchema, JSON.parse(line)) + } catch { + // Keep-alive or a shape we don't model. Counted rather than silently + // dropped: if the CLI's output format drifts, every event could start + // landing here and the turn would go quiet for no visible reason. + state.unparseableLines += 1 + return + } + + switch (evt.type) { + case 'system': { + if (evt.subtype !== 'init') return + state.sawInit = true + const tools = Array.isArray(evt.tools) ? evt.tools : [] + state.mcpToolCount = tools.filter( + (t) => typeof t === 'string' && t.startsWith('mcp__instatic__'), + ).length + const server = (evt.mcp_servers ?? []).find( + (s): s is { name?: unknown; status?: unknown } => + !!s && typeof s === 'object' && (s as { name?: unknown }).name === 'instatic', + ) + state.mcpServerStatus = typeof server?.status === 'string' ? server.status : null + console.info( + `[ai/claude-code] session ${evt.session_id ?? '?'} started — ` + + `${state.mcpToolCount} CMS tool(s), mcp_servers=${JSON.stringify(evt.mcp_servers ?? [])}`, + ) + return + } + case 'rate_limit_event': { + // Subscription limit. `allowed` is the normal steady-state ping; anything + // else means this turn is being throttled or refused, which otherwise + // shows up as the CLI going quiet. + const info = evt.rate_limit_info + if (!info || info.status === 'allowed') return + const resetsAt = typeof info.resetsAt === 'number' + ? new Date(info.resetsAt * 1000).toISOString().replace('T', ' ').slice(0, 16) + ' UTC' + : null + state.rateLimitMessage = + `Claude Code hit the ${info.rateLimitType ?? 'subscription'} limit on this machine's Claude ` + + `subscription (status: ${info.status})${resetsAt ? `. Resets at ${resetsAt}` : ''}.` + console.warn(`[ai/claude-code] ${state.rateLimitMessage}`) + return + } + case 'stream_event': { + // The nested `event` is a raw Anthropic SSE event. We only surface text + // deltas here; tool calls come from the aggregate `assistant` message. + const inner = evt.event + if (inner?.type === 'content_block_delta' && inner.delta?.type === 'text_delta') { + const text = inner.delta.text + if (typeof text === 'string' && text) { + state.sawText = true + yield { type: 'text', text } + } + } + return + } + case 'assistant': { + for (const block of evt.message?.content ?? []) { + if (block.type === 'tool_use' && typeof block.id === 'string') { + if (state.toolNames.has(block.id)) continue // already declared + const name = stripMcpPrefix(typeof block.name === 'string' ? block.name : 'tool') + state.toolNames.set(block.id, name) + state.toolCallCount += 1 + yield { + type: 'toolCall', + toolCallId: block.id, + toolName: name, + input: block.input ?? {}, + status: 'pending', + } + } + } + return + } + case 'user': { + // Tool results the CLI got back from the MCP server (executed against the + // live workspace). Emit a matching toolResult for each so the chat closes + // the tool-call badge. + for (const block of evt.message?.content ?? []) { + if (block.type === 'tool_result' && typeof block.tool_use_id === 'string') { + const name = state.toolNames.get(block.tool_use_id) ?? 'tool' + yield { + type: 'toolResult', + toolCallId: block.tool_use_id, + toolName: name, + ok: block.is_error !== true, + error: block.is_error === true ? toolErrorText(block.content) : undefined, + } + } + } + return + } + case 'result': { + state.resultSeen = true + const failed = evt.is_error === true || (evt.subtype != null && evt.subtype !== 'success') + if (failed) { + // Recorded, not yielded: the caller decides whether this is terminal or + // a recoverable session mismatch worth retrying. Yielding here would + // both pre-empt that choice and count as user-visible output. + state.resultErrorMessage = + `Claude Code turn failed${evt.subtype ? ` (${evt.subtype})` : ''}` + + `${evt.result ? `: ${evt.result}` : ''}.` + return + } + // Fallback: a response with no streamed text deltas — surface the final + // result string so the turn isn't silent. + if (!state.sawText && typeof evt.result === 'string' && evt.result) { + yield { type: 'text', text: evt.result } + } + const u = evt.usage + if (u) { + const promptTokens = u.input_tokens ?? 0 + const cacheReadTokens = u.cache_read_input_tokens ?? 0 + const cacheCreationTokens = u.cache_creation_input_tokens ?? 0 + yield { type: 'context', promptTokens, cacheReadTokens, cacheCreationTokens } + yield { + type: 'usage', + promptTokens, + completionTokens: u.output_tokens ?? 0, + cacheReadTokens, + cacheCreationTokens, + } + } + return + } + default: + return + } +} + +function stripMcpPrefix(name: string): string { + // mcp__instatic__site_insert_html → site_insert_html + const m = name.match(/^mcp__[^_]+(?:_[^_]+)*?__(.+)$/) + return m ? m[1]! : name.startsWith('mcp__') ? name.replace(/^mcp__[^_]*__/, '') : name +} + +function toolErrorText(content: unknown): string { + if (typeof content === 'string') return content + if (Array.isArray(content)) { + const text = content + .map((c) => (c && typeof c === 'object' && 'text' in c ? String((c as { text: unknown }).text) : '')) + .filter(Boolean) + .join(' ') + if (text) return text + } + return 'Tool call failed.' +} + +// --------------------------------------------------------------------------- + +// --------------------------------------------------------------------------- + +const AnthropicUsageShape = Type.Object( + { + input_tokens: Type.Optional(Type.Number()), + output_tokens: Type.Optional(Type.Number()), + cache_read_input_tokens: Type.Optional(Type.Number()), + cache_creation_input_tokens: Type.Optional(Type.Number()), + }, + { additionalProperties: true }, +) + +const ContentBlockShape = Type.Object( + { + type: Type.Optional(Type.String()), + id: Type.Optional(Type.String()), + name: Type.Optional(Type.String()), + input: Type.Optional(Type.Unknown()), + tool_use_id: Type.Optional(Type.String()), + is_error: Type.Optional(Type.Boolean()), + content: Type.Optional(Type.Unknown()), + text: Type.Optional(Type.String()), + }, + { additionalProperties: true }, +) + +/** `system`/`init` announces the toolset the model will actually see. */ +const RateLimitInfoShape = Type.Object( + { + status: Type.Optional(Type.String()), + resetsAt: Type.Optional(Type.Number()), + rateLimitType: Type.Optional(Type.String()), + }, + { additionalProperties: true }, +) + +const ClaudeCodeEventSchema = Type.Object( + { + type: Type.String(), + subtype: Type.Optional(Type.String()), + is_error: Type.Optional(Type.Boolean()), + result: Type.Optional(Type.String()), + usage: Type.Optional(AnthropicUsageShape), + session_id: Type.Optional(Type.String()), + tools: Type.Optional(Type.Array(Type.String())), + mcp_servers: Type.Optional(Type.Array(Type.Unknown())), + rate_limit_info: Type.Optional(RateLimitInfoShape), + event: Type.Optional( + Type.Object( + { + type: Type.Optional(Type.String()), + delta: Type.Optional( + Type.Object( + { type: Type.Optional(Type.String()), text: Type.Optional(Type.String()) }, + { additionalProperties: true }, + ), + ), + }, + { additionalProperties: true }, + ), + ), + message: Type.Optional( + Type.Object( + { content: Type.Optional(Type.Array(ContentBlockShape)) }, + { additionalProperties: true }, + ), + ), + }, + { additionalProperties: true }, +) diff --git a/server/ai/drivers/claudeCodeProcess.ts b/server/ai/drivers/claudeCodeProcess.ts new file mode 100644 index 000000000..019f9f10f --- /dev/null +++ b/server/ai/drivers/claudeCodeProcess.ts @@ -0,0 +1,199 @@ +/** + * One spawn of the `claude` CLI, and every way it can fail. + * + * Isolated from the driver because a subprocess has far more failure modes than + * an HTTP call, and all of them used to look identical from the composer: a few + * seconds of output, then nothing, forever. Everything here exists to turn one + * of those silences into a named, terminal outcome — see `AttemptOutcome`. + */ + +import type { AiStreamEvent } from '../runtime/types' +import { createTranslateState, translateLine } from './claudeCodeEvents' + +// --------------------------------------------------------------------------- + +export type SessionMode = 'start' | 'resume' + +export interface AttemptOutcome { + mode: SessionMode + /** Terminal message for the chat, or null when the turn ended cleanly. */ + error: string | null + /** Set when the turn failed in a way a different session mode would fix. */ + retryAs?: SessionMode + retryReason?: string +} + +/** + * How long the CLI may emit NOTHING before we treat the turn as wedged. It + * reports progress continuously while working — token-by-token deltas, a + * `system/thinking_tokens` tick while reasoning, tool traffic — so a gap this + * long means no output is coming, not that the model is busy. Without this a + * stuck child hangs the chat forever: the generator simply never yields again + * and the composer sits there with no error and no `done`. + */ +const IDLE_TIMEOUT_MS = Number(process.env.INSTATIC_CLAUDE_IDLE_TIMEOUT_MS ?? 180_000) + +/** Keep the tail of stderr for diagnostics without unbounded retention. */ +const STDERR_CAP = 4_000 + +export async function* runClaudeAttempt( + baseCmd: string[], + mode: SessionMode, + sessionUuid: string, + cwd: string, + env: Record, + signal: AbortSignal, +): AsyncGenerator { + const cmd = [ + ...baseCmd, + ...(mode === 'start' ? ['--session-id', sessionUuid] : ['--resume', sessionUuid]), + ] + + const proc = Bun.spawn(cmd, { cwd, env, stdin: 'ignore', stdout: 'pipe', stderr: 'pipe' }) + + // SIGTERM, then SIGKILL if it doesn't go. A CLI that ignores the first signal + // would otherwise keep `proc.exited` pending and wedge the turn anyway. + let killTimer: ReturnType | null = null + const kill = () => { + try { proc.kill() } catch { /* already gone */ } + if (killTimer) return + killTimer = setTimeout(() => { try { proc.kill('SIGKILL') } catch { /* gone */ } }, 5_000) + } + const onAbort = () => kill() + signal.addEventListener('abort', onAbort, { once: true }) + + // Drain stderr CONCURRENTLY. Reading it only after exit (or not at all) lets a + // chatty child fill the ~64KB pipe buffer and block forever on write — stdout + // then goes quiet with no error, which is indistinguishable from a hang. + let stderrText = '' + const stderrDone = (async () => { + try { + const dec = new TextDecoder() + for await (const chunk of proc.stderr as unknown as AsyncIterable) { + stderrText = (stderrText + dec.decode(chunk, { stream: true })).slice(-STDERR_CAP) + } + } catch { /* pipe torn down with the process */ } + })() + + const state = createTranslateState() + + let idleTimedOut = false + let idleTimer: ReturnType | null = null + const touch = () => { + if (idleTimer) clearTimeout(idleTimer) + idleTimer = setTimeout(() => { idleTimedOut = true; kill() }, IDLE_TIMEOUT_MS) + } + + let emitted = 0 + let drained = false + const decoder = new TextDecoder() + let buffer = '' + try { + touch() + for await (const chunk of proc.stdout as unknown as AsyncIterable) { + touch() + buffer += decoder.decode(chunk, { stream: true }) + let nl: number + while ((nl = buffer.indexOf('\n')) >= 0) { + const line = buffer.slice(0, nl) + buffer = buffer.slice(nl + 1) + if (!line.trim()) continue + for (const event of translateLine(line, state)) { emitted += 1; yield event } + // Fail fast on a toolless turn: the model cannot do anything useful and + // will bluff a tool call in prose, so stop before spending the turn. + if (state.sawInit && state.mcpToolCount === 0) { kill(); break } + } + if (state.sawInit && state.mcpToolCount === 0) break + } + if (buffer.trim() && !(state.sawInit && state.mcpToolCount === 0)) { + for (const event of translateLine(buffer, state)) { emitted += 1; yield event } + } + drained = true + } finally { + if (idleTimer) clearTimeout(idleTimer) + signal.removeEventListener('abort', onAbort) + // Abandoned before stdout ran out — the consumer stopped iterating (closed + // tab, handler teardown) and nothing else will ever reap this child, so the + // abort listener we just dropped can't do it either. + if (!drained) kill() + } + + const code = await proc.exited + await stderrDone + if (killTimer) clearTimeout(killTimer) + const stderr = stderrText.trim() + + if (state.unparseableLines > 0) { + console.warn( + `[ai/claude-code] skipped ${state.unparseableLines} unreadable output line(s) — CLI output format may have changed.`, + ) + } + + // --- terminal classification ------------------------------------------- + if (signal.aborted) return { mode, error: null } // user cancelled; runner handles it + + if (idleTimedOut) { + console.error(`[ai/claude-code] no output for ${IDLE_TIMEOUT_MS}ms — killed. stderr tail: ${stderr.slice(-400)}`) + return { + mode, + error: `Claude Code stopped responding (no output for ${Math.round(IDLE_TIMEOUT_MS / 1000)}s) and was stopped. Your message is saved — send it again to retry.`, + } + } + + if (state.sawInit && state.mcpToolCount === 0) { + console.error( + `[ai/claude-code] CLI exposed no mcp__instatic__* tools ` + + `(mcp server status: ${state.mcpServerStatus ?? 'unreported'}). stderr tail: ${stderr.slice(-400)}`, + ) + // The CLI tells us which of the two causes it was, so say so rather than + // offering the operator a choice of theories. + const cause = state.mcpServerStatus === 'failed' + ? "the CMS tool server (`instatic`) failed to start, so the CLI had nothing to call — its access token may have been revoked." + : 'the CLI did not expose them — most likely its tool-search feature hid the toolset behind a built-in that this integration strips.' + return { + mode, + error: `Claude Code started with none of the CMS tools available, so it could not act on the site: ${cause}` + + `${stderr ? ` CLI said: ${stderr.slice(-300)}` : ''}`, + } + } + + if (state.rateLimitMessage) return { mode, error: state.rateLimitMessage } + + if (code !== 0 && emitted === 0) { + // Nothing reached the user yet, so a different session mode can still be + // tried cleanly. Match on the CLI's own wording for the two recoverable cases. + if (mode === 'resume' && /no conversation found/i.test(stderr)) { + return { mode, error: null, retryAs: 'start', retryReason: 'session not on disk' } + } + if (mode === 'start' && /already in use/i.test(stderr)) { + return { mode, error: null, retryAs: 'resume', retryReason: 'session already exists' } + } + } + + if (state.resultErrorMessage) { + console.error(`[ai/claude-code] ${state.resultErrorMessage} stderr tail: ${stderr.slice(-400)}`) + return { + mode, + error: `${state.resultErrorMessage}${stderr ? ` CLI said: ${stderr.slice(-300)}` : ''}`, + } + } + + if (!state.resultSeen && code !== 0) { + console.error(`[ai/claude-code] exited ${code}. stderr tail: ${stderr.slice(-400)}`) + return { + mode, + error: `Claude Code exited ${code}${stderr ? `: ${stderr.slice(-500)}` : ''}. Is \`claude\` installed and signed in on the server?`, + } + } + + // Turn ended `success` without calling a single tool AND without saying + // anything — the other shape of a silent no-op turn. + if (state.resultSeen && emitted === 0) { + console.error(`[ai/claude-code] turn produced no output at all (exit ${code}). stderr tail: ${stderr.slice(-400)}`) + return { mode, error: 'Claude Code finished without producing a reply. Send your message again to retry.' } + } + + return { mode, error: null } +} + +// --------------------------------------------------------------------------- diff --git a/server/ai/drivers/index.ts b/server/ai/drivers/index.ts index 01472cef7..475ad5a5e 100644 --- a/server/ai/drivers/index.ts +++ b/server/ai/drivers/index.ts @@ -13,6 +13,7 @@ import { openaiDriver } from './openai' import { ollamaDriver } from './ollama' import { openrouterDriver } from './openrouter' import { openaiCompatibleDriver } from './openaiCompatible' +import { claudeCodeDriver } from './claudeCode' const DRIVERS: Record = { anthropic: anthropicDriver, @@ -20,6 +21,7 @@ const DRIVERS: Record = { ollama: ollamaDriver, openrouter: openrouterDriver, 'openai-compatible': openaiCompatibleDriver, + 'claude-code': claudeCodeDriver, } /** Returns the driver for a provider id, or throws if unknown. */ diff --git a/server/ai/handlers/credentials.ts b/server/ai/handlers/credentials.ts index cb70de721..de4039c16 100644 --- a/server/ai/handlers/credentials.ts +++ b/server/ai/handlers/credentials.ts @@ -42,6 +42,7 @@ const ProviderId = Type.Union([ Type.Literal('ollama'), Type.Literal('openrouter'), Type.Literal('openai-compatible'), + Type.Literal('claude-code'), ]) const CreateBodySchema = Type.Union([ diff --git a/server/ai/handlers/models.ts b/server/ai/handlers/models.ts index 12374db7a..c370ff0c6 100644 --- a/server/ai/handlers/models.ts +++ b/server/ai/handlers/models.ts @@ -20,7 +20,7 @@ import { getModelCatalogue, pricingKey } from '../pricing' import type { AiProviderModel } from '../drivers/types' import type { AiProviderId } from '../runtime/types' -const VALID_PROVIDERS: AiProviderId[] = ['anthropic', 'openai', 'ollama', 'openrouter', 'openai-compatible'] +const VALID_PROVIDERS: AiProviderId[] = ['anthropic', 'openai', 'ollama', 'openrouter', 'openai-compatible', 'claude-code'] export function tryHandleAiModels( req: Request, @@ -77,7 +77,7 @@ async function handleModels( id: '', providerId, authMode: - providerId === 'ollama' || providerId === 'openai-compatible' + providerId === 'ollama' || providerId === 'openai-compatible' || providerId === 'claude-code' ? ('baseUrl' as const) : ('apiKey' as const), apiKey: null, diff --git a/server/ai/mcp/connectors/internalConnector.ts b/server/ai/mcp/connectors/internalConnector.ts new file mode 100644 index 000000000..607e0db1b --- /dev/null +++ b/server/ai/mcp/connectors/internalConnector.ts @@ -0,0 +1,87 @@ +/** + * Internal MCP connector for the Claude Code driver. + * + * The Claude Code driver spawns the `claude` CLI and points it at Instatic's + * own MCP server (`/_instatic/mcp`). The CLI authenticates with a bearer token + * like any external MCP client — so the driver needs to mint one internally, + * without the HTTP connector API's step-up prompt (this is a server-internal, + * same-user delegation, not an operator minting a shareable secret). + * + * The connector is granted EXACTLY the chatting user's own capabilities, so the + * spawned CLI can do precisely what the user could through the built-in agent — + * never more (the privilege floor is satisfied by construction). + * + * Token lifecycle: the plaintext is only knowable at mint time (the DB stores a + * hash), so we cache it in-process, keyed by user. On the first use per server + * process we revoke any stale internal connectors from earlier boots (their + * plaintext is unrecoverable, hence unusable) and mint a fresh non-expiring one. + * A server restart rotates the token. + */ + +import type { DbClient } from '../../../db/client' +import type { CoreCapability } from '@core/capabilities' +import { createBearerConnection, listConnectorsForUser, revokeConnector } from './store' +import { generatePersonalAccessToken, hashMcpSecret } from './token' + +/** Reserved label marking a connector as driver-owned (not operator-created). */ +export const INTERNAL_CONNECTOR_LABEL = 'Claude Code (internal)' + +/** userId → plaintext bearer token, for this server process only. */ +const tokenByUser = new Map() + +/** In-flight mint per user, so concurrent turns don't double-mint. */ +const inflightByUser = new Map>() + +/** + * Return a usable bearer token for the internal Claude Code connector for + * `userId`, minting one on first use per process. `capabilities` must be the + * caller's own capability set — the connector is granted that exact subset. + */ +export async function getInternalConnectorToken( + db: DbClient, + userId: string, + capabilities: readonly CoreCapability[], +): Promise { + const cached = tokenByUser.get(userId) + if (cached) return cached + + const inflight = inflightByUser.get(userId) + if (inflight) return inflight + + const minting = mintForUser(db, userId, capabilities) + .then((token) => { + tokenByUser.set(userId, token) + return token + }) + .finally(() => { + inflightByUser.delete(userId) + }) + inflightByUser.set(userId, minting) + return minting +} + +async function mintForUser( + db: DbClient, + userId: string, + capabilities: readonly CoreCapability[], +): Promise { + // Revoke stale internal connectors from earlier boots — their plaintext is + // gone, so they're dead weight — to keep the MCP tab tidy. + const existing = await listConnectorsForUser(db, userId) + for (const c of existing) { + if (c.label === INTERNAL_CONNECTOR_LABEL && c.revokedAt === null) { + await revokeConnector(db, c.id, userId) + } + } + + const token = generatePersonalAccessToken() + const tokenHash = await hashMcpSecret(token) + await createBearerConnection(db, { + userId, + label: INTERNAL_CONNECTOR_LABEL, + capabilities, + tokenHash, + ttlDays: null, // never expires; rotated on each server restart + }) + return token +} diff --git a/server/ai/pricing/index.ts b/server/ai/pricing/index.ts index 17c1d068c..4f0064d02 100644 --- a/server/ai/pricing/index.ts +++ b/server/ai/pricing/index.ts @@ -55,7 +55,9 @@ export async function resolveCostUsd( modelId: string, usage: UsageTokens, ): Promise { - if (providerId === 'ollama') return 0 + // Ollama is self-hosted; Claude Code bills the flat-rate subscription, not + // per token — neither has a per-usage USD cost. + if (providerId === 'ollama' || providerId === 'claude-code') return 0 const catalogue = await ensureCatalogue(db) const entry = catalogue.get(pricingKey(modelId)) diff --git a/server/ai/runtime/types.ts b/server/ai/runtime/types.ts index afdc8f51b..f6b66f49e 100644 --- a/server/ai/runtime/types.ts +++ b/server/ai/runtime/types.ts @@ -24,7 +24,7 @@ export type { AiContentBlock, AiToolImage, AiToolOutput } from '@core/ai' // Provider identity + auth modes // --------------------------------------------------------------------------- -export type AiProviderId = 'anthropic' | 'openai' | 'ollama' | 'openrouter' | 'openai-compatible' +export type AiProviderId = 'anthropic' | 'openai' | 'ollama' | 'openrouter' | 'openai-compatible' | 'claude-code' /** * Credential auth modes. * diff --git a/src/__tests__/ai/claudeCodeMapping.test.ts b/src/__tests__/ai/claudeCodeMapping.test.ts new file mode 100644 index 000000000..b310793da --- /dev/null +++ b/src/__tests__/ai/claudeCodeMapping.test.ts @@ -0,0 +1,174 @@ +import { describe, test, expect } from 'bun:test' +import { + createTranslateState, + translateLine, + type TranslateState, +} from '../../../server/ai/drivers/claudeCodeEvents' +import type { AiStreamEvent } from '../../../server/ai/runtime/types' + +function feed(state: TranslateState, ...events: unknown[]): AiStreamEvent[] { + const out: AiStreamEvent[] = [] + for (const e of events) out.push(...translateLine(JSON.stringify(e), state)) + return out +} + +const INIT_WITH_TOOLS = { + type: 'system', + subtype: 'init', + session_id: 's1', + tools: ['mcp__instatic__get_context', 'mcp__instatic__site_get_tree'], + mcp_servers: [{ name: 'instatic', status: 'pending' }], +} + +describe('Claude Code stream-json translate', () => { + test('streams text deltas from the nested Anthropic event', () => { + const state = createTranslateState() + const events = feed( + state, + { type: 'stream_event', event: { type: 'content_block_delta', delta: { type: 'text_delta', text: 'Hel' } } }, + { type: 'stream_event', event: { type: 'content_block_delta', delta: { type: 'text_delta', text: 'lo' } } }, + ) + expect(events).toEqual([ + { type: 'text', text: 'Hel' }, + { type: 'text', text: 'lo' }, + ]) + expect(state.sawText).toBe(true) + }) + + test('thinking deltas produce no chat output', () => { + const state = createTranslateState() + const events = feed(state, { + type: 'stream_event', + event: { type: 'content_block_delta', delta: { type: 'thinking_delta', thinking: 'hmm' } }, + }) + expect(events).toEqual([]) + expect(state.sawText).toBe(false) + }) + + test('emits a toolCall per tool_use and names its result, stripping the mcp prefix', () => { + const state = createTranslateState() + const calls = feed(state, { + type: 'assistant', + message: { content: [{ type: 'tool_use', id: 'tu_1', name: 'mcp__instatic__site_get_tree', input: { a: 1 } }] }, + }) + expect(calls).toEqual([ + { type: 'toolCall', toolCallId: 'tu_1', toolName: 'site_get_tree', input: { a: 1 }, status: 'pending' }, + ]) + expect(state.toolCallCount).toBe(1) + + const results = feed(state, { + type: 'user', + message: { content: [{ type: 'tool_result', tool_use_id: 'tu_1' }] }, + }) + expect(results).toEqual([ + { type: 'toolResult', toolCallId: 'tu_1', toolName: 'site_get_tree', ok: true, error: undefined }, + ]) + }) + + // The bug this driver kept hitting: with tool search on, the CLI deferred every + // MCP tool behind a built-in that `--tools ''` strips, so the model got nothing + // to call, narrated a tool call in prose, and ended the turn `success`. + test('counts the CMS tools the CLI actually offered at init', () => { + const state = createTranslateState() + expect(feed(state, INIT_WITH_TOOLS)).toEqual([]) + expect(state.sawInit).toBe(true) + expect(state.mcpToolCount).toBe(2) + }) + + test('a toolless init is visible as zero CMS tools', () => { + const state = createTranslateState() + feed(state, { type: 'system', subtype: 'init', session_id: 's1', tools: [], mcp_servers: [{ name: 'instatic', status: 'pending' }] }) + expect(state.sawInit).toBe(true) + expect(state.mcpToolCount).toBe(0) + }) + + // The two toolless causes need different advice, and the CLI already tells us + // which one happened. + test('records the instatic MCP server status so the cause can be named', () => { + const connected = createTranslateState() + feed(connected, INIT_WITH_TOOLS) + expect(connected.mcpServerStatus).toBe('pending') + + const failed = createTranslateState() + feed(failed, { type: 'system', subtype: 'init', tools: [], mcp_servers: [{ name: 'instatic', status: 'failed' }] }) + expect(failed.mcpServerStatus).toBe('failed') + expect(failed.mcpToolCount).toBe(0) + + const absent = createTranslateState() + feed(absent, { type: 'system', subtype: 'init', tools: [], mcp_servers: [{ name: 'other', status: 'connected' }] }) + expect(absent.mcpServerStatus).toBeNull() + }) + + test('built-in tools do not count as CMS tools', () => { + const state = createTranslateState() + feed(state, { type: 'system', subtype: 'init', tools: ['Bash', 'ToolSearch'], mcp_servers: [] }) + expect(state.mcpToolCount).toBe(0) + }) + + test('an allowed rate-limit ping is not an error', () => { + const state = createTranslateState() + feed(state, { type: 'rate_limit_event', rate_limit_info: { status: 'allowed', rateLimitType: 'five_hour' } }) + expect(state.rateLimitMessage).toBeNull() + }) + + test('a blocking rate limit is captured with its reset time', () => { + const state = createTranslateState() + feed(state, { + type: 'rate_limit_event', + rate_limit_info: { status: 'rejected', rateLimitType: 'five_hour', resetsAt: 1790661000 }, + }) + expect(state.rateLimitMessage).toContain('five_hour') + expect(state.rateLimitMessage).toContain('rejected') + expect(state.rateLimitMessage).toContain('2026-09-29') + }) + + test('a successful result reports usage and context', () => { + const state = createTranslateState() + const events = feed(state, { + type: 'result', + subtype: 'success', + is_error: false, + usage: { input_tokens: 10, output_tokens: 41, cache_read_input_tokens: 6330, cache_creation_input_tokens: 5 }, + }) + expect(state.resultSeen).toBe(true) + expect(state.resultErrorMessage).toBeNull() + expect(events).toEqual([ + { type: 'context', promptTokens: 10, cacheReadTokens: 6330, cacheCreationTokens: 5 }, + { type: 'usage', promptTokens: 10, completionTokens: 41, cacheReadTokens: 6330, cacheCreationTokens: 5 }, + ]) + }) + + test('a result with no streamed text falls back to the result string', () => { + const state = createTranslateState() + const events = feed(state, { type: 'result', subtype: 'success', result: 'final answer' }) + expect(events).toEqual([{ type: 'text', text: 'final answer' }]) + }) + + // Held rather than yielded so the caller can decide between "terminal" and + // "recoverable session mismatch, retry as the other mode". + test('a failing result is recorded, not emitted as a stream error', () => { + const state = createTranslateState() + const events = feed(state, { + type: 'result', + subtype: 'error_during_execution', + is_error: true, + session_id: 'gone', + }) + expect(events).toEqual([]) + expect(state.resultSeen).toBe(true) + expect(state.resultErrorMessage).toContain('error_during_execution') + }) + + test('unreadable lines are counted so format drift is diagnosable', () => { + const state = createTranslateState() + expect([...translateLine('not json at all', state)]).toEqual([]) + expect([...translateLine('{"type":', state)]).toEqual([]) + expect(state.unparseableLines).toBe(2) + }) + + test('unknown event types are ignored without counting as drift', () => { + const state = createTranslateState() + expect(feed(state, { type: 'some_future_event', payload: 1 })).toEqual([]) + expect(state.unparseableLines).toBe(0) + }) +}) diff --git a/src/admin/ai/api.ts b/src/admin/ai/api.ts index 0eb0c165a..7664d8acb 100644 --- a/src/admin/ai/api.ts +++ b/src/admin/ai/api.ts @@ -39,6 +39,7 @@ const ProviderId = Type.Union([ Type.Literal('ollama'), Type.Literal('openrouter'), Type.Literal('openai-compatible'), + Type.Literal('claude-code'), ]) const AuthMode = Type.Union([ @@ -182,13 +183,13 @@ export async function listCredentials(signal?: AbortSignal): Promise { const key = `${providerId}\0${credentialId ?? ''}` diff --git a/src/admin/pages/ai/AiPage.module.css b/src/admin/pages/ai/AiPage.module.css index 992329750..ac0ac28b0 100644 --- a/src/admin/pages/ai/AiPage.module.css +++ b/src/admin/pages/ai/AiPage.module.css @@ -583,6 +583,19 @@ gap: var(--space-2xl); } +/* Stands in for the credential fields when a provider brings its own auth, so + the form still has something to say where the inputs would have been. */ +.providerSetupNote { + margin: 0; + padding: var(--space-m) var(--space-l); + border: 1px solid var(--border-subtle); + border-radius: var(--radius-m); + background: var(--surface-subtle); + color: var(--text-subtle); + font-size: var(--text-m); + line-height: 1.5; +} + .providerSetupField { display: grid; grid-template-columns: 150px minmax(0, 1fr); diff --git a/src/admin/pages/ai/ProviderMark.tsx b/src/admin/pages/ai/ProviderMark.tsx index 83b9dcb02..6d874ea03 100644 --- a/src/admin/pages/ai/ProviderMark.tsx +++ b/src/admin/pages/ai/ProviderMark.tsx @@ -5,6 +5,8 @@ import styles from './AiPage.module.css' const LOGO_CLASS: Partial> = { anthropic: styles.logoAnthropic, + // Same models, same mark — only the billing path differs. + 'claude-code': styles.logoAnthropic, openai: styles.logoOpenai, openrouter: styles.logoOpenrouter, ollama: styles.logoOllama, diff --git a/src/admin/pages/ai/providerCatalog.ts b/src/admin/pages/ai/providerCatalog.ts index 9f4c92af2..b66dc2ea9 100644 --- a/src/admin/pages/ai/providerCatalog.ts +++ b/src/admin/pages/ai/providerCatalog.ts @@ -1,4 +1,10 @@ -export type ProviderId = 'anthropic' | 'openai' | 'openrouter' | 'ollama' | 'openai-compatible' +export type ProviderId = + | 'anthropic' + | 'claude-code' + | 'openai' + | 'openrouter' + | 'ollama' + | 'openai-compatible' export type ProviderAuthMode = 'apiKey' | 'baseUrl' export interface ProviderSpec { @@ -8,6 +14,22 @@ export interface ProviderSpec { description: string authMode: ProviderAuthMode endpointLabel: string + /** + * The provider authenticates by some means outside Instatic, so the connect + * form collects no secret at all. The credential row still exists (a + * conversation must reference one) but carries only the inert sentinel below. + */ + credentialLess?: boolean + /** + * Shown in place of the credential fields when `credentialLess` — it has to + * explain where the auth actually comes from, since the form asks for nothing. + */ + credentialLessHint?: string + /** + * Stored as the credential's `baseUrl` when `credentialLess`. Never dialed — + * it exists only so the row satisfies the API's auth-shape check. + */ + sentinelBaseUrl?: string } export const PROVIDER_SPECS: ProviderSpec[] = [ @@ -19,6 +41,20 @@ export const PROVIDER_SPECS: ProviderSpec[] = [ authMode: 'apiKey', endpointLabel: 'api.anthropic.com', }, + { + id: 'claude-code', + label: 'Claude Code', + shortLabel: 'Your Claude subscription', + description: + 'Run the agent on this server’s Claude subscription instead of metered API credits.', + authMode: 'baseUrl', + endpointLabel: 'Local Claude Code CLI', + credentialLess: true, + credentialLessHint: + 'Claude Code runs on the Claude subscription this server is signed in to — no API key, ' + + 'and no metered API credits. Make sure the `claude` CLI is installed and signed in on the server.', + sentinelBaseUrl: 'claude-code://local', + }, { id: 'openai', label: 'OpenAI', diff --git a/src/admin/pages/ai/tabs/ProvidersTab.tsx b/src/admin/pages/ai/tabs/ProvidersTab.tsx index 88de6ce72..ac4cd89c5 100644 --- a/src/admin/pages/ai/tabs/ProvidersTab.tsx +++ b/src/admin/pages/ai/tabs/ProvidersTab.tsx @@ -350,9 +350,22 @@ function CredentialDetail({
- - - + + {!provider.credentialLess && ( + + )} +

Connection details

-

Instatic validates and encrypts the credential before storing it.

+

+ {provider.credentialLess + ? 'This provider brings its own authentication, so there is nothing to enter but a name.' + : 'Instatic validates and encrypts the credential before storing it.'} +

@@ -525,7 +542,16 @@ function AddCredentialForm({ event.preventDefault() setBusy(true) try { - const body: CreateCredentialBody = provider.authMode === 'apiKey' + const body: CreateCredentialBody = provider.credentialLess + ? { + // Nothing was collected: the provider authenticates elsewhere. The + // row exists only so a conversation has a credential to reference. + providerId: provider.id, + authMode: 'baseUrl', + displayLabel, + baseUrl: provider.sentinelBaseUrl ?? '', + } + : provider.authMode === 'apiKey' ? { providerId: provider.id, authMode: 'apiKey', @@ -569,47 +595,53 @@ function AddCredentialForm({ - {provider.authMode === 'baseUrl' && ( -
- -
- setBaseUrl(event.currentTarget.value)} - placeholder={baseUrlPlaceholder} - required - /> -

The root URL of the compatible API.

+ {provider.credentialLess ? ( +

{provider.credentialLessHint}

+ ) : ( + <> + {provider.authMode === 'baseUrl' && ( +
+ +
+ setBaseUrl(event.currentTarget.value)} + placeholder={baseUrlPlaceholder} + required + /> +

The root URL of the compatible API.

+
+
+ )} + +
+ +
+ setApiKey(event.currentTarget.value)} + placeholder={API_KEY_PLACEHOLDER[provider.id] ?? 'Leave blank if no auth'} + autoComplete="new-password" + data-1p-ignore="true" + data-lpignore="true" + data-bwignore="true" + data-form-type="other" + required={provider.authMode === 'apiKey'} + /> +

{provider.authMode === 'apiKey' ? 'Stored encrypted and never displayed again.' : 'Leave blank when the endpoint does not require authentication.'}

+
-
+ )} -
- -
- setApiKey(event.currentTarget.value)} - placeholder={API_KEY_PLACEHOLDER[provider.id] ?? 'Leave blank if no auth'} - autoComplete="new-password" - data-1p-ignore="true" - data-lpignore="true" - data-bwignore="true" - data-form-type="other" - required={provider.authMode === 'apiKey'} - /> -

{provider.authMode === 'apiKey' ? 'Stored encrypted and never displayed again.' : 'Leave blank when the endpoint does not require authentication.'}

-
-
-