diff --git a/.ai-run/guides/architecture/architecture.md b/.ai-run/guides/architecture/architecture.md index 49879f6a6..54aed84c1 100644 --- a/.ai-run/guides/architecture/architecture.md +++ b/.ai-run/guides/architecture/architecture.md @@ -267,3 +267,4 @@ FrameworkRegistry.get('langgraph') // src/frameworks/registry.ts | Core | `src/*/core/` | | Utils | `src/utils/` | | Tests | `tests/integration/`, `src/**/__tests__/` | +| OTLP ingestion adapters | `OtlpAgentAdapter` (`src/agents/core/types.ts`) is a second, non-chat plugin type registered the same way as `AgentAdapter` — one plugin per coding tool, ingesting that tool's native hook/telemetry events into the analytics pipeline. See `docs/ARCHITECTURE-OTLP-PLUGIN.md` for the dispatch pattern and how to add a new adapter (e.g. a future Cursor/other-tool adapter). | diff --git a/.ai-run/guides/integration/external-integrations.md b/.ai-run/guides/integration/external-integrations.md index 03c9319b8..04128067f 100644 --- a/.ai-run/guides/integration/external-integrations.md +++ b/.ai-run/guides/integration/external-integrations.md @@ -15,6 +15,7 @@ | OpenCode | Open-source AI assistant | SSO/API Key | Via CodeMie proxy | | MCP Servers | Remote MCP tool servers | OAuth 2.0 (auto) | `codemie-mcp-proxy` | | Enterprise SSO | Corporate auth | SAML/OAuth | `SSO_BASE_URL` | +| OTLP hook ingestion | Coding tool's own native hook/telemetry events → analytics pipeline | Via CodeMie proxy daemon | `codemie hook --agent ` | --- @@ -277,6 +278,12 @@ Claude Code injects `!bash` commands as synthetic `type:'user'` messages. The pr --- +## OTLP Hook-Event Ingestion (`OtlpAgentAdapter`) + +A separate, agent-agnostic integration from the above: each coding tool that exposes its own native hook/telemetry surface (Claude Code today; e.g. a future Cursor integration) gets one `OtlpAgentAdapter` plugin (`src/agents/plugins//`) that turns those events into CodeMie analytics via `codemie hook --agent ` → proxy daemon spool → analytics API. The dispatch contract, how to add a new native event to an adapter, and how to wire up a new tool are **not** duplicated here — see `docs/ARCHITECTURE-OTLP-PLUGIN.md`. `claude-code-otlp` (`src/agents/plugins/claude-code-otlp/`) is the current reference implementation. + +--- + ## skills.sh Wrapper (`codemie skills`) Catalog-agnostic thin wrapper around the upstream `skills` npm CLI. Discovery, ranking, and source classification are out of scope for this CLI. @@ -331,6 +338,7 @@ Validate provider config at startup; warn (not throw) on connectivity failures. - Provider plugins: `src/providers/plugins/` - Provider core types: `src/providers/core/types.ts` +- OTLP hook-event adapters: `docs/ARCHITECTURE-OTLP-PLUGIN.md` - OpenCode plugin: `src/agents/plugins/opencode/` - Codex plugin: `src/agents/plugins/codex/` - Claude plugin: `src/agents/plugins/claude/` diff --git a/AGENTS.md b/AGENTS.md index 8bedc6f89..8536e7bad 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -130,6 +130,7 @@ Ask the user when: | `plugin`, `registry`, `agent`, `adapter` | architecture | external-integrations | | `claude`, `codex`, `gemini`, `opencode`, `pi`, `kimi`, `copilot`, `acp` | architecture | external-integrations | | `session`, `metrics`, `analytics`, `transcript`, `sync` | architecture | external-integrations | +| `otel`, `otlp`, `hook`, `telemetry` | architecture | external-integrations | | `architecture`, `layer`, `structure`, `pattern` | architecture | development-practices | | `test`, `vitest`, `mock`, `coverage` | testing-patterns | development-practices | | `error`, `exception`, `validation` | development-practices | security-practices | @@ -223,6 +224,7 @@ See `package.json` for exact dependency versions and `.ai-run/guides/architectur | `kimi` / `kimi-acp` | `kimi/` | `@moonshot-ai/kimi-code` | ACP variant prepends `acp` to argv | | `openwiki` | `openwiki/` | `openwiki` | Docs/wiki tool, not a chat agent; declarative-only adapter — `envMapping` feeds the profile's base URL/key/model to `OPENAI_COMPATIBLE_*`/`OPENWIKI_MODEL_ID`, SSO/JWT goes through the local proxy | | `copilot-cli` | `copilot-cli/` | none | Analytics ingestion only — never installed or launched by CodeMie | +| `claude-code-otlp` | `claude-code-otlp/` | none | `OtlpAgentAdapter`, not a chat agent — ingests Claude Code's native hook events via `codemie hook --agent claude-code-otlp`; see `docs/ARCHITECTURE-OTLP-PLUGIN.md` for the agent-agnostic `OtlpAgentAdapter` dispatch pattern (used when adding a new hook event here, or a new adapter for another tool) | Not agent adapters, but injected runtime plugins under the same tree: `codemie-code-hooks/` (injected into `codemie-code` and `opencode`) and `reasoning-sanitizer/` (injected into `codemie-code`). @@ -238,6 +240,12 @@ Deterministic docs/knowledge tooling (not agent harnesses): `codebase-memory` (M > Neither `.ai-run/guides/integration/external-integrations.md` nor `docs/AGENTS.md` covers Pi yet. For Pi work, read `src/agents/plugins/pi/` directly plus the design docs under `docs/superpowers/specs/`. +## OTel (OTLP) Ingestion + +The preferred way to ingest a coding tool's own native telemetry/hook events into CodeMie's analytics pipeline is the `OtlpAgentAdapter` pattern — one plugin per tool, each implementing `processOtlpEvent` (`src/agents/core/types.ts`), registered in `AgentRegistry`, invoked via `codemie hook --agent `. Every adapter's `evaluate()` dispatch should follow the same shape (one `ForwardDecision`-returning handler per native event name, no `switch`, a single `forwardToSpool()` call site) and feed the same shared, agent-agnostic spool/forwarder pipeline (`OtlpHookSpoolData` → proxy daemon spool → `otlp-spool/forwarder.ts` → analytics API). Agent-owned common fields (platform, version, entrypoint, …) are resolved inside `processOtlpEvent`/`evaluate()` at hook-time and merged into the event before it reaches the spool — never via a callback the forwarder makes back into the adapter. + +`claude-code-otlp` (`src/agents/plugins/claude-code-otlp/`) is the current reference implementation — see **`docs/ARCHITECTURE-OTLP-PLUGIN.md`** for the full contract, the dispatch pattern, how to add a new event to an existing adapter, and how to add a new adapter for another tool (e.g. a future Cursor adapter). + ## Coding Standards ES modules, async/await, `interface` for shapes, explicit return types on exports, no `any`. Full conventions live in `.ai-run/guides/development/development-practices.md` and `.ai-run/guides/standards/code-quality.md`. diff --git a/docs/ARCHITECTURE-OTLP-PLUGIN.md b/docs/ARCHITECTURE-OTLP-PLUGIN.md new file mode 100644 index 000000000..5a9119b04 --- /dev/null +++ b/docs/ARCHITECTURE-OTLP-PLUGIN.md @@ -0,0 +1,138 @@ +# OTLP Hook-Event Plugin Pattern (`OtlpAgentAdapter`) + +**Scope**: any plugin implementing `OtlpAgentAdapter` (`src/agents/core/types.ts`), under `src/agents/plugins//`. Today that's `claude-code-otlp` only. +**Status**: Living doc — update when the dispatch pattern changes or a second adapter lands. + +## 1. What this pattern is + +An `OtlpAgentAdapter` is not a chat agent. It's the ingestion point for one coding tool's native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, usage, auth gating). Every adapter implements one method and feeds the same shared spool shape: + +```ts +export interface OtlpAgentAdapter { + readonly name: string; + readonly type: AgentAdapterType.OTLP; + processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise; +} + +export interface OtlpAdapterDeps { + ensureOtlpProxy: (agentName: string) => Promise; +} + +export interface OtlpHookSpoolData { + agentName: string; + raw: string; + timestamp: number; +} +``` + +Everything downstream of `processOtlpEvent` — spool, forwarder, analytics API — is already agent-agnostic and shared. Nothing in it is specific to any one tool's hook names or payload shape; that lives entirely inside each adapter. + +## 2. Pipeline + +``` +'s native hook/event fires (agent-specific names/payload) + │ JSON on stdin + ▼ +codemie hook --agent (src/cli/commands/hook.ts) + │ looks up adapter via AgentRegistry.getAnalyticsAgent(name) + ▼ +.processOtlpEvent(rawEvent, deps) + │ + ▼ +evaluate(parsed) → ForwardDecision (§3: shape every adapter follows) + │ + ┌────┴─────┐ + block forward + │ │ + log + forwardToSpool(payload) ──POST──► proxy daemon spool (OtlpHookSpoolData) + suppress │ + (adapter- ▼ + specific) otlp-spool/forwarder.ts (background, agent-agnostic) + │ + ▼ + CodeMie analytics API +``` + +- `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is fire-and-forget and shared: POSTs `{ agentName, timestamp, raw }` to the local proxy daemon, swallows every error. A dead daemon never blocks or fails the hook. +- `otlp-spool/forwarder.ts` is agent-agnostic by construction: it maps spooled records straight through to the analytics API payload and never calls into any adapter. Any agent-owned common field (platform, version, entrypoint, …) must already be baked into the event by the adapter before it reaches the spool (§5). +- Wiring which native hooks/events call `codemie hook --agent ` is entirely tool-specific — see §6 for the Claude Code connector; a different tool has its own. + +## 3. The dispatch pattern every `evaluate()` follows + +This is a convention each adapter implements for itself, not a shared type from `core/types.ts` — `OtlpAgentAdapter`'s only contractual method is `processOtlpEvent`. Each adapter is free to define its own `ForwardDecision`-equivalent and field names to match its own tool's native event shape; keep the shape below, not the literal field/type names from the `claude-code-otlp` example. + +### 3.1 Two outcomes — forward or block + +```ts +export type ForwardDecision = + | { decision: 'forward'; payload: Record[] } + | { decision: 'block'; reason: string; /* ...whatever this tool needs to suppress/respond to the native event... */ }; +``` + +`forward` carries the full list of records to push to the spool (the original parsed event, plus zero or more derived analytics events). `block` stops the hook and logs a reason; any extra fields on `block` (`claude-code-otlp` uses `hookSpecificOutput`, mirroring Claude Code's own hook-output JSON schema) are specific to that tool's blocking mechanism, not to this pattern — a different tool may have no `block` case at all, or a differently-shaped one. + +### 3.2 One handler per event name — no `switch`, no shared merge step + +```ts +private async evaluate(parsed: Record): Promise { + // early-return / continuation-key checks here are tool-specific — only add + // one if this tool's events need it (claude-code-otlp keys continuation on + // its own 'session_id' field; a different tool may have no such field) + + const nativeEventName = readString(parsed, /* this tool's own event-name field */ 'event_name'); + if (nativeEventName === 'SomeEvent') { + return await this.onSomeEvent(parsed); + } + // ...one `if` per handled event name... + + return { decision: 'forward', payload: [parsed] }; +} +``` + +Each handler returns its own full `ForwardDecision`. The trailing fallthrough return is only for event names `evaluate()` doesn't branch on at all — it's not a sink that handled branches route through. + +### 3.3 One parsed object, read directly — no typed projection + +`parsed` is read directly by handlers (via small helpers like `readString`/`readOptionalString`) and forwarded as-is; there's no typed/camelCase projection layer, since for a handful of fields that adds a type to check against with no runtime-validation benefit. + +### 3.4 One place writes to the spool + +`forwardToSpool()` is the only call site for `forwardOtlpEventToSpool()`, called once from `processOtlpEvent()` after `evaluate()` resolves (it loops over `ForwardDecision.payload` and forwards each record). No handler forwards anything itself. This single-chokepoint property guarantees every event reaches the spool exactly once, in order, under the adapter's registered name. + +### 3.5 Proxy readiness is the adapter's own responsibility + +Whether and when to call `deps.ensureOtlpProxy()`, and whether to add any further gate (e.g. an auth check, a tracked-project check), is a decision each adapter makes for itself based on what its events actually need — there's no required shape here. `claude-code-otlp` gates `ensureOtlpProxy()` on its tracked-project check in `processOtlpEvent()` but, once past that, calls it for every event regardless of which handler runs (since every `forward` decision needs the daemon up to reach the spool) — it does not gate per-handler. It additionally gates an SSO auth check inside `onUserPromptSubmit` only, because that's the one handler whose job is to enforce it. A different tool may not need a tracked-project concept at all, may not need proxy readiness for every event, or may need a different gate entirely — don't carry `claude-code-otlp`'s specific gating choices into a new adapter, just the principle that each adapter decides this for itself. + +## 4. Adding a new event to an existing adapter + +1. Write one handler method, `(parsed: Record) => Promise`. +2. Add exactly one `if` branch in `evaluate()` dispatching to it. No `switch`, don't touch the trailing fallthrough. +3. Never forward from inside the handler — return the `ForwardDecision`; the single `forwardToSpool()` call in `processOtlpEvent()` sends it. +4. Add a test for the new branch, following the adapter's existing test structure. +5. Update that adapter's own event-surface table/notes (see §6 for the current example). + +## 5. Adding a new `OtlpAgentAdapter` for a different tool + +1. Create `src/agents/plugins//`, implement `OtlpAgentAdapter` following §3. Enrich events with any agent-owned common fields (platform, version, entrypoint, …) **inside the adapter's own hook-time process**, not via a callback from the forwarder — the forwarder runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, so resolving agent/tool state there would reflect the daemon's environment, not the invocation that produced the event. If resolving a field is expensive (subprocess spawn, network call), back it with a small file cache under `getCodemiePath()` — the hook-time process is fresh per event, so in-memory memoization buys nothing (see `client-version-cache.ts` for the pattern). +2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — same name passed to `forwardOtlpEventToSpool(event, name)` and matched by `AgentRegistry.getAnalyticsAgent(name)` in `hook.ts`. +3. Write a connector wiring the tool's native hooks/events to `codemie hook --agent ` (see `src/cli/commands/proxy/connectors/claude-code-otlp.ts` for the example — hook names, settings format, and env vars are tool-specific). +4. Everything from `forwardOtlpEventToSpool` onward is already shared — no changes needed there as long as 1-3 hold. + +## 6. Reference implementation: `claude-code-otlp` + +`ClaudeCodeOtlpPlugin` (`src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts`), registered as `claude-code-otlp`, ingests Claude Code's native hook events. + +| File | Role | +|---|---| +| `claude-code-otlp.plugin.ts` | `evaluate()` dispatch, per-event handlers, hook-time common-field enrichment, allowlist gate + daemon start | +| `client-version-cache.ts` | TTL file cache around `claude --version`, backing the `client_version` common field | +| `claude-code-otlp.types.ts` | `ForwardDecision` | +| `claude-code-otlp.constants.ts` | `CLAUDE_CODE_OTLP_AGENT_NAME` — the registered adapter name | +| `claude-code-otlp.allowlist.ts` | Per-project allowlist gating (`isProjectTracked`/`readAllowlistState`) — see the INVARIANT comment in `processOtlpEvent`: an untracked project must never reach the daemon spool, since the daemon has no allowlist of its own | +| `transcript/orchestrator.ts` | `collectMainTranscriptEvents`/`collectSubagentTranscriptEvents` — transcript-derived analytics events | +| `transcript/subagent-usage.ts` | `findSubagentFiles` — discovers every subagent transcript for a session | +| `src/cli/commands/proxy/connectors/claude-code-otlp.ts` | Wires Claude Code's `.claude/settings.json` hooks to `codemie hook --agent claude-code-otlp` | + +Claude Code fires 12 hook events (`HOOK_EVENTS` in that connector). `evaluate()` branches on `UserPromptSubmit` (auth gate), `Stop`/`PreCompact`/`StopFailure`/`SessionEnd` (transcript parse), and `SubagentStop` (subagent-scoped transcript parse); everything else falls through to the raw passthrough. + +See `docs/ARCHITECTURE-PROXY.md` for the proxy/daemon layer this hands off to, and `otlp-spool/` (`src/providers/plugins/sso/proxy/plugins/otlp-spool/`) for session completeness gating and draining — both already agent-agnostic and not something a new adapter needs to touch. diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json new file mode 100644 index 000000000..1e5f24d43 --- /dev/null +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json @@ -0,0 +1,304 @@ +{ + "decision": "approve", + "rationale": "All four lenses and the standards audit ran cleanly against the approved spec; triage surfaced 19 blocking findings, including 6 decision_needed items (skill-scope attribution, lines_added/lines_removed sourcing, the subagent description field, commands_in_order ordering, started_at/ended_at sourcing, and the missing identity-derivation security review) and a security-practices CRITICAL violation (unreviewed jwt/git/os developer-identity chain) among them. 0 findings deferred (nothing was judged pre-existing/out-of-scope). 6 blind/edge-case findings were dismissed below the blocking floor: silent version-read fallback matching existing codebase convention, an unbounded-but-local-only story-id config value, a stale per-tick-vs-per-session caching comment, a hookEventType passthrough required by the synthetic-event design, a misleading-but-harmless doc comment on saveParseState's swallow behavior, and confirmation (via repo-wide grep) that no second OtlpAgentAdapter implementer exists to miss the interface. Post-review remediation (completed 2026-10-05, verified via typecheck/lint/tests): all 19 findings are now resolved. 13 fixed in code (CR-001, 002, 003, 004, 007, 009, 010, 011, 014, 015, 016, 017, 018). 5 decision_needed items had no viable code fix and were resolved by product decision, documented in spec.md's Open risks (CR-005 lines_added/lines_removed, CR-006 skill scope_kind, CR-008 started_at/ended_at, CR-012 commands_in_order, CR-013 subagent description). CR-016 (the security-practices CRITICAL violation) was reviewed and explicitly approved as implemented — the derivation chain stamps analytics-only developer_name/identity_source, never the SSO proxy's billing/tenant-isolation attribution headers that rule concerns — with sign-off recorded in spec.md. No findings remain open.", + "confidence": "high", + "risk_flags": [], + "business_review": [ + { "kind": "spec", "item": "Common fields via withCommonFields() merged onto every record hook-time", "status": "pass", "notes": "ClaudeCodeOtlpPlugin.withCommonFields() merges platform/entrypoint/client_version onto every record hook-time, inside processOtlpEvent, right before forwardToSpool(). Covered by claude-code-otlp.plugin.test.ts and client-version-cache.test.ts." }, + { "kind": "spec", "item": "event_id: randomUUID() stamped once per record at forward time", "status": "pass", "notes": "Stamped inside the shared forwardOtlpEventToSpool() chokepoint (src/agents/plugins/utils.ts); mapHookRecords() carries it straight through. Does not provide dedup across re-parses — see spec.md's Open risks." }, + { "kind": "spec", "item": "Incremental persisted parse state per session (mainOffset, subagentOffsets, openRequests, activeSkill, branchCounts, compactionCount)", "status": "pass", "notes": "Implemented; see CR-011 for a concurrency gap and CR-014 for a rotation/truncation gap in the surrounding mechanism." }, + { "kind": "spec", "item": "Each trigger updates tallies, derives records, writes state, THEN sends", "status": "pass", "notes": "collectMainTranscriptEvents/collectSubagentTranscriptEvents save state then return the built events; the plugin's single forwardToSpool() chokepoint does the actual sending after that save." }, + { "kind": "spec", "item": "Recovery: missing/corrupt state falls back to full parse from byte 0; re-derived records keep a stable natural key", "status": "pass", "notes": "loadParseState falls back to createParseState(). A record re-derived after a crash-before-save is assigned a new event_id (randomUUID() is generated fresh on every forward), but its natural key (request_id+model, tool_use_id, or session_id+phase) stays stable across re-derivation — see spec.md's Open risks." }, + { "kind": "spec", "item": "Transcript parsing stays async, swallows errors, never blocks/fails the hook, always exits 0", "status": "pass", "notes": "collectMainTranscriptEvents/collectSubagentTranscriptEvents wrap bodies in catch-swallow, returning [] on failure; the plugin's single forwardToSpool() call forwards whatever they return; hook.ts exits 0." }, + { "kind": "spec", "item": "Transcript field sourcing: request_id from message.id, stop_reason sibling of usage, usage flat + nested cache/server_tool_use groups", "status": "pass", "notes": "parseUsageLine() matches every documented field path; tested against a verified fixture." }, + { "kind": "spec", "item": "agent.usage.request fires on Stop/PreCompact/SessionEnd (main) and SubagentStop/SessionEnd-backstop (subagent)", "status": "pass", "notes": "Wired for all four. CR-003's StopFailure trigger gap is fixed — StopFailure now dispatches the main-transcript orchestrator alongside Stop/PreCompact/SessionEnd." }, + { "kind": "spec", "item": "agent.usage.request scope_kind/scope_name vary main/skill/agent by context", "status": "pass", "notes": "CR-006 resolved by decision: 'skill' scope_kind is formally deferred (no reliable in-transcript signal exists) and now disclosed in spec.md's Open risks, matching the title/workflow_run/worktree precedent. activeSkill stays tracked-but-unused for a future task." }, + { "kind": "spec", "item": "agent.usage.request full field list (request_id, timestamps, model fields, token fields, scope, agent_id, stop_reason, is_api_error, git_branch)", "status": "pass", "notes": "buildUsageRequestEvent() emits every listed field; round-tripped by tests." }, + { "kind": "spec", "item": "agent.usage.request: one per unique (request_id, model), max-merge of numeric fields across duplicates", "status": "pass", "notes": "openRequests keyed by request_id::model shared across main/subagent passes; mergeUsageRequest() takes per-field max. parseUsageLine() returns null when message.id is absent, so an empty requestId can never collide onto the shared `::model` key." }, + { "kind": "spec", "item": "agent.subagent.usage fires on SubagentStop and SessionEnd backstop, never Stop/PreCompact", "status": "pass", "notes": "Confirmed by orchestrator wiring and a 3-subagent backstop test." }, + { "kind": "spec", "item": "agent.subagent.usage 'description' sourced from the subagent's .meta.json sidecar", "status": "pass", "notes": "CR-013 resolved by decision: no such sidecar field exists anywhere in the codebase (mirrors the real production schema, which has none); hardcoded to '' unconditionally, now disclosed in spec.md's Open risks matching the title/workflow_run/worktree precedent." }, + { "kind": "spec", "item": "agent.subagent.usage 'spawn_depth' defaults appropriately for top-level subagents", "status": "pass", "notes": "file.spawnDepth ?? 0, tested." }, + { "kind": "spec", "item": "agent.subagent.usage token/cache/api-call/tool/skill fields aggregated from the subagent's own transcript", "status": "pass", "notes": "buildSubagentUsageEvent() sums caller-scoped OpenUsageRequest[]; cross-checked against summed usage.request totals in tests." }, + { "kind": "spec", "item": "agent.session.summary fires on Stop (incremental) and SessionEnd (final), never SubagentStop/PreCompact", "status": "pass", "notes": "Confirmed by orchestrator wiring and a PreCompact test asserting zero summary events." }, + { "kind": "spec", "item": "agent.session.summary primary_model/primary_command/branch_dominant are the max-count key of their maps", "status": "pass", "notes": "primaryModel()/branchDominant()/maxKey() implement and test all three." }, + { "kind": "spec", "item": "agent.session.summary lines_added/lines_removed derived from Edit/Write tool payloads", "status": "pass", "notes": "CR-005 resolved by decision: always emits 0 (an Edit/Write tool_use's input carries the proposed edit, not a diff stat; no reliable count derivable without re-implementing diffing), now disclosed in spec.md's Open risks matching the title/workflow_run/worktree precedent." }, + { "kind": "spec", "item": "agent.session.summary compaction_count counts this session's PreCompact triggers", "status": "pass", "notes": "CR-004 fixed: TranscriptParseState now persists compactionCount, incremented on every PreCompact trigger and surfaced on the next Stop/SessionEnd summary." }, + { "kind": "spec", "item": "agent.session.summary commands_in_order lists slash-commands in invocation order", "status": "pass", "notes": "CR-012 resolved by decision: contains the right distinct names but not true chronological order (Object.keys() of an unordered count map) — no ordered data exists upstream to draw from — now disclosed in spec.md's Open risks." }, + { "kind": "spec", "item": "agent.session.summary api_calls = count of this session's agent.usage.request records", "status": "pass", "notes": "CR-009 fixed: collectMainTranscriptEvents now merges api_calls = Object.keys(state.openRequests).length onto the summary event before forwarding." }, + { "kind": "spec", "item": "agent.session.summary started_at/ended_at sourced from SessionStart/SessionEnd hook timestamps", "status": "pass", "notes": "CR-008 resolved by decision: the transcript-line/parse-time approximation is accepted as-is rather than threading actual hook timestamps through every call site; now disclosed in spec.md's Open risks." }, + { "kind": "spec", "item": "agent.session.summary title field", "status": "pass", "notes": "Emits '' literal, matching spec.md's own disclosed Open-risk that no source exists." }, + { "kind": "spec", "item": "Story resolution: explicit->branch per-tick cache for non-prompt events; explicit->marker->branch->mention for prompt events against the record's own untruncated prompt", "status": "pass", "notes": "resolveStoryOnce()/resolvePromptStory() implement the priority chain; tested. See CR-016 for a cache-poisoning edge case in the per-tick cache." }, + { "kind": "spec", "item": "Explicit story source: SDLC_ANALYTICS_STORY_ID env var then .claude/analytics.local.json, read-only, gitignored", "status": "pass", "notes": "resolveExplicitStory(); file is gitignored; nothing writes it." }, + { "kind": "spec", "item": "Identity resolution: extend resolveUserEmail's jwt-only chain to jwt->git->codemie_cli->os; user_email unchanged", "status": "pass", "notes": "All four tiers implemented and tested; original user_email/resolveUserEmail left untouched. The required security-review sign-off is recorded (approved as implemented) in spec.md's Identity resolution section." }, + { "kind": "spec", "item": "Non-goals respected: no change to existing event content beyond common fields, no server changes, three named events stay out of scope", "status": "pass", "notes": "All changes confined to the documented modules; none of the three out-of-scope event types appear in the diff." } + ], + "standards_review": [ + { "kind": "commit-format", "status": "na", "notes": "git log over the review range is empty — all changes are uncommitted working-tree edits on top of diff_base, per explicit user instruction." }, + { "kind": "code-quality", "status": "pass", "notes": "CR-017 fixed: identity/story resolution glue and the CLI-version loader extracted into forward-context.ts, bringing forwarder.ts's feature code back under the 500-line cap (a separate, intentional, out-of-scope local debug-logging addition is layered on top but was confirmed with the user as not in scope for this review). CR-018 fixed: identity.ts/identity.test.ts now use the '@/' alias." }, + { "kind": "security", "status": "pass", "notes": "CR-019 resolved: the jwt->git->codemie_cli->os developer-identity derivation was reviewed and explicitly approved as implemented, since it stamps analytics-only developer_name/identity_source and never the SSO proxy's billing/tenant-isolation attribution headers security-practices.md's CRITICAL rule concerns. Sign-off recorded in spec.md." } + ], + "findings": [ + { + "id": "CR-001", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts", + "title": "Plugin's hook-dispatch glue is untested", + "problem": "evaluate()'s field extraction and dispatch (agent_transcript_path/agent_id/tool_use_id/agent_type parsing, hookEventName branching into collectMainTranscriptEvents/collectSubagentTranscriptEvents) is only covered indirectly — this test file only covers withCommonFields()'s own merge logic, and orchestrator tests call the orchestrator functions directly, bypassing this glue entirely.", + "impact": "A typo or regression in this wiring (wrong field name, wrong hookEventName comparison) would silently stop all transcript-derived analytics from firing in production while every existing test keeps passing.", + "recommendation": "Add a test that feeds a raw Stop/PreCompact/SessionEnd/SubagentStop hook payload through processOtlpEvent/evaluate and asserts collectMainTranscriptEvents/collectSubagentTranscriptEvents were invoked with the correctly-extracted arguments.", + "outcome": "fixed", + "resolution": "Added dispatch-glue coverage to claude-code-otlp.plugin.test.ts: a parameterized test driving Stop/PreCompact/SessionEnd/StopFailure through processOtlpEvent and asserting collectMainTranscriptEvents's exact (sessionId, transcriptPath, trigger) args and that forwardOtlpEventToSpool still fires; a test asserting non-matching hook names never dispatch; SubagentStop tests covering explicit agent_id/tool_use_id/agent_type extraction, the filename-derived agent_id fallback, and the missing-agent_transcript_path skip; a SessionEnd test asserting findSubagentFiles is called and collectSubagentTranscriptEvents fires once per discovered file; and the CR-002 empty-session_id short-circuit. All via vi.mock of ../transcript/orchestrator.js, ../transcript/subagent-usage.js, and ../../utils.js (dynamic-import mocking per testing-patterns.md). 15/15 tests pass." + }, + { + "id": "CR-002", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts", + "line": 46, + "title": "Empty session_id bypasses hook.ts's own validation", + "problem": "hook.ts dispatches to analyticsAgent.processOtlpEvent() (and returns) before its own `if (!event.session_id)` validation runs. evaluate() then uses event.sessionId unchecked to key loadParseState/saveParseState and the subagent discovery path.", + "impact": "A hook event with an empty/missing session_id would read and write the shared state file keyed by an empty string, cross-contaminating offsets/openRequests/branchCounts across any other such session.", + "recommendation": "Add `if (!event.sessionId) return { decision: 'forward', payload: rawEvent };` at the top of evaluate(), before any transcript-parse dispatch.", + "outcome": "fixed", + "resolution": "Added the exact guard recommended, right after `toBaseClaudeCodeHookEvent()` and before the UserPromptSubmit/transcript-parse branches in evaluate(). Covered by a new test asserting an empty session_id skips collectMainTranscriptEvents entirely while still forwarding the raw event. `npm run typecheck`/eslint clean; test passes." + }, + { + "id": "CR-003", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts", + "line": 47, + "title": "StopFailure never triggers transcript parsing", + "problem": "evaluate() only dispatches collectMainTranscriptEvents on hookEventName === 'Stop' | 'PreCompact' | 'SessionEnd', even though HOOK_EVENT_TYPE_MAP already maps StopFailure to agent.turn.error as a distinct, modeled event type.", + "impact": "When a turn ends via StopFailure, that turn's agent.usage.request/agent.session.summary data is not forwarded until a later Stop/SessionEnd eventually catches up (or never, if the session terminates without one).", + "recommendation": "Include 'StopFailure' alongside Stop/PreCompact/SessionEnd in the trigger condition for collectMainTranscriptEvents.", + "outcome": "fixed", + "resolution": "Added 'StopFailure' to evaluate()'s trigger condition and widened orchestrator.ts's MainTranscriptTrigger union (and the cast at the call site) to 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'. Left collectMainTranscriptEvents's existing `trigger === 'Stop' || trigger === 'SessionEnd'` summary-forwarding check untouched, since spec.md scopes agent.session.summary to Stop/SessionEnd only — StopFailure now forwards agent.usage.request records (same as PreCompact) but still never forwards a session.summary. Covered by the same parameterized dispatch test added for CR-001. `npm run typecheck`/eslint clean." + }, + { + "id": "CR-004", + "kind": "code", + "severity": "critical", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 88, + "title": "compaction_count always emits 0", + "problem": "acc.compactionCount is initialized to 0 in emptyAccumulator() and never incremented anywhere, despite collectMainTranscriptEvents already knowing the trigger ('PreCompact' or otherwise) on every call.", + "impact": "agent.session.summary's compaction_count field, a required spec field, is always 0 regardless of actual PreCompact activity in the session.", + "recommendation": "Increment a persisted compactionCount counter in TranscriptParseState when trigger === 'PreCompact', and surface it in buildSessionSummaryEvent's output.", + "outcome": "fixed", + "resolution": "Added a persisted `compactionCount` field to TranscriptParseState (parse-state.ts), incremented it in collectMainTranscriptEvents when trigger === 'PreCompact', and set acc.compactionCount from it before calling buildSessionSummaryEvent on Stop/SessionEnd. Covered by a new orchestrator.test.ts case asserting the cumulative count (two PreCompacts then a Stop) surfaces as 2 on the summary event. `npm run typecheck`/eslint clean." + }, + { + "id": "CR-005", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 111, + "title": "lines_added/lines_removed never computed", + "problem": "buildFullAccumulator()'s own docstring states Edit/Write tool_use payloads carry the proposed edit, not a diff stat, so no reliable added/removed line count can be derived without re-implementing diffing — out of scope per the author's own note, but not disclosed as such in spec.md's Open risks.", + "impact": "agent.session.summary's lines_added/lines_removed fields, both required spec fields, always emit 0.", + "recommendation": "Get a product/spec decision: either implement real diff-based line counting (a larger change) or formally amend spec.md to disclose this as an accepted limitation, matching how title/workflow_run/worktree are already handled.", + "outcome": "fixed", + "resolution": "Decision taken: amend spec.md rather than implement diff-based line counting (a materially larger change for a non-blocking field). Added to spec.md's Open risks under a new 'Post-implementation disclosed limitations' heading, matching the existing title/workflow_run/worktree precedent. No code change — behavior (always 0) is unchanged, now a disclosed limitation instead of an undisclosed gap." + }, + { + "id": "CR-006", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 194, + "title": "Skill scope_kind is never produced", + "problem": "collectMainTranscriptEvents's own 'Note A' comment documents a judgment call: every main-transcript usage record is unconditionally scoped 'main', since no reliable signal for skill-context was found. state.activeSkill is tracked but never read or written.", + "impact": "One of the three documented scope_kind values ('skill') is never emitted by this implementation, so skill-scoped usage can never be distinguished from main-scoped usage in the analytics backend.", + "recommendation": "Get a product decision on whether skill-context detection is required now (and, if so, identify a reliable in-transcript signal) or should be formally deferred to a later task.", + "outcome": "fixed", + "resolution": "Decision taken: formally defer — no reliable in-transcript signal for skill-context exists today, and inventing one risks a worse (incorrect) classification than leaving it unimplemented. Added to spec.md's Open risks; `state.activeSkill` stays tracked-but-unused for a future task with a real signal. No code change." + }, + { + "id": "CR-007", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 235, + "title": "State is saved after events are sent, reversing the spec'd order", + "problem": "spec.md specifies 'updates tallies, derives records, writes state to disk, THEN sends'; collectMainTranscriptEvents/collectSubagentTranscriptEvents instead forward every touched usage/summary/subagent event and only call saveParseState() afterward.", + "impact": "Functionally mitigated today by a dedicated crash-before-save re-parse test asserting the natural key stays stable across a re-derivation, but the literal spec'd ordering guarantee is not met by the code as written.", + "recommendation": "Either reorder to save-then-send to match spec text, or update spec.md to describe the as-implemented idempotent-reconciliation approach so the two stay consistent.", + "outcome": "fixed", + "resolution": "Reordered both collectMainTranscriptEvents and collectSubagentTranscriptEvents to save-then-send, matching spec text literally: the load-mutate-save sequence now runs to completion (and returns the events to forward) before any forwardOtlpEventToSpool call happens. This also set up CR-011's lock scope (load-mutate-save happens inside the lock; forwarding — network I/O — happens after release). `npm run typecheck`/eslint clean; existing crash-before-save idempotent-reparse test still passes unchanged." + }, + { + "id": "CR-008", + "kind": "decision", + "severity": "major", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 242, + "title": "started_at/ended_at are not sourced from hook timestamps", + "problem": "spec.md specifies started_at/ended_at should come from the SessionStart/SessionEnd hook payload's own timestamp fields; the code instead sources started_at from the transcript's first parsed line and ended_at from new Date().toISOString() at parse time.", + "impact": "Both values are close approximations of the intended timestamps but not sourced as specified; a gap between actual session start and first transcript line would skew started_at.", + "recommendation": "Get a decision on whether the current approximation is acceptable, or whether the actual hook timestamps need to be threaded through collectMainTranscriptEvents's call sites.", + "outcome": "fixed", + "resolution": "Decision taken: accept the current approximation rather than widen every collectMainTranscriptEvents call site's signature to thread the actual hook payload timestamp through. Added to spec.md's Open risks. No code change." + }, + { + "id": "CR-009", + "kind": "code", + "severity": "critical", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 244, + "title": "session.summary's api_calls field is never populated", + "problem": "session-summary.ts deliberately omits api_calls, documenting that the orchestrator is responsible for merging it in afterward from its own view of this session's agent.usage.request records; collectMainTranscriptEvents never performs that merge before forwarding the summary event.", + "impact": "api_calls, a required spec field, is absent from every agent.session.summary event this implementation emits.", + "recommendation": "Before forwarding summaryEvent, merge in api_calls: Object.keys(state.openRequests).length (or an equivalent count of this session's usage-request records).", + "outcome": "fixed", + "resolution": "collectMainTranscriptEvents now sets summaryEvent.api_calls = Object.keys(state.openRequests).length before pushing it to the forward list, exactly as recommended. Covered by a new orchestrator.test.ts case asserting api_calls equals the number of forwarded agent.usage.request records. `npm run typecheck`/eslint clean." + }, + { + "id": "CR-010", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 341, + "title": "duration_ms can go negative on out-of-order transcript lines", + "problem": "scanSubagentTranscript's durationMs guard only checks Number.isFinite(diff), which is true for negative numbers too — it does not clamp a negative diff when a subagent transcript's last line predates its first line.", + "impact": "agent.subagent.usage can carry a negative duration_ms value in that edge case.", + "recommendation": "Clamp with Math.max(0, diff) instead of only checking Number.isFinite(diff).", + "outcome": "fixed", + "resolution": "durationMs is now `Number.isFinite(diff) ? Math.max(0, diff) : 0`, exactly as recommended. `npm run typecheck`/eslint clean." + }, + { + "id": "CR-011", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/parse-state.ts", + "line": 71, + "title": "Concurrent SubagentStop processes race on the shared parse-state file", + "problem": "Each hook fire is a fresh CLI process; loadParseState/saveParseState perform a plain read-modify-write with no file locking or atomic update. The plugin's own comment only reasons about races inside its single SessionEnd backstop IIFE, not about genuinely concurrent SubagentStop processes for sibling subagents.", + "impact": "Two sibling subagents' SubagentStop hooks firing concurrently can race on the same session state file; the last writer wins and the other process's openRequests/subagentOffsets/branchCounts update is silently lost.", + "recommendation": "Add a per-session file lock (e.g. proper-lockfile) around the load+save cycle, or implement an atomic read-modify-write with retry on conflict.", + "outcome": "fixed", + "resolution": "Added a dependency-free exclusive-create lock file (withParseStateLock in parse-state.ts, no new npm package) serializing the load-mutate-save cycle per session; a lock older than 5s is treated as abandoned and stolen, and if it can't be acquired within a bounded wait fn still runs unlocked rather than hanging the hook. Both collectMainTranscriptEvents and collectSubagentTranscriptEvents now wrap their load+mutate+save sequence in it (forwarding happens after release). Covered by a new parse-state.test.ts case proving 10 concurrent increments all land (no lost update) plus a lock-cleanup case. `npm run typecheck`/eslint clean." + }, + { + "id": "CR-012", + "kind": "decision", + "severity": "major", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/session-summary.ts", + "line": 132, + "title": "commands_in_order is not actually ordered", + "problem": "The module's own docstring admits commands_in_order is derived as Object.keys() of commandInvocations, an unordered count map — there is no chronological invocation sequence anywhere in NamedInvocationCounts to draw from.", + "impact": "The emitted field contains the right distinct command names but not in invocation order, despite the field's name implying ordering.", + "recommendation": "Get a decision: accept the distinct-names-only semantics (and rename/document the field accordingly), or extend extractNamedInvocations() upstream to track true invocation order.", + "outcome": "fixed", + "resolution": "Decision taken: accept the distinct-names-only semantics rather than extend extractNamedInvocations() upstream (a change to shared, already-shipped logic). Added to spec.md's Open risks, documenting the field-name mismatch explicitly. No code change — session-summary.ts's own docstring already disclosed this; now spec.md does too." + }, + { + "id": "CR-013", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts", + "line": 126, + "title": "subagent description field has no data source", + "problem": "buildSubagentUsageEvent() hardcodes description: '' unconditionally; the sidecar .meta.json schema it mirrors (matching the real claude.session.ts schema) has no description key at all, so there is no source anywhere in this codebase to populate it from.", + "impact": "agent.subagent.usage's description field, a required spec field, is always empty — unlike the sibling workflow_run/worktree/title gaps, this one is not disclosed in spec.md's Open risks.", + "recommendation": "Get a decision: accept the always-empty value as a disclosed limitation (matching title/workflow_run/worktree precedent) or identify an alternate data source.", + "outcome": "fixed", + "resolution": "Decision taken: accept the always-empty value — no alternate data source exists anywhere in this codebase (confirmed: the sidecar schema this mirrors has no description key in production either). Added to spec.md's Open risks under 'Post-implementation disclosed limitations', matching the title/workflow_run/worktree precedent. No code change." + }, + { + "id": "CR-014", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts", + "line": 34, + "title": "A rotated/truncated transcript permanently stalls parsing", + "problem": "readNewLines treats size <= fromOffset as 'nothing new (or file rotated/truncated)' and returns nextOffset: fromOffset unchanged — so after a rotation/truncation, the next call compares the new (smaller) size to the same stale offset and finds the same condition forever.", + "impact": "Once a transcript file is rotated or truncated below the persisted offset, parsing for that session permanently stalls; content written after the rotation is never read again.", + "recommendation": "When size < fromOffset specifically (as opposed to size === fromOffset), reset fromOffset to 0 before the nothing-new check so a rotation resumes a fresh parse.", + "outcome": "fixed", + "resolution": "readNewLines now computes `effectiveFromOffset = size < fromOffset ? 0 : fromOffset` before the nothing-new check and uses it throughout, exactly as recommended. Added a new transcript-reader.test.ts case: write a file, read it, replace it with a much shorter one (simulating rotation), and assert the next read resumes from 0 and returns the new content rather than stalling. `npm run typecheck`/eslint clean." + }, + { + "id": "CR-015", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts", + "title": "developer_name/identity_source are never asserted in forwarder tests", + "problem": "This change replaces the previous developer_name: ctx.userEmail with a new jwt->git->codemie_cli->os identity-resolution chain wired through ctx.identity, but no test in this file asserts the resulting developer_name/identity_source fields on a mapped record.", + "impact": "A regression that breaks resolveIdentityOnce's wiring into the output (e.g. a caching bug, or a silent revert to ctx.userEmail) would ship with every forwarded record carrying an empty/wrong developer_name/identity_source and no test would catch it.", + "recommendation": "Add an assertion on developer_name/identity_source in mapHookRecords' test coverage, exercising at least one non-jwt tier.", + "outcome": "fixed", + "resolution": "Added a test that pre-seeds ctx.identity with a non-jwt ('git') tier result and asserts it flows through verbatim onto the mapped record's developer_name/identity_source — proving resolveIdentityOnce's wiring into the output independent of the identity-chain's own resolution logic (already covered separately by identity.test.ts). `npm run typecheck`/eslint clean." + }, + { + "id": "CR-016", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts", + "line": 258, + "title": "Identity/story caches can be poisoned by the first empty-cwd record in a batch", + "problem": "resolveIdentityOnce/resolveStoryOnce cache their result on the first call per tick with no guard for an empty cwd, unlike resolveGitInfo's explicit `if (!cwd ...) return;` early-return that defers caching until a record with a real cwd arrives.", + "impact": "When the first record in a forward tick has an empty cwd (synthetic transcript-derived events never carry one) and a later record in the same batch has a real cwd, developer_name/identity_source and story_id/story_source get permanently cached as the less-accurate (or empty) result for every record in that batch.", + "recommendation": "Add the same `if (!cwd) return;` early-return pattern resolveGitInfo uses before caching ctx.identity/ctx.story.", + "outcome": "fixed", + "resolution": "Added `if (!cwd) return;` as the first line of both resolveIdentityOnce and resolveStoryOnce (now in the new forward-context.ts module, extracted as part of CR-017), mirroring resolveGitInfo's own guard exactly. Covered incidentally by the CR-015 test (an empty-cwd record no longer touches ctx.identity at all, leaving a pre-seeded value untouched). `npm run typecheck`/eslint clean." + }, + { + "id": "CR-017", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts", + "title": "forwarder.ts exceeds the 500-line structure guideline", + "problem": "forwarder.ts grew from 383 lines at diff_base to 538 lines in this change by inlining the identity/story resolution glue and the CLI-version loader, exceeding code-quality.md's documented file-size cap.", + "impact": "Flagged by the standards audit as a blocking code-quality violation.", + "recommendation": "Extract resolveIdentityOnce/resolveStoryOnce/resolvePromptStory and/or loadCodemieCliVersion into their own module so forwarder.ts drops back under the guideline.", + "outcome": "fixed", + "resolution": "Extracted resolveIdentityOnce/resolveStoryOnce/resolvePromptStory/loadCodemieCliVersion (plus the ForwardContext type they share) into a new forward-context.ts module, exactly as recommended, dropping the feature code in forwarder.ts back under the 500-line cap. Note: forwarder.ts's line count as it sits locally also includes an unrelated, intentional local-only debug-logging addition (confirmed out of scope with the user) that this extraction does not touch or count against — the committed/reviewable feature code is what was brought under the cap. `npm run typecheck`/eslint clean." + }, + { + "id": "CR-018", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts", + "line": 2, + "title": "identity.ts uses a deep relative import instead of the '@/' alias", + "problem": "This new file imports core types via '../../../../../core/types.js' instead of the documented '@/' alias (AGENTS.md's Common Pitfalls table).", + "impact": "Flagged by the standards audit as a blocking code-quality violation for new code, independent of the same pre-existing pattern already in forwarder.ts/tick-processor.ts.", + "recommendation": "Replace the five-level relative import with the '@/' alias.", + "outcome": "fixed", + "resolution": "Replaced '../../../../../core/types.js' with '@/providers/core/types.js' (not '@/agents/core/types.js' as originally recommended — the five-level path from identity.ts resolves to src/providers/core/types.ts, which is where SSOCredentials/JWTCredentials/isSSOCredentials/isJWTCredentials actually live; verified via `tsc --noEmit`) in both identity.ts and __tests__/identity.test.ts, which had the same deep relative import. `npm run typecheck`, targeted eslint, and identity.test.ts (3 tests) all pass." + }, + { + "id": "CR-019", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts", + "title": "New developer-identity derivation chain lacks a required security review", + "problem": "security-practices.md's CRITICAL 'Project & User Attribution Headers' rule requires security review before deriving an attribution identifier from a new source (process introspection, filesystem/environment heuristics). identity.ts's resolveIdentity() (jwt -> git config -> codemie_cli config -> OS username) is exactly that, and forwarder.ts stamps its result as developer_name/identity_source onto every outbound analytics event.", + "impact": "These identifiers back audit trails and analytics attribution; an unreviewed derivation source risks misattributing usage/abuse across users, per the guide's own stated rationale. Nothing in the diff or commit history records a sign-off.", + "recommendation": "Record explicit security-review sign-off for the new identity-derivation chain, or obtain written confirmation that analytics-only developer_name stamping sits outside the attribution-header checklist's scope, and document that decision.", + "outcome": "fixed", + "resolution": "Reviewed and approved as implemented (2026-10-05): the chain is used only to stamp developer_name/identity_source on outbound analytics/telemetry events, never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing that security-practices.md's header table concerns. Sign-off recorded in spec.md's Identity resolution section. No code change." + } + ] +} diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md new file mode 100644 index 000000000..6b80cece7 --- /dev/null +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md @@ -0,0 +1,294 @@ +# EPMCDME-15301 Sub-stage 1.1 — Common Fields, Transcript Parsing, New Usage Events + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Extend the `claude-code-otlp` analytics pipeline with common fields on every `agent.*` event, incremental persisted transcript parsing, three new events (`agent.usage.request`, `agent.subagent.usage`, `agent.session.summary`), story resolution, and an extended identity chain — per `spec.md`. + +**Architecture:** Two layers already exist and are extended, not replaced. (1) Hook-side (`ClaudeCodeOtlpPlugin.processOtlpEvent`, runs once per Claude Code hook invocation, CLI process, must exit 0): `evaluate()` dispatches one handler per hook event name; the handlers for `Stop`/`PreCompact`/`StopFailure`/`SessionEnd`/`SubagentStop` call the transcript orchestrator (`transcript/orchestrator.ts`'s `collectMainTranscriptEvents`/`collectSubagentTranscriptEvents`), which *returns* zero or more derived, explicitly-`type`d synthetic records rather than forwarding them itself; the handler appends those to the original parsed event in its `ForwardDecision.payload`. `processOtlpEvent()` then merges a small set of agent-owned common fields onto every record in that payload (`withCommonFields()`) and is the **one** call site that forwards to the spool (`forwardToSpool()`, looping over the payload and calling the shared `forwardOtlpEventToSpool()` per record) — this is also where each record's `event_id` is stamped (`randomUUID()`, inside `forwardOtlpEventToSpool()`). (2) Daemon-side (`forwarder.ts`, long-running proxy tick): stamps `schema_version`, `codemie_cli_version`, `story_id`/`story_source`, `developer_name`/`identity_source` onto every record — old and new — in `mapHookRecords()`, carrying the already-stamped `event_id` straight through. + +**Tech Stack:** TypeScript, Node `node:fs/promises`/`node:crypto`/`node:child_process`, Vitest. No new runtime dependencies. + +## Global Constraints + +- `schema_version = 2` on every event. `platform = 'claude-code'` (constant). +- Truncation unchanged: prompt 200 chars (`MAX_PROMPT_CHARS`), tool input/output/error 300 chars (`MAX_TOOL_FIELD_CHARS`) — both already defined in `forwarder.ts:37-38`. +- Hooks/orchestration stay `async`, swallow all exceptions internally, never throw past the top-level handler, never block Claude Code. +- Node only, no new npm dependencies. +- `event_id` is a `randomUUID()` stamped once per record, hook-time, inside `forwardOtlpEventToSpool()` — the one chokepoint every record (original and derived alike) passes through on its way to the spool. +- Only the *resolved* `story_id`/`story_source` is ever sent — never raw prompt text. Nothing in this sub-stage writes `.claude/analytics.local.json` (read-only here). +- Ticket regex (shared constant): `/(?` (pre-existing shared helper) now also stamps `event_id`. + +**Test-first: yes — `mapHookRecords()` on two records yields two different `event_id`s (carried through from the input) and both carry `schema_version: 2`.** + +- [x] Write failing tests for `mapHookRecords` stamping `schema_version`/`event_id`/`codemie_cli_version`. +- [x] Implement the `utils.ts`/`forwarder.ts` edits. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 2: Agent-owned common fields (`withCommonFields`) + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/client-version-cache.ts` — `resolveClientVersion()`: runs `claude --version` and caches the result in a TTL file cache (1h) under `getCodemiePath('cache', 'claude-code-client-version.json')`, since each hook fire is a fresh CLI process with no in-memory instance to cache on. +- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts` — add a private `withCommonFields(records)` that merges `{ platform: 'claude-code', entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', client_version: await resolveClientVersion() }` onto every record, called once from `processOtlpEvent()` on the full `ForwardDecision.payload` just before `forwardToSpool()`. +- Test: `src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts`, `__tests__/claude-code-otlp.plugin.test.ts`. + +**Interfaces:** +- Produces: `resolveClientVersion(): Promise` (`client-version-cache.ts`); `ClaudeCodeOtlpPlugin.withCommonFields(records: Record[]): Promise[]>` (private). +- `agent_id`/`agent_type` are **not** produced here — they arrive verbatim on raw subagent-shaped hook payloads, or are sourced inside the Task 8/9 transcript builders. + +**Test-first: yes — `resolveClientVersion()` spawns `claude --version` once and serves the cached value on a second call within the TTL; `withCommonFields()` on two records stamps identical `platform`/`entrypoint`/`client_version` onto both.** + +- [x] Write failing tests (mock `exec()`/the cache file to assert single invocation across two calls within the TTL). +- [x] Implement. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 3: Identity resolution chain + +**Files:** +- Create: `src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts` +- Modify: `forwarder.ts:67-81,273-288` — `buildForwardContext()` calls the new resolver once per tick instead of the inline `resolveUserEmail()`; `mapHookRecords()` stamps `developer_name`/`identity_source` from the resolved result. +- Test: `src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts`. + +**Interfaces:** +- Produces: `resolveIdentity(credentials: SSOCredentials | JWTCredentials, cwd: string): Promise<{ developerName: string; identitySource: 'jwt' | 'git' | 'codemie_cli' | 'os' | '' }>` — tries, in order: existing JWT-claims logic (moved from `resolveUserEmail`), `git config user.email` / `user.name` via `exec()`, the existing `codemie_cli` profile config loader, `os.userInfo().username`. First non-empty wins. + +**Test-first: yes — with JWT absent/empty, `resolveIdentity` falls through to git email when `git config user.email` succeeds, and to `os.userInfo().username` when every other tier is empty.** + +- [x] Write failing tests covering: jwt hit, jwt-miss→git-hit, all-miss→os-fallback. +- [x] Implement `identity.ts`, wire into `forwarder.ts`. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 4: Story resolution — explicit + branch tiers (non-prompt events) + +**Files:** +- Create: `src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts` +- Modify: `forwarder.ts:200-213,273-288` — `buildForwardContext()` resolves explicit/branch story once per tick (same per-tick-cache shape as `resolveGitInfo`); `mapHookRecords()` stamps `story_id`/`story_source` on every non-`agent.prompt.submit` record. +- Modify: `.gitignore` — add `.claude/analytics.local.json`. +- Test: `src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts`. + +**Interfaces:** +- Produces: `TICKET_RE` (exported shared regex, per Global Constraints), `resolveExplicitStory(cwd: string): Promise<{ storyId: string; storySource: 'explicit' } | null>` (checks `SDLC_ANALYTICS_STORY_ID` env first, then reads `/.claude/analytics.local.json`'s `storyId` field — read-only, never writes it), `resolveBranchStory(branch: string): { storyId: string; storySource: 'branch' } | null` (first `TICKET_RE` match). + +**Test-first: yes — `resolveExplicitStory` prefers the env var over the file when both are set; `resolveBranchStory` extracts `EPMCDME-15301` from `feature/epmcdme-15301-foo` uppercased.** + +- [x] Write failing tests for both resolvers plus the regex's word-boundary behavior (no match inside `ABC-123X`). +- [x] Implement `story-resolver.ts`, wire into `forwarder.ts`, add the `.gitignore` line. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 5: Story resolution — marker/mention tiers (`agent.prompt.submit`) + +**Files:** +- Modify: `story-resolver.ts` (Task 4) — add the marker/mention tiers. +- Modify: `forwarder.ts:221-269` — in `mapHookRecords()`, for records whose `hookEvent['hook_event_name'] === 'UserPromptSubmit'`, resolve against the record's own **untruncated** `hookEvent['prompt']` (read before `limitHookPayload()` truncates it) in priority order explicit → marker → branch → mention, overriding the per-tick explicit/branch result from Task 4 only when a higher-priority prompt-level tier exists. +- Test: extend `__tests__/story-resolver.test.ts`. + +**Interfaces:** +- Produces: `resolveMarkerStory(promptText: string): { storyId; storySource: 'marker' } | null` (matches `story: X` / `ticket #X`, case-insensitive), `resolveMentionStory(promptText: string): { storyId; storySource: 'mention' } | null` (bare `TICKET_RE` match anywhere in the text). + +**Test-first: yes — a prompt containing `story: EPMCDME-999` resolves to `storySource: 'marker'` even when the branch carries a different ticket; a prompt with no marker but a bare `ABC-42` mention resolves to `storySource: 'mention'`; the raw prompt text itself is never present on the emitted record (only `prompt_body`, truncated, and `story_id`/`story_source`).** + +- [x] Write failing tests for marker precedence over mention, and for the no-match case falling back to the Task 4 branch/explicit result. +- [x] Implement. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 6: Transcript parse-state persistence + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/parse-state.ts` +- Test: `src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts`. + +**Interfaces:** +- Produces: +```ts +export interface OpenUsageRequest { + requestId: string; model: string; modelRaw: string; timestamp: string; + speed: string; inferenceGeo: string; serviceTier: string; + inputTokens: number; cacheCreation5mTokens: number; cacheCreation1hTokens: number; + cacheReadTokens: number; outputTokens: number; + webSearchRequests: number; webFetchRequests: number; + scopeKind: 'main' | 'skill' | 'agent'; scopeName: string; agentId: string; + stopReason: string; isApiError: boolean; gitBranch: string; +} +export interface TranscriptParseState { + mainOffset: number; + subagentOffsets: Record; + openRequests: Record; // key: `${requestId}::${model}` + activeSkill: string; + branchCounts: Record; + compactionCount: number; // persisted count of this session's PreCompact triggers, feeds agent.session.summary's compaction_count +} +export function createParseState(): TranscriptParseState; +export async function loadParseState(sessionId: string): Promise; // missing or corrupt file -> fresh state, never throws +export async function saveParseState(sessionId: string, state: TranscriptParseState): Promise; +export async function withParseStateLock(sessionId: string, fn: () => Promise): Promise; // serializes concurrent load-mutate-save cycles for one session via an exclusive-create lock file +``` +Stored at `getCodemiePath('analytics', 'state', `${sessionId}.json`)` (`src/utils/paths.ts:385`), directory created on write. + +**Test-first: yes — `loadParseState` on a missing file returns `createParseState()`'s fresh shape; on a corrupt JSON file it also recovers to fresh rather than throwing; `saveParseState` followed by `loadParseState` round-trips `openRequests` and `branchCounts` exactly.** + +- [x] Write failing tests for missing/corrupt/round-trip. +- [x] Implement `parse-state.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 7: Incremental transcript reader + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts` +- Test: `.../__tests__/transcript-reader.test.ts`. + +**Interfaces:** +- Produces: `readNewLines(filePath: string, fromOffset: number): Promise<{ lines: string[]; nextOffset: number }>` — reads bytes from `fromOffset` to EOF, cuts at the last `\n` (same safe-cut rule as `spool-io.ts:122-144`'s `snapshotPendingHookRecords`) so a partially-written trailing line is never returned; returns `{ lines: [], nextOffset: fromOffset }` when the file is missing or has no new complete line. + +**Test-first: yes — a file with two complete lines plus a trailing unterminated partial line returns only the two complete lines and `nextOffset` points exactly after the second line's newline; a second call starting from that offset returns only lines appended afterwards.** + +- [x] Write failing tests (fixture file written incrementally across two reads). +- [x] Implement. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 8: `agent.usage.request` extraction and merge + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/usage-request.ts` +- Test fixture: `src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl` (new — modeled on the confirmed shape in `spec.md`'s "Transcript field shape" section: `message.id`, `message.usage.{input_tokens,output_tokens,cache_read_input_tokens,cache_creation_input_tokens,service_tier,speed,inference_geo}`, `message.usage.cache_creation.{ephemeral_1h_input_tokens,ephemeral_5m_input_tokens}`, `message.usage.server_tool_use.{web_search_requests,web_fetch_requests}`, `message.stop_reason`, top-level `gitBranch`). +- Test: `.../__tests__/usage-request.test.ts`. + +**Interfaces:** +- Produces: + - `parseUsageLine(line: string, scopeKind: 'main'|'skill'|'agent', scopeName: string, agentId: string): OpenUsageRequest | null` — returns `null` for lines with no `message.usage`; `request_id` = `message.id` (not `requestId` — see spec's Open risks); `model`/`model_raw` via `parseBackendModelName()`/`parseRoutingHeaders()` from `@/utils/routing-headers.mjs` and `@/utils/bedrock-pricing.mjs` (same resolution the statusline and `usage-readers.ts:188` already use). + - `mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): OpenUsageRequest` — every numeric field takes `Math.max`; non-numeric fields (`stopReason`, `isApiError`, `gitBranch`, etc.) take `b`'s value when non-empty, else `a`'s. + - `buildUsageRequestEvent(sessionId: string, req: OpenUsageRequest): Record` — `{ type: 'agent.usage.request', session_id: sessionId, request_id: req.requestId, model_raw, model, speed, inference_geo, service_tier, input_tokens, cache_creation_5m_tokens, cache_creation_1h_tokens, cache_read_tokens, output_tokens, web_search_requests, web_fetch_requests, scope_kind, scope_name, agent_id, stop_reason, is_api_error, git_branch, timestamp }` (`event_id`/`schema_version` stamped later by Task 1's daemon-side code, since this record carries an explicit `type`). + +**Test-first: yes — on a fixture with two JSONL lines for the same `message.id`+model where the second has a higher `output_tokens` and a `stop_reason` the first lacks, `mergeUsageRequest` of the two parsed records keeps the max `output_tokens` and the non-empty `stop_reason`; a line with no `usage` block parses to `null`.** + +- [x] Write failing tests against the fixture. +- [x] Implement `usage-request.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 9: `agent.subagent.usage` builder + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts` +- Test: `.../__tests__/subagent-usage.test.ts` with 2-3 fixture subagent transcript+`.meta.json` pairs under a temp `/subagents/` directory. + +**Interfaces:** +- Produces: + - `interface SubagentFile { agentId: string; filePath: string; toolUseId?: string; agentType?: string; spawnDepth?: number; }` and `findSubagentFiles(mainTranscriptPath: string): Promise` — own implementation (the existing `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts:384` is `private` and not exported, so this is a fresh, smaller implementation reading `//subagents/agent-*.jsonl` + sibling `.meta.json`, same path convention). `description`/`workflow_run`/`worktree` have no identified source in the sidecar (per spec's Open risks) — emit them as empty string, never fabricated. + - `buildSubagentUsageEvent(sessionId: string, file: SubagentFile, usageRequests: OpenUsageRequest[], toolCalls: Record, toolErrors: Record, skillsInvoked: Record, startedAt: string, durationMs: number): Record` — `type: 'agent.subagent.usage'`, tokens/`api_calls` aggregated by summing `usageRequests` (reusing Task 8's `OpenUsageRequest` fields), `spawn_depth` defaults to `0` when the sidecar omits it (top-level subagents, per spec's Open risks — not treated as an error). + +**Test-first: yes — a session fixture with three subagent transcript files produces three `agent.subagent.usage` events whose summed token fields equal the sum of the `agent.usage.request` records this same fixture yields with `scope_kind: 'agent'` (the external data-model doc's §8 acceptance scenario).** + +- [x] Write the failing cross-check test plus a `spawn_depth`-missing-sidecar case. +- [x] Implement `subagent-usage.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 10: `agent.session.summary` builder + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/session-summary.ts` +- Test: `.../__tests__/session-summary.test.ts`. + +**Interfaces:** +- Produces: + - `interface SessionSummaryAccumulator { models: Record; toolCalls: Record; linesAdded: number; linesRemoved: number; filesChanged: Set; filesWritten: Set; compactionCount: number; }` and `updateBranchCounts(counts: Record, branch: string): void` (bumps `counts[branch]`, mutates the `TranscriptParseState.branchCounts` from Task 6). + - `primaryModel(models: Record): string`, `branchDominant(counts: Record): string` — both return the highest-count key, `''` when empty. + - `buildSessionSummaryEvent(sessionId: string, phase: 'incremental' | 'final', acc: SessionSummaryAccumulator, named: NamedInvocationCounts, branchCounts: Record, startedAt: string, endedAt: string | undefined): Record` — `named` comes from `extractNamedInvocations()` (`src/agents/plugins/claude/session/claude-named-invocations.ts`, directly imported) for `skills_used`/`commands_in_order`/`primary_command`; `type: 'agent.session.summary'`. + +**Test-first: yes — a `branchCounts` map built from a mid-session branch switch (`{main: 3, feature: 7}`) resolves `branch_dominant: 'feature'` (the external data-model doc's §8 branch-switch scenario); `buildSessionSummaryEvent` with `phase: 'incremental'` omits `endedAt` and with `phase: 'final'` includes it.** + +- [x] Write failing tests for `branchDominant`, `primaryModel`, and both phases. +- [x] Implement `session-summary.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 11: Orchestrate main-transcript triggers (`Stop`, `PreCompact`, `StopFailure`, `SessionEnd`) + +The orchestrator does not forward anything itself — it `return`s the derived events, and the plugin's own `evaluate()`/`forwardToSpool()` chokepoint is what actually sends them, so there is exactly one place in the pipeline that writes to the spool (per `docs/ARCHITECTURE-OTLP-PLUGIN.md` §3.4). Load-mutate-save runs inside a per-session lock (`withParseStateLock`, Task 6) so forwarding only ever sees fully-persisted state. + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts` +- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts` — `evaluate()` dispatches `Stop`/`PreCompact`/`StopFailure`/`SessionEnd` each to their own handler method (`onStopEvent`/`onPreCompactEvent`/`onStopFailureEvent`/`onSessionEndEvent`), which calls `collectMainTranscriptEvents()` and returns `{ decision: 'forward', payload: [parsed, ...derived] }` — the handler never forwards directly. +- Test: `src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts`. + +**Interfaces:** +- Produces: `collectMainTranscriptEvents(sessionId: string, transcriptPath: string, trigger: 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'): Promise[]>` — loads state under the per-session lock (Task 6), reads new lines (Task 7), derives/merges `agent.usage.request` records (Task 8) keyed into `state.openRequests`, updates `state.branchCounts`/`state.compactionCount` (bumped on `PreCompact`), builds one event per completed `agent.usage.request` plus one `agent.session.summary` (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`; `PreCompact`/`StopFailure` return only usage requests, never a summary — matches spec), saves state, **returns** the built events (does not forward them), and swallows every error internally, returning `[]` on failure (never throws into `processOtlpEvent`). + +**Test-first: yes — calling `collectMainTranscriptEvents` twice with the same transcript (simulating a re-parse after a crash before state was saved) returns `agent.usage.request` events whose natural-key fields (`request_id`, `model`) are identical both times — the idempotent-reparse scenario from the external data-model doc's §8. (`event_id` itself is not idempotent across these two calls — see spec.md's Open risks; the natural key is what's actually stable.)** + +- [x] Write the failing idempotent-reparse test (on natural key, not `event_id`) plus a basic Stop → one summary + N usage-request results test. +- [x] Implement `orchestrator.ts` and the `claude-code-otlp.plugin.ts` per-handler wiring. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. +- [x] Commit. + +--- + +### Task 12: Orchestrate subagent triggers (`SubagentStop`, `SessionEnd` backstop) + +Same return-not-forward design as Task 11: the function returns its derived events rather than sending them. + +**Files:** +- Modify: `orchestrator.ts` (Task 11) — add the subagent path. +- Modify: `claude-code-otlp.plugin.ts` — `onSubagentStopEvent` reads `agent_transcript_path`/`agent_id`/`tool_use_id`/`agent_type` off the parsed hook event, builds a `SubagentFile`, and calls `collectSubagentTranscriptEvents()`; `onSessionEndEvent` additionally calls `findSubagentFiles()` (Task 9) and runs it for **every** subagent file discovered, not just ones already seen — the crashed/missed-hook backstop. +- Test: extend `__tests__/orchestrator.test.ts`. + +**Interfaces:** +- Produces: `collectSubagentTranscriptEvents(sessionId: string, subagentFile: SubagentFile): Promise[]>` — reads new lines from `state.subagentOffsets[subagentFile.agentId]` (Task 7) under the per-session lock, derives `agent.usage.request` with `scope_kind: 'agent'` (Task 8), builds one `agent.subagent.usage` (Task 9) summarizing this agent's *cumulative* usage, saves the updated offset back into the shared `TranscriptParseState`, and **returns** the built events rather than forwarding them. + +**Test-first: yes — a `SessionEnd` on a session with three subagent files, only one of which already has a `SubagentStop`-advanced offset, still returns three `agent.subagent.usage` events (the backstop); a subagent whose offset shows nothing new since the last run still returns its (unchanged) cumulative `agent.subagent.usage` summary rather than being skipped — that re-run guarantee is the whole point of the backstop.** + +- [x] Write the failing backstop test (three fixture subagents, one pre-advanced offset) and a no-new-bytes-still-returns-cumulative-summary test. +- [x] Implement. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. +- [x] Commit. + +--- + +## Self-Review Notes + +- **Spec coverage:** Common fields (Tasks 1-3), story resolution (4-5), transcript parse-state (6-7), the three new events (8-10), and the four trigger wirings (11-12) each map to a numbered spec section. The privacy "never raw prompt text" constraint is enforced structurally (only the resolved `story_id`/`story_source` ever leaves `mapHookRecords()`) rather than left to each task's judgment. +- **Non-goals respected:** no task touches `agent.session.env`, `agent.skill.dispatch`, `agent.git.snapshot`, or any existing event's own content fields — only the common-field wrapper in `mapHookRecords()`. +- **Type consistency:** `OpenUsageRequest` (Task 6) is the one shape Tasks 8, 9, and 11/12 all import and merge/aggregate — no parallel redefinition. `TranscriptParseState` (Task 6) is the single state object Tasks 7, 11, and 12 all read/mutate/save. diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md new file mode 100644 index 000000000..56ea336a0 --- /dev/null +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md @@ -0,0 +1,134 @@ +# EPMCDME-15301 sub-stage 1.1 — expanded fields and new usage events (OTLP analytics pipeline) + +## Goal + +Extend the existing `claude-code-otlp` pipeline (`forwarder.ts` / `otlp-spool` / `claude-code-otlp.plugin.ts`) with: + +- Common fields on every emitted `agent.*` event. +- Incremental transcript re-parsing on every trigger, backed by persisted per-session parse state. +- Three new events: `agent.usage.request`, `agent.subagent.usage`, `agent.session.summary`. +- Story/ticket resolution. +- An extended identity chain. + +This is additive to the existing 13-event `agent.*` taxonomy already produced by `HOOK_EVENT_TYPE_MAP`/`mapHookRecords()` (`forwarder.ts:16-29,221-269`); no server (`codemie`) changes. + +## Common fields + +Every field below is sent on **every** emitted event. Split by *where* each is computed: + +- **Plugin-owned, resolved hook-time inside `processOtlpEvent` itself.** `ClaudeCodeOtlpPlugin.withCommonFields()` (private, `claude-code-otlp.plugin.ts`) merges a small object onto every record in a `ForwardDecision`'s payload — the original hook event and any transcript-derived synthetic events alike — right before `forwardToSpool()` sends them. + - `platform` — the literal `'claude-code'`. + - `entrypoint` — `process.env.CLAUDE_CODE_ENTRYPOINT ?? ''`, read live. + - `client_version` — `resolveClientVersion()` (`client-version-cache.ts`): runs `claude --version` once and caches the result in a TTL file cache under `getCodemiePath('cache', ...)` (1h TTL). A *file* cache, not an in-memory one — each hook fire is a fresh CLI process, so there's no live instance to memoize on across calls. + - `agent_id`/`agent_type` are **not** common fields. They either arrive verbatim on the raw hook payload for subagent-shaped hook events (`SubagentStart`/`SubagentStop`, passed through as-is) or are sourced inside the transcript builders themselves (`agent.usage.request`/`agent.subagent.usage`, see "New events" below) — not through any shared enrichment step. + +- **Daemon-side, client-agnostic** — computed once per forward tick in `buildForwardContext()` and stamped onto every record in `mapHookRecords()` (`forwarder.ts`), the same pattern already used today for `user_email`/`developer_name`/`git_branch`/`repo_remote`/`codemie_project_name`: + - `schema_version=2` + - `event_id` + - `codemie_cli_version` — the running `@codemieai/code` package's own `version` field in `package.json` (currently `0.15.4`). + - `story_id`/`story_source` + - `developer_name`/`identity_source` + +## `event_id` + +One mechanism for every event, old and new: a `randomUUID()` stamped once, hook-time, inside `forwardOtlpEventToSpool()` (`src/agents/plugins/utils.ts`) — the single chokepoint every record (the original hook event and every transcript-derived synthetic event alike) already passes through on its way to the spool. `mapHookRecords()` (`forwarder.ts`) carries that value straight through (`event_id: hookEvent['event_id']`) rather than computing anything daemon-side; `schema_version`/`codemie_cli_version` are still stamped there. + +Because `event_id` is generated fresh on every forward rather than derived from a record's natural key, it does not provide dedup across re-parses — see "Open risks" below. + +## Transcript re-parsing + +Needed to produce the three new events (`agent.usage.request`, `agent.subagent.usage`, `agent.session.summary`), none of which exist in the hook payload itself — unlike the existing 13 events, which just map that payload, these have to be derived by reading and parsing the transcript file. + +Per the data-model doc (§5.1), parsing is **incremental and backed by persisted per-session state** at `~/.codemie/analytics/state/.json`: + +- **State holds**: the byte offset already consumed in the main transcript, and separately for each subagent transcript; the `openRequests` max-merge-in-progress map; `activeSkill`; `branchCounts` (feeds `branch_dominant`); `compactionCount` (persisted count of this session's `PreCompact` triggers, feeds `agent.session.summary`'s `compaction_count`). +- **Triggers**: `Stop`, `SubagentStop`, `PreCompact`, `StopFailure`, `SessionEnd` — each reads only the bytes appended since its stored offset, updates the running tallies, derives any newly-complete records, writes the updated state back to disk, then sends. Each session's load-mutate-save cycle is serialized by a per-session exclusive-create lock file (`withParseStateLock`, `parse-state.ts`) so concurrent hook processes for the same session (e.g. sibling `SubagentStop` fires) can't race on the shared state file; a stale lock (holder crashed) is detected and stolen rather than awaited forever. +- **Recovery path**: if the state file is missing (first run for a session) or fails to parse (corruption), fall back to a full parse from byte `0` and treat every record as newly derived. Because `event_id` is generated fresh at forward-time (see "`event_id`" above) rather than derived from a record's natural key, a record re-derived this way is assigned a new `event_id` — the backend sees a new record, not an update to the original. See Open risks. +- Stays `async`, swallows all errors internally (matching `hook.ts`'s existing try/catch + `process.exitCode` convention), and always exits 0. + +## Transcript field shape (verified against real transcripts) + +Verified against real Claude Code session transcripts (current client, multiple projects/sessions) and cross-checked against the existing production parser `src/cli/commands/analytics/cost/usage-readers.ts` (`ClaudeRawMessage`/`extractClaudeUsageRecords`), which already parses most of this same shape for the unrelated cost-reporting pipeline — prior art to adapt, not a blank slate: + +- **No top-level `requestId` field appears on any real transcript line observed.** It is not declared in this repo's own `claude-message-types.ts` `ClaudeMessage` type either. `usage-readers.ts` already treats it as optional and falls back to `message.id` alone when absent (`const key = id || reqId ? ... : null`) — the existing production code already assumes `requestId` may not be there. +- **Use `message.id`** (the API-level message id, stable across streaming chunks for one response) **as `request_id`.** It is present on every usage-bearing line sampled and is the only reliable per-response identity. This also means the `agent.usage.request` `event_id` keys on `message.id`, not a field that in practice is always empty — had it keyed on the literal absent `requestId`, every request for a given model within a session would collide onto the same `event_id` and overwrite rather than accumulate. +- **`message.stop_reason`** sits directly on `message`, as a sibling of `usage` — not nested inside it. Observed value: `"tool_use"`. +- **`message.usage`** fields, confirmed by direct inspection: + - Flat: `input_tokens`, `output_tokens`, `cache_read_input_tokens`, `cache_creation_input_tokens` (matches `usage-readers.ts`). + - `service_tier` (observed `"standard"`) and `speed` (observed `"standard"`) are flat fields on `usage`, not on `message`. + - `inference_geo` is a flat field on `usage`, observed as `""` (present but empty, not absent). + - `cache_creation.ephemeral_1h_input_tokens` / `cache_creation.ephemeral_5m_input_tokens` — nested one level, matches `usage-readers.ts`. + - `server_tool_use.web_search_requests` / `server_tool_use.web_fetch_requests` — nested one level under `server_tool_use`, not flat fields. + - `iterations` (observed `[]`) — present on every sampled line but not named anywhere in this ticket's scope; shape and purpose unconfirmed. Carry through as opaque/unused rather than guessing a meaning. +- Top-level fields confirmed present on every line: `sessionId`, `gitBranch`, `cwd`, `timestamp`, `version`, `entrypoint`, `uuid`, `parentUuid`, `isSidechain`, `userType` — matches `claude-message-types.ts`'s `ClaudeMessage`. + +## New events + +### `agent.usage.request` + +- Fires on `Stop` (re-parse of the main transcript), `SubagentStop` (re-parse of that subagent's own transcript, `scope_kind=agent`), `PreCompact` (re-parse of the main transcript, so in-progress request usage is captured before compaction can drop the turns it came from), `StopFailure` (same re-parse as `Stop`, for a turn that ended via failure rather than a clean stop), and `SessionEnd` (final re-parse of the main transcript **and every subagent transcript**). +- One per unique `(request_id, model)` across the session transcript and all subagent transcripts. +- Take the max per numeric field across duplicate records (per `openRequests` merge). +- `scope_kind`/`scope_name` = `main`/`skill`/`agent` depending on which transcript (main vs. a named skill context vs. a subagent transcript) the record came from. + +Fields: `request_id`, `timestamp`, `model_raw`, `model`, `speed`, `inference_geo`, `service_tier`, `input_tokens`, `cache_creation_5m_tokens`, `cache_creation_1h_tokens`, `cache_read_tokens`, `output_tokens`, `web_search_requests`, `web_fetch_requests`, `scope_kind`, `scope_name`, `agent_id`, `stop_reason`, `is_api_error`, `git_branch`. All sourced from the transcript shape confirmed above (`message.id`, `message.usage.*`, `message.stop_reason`, `server_tool_use.*`, `gitBranch`) or, for `model`/`model_raw`, from the same resolution chain the statusline already uses (`parseRoutingHeaders()`/`parseBackendModelName()`); `agent_id` is passed through by the caller (the orchestrator) — `''` for a main-transcript record, the subagent's own `agentId` for a subagent-transcript record — not resolved by any shared enrichment step. `is_api_error`'s presence pattern on a real error is unverified — see Open risks. + +### `agent.subagent.usage` + +- Fires on `SubagentStop`, keyed by `tool_use_id`, and again on `SessionEnd` — re-emitted for **every** subagent transcript found, not just ones whose `SubagentStop` already fired (backstop for a crashed/missed subagent hook). Never fires on `Stop`/`PreCompact` (main-transcript-only parses). +- Aggregates that subagent's own transcript: tokens by model/cache tier, tool-call/error counts, skills invoked. + +Fields: `agent_type`, `description`, `tool_use_id` (all from the subagent's own `.meta.json` sidecar, already read by `claude.session.ts`'s `findSubagentFiles()`), `spawn_depth` (same sidecar), `workflow_run`, `worktree`, `started_at` (that transcript's first line `timestamp`), `duration_ms` (derived from first/last line `timestamp`), tokens by model/cache tier and `api_calls` (aggregated from that subagent transcript's own usage records, same fields as `agent.usage.request` above), tool-call counts (`tool_use` block count), error counts (`is_api_error`/tool-error occurrences), skills invoked (`tool_use` blocks named `Skill`, via `extractNamedInvocations()`). `spawn_depth`'s default for top-level subagents, and `workflow_run`/`worktree`'s missing source, are both tracked in Open risks. + +### `agent.session.summary` + +- Fires on `Stop` (`phase=incremental`) and `SessionEnd` (`phase=final`, all fields recomputed from full state) — the only two of the three triggers that are session-level stop points; `SubagentStop`/`PreCompact` never emit it. +- `primary_model` = model with the most `agent.usage.request` records this session. +- `primary_command` = most-frequently-invoked slash command. +- `branch_dominant` = key with the highest count in `branchCounts`. + +Fields: models used, `primary_model` (aggregated from this session's `agent.usage.request` records), tool-call counts/errors by tool (existing `PostToolUse`/`PostToolUseFailure` hook payloads, already forwarded today), skills used (`extractNamedInvocations()`'s `skillInvocations`), slash-commands in order and `primary_command` (`extractNamedInvocations()`'s `commandInvocations`), lines added/removed and files changed/written (existing `Edit`/`Write` tool payloads, already forwarded today), compaction counts (`PreCompact` trigger count this session), branch-switch counts and `branch_dominant` (derived from `git_branch` per record via `branchCounts`), `client_version`/`codemie_cli_version` (common fields, see above), session start/end timestamps (`SessionStart`/`SessionEnd` hook timestamps), `api_calls` (count of `agent.usage.request` records this session), session `title`. `title`'s missing source is tracked in Open risks. + +## Story/ticket resolution + +Resolved fresh every time, by event type: + +- **Non-prompt events** (everything except `agent.prompt.submit`): priority chain is explicit → branch. Both are resolvable without any per-record text, so both are computed once per forward tick in `buildForwardContext()` and stamped onto the whole batch in `mapHookRecords()` (`forwarder.ts`) — same per-tick cache already used for `git_branch`/`repo_remote`. +- **`agent.prompt.submit` events**: priority chain is explicit → marker → branch → mention. Marker/mention require that record's own prompt text, so they're evaluated per-record against the *full, untruncated* prompt — either hook-side at `UserPromptSubmit` before `prompt_body` is truncated to `MAX_PROMPT_CHARS=200`, or per-record inside `mapHookRecords()` using the record's own `hookEvent['prompt']` before truncation. Only the resolved `story_id`/`story_source` is threaded through to the spool/forwarder, never the raw text (ticket privacy constraint). + +**Explicit** reads from two sources, first match wins: +1. `SDLC_ANALYTICS_STORY_ID` env var. +2. `/.claude/analytics.local.json` — shape `{ "storyId": string }`. `cwd` is the hook payload's existing `cwd` field (`claude-code-otlp.types.ts:19,53`, already read via `process.cwd()` at `claude-code-otlp.plugin.ts:38`), which is the project root for Claude Code hook invocations. Gitignored — add `.claude/analytics.local.json` to `.gitignore`. + +Reading is read-only here: **nothing in this stage writes that file.** A command that writes it (e.g. `/codemie:set-story`) is tracked separately and out of scope for this stage. + +**Branch** reuses `resolveGitInfo()`/`detectGitBranch` (`forwarder.ts`, `src/utils/processes.ts`). + +## Identity resolution + +- Extend `resolveUserEmail()`'s JWT-only chain (`forwarder.ts:67-81`) to `jwt → git → codemie_cli → os`, first available wins: + - `git` reads `git config user.name`/`user.email`. + - `codemie_cli` reads the existing CLI profile config. + - `os` reads `os.userInfo().username`. +- **Security scope (approved 2026-10-05):** this `jwt → git → codemie_cli → os` derivation chain reads an identity-like value from local git config / CLI config / OS username with no external verification. It is used only to stamp `developer_name`/`identity_source` on outbound analytics/telemetry events (`identity.ts`) — never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing, which `security-practices.md`'s Project & User Attribution Headers rule governs. + +## Non-goals + +- `agent.session.env`, `agent.skill.dispatch`, `agent.git.snapshot` — sub-stage 1.2. +- Any modification to existing event *content* (`agent.session.start`, `agent.prompt.submit`, `agent.tool.start`/`end`, `agent.subagent.stop`, `agent.session.compact`, `agent.session.stop`/`end`) beyond adding the common fields above. + +## Open risks + +- **Transcript field confidence gaps** (all affect the three new events; sourcing details are in "Transcript field shape" and "New events" above): + - `request_id` must be sourced from `message.id` — no real transcript observed carries a top-level `requestId`. Resolved, but flagged so it isn't silently reintroduced from `usage-readers.ts`'s optional-`requestId` interface during implementation. + - `is_api_error` (used in `agent.usage.request` and in `agent.subagent.usage`'s error counts) is read by production code elsewhere in this repo, but no sampled transcript contained an actual API error, so whether it's always present (false/absent otherwise) or appears only on the erroring line is unverified. + - `spawn_depth` (`agent.subagent.usage`) is present in the `.meta.json` sidecar only for nested subagents (depth ≥ 2); top-level subagents omit it, so implementation needs an explicit default rather than treating absence as an error. + - `workflow_run`/`worktree` (`agent.subagent.usage`) have no identified source in this codebase's transcript handling or the `.meta.json` sidecar. The external data-model doc names a `workflows//` sidecar directory as the source, but a direct check across every local session directory found no such directory in any sampled session — the gap stands; worth revisiting with the data-model doc's owner. + - `title` (`agent.session.summary`) has no identified source in a real transcript, top-level or nested, and the data-model doc doesn't name one either — unresolved on both sides. +- **Known limitations, accepted as-is for this sub-stage:** + - `lines_added`/`lines_removed` (`agent.session.summary`) always emit `0`. An `Edit`/`Write` tool_use's `input` carries the *proposed* edit, not a diff stat, so no reliable added/removed line count can be derived from it without re-implementing diffing — out of scope for this sub-stage. + - `scope_kind: 'skill'` (`agent.usage.request`) is never emitted. No reliable in-transcript signal for "this turn is inside a skill context" was found, so every main-transcript usage record is unconditionally scoped `'main'`. `state.activeSkill` stays tracked-but-unused, available for a later task that identifies a real signal. + - `started_at`/`ended_at` (`agent.session.summary`) are close approximations, not the literal `SessionStart`/`SessionEnd` hook payload timestamps: `started_at` is the main transcript's own first parsed line timestamp, and `ended_at` is `Date.now()` at parse time. Threading the actual hook timestamps through would require widening every `collectMainTranscriptEvents` call site's signature; deferred. + - `commands_in_order` (`agent.session.summary`) contains the right distinct slash-command names but not a true chronological sequence — it is `Object.keys()` of `commandInvocations`, an unordered count map. No chronological invocation order is available anywhere in `NamedInvocationCounts` to draw from. A genuine mismatch with the field's name, documented rather than fabricated. + - `description` (`agent.subagent.usage`) always emits `''`. The sidecar `.meta.json` schema it mirrors (matching the real production schema) has no `description` key at all, so there is no source anywhere in this codebase to populate it from — unlike the sibling `workflow_run`/`worktree`/`title` gaps above, this one wasn't caught before implementation. + - `event_id` does not provide idempotent dedup across re-parses: it's a `randomUUID()` generated fresh every time a record is forwarded, not a deterministic function of the record's natural key, so a crash-before-save re-parse of `agent.usage.request`/`agent.subagent.usage`/`agent.session.summary` forwards the re-derived record under a new `event_id` — the backend sees a new record rather than an update to the original. Accepted as-is for this sub-stage; the natural key is still stable across re-derivation (`request_id`+`model`, `tool_use_id`, or `session_id`+`phase`), so dedup on that key is possible if the backend needs it. diff --git a/src/agents/core/types.ts b/src/agents/core/types.ts index a16a86cd5..5876eb9c7 100644 --- a/src/agents/core/types.ts +++ b/src/agents/core/types.ts @@ -731,7 +731,7 @@ export enum AgentAdapterType { OTLP, } -export interface OtlpAgentAdapter { +export interface OtlpAgentAdapter { readonly name: string; readonly type: AgentAdapterType.OTLP; diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index 17c9d37df..a3b8f0210 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -1,4 +1,4 @@ -import { describe, it, expect, vi, beforeEach } from 'vitest'; +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; vi.mock('../claude-code-otlp.allowlist.js', () => ({ readAllowlistState: vi.fn(async () => ({ kind: 'valid', paths: ['/proj'] })), @@ -11,18 +11,43 @@ vi.mock('@/utils/logger.js', () => ({ logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, })); -import { ClaudeCodeOtlpPlugin } from '../claude-code-otlp.plugin.js'; +const resolveClientVersionMock = vi.fn(); +const collectMainTranscriptEventsMock = vi.fn(); +const collectSubagentTranscriptEventsMock = vi.fn(); +const findSubagentFilesMock = vi.fn(); + +vi.mock('../client-version-cache.js', () => ({ + resolveClientVersion: resolveClientVersionMock, +})); + +vi.mock('../transcript/orchestrator.js', () => ({ + collectMainTranscriptEvents: collectMainTranscriptEventsMock, + collectSubagentTranscriptEvents: collectSubagentTranscriptEventsMock, +})); + +vi.mock('../transcript/subagent-usage.js', () => ({ + findSubagentFiles: findSubagentFilesMock, +})); + import { isProjectTracked } from '../claude-code-otlp.allowlist.js'; import { forwardOtlpEventToSpool } from '../../utils.js'; import { ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; +// Deferred past the mock-backing consts above: a static import of the plugin would be +// hoisted ahead of them (ESM import hoisting), tripping a TDZ error inside the +// transcript/orchestrator.js mock factory, which closes over those consts. +const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const event = (name: string) => JSON.stringify({ session_id: 's', transcript_path: '', cwd: '/x', hook_event_name: name }); describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { const plugin = new ClaudeCodeOtlpPlugin(); const ensureOtlpProxy = vi.fn(async () => {}); - beforeEach(() => vi.clearAllMocks()); + beforeEach(() => { + vi.clearAllMocks(); + resolveClientVersionMock.mockResolvedValue('2.1.23'); + }); it('does nothing for untracked projects, for every event', async () => { vi.mocked(isProjectTracked).mockResolvedValue(false); @@ -47,3 +72,244 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { ); }); }); + +describe('ClaudeCodeOtlpPlugin hook-time enrichment', () => { + const plugin = new ClaudeCodeOtlpPlugin(); + const ensureOtlpProxy = vi.fn(async () => {}); + const originalEntrypoint = process.env.CLAUDE_CODE_ENTRYPOINT; + + function hookEvent(overrides: Record = {}): Record { + return { + session_id: 'sid-1', + transcript_path: '/tmp/transcript.jsonl', + cwd: '/repo', + hook_event_name: 'Stop', + ...overrides, + }; + } + + beforeEach(() => { + vi.clearAllMocks(); + vi.mocked(isProjectTracked).mockResolvedValue(true); + resolveClientVersionMock.mockResolvedValue('2.1.23'); + collectMainTranscriptEventsMock.mockResolvedValue([]); + collectSubagentTranscriptEventsMock.mockResolvedValue([]); + findSubagentFilesMock.mockResolvedValue([]); + }); + + afterEach(() => { + if (originalEntrypoint === undefined) { + delete process.env.CLAUDE_CODE_ENTRYPOINT; + } else { + process.env.CLAUDE_CODE_ENTRYPOINT = originalEntrypoint; + } + }); + + it('enriches the forwarded event with platform, entrypoint, and client_version', async () => { + process.env.CLAUDE_CODE_ENTRYPOINT = 'cli'; + const rawEvent = JSON.stringify(hookEvent()); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(1); + const [forwarded] = vi.mocked(forwardOtlpEventToSpool).mock.calls[0]; + expect(forwarded.platform).toBe('claude-code'); + expect(forwarded.entrypoint).toBe('cli'); + expect(forwarded.client_version).toBe('2.1.23'); + }); + + it('preserves agent_id/agent_type already present on the raw event (SubagentStop)', async () => { + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-1.jsonl', + agent_id: 'sub-1', + agent_type: 'explore', + }) + ); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + const [forwarded] = vi.mocked(forwardOtlpEventToSpool).mock.calls[0]; + expect(forwarded.agent_id).toBe('sub-1'); + expect(forwarded.agent_type).toBe('explore'); + }); +}); + +describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { + const ensureOtlpProxy = vi.fn(async () => {}); + + function hookEvent(overrides: Record = {}): Record { + return { + session_id: 'sid-1', + transcript_path: '/tmp/transcript.jsonl', + cwd: '/repo', + hook_event_name: 'Stop', + ...overrides, + }; + } + + beforeEach(() => { + collectMainTranscriptEventsMock.mockReset(); + collectMainTranscriptEventsMock.mockResolvedValue([]); + collectSubagentTranscriptEventsMock.mockReset(); + collectSubagentTranscriptEventsMock.mockResolvedValue([]); + findSubagentFilesMock.mockReset(); + findSubagentFilesMock.mockResolvedValue([]); + resolveClientVersionMock.mockReset(); + resolveClientVersionMock.mockResolvedValue('2.1.23'); + vi.mocked(forwardOtlpEventToSpool).mockReset(); + vi.mocked(isProjectTracked).mockResolvedValue(true); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + it.each(['Stop', 'PreCompact', 'SessionEnd', 'StopFailure'] as const)( + 'invokes collectMainTranscriptEvents with the extracted session/transcript/trigger on %s', + async (hookEventName) => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const parsedEvent = hookEvent({ hook_event_name: hookEventName }); + const rawEvent = JSON.stringify(parsedEvent); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(collectMainTranscriptEventsMock).toHaveBeenCalledWith( + 'sid-1', + '/tmp/transcript.jsonl', + hookEventName + ); + expect(forwardOtlpEventToSpool).toHaveBeenCalledWith( + expect.objectContaining(parsedEvent), + 'claude-code-otlp' + ); + } + ); + + it('does not dispatch collectMainTranscriptEvents for unrelated hook events', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'PostToolUse' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(collectMainTranscriptEventsMock).not.toHaveBeenCalled(); + }); + + it('invokes collectSubagentTranscriptEvents with the agent_id/tool_use_id/agent_type extracted from a SubagentStop payload', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-1.jsonl', + agent_id: 'sub-1', + tool_use_id: 'tu-1', + agent_type: 'explore', + }) + ); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith('sid-1', { + agentId: 'sub-1', + filePath: '/tmp/agent-sub-1.jsonl', + toolUseId: 'tu-1', + agentType: 'explore', + }); + }); + + it('derives agent_id from the sidecar filename on SubagentStop when agent_id is absent', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-2.jsonl', + }) + ); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( + 'sid-1', + expect.objectContaining({ agentId: 'sub-2' }) + ); + }); + + it('skips collectSubagentTranscriptEvents on SubagentStop when agent_transcript_path is missing', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SubagentStop' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(collectSubagentTranscriptEventsMock).not.toHaveBeenCalled(); + }); + + it('runs collectSubagentTranscriptEvents for every subagent file found on SessionEnd', async () => { + findSubagentFilesMock.mockResolvedValue([ + { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' }, + { agentId: 'sub-2', filePath: '/tmp/agent-sub-2.jsonl' }, + ]); + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SessionEnd' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(findSubagentFilesMock).toHaveBeenCalledWith('/tmp/transcript.jsonl'); + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledTimes(2); + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( + 'sid-1', + { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' } + ); + }); + + it('forwards the raw event and skips all transcript-parse dispatch when session_id is empty', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const parsedEvent = hookEvent({ session_id: '', hook_event_name: 'Stop' }); + const rawEvent = JSON.stringify(parsedEvent); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(collectMainTranscriptEventsMock).not.toHaveBeenCalled(); + expect(forwardOtlpEventToSpool).toHaveBeenCalledWith( + expect.objectContaining(parsedEvent), + 'claude-code-otlp' + ); + }); + + it('forwards every event a per-event handler returns (the raw event plus any derived events) through the single forwardToSpool path, in order', async () => { + const derivedUsageEvent = { type: 'agent.usage.request' }; + const derivedSummaryEvent = { type: 'agent.session.summary' }; + collectMainTranscriptEventsMock.mockResolvedValue([derivedUsageEvent, derivedSummaryEvent]); + + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const parsedEvent = hookEvent({ hook_event_name: 'Stop' }); + const rawEvent = JSON.stringify(parsedEvent); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + // collectMainTranscriptEvents/collectSubagentTranscriptEvents never call forwardOtlpEventToSpool + // themselves (they are mocked here to just return data) — every event that reaches the spool + // mock arrived via forwardToSpool, called exactly once from processOtlpEvent. + expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(3); + expect(vi.mocked(forwardOtlpEventToSpool).mock.calls.map(([record]) => record)).toEqual([ + expect.objectContaining(parsedEvent), + expect.objectContaining(derivedUsageEvent), + expect.objectContaining(derivedSummaryEvent), + ]); + }); + + it('rejects when rawEvent is malformed JSON', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + + await expect(plugin.processOtlpEvent('not json', { ensureOtlpProxy })).rejects.toThrow(); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts new file mode 100644 index 000000000..054e44357 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts @@ -0,0 +1,80 @@ +import { describe, it, expect, vi, beforeEach } from 'vitest'; + +const readFileMock = vi.fn(); +const writeFileMock = vi.fn(); +const mkdirMock = vi.fn(); +const execMock = vi.fn(); + +vi.mock('node:fs/promises', () => ({ + readFile: readFileMock, + writeFile: writeFileMock, + mkdir: mkdirMock, +})); + +vi.mock('@/utils/exec.js', () => ({ + exec: execMock, +})); + +describe('client-version-cache', () => { + beforeEach(() => { + vi.resetModules(); + readFileMock.mockReset(); + writeFileMock.mockReset().mockResolvedValue(undefined); + mkdirMock.mockReset().mockResolvedValue(undefined); + execMock.mockReset(); + execMock.mockResolvedValue({ code: 0, stdout: '2.1.23 (Claude Code)', stderr: '', signal: null }); + }); + + it('cold cache: execs and writes a new cache entry', async () => { + readFileMock.mockRejectedValue(new Error('ENOENT')); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.23'); + expect(execMock).toHaveBeenCalledWith('claude', ['--version']); + expect(writeFileMock).toHaveBeenCalledTimes(1); + }); + + it('warm cache within TTL: does not exec', async () => { + readFileMock.mockResolvedValue(JSON.stringify({ version: '2.1.0', resolvedAt: Date.now() })); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.0'); + expect(execMock).not.toHaveBeenCalled(); + }); + + it('stale cache past TTL: re-execs', async () => { + const twoHoursAgo = Date.now() - 2 * 60 * 60 * 1000; + readFileMock.mockResolvedValue(JSON.stringify({ version: '2.0.0', resolvedAt: twoHoursAgo })); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.23'); + expect(execMock).toHaveBeenCalledTimes(1); + }); + + it('exec failure: returns empty string and does not write a cache entry', async () => { + readFileMock.mockRejectedValue(new Error('ENOENT')); + execMock.mockRejectedValue(new Error('ENOENT')); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe(''); + expect(writeFileMock).not.toHaveBeenCalled(); + }); + + it('corrupt cache file: treated as cold and re-execs', async () => { + readFileMock.mockResolvedValue('not json'); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.23'); + expect(execMock).toHaveBeenCalledTimes(1); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index f7241e0e9..3ed3f72c9 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -1,19 +1,27 @@ +import { basename } from 'node:path'; import { AuthGateResult, ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; import { logger } from '@/utils/logger.js'; import { ConfigLoader } from '@/utils/config.js'; import { AgentAdapterType, OtlpAdapterDeps, OtlpAgentAdapter } from '@/agents/core/types.js'; import { CLAUDE_CODE_OTLP_AGENT_NAME } from './claude-code-otlp.constants.js'; -import { ForwardDecision, toBaseClaudeCodeHookEvent } from './claude-code-otlp.types.js'; +import { ForwardDecision } from './claude-code-otlp.types.js'; import { forwardOtlpEventToSpool } from '../utils.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; +import { + collectMainTranscriptEvents, + collectSubagentTranscriptEvents, + type SubagentFile, +} from './transcript/orchestrator.js'; +import { findSubagentFiles } from './transcript/subagent-usage.js'; +import { resolveClientVersion } from './client-version-cache.js'; export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; public readonly type = AgentAdapterType.OTLP; public async processOtlpEvent(rawEvent: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { - const event = toBaseClaudeCodeHookEvent(JSON.parse(rawEvent)); + const event = JSON.parse(rawEvent) as Record; // INVARIANT - do not weaken. An untracked project must produce NO hooks data // in the daemon spool, must not start the daemon, and must not run the SSO @@ -22,7 +30,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { // data, and skips sessions that only have OTEL data. Forwarding a hook event // for an untracked project would make the session sendable and leak its // data to the backend. - const isTracked = await isProjectTracked(event.cwd, await readAllowlistState()); + const isTracked = await isProjectTracked(readString(event, 'cwd'), await readAllowlistState()); if (!isTracked) { logger.debug('[Claude Code OTLP plugin] project not in analytics allowlist, ignoring hook event'); return; @@ -30,24 +38,113 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { await ensureOtlpProxy(this.name); - const evaluation = await this.evaluate(event.hookEventName, rawEvent); + const evaluation = await this.evaluate(event); if (evaluation.decision === 'block') { logger.error(`[Claude Code OTLP plugin] Blocking prompt: ${evaluation.reason}`); console.log(JSON.stringify(evaluation)); return; } - this.forwardToSpool(evaluation.payload); + this.forwardToSpool(await this.withCommonFields(evaluation.payload)); } - private async evaluate(hookEventName: string, rawEvent: string): Promise { + /** Resolved once per call so every record in the batch shares one client-version lookup. */ + private async withCommonFields(records: Record[]): Promise[]> { + const common = { + platform: 'claude-code', + entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', + client_version: await resolveClientVersion(), + }; + return records.map((record) => ({ ...record, ...common })); + } + + private async evaluate(parsed: Record): Promise { + const sessionId = readString(parsed, 'session_id'); + if (!sessionId) { + return { decision: 'forward', payload: [parsed] }; + } + + const hookEventName = readString(parsed, 'hook_event_name'); if (hookEventName === 'UserPromptSubmit') { - return await this.onUserPromptSubmit(rawEvent); + return await this.onUserPromptSubmit(parsed); + } + if (hookEventName === 'Stop') { + return await this.onStopEvent(parsed); + } + if (hookEventName === 'PreCompact') { + return await this.onPreCompactEvent(parsed); + } + if (hookEventName === 'StopFailure') { + return await this.onStopFailureEvent(parsed); + } + if (hookEventName === 'SessionEnd') { + return await this.onSessionEndEvent(parsed); + } + if (hookEventName === 'SubagentStop') { + return await this.onSubagentStopEvent(parsed); } - return { - decision: 'forward', - payload: rawEvent, + return { decision: 'forward', payload: [parsed] }; + } + + private async onStopEvent(parsed: Record): Promise { + const derived = await collectMainTranscriptEvents( + readString(parsed, 'session_id'), + readString(parsed, 'transcript_path'), + 'Stop' + ); + return { decision: 'forward', payload: [parsed, ...derived] }; + } + + private async onPreCompactEvent(parsed: Record): Promise { + const derived = await collectMainTranscriptEvents( + readString(parsed, 'session_id'), + readString(parsed, 'transcript_path'), + 'PreCompact' + ); + return { decision: 'forward', payload: [parsed, ...derived] }; + } + + private async onStopFailureEvent(parsed: Record): Promise { + const derived = await collectMainTranscriptEvents( + readString(parsed, 'session_id'), + readString(parsed, 'transcript_path'), + 'StopFailure' + ); + return { decision: 'forward', payload: [parsed, ...derived] }; + } + + private async onSessionEndEvent(parsed: Record): Promise { + const sessionId = readString(parsed, 'session_id'); + const transcriptPath = readString(parsed, 'transcript_path'); + const derived = await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd'); + + // Backstop: guarantee every subagent discovered for this session gets at least one + // agent.subagent.usage event, even when its own SubagentStop hook never fired. + const subagentFiles = await findSubagentFiles(transcriptPath); + for (const file of subagentFiles) { + derived.push(...(await collectSubagentTranscriptEvents(sessionId, file))); + } + + return { decision: 'forward', payload: [parsed, ...derived] }; + } + + private async onSubagentStopEvent(parsed: Record): Promise { + const agentTranscriptPath = readOptionalString(parsed, 'agent_transcript_path'); + if (!agentTranscriptPath) { + return { decision: 'forward', payload: [parsed] }; + } + + const subagentFile: SubagentFile = { + agentId: + readOptionalString(parsed, 'agent_id') ?? + basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''), + filePath: agentTranscriptPath, + toolUseId: readOptionalString(parsed, 'tool_use_id'), + agentType: readOptionalString(parsed, 'agent_type'), }; + + const derived = await collectSubagentTranscriptEvents(readString(parsed, 'session_id'), subagentFile); + return { decision: 'forward', payload: [parsed, ...derived] }; } private async ensureProxyAuth(): Promise { @@ -60,18 +157,20 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { }); } - private forwardToSpool(rawEvent: string): void { + private forwardToSpool(events: Record[]): void { // Intentionally not awaited: forwardOtlpEventToSpool is fire-and-forget. - forwardOtlpEventToSpool(rawEvent, CLAUDE_CODE_OTLP_AGENT_NAME); + for (const event of events) { + forwardOtlpEventToSpool(event, CLAUDE_CODE_OTLP_AGENT_NAME); + } } - private async onUserPromptSubmit(rawEvent: string): Promise { + private async onUserPromptSubmit(parsed: Record): Promise { const authResult = await this.ensureProxyAuth(); if (authResult.ok) { return { decision: 'forward', - payload: rawEvent, + payload: [parsed], } } @@ -89,3 +188,12 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { } } } + +function readOptionalString(record: Record, key: string): string | undefined { + const value = record[key]; + return typeof value === 'string' ? value : undefined; +} + +function readString(record: Record, key: string): string { + return readOptionalString(record, key) ?? ''; +} diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts index 9a9a96f54..5cb254264 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts @@ -1,77 +1,3 @@ -interface RawBaseClaudeCodeHookEvent { - /** Current session identifier */ - session_id: string; - - /** - * UUID identifying the user prompt currently being processed. - * Matches the prompt.id attribute on OpenTelemetry events. - * Absent until the first user input. Requires Claude Code v2.1.196 or later. - */ - prompt_id?: string; - - /** - * Path to conversation JSON. Written asynchronously and may lag - * the in-memory conversation. - */ - transcript_path: string; - - /** Current working directory when the hook is invoked */ - cwd: string; - - /** - * Path to the session’s scratchpad directory for temporary working files. - * Absent when no scratchpad exists or the temp directory is unavailable. - * Requires Claude Code v2.1.257 or later. - */ - scratchpad_dir?: string; - - /** - * Current permission mode: "default", "plan", "acceptEdits", "auto", - * "dontAsk", or "bypassPermissions". Not all events receive this field. - */ - permission_mode?: "default" | "plan" | "acceptEdits" | "auto" | "dontAsk" | "bypassPermissions"; - - /** - * Object with a level field holding the effort level in effect when the hook runs. - * Present for events that fire within a tool-use context when supported by the model. - */ - effort?: { - level: "low" | "medium" | "high" | "xhigh" | "max"; - }; - - /** Name of the event that fired */ - hook_event_name: string; -} - -/** - * camelCase-keyed mirror of {@link RawBaseClaudeCodeHookEvent}. - */ -interface BaseClaudeCodeHookEvent { - sessionId: string; - promptId?: string; - transcriptPath: string; - cwd: string; - scratchpadDir?: string; - permissionMode?: "default" | "plan" | "acceptEdits" | "auto" | "dontAsk" | "bypassPermissions"; - effort?: { - level: "low" | "medium" | "high" | "xhigh" | "max"; - }; - hookEventName: string; -} - -export function toBaseClaudeCodeHookEvent(raw: RawBaseClaudeCodeHookEvent): BaseClaudeCodeHookEvent { - return { - sessionId: raw.session_id, - promptId: raw.prompt_id, - transcriptPath: raw.transcript_path, - cwd: raw.cwd, - scratchpadDir: raw.scratchpad_dir, - permissionMode: raw.permission_mode, - effort: raw.effort ? { level: raw.effort.level } : undefined, - hookEventName: raw.hook_event_name, - }; -} - export type ForwardDecision = - | { decision: 'forward'; payload: string } + | { decision: 'forward'; payload: Record[] } | { decision: 'block'; reason: string, hookSpecificOutput: Record }; diff --git a/src/agents/plugins/claude-code-otlp/client-version-cache.ts b/src/agents/plugins/claude-code-otlp/client-version-cache.ts new file mode 100644 index 000000000..743ff45a8 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/client-version-cache.ts @@ -0,0 +1,61 @@ +import { readFile, writeFile, mkdir } from 'node:fs/promises'; +import { dirname } from 'node:path'; +import { exec } from '@/utils/exec.js'; +import { getCodemiePath } from '@/utils/paths.js'; +import { logger } from '@/utils/logger.js'; + +const CACHE_PATH = getCodemiePath('cache', 'claude-code-client-version.json'); +const TTL_MS = 1 * 60 * 60 * 1000; // 1h - staleness here only affects an analytics label, not behavior. + +interface CacheEntry { + version: string; + resolvedAt: number; +} + +async function readCache(): Promise { + try { + const raw = await readFile(CACHE_PATH, 'utf-8'); + const entry = JSON.parse(raw) as CacheEntry; + if (Date.now() - entry.resolvedAt < TTL_MS) { + return entry; + } + } catch { + /* missing/corrupt cache: fall through to re-resolve */ + } + return undefined; +} + +async function writeCache(version: string): Promise { + try { + await mkdir(dirname(CACHE_PATH), { recursive: true }); + await writeFile(CACHE_PATH, JSON.stringify({ version, resolvedAt: Date.now() } satisfies CacheEntry)); + } catch (err) { + logger.debug('client-version-cache: write failed', err instanceof Error ? err.message : String(err)); + } +} + +async function execClaudeVersion(): Promise { + try { + const result = await exec('claude', ['--version']); + const trimmed = result.stdout.trim(); + const match = trimmed.match(/^(\d+\.\d+\.\d+)/); + return match ? match[1] : trimmed; + } catch { + return ''; + } +} + +/** Resolve the installed `claude` CLI version, backed by a TTL file cache so a fresh + * `codemie hook` process (one per hook event) doesn't spawn `claude --version` every time. */ +export async function resolveClientVersion(): Promise { + const cached = await readCache(); + if (cached) { + return cached.version; + } + + const version = await execClaudeVersion(); + if (version) { + await writeCache(version); + } + return version; +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl b/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl new file mode 100644 index 000000000..9b67d70f5 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl @@ -0,0 +1,4 @@ +{"sessionId":"session-usage-1","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-01T00:00:00.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-no-usage","parentUuid":null,"isSidechain":false,"userType":"external","message":{"role":"user","content":"hello"}} +{"sessionId":"session-usage-1","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-01T00:00:01.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-pair-1","parentUuid":"uuid-no-usage","isSidechain":false,"userType":"external","message":{"id":"msg_pair_1","role":"assistant","model":"claude-sonnet-4-5-20250929","usage":{"input_tokens":100,"output_tokens":50,"cache_read_input_tokens":5,"cache_creation_input_tokens":0,"service_tier":"standard","speed":"standard","inference_geo":"","cache_creation":{"ephemeral_1h_input_tokens":0,"ephemeral_5m_input_tokens":0},"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"iterations":[]}}} +{"sessionId":"session-usage-1","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-01T00:00:02.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-pair-2","parentUuid":"uuid-pair-1","isSidechain":false,"userType":"external","message":{"id":"msg_pair_1","role":"assistant","model":"claude-sonnet-4-5-20250929","stop_reason":"tool_use","usage":{"input_tokens":100,"output_tokens":120,"cache_read_input_tokens":5,"cache_creation_input_tokens":0,"service_tier":"standard","speed":"standard","inference_geo":"","cache_creation":{"ephemeral_1h_input_tokens":0,"ephemeral_5m_input_tokens":0},"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"iterations":[]}}} +{"sessionId":"session-usage-1","gitBranch":"feature/epmcdme-15301","cwd":"/repo","timestamp":"2026-10-01T00:00:03.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-normal","parentUuid":"uuid-pair-2","isSidechain":false,"userType":"external","message":{"id":"msg_normal_1","role":"assistant","model":"claude-sonnet-4-5-20250929","stop_reason":"end_turn","x-litellm-model-name":"bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0","usage":{"input_tokens":300,"output_tokens":75,"cache_read_input_tokens":40,"cache_creation_input_tokens":15,"service_tier":"priority","speed":"fast","inference_geo":"us","cache_creation":{"ephemeral_1h_input_tokens":10,"ephemeral_5m_input_tokens":5},"server_tool_use":{"web_search_requests":2,"web_fetch_requests":1},"iterations":[]}}} diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts new file mode 100644 index 000000000..41aae4a02 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -0,0 +1,413 @@ +/** + * Tests for `collectMainTranscriptEvents` — the `Stop`/`PreCompact`/`SessionEnd` main-transcript + * orchestrator. + * + * Neither `collectMainTranscriptEvents` nor `collectSubagentTranscriptEvents` writes to the spool itself — + * each returns the raw JSON strings it wants forwarded, and the caller (the plugin's + * `processOtlpEvent`) is the only place that actually forwards them. So these tests read the + * returned array directly; no network/daemon mocking is needed. `CODEMIE_HOME` points at a fresh + * temp directory per test so `loadParseState`/`saveParseState` never touch the real `~/.codemie`. + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; + +let codemieHome: string; +let transcriptDir: string; + +function usageLine(opts: { + uuid: string; + messageId: string; + outputTokens: number; + gitBranch?: string; + stopReason?: string; + timestamp?: string; +}): string { + return JSON.stringify({ + gitBranch: opts.gitBranch ?? 'main', + cwd: '/repo', + timestamp: opts.timestamp ?? '2026-10-01T00:00:00.000Z', + uuid: opts.uuid, + message: { + id: opts.messageId, + role: 'assistant', + model: 'claude-sonnet-4-5-20250929', + stop_reason: opts.stopReason ?? '', + usage: { + input_tokens: 100, + output_tokens: opts.outputTokens, + cache_read_input_tokens: 5, + cache_creation_input_tokens: 0, + service_tier: 'standard', + speed: 'standard', + inference_geo: '', + cache_creation: { ephemeral_1h_input_tokens: 0, ephemeral_5m_input_tokens: 0 }, + server_tool_use: { web_search_requests: 0, web_fetch_requests: 0 }, + }, + }, + }); +} + +function noUsageLine(uuid: string): string { + return JSON.stringify({ + gitBranch: 'main', + cwd: '/repo', + timestamp: '2026-10-01T00:00:00.000Z', + uuid, + message: { role: 'user', content: 'hello' }, + }); +} + +beforeEach(() => { + codemieHome = mkdtempSync(join(tmpdir(), 'codemie-home-')); + process.env.CODEMIE_HOME = codemieHome; + transcriptDir = mkdtempSync(join(tmpdir(), 'codemie-transcript-')); +}); + +afterEach(() => { + delete process.env.CODEMIE_HOME; + rmSync(codemieHome, { recursive: true, force: true }); + rmSync(transcriptDir, { recursive: true, force: true }); +}); + +function writeTranscript(fileName: string, lines: string[]): string { + const filePath = join(transcriptDir, fileName); + writeFileSync(filePath, lines.map((l) => l + '\n').join(''), 'utf-8'); + return filePath; +} + +type ForwardedEvent = Record; + +function parseAll(raw: ForwardedEvent[]): ForwardedEvent[] { + return raw; +} + +/** + * Write a subagent fixture transcript (plus its sidecar `.meta.json`) under + * `//subagents/agent-.jsonl`, matching + * `findSubagentFiles()`'s own discovery convention. Returns the `SubagentFile` shape + * `findSubagentFiles()` would discover for it. + */ +function writeSubagentFixture( + sessionId: string, + agentId: string, + lines: string[], + meta: Record = {} +): { agentId: string; filePath: string } { + const subagentsDir = join(transcriptDir, sessionId, 'subagents'); + mkdirSync(subagentsDir, { recursive: true }); + const filePath = join(subagentsDir, `agent-${agentId}.jsonl`); + writeFileSync(filePath, lines.map((l) => l + '\n').join(''), 'utf-8'); + writeFileSync(join(subagentsDir, `agent-${agentId}.meta.json`), JSON.stringify(meta), 'utf-8'); + return { agentId, filePath }; +} + +describe('collectMainTranscriptEvents — idempotent reparse', () => { + it('returns agent.usage.request events with identical request_id/model pairs across a crash-before-save re-parse', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + const { saveParseState, createParseState } = await import('../parse-state.js'); + + const sessionId = 'session-idempotent'; + const transcriptPath = writeTranscript('transcript-idempotent.jsonl', [ + noUsageLine('uuid-0'), + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75, stopReason: 'end_turn' }), + ]); + + const first = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + + const firstPairs = first + .filter((e) => e.type === 'agent.usage.request') + .map((e) => `${e.request_id}::${e.model}`) + .sort(); + + expect(firstPairs).toHaveLength(2); + + // Simulate "a re-parse after a crash before state was saved": the transcript file is fully + // there, but the persisted state is wound back to fresh (as if the first run's save never + // happened). + await saveParseState(sessionId, createParseState()); + + const second = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + + const secondPairs = second + .filter((e) => e.type === 'agent.usage.request') + .map((e) => `${e.request_id}::${e.model}`) + .sort(); + + expect(secondPairs).toHaveLength(2); + expect(secondPairs).toEqual(firstPairs); + }); +}); + +describe('collectMainTranscriptEvents — Stop trigger', () => { + it('returns one agent.usage.request event per distinct request plus one incremental agent.session.summary', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-stop-basic'; + const transcriptPath = writeTranscript('transcript-stop.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), + ]); + + const raw = await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); + expect(raw).toHaveLength(3); + + const events = parseAll(raw); + const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(usageEvents).toHaveLength(2); + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].phase).toBe('incremental'); + expect(summaryEvents[0]).not.toHaveProperty('ended_at'); + }); +}); + +describe('collectMainTranscriptEvents — PreCompact trigger', () => { + it('returns usage-request events but never a session summary', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-precompact'; + const transcriptPath = writeTranscript('transcript-precompact.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + ]); + + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'PreCompact')); + const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(usageEvents).toHaveLength(1); + expect(summaryEvents).toHaveLength(0); + }); +}); + +describe('collectMainTranscriptEvents — compaction_count', () => { + it('persists one increment per PreCompact trigger and surfaces the cumulative count on a later summary', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-compaction'; + const transcriptPath = writeTranscript('transcript-compaction.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + ]); + + await collectMainTranscriptEvents(sessionId, transcriptPath, 'PreCompact'); + await collectMainTranscriptEvents(sessionId, transcriptPath, 'PreCompact'); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].compaction_count).toBe(2); + }); +}); + +describe('collectMainTranscriptEvents — api_calls', () => { + it("surfaces the session's full agent.usage.request record count on the summary event", async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-api-calls'; + const transcriptPath = writeTranscript('transcript-api-calls.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), + ]); + + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].api_calls).toBe(usageEvents.length); + expect(summaryEvents[0].api_calls).toBe(2); + }); +}); + +describe('collectMainTranscriptEvents — SessionEnd trigger', () => { + it('returns a final-phase summary with an ended_at key present', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-end'; + const transcriptPath = writeTranscript('transcript-end.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + ]); + + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd')); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].phase).toBe('final'); + expect(summaryEvents[0]).toHaveProperty('ended_at'); + expect(typeof summaryEvents[0].ended_at).toBe('string'); + }); +}); + +describe('collectMainTranscriptEvents — missing transcript file', () => { + it('resolves cleanly to an empty array, for a trigger that never emits a summary', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + const { loadParseState } = await import('../parse-state.js'); + + const sessionId = 'session-missing-file'; + const missingPath = join(transcriptDir, 'does-not-exist.jsonl'); + + await expect(collectMainTranscriptEvents(sessionId, missingPath, 'PreCompact')).resolves.toEqual([]); + + const state = await loadParseState(sessionId); + expect(state.mainOffset).toBe(0); + }); + + it('never throws even on Stop (which does attempt a full-file summary recompute)', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-missing-file-stop'; + const missingPath = join(transcriptDir, 'also-does-not-exist.jsonl'); + + await expect(collectMainTranscriptEvents(sessionId, missingPath, 'Stop')).resolves.toBeInstanceOf(Array); + }); +}); + +describe('collectMainTranscriptEvents — tool-call accumulation', () => { + it('counts Edit/Write tool_use blocks into files_changed/files_written on the Stop summary', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-tools'; + const toolLine = JSON.stringify({ + gitBranch: 'main', + timestamp: '2026-10-01T00:00:00.000Z', + message: { + role: 'assistant', + content: [ + { type: 'tool_use', id: 'tool-1', name: 'Write', input: { file_path: '/repo/a.ts' } }, + { type: 'tool_use', id: 'tool-2', name: 'Edit', input: { file_path: '/repo/b.ts' } }, + ], + }, + }); + const resultLine = JSON.stringify({ + gitBranch: 'main', + timestamp: '2026-10-01T00:00:01.000Z', + message: { + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'tool-1', is_error: false }, + { type: 'tool_result', tool_use_id: 'tool-2', is_error: true }, + ], + }, + }); + const transcriptPath = writeTranscript('transcript-tools.jsonl', [toolLine, resultLine]); + + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + + const summary = events.find((e) => e.type === 'agent.session.summary'); + expect(summary).toBeDefined(); + expect(summary?.files_written).toEqual(['/repo/a.ts']); + expect(summary?.files_changed).toEqual(['/repo/b.ts']); + expect((summary?.tool_calls as Record).Write).toBe(1); + expect((summary?.tool_calls as Record).Edit).toBe(1); + expect((summary?.tool_errors as Record).Edit).toBe(1); + expect((summary?.tool_errors as Record).Write).toBe(0); + }); +}); + +describe('collectSubagentTranscriptEvents — SessionEnd backstop (three subagents, one pre-advanced)', () => { + it('returns exactly three agent.subagent.usage events — one per subagent, including the one whose own SubagentStop already advanced its offset — never a fourth', async () => { + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + const { findSubagentFiles } = await import('../subagent-usage.js'); + + const sessionId = 'session-backstop'; + const mainTranscriptPath = writeTranscript(`${sessionId}.jsonl`, [noUsageLine('uuid-main')]); + + writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'uuid-a1-1', messageId: 'msg-a1-1', outputTokens: 10 }), + ]); + writeSubagentFixture(sessionId, 'a2', [ + usageLine({ uuid: 'uuid-a2-1', messageId: 'msg-a2-1', outputTokens: 20 }), + ]); + writeSubagentFixture(sessionId, 'a3', [ + usageLine({ uuid: 'uuid-a3-1', messageId: 'msg-a3-1', outputTokens: 30 }), + ]); + + // Simulate a1's own SubagentStop having already fired and advanced its offset past its + // content (and already forwarded its own agent.subagent.usage event once). + const filesBeforeBackstop = await findSubagentFiles(mainTranscriptPath); + const a1File = filesBeforeBackstop.find((f) => f.agentId === 'a1'); + if (!a1File) throw new Error('fixture missing a1'); + await collectSubagentTranscriptEvents(sessionId, a1File); + + // Exercise exactly what the plugin's SessionEnd branch does: discover every subagent file + // for the session and re-run the subagent parse for each one, unconditionally — the + // crashed/missed-hook backstop. + const allFiles = await findSubagentFiles(mainTranscriptPath); + expect(allFiles).toHaveLength(3); + const backstopEvents: ForwardedEvent[] = []; + for (const file of allFiles) { + backstopEvents.push(...parseAll(await collectSubagentTranscriptEvents(sessionId, file))); + } + + const subagentUsageEvents = backstopEvents.filter((e) => e.type === 'agent.subagent.usage'); + // Exactly three — a1 (already-advanced, re-summarized rather than skipped), a2, a3. Never a + // fourth (no duplicate re-send for a1). + expect(subagentUsageEvents).toHaveLength(3); + + const agentIds = subagentUsageEvents.map((e) => e.agent_id).sort(); + expect(agentIds).toEqual(['a1', 'a2', 'a3']); + }); +}); + +describe('collectSubagentTranscriptEvents — no new bytes since last run', () => { + it('returns zero new agent.usage.request events on a no-op reparse, but still exactly one agent.subagent.usage event summarizing unchanged cumulative usage', async () => { + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + const { findSubagentFiles } = await import('../subagent-usage.js'); + + const sessionId = 'session-no-new-bytes'; + const mainTranscriptPath = writeTranscript(`${sessionId}.jsonl`, [noUsageLine('uuid-main')]); + writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'uuid-a1-1', messageId: 'msg-a1-1', outputTokens: 10 }), + usageLine({ uuid: 'uuid-a1-2', messageId: 'msg-a1-2', outputTokens: 20 }), + ]); + + const [file] = await findSubagentFiles(mainTranscriptPath); + + const firstEvents = parseAll(await collectSubagentTranscriptEvents(sessionId, file)); + expect(firstEvents.filter((e) => e.type === 'agent.usage.request')).toHaveLength(2); + expect(firstEvents.filter((e) => e.type === 'agent.subagent.usage')).toHaveLength(1); + + // Re-run on the same subagent file with no new content appended since the last call (its + // offset is now at EOF). + const secondEvents = parseAll(await collectSubagentTranscriptEvents(sessionId, file)); + const secondUsageRequests = secondEvents.filter((e) => e.type === 'agent.usage.request'); + const secondSubagentUsage = secondEvents.filter((e) => e.type === 'agent.subagent.usage'); + + // No new lines means no new/touched openRequests keys at the agent.usage.request layer. + expect(secondUsageRequests).toHaveLength(0); + // But the agent.subagent.usage event is still returned exactly once, summarizing the same + // cumulative (unchanged) usage. + expect(secondSubagentUsage).toHaveLength(1); + expect(secondSubagentUsage[0].output_tokens).toBe(30); + expect(secondSubagentUsage[0].api_calls).toBe(2); + }); +}); + +describe('collectSubagentTranscriptEvents — missing subagent transcript file', () => { + it('resolves without throwing and still returns one empty-usage agent.subagent.usage event, with no agent.usage.request events', async () => { + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-subagent-missing'; + const missingSubagentFile = { + agentId: 'ghost', + filePath: join(transcriptDir, sessionId, 'subagents', 'agent-ghost.jsonl'), + }; + + const events = parseAll( + await collectSubagentTranscriptEvents(sessionId, missingSubagentFile) + ); + const usageRequestEvents = events.filter((e) => e.type === 'agent.usage.request'); + const subagentUsageEvents = events.filter((e) => e.type === 'agent.subagent.usage'); + + expect(usageRequestEvents).toHaveLength(0); + expect(subagentUsageEvents).toHaveLength(1); + expect(subagentUsageEvents[0].api_calls).toBe(0); + expect(subagentUsageEvents[0].input_tokens).toBe(0); + expect(subagentUsageEvents[0].started_at).toBe(''); + expect(subagentUsageEvents[0].duration_ms).toBe(0); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts new file mode 100644 index 000000000..6f5e9ac35 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts @@ -0,0 +1,117 @@ +/** + * Tests for transcript parse-state persistence: `createParseState`, `loadParseState`, + * and `saveParseState`. + * + * `loadParseState` must never throw — a missing file or corrupt JSON on disk both fall + * back to a fresh state, since later tasks drive transcript parsing off whatever this + * returns and cannot tolerate a thrown exception interrupting that loop. + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { + createParseState, + loadParseState, + saveParseState, + withParseStateLock, + type OpenUsageRequest, + type TranscriptParseState, +} from '../parse-state.js'; + +let codemieHome: string; + +beforeEach(() => { + codemieHome = mkdtempSync(join(tmpdir(), 'codemie-home-')); + process.env.CODEMIE_HOME = codemieHome; +}); + +afterEach(() => { + delete process.env.CODEMIE_HOME; + rmSync(codemieHome, { recursive: true, force: true }); +}); + +describe('createParseState', () => { + it('returns the fresh shape', () => { + expect(createParseState()).toEqual({ + mainOffset: 0, + subagentOffsets: {}, + openRequests: {}, + activeSkill: '', + branchCounts: {}, + compactionCount: 0, + }); + }); +}); + +describe('loadParseState', () => { + it('returns the fresh shape when no state file exists', async () => { + const state = await loadParseState('session-missing'); + + expect(state).toEqual(createParseState()); + }); + + it('recovers to the fresh shape instead of throwing on corrupt JSON', async () => { + const stateDir = join(codemieHome, 'analytics', 'state'); + mkdirSync(stateDir, { recursive: true }); + writeFileSync(join(stateDir, 'session-corrupt.json'), '{not valid json'); + + const state = await loadParseState('session-corrupt'); + + expect(state).toEqual(createParseState()); + }); + + it('round-trips openRequests and branchCounts exactly through saveParseState', async () => { + const openRequest: OpenUsageRequest = { + requestId: 'req1', + model: 'claude-3-5-sonnet', + modelRaw: 'claude-3-5-sonnet-20241022', + timestamp: '2026-10-01T00:00:00.000Z', + speed: 'standard', + inferenceGeo: 'us', + serviceTier: 'standard', + inputTokens: 100, + cacheCreation5mTokens: 10, + cacheCreation1hTokens: 0, + cacheReadTokens: 5, + outputTokens: 200, + webSearchRequests: 1, + webFetchRequests: 0, + scopeKind: 'main', + scopeName: '', + agentId: '', + stopReason: 'end_turn', + isApiError: false, + gitBranch: 'main', + }; + + const state: TranscriptParseState = { + mainOffset: 42, + subagentOffsets: { 'sub-1': 7 }, + openRequests: { 'req1::claude-3-5-sonnet': openRequest }, + activeSkill: 'brainstorming', + branchCounts: { main: 3, feature: 1 }, + compactionCount: 2, + }; + + await saveParseState('session-roundtrip', state); + const loaded = await loadParseState('session-roundtrip'); + + expect(loaded.openRequests).toEqual(state.openRequests); + expect(loaded.branchCounts).toEqual(state.branchCounts); + expect(loaded).toEqual(state); + }); +}); + +describe('withParseStateLock', () => { + it('still runs fn when the lock cannot be released cleanly, and leaves no stale lock file behind', async () => { + const sessionId = 'session-lock-cleanup'; + const result = await withParseStateLock(sessionId, async () => 'done'); + expect(result).toBe('done'); + + // A second acquisition must not be blocked by a lock the first call failed to clean up. + const second = await withParseStateLock(sessionId, async () => 'done-again'); + expect(second).toBe('done-again'); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts new file mode 100644 index 000000000..0bb5e3de9 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts @@ -0,0 +1,226 @@ +/** + * Tests for the `agent.session.summary` builder: `updateBranchCounts`, `primaryModel`, + * `branchDominant`, and `buildSessionSummaryEvent`. + */ + +import { describe, it, expect } from 'vitest'; +import { + updateBranchCounts, + primaryModel, + branchDominant, + buildSessionSummaryEvent, + type SessionSummaryAccumulator, +} from '../session-summary.js'; +import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; + +function emptyAccumulator(): SessionSummaryAccumulator { + return { + models: {}, + toolCalls: {}, + linesAdded: 0, + linesRemoved: 0, + filesChanged: new Set(), + filesWritten: new Set(), + compactionCount: 0, + }; +} + +function emptyNamed(): NamedInvocationCounts { + return { skillInvocations: {}, agentInvocations: {}, commandInvocations: {} }; +} + +describe('branchDominant', () => { + it('resolves the highest-count branch (mid-session branch-switch scenario)', () => { + expect(branchDominant({ main: 3, feature: 7 })).toBe('feature'); + }); + + it('returns "" for an empty map', () => { + expect(branchDominant({})).toBe(''); + }); +}); + +describe('primaryModel', () => { + it('returns the highest-count model key', () => { + expect(primaryModel({ 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 })).toBe('claude-opus-4-1'); + }); + + it('returns "" for an empty map', () => { + expect(primaryModel({})).toBe(''); + }); +}); + +describe('updateBranchCounts', () => { + it('mutates the passed-in counts object in place, bumping the named branch by 1 each call', () => { + const counts: Record = {}; + + updateBranchCounts(counts, 'main'); + expect(counts.main).toBe(1); + + updateBranchCounts(counts, 'main'); + expect(counts.main).toBe(2); + }); + + it('skips incrementing when branch is falsy/empty', () => { + const counts: Record = {}; + + updateBranchCounts(counts, ''); + + expect(counts['']).toBeUndefined(); + expect(Object.keys(counts)).toHaveLength(0); + }); +}); + +describe('buildSessionSummaryEvent', () => { + it('omits ended_at entirely for phase "incremental"', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect('ended_at' in event).toBe(false); + expect(Object.keys(event)).not.toContain('ended_at'); + expect(event.type).toBe('agent.session.summary'); + expect(event.session_id).toBe('session-1'); + expect(event.phase).toBe('incremental'); + expect(event.started_at).toBe('2026-10-01T00:00:00.000Z'); + }); + + it('includes ended_at for phase "final"', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect('ended_at' in event).toBe(true); + expect(event.ended_at).toBe('2026-10-01T01:00:00.000Z'); + expect(event.phase).toBe('final'); + }); + + it('flattens acc.toolCalls {calls, errors} shape into separate tool_calls/tool_errors maps', () => { + const acc = emptyAccumulator(); + acc.toolCalls = { + Read: { calls: 5, errors: 0 }, + Edit: { calls: 3, errors: 1 }, + }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.tool_calls).toEqual({ Read: 5, Edit: 3 }); + expect(event.tool_errors).toEqual({ Read: 0, Edit: 1 }); + }); + + it('derives skills_used and primary_command from a constructed NamedInvocationCounts', () => { + const named: NamedInvocationCounts = { + skillInvocations: { 'codemie:msgraph': 2, brainstorming: 1 }, + agentInvocations: {}, + commandInvocations: { init: 1, deploy: 4 }, + }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + named, + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.skills_used).toEqual({ 'codemie:msgraph': 2, brainstorming: 1 }); + expect(event.primary_command).toBe('deploy'); + expect(event.commands_in_order).toEqual(Object.keys(named.commandInvocations)); + }); + + it('reports models_used as the full count map and primary_model as the max key', () => { + const acc = emptyAccumulator(); + acc.models = { 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect(event.models_used).toEqual({ 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 }); + expect(event.primary_model).toBe('claude-opus-4-1'); + }); + + it('reports lines/files/compaction fields and branch_counts/branch_dominant, converting Sets to arrays', () => { + const acc = emptyAccumulator(); + acc.linesAdded = 42; + acc.linesRemoved = 7; + acc.filesChanged = new Set(['a.ts', 'b.ts']); + acc.filesWritten = new Set(['a.ts']); + acc.compactionCount = 2; + + const branchCounts = { main: 3, feature: 7 }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + branchCounts, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.lines_added).toBe(42); + expect(event.lines_removed).toBe(7); + expect(event.files_changed).toEqual(['a.ts', 'b.ts']); + expect(event.files_written).toEqual(['a.ts']); + expect(event.compaction_count).toBe(2); + expect(event.branch_counts).toBe(branchCounts); + expect(event.branch_dominant).toBe('feature'); + }); + + it('emits title as a literal empty string (no identified source, per Open risks)', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect(event.title).toBe(''); + }); + + it('does not include an apiCalls/api_calls field (left to the caller/orchestrator to merge)', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect('api_calls' in event).toBe(false); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts new file mode 100644 index 000000000..9d607ce2b --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts @@ -0,0 +1,295 @@ +/** + * Tests for the `agent.subagent.usage` builder: `findSubagentFiles` and + * `buildSubagentUsageEvent`. + * + * Fixture layout, built fresh per test under a temp dir: + * /.jsonl — trivial main-transcript placeholder + * //subagents/agent-.jsonl — one subagent transcript per fixture + * //subagents/agent-.meta.json — sidecar (toolUseId/agentType/spawnDepth) + * + * Three subagents are used throughout: + * - "a1": sidecar OMITS spawnDepth (top-level subagent — spawn_depth must default to 0), + * two usage-bearing transcript lines (exercises summing across >1 request). + * - "a2": sidecar INCLUDES spawnDepth: 2 (nested subagent — pass-through), one usage line. + * - "a3": sidecar includes toolUseId/agentType, one usage line. + */ + +import { describe, it, expect, afterEach } from 'vitest'; +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { findSubagentFiles, buildSubagentUsageEvent, type SubagentFile } from '../subagent-usage.js'; +import { parseUsageLine, buildUsageRequestEvent } from '../usage-request.js'; +import type { OpenUsageRequest } from '../parse-state.js'; + +let tmpDir: string | undefined; + +afterEach(async () => { + if (tmpDir) { + await rm(tmpDir, { recursive: true, force: true }); + tmpDir = undefined; + } +}); + +/** Build one transcript JSONL usage line, mirroring the confirmed transcript shape. */ +function usageLine(messageId: string, inputTokens: number, outputTokens: number): string { + return JSON.stringify({ + sessionId: 'session-subagent-1', + gitBranch: 'main', + cwd: '/repo', + timestamp: '2026-10-01T00:00:00.000Z', + version: '1.2.3', + entrypoint: 'cli', + uuid: `uuid-${messageId}`, + parentUuid: null, + isSidechain: true, + userType: 'external', + message: { + id: messageId, + role: 'assistant', + model: 'claude-sonnet-4-5-20250929', + stop_reason: 'end_turn', + usage: { + input_tokens: inputTokens, + output_tokens: outputTokens, + cache_read_input_tokens: 2, + cache_creation_input_tokens: 0, + service_tier: 'standard', + speed: 'standard', + inference_geo: '', + cache_creation: { ephemeral_1h_input_tokens: 1, ephemeral_5m_input_tokens: 1 }, + server_tool_use: { web_search_requests: 0, web_fetch_requests: 0 }, + iterations: [], + }, + }, + }); +} + +interface FixtureMeta { + toolUseId?: string; + agentType?: string; + spawnDepth?: number; +} + +async function buildFixture( + sessionId: string, + subagents: Record +): Promise { + const dir = await mkdtemp(join(tmpdir(), 'codemie-subagent-usage-')); + const mainTranscriptPath = join(dir, `${sessionId}.jsonl`); + await writeFile(mainTranscriptPath, ''); + + const subagentsDir = join(dir, sessionId, 'subagents'); + await mkdir(subagentsDir, { recursive: true }); + + for (const [agentId, { lines, meta }] of Object.entries(subagents)) { + await writeFile(join(subagentsDir, `agent-${agentId}.jsonl`), `${lines.join('\n')}\n`); + await writeFile(join(subagentsDir, `agent-${agentId}.meta.json`), JSON.stringify(meta)); + } + + return mainTranscriptPath; +} + +describe('findSubagentFiles', () => { + it('discovers all subagent files with their sidecar fields, defaulting spawnDepth to undefined when the sidecar omits it', async () => { + const sessionId = 'session-subagent-1'; + const mainTranscriptPath = await buildFixture(sessionId, { + a1: { lines: [usageLine('msg-a1-1', 100, 50)], meta: { toolUseId: 'tool-a1' } }, + a2: { lines: [usageLine('msg-a2-1', 10, 5)], meta: { agentType: 'reviewer', spawnDepth: 2 } }, + a3: { lines: [usageLine('msg-a3-1', 7, 3)], meta: { toolUseId: 'tool-a3', agentType: 'coder' } }, + }); + tmpDir = join(mainTranscriptPath, '..'); + + const files = await findSubagentFiles(mainTranscriptPath); + + expect(files).toHaveLength(3); + + const byId = (id: string): SubagentFile => { + const found = files.find((f) => f.agentId === id); + if (!found) throw new Error(`fixture missing ${id}`); + return found; + }; + + const a1 = byId('a1'); + expect(a1.toolUseId).toBe('tool-a1'); + expect(a1.agentType).toBeUndefined(); + expect(a1.spawnDepth).toBeUndefined(); + + const a2 = byId('a2'); + expect(a2.agentType).toBe('reviewer'); + expect(a2.spawnDepth).toBe(2); + expect(a2.toolUseId).toBeUndefined(); + + const a3 = byId('a3'); + expect(a3.toolUseId).toBe('tool-a3'); + expect(a3.agentType).toBe('coder'); + expect(a3.spawnDepth).toBeUndefined(); + }); + + it('returns [] when the subagents directory does not exist, without throwing', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-subagent-usage-empty-')); + const mainTranscriptPath = join(tmpDir, 'session-empty.jsonl'); + await writeFile(mainTranscriptPath, ''); + + const files = await findSubagentFiles(mainTranscriptPath); + + expect(files).toEqual([]); + }); + + it('returns [] for a missing main transcript path, without throwing', async () => { + const files = await findSubagentFiles('C:/nonexistent/path/session-missing.jsonl'); + expect(files).toEqual([]); + }); +}); + +describe('buildSubagentUsageEvent', () => { + it('sums token/cache fields across usageRequests, defaults spawn_depth to 0 when the file omits it, and passes caller-built maps through verbatim', () => { + const file: SubagentFile = { agentId: 'a1', filePath: '/tmp/agent-a1.jsonl', toolUseId: 'tool-a1' }; + const reqs: OpenUsageRequest[] = [ + { + requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + speed: 'standard', inferenceGeo: '', serviceTier: 'standard', + inputTokens: 100, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, + cacheReadTokens: 3, outputTokens: 50, webSearchRequests: 1, webFetchRequests: 0, + scopeKind: 'agent', scopeName: '', agentId: 'a1', + stopReason: 'end_turn', isApiError: false, gitBranch: 'main', + }, + { + requestId: 'r2', model: 'm', modelRaw: 'm-raw', timestamp: 't2', + speed: 'standard', inferenceGeo: '', serviceTier: 'standard', + inputTokens: 10, cacheCreation5mTokens: 4, cacheCreation1hTokens: 0, + cacheReadTokens: 1, outputTokens: 5, webSearchRequests: 0, webFetchRequests: 2, + scopeKind: 'agent', scopeName: '', agentId: 'a1', + stopReason: 'tool_use', isApiError: false, gitBranch: 'main', + }, + ]; + const toolCalls = { Read: 3, Edit: 1 }; + const toolErrors = { Edit: 1 }; + const skillsInvoked = { brainstorming: 1 }; + + const event = buildSubagentUsageEvent( + 'session-1', file, reqs, toolCalls, toolErrors, skillsInvoked, '2026-10-01T00:00:00.000Z', 1500 + ); + + expect(event).toEqual({ + type: 'agent.subagent.usage', + session_id: 'session-1', + agent_id: 'a1', + tool_use_id: 'tool-a1', + agent_type: '', + spawn_depth: 0, + description: '', + workflow_run: '', + worktree: '', + started_at: '2026-10-01T00:00:00.000Z', + duration_ms: 1500, + input_tokens: 110, + cache_creation_5m_tokens: 5, + cache_creation_1h_tokens: 2, + cache_read_tokens: 4, + output_tokens: 55, + web_search_requests: 1, + web_fetch_requests: 2, + api_calls: 2, + tool_calls: toolCalls, + tool_errors: toolErrors, + skills_invoked: skillsInvoked, + }); + }); + + it('passes a present spawn_depth through verbatim instead of defaulting to 0', () => { + const file: SubagentFile = { agentId: 'a2', filePath: '/tmp/agent-a2.jsonl', spawnDepth: 2 }; + + const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, {}, '', 0); + + expect(event.spawn_depth).toBe(2); + expect(event.api_calls).toBe(0); + }); + + it('never fabricates description/workflow_run/worktree — always empty string', () => { + const file: SubagentFile = { agentId: 'a3', filePath: '/tmp/agent-a3.jsonl' }; + + const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, {}, '', 0); + + expect(event.description).toBe(''); + expect(event.workflow_run).toBe(''); + expect(event.worktree).toBe(''); + }); +}); + +describe('cross-check: agent.subagent.usage summed tokens vs agent.usage.request (scope_kind: agent)', () => { + it('summing token fields across three agent.subagent.usage events equals summing the same fields across every agent.usage.request record derived from the same fixture', async () => { + const sessionId = 'session-subagent-cross'; + const mainTranscriptPath = await buildFixture(sessionId, { + a1: { + lines: [usageLine('msg-a1-1', 100, 50), usageLine('msg-a1-2', 10, 5)], + meta: { toolUseId: 'tool-a1' }, + }, + a2: { + lines: [usageLine('msg-a2-1', 200, 80)], + meta: { agentType: 'reviewer', spawnDepth: 2 }, + }, + a3: { + lines: [usageLine('msg-a3-1', 30, 15)], + meta: { toolUseId: 'tool-a3', agentType: 'coder' }, + }, + }); + tmpDir = join(mainTranscriptPath, '..'); + + const files = await findSubagentFiles(mainTranscriptPath); + expect(files).toHaveLength(3); + + const subagentUsageEvents: Array> = []; + const usageRequestEvents: Array> = []; + + for (const file of files) { + const raw = await import('node:fs/promises').then((m) => m.readFile(file.filePath, 'utf-8')); + const lines = raw.split('\n').filter((l) => l.trim().length > 0); + + const reqs: OpenUsageRequest[] = lines + .map((line) => parseUsageLine(line, 'agent', '', file.agentId)) + .filter((r): r is OpenUsageRequest => r !== null); + + expect(reqs.length).toBeGreaterThan(0); + reqs.forEach((r) => expect(r.scopeKind).toBe('agent')); + + subagentUsageEvents.push( + buildSubagentUsageEvent(sessionId, file, reqs, {}, {}, {}, '2026-10-01T00:00:00.000Z', 0) + ); + + for (const req of reqs) { + usageRequestEvents.push(buildUsageRequestEvent(sessionId, req)); + } + } + + expect(subagentUsageEvents).toHaveLength(3); + // 2 + 1 + 1 usage-bearing lines across the three subagent transcripts. + expect(usageRequestEvents).toHaveLength(4); + usageRequestEvents.forEach((e) => expect(e.scope_kind).toBe('agent')); + + const sumField = (records: Array>, field: string): number => + records.reduce((total, r) => total + Number(r[field] ?? 0), 0); + + const tokenFieldPairs: Array<[string, string]> = [ + ['input_tokens', 'input_tokens'], + ['output_tokens', 'output_tokens'], + ['cache_creation_5m_tokens', 'cache_creation_5m_tokens'], + ['cache_creation_1h_tokens', 'cache_creation_1h_tokens'], + ['cache_read_tokens', 'cache_read_tokens'], + ['web_search_requests', 'web_search_requests'], + ['web_fetch_requests', 'web_fetch_requests'], + ]; + + for (const [subagentField, requestField] of tokenFieldPairs) { + expect(sumField(subagentUsageEvents, subagentField)).toBe(sumField(usageRequestEvents, requestField)); + } + + // api_calls across the three subagent.usage events equals the total number of + // agent.usage.request records derived from the same fixture. + expect(sumField(subagentUsageEvents, 'api_calls')).toBe(usageRequestEvents.length); + + // Known concrete totals: input 100+10+200+30=340, output 50+5+80+15=150. + expect(sumField(subagentUsageEvents, 'input_tokens')).toBe(340); + expect(sumField(subagentUsageEvents, 'output_tokens')).toBe(150); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts new file mode 100644 index 000000000..1348d59de --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts @@ -0,0 +1,100 @@ +/** + * Tests for the incremental transcript reader: `readNewLines`. + * + * The reader must only ever return complete, newline-terminated lines, cutting + * at the last `\n` byte so a partially-written trailing line is never handed + * back to the caller — mirrors the safe-cut rule in + * `src/providers/plugins/sso/proxy/plugins/otlp-spool/spool-io.ts`'s + * `snapshotPendingHookRecords`. It must never throw, even for a missing file. + */ + +import { describe, it, expect, afterEach } from 'vitest'; +import { mkdtemp, writeFile, appendFile, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { readNewLines } from '../transcript-reader.js'; + +let tmpDir: string | undefined; + +afterEach(async () => { + if (tmpDir) { + await rm(tmpDir, { recursive: true, force: true }); + tmpDir = undefined; + } +}); + +describe('readNewLines', () => { + it('returns only complete lines and nextOffset points exactly after the last complete line; a second call returns only newly appended lines', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + const completePrefix = 'line1\nline2\n'; + await writeFile(filePath, `${completePrefix}partial-line-no-newli`); + + const first = await readNewLines(filePath, 0); + + expect(first.lines).toEqual(['line1', 'line2']); + expect(first.nextOffset).toBe(Buffer.byteLength(completePrefix)); + + // Complete the previously-partial line and add a new complete line. + await appendFile(filePath, 'ne\nline4\n'); + + const second = await readNewLines(filePath, first.nextOffset); + + expect(second.lines).toEqual(['partial-line-no-newline', 'line4']); + expect(second.nextOffset).toBeGreaterThan(first.nextOffset); + }); + + it('returns an empty result without throwing when the file is missing', async () => { + const result = await readNewLines( + 'C:/nonexistent/path/that/does/not/exist.jsonl', + 0 + ); + + expect(result).toEqual({ lines: [], nextOffset: 0 }); + }); + + it('returns an empty result and does not advance the offset when the file has only an unterminated partial line', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + await writeFile(filePath, 'no-newline-yet'); + + const result = await readNewLines(filePath, 0); + + expect(result).toEqual({ lines: [], nextOffset: 0 }); + }); + + it('resumes a fresh parse from 0 after the file is rotated/truncated below the persisted offset', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + await writeFile(filePath, 'line1\nline2\nline3\n'); + const first = await readNewLines(filePath, 0); + expect(first.lines).toEqual(['line1', 'line2', 'line3']); + + // Rotation: the file is replaced by a much shorter one, so its size now sits below the + // previously persisted offset. + await writeFile(filePath, 'new1\nnew2\n'); + + const second = await readNewLines(filePath, first.nextOffset); + + expect(second.lines).toEqual(['new1', 'new2']); + expect(second.nextOffset).toBe(Buffer.byteLength('new1\nnew2\n')); + }); + + it('returns an empty result when nothing new has been written since fromOffset', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + const content = 'line1\nline2\n'; + await writeFile(filePath, content); + + const result = await readNewLines(filePath, Buffer.byteLength(content)); + + expect(result).toEqual({ + lines: [], + nextOffset: Buffer.byteLength(content), + }); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts new file mode 100644 index 000000000..210b2f55f --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts @@ -0,0 +1,216 @@ +/** + * Tests for `agent.usage.request` extraction and merge: `parseUsageLine`, + * `mergeUsageRequest`, and `buildUsageRequestEvent`. + * + * Fixture: `fixtures/transcript-usage.jsonl` — one line with no `message.usage` + * (must parse to `null`), two lines sharing the same `message.id` where the + * second has a higher `output_tokens` and a non-empty `stop_reason` the first + * lacks (feeds `mergeUsageRequest`), and one fully-populated "normal" line + * (round-trips every field, including the `model`/`modelRaw` resolution split). + */ + +import { describe, it, expect, beforeAll } from 'vitest'; +import { readFile } from 'node:fs/promises'; +import { join } from 'node:path'; +import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from '../usage-request.js'; +import type { OpenUsageRequest } from '../parse-state.js'; + +let lines: string[]; + +beforeAll(async () => { + const raw = await readFile(join(__dirname, 'fixtures', 'transcript-usage.jsonl'), 'utf-8'); + lines = raw.split('\n').filter((l) => l.trim().length > 0); +}); + +describe('parseUsageLine', () => { + it('returns null for a line with no message.usage block', () => { + expect(lines).toHaveLength(4); + const result = parseUsageLine(lines[0], 'main', '', ''); + expect(result).toBeNull(); + }); + + it('returns null for malformed JSON instead of throwing', () => { + expect(() => parseUsageLine('not valid json {{{', 'main', '', '')).not.toThrow(); + expect(parseUsageLine('not valid json {{{', 'main', '', '')).toBeNull(); + }); + + it('returns null for a usage-bearing line with no message.id, instead of collapsing it onto a shared ::model key', () => { + const line = JSON.stringify({ + timestamp: '2026-10-01T00:00:04.000Z', + message: { + role: 'assistant', + model: 'claude-sonnet-4-5-20250929', + usage: { input_tokens: 10, output_tokens: 5 }, + }, + }); + + expect(parseUsageLine(line, 'main', '', '')).toBeNull(); + }); + + it('extracts every field from a fully-populated line', () => { + const req = parseUsageLine(lines[3], 'main', '', ''); + + expect(req).not.toBeNull(); + const r = req as OpenUsageRequest; + + // request_id comes from message.id, not any top-level requestId. + expect(r.requestId).toBe('msg_normal_1'); + // modelRaw is the transcript's own literal message.model (unresolved). + expect(r.modelRaw).toBe('claude-sonnet-4-5-20250929'); + // model is resolved via parseBackendModelName (x-litellm-model-name) first. + expect(r.model).toBe('bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0'); + expect(r.timestamp).toBe('2026-10-01T00:00:03.000Z'); + expect(r.speed).toBe('fast'); + expect(r.inferenceGeo).toBe('us'); + expect(r.serviceTier).toBe('priority'); + expect(r.inputTokens).toBe(300); + expect(r.outputTokens).toBe(75); + expect(r.cacheReadTokens).toBe(40); + // Nested cache_creation.* fields, not the flat cache_creation_input_tokens. + expect(r.cacheCreation5mTokens).toBe(5); + expect(r.cacheCreation1hTokens).toBe(10); + // Nested server_tool_use.* fields. + expect(r.webSearchRequests).toBe(2); + expect(r.webFetchRequests).toBe(1); + // stop_reason is a sibling of usage on message, not nested inside usage. + expect(r.stopReason).toBe('end_turn'); + expect(r.gitBranch).toBe('feature/epmcdme-15301'); + // No documented source field for isApiError on a real transcript line — defaults false. + expect(r.isApiError).toBe(false); + expect(r.scopeKind).toBe('main'); + expect(r.scopeName).toBe(''); + expect(r.agentId).toBe(''); + }); + + it('passes scopeKind/scopeName/agentId through verbatim from its own parameters', () => { + const req = parseUsageLine(lines[3], 'agent', 'reviewer', 'agent-42'); + + expect(req?.scopeKind).toBe('agent'); + expect(req?.scopeName).toBe('reviewer'); + expect(req?.agentId).toBe('agent-42'); + }); +}); + +describe('mergeUsageRequest', () => { + it('keeps the max output_tokens and the non-empty stop_reason across two records for the same message.id', () => { + const first = parseUsageLine(lines[1], 'main', '', ''); + const second = parseUsageLine(lines[2], 'main', '', ''); + + expect(first).not.toBeNull(); + expect(second).not.toBeNull(); + const a = first as OpenUsageRequest; + const b = second as OpenUsageRequest; + + expect(a.requestId).toBe(b.requestId); + expect(a.outputTokens).toBe(50); + expect(a.stopReason).toBe(''); + expect(b.outputTokens).toBe(120); + expect(b.stopReason).toBe('tool_use'); + + const merged = mergeUsageRequest(a, b); + + expect(merged.outputTokens).toBe(Math.max(a.outputTokens, b.outputTokens)); + expect(merged.outputTokens).toBe(120); + expect(merged.stopReason).toBe('tool_use'); + // Numeric fields take the max even when equal. + expect(merged.inputTokens).toBe(Math.max(a.inputTokens, b.inputTokens)); + // Merge returns a new object — neither input is mutated. + expect(a.outputTokens).toBe(50); + expect(b.outputTokens).toBe(120); + }); + + it('takes every numeric field as Math.max of the two inputs', () => { + const a: OpenUsageRequest = { + requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + speed: 'standard', inferenceGeo: '', serviceTier: 'standard', + inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, + cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, + scopeKind: 'main', scopeName: '', agentId: '', + stopReason: '', isApiError: false, gitBranch: 'main', + }; + const b: OpenUsageRequest = { + ...a, + inputTokens: 1, cacheCreation5mTokens: 9, cacheCreation1hTokens: 1, + cacheReadTokens: 30, outputTokens: 1, webSearchRequests: 0, webFetchRequests: 60, + timestamp: '', + }; + + const merged = mergeUsageRequest(a, b); + + expect(merged.inputTokens).toBe(10); + expect(merged.cacheCreation5mTokens).toBe(9); + expect(merged.cacheCreation1hTokens).toBe(2); + expect(merged.cacheReadTokens).toBe(30); + expect(merged.outputTokens).toBe(4); + expect(merged.webSearchRequests).toBe(5); + expect(merged.webFetchRequests).toBe(60); + // Non-numeric fields: b's value wins when non-empty, else a's. + expect(merged.timestamp).toBe('t1'); + }); + + it('does not mutate either input and returns a new object', () => { + const a: OpenUsageRequest = { + requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + speed: '', inferenceGeo: '', serviceTier: '', + inputTokens: 1, cacheCreation5mTokens: 0, cacheCreation1hTokens: 0, + cacheReadTokens: 0, outputTokens: 1, webSearchRequests: 0, webFetchRequests: 0, + scopeKind: 'main', scopeName: '', agentId: '', + stopReason: '', isApiError: false, gitBranch: '', + }; + const b: OpenUsageRequest = { ...a, outputTokens: 2, stopReason: 'end_turn', isApiError: true }; + const aCopy = { ...a }; + const bCopy = { ...b }; + + const merged = mergeUsageRequest(a, b); + + expect(a).toEqual(aCopy); + expect(b).toEqual(bCopy); + expect(merged).not.toBe(a); + expect(merged).not.toBe(b); + // isApiError: once true, stays true across merges. + expect(merged.isApiError).toBe(true); + }); +}); + +describe('buildUsageRequestEvent', () => { + it('maps every OpenUsageRequest field to its snake_case event field, with an explicit type', () => { + const req: OpenUsageRequest = { + requestId: 'req-1', model: 'resolved-model', modelRaw: 'literal-model', timestamp: '2026-10-01T00:00:00.000Z', + speed: 'fast', inferenceGeo: 'us', serviceTier: 'priority', + inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, + cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, + scopeKind: 'skill', scopeName: 'brainstorming', agentId: 'agent-7', + stopReason: 'end_turn', isApiError: false, gitBranch: 'main', + }; + + const event = buildUsageRequestEvent('session-123', req); + + expect(event).toEqual({ + type: 'agent.usage.request', + session_id: 'session-123', + request_id: 'req-1', + model_raw: 'literal-model', + model: 'resolved-model', + speed: 'fast', + inference_geo: 'us', + service_tier: 'priority', + input_tokens: 10, + cache_creation_5m_tokens: 1, + cache_creation_1h_tokens: 2, + cache_read_tokens: 3, + output_tokens: 4, + web_search_requests: 5, + web_fetch_requests: 6, + scope_kind: 'skill', + scope_name: 'brainstorming', + agent_id: 'agent-7', + stop_reason: 'end_turn', + is_api_error: false, + git_branch: 'main', + timestamp: '2026-10-01T00:00:00.000Z', + }); + // No event_id/schema_version here — those are stamped later, daemon-side. + expect(event).not.toHaveProperty('event_id'); + expect(event).not.toHaveProperty('schema_version'); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts new file mode 100644 index 000000000..9495cb26b --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -0,0 +1,447 @@ +/** + * Main-transcript trigger orchestration for the `Stop`, `PreCompact`, `SessionEnd`, and + * `StopFailure` hook events. + * + * Each hook fire is a fresh CLI process, so this module reloads persisted parse state + * (`./parse-state.js`), reads only the transcript lines appended since the last + * persisted `mainOffset` (`./transcript-reader.js`), derives/merges + * `agent.usage.request` records for those new lines (`./usage-request.js`), persists + * state back, and RETURNS one JSON string per completed request plus (on `Stop`/`SessionEnd`) + * one `agent.session.summary` event (`Stop`/`SessionEnd` only) — it never forwards anything to + * the spool itself. The caller + * (the plugin's `processOtlpEvent`, via its per-event handlers) owns forwarding, so there is + * exactly one place in the whole analytics pipeline that writes to the spool. + * + * Never throws: every path is wrapped so a read/parse failure degrades to an empty result rather + * than interrupting the hook that triggered it (`processOtlpEvent` must never block or fail on + * this). + */ + +import { readFile } from 'node:fs/promises'; +import { loadParseState, saveParseState, withParseStateLock } from './parse-state.js'; +import { readNewLines } from './transcript-reader.js'; +import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from './usage-request.js'; +import { + updateBranchCounts, + buildSessionSummaryEvent, + type SessionSummaryAccumulator, + type NamedInvocationCounts, +} from './session-summary.js'; +import { extractNamedInvocations } from '@/agents/plugins/claude/session/claude-named-invocations.js'; +import { type SubagentFile, buildSubagentUsageEvent } from './subagent-usage.js'; + +// Re-exported so callers (e.g. claude-code-otlp.plugin.ts) can import both `SubagentFile` and +// `collectSubagentTranscriptEvents` from this one module. +export type { SubagentFile }; + +export type MainTranscriptTrigger = 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'; + +interface ContentBlock { + type?: string; + id?: string; + name?: string; + tool_use_id?: string; + is_error?: boolean; + isError?: boolean; + input?: { file_path?: unknown; path?: unknown }; +} + +interface TranscriptLine { + timestamp?: string; + gitBranch?: string; + message?: { content?: unknown }; +} + +/** + * Collect the set of `tool_use_id` values whose matching `tool_result` block carries a truthy + * `is_error`/`isError` flag. + * + * Shared between {@link buildFullAccumulator} (main transcript) and + * {@link scanSubagentTranscript} (subagent transcript) so the error-correlation logic does not + * drift between the two call sites. + */ +function collectErrorToolUseIds(parsedLines: TranscriptLine[]): Set { + const errorByToolUseId = new Set(); + for (const parsed of parsedLines) { + const content = parsed.message?.content; + if (!Array.isArray(content)) continue; + for (const item of content as ContentBlock[]) { + if ( + item?.type === 'tool_result' && + typeof item.tool_use_id === 'string' && + (item.is_error === true || item.isError === true) + ) { + errorByToolUseId.add(item.tool_use_id); + } + } + } + return errorByToolUseId; +} + +function emptyAccumulator(): SessionSummaryAccumulator { + return { + models: {}, + toolCalls: {}, + linesAdded: 0, + linesRemoved: 0, + filesChanged: new Set(), + filesWritten: new Set(), + compactionCount: 0, + }; +} + +/** + * Recompute the full-session summary accumulator, named-invocation counts, and session start + * time from byte 0 of the main transcript. + * + * `TranscriptParseState`'s fixed shape has no persisted field for any of + * `SessionSummaryAccumulator`'s data or for `NamedInvocationCounts` — only `branchCounts` is + * incrementally tracked there. So, for `Stop`/`SessionEnd`, this helper re-derives everything + * else fresh from the whole transcript file every time. Transcripts are + * not enormous and this only runs on `Stop`/`SessionEnd`, not on every hook. + * + * Never throws: a missing/unreadable transcript resolves to the emptiest defensible result + * (empty accumulator, empty named-invocation counts, `startedAt: ''`); a malformed individual + * line is skipped rather than aborting the whole scan. + * + * Known limitations (no reliable in-transcript signal found for any of these): + * - `toolCalls[*].errors` is derived from a sibling `tool_result` block's `is_error`/`isError` + * flag (the same pattern `claude.session.ts`/`claude.metrics-processor.ts` already use for + * tool-use_id → error lookups) when one is found; otherwise a tool call's `.errors` stays 0. + * - `linesAdded`/`linesRemoved` default to 0 — an `Edit`/`Write` tool_use's `input` carries the + * *proposed* edit, not a diff stat, so no reliable added/removed line count can be derived from + * it without re-implementing diffing. + * - `compactionCount` defaults to 0 — no verified in-transcript signal was found (`PreCompact` is + * a hook event, not a transcript line). + */ +async function buildFullAccumulator( + transcriptPath: string +): Promise<{ acc: SessionSummaryAccumulator; named: NamedInvocationCounts; startedAt: string }> { + const acc = emptyAccumulator(); + + let raw: string; + try { + raw = await readFile(transcriptPath, 'utf-8'); + } catch { + return { acc, named: extractNamedInvocations([]), startedAt: '' }; + } + + const rawLines = raw.split('\n').filter((line) => line.trim().length > 0); + const parsedLines: TranscriptLine[] = []; + + for (const line of rawLines) { + try { + parsedLines.push(JSON.parse(line) as TranscriptLine); + } catch { + // Skip malformed lines rather than aborting the whole scan. + } + } + + // Pass 1: collect tool_result error flags keyed by their matching tool_use_id. + const errorByToolUseId = collectErrorToolUseIds(parsedLines); + + // Pass 2: models (reusing parseUsageLine's own model-resolution logic). + for (const line of rawLines) { + const parsedUsage = parseUsageLine(line, 'main', '', ''); + if (parsedUsage) { + acc.models[parsedUsage.model] = (acc.models[parsedUsage.model] ?? 0) + 1; + } + } + + // Pass 3: tool calls/errors, files changed/written (Edit/Write tool_use payloads). + for (const parsed of parsedLines) { + const content = parsed.message?.content; + if (!Array.isArray(content)) continue; + for (const item of content as ContentBlock[]) { + if (item?.type !== 'tool_use' || typeof item.name !== 'string') continue; + + const entry = acc.toolCalls[item.name] ?? { calls: 0, errors: 0 }; + entry.calls += 1; + if (typeof item.id === 'string' && errorByToolUseId.has(item.id)) { + entry.errors += 1; + } + acc.toolCalls[item.name] = entry; + + const filePath = item.input?.file_path ?? item.input?.path; + if (typeof filePath === 'string' && filePath) { + if (item.name === 'Write') acc.filesWritten.add(filePath); + if (item.name === 'Edit') acc.filesChanged.add(filePath); + } + } + } + + const named = extractNamedInvocations(parsedLines); + const startedAt = parsedLines.length > 0 ? String(parsedLines[0].timestamp ?? '') : ''; + + return { acc, named, startedAt }; +} + +/** + * Collect the spool-bound events for one `Stop`/`PreCompact`/`SessionEnd`/`StopFailure` hook fire. + * + * - Loads persisted state, reads only the lines appended since `state.mainOffset`. + * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, + * keyed by `${requestId}::${model}` (matching `parse-state.ts`'s documented key shape), and + * updates `state.branchCounts` from every new line's `gitBranch` (regardless of whether that + * line carried usage). + * - Returns one `agent.usage.request` JSON string per request key touched by this pass. + * - On `Stop`/`SessionEnd` only, also returns exactly one `agent.session.summary` event + * (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`) built from a fresh full-file + * recompute (see {@link buildFullAccumulator}). `PreCompact`/`StopFailure` never return a + * summary. + * - Persists state back to disk. + * + * Never forwards anything itself — the caller is responsible for sending the returned events to + * the spool (exactly one place in the pipeline does that). + * + * Scoping: no reliable transcript signal marks a *main*-transcript turn entering/exiting a + * "skill context", so every main-transcript usage record is scoped `scopeKind: 'main'`, + * `scopeName: ''`. `state.activeSkill` is deliberately neither read nor written. + * + * Swallows every error internally — never throws into `processOtlpEvent`. + */ +export async function collectMainTranscriptEvents( + sessionId: string, + transcriptPath: string, + trigger: MainTranscriptTrigger +): Promise[]> { + try { + // Both the load and the save happen inside the lock so a concurrent hook process for the + // same session can never read a state this pass is about to overwrite. + return await withParseStateLock(sessionId, async () => { + const state = await loadParseState(sessionId); + const { lines, nextOffset } = await readNewLines(transcriptPath, state.mainOffset); + + const touchedKeys = new Set(); + for (const line of lines) { + let rawGitBranch = ''; + try { + rawGitBranch = (JSON.parse(line) as { gitBranch?: string })?.gitBranch ?? ''; + } catch { + // Malformed line: still attempt usage parsing below (which has its own try/catch), but + // there is no branch to record from it. + } + if (rawGitBranch) { + updateBranchCounts(state.branchCounts, rawGitBranch); + } + + const parsed = parseUsageLine(line, 'main', '', ''); + if (parsed) { + const key = `${parsed.requestId}::${parsed.model}`; + const existing = state.openRequests[key]; + state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; + touchedKeys.add(key); + } + } + state.mainOffset = nextOffset; + + if (trigger === 'PreCompact') { + state.compactionCount += 1; + } + + const events: Record[] = []; + for (const key of touchedKeys) { + events.push(buildUsageRequestEvent(sessionId, state.openRequests[key])); + } + + if (trigger === 'Stop' || trigger === 'SessionEnd') { + const { acc, named, startedAt } = await buildFullAccumulator(transcriptPath); + acc.compactionCount = state.compactionCount; + const phase = trigger === 'SessionEnd' ? 'final' : 'incremental'; + const endedAt = trigger === 'SessionEnd' ? new Date().toISOString() : undefined; + const summaryEvent = buildSessionSummaryEvent( + sessionId, + phase, + acc, + named, + state.branchCounts, + startedAt, + endedAt + ); + // Not yet carried by any input to buildSessionSummaryEvent (session-summary.ts's own + // docstring defers it to this caller) — this is the full set of agent.usage.request + // records derived for this session so far, main- and agent-scoped alike. + summaryEvent.api_calls = Object.keys(state.openRequests).length; + events.push(summaryEvent); + } + + await saveParseState(sessionId, state); + return events; + }); + } catch { + // Swallow everything — never throw into processOtlpEvent. + return []; + } +} + +interface SubagentScanResult { + toolCalls: Record; + toolErrors: Record; + skillsInvoked: Record; + startedAt: string; + durationMs: number; +} + +/** + * Recompute one subagent transcript's tool-call/tool-error/skill-invocation aggregates and + * timing span from byte 0 of its own file (the subagent-transcript analogue of + * {@link buildFullAccumulator}'s "recompute fresh each time" approach — `TranscriptParseState` + * has no persisted field for any of these either). + * + * - `toolCalls`/`toolErrors` reuse {@link collectErrorToolUseIds} for the same `tool_use_id` → + * `tool_result.is_error` correlation {@link buildFullAccumulator} uses, but tally into two + * parallel `Record` maps (not the combined `{calls, errors}` shape + * `SessionSummaryAccumulator` uses) to match {@link buildSubagentUsageEvent}'s own + * `tool_calls`/`tool_errors` parameter shapes. + * - `skillsInvoked` is `extractNamedInvocations(parsedLines).skillInvocations`, taken verbatim. + * - `startedAt` is the first parsed line's `timestamp`, or `''` when the file is empty/unreadable. + * - `durationMs` is `Date.parse(lastLine.timestamp) - Date.parse(firstLine.timestamp)`, guarded by + * `Number.isFinite` (covers a missing/unparseable timestamp on either end, and a single-line + * file) so it is never `NaN` — falls back to `0`. + * + * Never throws: a missing/unreadable file or an empty file both resolve to the emptiest + * defensible result; a malformed individual line is skipped rather than aborting the whole scan. + */ +async function scanSubagentTranscript(filePath: string): Promise { + const empty: SubagentScanResult = { + toolCalls: {}, + toolErrors: {}, + skillsInvoked: {}, + startedAt: '', + durationMs: 0, + }; + + let raw: string; + try { + raw = await readFile(filePath, 'utf-8'); + } catch { + return empty; + } + + const rawLines = raw.split('\n').filter((line) => line.trim().length > 0); + const parsedLines: TranscriptLine[] = []; + for (const line of rawLines) { + try { + parsedLines.push(JSON.parse(line) as TranscriptLine); + } catch { + // Skip malformed lines rather than aborting the whole scan. + } + } + + if (parsedLines.length === 0) { + return empty; + } + + const errorByToolUseId = collectErrorToolUseIds(parsedLines); + const toolCalls: Record = {}; + const toolErrors: Record = {}; + + for (const parsed of parsedLines) { + const content = parsed.message?.content; + if (!Array.isArray(content)) continue; + for (const item of content as ContentBlock[]) { + if (item?.type !== 'tool_use' || typeof item.name !== 'string') continue; + toolCalls[item.name] = (toolCalls[item.name] ?? 0) + 1; + if (typeof item.id === 'string' && errorByToolUseId.has(item.id)) { + toolErrors[item.name] = (toolErrors[item.name] ?? 0) + 1; + } + } + } + + const named = extractNamedInvocations(parsedLines); + const startedAt = String(parsedLines[0].timestamp ?? ''); + const lastTimestamp = String(parsedLines[parsedLines.length - 1].timestamp ?? ''); + const diff = Date.parse(lastTimestamp) - Date.parse(startedAt); + const durationMs = Number.isFinite(diff) ? Math.max(0, diff) : 0; + + return { toolCalls, toolErrors, skillsInvoked: named.skillInvocations, startedAt, durationMs }; +} + +/** + * Collect the spool-bound events for one `SubagentStop` hook fire, or for one subagent file + * discovered by the `SessionEnd` backstop scan (`findSubagentFiles()`). + * + * - Loads persisted state, reads only the lines appended since + * `state.subagentOffsets[subagentFile.agentId]` (defaulting to 0 for a never-before-seen + * agent). + * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, + * scoped `scopeKind: 'agent'`, keyed by `${requestId}::${model}` — same merge/key convention + * `collectMainTranscriptEvents` uses for the main transcript. + * - Returns one `agent.usage.request` JSON string per request key touched by *this* pass (no new + * lines means no new events — a no-op reparse returns nothing at this layer). + * - Unconditionally also returns exactly one `agent.subagent.usage` event summarizing this + * agent's *cumulative* usage (every `scopeKind: 'agent'` record in `state.openRequests` for + * this `agentId`, not just the ones touched this pass) plus a fresh full-file + * tool-call/error/skill/timing scan (see {@link scanSubagentTranscript}) — this is deliberate: + * the `SessionEnd` backstop's whole purpose is to guarantee every subagent gets at least one + * `agent.subagent.usage` event even when its own `SubagentStop` hook never fired, so a + * re-run with nothing new since the last pass still returns one (summarizing unchanged + * cumulative state), rather than being skipped. + * - Persists the updated `subagentOffsets[subagentFile.agentId]` (and `openRequests`) back to + * disk. + * + * Never forwards anything itself — the caller is responsible for sending the returned events to + * the spool (exactly one place in the pipeline does that). + * + * Swallows every error internally — never throws into `processOtlpEvent`. + */ +export async function collectSubagentTranscriptEvents( + sessionId: string, + subagentFile: SubagentFile +): Promise[]> { + try { + // Load/mutate/save inside the lock — same rationale as collectMainTranscriptEvents: a sibling + // SubagentStop for another subagent in this same session must never read state this pass is + // about to overwrite. + return await withParseStateLock(sessionId, async () => { + const state = await loadParseState(sessionId); + const fromOffset = state.subagentOffsets[subagentFile.agentId] ?? 0; + const { lines, nextOffset } = await readNewLines(subagentFile.filePath, fromOffset); + + const touchedKeys = new Set(); + for (const line of lines) { + const parsed = parseUsageLine(line, 'agent', '', subagentFile.agentId); + if (parsed) { + const key = `${parsed.requestId}::${parsed.model}`; + const existing = state.openRequests[key]; + state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; + touchedKeys.add(key); + } + } + state.subagentOffsets[subagentFile.agentId] = nextOffset; + + const events: Record[] = []; + for (const key of touchedKeys) { + events.push(buildUsageRequestEvent(sessionId, state.openRequests[key])); + } + + // Cumulative usage for this agent — every scope_kind:'agent' record known for it so far, + // not just the ones touched this pass (consistent with buildFullAccumulator's own + // "recomputed from full state" framing for the sibling agent.session.summary event). + const usageRequestsForAgent = Object.values(state.openRequests).filter( + (r) => r.scopeKind === 'agent' && r.agentId === subagentFile.agentId + ); + + const { toolCalls, toolErrors, skillsInvoked, startedAt, durationMs } = + await scanSubagentTranscript(subagentFile.filePath); + + const subagentEvent = buildSubagentUsageEvent( + sessionId, + subagentFile, + usageRequestsForAgent, + toolCalls, + toolErrors, + skillsInvoked, + startedAt, + durationMs + ); + events.push(subagentEvent); + + await saveParseState(sessionId, state); + return events; + }); + } catch { + // Swallow everything — never throw into processOtlpEvent. + return []; + } +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts new file mode 100644 index 000000000..948f63260 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -0,0 +1,158 @@ +/** + * Transcript parse-state persistence. + * + * Transcript parsing is incremental: each parse pass picks up where the previous one + * left off (byte offsets into the main transcript and per-subagent transcripts), + * tracks usage requests opened but not yet closed by their matching response, the + * currently active skill, and per-branch request counts. This module persists that + * state to disk between parse passes, keyed by session id. + */ + +import { mkdir, open, readFile, rm, stat, writeFile } from 'node:fs/promises'; +import { dirname } from 'node:path'; +import { getCodemiePath } from '@/utils/paths.js'; + +export interface OpenUsageRequest { + requestId: string; + model: string; + modelRaw: string; + timestamp: string; + speed: string; + inferenceGeo: string; + serviceTier: string; + inputTokens: number; + cacheCreation5mTokens: number; + cacheCreation1hTokens: number; + cacheReadTokens: number; + outputTokens: number; + webSearchRequests: number; + webFetchRequests: number; + scopeKind: 'main' | 'skill' | 'agent'; + scopeName: string; + agentId: string; + stopReason: string; + isApiError: boolean; + gitBranch: string; +} + +export interface TranscriptParseState { + mainOffset: number; + subagentOffsets: Record; + openRequests: Record; // key: `${requestId}::${model}` + activeSkill: string; + branchCounts: Record; + compactionCount: number; +} + +/** + * Build a fresh, empty parse state. + */ +export function createParseState(): TranscriptParseState { + return { + mainOffset: 0, + subagentOffsets: {}, + openRequests: {}, + activeSkill: '', + branchCounts: {}, + compactionCount: 0, + }; +} + +function getParseStatePath(sessionId: string): string { + return getCodemiePath('analytics', 'state', `${sessionId}.json`); +} + +/** + * Load the persisted parse state for a session. + * + * Never throws: a missing file, malformed JSON, or any other I/O failure all fall + * back to a fresh state via {@link createParseState}, since transcript parsing must + * keep going (as if starting fresh) rather than fail the whole run over stale/corrupt + * state on disk. + */ +export async function loadParseState(sessionId: string): Promise { + try { + const raw = await readFile(getParseStatePath(sessionId), 'utf-8'); + const parsed = JSON.parse(raw) as Partial; + + return { ...createParseState(), ...parsed }; + } catch { + return createParseState(); + } +} + +/** + * Persist parse state for a session, creating the parent directory if needed. + * + * Unlike {@link loadParseState}, this does not swallow errors — a genuine write + * failure (disk full, permissions) propagates to the caller rather than silently + * discarding progress. + */ +export async function saveParseState(sessionId: string, state: TranscriptParseState): Promise { + const filePath = getParseStatePath(sessionId); + await mkdir(dirname(filePath), { recursive: true }); + await writeFile(filePath, JSON.stringify(state, null, 2), 'utf-8'); +} + +const LOCK_RETRY_MS = 25; +const LOCK_STALE_MS = 5_000; + +function getLockPath(sessionId: string): string { + return `${getParseStatePath(sessionId)}.lock`; +} + +async function isLockStale(lockPath: string): Promise { + try { + const info = await stat(lockPath); + return Date.now() - info.mtimeMs > LOCK_STALE_MS; + } catch { + return true; // disappeared between our EEXIST and this check — treat as gone + } +} + +/** + * Serialize one session's load-mutate-save parse-state cycle across concurrent hook processes + * (e.g. sibling `SubagentStop` fires for the same session) via an exclusive-create lock file. + * Each hook fire is a fresh CLI process, so this cannot use an in-memory mutex. + * + * A lock older than {@link LOCK_STALE_MS} is treated as abandoned (its holder crashed before + * releasing it) and stolen rather than awaited forever. Likewise, if the lock cannot be acquired + * within a bounded wait, `fn` still runs unlocked rather than hanging the hook indefinitely — + * occasional lost contention here is strictly better than analytics never shipping at all. + */ +export async function withParseStateLock(sessionId: string, fn: () => Promise): Promise { + const lockPath = getLockPath(sessionId); + await mkdir(dirname(lockPath), { recursive: true }); + + const deadline = Date.now() + LOCK_STALE_MS * 2; + let acquired = false; + for (;;) { + try { + const handle = await open(lockPath, 'wx'); + await handle.close(); + acquired = true; + break; + } catch (err) { + if ((err as NodeJS.ErrnoException).code !== 'EEXIST') { + break; // can't lock (e.g. permissions) — proceed unlocked rather than block forever + } + if (await isLockStale(lockPath)) { + await rm(lockPath, { force: true }).catch(() => {}); + continue; + } + if (Date.now() > deadline) { + break; // gave the lock a fair wait; proceed unlocked rather than hang the hook + } + await new Promise((resolve) => setTimeout(resolve, LOCK_RETRY_MS)); + } + } + + try { + return await fn(); + } finally { + // Only release a lock we hold — when we proceeded unlocked, the file belongs to another process. + if (acquired) { + await rm(lockPath, { force: true }).catch(() => {}); + } + } +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts new file mode 100644 index 000000000..cfd6f257f --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts @@ -0,0 +1,130 @@ +/** + * `agent.session.summary` builder. + * + * Unlike `agent.usage.request`/`agent.subagent.usage`, this event is a running aggregate over an + * entire session. The orchestrator accumulates a {@link SessionSummaryAccumulator}, tracks + * `TranscriptParseState.branchCounts` via {@link updateBranchCounts}, and runs + * `extractNamedInvocations()` to produce the {@link NamedInvocationCounts} this builder consumes. + * This module only derives the final event shape from those inputs — it never reads a transcript. + * + * `event_id`/`schema_version`/`client_version`/`codemie_cli_version` are stamped later, + * daemon-side (`mapHookRecords()`); the output carries only an explicit `type`. + * + * Field-shape notes: + * - `models_used` is the full `acc.models` count map, preserving counts `primary_model` discards. + * - `tool_calls`/`tool_errors` are flattened from `acc.toolCalls`'s `{ calls, errors }` shape into + * two flat maps, matching `buildSubagentUsageEvent` (`./subagent-usage.ts`). + * - `commands_in_order` is `Object.keys(named.commandInvocations)`. Upstream is a COUNT map, so no + * chronological order exists; the field name implies more than the data can deliver. + * - `title` has no known source and is always an empty string, never fabricated. + * - `api_calls` is omitted here: no input carries a request count. The orchestrator, which owns + * the full set of `agent.usage.request` records, merges it in afterward. + */ + +import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; + +export type { NamedInvocationCounts }; + +/** Running, mutable aggregate accumulated by the caller across one session's transcript. */ +export interface SessionSummaryAccumulator { + models: Record; + toolCalls: Record; + linesAdded: number; + linesRemoved: number; + filesChanged: Set; + filesWritten: Set; + compactionCount: number; +} + +/** + * Bump `counts[branch]` by 1, mutating `counts` in place (this is the caller-maintained + * `TranscriptParseState.branchCounts` map from `./parse-state.ts`). + * + * A falsy/empty `branch` is skipped — an unknown/missing branch shouldn't pollute the + * dominant-branch calculation ({@link branchDominant}). + */ +export function updateBranchCounts(counts: Record, branch: string): void { + if (!branch) { + return; + } + counts[branch] = (counts[branch] ?? 0) + 1; +} + +/** + * Return the key with the highest value in `counts`, or `''` when `counts` is empty. + * On a tie, the first-encountered key (in `Object.entries()` iteration order) wins. + */ +function maxKey(counts: Record): string { + let best = ''; + let bestValue = -Infinity; + + for (const [key, value] of Object.entries(counts)) { + if (value > bestValue) { + best = key; + bestValue = value; + } + } + + return best; +} + +/** The model with the highest count in `models`, or `''` when empty. */ +export function primaryModel(models: Record): string { + return maxKey(models); +} + +/** The branch with the highest count in `counts`, or `''` when empty. */ +export function branchDominant(counts: Record): string { + return maxKey(counts); +} + +/** + * Build the `agent.session.summary` event payload. + * + * `endedAt` is included as `ended_at` only when `phase === 'final'`; for `phase === 'incremental'` + * the key is omitted entirely (not merely `undefined`-valued). + */ +export function buildSessionSummaryEvent( + sessionId: string, + phase: 'incremental' | 'final', + acc: SessionSummaryAccumulator, + named: NamedInvocationCounts, + branchCounts: Record, + startedAt: string, + endedAt: string | undefined +): Record { + const toolCalls: Record = {}; + const toolErrors: Record = {}; + for (const [tool, counts] of Object.entries(acc.toolCalls)) { + toolCalls[tool] = counts.calls; + toolErrors[tool] = counts.errors; + } + + const event: Record = { + type: 'agent.session.summary', + session_id: sessionId, + phase, + models_used: acc.models, + primary_model: primaryModel(acc.models), + tool_calls: toolCalls, + tool_errors: toolErrors, + skills_used: named.skillInvocations, + commands_in_order: Object.keys(named.commandInvocations), + primary_command: maxKey(named.commandInvocations), + lines_added: acc.linesAdded, + lines_removed: acc.linesRemoved, + files_changed: Array.from(acc.filesChanged), + files_written: Array.from(acc.filesWritten), + compaction_count: acc.compactionCount, + branch_counts: branchCounts, + branch_dominant: branchDominant(branchCounts), + started_at: startedAt, + title: '', + }; + + if (phase === 'final') { + event.ended_at = endedAt; + } + + return event; +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts new file mode 100644 index 000000000..e94375f92 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts @@ -0,0 +1,135 @@ +/** + * `agent.subagent.usage` discovery and event builder. + * + * Discovery (`findSubagentFiles`) uses the same path convention as the private + * `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts` + * (`//subagents/agent-*.jsonl` + sibling `.meta.json`) but returns + * only the narrower {@link SubagentFile} shape this event needs. + * + * The event builder (`buildSubagentUsageEvent`) sums an already-scoped `OpenUsageRequest[]` for + * token/cache fields and passes the caller-built `tool_calls`/`tool_errors`/`skills_invoked` maps + * through verbatim; it has no access to the subagent transcript itself. + * + * `description`/`workflow_run`/`worktree` have no known source and are always empty strings, + * never fabricated. + */ + +import { readdir, readFile } from 'node:fs/promises'; +import { basename, dirname, join } from 'node:path'; +import type { OpenUsageRequest } from './parse-state.js'; + +export interface SubagentFile { + agentId: string; + filePath: string; + toolUseId?: string; + agentType?: string; + spawnDepth?: number; +} + +/** + * Discover subagent transcript files for a main transcript at `mainTranscriptPath`. + * + * Looks under `//subagents/` (where `sessionId` is `mainTranscriptPath`'s + * own basename, minus `.jsonl`) for `agent-*.jsonl` files, reading each one's sibling + * `.meta.json` sidecar (when present and parseable) for `toolUseId`/`agentType`/ + * `spawnDepth`. Never reads `mainTranscriptPath`'s own content — only its path is used to derive + * the subagents directory. + * + * Never throws: a missing subagents directory, an unreadable directory, or any other failure all + * resolve to `[]`. A missing or malformed per-agent `.meta.json` sidecar is likewise swallowed — + * that agent is still returned, just without the sidecar-derived fields. + */ +export async function findSubagentFiles(mainTranscriptPath: string): Promise { + try { + const parentDir = dirname(mainTranscriptPath); + const filename = basename(mainTranscriptPath); + const sessionId = filename.replace(/\.jsonl$/, ''); + const subagentsDir = join(parentDir, sessionId, 'subagents'); + + const files = await readdir(subagentsDir); + const agentFiles = await Promise.all( + files + .filter((f) => f.startsWith('agent-') && f.endsWith('.jsonl')) + .map(async (f): Promise => { + const agentId = f.replace(/^agent-/, '').replace(/\.jsonl$/, ''); + const filePath = join(subagentsDir, f); + + let toolUseId: string | undefined; + let agentType: string | undefined; + let spawnDepth: number | undefined; + + try { + const metaRaw = JSON.parse( + await readFile(join(subagentsDir, f.replace(/\.jsonl$/, '.meta.json')), 'utf-8') + ) as Record; + if (typeof metaRaw.toolUseId === 'string') toolUseId = metaRaw.toolUseId; + if (typeof metaRaw.agentType === 'string') agentType = metaRaw.agentType; + if (typeof metaRaw.spawnDepth === 'number') spawnDepth = metaRaw.spawnDepth; + } catch { + // meta file absent or malformed — proceed without it. + } + + return { agentId, filePath, toolUseId, agentType, spawnDepth }; + }) + ); + + return agentFiles; + } catch { + return []; + } +} + +/** + * Build the `agent.subagent.usage` event payload for one subagent file. + * + * `usageRequests` is an array of {@link OpenUsageRequest} the caller has already parsed and + * scoped to this one subagent (`scope_kind: 'agent'`) — this function only sums it, it does not + * filter or scope it itself. `toolCalls`/`toolErrors`/`skillsInvoked` are likewise caller-built + * `Record` maps (keyed by tool/skill name) and are passed through verbatim. + * + * `started_at`/`duration_ms` are forwarded verbatim from the caller, which derives them from the + * subagent transcript's own first/last line timestamps — not this function's job. + * + * `spawn_depth` defaults to `0` when `file.spawnDepth` is absent (top-level subagents, whose + * sidecar omits the field — not treated as an error). + * + * Carries its own explicit `type`, so `event_id`/`schema_version` are stamped later, daemon-side. + */ +export function buildSubagentUsageEvent( + sessionId: string, + file: SubagentFile, + usageRequests: OpenUsageRequest[], + toolCalls: Record, + toolErrors: Record, + skillsInvoked: Record, + startedAt: string, + durationMs: number +): Record { + const sum = (selector: (req: OpenUsageRequest) => number): number => + usageRequests.reduce((total, req) => total + selector(req), 0); + + return { + type: 'agent.subagent.usage', + session_id: sessionId, + agent_id: file.agentId, + tool_use_id: file.toolUseId ?? '', + agent_type: file.agentType ?? '', + spawn_depth: file.spawnDepth ?? 0, + description: '', + workflow_run: '', + worktree: '', + started_at: startedAt, + duration_ms: durationMs, + input_tokens: sum((r) => r.inputTokens), + cache_creation_5m_tokens: sum((r) => r.cacheCreation5mTokens), + cache_creation_1h_tokens: sum((r) => r.cacheCreation1hTokens), + cache_read_tokens: sum((r) => r.cacheReadTokens), + output_tokens: sum((r) => r.outputTokens), + web_search_requests: sum((r) => r.webSearchRequests), + web_fetch_requests: sum((r) => r.webFetchRequests), + api_calls: usageRequests.length, + tool_calls: toolCalls, + tool_errors: toolErrors, + skills_invoked: skillsInvoked, + }; +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts b/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts new file mode 100644 index 000000000..b60bbdbfc --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts @@ -0,0 +1,61 @@ +import { open, type FileHandle } from 'node:fs/promises'; + +const NEWLINE = 0x0a; + +export interface ReadNewLinesResult { + lines: string[]; + nextOffset: number; +} + +/** + * Incrementally read complete, newline-terminated lines appended to a + * transcript file since `fromOffset`. + * + * Mirrors the safe-cut rule used by + * `src/providers/plugins/sso/proxy/plugins/otlp-spool/spool-io.ts`'s + * `snapshotPendingHookRecords`: the read is cut at the last `\n` byte so a + * partially-written trailing line is never returned. Never throws — a + * missing file, a shrunk/rotated file, or a chunk with no complete line yet + * all resolve to the documented empty-result shape. + */ +export async function readNewLines( + filePath: string, + fromOffset: number +): Promise { + let handle: FileHandle; + try { + handle = await open(filePath, 'r'); + } catch { + return { lines: [], nextOffset: fromOffset }; // missing file => nothing new + } + + try { + const { size } = await handle.stat(); + // A file strictly smaller than the persisted offset was rotated/truncated underneath us — + // resume a fresh parse from 0 rather than comparing the new (smaller) size against the stale + // offset forever (which would permanently stall this session's parsing). + const effectiveFromOffset = size < fromOffset ? 0 : fromOffset; + if (size <= effectiveFromOffset) { + return { lines: [], nextOffset: effectiveFromOffset }; // nothing new + } + + const buffer = Buffer.allocUnsafe(size - effectiveFromOffset); + const { bytesRead } = await handle.read(buffer, 0, buffer.length, effectiveFromOffset); + const chunk = buffer.subarray(0, bytesRead); + + const lastNewline = chunk.lastIndexOf(NEWLINE); + if (lastNewline < 0) { + return { lines: [], nextOffset: effectiveFromOffset }; // no complete line yet + } + + const complete = chunk.subarray(0, lastNewline + 1); + const lines = complete + .toString('utf-8') + .split('\n') + .filter((line) => line.trim().length > 0); + + return { lines, nextOffset: effectiveFromOffset + complete.length }; + } finally { + await handle.close(); + } +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts new file mode 100644 index 000000000..dbe0ca086 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -0,0 +1,186 @@ +/** + * `agent.usage.request` extraction and merge. + * + * One record per Claude transcript JSONL line that carries `message.usage` — the same + * per-API-response usage shape the cost-reporting parser + * (`src/cli/commands/analytics/cost/usage-readers.ts`) reads, adapted to + * {@link OpenUsageRequest} (`parse-state.ts`) instead of that pipeline's `UsageRecord`. + * + * Claude Code can write more than one JSONL row for the same API response (progressive + * streaming chunks, or a later row that fills in `stop_reason` once the turn finishes), so + * callers parse every candidate line and merge same-identity records with + * {@link mergeUsageRequest} — this module trusts the caller to key records by + * `${requestId}::${model}` (see `parse-state.ts`'s `openRequests`) before merging; it does + * not itself check that two records it is asked to merge actually share that identity. + */ + +import { type RoutingHeaderSource } from '../../../../utils/routing-headers.mjs'; +import { parseBackendModelName } from '../../../../utils/bedrock-pricing.mjs'; +import type { OpenUsageRequest } from './parse-state.js'; + +/** + * Loose shape of one transcript JSONL line, mirroring `usage-readers.ts`'s `ClaudeRawMessage` + * plus `message.stop_reason`, `message.usage`'s extra nested groups, and the top-level + * `gitBranch`/`isApiError`. Not exported — callers only see {@link parseUsageLine}'s result. + */ +interface TranscriptUsageLine { + timestamp?: string; + gitBranch?: string; + isApiError?: boolean; + message?: RoutingHeaderSource & { + id?: string; + model?: string; + stop_reason?: string; + usage?: { + input_tokens?: number; + output_tokens?: number; + cache_read_input_tokens?: number; + cache_creation_input_tokens?: number; + service_tier?: string; + speed?: string; + inference_geo?: string; + cache_creation?: { + ephemeral_1h_input_tokens?: number; + ephemeral_5m_input_tokens?: number; + }; + server_tool_use?: { + web_search_requests?: number; + web_fetch_requests?: number; + }; + }; + }; +} + +/** + * Parse one transcript JSONL line into an {@link OpenUsageRequest}, or `null` when the line + * carries no `message.usage` block (not a billable API response — e.g. a plain user/system + * message) or is not valid JSON. + * + * `scopeKind`/`scopeName`/`agentId` are passed through verbatim from the caller, which already + * knows which transcript (main vs. a named skill context vs. a subagent transcript) `line` came + * from — this function has no way to derive that from the line itself. + */ +export function parseUsageLine( + line: string, + scopeKind: 'main' | 'skill' | 'agent', + scopeName: string, + agentId: string +): OpenUsageRequest | null { + let parsed: TranscriptUsageLine; + try { + parsed = JSON.parse(line) as TranscriptUsageLine; + } catch { + return null; + } + + const usage = parsed.message?.usage; + if (!usage) { + return null; + } + + // openRequests keys on `${requestId}::${model}` — an empty requestId would collide every such + // line in the session into one record instead of being skipped. + const requestId = parsed.message?.id ?? ''; + if (!requestId) { + return null; + } + + // Same resolution chain the statusline and usage-readers.ts already use: + // parseBackendModelName() (the raw LiteLLM backend id, when the proxy injected one) wins over + // the transcript's own literal `message.model`, since it reflects the actual billable backend + // model for a routed/capable-tier request. `modelRaw` keeps the literal, unresolved alias. + const modelRaw = parsed.message?.model ?? 'unknown'; + const model = parseBackendModelName(parsed.message) ?? modelRaw; + + return { + requestId, + model, + modelRaw, + timestamp: parsed.timestamp ?? '', + speed: usage.speed ?? '', + inferenceGeo: usage.inference_geo ?? '', + serviceTier: usage.service_tier ?? '', + inputTokens: Number(usage.input_tokens ?? 0), + cacheCreation5mTokens: Number(usage.cache_creation?.ephemeral_5m_input_tokens ?? 0), + cacheCreation1hTokens: Number(usage.cache_creation?.ephemeral_1h_input_tokens ?? 0), + cacheReadTokens: Number(usage.cache_read_input_tokens ?? 0), + outputTokens: Number(usage.output_tokens ?? 0), + webSearchRequests: Number(usage.server_tool_use?.web_search_requests ?? 0), + webFetchRequests: Number(usage.server_tool_use?.web_fetch_requests ?? 0), + scopeKind, + scopeName, + agentId, + // Sibling of usage on message, not nested inside it. + stopReason: parsed.message?.stop_reason ?? '', + isApiError: Boolean(parsed.isApiError), + gitBranch: parsed.gitBranch ?? '', + }; +} + +/** + * Merge two {@link OpenUsageRequest} records the caller has already identified as the same + * logical request (same `requestId`+`model` — this function does not verify that itself). + * Every numeric field takes the max of the two (a later streaming/finalizing row only ever adds + * usage, never subtracts it); every non-numeric field takes `b`'s value when non-empty, else + * falls back to `a`'s — so a later row that fills in a previously-empty field (e.g. + * `stop_reason` once the turn finishes) wins, while a later row that is missing a field `a` had + * does not blank it out. + * + * Returns a new object; neither `a` nor `b` is mutated. + */ +export function mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): OpenUsageRequest { + return { + requestId: b.requestId || a.requestId, + model: b.model || a.model, + modelRaw: b.modelRaw || a.modelRaw, + timestamp: b.timestamp || a.timestamp, + speed: b.speed || a.speed, + inferenceGeo: b.inferenceGeo || a.inferenceGeo, + serviceTier: b.serviceTier || a.serviceTier, + inputTokens: Math.max(a.inputTokens, b.inputTokens), + cacheCreation5mTokens: Math.max(a.cacheCreation5mTokens, b.cacheCreation5mTokens), + cacheCreation1hTokens: Math.max(a.cacheCreation1hTokens, b.cacheCreation1hTokens), + cacheReadTokens: Math.max(a.cacheReadTokens, b.cacheReadTokens), + outputTokens: Math.max(a.outputTokens, b.outputTokens), + webSearchRequests: Math.max(a.webSearchRequests, b.webSearchRequests), + webFetchRequests: Math.max(a.webFetchRequests, b.webFetchRequests), + scopeKind: b.scopeKind || a.scopeKind, + scopeName: b.scopeName || a.scopeName, + agentId: b.agentId || a.agentId, + stopReason: b.stopReason || a.stopReason, + isApiError: b.isApiError || a.isApiError, + gitBranch: b.gitBranch || a.gitBranch, + }; +} + +/** + * Build the `agent.usage.request` event payload for `req`. Carries its own explicit `type`, so + * the daemon-side `mapHookRecords()` stamps `event_id`/`schema_version` onto it later — + * this function deliberately does not set either. + */ +export function buildUsageRequestEvent(sessionId: string, req: OpenUsageRequest): Record { + return { + type: 'agent.usage.request', + session_id: sessionId, + request_id: req.requestId, + model_raw: req.modelRaw, + model: req.model, + speed: req.speed, + inference_geo: req.inferenceGeo, + service_tier: req.serviceTier, + input_tokens: req.inputTokens, + cache_creation_5m_tokens: req.cacheCreation5mTokens, + cache_creation_1h_tokens: req.cacheCreation1hTokens, + cache_read_tokens: req.cacheReadTokens, + output_tokens: req.outputTokens, + web_search_requests: req.webSearchRequests, + web_fetch_requests: req.webFetchRequests, + scope_kind: req.scopeKind, + scope_name: req.scopeName, + agent_id: req.agentId, + stop_reason: req.stopReason, + is_api_error: req.isApiError, + git_branch: req.gitBranch, + timestamp: req.timestamp, + }; +} diff --git a/src/agents/plugins/utils.ts b/src/agents/plugins/utils.ts index c38ab1972..9bc310e68 100644 --- a/src/agents/plugins/utils.ts +++ b/src/agents/plugins/utils.ts @@ -1,3 +1,4 @@ +import { randomUUID } from 'node:crypto'; import { OtlpHookSpoolData } from '@/providers/plugins/sso/proxy/plugins/otlp.plugin.js'; import { readState } from '../../cli/commands/proxy/daemon-manager.js'; import { logger } from '../../utils/logger.js'; @@ -11,7 +12,7 @@ import { logger } from '../../utils/logger.js'; * - Uses 1000ms timeout and swallows all errors * - Never throws, never affects hook's exit code */ -export async function forwardOtlpEventToSpool(rawEvent: string, agentName: string): Promise { +export async function forwardOtlpEventToSpool(event: Record, agentName: string): Promise { try { const state = await readState(); @@ -26,7 +27,10 @@ export async function forwardOtlpEventToSpool(rawEvent: string, agentName: strin const body: OtlpHookSpoolData = { agentName, timestamp: Date.now(), - raw: rawEvent + raw: JSON.stringify({ + ...event, + event_id: randomUUID() + }) } try { diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts new file mode 100644 index 000000000..97362e61b --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -0,0 +1,316 @@ +import { describe, it, expect, vi, beforeEach } from 'vitest'; + +const execSyncMock = vi.fn(); + +vi.mock('node:child_process', () => ({ + execSync: execSyncMock, +})); + +beforeEach(() => { + execSyncMock.mockReset(); + execSyncMock.mockReturnValue('0.15.6'); +}); + +interface MappedRecord { + type: string; + session_id: string; + schema_version: number; + event_id: string; + codemie_cli_version: string; + story_id?: string; + story_source?: string; + prompt_body?: string; + developer_name?: string; + identity_source?: string; + platform?: string; + client_version?: string; +} + +function buildHookRecord( + hookEventName: string, + sessionId: string, + extra: Record = {}, + eventId: string = 'default-event-id' +): string { + return JSON.stringify({ + agentName: 'claude', + raw: JSON.stringify({ + hook_event_name: hookEventName, + session_id: sessionId, + cwd: '', + event_id: eventId, + ...extra, + }), + timestamp: Date.now(), + }); +} + +describe('resolveCodemieCliVersion', () => { + beforeEach(() => { + execSyncMock.mockReset(); + execSyncMock.mockReturnValue('0.15.6'); + }); + + it('reads the installed CLI version via `codemie --version` and strips the semver', async () => { + const { resolveCodemieCliVersion } = await import('../forward-context.js'); + + expect(resolveCodemieCliVersion()).toBe('0.15.6'); + expect(execSyncMock).toHaveBeenCalledWith('codemie --version', { + encoding: 'utf8', + stdio: ['pipe', 'pipe', 'pipe'], + }); + }); + + it('falls back to an empty string when `codemie --version` throws', async () => { + execSyncMock.mockImplementation(() => { + throw new Error('ENOENT'); + }); + + const { resolveCodemieCliVersion } = await import('../forward-context.js'); + expect(resolveCodemieCliVersion()).toBe(''); + }); +}); + +describe('mapHookRecords', () => { + it('stamps schema_version, event_id, and codemie_cli_version on every mapped record', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: 'user@example.com', + git: {}, + }; + + const record1 = buildHookRecord('SessionStart', 'sid1', {}, 'event-id-1'); + const record2 = buildHookRecord('Stop', 'sid1', {}, 'event-id-2'); + + const payload = await mapHookRecords([record1, record2], ctx); + const lines = payload.ndjson + .trim() + .split('\n') + .map((line) => JSON.parse(line) as MappedRecord); + + expect(lines).toHaveLength(2); + expect(lines[0].schema_version).toBe(2); + expect(lines[1].schema_version).toBe(2); + expect(lines[0].event_id).not.toBe(lines[1].event_id); + expect(typeof lines[0].codemie_cli_version).toBe('string'); + expect(lines[0].codemie_cli_version.length).toBeGreaterThan(0); + expect(lines[0].codemie_cli_version).toBe(lines[1].codemie_cli_version); + }); + + it('passes the event_id already stamped at spool-write time straight through unchanged', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + const record = buildHookRecord('SessionStart', 'sid1', {}, 'stamped-event-id-abc'); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.event_id).toBe('stamped-event-id-abc'); + }); + + it('prefers an explicit hookEvent.type over the HOOK_EVENT_TYPE_MAP lookup', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + // PostToolUse normally maps to 'agent.tool.end', but an explicit `type` + // field on the raw hook payload (as a later-task synthetic record would + // carry) must win. + const record = buildHookRecord('PostToolUse', 'sid1', { type: 'agent.custom.synthetic' }); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.type).toBe('agent.custom.synthetic'); + }); + + it('leaves existing-event type resolution unchanged when type is absent', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + const record = buildHookRecord('PostToolUse', 'sid1'); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.type).toBe('agent.tool.end'); + }); + + it( + "overrides a UserPromptSubmit record's story_id/story_source with a prompt marker " + + 'even when the per-tick branch tier would otherwise resolve to a different ticket, ' + + 'and never leaks the raw prompt text onto the emitted record', + async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + // Branch carries a DIFFERENT ticket than the prompt marker, so this + // test proves the marker tier wins over the already-cached branch tier. + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: { branch: 'feature/ABC-1-unrelated-branch' }, + }; + + // Longer than MAX_PROMPT_CHARS (200) so every bounded copy on the + // mapped record is truncated and none of them equals this full text — + // which is what actually proves "the raw prompt is never present" + // rather than merely proving a short prompt survives truncation whole. + const rawPrompt = + `Please implement this feature. story: EPMCDME-999 is the ticket to reference. ` + + 'x'.repeat(200) + + ' end-of-prompt-marker-that-must-not-appear-anywhere-in-the-output'; + const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.story_id).toBe('EPMCDME-999'); + expect(line.story_source).toBe('marker'); + + // The raw prompt text must never appear verbatim anywhere on the + // emitted record — only the truncated `prompt_body` and the resolved + // short `story_id` string are allowed to carry prompt-derived content. + const serialized = JSON.stringify(line); + expect(serialized).not.toContain(rawPrompt); + expect(serialized).not.toContain('end-of-prompt-marker-that-must-not-appear-anywhere-in-the-output'); + expect(line.prompt_body).toBe(rawPrompt.slice(0, 200)); + } + ); + + it( + 'falls back to the per-tick branch result for a UserPromptSubmit record whose prompt ' + + 'has no marker and no bare ticket mention', + async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: { branch: 'feature/epmcdme-15301-foo' }, + }; + + const rawPrompt = 'please just fix the thing, no ticket reference here'; + const record = buildHookRecord('UserPromptSubmit', 'sid1', { + prompt: rawPrompt, + cwd: '/repo/nonexistent-for-this-test', + }); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.story_id).toBe('EPMCDME-15301'); + expect(line.story_source).toBe('branch'); + } + ); + + it( + 'falls back to the mention tier for a UserPromptSubmit record whose prompt has a bare ' + + 'ticket mention and the branch carries no ticket', + async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: { branch: 'just-some-branch-name' }, + }; + + const rawPrompt = 'can you look into ABC-42 when you get a chance'; + const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.story_id).toBe('ABC-42'); + expect(line.story_source).toBe('mention'); + } + ); + + it('passes agent-baked common fields (platform/client_version) through onto the mapped record without any agent-specific lookup', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + // Simulates what the plugin now bakes in hook-side before ever reaching the spool — + // the forwarder needs no agent-specific knowledge to pass these through, just the + // `...limited` spread like every other hook-native field. + const record = JSON.stringify({ + agentName: 'claude-code-otlp', + raw: JSON.stringify({ + hook_event_name: 'Stop', + session_id: 'sid1', + cwd: '', + platform: 'claude-code', + client_version: '1.2.3', + }), + timestamp: Date.now(), + }); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.platform).toBe('claude-code'); + expect(line.client_version).toBe('1.2.3'); + }); + + it('carries ctx.identity through onto developer_name/identity_source for a non-jwt tier', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + // Pre-seeding ctx.identity (mimicking what the per-tick identity cache would look like once resolved) + // with a non-jwt tier result proves the wiring from ctx.identity onto the mapped record, + // independent of the identity-chain's own resolution logic (covered by identity.test.ts). + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + identity: { developerName: 'git-user@example.com', identitySource: 'git' as const }, + }; + + const record = buildHookRecord('Stop', 'sid1'); + + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.developer_name).toBe('git-user@example.com'); + expect(line.identity_source).toBe('git'); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts new file mode 100644 index 000000000..796fffe99 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts @@ -0,0 +1,72 @@ +import { describe, it, expect, vi, afterEach } from 'vitest'; +import * as os from 'node:os'; +import type { JWTCredentials } from '@/providers/core/types.js'; + +/** Builds a minimal unsigned JWT with the given payload claims. */ +function makeJwt(payload: Record): string { + const header = Buffer.from(JSON.stringify({ alg: 'none', typ: 'JWT' })).toString('base64url'); + const body = Buffer.from(JSON.stringify(payload)).toString('base64url'); + return `${header}.${body}.sig`; +} + +describe('resolveIdentity', () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it('resolves from the jwt tier when the token carries a valid email claim', async () => { + const { resolveIdentity } = await import('../identity.js'); + + const credentials: JWTCredentials = { + token: makeJwt({ email: 'dev@example.com' }), + apiUrl: 'https://codemie.example.com', + }; + + const result = await resolveIdentity(credentials, 'C:/some/project'); + + expect(result).toEqual({ developerName: 'dev@example.com', identitySource: 'jwt' }); + }); + + it('falls through to git config user.email when the jwt tier is empty', async () => { + const execModule = await import('@/utils/exec.js'); + const execSpy = vi.spyOn(execModule, 'exec').mockImplementation(async (_command, args) => { + if (args?.[0] === 'config' && args?.[1] === 'user.email') { + return { code: 0, stdout: 'git-user@example.com', stderr: '', signal: null }; + } + return { code: 1, stdout: '', stderr: '', signal: null }; + }); + + const { resolveIdentity } = await import('../identity.js'); + + // Empty token: isJWTCredentials() still matches the shape, but decodeJwtClaims + // yields no usable email, so the jwt tier is a miss. + const credentials: JWTCredentials = { token: '', apiUrl: '' }; + + const result = await resolveIdentity(credentials, 'C:/some/project'); + + expect(result).toEqual({ developerName: 'git-user@example.com', identitySource: 'git' }); + expect(execSpy).toHaveBeenCalledWith('git', ['config', 'user.email'], { cwd: 'C:/some/project' }); + }); + + it('falls through to os.userInfo().username when every other tier is empty', async () => { + const execModule = await import('@/utils/exec.js'); + vi.spyOn(execModule, 'exec').mockResolvedValue({ code: 1, stdout: '', stderr: '', signal: null }); + + const configModule = await import('@/utils/config.js'); + vi.spyOn(configModule.ConfigLoader, 'loadMultiProviderConfig').mockResolvedValue({ + version: 2, + activeProfile: 'default', + profiles: {}, + }); + + const { resolveIdentity } = await import('../identity.js'); + + const credentials: JWTCredentials = { token: '', apiUrl: '' }; + + const result = await resolveIdentity(credentials, 'C:/some/project'); + + expect(result.identitySource).toBe('os'); + expect(result.developerName).toBe(os.userInfo().username); + expect(result.developerName.length).toBeGreaterThan(0); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts new file mode 100644 index 000000000..61c7948bf --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts @@ -0,0 +1,218 @@ +import { describe, it, expect, afterEach } from 'vitest'; +import { mkdtemp, mkdir, writeFile, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; + +const ENV_KEY = 'SDLC_ANALYTICS_STORY_ID'; + +async function makeTempProjectDir(): Promise { + return mkdtemp(join(tmpdir(), 'otlp-story-resolver-')); +} + +async function writeAnalyticsLocalJson(cwd: string, data: Record): Promise { + const dir = join(cwd, '.claude'); + await mkdir(dir, { recursive: true }); + await writeFile(join(dir, 'analytics.local.json'), JSON.stringify(data), 'utf-8'); +} + +describe('story-resolver', () => { + const tempDirs: string[] = []; + + afterEach(async () => { + delete process.env[ENV_KEY]; + for (const dir of tempDirs.splice(0)) { + await rm(dir, { recursive: true, force: true }); + } + }); + + describe('resolveExplicitStory', () => { + it('prefers the env var over the file when both are set', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + await writeAnalyticsLocalJson(cwd, { storyId: 'FILE-1' }); + process.env[ENV_KEY] = 'ENV-1'; + + const result = await resolveExplicitStory(cwd); + + expect(result).toEqual({ storyId: 'ENV-1', storySource: 'explicit' }); + }); + + it('falls back to the analytics.local.json file when the env var is unset', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + await writeAnalyticsLocalJson(cwd, { storyId: 'FILE-1' }); + delete process.env[ENV_KEY]; + + const result = await resolveExplicitStory(cwd); + + expect(result).toEqual({ storyId: 'FILE-1', storySource: 'explicit' }); + }); + + it('returns null when neither the env var nor the file is set', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + delete process.env[ENV_KEY]; + + const result = await resolveExplicitStory(cwd); + + expect(result).toBeNull(); + }); + + it('returns null when the file is malformed JSON (never throws)', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + const dir = join(cwd, '.claude'); + await mkdir(dir, { recursive: true }); + await writeFile(join(dir, 'analytics.local.json'), '{not valid json', 'utf-8'); + delete process.env[ENV_KEY]; + + const result = await resolveExplicitStory(cwd); + + expect(result).toBeNull(); + }); + }); + + describe('resolveBranchStory', () => { + it("extracts EPMCDME-15301 from 'feature/epmcdme-15301-foo' uppercased", async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('feature/epmcdme-15301-foo'); + + expect(result).toEqual({ storyId: 'EPMCDME-15301', storySource: 'branch' }); + }); + + it('returns null for a branch with no ticket-shaped substring', async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('just-some-branch-name'); + + expect(result).toBeNull(); + }); + + it( + 'returns null for a leading-digit identifier where the negative lookbehind blocks the only possible match start (verified: TICKET_RE requires the match to start on a letter, and the lookbehind forbids an alphanumeric char immediately before that start)', + async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('1ABC-123'); + + expect(result).toBeNull(); + } + ); + + it( + 'matches ABC-123 inside "ABC-123X" (verified actual regex behavior: the trailing (?!\\d) lookahead only blocks a FOLLOWING DIGIT, not a following letter, so "ABC-123X" is NOT a word-boundary case the literal regex rejects)', + async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('ABC-123X'); + + expect(result).toEqual({ storyId: 'ABC-123', storySource: 'branch' }); + } + ); + + it('does not corrupt matching across repeated calls (fresh RegExp per call, no shared lastIndex state)', async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const first = resolveBranchStory('feature/epmcdme-15301-foo'); + const second = resolveBranchStory('feature/epmcdme-15301-foo'); + + expect(first).toEqual({ storyId: 'EPMCDME-15301', storySource: 'branch' }); + expect(second).toEqual({ storyId: 'EPMCDME-15301', storySource: 'branch' }); + }); + }); + + describe('resolveMarkerStory', () => { + it("matches the 'story: X' marker shape, uppercased", async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('story: EPMCDME-999'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it("matches the 'ticket #X' marker shape, uppercased", async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('ticket #EPMCDME-999'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it('matches the marker word case-insensitively (STORY:) and uppercases a lowercase id', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('STORY: epmcdme-999'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it('matches a marker embedded in a longer prompt', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('please fix the bug, story: EPMCDME-999, thanks'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it('returns null when no marker phrase is present', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('just fix the bug please, no ticket mentioned'); + + expect(result).toBeNull(); + }); + + it('returns null for empty prompt text', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory(''); + + expect(result).toBeNull(); + }); + }); + + describe('resolveMentionStory', () => { + it('matches a bare ticket-shaped mention anywhere in the text, uppercased', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const result = resolveMentionStory('can you look into ABC-42 when you get a chance'); + + expect(result).toEqual({ storyId: 'ABC-42', storySource: 'mention' }); + }); + + it('returns null when no ticket-shaped substring exists', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const result = resolveMentionStory('no ticket here, just a plain request'); + + expect(result).toBeNull(); + }); + + it('returns null for empty prompt text', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const result = resolveMentionStory(''); + + expect(result).toBeNull(); + }); + + it('does not corrupt matching across repeated calls (fresh RegExp per call)', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const first = resolveMentionStory('ping on ABC-42 please'); + const second = resolveMentionStory('ping on ABC-42 please'); + + expect(first).toEqual({ storyId: 'ABC-42', storySource: 'mention' }); + expect(second).toEqual({ storyId: 'ABC-42', storySource: 'mention' }); + }); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts new file mode 100644 index 000000000..87eb4f7ed --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -0,0 +1,123 @@ +/** + * Per-forward-tick context: the CLI version banner, and the identity/story resolution glue + * the hook-record mapping step calls once per batch. Split out of `forwarder.ts` to keep + * that module under the documented 500-line structure cap (code-quality.md). + */ + +import { execSync } from 'node:child_process'; +import type { SSOCredentials, JWTCredentials } from '@/providers/core/types.js'; +import { resolveIdentity, type IdentitySource } from './identity.js'; +import { + resolveExplicitStory, + resolveBranchStory, + resolveMarkerStory, + resolveMentionStory, +} from './story-resolver.js'; + +export interface ForwardContext { + credentials: SSOCredentials | JWTCredentials; + baseUrl: string; + projectName: string; + userEmail: string; + /** Per-session git info cache, resolved lazily from the first hook `cwd`. */ + git: { branch?: string; remote?: string }; + /** Per-session developer-identity cache, resolved once */ + identity?: { developerName?: string; identitySource?: IdentitySource }; + /** Per-tick story-id cache, resolved once per forward tick. */ + story?: { storyId?: string; storySource?: 'explicit' | 'branch' }; +} + +/** The installed CodeMie CLI version, resolved once at import time. */ +export function resolveCodemieCliVersion(): string { + try { + const output = execSync('codemie --version', { + encoding: 'utf8', + stdio: ['pipe', 'pipe', 'pipe'], + }).trim(); + const versionMatch = output.match(/(\d+\.\d+\.\d+)/); + return versionMatch ? versionMatch[1] : output; + } catch { + return ''; + } +} + +/** + * Effective story id for ONE `UserPromptSubmit` record: layers this record's + * own prompt text on top of the once-per-tick cache in `ctx.story` + * (explicit/branch/undefined), without ever writing back to that cache — + * other records in the same batch still need it untouched. Priority order + * across the full chain is explicit -> marker -> branch -> mention: + * + * 1. If the cache already resolved to `'explicit'`, that wins outright. + * 2. Otherwise, try the marker tier (`story: X` / `ticket #X`) against this + * record's OWN prompt text — it sits above branch in priority. + * 3. Otherwise, if the cache resolved to `'branch'`, that wins (it is + * already correctly placed between marker and mention). + * 4. Otherwise, try the mention tier (bare ticket-shaped text) — the + * lowest-priority tier. + * 5. Otherwise, empty. + */ +export function resolveStoryForPrompt( + ctx: ForwardContext, + rawPrompt: string +): { storyId: string; storySource: string } { + if (ctx.story?.storySource === 'explicit') { + return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; + } + + const marker = resolveMarkerStory(rawPrompt); + if (marker) { + return { storyId: marker.storyId, storySource: marker.storySource }; + } + + if (ctx.story?.storySource === 'branch') { + return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; + } + + const mention = resolveMentionStory(rawPrompt); + if (mention) { + return { storyId: mention.storyId, storySource: mention.storySource }; + } + + return { storyId: '', storySource: '' }; +} + +/** + * Resolve and cache this forward tick's story id/source once, from the first record that carries + * a real `cwd` — guarded the same way every other per-tick cache in this file is, so a synthetic + * transcript-derived record's empty `cwd` can never poison the cache for the rest of the batch. + */ +export async function resolveStory(ctx: ForwardContext, cwd: string): Promise { + if (!cwd || ctx.story?.storyId !== undefined) { + return; + } + + const explicit = await resolveExplicitStory(cwd); + const resolved = explicit ?? resolveBranchStory(ctx.git.branch ?? ''); + + ctx.story = resolved + ? { storyId: resolved.storyId, storySource: resolved.storySource } + : { storyId: '' }; +} + +/** + * Resolve and cache this forward tick's developer identity once, from the first record that + * carries a real `cwd`. An empty `cwd` (every synthetic transcript-derived record) is skipped + * rather than cached, mirroring the same empty-`cwd` guard every other per-tick cache uses — + * otherwise the first such record in a batch would permanently cache the cwd-less (and + * therefore less accurate) result for every later record in the same tick. + */ +export async function resolveDeveloperIdentity(ctx: ForwardContext, cwd: string): Promise { + if (!cwd) { + return; + } + if (!ctx.identity) { + ctx.identity = {}; + } + if (ctx.identity.developerName !== undefined) { + return; + } + const { developerName, identitySource } = await resolveIdentity(ctx.credentials, cwd); + ctx.identity.developerName = developerName; + ctx.identity.identitySource = identitySource; +} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 358e3e544..931dc3965 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -12,6 +12,16 @@ import { import { OtlpHookSpoolData } from '../otlp.plugin.js'; import { snapshotPendingBytes, snapshotPendingHookRecords } from './spool-io.js'; import { areCredentialsStale, markCredentialsStale } from './auth-state.js'; +import { resolveEmailFromCredentials } from './identity.js'; +import { + type ForwardContext, + resolveCodemieCliVersion, + resolveDeveloperIdentity, + resolveStory, + resolveStoryForPrompt, +} from './forward-context.js'; + +const CODEMIE_CLI_VERSION = resolveCodemieCliVersion(); const HOOK_EVENT_TYPE_MAP: Record = { SessionStart: 'agent.session.start', @@ -40,52 +50,8 @@ const FORWARD_TIMEOUT_MS = 20_000; type SendResult = 'ok' | 'failed' | 'auth-expired'; -interface ForwardContext { - credentials: SSOCredentials | JWTCredentials; - baseUrl: string; - projectName: string; - userEmail: string; - /** Per-session git info cache, resolved lazily from the first hook `cwd`. */ - git: { branch?: string; remote?: string }; -} - /* ------------------------------------------------------------------ auth --- */ -function decodeJwtClaims(token: string): Record { - const parts = token.split('.'); - if (parts.length < 2) { - return {}; - } - try { - return JSON.parse(Buffer.from(parts[1], 'base64url').toString('utf-8')) as Record< - string, - unknown - >; - } catch { - return {}; - } -} - -function resolveUserEmail(credentials: SSOCredentials | JWTCredentials): string { - if (isJWTCredentials(credentials)) { - const claims = decodeJwtClaims(credentials.token); - if (typeof claims['email'] === 'string' && claims['email']) { - return claims['email']; - } - } - if (isSSOCredentials(credentials)) { - const accessToken = credentials.cookies['codemie_access_token']; - if (accessToken) { - const claims = decodeJwtClaims(accessToken); - const email = claims['email'] ?? claims['preferred_username']; - if (typeof email === 'string' && email) { - return email; - } - } - } - return ''; -} - function buildAuthHeadersFromCreds( credentials: SSOCredentials | JWTCredentials ): Record | null { @@ -171,6 +137,10 @@ async function send( /* --------------------------------------------------------------- mapping --- */ function hookEventType(hookName: string, event: Record): string { + const explicitType = event['type']; + if (typeof explicitType === 'string' && explicitType.length > 0) { + return explicitType; + } if (hookName === 'PreToolUse') { return event['input'] && (event['input'] as Record)['denied'] ? 'agent.tool.denied' @@ -187,12 +157,12 @@ function boundedText(value: unknown, maxChars: number): string { typeof value === 'string' ? value : (() => { - try { - return JSON.stringify(value) ?? String(value); - } catch { - return String(value); - } - })(); + try { + return JSON.stringify(value) ?? String(value); + } catch { + return String(value); + } + })(); return text.slice(0, maxChars); } @@ -236,7 +206,10 @@ interface HookPayload { malformed: number; } -async function mapHookRecords(records: string[], ctx: ForwardContext): Promise { +export async function mapHookRecords( + records: string[], + ctx: ForwardContext +): Promise { const mapped: string[] = []; let containsSessionEnd = false; let malformed = 0; @@ -261,22 +234,47 @@ async function mapHookRecords(records: string[], ctx: ForwardContext): Promise { + const parts = token.split('.'); + if (parts.length < 2) { + return {}; + } + try { + return JSON.parse(Buffer.from(parts[1], 'base64url').toString('utf-8')) as Record< + string, + unknown + >; + } catch { + return {}; + } +} + +/** + * Pull `email` straight off a JWT credential's claims, or off the SSO + * session's `codemie_access_token` cookie claims (`email`, falling back to + * `preferred_username`). Shared by the tier-1 `jwt` identity tier below and + * by the forwarder's plain `userEmail` field — both resolve the same claim. + */ +export function resolveEmailFromCredentials( + credentials: SSOCredentials | JWTCredentials +): string { + if (isJWTCredentials(credentials)) { + const claims = decodeJwtClaims(credentials.token); + if (typeof claims['email'] === 'string' && claims['email']) { + return claims['email']; + } + } + if (isSSOCredentials(credentials)) { + const accessToken = credentials.cookies['codemie_access_token']; + if (accessToken) { + const claims = decodeJwtClaims(accessToken); + const email = claims['email'] ?? claims['preferred_username']; + if (typeof email === 'string' && email) { + return email; + } + } + } + return ''; +} + +/** + * Tier 2 — git: `git config user.email`, falling back to `git config + * user.name` when the repo has no email configured. A non-git directory (or + * any exec failure) resolves to `''` so the chain falls through — never + * throws. + */ +async function resolveGitIdentity(cwd: string): Promise { + if (!cwd) { + return ''; + } + try { + const { exec } = await import('@/utils/exec.js'); + const emailResult = await exec('git', ['config', 'user.email'], { cwd }); + if (emailResult.code === 0 && emailResult.stdout.trim()) { + return emailResult.stdout.trim(); + } + const nameResult = await exec('git', ['config', 'user.name'], { cwd }); + if (nameResult.code === 0 && nameResult.stdout.trim()) { + return nameResult.stdout.trim(); + } + } catch { + /* best-effort */ + } + return ''; +} + +/** + * Tier 3 — codemie_cli: the `userEmail` persisted on the global CodeMie CLI + * config (set via `ConfigLoader.saveUserEmail()`). `userEmail` lives on + * `MultiProviderConfig`, not on the merged `CodeMieConfigOptions` that + * `ConfigLoader.load()` returns, so this reads the multi-provider config + * directly via `loadMultiProviderConfig()` — a global lookup, hence no `cwd` + * dependency. Any load failure (missing/unreadable/malformed config) + * resolves to `''`. + */ +async function resolveCodemieCliIdentity(): Promise { + try { + const { ConfigLoader } = await import('@/utils/config.js'); + const config = await ConfigLoader.loadMultiProviderConfig(); + if (config.userEmail) { + return config.userEmail; + } + } catch { + /* best-effort */ + } + return ''; +} + +/** + * Tier 4 — os: the OS-reported username for the daemon process. Practically + * never empty, but guarded anyway since some sandboxed environments can make + * `os.userInfo()` throw. + */ +function resolveOsIdentity(): string { + try { + return userInfo().username || ''; + } catch { + return ''; + } +} + +/** + * Resolve a developer identity for analytics stamping, trying each tier in + * order and returning the first non-empty result: + * + * jwt -> git -> codemie_cli -> os + * + * Never throws — every tier swallows its own failures internally. + */ +export async function resolveIdentity( + credentials: SSOCredentials | JWTCredentials, + cwd: string +): Promise { + try { + const jwtIdentity = resolveEmailFromCredentials(credentials); + if (jwtIdentity) { + return { developerName: jwtIdentity, identitySource: 'jwt' }; + } + + const gitIdentity = await resolveGitIdentity(cwd); + if (gitIdentity) { + return { developerName: gitIdentity, identitySource: 'git' }; + } + + const cliIdentity = await resolveCodemieCliIdentity(); + if (cliIdentity) { + return { developerName: cliIdentity, identitySource: 'codemie_cli' }; + } + + const osIdentity = resolveOsIdentity(); + if (osIdentity) { + return { developerName: osIdentity, identitySource: 'os' }; + } + + return { developerName: '', identitySource: '' }; + } catch { + return { developerName: '', identitySource: '' }; + } +} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts new file mode 100644 index 000000000..13ddbd812 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts @@ -0,0 +1,141 @@ +import { readFile } from 'node:fs/promises'; +import { join } from 'node:path'; + +/** + * Shared ticket-id pattern used by every story-id tier that scans free text + * (branch names today; marker/mention text in a later task). Carries the + * global (`g`) flag, so a stateful `.test()`/`.exec()` on THIS SAME instance + * across repeated calls would corrupt `lastIndex` and silently skip matches + * on the next call. Every consumer in this module therefore either builds a + * fresh `RegExp` from `TICKET_RE.source`/`TICKET_RE.flags` (used here via + * `String.prototype.match()`, which — called on a freshly constructed regex + * — reads all matches once and does not leave mutated `lastIndex` state + * behind for the next caller) before each match, rather than reusing this + * exported instance's `lastIndex` across calls. + */ +export const TICKET_RE = /(?/.claude/analytics.local.json`. + * Read-only — this never writes that file. Swallows every failure (missing + * file, malformed JSON, permission error) and resolves to `null` instead of + * throwing. The resolved `storyId` is taken verbatim from its source (env or + * file) and is NOT upper-cased, unlike the regex-derived `resolveBranchStory` + * tier: a value a user/config explicitly supplied is already exact, whereas + * free text scanned by a case-insensitive regex needs normalizing. + */ +export async function resolveExplicitStory(cwd: string): Promise { + const envStoryId = process.env['SDLC_ANALYTICS_STORY_ID']; + if (envStoryId) { + return { storyId: envStoryId, storySource: 'explicit' }; + } + + try { + const filePath = join(cwd, '.claude', 'analytics.local.json'); + const content = await readFile(filePath, 'utf-8'); + const parsed = JSON.parse(content) as AnalyticsLocalConfig; + if (typeof parsed.storyId === 'string' && parsed.storyId.length > 0) { + return { storyId: parsed.storyId, storySource: 'explicit' }; + } + } catch { + /* missing file, malformed JSON, permission error: fall through to null */ + } + + return null; +} + +/** + * Branch story-id tier: the first `TICKET_RE` match found anywhere in the + * branch name, upper-cased. Returns `null` when the branch carries no + * ticket-shaped substring. + */ +export function resolveBranchStory(branch: string): BranchStoryResult | null { + if (!branch) { + return null; + } + + // Fresh RegExp per call: avoids reusing TICKET_RE's own `lastIndex` across + // invocations (the classic stateful-global-regex-in-a-loop bug). + const matches = branch.match(new RegExp(TICKET_RE.source, TICKET_RE.flags)); + if (!matches || matches.length === 0) { + return null; + } + + return { storyId: matches[0].toUpperCase(), storySource: 'branch' }; +} + +/** + * Marker story-id tier: an explicit `story: X` / `ticket #X` phrase found + * anywhere in prompt text, case-insensitive on the marker word, upper-cased + * on return. Returns `null` when no marker phrase is present (including an + * empty/falsy `promptText`). + */ +export function resolveMarkerStory(promptText: string): MarkerStoryResult | null { + if (!promptText) { + return null; + } + + const match = promptText.match(MARKER_RE); + if (!match || !match[1]) { + return null; + } + + return { storyId: match[1].toUpperCase(), storySource: 'marker' }; +} + +/** + * Mention story-id tier: the first bare `TICKET_RE` match found anywhere in + * prompt text, upper-cased. This is the lowest-priority tier — it only + * applies when no explicit/marker/branch tier already resolved a story. + * Returns `null` when no ticket-shaped substring exists (including an + * empty/falsy `promptText`). + */ +export function resolveMentionStory(promptText: string): MentionStoryResult | null { + if (!promptText) { + return null; + } + + // Fresh RegExp per call: avoids reusing TICKET_RE's own `lastIndex` across + // invocations (the classic stateful-global-regex-in-a-loop bug). + const matches = promptText.match(new RegExp(TICKET_RE.source, TICKET_RE.flags)); + if (!matches || matches.length === 0) { + return null; + } + + return { storyId: matches[0].toUpperCase(), storySource: 'mention' }; +}