From cda5c7502b6d4c9fee5aad22f49c3e44a5d42190 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Mon, 5 Oct 2026 11:53:39 +0300 Subject: [PATCH 01/35] feat(proxy): expanded fields and new usage events (OTLP analytics pipeline) --- .../code-review-final.json | 316 +++++++++++++ .../plan.md | 287 ++++++++++++ .../spec.md | 137 ++++++ src/agents/core/types.ts | 7 +- .../__tests__/claude-code-otlp.plugin.test.ts | 78 +++- .../claude-code-otlp.plugin.ts | 87 ++++ .../__tests__/fixtures/transcript-usage.jsonl | 4 + .../transcript/__tests__/orchestrator.test.ts | 398 ++++++++++++++++ .../transcript/__tests__/parse-state.test.ts | 102 +++++ .../__tests__/session-summary.test.ts | 226 +++++++++ .../__tests__/subagent-usage.test.ts | 295 ++++++++++++ .../__tests__/transcript-reader.test.ts | 82 ++++ .../__tests__/usage-request.test.ts | 203 ++++++++ .../transcript/orchestrator.ts | 433 ++++++++++++++++++ .../transcript/parse-state.ts | 99 ++++ .../transcript/session-summary.ts | 150 ++++++ .../transcript/subagent-usage.ts | 143 ++++++ .../transcript/transcript-reader.ts | 57 +++ .../transcript/usage-request.ts | 195 ++++++++ .../otlp-spool/__tests__/event-id.test.ts | 44 ++ .../otlp-spool/__tests__/forwarder.test.ts | 213 +++++++++ .../otlp-spool/__tests__/identity.test.ts | 72 +++ .../__tests__/story-resolver.test.ts | 218 +++++++++ .../sso/proxy/plugins/otlp-spool/event-id.ts | 28 ++ .../sso/proxy/plugins/otlp-spool/forwarder.ts | 172 +++++-- .../sso/proxy/plugins/otlp-spool/identity.ts | 153 +++++++ .../plugins/otlp-spool/story-resolver.ts | 129 ++++++ 27 files changed, 4300 insertions(+), 28 deletions(-) create mode 100644 docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json create mode 100644 docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md create mode 100644 docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/parse-state.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/session-summary.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/usage-request.ts create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json new file mode 100644 index 000000000..59a1a254c --- /dev/null +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json @@ -0,0 +1,316 @@ +{ + "decision": "request-changes", + "rationale": "All four lenses and the standards audit ran cleanly against the approved spec; triage surfaced 23 blocking findings, including 6 decision_needed items (skill-scope attribution, lines_added/lines_removed sourcing, the subagent description field, commands_in_order ordering, started_at/ended_at sourcing, and the missing identity-derivation security review) and a security-practices CRITICAL violation (unreviewed jwt/git/os developer-identity chain). 0 findings deferred (nothing was judged pre-existing/out-of-scope). 6 blind/edge-case findings were dismissed below the blocking floor: silent version-read fallback matching existing codebase convention, an unbounded-but-local-only story-id config value, a stale per-tick-vs-per-session caching comment, a hookEventType passthrough required by the new synthetic-event design, a misleading-but-harmless doc comment on saveParseState's swallow behavior, and confirmation (via repo-wide grep) that no second OtlpAgentAdapter implementer exists to miss the new interface method.", + "confidence": "low", + "risk_flags": ["security"], + "business_review": [ + { "kind": "spec", "item": "Common fields via prepareAnalyticsFields()/AgentRegistry.getAnalyticsAgent merged into every mapped record", "status": "pass", "notes": "Matches spec; minor deviation: Record return type and an extra undocumented plugin_version field." }, + { "kind": "spec", "item": "event_id: deterministic per-type composed id (hooks by byte offset; usage.request by request_id+model; subagent.usage by tool_use_id; session.summary by phase)", "status": "pass", "notes": "All four formulas implemented and tested as specified; see CR-015/CR-018 for collision edge cases on two of the formulas." }, + { "kind": "spec", "item": "Incremental persisted parse state per session (mainOffset, subagentOffsets, openRequests, activeSkill, branchCounts)", "status": "pass", "notes": "Implemented; see CR-011 for a concurrency gap and CR-014 for a rotation/truncation gap in the surrounding mechanism." }, + { "kind": "spec", "item": "Each trigger updates tallies, derives records, writes state, THEN sends", "status": "partial", "notes": "Code forwards touched records/summary before saveParseState — order reversed vs spec text. Mitigated by idempotent event_id design and a crash-before-save test." }, + { "kind": "spec", "item": "Recovery: missing/corrupt state falls back to full parse from byte 0; re-derived records share event_id with originals", "status": "pass", "notes": "loadParseState falls back to createParseState(); idempotent-reparse test confirms matching ids across simulated crash." }, + { "kind": "spec", "item": "Transcript parsing stays async, swallows errors, never blocks/fails the hook, always exits 0", "status": "pass", "notes": "runMainTranscriptParse/runSubagentTranscriptParse wrap bodies in catch-swallow; fire-and-forget from evaluate(); hook.ts exits 0." }, + { "kind": "spec", "item": "Transcript field sourcing: request_id from message.id, stop_reason sibling of usage, usage flat + nested cache/server_tool_use groups", "status": "pass", "notes": "parseUsageLine() matches every documented field path; tested against a verified fixture." }, + { "kind": "spec", "item": "agent.usage.request fires on Stop/PreCompact/SessionEnd (main) and SubagentStop/SessionEnd-backstop (subagent)", "status": "pass", "notes": "Wired for all four; see CR-003 for the StopFailure gap in trigger coverage." }, + { "kind": "spec", "item": "agent.usage.request scope_kind/scope_name vary main/skill/agent by context", "status": "partial", "notes": "Main-transcript records are unconditionally hardcoded scopeKind:'main' ('Note A' judgment call); 'skill' scope is never produced; activeSkill tracked but unused." }, + { "kind": "spec", "item": "agent.usage.request full field list (request_id, timestamps, model fields, token fields, scope, agent_id, stop_reason, is_api_error, git_branch)", "status": "pass", "notes": "buildUsageRequestEvent() emits every listed field; round-tripped by tests." }, + { "kind": "spec", "item": "agent.usage.request: one per unique (request_id, model), max-merge of numeric fields across duplicates", "status": "pass", "notes": "openRequests keyed by request_id::model shared across main/subagent passes; mergeUsageRequest() takes per-field max. See CR-015 for a request_id='' collision edge case." }, + { "kind": "spec", "item": "agent.subagent.usage fires on SubagentStop and SessionEnd backstop, never Stop/PreCompact", "status": "pass", "notes": "Confirmed by orchestrator wiring and a 3-subagent backstop test." }, + { "kind": "spec", "item": "agent.subagent.usage 'description' sourced from the subagent's .meta.json sidecar", "status": "fail", "notes": "No such sidecar field exists anywhere in the codebase (mirrors the real production schema, which has none); hardcoded to '' unconditionally. Undisclosed in spec.md's Open risks." }, + { "kind": "spec", "item": "agent.subagent.usage 'spawn_depth' defaults appropriately for top-level subagents", "status": "pass", "notes": "file.spawnDepth ?? 0, tested." }, + { "kind": "spec", "item": "agent.subagent.usage token/cache/api-call/tool/skill fields aggregated from the subagent's own transcript", "status": "pass", "notes": "buildSubagentUsageEvent() sums caller-scoped OpenUsageRequest[]; cross-checked against summed usage.request totals in tests." }, + { "kind": "spec", "item": "agent.session.summary fires on Stop (incremental) and SessionEnd (final), never SubagentStop/PreCompact", "status": "pass", "notes": "Confirmed by orchestrator wiring and a PreCompact test asserting zero summary events." }, + { "kind": "spec", "item": "agent.session.summary primary_model/primary_command/branch_dominant are the max-count key of their maps", "status": "pass", "notes": "primaryModel()/branchDominant()/maxKey() implement and test all three." }, + { "kind": "spec", "item": "agent.session.summary lines_added/lines_removed derived from Edit/Write tool payloads", "status": "fail", "notes": "buildFullAccumulator() never increments either field; always emits 0. Code comment discloses the limitation; spec.md's Open risks does not." }, + { "kind": "spec", "item": "agent.session.summary compaction_count counts this session's PreCompact triggers", "status": "fail", "notes": "Never incremented anywhere despite trigger already being known per call; always emits 0. Not disclosed in spec.md's Open risks; trivially fixable." }, + { "kind": "spec", "item": "agent.session.summary commands_in_order lists slash-commands in invocation order", "status": "partial", "notes": "Emits Object.keys() of an unordered count map, not a chronological sequence. Code comment calls this 'a genuine mismatch' with the field's name." }, + { "kind": "spec", "item": "agent.session.summary api_calls = count of this session's agent.usage.request records", "status": "fail", "notes": "session-summary.ts deliberately omits it pending an orchestrator merge; orchestrator.ts never performs that merge. Field is absent from every emitted event." }, + { "kind": "spec", "item": "agent.session.summary started_at/ended_at sourced from SessionStart/SessionEnd hook timestamps", "status": "partial", "notes": "started_at comes from the transcript's own first-line timestamp; ended_at from Date.now() at parse time — neither reads the actual hook payload timestamp." }, + { "kind": "spec", "item": "agent.session.summary title field", "status": "pass", "notes": "Emits '' literal, matching spec.md's own disclosed Open-risk that no source exists." }, + { "kind": "spec", "item": "Story resolution: explicit->branch per-tick cache for non-prompt events; explicit->marker->branch->mention for prompt events against the record's own untruncated prompt", "status": "pass", "notes": "resolveStoryOnce()/resolvePromptStory() implement the priority chain; tested. See CR-019 for a cache-poisoning edge case in the per-tick cache." }, + { "kind": "spec", "item": "Explicit story source: SDLC_ANALYTICS_STORY_ID env var then .claude/analytics.local.json, read-only, gitignored", "status": "pass", "notes": "resolveExplicitStory(); file is gitignored; nothing writes it." }, + { "kind": "spec", "item": "Identity resolution: extend resolveUserEmail's jwt-only chain to jwt->git->codemie_cli->claude_account->os; user_email unchanged", "status": "pass", "notes": "All five tiers implemented and tested; original user_email/resolveUserEmail left untouched. See CR-023 for the required security-review gap on this new chain." }, + { "kind": "spec", "item": "Non-goals respected: no change to existing event content beyond common fields, no server changes, three named events stay out of scope", "status": "pass", "notes": "All changes confined to the documented modules; none of the three out-of-scope event types appear in the diff." } + ], + "standards_review": [ + { "kind": "commit-format", "status": "na", "notes": "git log over the review range is empty — all changes are uncommitted working-tree edits on top of diff_base, per explicit user instruction." }, + { "kind": "code-quality", "status": "partial", "notes": "forwarder.ts grew to 538 lines (>500-line structure cap); identity.ts (new file) uses a deep relative import instead of the documented '@/' alias." }, + { "kind": "security", "status": "fail", "notes": "New jwt->git->codemie_cli->os developer-identity derivation stamped on every analytics event with no recorded security-review sign-off, per security-practices.md's CRITICAL attribution-header rule." } + ], + "findings": [ + { + "id": "CR-001", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts", + "title": "Plugin's hook-dispatch glue is untested", + "problem": "evaluate()'s field extraction and dispatch (agent_transcript_path/agent_id/tool_use_id/agent_type parsing, hookEventName branching into runMainTranscriptParse/runSubagentTranscriptParse) is only covered indirectly — only prepareAnalyticsFields is tested in this file, and orchestrator tests call the orchestrator functions directly, bypassing this glue entirely.", + "impact": "A typo or regression in this wiring (wrong field name, wrong hookEventName comparison) would silently stop all transcript-derived analytics from firing in production while every existing test keeps passing.", + "recommendation": "Add a test that feeds a raw Stop/PreCompact/SessionEnd/SubagentStop hook payload through processOtlpEvent/evaluate and asserts runMainTranscriptParse/runSubagentTranscriptParse were invoked with the correctly-extracted arguments." + }, + { + "id": "CR-002", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts", + "line": 46, + "title": "Empty session_id bypasses hook.ts's own validation", + "problem": "hook.ts dispatches to analyticsAgent.processOtlpEvent() (and returns) before its own `if (!event.session_id)` validation runs. evaluate() then uses event.sessionId unchecked to key loadParseState/saveParseState and the subagent discovery path.", + "impact": "A hook event with an empty/missing session_id would read and write the shared state file keyed by an empty string, cross-contaminating offsets/openRequests/branchCounts across any other such session.", + "recommendation": "Add `if (!event.sessionId) return { decision: 'forward', payload: rawEvent };` at the top of evaluate(), before any transcript-parse dispatch." + }, + { + "id": "CR-003", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts", + "line": 47, + "title": "StopFailure never triggers transcript parsing", + "problem": "evaluate() only dispatches runMainTranscriptParse on hookEventName === 'Stop' | 'PreCompact' | 'SessionEnd', even though HOOK_EVENT_TYPE_MAP already maps StopFailure to agent.turn.error as a distinct, modeled event type.", + "impact": "When a turn ends via StopFailure, that turn's agent.usage.request/agent.session.summary data is not forwarded until a later Stop/SessionEnd eventually catches up (or never, if the session terminates without one).", + "recommendation": "Include 'StopFailure' alongside Stop/PreCompact/SessionEnd in the trigger condition for runMainTranscriptParse." + }, + { + "id": "CR-004", + "kind": "code", + "severity": "critical", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 88, + "title": "compaction_count always emits 0", + "problem": "acc.compactionCount is initialized to 0 in emptyAccumulator() and never incremented anywhere, despite runMainTranscriptParse already knowing the trigger ('PreCompact' or otherwise) on every call.", + "impact": "agent.session.summary's compaction_count field, a required spec field, is always 0 regardless of actual PreCompact activity in the session.", + "recommendation": "Increment a persisted compactionCount counter in TranscriptParseState when trigger === 'PreCompact', and surface it in buildSessionSummaryEvent's output." + }, + { + "id": "CR-005", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 111, + "title": "lines_added/lines_removed never computed", + "problem": "buildFullAccumulator()'s own docstring states Edit/Write tool_use payloads carry the proposed edit, not a diff stat, so no reliable added/removed line count can be derived without re-implementing diffing — out of scope per the author's own note, but not disclosed as such in spec.md's Open risks.", + "impact": "agent.session.summary's lines_added/lines_removed fields, both required spec fields, always emit 0.", + "recommendation": "Get a product/spec decision: either implement real diff-based line counting (a larger change) or formally amend spec.md to disclose this as an accepted limitation, matching how title/workflow_run/worktree are already handled." + }, + { + "id": "CR-006", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 194, + "title": "Skill scope_kind is never produced", + "problem": "runMainTranscriptParse's own 'Note A' comment documents a judgment call: every main-transcript usage record is unconditionally scoped 'main', since no reliable signal for skill-context was found. state.activeSkill is tracked but never read or written.", + "impact": "One of the three documented scope_kind values ('skill') is never emitted by this implementation, so skill-scoped usage can never be distinguished from main-scoped usage in the analytics backend.", + "recommendation": "Get a product decision on whether skill-context detection is required now (and, if so, identify a reliable in-transcript signal) or should be formally deferred to a later task." + }, + { + "id": "CR-007", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 235, + "title": "State is saved after events are sent, reversing the spec'd order", + "problem": "spec.md specifies 'updates tallies, derives records, writes state to disk, THEN sends'; runMainTranscriptParse/runSubagentTranscriptParse instead forward every touched usage/summary/subagent event and only call saveParseState() afterward.", + "impact": "Functionally mitigated today by the idempotent event_id design and a dedicated crash-before-save re-parse test, but the literal spec'd ordering guarantee is not met by the code as written.", + "recommendation": "Either reorder to save-then-send to match spec text, or update spec.md to describe the as-implemented idempotent-reconciliation approach so the two stay consistent." + }, + { + "id": "CR-008", + "kind": "decision", + "severity": "major", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 242, + "title": "started_at/ended_at are not sourced from hook timestamps", + "problem": "spec.md specifies started_at/ended_at should come from the SessionStart/SessionEnd hook payload's own timestamp fields; the code instead sources started_at from the transcript's first parsed line and ended_at from new Date().toISOString() at parse time.", + "impact": "Both values are close approximations of the intended timestamps but not sourced as specified; a gap between actual session start and first transcript line would skew started_at.", + "recommendation": "Get a decision on whether the current approximation is acceptable, or whether the actual hook timestamps need to be threaded through runMainTranscriptParse's call sites." + }, + { + "id": "CR-009", + "kind": "code", + "severity": "critical", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 244, + "title": "session.summary's api_calls field is never populated", + "problem": "session-summary.ts deliberately omits api_calls, documenting that the orchestrator is responsible for merging it in afterward from its own view of this session's agent.usage.request records; runMainTranscriptParse never performs that merge before forwarding the summary event.", + "impact": "api_calls, a required spec field, is absent from every agent.session.summary event this implementation emits.", + "recommendation": "Before forwarding summaryEvent, merge in api_calls: Object.keys(state.openRequests).length (or an equivalent count of this session's usage-request records)." + }, + { + "id": "CR-010", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", + "line": 341, + "title": "duration_ms can go negative on out-of-order transcript lines", + "problem": "scanSubagentTranscript's durationMs guard only checks Number.isFinite(diff), which is true for negative numbers too — it does not clamp a negative diff when a subagent transcript's last line predates its first line.", + "impact": "agent.subagent.usage can carry a negative duration_ms value in that edge case.", + "recommendation": "Clamp with Math.max(0, diff) instead of only checking Number.isFinite(diff)." + }, + { + "id": "CR-011", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/parse-state.ts", + "line": 71, + "title": "Concurrent SubagentStop processes race on the shared parse-state file", + "problem": "Each hook fire is a fresh CLI process; loadParseState/saveParseState perform a plain read-modify-write with no file locking or atomic update. The plugin's own comment only reasons about races inside its single SessionEnd backstop IIFE, not about genuinely concurrent SubagentStop processes for sibling subagents.", + "impact": "Two sibling subagents' SubagentStop hooks firing concurrently can race on the same session state file; the last writer wins and the other process's openRequests/subagentOffsets/branchCounts update is silently lost.", + "recommendation": "Add a per-session file lock (e.g. proper-lockfile) around the load+save cycle, or implement an atomic read-modify-write with retry on conflict." + }, + { + "id": "CR-012", + "kind": "decision", + "severity": "major", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/session-summary.ts", + "line": 132, + "title": "commands_in_order is not actually ordered", + "problem": "The module's own docstring admits commands_in_order is derived as Object.keys() of commandInvocations, an unordered count map — there is no chronological invocation sequence anywhere in NamedInvocationCounts to draw from.", + "impact": "The emitted field contains the right distinct command names but not in invocation order, despite the field's name implying ordering.", + "recommendation": "Get a decision: accept the distinct-names-only semantics (and rename/document the field accordingly), or extend extractNamedInvocations() upstream to track true invocation order." + }, + { + "id": "CR-013", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts", + "line": 126, + "title": "subagent description field has no data source", + "problem": "buildSubagentUsageEvent() hardcodes description: '' unconditionally; the sidecar .meta.json schema it mirrors (matching the real claude.session.ts schema) has no description key at all, so there is no source anywhere in this codebase to populate it from.", + "impact": "agent.subagent.usage's description field, a required spec field, is always empty — unlike the sibling workflow_run/worktree/title gaps, this one is not disclosed in spec.md's Open risks.", + "recommendation": "Get a decision: accept the always-empty value as a disclosed limitation (matching title/workflow_run/worktree precedent) or identify an alternate data source." + }, + { + "id": "CR-014", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts", + "line": 34, + "title": "A rotated/truncated transcript permanently stalls parsing", + "problem": "readNewLines treats size <= fromOffset as 'nothing new (or file rotated/truncated)' and returns nextOffset: fromOffset unchanged — so after a rotation/truncation, the next call compares the new (smaller) size to the same stale offset and finds the same condition forever.", + "impact": "Once a transcript file is rotated or truncated below the persisted offset, parsing for that session permanently stalls; content written after the rotation is never read again.", + "recommendation": "When size < fromOffset specifically (as opposed to size === fromOffset), reset fromOffset to 0 before the nothing-new check so a rotation resumes a fresh parse." + }, + { + "id": "CR-015", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/agents/plugins/claude-code-otlp/transcript/usage-request.ts", + "line": 98, + "title": "usage.request event_id collides when message.id is absent", + "problem": "parseUsageLine defaults requestId to '' when message.id is absent; both openRequests' key (${requestId}::${model}) and computeEventId's agent.usage.request formula key on request_id+model, with no guard against an empty request_id.", + "impact": "Two distinct requests to the same model that both omit message.id collide into one openRequests entry/event_id; mergeUsageRequest's max-merge then blends unrelated usage into a single reported record, losing or misattributing part of the real usage.", + "recommendation": "Skip lines with an empty requestId (do not key openRequests/event_id on `::model` alone), rather than silently merging them." + }, + { + "id": "CR-016", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts", + "line": 16, + "title": "prepareAnalyticsFields merge path is never exercised by tests", + "problem": "Every mapHookRecords test in this file hardcodes agentName: 'claude', which is not a registered OTLP agent name (the registered name is 'claude-code-otlp'), so AgentRegistry.getAnalyticsAgent always resolves undefined and commonFields is always {} via the '?? {}' fallback.", + "impact": "A regression in the AgentRegistry lookup or in merging prepareAnalyticsFields's result (wrong key, dropped field, wrong agent-name string) would ship silently, since no existing test exercises the real merge path or asserts platform/client_version/plugin_version/agent_id/agent_type on the output.", + "recommendation": "Use the real registered agent name ('claude-code-otlp') in at least one test, and assert the common fields appear on the mapped record output." + }, + { + "id": "CR-017", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts", + "title": "developer_name/identity_source are never asserted in forwarder tests", + "problem": "This change replaces the previous developer_name: ctx.userEmail with a new jwt->git->codemie_cli->claude_account->os identity-resolution chain wired through ctx.identity, but no test in this file asserts the resulting developer_name/identity_source fields on a mapped record.", + "impact": "A regression that breaks resolveIdentityOnce's wiring into the output (e.g. a caching bug, or a silent revert to ctx.userEmail) would ship with every forwarded record carrying an empty/wrong developer_name/identity_source and no test would catch it.", + "recommendation": "Add an assertion on developer_name/identity_source in mapHookRecords' test coverage, exercising at least one non-jwt tier." + }, + { + "id": "CR-018", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts", + "line": 35, + "title": "subagent.usage event_id collides when tool_use_id is absent", + "problem": "computeEventId's agent.subagent.usage case keys solely on tool_use_id; buildSubagentUsageEvent emits tool_use_id: file.toolUseId ?? '', and findSubagentFiles leaves toolUseId undefined whenever the sidecar .meta.json omits it — true for top-level subagents per this module's own doc comment.", + "impact": "Any two subagents in the same session that both lack a sidecar tool_use_id collide on the identical event_id (${sessionId}:agent.subagent.usage:); the backend's at-least-once dedup keeps only one and silently drops the other's usage event.", + "recommendation": "Fall back to agent_id when tool_use_id is empty, e.g. `${sessionId}:agent.subagent.usage:${toolUseId || agentId}`." + }, + { + "id": "CR-019", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts", + "line": 258, + "title": "Identity/story caches can be poisoned by the first empty-cwd record in a batch", + "problem": "resolveIdentityOnce/resolveStoryOnce cache their result on the first call per tick with no guard for an empty cwd, unlike resolveGitInfo's explicit `if (!cwd ...) return;` early-return that defers caching until a record with a real cwd arrives.", + "impact": "When the first record in a forward tick has an empty cwd (synthetic transcript-derived events never carry one) and a later record in the same batch has a real cwd, developer_name/identity_source and story_id/story_source get permanently cached as the less-accurate (or empty) result for every record in that batch.", + "recommendation": "Add the same `if (!cwd) return;` early-return pattern resolveGitInfo uses before caching ctx.identity/ctx.story." + }, + { + "id": "CR-020", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts", + "line": 387, + "title": "prepareAnalyticsFields call has no try/catch despite a documented 'must never throw' contract", + "problem": "mapHookRecords calls analyticsAgent?.prepareAnalyticsFields(hookEvent) with no surrounding try/catch; OtlpAgentAdapter.prepareAnalyticsFields's 'must never throw' contract is only a doc comment, not enforced at this one call site that crosses the plugin boundary.", + "impact": "The current (only) implementation is written defensively and never throws, but a future/alternate adapter implementation that violates the documented contract would throw out of the per-record loop and abort processing of every remaining record in that forward tick.", + "recommendation": "Wrap the call in try/catch, defaulting commonFields to {} on failure, so one misbehaving adapter can never abort the whole batch." + }, + { + "id": "CR-021", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts", + "title": "forwarder.ts exceeds the 500-line structure guideline", + "problem": "forwarder.ts grew from 383 lines at diff_base to 538 lines in this change by inlining the identity/story resolution glue and the CLI-version loader, exceeding code-quality.md's documented file-size cap.", + "impact": "Flagged by the standards audit as a blocking code-quality violation.", + "recommendation": "Extract resolveIdentityOnce/resolveStoryOnce/resolvePromptStory and/or loadCodemieCliVersion into their own module so forwarder.ts drops back under the guideline." + }, + { + "id": "CR-022", + "kind": "code", + "severity": "major", + "triage": "patch", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts", + "line": 2, + "title": "identity.ts uses a deep relative import instead of the '@/' alias", + "problem": "This new file imports core types via '../../../../../core/types.js' instead of the documented '@/' alias (AGENTS.md's Common Pitfalls table).", + "impact": "Flagged by the standards audit as a blocking code-quality violation for new code, independent of the same pre-existing pattern already in forwarder.ts/tick-processor.ts.", + "recommendation": "Replace the five-level relative import with the '@/' alias.", + "outcome": "fixed", + "resolution": "Replaced '../../../../../core/types.js' with '@/providers/core/types.js' (not '@/agents/core/types.js' as originally recommended — the five-level path from identity.ts resolves to src/providers/core/types.ts, which is where SSOCredentials/JWTCredentials/isSSOCredentials/isJWTCredentials actually live; verified via `tsc --noEmit`) in both identity.ts and __tests__/identity.test.ts, which had the same deep relative import. `npm run typecheck`, targeted eslint, and identity.test.ts (3 tests) all pass." + }, + { + "id": "CR-023", + "kind": "decision", + "severity": "critical", + "triage": "decision_needed", + "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts", + "title": "New developer-identity derivation chain lacks a required security review", + "problem": "security-practices.md's CRITICAL 'Project & User Attribution Headers' rule requires security review before deriving an attribution identifier from a new source (process introspection, filesystem/environment heuristics). identity.ts's resolveIdentity() (jwt -> git config -> codemie_cli config -> OS username) is exactly that, and forwarder.ts stamps its result as developer_name/identity_source onto every outbound analytics event.", + "impact": "These identifiers back audit trails and analytics attribution; an unreviewed derivation source risks misattributing usage/abuse across users, per the guide's own stated rationale. Nothing in the diff or commit history records a sign-off.", + "recommendation": "Record explicit security-review sign-off for the new identity-derivation chain, or obtain written confirmation that analytics-only developer_name stamping sits outside the attribution-header checklist's scope, and document that decision." + } + ] +} diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md new file mode 100644 index 000000000..1f196440a --- /dev/null +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md @@ -0,0 +1,287 @@ +# EPMCDME-15301 Sub-stage 1.1 — Common Fields, Transcript Parsing, New Usage Events + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Extend the `claude-code-otlp` analytics pipeline with common fields on every `agent.*` event, incremental persisted transcript parsing, three new events (`agent.usage.request`, `agent.subagent.usage`, `agent.session.summary`), story resolution, and an extended identity chain — per `spec.md`. + +**Architecture:** Two layers already exist and are extended, not replaced. (1) Hook-side (`ClaudeCodeOtlpPlugin.processOtlpEvent`, runs once per Claude Code hook invocation, CLI process, must exit 0): gains a `prepareAnalyticsFields()` method for per-record agent-owned fields, and — on `Stop`/`SubagentStop`/`PreCompact`/`SessionEnd` — an orchestration step that incrementally parses the transcript(s) and forwards each derived event as its own synthetic hook-shaped record via the existing `forwardOtlpEventToSpool()`, tagged with an explicit `type` so it is **not** re-mapped by the hook-name table. (2) Daemon-side (`forwarder.ts`, long-running proxy tick): stamps `schema_version`, `event_id`, `codemie_cli_version`, `story_id`/`story_source`, `developer_name`/`identity_source` onto every record — old and new — in `mapHookRecords()`. + +**Tech Stack:** TypeScript, Node `node:fs/promises`/`node:crypto`/`node:child_process`, Vitest. No new runtime dependencies. + +## Global Constraints + +- `schema_version = 2` on every event. `platform = 'claude-code'` (constant). +- Truncation unchanged: prompt 200 chars (`MAX_PROMPT_CHARS`), tool input/output/error 300 chars (`MAX_TOOL_FIELD_CHARS`) — both already defined in `forwarder.ts:37-38`. +- Hooks/orchestration stay `async`, swallow all exceptions internally, never throw past the top-level handler, never block Claude Code. +- Node only, no new npm dependencies. +- `event_id` is a pure string function of fields already on the record — never a generated/stored UUID. +- Only the *resolved* `story_id`/`story_source` is ever sent — never raw prompt text. Nothing in this sub-stage writes `.claude/analytics.local.json` (read-only here). +- Ticket regex (shared constant): `/(?): string` — branches on `type`: existing 13 types use `${sessionId}:${type}:${byteOffset}` (`fields.byteOffset: number`); `agent.usage.request` uses `${sessionId}:agent.usage.request:${request_id}:${model}`; `agent.subagent.usage` uses `${sessionId}:agent.subagent.usage:${tool_use_id}`; `agent.session.summary` uses `${sessionId}:agent.session.summary:${phase}`. + +**Test-first: yes — `computeEventId` returns the exact byte-offset formula for an existing-event type and the request/model-keyed formula for `agent.usage.request`; `mapHookRecords()` on two records yields two different `event_id`s and both carry `schema_version: 2`.** + +- [ ] Write failing tests for `computeEventId`'s four branches and for `mapHookRecords` stamping `schema_version`/`event_id`/`codemie_cli_version`. +- [ ] Implement `event-id.ts` and the `forwarder.ts` edits. +- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 2: Agent-owned common fields (`prepareAnalyticsFields`) + +**Files:** +- Modify: `src/agents/core/types.ts:734-739` — add `prepareAnalyticsFields(hookEvent: Record): Promise>` to `OtlpAgentAdapter` (return type kept generic so core/types.ts has no dependency on a leaf plugin's types). +- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts` — implement it: `platform: 'claude-code'`, `entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? ''`, `client_version` from `claude --version` (reuse the `exec()` helper from `src/utils/exec.ts`, same approach as `ClaudeAgentAdapter.getVersion()` at `src/agents/plugins/claude/claude.plugin.ts:633`; cache the resolved version on the instance so a high-frequency hook like `PostToolUse` doesn't spawn a subprocess per call), `plugin_version` from `src/agents/plugins/claude/plugin/.claude-plugin/plugin.json`'s `version` field, `agent_id`/`agent_type` read directly off `hookEvent['agent_id']`/`hookEvent['agent_type']` when present (subagent context only). +- Modify: `src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts` — in `mapHookRecords()`, resolve `AgentRegistry.getAnalyticsAgent(spoolData.agentName)` and merge `await analyticsAgent?.prepareAnalyticsFields(hookEvent) ?? {}` into the mapped record before the daemon-side fields. +- Test: `src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts` (new). + +**Interfaces:** +- Consumes: `AgentRegistry.getAnalyticsAgent(name): OtlpAgentAdapter | undefined` (`src/agents/registry.ts:76`). +- Produces: `ClaudeCodeOtlpPlugin.prepareAnalyticsFields()` — relied on by Task 11/12's orchestrator for `agent_id` on `agent.usage.request`/`agent.subagent.usage`. + +**Test-first: yes — `prepareAnalyticsFields()` on a hook event carrying `agent_id`/`agent_type` returns both; on one without, it omits them; `client_version` is only spawned once across two calls.** + +- [ ] Write failing tests (mock `exec()` to assert single invocation across two calls). +- [ ] Implement. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 3: Identity resolution chain + +**Files:** +- Create: `src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts` +- Modify: `forwarder.ts:67-81,273-288` — `buildForwardContext()` calls the new resolver once per tick instead of the inline `resolveUserEmail()`; `mapHookRecords()` stamps `developer_name`/`identity_source` from the resolved result. +- Test: `src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts`. + +**Interfaces:** +- Produces: `resolveIdentity(credentials: SSOCredentials | JWTCredentials, cwd: string): Promise<{ developerName: string; identitySource: 'jwt' | 'git' | 'codemie_cli' | 'claude_account' | 'os' | '' }>` — tries, in order: existing JWT-claims logic (moved from `resolveUserEmail`), `git config user.email` / `user.name` via `exec()`, the existing `codemie_cli` profile config loader, a best-effort `claude_account` lookup that returns nothing if unavailable (documented limitation, falls through), `os.userInfo().username`. First non-empty wins. + +**Test-first: yes — with JWT absent/empty, `resolveIdentity` falls through to git email when `git config user.email` succeeds, and to `os.userInfo().username` when every other tier is empty.** + +- [ ] Write failing tests covering: jwt hit, jwt-miss→git-hit, all-miss→os-fallback. +- [ ] Implement `identity.ts`, wire into `forwarder.ts`. +- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 4: Story resolution — explicit + branch tiers (non-prompt events) + +**Files:** +- Create: `src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts` +- Modify: `forwarder.ts:200-213,273-288` — `buildForwardContext()` resolves explicit/branch story once per tick (same per-tick-cache shape as `resolveGitInfo`); `mapHookRecords()` stamps `story_id`/`story_source` on every non-`agent.prompt.submit` record. +- Modify: `.gitignore` — add `.claude/analytics.local.json`. +- Test: `src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts`. + +**Interfaces:** +- Produces: `TICKET_RE` (exported shared regex, per Global Constraints), `resolveExplicitStory(cwd: string): Promise<{ storyId: string; storySource: 'explicit' } | null>` (checks `SDLC_ANALYTICS_STORY_ID` env first, then reads `/.claude/analytics.local.json`'s `storyId` field — read-only, never writes it), `resolveBranchStory(branch: string): { storyId: string; storySource: 'branch' } | null` (first `TICKET_RE` match). + +**Test-first: yes — `resolveExplicitStory` prefers the env var over the file when both are set; `resolveBranchStory` extracts `EPMCDME-15301` from `feature/epmcdme-15301-foo` uppercased.** + +- [ ] Write failing tests for both resolvers plus the regex's word-boundary behavior (no match inside `ABC-123X`). +- [ ] Implement `story-resolver.ts`, wire into `forwarder.ts`, add the `.gitignore` line. +- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 5: Story resolution — marker/mention tiers (`agent.prompt.submit`) + +**Files:** +- Modify: `story-resolver.ts` (Task 4) — add the marker/mention tiers. +- Modify: `forwarder.ts:221-269` — in `mapHookRecords()`, for records whose `hookEvent['hook_event_name'] === 'UserPromptSubmit'`, resolve against the record's own **untruncated** `hookEvent['prompt']` (read before `limitHookPayload()` truncates it) in priority order explicit → marker → branch → mention, overriding the per-tick explicit/branch result from Task 4 only when a higher-priority prompt-level tier exists. +- Test: extend `__tests__/story-resolver.test.ts`. + +**Interfaces:** +- Produces: `resolveMarkerStory(promptText: string): { storyId; storySource: 'marker' } | null` (matches `story: X` / `ticket #X`, case-insensitive), `resolveMentionStory(promptText: string): { storyId; storySource: 'mention' } | null` (bare `TICKET_RE` match anywhere in the text). + +**Test-first: yes — a prompt containing `story: EPMCDME-999` resolves to `storySource: 'marker'` even when the branch carries a different ticket; a prompt with no marker but a bare `ABC-42` mention resolves to `storySource: 'mention'`; the raw prompt text itself is never present on the emitted record (only `prompt_body`, truncated, and `story_id`/`story_source`).** + +- [ ] Write failing tests for marker precedence over mention, and for the no-match case falling back to the Task 4 branch/explicit result. +- [ ] Implement. +- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 6: Transcript parse-state persistence + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/parse-state.ts` +- Test: `src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts`. + +**Interfaces:** +- Produces: +```ts +export interface OpenUsageRequest { + requestId: string; model: string; modelRaw: string; timestamp: string; + speed: string; inferenceGeo: string; serviceTier: string; + inputTokens: number; cacheCreation5mTokens: number; cacheCreation1hTokens: number; + cacheReadTokens: number; outputTokens: number; + webSearchRequests: number; webFetchRequests: number; + scopeKind: 'main' | 'skill' | 'agent'; scopeName: string; agentId: string; + stopReason: string; isApiError: boolean; gitBranch: string; +} +export interface TranscriptParseState { + mainOffset: number; + subagentOffsets: Record; + openRequests: Record; // key: `${requestId}::${model}` + activeSkill: string; + branchCounts: Record; +} +export function createParseState(): TranscriptParseState; +export async function loadParseState(sessionId: string): Promise; // missing or corrupt file -> fresh state, never throws +export async function saveParseState(sessionId: string, state: TranscriptParseState): Promise; +``` +Stored at `getCodemiePath('analytics', 'state', `${sessionId}.json`)` (`src/utils/paths.ts:385`), directory created on write. + +**Test-first: yes — `loadParseState` on a missing file returns `createParseState()`'s fresh shape; on a corrupt JSON file it also recovers to fresh rather than throwing; `saveParseState` followed by `loadParseState` round-trips `openRequests` and `branchCounts` exactly.** + +- [ ] Write failing tests for missing/corrupt/round-trip. +- [ ] Implement `parse-state.ts`. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 7: Incremental transcript reader + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts` +- Test: `.../__tests__/transcript-reader.test.ts`. + +**Interfaces:** +- Produces: `readNewLines(filePath: string, fromOffset: number): Promise<{ lines: string[]; nextOffset: number }>` — reads bytes from `fromOffset` to EOF, cuts at the last `\n` (same safe-cut rule as `spool-io.ts:122-144`'s `snapshotPendingHookRecords`) so a partially-written trailing line is never returned; returns `{ lines: [], nextOffset: fromOffset }` when the file is missing or has no new complete line. + +**Test-first: yes — a file with two complete lines plus a trailing unterminated partial line returns only the two complete lines and `nextOffset` points exactly after the second line's newline; a second call starting from that offset returns only lines appended afterwards.** + +- [ ] Write failing tests (fixture file written incrementally across two reads). +- [ ] Implement. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 8: `agent.usage.request` extraction and merge + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/usage-request.ts` +- Test fixture: `src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl` (new — modeled on the confirmed shape in `spec.md`'s "Transcript field shape" section: `message.id`, `message.usage.{input_tokens,output_tokens,cache_read_input_tokens,cache_creation_input_tokens,service_tier,speed,inference_geo}`, `message.usage.cache_creation.{ephemeral_1h_input_tokens,ephemeral_5m_input_tokens}`, `message.usage.server_tool_use.{web_search_requests,web_fetch_requests}`, `message.stop_reason`, top-level `gitBranch`). +- Test: `.../__tests__/usage-request.test.ts`. + +**Interfaces:** +- Produces: + - `parseUsageLine(line: string, scopeKind: 'main'|'skill'|'agent', scopeName: string, agentId: string): OpenUsageRequest | null` — returns `null` for lines with no `message.usage`; `request_id` = `message.id` (not `requestId` — see spec's Open risks); `model`/`model_raw` via `parseBackendModelName()`/`parseRoutingHeaders()` from `@/utils/routing-headers.mjs` and `@/utils/bedrock-pricing.mjs` (same resolution the statusline and `usage-readers.ts:188` already use). + - `mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): OpenUsageRequest` — every numeric field takes `Math.max`; non-numeric fields (`stopReason`, `isApiError`, `gitBranch`, etc.) take `b`'s value when non-empty, else `a`'s. + - `buildUsageRequestEvent(sessionId: string, req: OpenUsageRequest): Record` — `{ type: 'agent.usage.request', session_id: sessionId, request_id: req.requestId, model_raw, model, speed, inference_geo, service_tier, input_tokens, cache_creation_5m_tokens, cache_creation_1h_tokens, cache_read_tokens, output_tokens, web_search_requests, web_fetch_requests, scope_kind, scope_name, agent_id, stop_reason, is_api_error, git_branch, timestamp }` (`event_id`/`schema_version` stamped later by Task 1's daemon-side code, since this record carries an explicit `type`). + +**Test-first: yes — on a fixture with two JSONL lines for the same `message.id`+model where the second has a higher `output_tokens` and a `stop_reason` the first lacks, `mergeUsageRequest` of the two parsed records keeps the max `output_tokens` and the non-empty `stop_reason`; a line with no `usage` block parses to `null`.** + +- [ ] Write failing tests against the fixture. +- [ ] Implement `usage-request.ts`. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 9: `agent.subagent.usage` builder + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts` +- Test: `.../__tests__/subagent-usage.test.ts` with 2-3 fixture subagent transcript+`.meta.json` pairs under a temp `/subagents/` directory. + +**Interfaces:** +- Produces: + - `interface SubagentFile { agentId: string; filePath: string; toolUseId?: string; agentType?: string; spawnDepth?: number; }` and `findSubagentFiles(mainTranscriptPath: string): Promise` — own implementation (the existing `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts:384` is `private` and not exported, so this is a fresh, smaller implementation reading `//subagents/agent-*.jsonl` + sibling `.meta.json`, same path convention). `description`/`workflow_run`/`worktree` have no identified source in the sidecar (per spec's Open risks) — emit them as empty string, never fabricated. + - `buildSubagentUsageEvent(sessionId: string, file: SubagentFile, usageRequests: OpenUsageRequest[], toolCalls: Record, toolErrors: Record, skillsInvoked: Record, startedAt: string, durationMs: number): Record` — `type: 'agent.subagent.usage'`, tokens/`api_calls` aggregated by summing `usageRequests` (reusing Task 8's `OpenUsageRequest` fields), `spawn_depth` defaults to `0` when the sidecar omits it (top-level subagents, per spec's Open risks — not treated as an error). + +**Test-first: yes — a session fixture with three subagent transcript files produces three `agent.subagent.usage` events whose summed token fields equal the sum of the `agent.usage.request` records this same fixture yields with `scope_kind: 'agent'` (the external data-model doc's §8 acceptance scenario).** + +- [ ] Write the failing cross-check test plus a `spawn_depth`-missing-sidecar case. +- [ ] Implement `subagent-usage.ts`. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 10: `agent.session.summary` builder + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/session-summary.ts` +- Test: `.../__tests__/session-summary.test.ts`. + +**Interfaces:** +- Produces: + - `interface SessionSummaryAccumulator { models: Record; toolCalls: Record; linesAdded: number; linesRemoved: number; filesChanged: Set; filesWritten: Set; compactionCount: number; }` and `updateBranchCounts(counts: Record, branch: string): void` (bumps `counts[branch]`, mutates the `TranscriptParseState.branchCounts` from Task 6). + - `primaryModel(models: Record): string`, `branchDominant(counts: Record): string` — both return the highest-count key, `''` when empty. + - `buildSessionSummaryEvent(sessionId: string, phase: 'incremental' | 'final', acc: SessionSummaryAccumulator, named: NamedInvocationCounts, branchCounts: Record, startedAt: string, endedAt: string | undefined): Record` — `named` comes from `extractNamedInvocations()` (`src/agents/plugins/claude/session/claude-named-invocations.ts`, directly imported) for `skills_used`/`commands_in_order`/`primary_command`; `type: 'agent.session.summary'`. + +**Test-first: yes — a `branchCounts` map built from a mid-session branch switch (`{main: 3, feature: 7}`) resolves `branch_dominant: 'feature'` (the external data-model doc's §8 branch-switch scenario); `buildSessionSummaryEvent` with `phase: 'incremental'` omits `endedAt` and with `phase: 'final'` includes it.** + +- [ ] Write failing tests for `branchDominant`, `primaryModel`, and both phases. +- [ ] Implement `session-summary.ts`. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 11: Orchestrate main-transcript triggers (`Stop`, `PreCompact`, `SessionEnd`) + +**Files:** +- Create: `src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts` +- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts:24-35` — in `evaluate()`, after the existing `UserPromptSubmit` branch, when `event.hookEventName` is `Stop`, `PreCompact`, or `SessionEnd`, call the orchestrator (fire-and-forget, matching `forwardToSpool`'s pattern) before returning the normal `forward` decision. +- Test: `src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts`. + +**Interfaces:** +- Produces: `runMainTranscriptParse(sessionId: string, transcriptPath: string, trigger: 'Stop' | 'PreCompact' | 'SessionEnd'): Promise` — loads state (Task 6), reads new lines (Task 7), derives/merges `agent.usage.request` records (Task 8) keyed into `state.openRequests`, updates `state.branchCounts`/summary accumulator, emits one `forwardOtlpEventToSpool(JSON.stringify(event), CLAUDE_CODE_OTLP_AGENT_NAME)` call per completed `agent.usage.request` plus one `agent.session.summary` (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`; `PreCompact` emits only usage requests, never a summary — matches spec), saves state, and swallows every error internally (never throws into `processOtlpEvent`). + +**Test-first: yes — calling `runMainTranscriptParse` twice with the same transcript (simulating a re-parse after a crash before state was saved) forwards `agent.usage.request` events whose `event_id`-determining fields (`request_id`, `model`) are identical both times — the idempotent-reparse scenario from the external data-model doc's §8.** + +- [ ] Write the failing idempotent-reparse test plus a basic Stop → one summary + N usage-request forwards test (mock `forwardOtlpEventToSpool`). +- [ ] Implement `orchestrator.ts` and the `claude-code-otlp.plugin.ts` wiring. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. +- [ ] Commit. + +--- + +### Task 12: Orchestrate subagent triggers (`SubagentStop`, `SessionEnd` backstop) + +**Files:** +- Modify: `orchestrator.ts` (Task 11) — add the subagent path. +- Modify: `claude-code-otlp.plugin.ts` — on `SubagentStop`, call the new function with that record's own subagent transcript path (`hookEvent['agent_transcript_path']`, read loosely off the parsed JSON since it is outside `BaseClaudeCodeHookEvent`'s modeled fields) instead of the main one; `SessionEnd` additionally re-runs it for **every** subagent file `findSubagentFiles()` (Task 9) discovers, not just ones already seen — the crashed/missed-hook backstop. +- Test: extend `__tests__/orchestrator.test.ts`. + +**Interfaces:** +- Produces: `runSubagentTranscriptParse(sessionId: string, mainTranscriptPath: string, subagentFile: SubagentFile): Promise` — reads new lines from `state.subagentOffsets[subagentFile.agentId]` (Task 7), derives `agent.usage.request` with `scope_kind: 'agent'` (Task 8), builds and forwards one `agent.subagent.usage` (Task 9), saves the updated offset back into the shared `TranscriptParseState`. + +**Test-first: yes — a `SessionEnd` on a session with three subagent files, only one of which already has a `SubagentStop`-advanced offset, still forwards three `agent.subagent.usage` events (the backstop), and never forwards a fourth for a subagent whose offset shows nothing new since the last run.** + +- [ ] Write the failing backstop test (three fixture subagents, one pre-advanced offset) and a no-new-bytes-means-no-resend test. +- [ ] Implement. +- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. +- [ ] Commit. + +--- + +## Self-Review Notes + +- **Spec coverage:** Common fields (Tasks 1-3), story resolution (4-5), transcript parse-state (6-7), the three new events (8-10), and the four trigger wirings (11-12) each map to a numbered spec section. `event_id`'s "no `generateUUID()`" and the privacy "never raw prompt text" constraints are enforced structurally (pure-function `event_id`, resolved-only story fields) rather than left to each task's judgment. +- **Non-goals respected:** no task touches `agent.session.env`, `agent.skill.dispatch`, `agent.git.snapshot`, or any existing event's own content fields — only the common-field wrapper in `mapHookRecords()`. +- **Type consistency:** `OpenUsageRequest` (Task 6) is the one shape Tasks 8, 9, and 11/12 all import and merge/aggregate — no parallel redefinition. `TranscriptParseState` (Task 6) is the single state object Tasks 7, 11, and 12 all read/mutate/save. diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md new file mode 100644 index 000000000..d2289eabe --- /dev/null +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md @@ -0,0 +1,137 @@ +# EPMCDME-15301 sub-stage 1.1 — expanded fields and new usage events (OTLP analytics pipeline) + +## Goal + +Extend the existing `claude-code-otlp` pipeline (`forwarder.ts` / `otlp-spool` / `claude-code-otlp.plugin.ts`) with: + +- Common fields on every emitted `agent.*` event. +- Incremental transcript re-parsing on every trigger, backed by persisted per-session parse state. +- Three new events: `agent.usage.request`, `agent.subagent.usage`, `agent.session.summary`. +- Story/ticket resolution. +- An extended identity chain. + +This is additive to the existing 13-event `agent.*` taxonomy already produced by `HOOK_EVENT_TYPE_MAP`/`mapHookRecords()` (`forwarder.ts:16-29,221-269`); no server (`codemie`) changes. + +## Common fields + +Every field below is sent on **every** emitted event. Split by *where* each is computed: + +- **Plugin-owned, resolved live in the forwarder — no hook-side capture, nothing new added to the spool.** A new method on the `OtlpAgentAdapter` interface (`src/agents/core/types.ts:734-739`, alongside its existing sole method `processOtlpEvent`): `prepareAnalyticsFields(hookEvent: Record): Promise`, implemented by `ClaudeCodeOtlpPlugin`. Called from the forwarder, resolving the right plugin per record off the spooled `OtlpHookSpoolData.agentName`: + ```ts + const analyticsAgent = AgentRegistry.getAnalyticsAgent(spoolData.agentName); + const commonFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; + ``` + (mirrors the existing `AgentRegistry.getAnalyticsAgent(otlpHookSpoolData.agentName)` lookup already used in `otlp.plugin.ts`'s `handleHooks()` to validate `agentName` before spooling.) `ClaudeCodeOtlpPlugin.processOtlpEvent()` (the hook CLI process) is unchanged — nothing new is captured there and nothing new is added to the spooled record. Everything below is computed autocalculated on the fly, inside this method, when the forwarder calls it: + - `platform` — the plugin's own `platform` class field (`'claude-code' as const`). + - `client_version` — `const { stdout: claudeVersion } = await execPromise('claude --version');`, run live by the method itself. + - `entrypoint` — `process.env.CLAUDE_CODE_ENTRYPOINT`, read live by the method itself (empty string if unset, not blocking). + - `agent_id`/`agent_type` — read off this record's own `hookEvent` (already spooled as `raw` today), because interpreting which fields identify an agent/subagent is specific to that agent's hook-payload shape, not something a generic, agent-agnostic forwarder should own. Each OTLP plugin (today: `claude-code-otlp`; future: `cursor`/`codex`/`copilot`) implements its own interpretation. + +- **Daemon-side, client-agnostic** — computed once per forward tick in `buildForwardContext()` and stamped onto every record in `mapHookRecords()` (`forwarder.ts`), the same pattern already used today for `user_email`/`developer_name`/`git_branch`/`repo_remote`/`codemie_project_name`: + - `schema_version=2` + - `event_id` + - `codemie_cli_version` — the running `@codemieai/code` package's own `version` field in `package.json` (currently `0.15.4`). + - `story_id`/`story_source` + - `developer_name`/`identity_source` + +## `event_id` + +One mechanism for every event, old and new: a plain string composed from fields already on the record, computed daemon-side, no `generateUUID()`, no stored/persisted id anywhere. + +| Event | `event_id` | +|---|---| +| Existing 13 hook-mapped events | `${session_id}:${type}:` — computed in `mapHookRecords()` (`forwarder.ts`) from fields already available there; nothing new added to `OtlpHookSpoolData`. | +| `agent.usage.request` | `${session_id}:agent.usage.request:${request_id}:${model}` — `model` included to match the `(request_id, model)` uniqueness key stated in "New events" below. | +| `agent.subagent.usage` | `${session_id}:agent.subagent.usage:${tool_use_id}` | +| `agent.session.summary` | `${session_id}:agent.session.summary:${phase}` (`phase` is `incremental` or `final`; every `Stop`-triggered re-emission this session reuses the same `incremental` id — intentional, a running summary is one record that gets refreshed) | + +## Transcript re-parsing + +Needed to produce the three new events (`agent.usage.request`, `agent.subagent.usage`, `agent.session.summary`), none of which exist in the hook payload itself — unlike the existing 13 events, which just map that payload, these have to be derived by reading and parsing the transcript file. + +Per the data-model doc (§5.1), parsing is **incremental and backed by persisted per-session state** at `~/.codemie/analytics/state/.json`: + +- **State holds**: the byte offset already consumed in the main transcript, and separately for each subagent transcript; the `openRequests` max-merge-in-progress map; `activeSkill`; `branchCounts` (feeds `branch_dominant`). +- **Triggers**: `Stop`, `SubagentStop`, `PreCompact`, `SessionEnd` — each reads only the bytes appended since its stored offset, updates the running tallies, derives any newly-complete records, writes the updated state back to disk, then sends. +- **Recovery path**: if the state file is missing (first run for a session) or fails to parse (corruption), fall back to a full parse from byte `0` and treat every record as newly derived. Each record's `event_id` is a pure function of its natural key (see "`event_id`" above), so a record re-derived this way carries the exact same id as before — the backend sees an update, not a duplicate. +- Stays `async`, swallows all errors internally (matching `hook.ts`'s existing try/catch + `process.exitCode` convention), and always exits 0. + +## Transcript field shape (verified against real transcripts) + +Verified against real Claude Code session transcripts (current client, multiple projects/sessions) and cross-checked against the existing production parser `src/cli/commands/analytics/cost/usage-readers.ts` (`ClaudeRawMessage`/`extractClaudeUsageRecords`), which already parses most of this same shape for the unrelated cost-reporting pipeline — prior art to adapt, not a blank slate: + +- **No top-level `requestId` field appears on any real transcript line observed.** It is not declared in this repo's own `claude-message-types.ts` `ClaudeMessage` type either. `usage-readers.ts` already treats it as optional and falls back to `message.id` alone when absent (`const key = id || reqId ? ... : null`) — the existing production code already assumes `requestId` may not be there. +- **Use `message.id`** (the API-level message id, stable across streaming chunks for one response) **as `request_id`.** It is present on every usage-bearing line sampled and is the only reliable per-response identity. This also means the `agent.usage.request` `event_id` keys on `message.id`, not a field that in practice is always empty — had it keyed on the literal absent `requestId`, every request for a given model within a session would collide onto the same `event_id` and overwrite rather than accumulate. +- **`message.stop_reason`** sits directly on `message`, as a sibling of `usage` — not nested inside it. Observed value: `"tool_use"`. +- **`message.usage`** fields, confirmed by direct inspection: + - Flat: `input_tokens`, `output_tokens`, `cache_read_input_tokens`, `cache_creation_input_tokens` (matches `usage-readers.ts`). + - `service_tier` (observed `"standard"`) and `speed` (observed `"standard"`) are flat fields on `usage`, not on `message`. + - `inference_geo` is a flat field on `usage`, observed as `""` (present but empty, not absent). + - `cache_creation.ephemeral_1h_input_tokens` / `cache_creation.ephemeral_5m_input_tokens` — nested one level, matches `usage-readers.ts`. + - `server_tool_use.web_search_requests` / `server_tool_use.web_fetch_requests` — nested one level under `server_tool_use`, not flat fields. + - `iterations` (observed `[]`) — present on every sampled line but not named anywhere in this ticket's scope; shape and purpose unconfirmed. Carry through as opaque/unused rather than guessing a meaning. +- Top-level fields confirmed present on every line: `sessionId`, `gitBranch`, `cwd`, `timestamp`, `version`, `entrypoint`, `uuid`, `parentUuid`, `isSidechain`, `userType` — matches `claude-message-types.ts`'s `ClaudeMessage`. + +## New events + +### `agent.usage.request` + +- Fires on `Stop` (re-parse of the main transcript), `SubagentStop` (re-parse of that subagent's own transcript, `scope_kind=agent`), `PreCompact` (re-parse of the main transcript, so in-progress request usage is captured before compaction can drop the turns it came from), and `SessionEnd` (final re-parse of the main transcript **and every subagent transcript**). +- One per unique `(request_id, model)` across the session transcript and all subagent transcripts. +- Take the max per numeric field across duplicate records (per `openRequests` merge). +- `scope_kind`/`scope_name` = `main`/`skill`/`agent` depending on which transcript (main vs. a named skill context vs. a subagent transcript) the record came from. + +Fields: `request_id`, `timestamp`, `model_raw`, `model`, `speed`, `inference_geo`, `service_tier`, `input_tokens`, `cache_creation_5m_tokens`, `cache_creation_1h_tokens`, `cache_read_tokens`, `output_tokens`, `web_search_requests`, `web_fetch_requests`, `scope_kind`, `scope_name`, `agent_id`, `stop_reason`, `is_api_error`, `git_branch`. All sourced from the transcript shape confirmed above (`message.id`, `message.usage.*`, `message.stop_reason`, `server_tool_use.*`, `gitBranch`) or, for `model`/`model_raw`, from the same resolution chain the statusline already uses (`parseRoutingHeaders()`/`parseBackendModelName()`); `agent_id` comes from the new plugin method above. `is_api_error`'s presence pattern on a real error is unverified — see Open risks. + +### `agent.subagent.usage` + +- Fires on `SubagentStop`, keyed by `tool_use_id`, and again on `SessionEnd` — re-emitted for **every** subagent transcript found, not just ones whose `SubagentStop` already fired (backstop for a crashed/missed subagent hook). Never fires on `Stop`/`PreCompact` (main-transcript-only parses). +- Aggregates that subagent's own transcript: tokens by model/cache tier, tool-call/error counts, skills invoked. + +Fields: `agent_type`, `description`, `tool_use_id` (all from the subagent's own `.meta.json` sidecar, already read by `claude.session.ts`'s `findSubagentFiles()`), `spawn_depth` (same sidecar), `workflow_run`, `worktree`, `started_at` (that transcript's first line `timestamp`), `duration_ms` (derived from first/last line `timestamp`), tokens by model/cache tier and `api_calls` (aggregated from that subagent transcript's own usage records, same fields as `agent.usage.request` above), tool-call counts (`tool_use` block count), error counts (`is_api_error`/tool-error occurrences), skills invoked (`tool_use` blocks named `Skill`, via `extractNamedInvocations()`). `spawn_depth`'s default for top-level subagents, and `workflow_run`/`worktree`'s missing source, are both tracked in Open risks. + +### `agent.session.summary` + +- Fires on `Stop` (`phase=incremental`) and `SessionEnd` (`phase=final`, all fields recomputed from full state) — the only two of the three triggers that are session-level stop points; `SubagentStop`/`PreCompact` never emit it. +- `primary_model` = model with the most `agent.usage.request` records this session. +- `primary_command` = most-frequently-invoked slash command. +- `branch_dominant` = key with the highest count in `branchCounts`. + +Fields: models used, `primary_model` (aggregated from this session's `agent.usage.request` records), tool-call counts/errors by tool (existing `PostToolUse`/`PostToolUseFailure` hook payloads, already forwarded today), skills used (`extractNamedInvocations()`'s `skillInvocations`), slash-commands in order and `primary_command` (`extractNamedInvocations()`'s `commandInvocations`), lines added/removed and files changed/written (existing `Edit`/`Write` tool payloads, already forwarded today), compaction counts (`PreCompact` trigger count this session), branch-switch counts and `branch_dominant` (derived from `git_branch` per record via `branchCounts`), `client_version`/`codemie_cli_version` (common fields, see above), session start/end timestamps (`SessionStart`/`SessionEnd` hook timestamps), `api_calls` (count of `agent.usage.request` records this session), session `title`. `title`'s missing source is tracked in Open risks. + +## Story/ticket resolution + +Resolved fresh every time, by event type: + +- **Non-prompt events** (everything except `agent.prompt.submit`): priority chain is explicit → branch. Both are resolvable without any per-record text, so both are computed once per forward tick in `buildForwardContext()` and stamped onto the whole batch in `mapHookRecords()` (`forwarder.ts`) — same per-tick cache already used for `git_branch`/`repo_remote`. +- **`agent.prompt.submit` events**: priority chain is explicit → marker → branch → mention. Marker/mention require that record's own prompt text, so they're evaluated per-record against the *full, untruncated* prompt — either hook-side at `UserPromptSubmit` before `prompt_body` is truncated to `MAX_PROMPT_CHARS=200`, or per-record inside `mapHookRecords()` using the record's own `hookEvent['prompt']` before truncation. Only the resolved `story_id`/`story_source` is threaded through to the spool/forwarder, never the raw text (ticket privacy constraint). + +**Explicit** reads from two sources, first match wins: +1. `SDLC_ANALYTICS_STORY_ID` env var. +2. `/.claude/analytics.local.json` — shape `{ "storyId": string }`. `cwd` is the hook payload's existing `cwd` field (`claude-code-otlp.types.ts:19,53`, already read via `process.cwd()` at `claude-code-otlp.plugin.ts:38`), which is the project root for Claude Code hook invocations. Gitignored — add `.claude/analytics.local.json` to `.gitignore`. + +Reading is read-only here: **nothing in this stage writes that file.** A command that writes it (e.g. `/codemie:set-story`) is tracked separately and out of scope for this stage. + +**Branch** reuses `resolveGitInfo()`/`detectGitBranch` (`forwarder.ts`, `src/utils/processes.ts`). + +## Identity resolution + +- Extend `resolveUserEmail()`'s JWT-only chain (`forwarder.ts:67-81`) to `jwt → git → codemie_cli → claude_account → os`, first available wins: + - `git` reads `git config user.name`/`user.email`. + - `codemie_cli` reads the existing CLI profile config. + - `os` reads `os.userInfo().username`. + - `claude_account` has no precedent in this repo — implement it as a best-effort lookup that yields nothing if unavailable, falling through to `os` (documented limitation, not a blocker). + +## Non-goals + +- `agent.session.env`, `agent.skill.dispatch`, `agent.git.snapshot` — sub-stage 1.2. +- Any modification to existing event *content* (`agent.session.start`, `agent.prompt.submit`, `agent.tool.start`/`end`, `agent.subagent.stop`, `agent.session.compact`, `agent.session.stop`/`end`) beyond adding the common fields above. + +## Open risks + +- **Transcript field confidence gaps** (all affect the three new events; sourcing details are in "Transcript field shape" and "New events" above): + - `request_id` must be sourced from `message.id` — no real transcript observed carries a top-level `requestId`. Resolved, but flagged so it isn't silently reintroduced from `usage-readers.ts`'s optional-`requestId` interface during implementation. + - `is_api_error` (used in `agent.usage.request` and in `agent.subagent.usage`'s error counts) is read by production code elsewhere in this repo, but no sampled transcript contained an actual API error, so whether it's always present (false/absent otherwise) or appears only on the erroring line is unverified. + - `spawn_depth` (`agent.subagent.usage`) is present in the `.meta.json` sidecar only for nested subagents (depth ≥ 2); top-level subagents omit it, so implementation needs an explicit default rather than treating absence as an error. + - `workflow_run`/`worktree` (`agent.subagent.usage`) have no identified source in this codebase's transcript handling or the `.meta.json` sidecar. The external data-model doc names a `workflows//` sidecar directory as the source, but a direct check across every local session directory found no such directory in any sampled session — the gap stands; worth revisiting with the data-model doc's owner. + - `title` (`agent.session.summary`) has no identified source in a real transcript, top-level or nested, and the data-model doc doesn't name one either — unresolved on both sides. diff --git a/src/agents/core/types.ts b/src/agents/core/types.ts index a16a86cd5..fa171742f 100644 --- a/src/agents/core/types.ts +++ b/src/agents/core/types.ts @@ -731,7 +731,7 @@ export enum AgentAdapterType { OTLP, } -export interface OtlpAgentAdapter { +export interface OtlpAgentAdapter { readonly name: string; readonly type: AgentAdapterType.OTLP; @@ -744,6 +744,11 @@ export interface OtlpAgentAdapter { * spool. */ processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise; + + /** + * Resolve agent-owned common fields for a single hook event + */ + prepareAnalyticsFields(hookEvent: Record): Promise>; } export interface OtlpAdapterDeps { diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index 17c9d37df..e8a3879eb 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -1,4 +1,4 @@ -import { describe, it, expect, vi, beforeEach } from 'vitest'; +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; vi.mock('../claude-code-otlp.allowlist.js', () => ({ readAllowlistState: vi.fn(async () => ({ kind: 'valid', paths: ['/proj'] })), @@ -11,6 +11,11 @@ vi.mock('@/utils/logger.js', () => ({ logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, })); +const execMock = vi.fn(); +vi.mock('@/utils/exec.js', () => ({ + exec: execMock, +})); + import { ClaudeCodeOtlpPlugin } from '../claude-code-otlp.plugin.js'; import { isProjectTracked } from '../claude-code-otlp.allowlist.js'; import { forwardOtlpEventToSpool } from '../../utils.js'; @@ -47,3 +52,74 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { ); }); }); + +describe('ClaudeCodeOtlpPlugin.prepareAnalyticsFields', () => { + const originalEntrypoint = process.env.CLAUDE_CODE_ENTRYPOINT; + + beforeEach(() => { + execMock.mockReset(); + execMock.mockResolvedValue({ code: 0, stdout: '2.1.23 (Claude Code)', stderr: '', signal: null }); + }); + + afterEach(() => { + vi.restoreAllMocks(); + if (originalEntrypoint === undefined) { + delete process.env.CLAUDE_CODE_ENTRYPOINT; + } else { + process.env.CLAUDE_CODE_ENTRYPOINT = originalEntrypoint; + } + }); + + it('returns platform, entrypoint, and client_version', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + process.env.CLAUDE_CODE_ENTRYPOINT = 'cli'; + + const plugin = new ClaudeCodeOtlpPlugin(); + const fields = await plugin.prepareAnalyticsFields({}); + + expect(fields.platform).toBe('claude-code'); + expect(fields.entrypoint).toBe('cli'); + expect(fields.client_version).toBe('2.1.23'); + }); + + it('includes agent_id/agent_type when present on the hook event', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + + const plugin = new ClaudeCodeOtlpPlugin(); + const fields = await plugin.prepareAnalyticsFields({ agent_id: 'sub-1', agent_type: 'explore' }); + + expect(fields.agent_id).toBe('sub-1'); + expect(fields.agent_type).toBe('explore'); + }); + + it('omits agent_id/agent_type when absent from the hook event', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + + const plugin = new ClaudeCodeOtlpPlugin(); + const fields = await plugin.prepareAnalyticsFields({}); + + expect(fields).not.toHaveProperty('agent_id'); + expect(fields).not.toHaveProperty('agent_type'); + }); + + it('spawns `claude --version` only once across two prepareAnalyticsFields calls', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + + const plugin = new ClaudeCodeOtlpPlugin(); + await plugin.prepareAnalyticsFields({}); + await plugin.prepareAnalyticsFields({ agent_id: 'sub-2' }); + + expect(execMock).toHaveBeenCalledTimes(1); + expect(execMock).toHaveBeenCalledWith('claude', ['--version']); + }); + + it('falls back to an empty client_version when `claude --version` throws', async () => { + execMock.mockRejectedValue(new Error('ENOENT')); + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + + const plugin = new ClaudeCodeOtlpPlugin(); + const fields = await plugin.prepareAnalyticsFields({}); + + expect(fields.client_version).toBe(''); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index f7241e0e9..fe40593f2 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -1,4 +1,5 @@ +import { basename } from 'node:path'; import { AuthGateResult, ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; import { logger } from '@/utils/logger.js'; import { ConfigLoader } from '@/utils/config.js'; @@ -7,10 +8,16 @@ import { CLAUDE_CODE_OTLP_AGENT_NAME } from './claude-code-otlp.constants.js'; import { ForwardDecision, toBaseClaudeCodeHookEvent } from './claude-code-otlp.types.js'; import { forwardOtlpEventToSpool } from '../utils.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; +import { runMainTranscriptParse, runSubagentTranscriptParse, type SubagentFile } from './transcript/orchestrator.js'; +import { findSubagentFiles } from './transcript/subagent-usage.js'; export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; public readonly type = AgentAdapterType.OTLP; + public readonly platform = 'claude-code' as const; + + /** Cached across calls so a high-frequency hook (e.g. PostToolUse) doesn't spawn a subprocess per call. */ + private cachedClientVersion: string | undefined; public async processOtlpEvent(rawEvent: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { const event = toBaseClaudeCodeHookEvent(JSON.parse(rawEvent)); @@ -44,6 +51,49 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { return await this.onUserPromptSubmit(rawEvent); } + const rawParsed = JSON.parse(rawEvent); + const event = toBaseClaudeCodeHookEvent(rawParsed); + + if ( + event.hookEventName === 'Stop' || + event.hookEventName === 'PreCompact' || + event.hookEventName === 'SessionEnd' + ) { + void runMainTranscriptParse( + event.sessionId, + event.transcriptPath, + event.hookEventName as 'Stop' | 'PreCompact' | 'SessionEnd' + ); + } + + if (event.hookEventName === 'SessionEnd') { + void (async () => { + const subagentFiles = await findSubagentFiles(event.transcriptPath); + for (const file of subagentFiles) { + await runSubagentTranscriptParse(event.sessionId, event.transcriptPath, file); + } + })(); + } + + if (event.hookEventName === 'SubagentStop') { + const rawRecord = rawParsed as Record; + const agentTranscriptPath = + typeof rawRecord['agent_transcript_path'] === 'string' ? rawRecord['agent_transcript_path'] : ''; + if (agentTranscriptPath) { + const agentId = + typeof rawRecord['agent_id'] === 'string' + ? rawRecord['agent_id'] + : basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''); + const subagentFile: SubagentFile = { + agentId, + filePath: agentTranscriptPath, + toolUseId: typeof rawRecord['tool_use_id'] === 'string' ? rawRecord['tool_use_id'] : undefined, + agentType: typeof rawRecord['agent_type'] === 'string' ? rawRecord['agent_type'] : undefined, + }; + void runSubagentTranscriptParse(event.sessionId, event.transcriptPath, subagentFile); + } + } + return { decision: 'forward', payload: rawEvent, @@ -65,6 +115,43 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { forwardOtlpEventToSpool(rawEvent, CLAUDE_CODE_OTLP_AGENT_NAME); } + public async prepareAnalyticsFields( + hookEvent: Record + ): Promise> { + const fields: Record = { + platform: this.platform, + entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', + client_version: await this.getClientVersion(), + }; + + if (typeof hookEvent['agent_id'] === 'string') { + fields.agent_id = hookEvent['agent_id']; + } + if (typeof hookEvent['agent_type'] === 'string') { + fields.agent_type = hookEvent['agent_type']; + } + + return fields; + } + + private async getClientVersion(): Promise { + if (this.cachedClientVersion !== undefined) { + return this.cachedClientVersion; + } + + try { + const { exec } = await import('@/utils/exec.js'); + const result = await exec('claude', ['--version']); + const trimmed = result.stdout.trim(); + const versionMatch = trimmed.match(/^(\d+\.\d+\.\d+)/); + this.cachedClientVersion = versionMatch ? versionMatch[1] : trimmed; + } catch { + this.cachedClientVersion = ''; + } + + return this.cachedClientVersion; + } + private async onUserPromptSubmit(rawEvent: string): Promise { const authResult = await this.ensureProxyAuth(); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl b/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl new file mode 100644 index 000000000..9b67d70f5 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage.jsonl @@ -0,0 +1,4 @@ +{"sessionId":"session-usage-1","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-01T00:00:00.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-no-usage","parentUuid":null,"isSidechain":false,"userType":"external","message":{"role":"user","content":"hello"}} +{"sessionId":"session-usage-1","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-01T00:00:01.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-pair-1","parentUuid":"uuid-no-usage","isSidechain":false,"userType":"external","message":{"id":"msg_pair_1","role":"assistant","model":"claude-sonnet-4-5-20250929","usage":{"input_tokens":100,"output_tokens":50,"cache_read_input_tokens":5,"cache_creation_input_tokens":0,"service_tier":"standard","speed":"standard","inference_geo":"","cache_creation":{"ephemeral_1h_input_tokens":0,"ephemeral_5m_input_tokens":0},"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"iterations":[]}}} +{"sessionId":"session-usage-1","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-01T00:00:02.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-pair-2","parentUuid":"uuid-pair-1","isSidechain":false,"userType":"external","message":{"id":"msg_pair_1","role":"assistant","model":"claude-sonnet-4-5-20250929","stop_reason":"tool_use","usage":{"input_tokens":100,"output_tokens":120,"cache_read_input_tokens":5,"cache_creation_input_tokens":0,"service_tier":"standard","speed":"standard","inference_geo":"","cache_creation":{"ephemeral_1h_input_tokens":0,"ephemeral_5m_input_tokens":0},"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"iterations":[]}}} +{"sessionId":"session-usage-1","gitBranch":"feature/epmcdme-15301","cwd":"/repo","timestamp":"2026-10-01T00:00:03.000Z","version":"1.2.3","entrypoint":"cli","uuid":"uuid-normal","parentUuid":"uuid-pair-2","isSidechain":false,"userType":"external","message":{"id":"msg_normal_1","role":"assistant","model":"claude-sonnet-4-5-20250929","stop_reason":"end_turn","x-litellm-model-name":"bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0","usage":{"input_tokens":300,"output_tokens":75,"cache_read_input_tokens":40,"cache_creation_input_tokens":15,"service_tier":"priority","speed":"fast","inference_geo":"us","cache_creation":{"ephemeral_1h_input_tokens":10,"ephemeral_5m_input_tokens":5},"server_tool_use":{"web_search_requests":2,"web_fetch_requests":1},"iterations":[]}}} diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts new file mode 100644 index 000000000..99c3e847b --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -0,0 +1,398 @@ +/** + * Tests for `runMainTranscriptParse` — the `Stop`/`PreCompact`/`SessionEnd` main-transcript + * orchestrator. + * + * `forwardOtlpEventToSpool` (`@/agents/plugins/utils.js`) is mocked so no real network/daemon + * call happens; `CODEMIE_HOME` points at a fresh temp directory per test so `loadParseState`/ + * `saveParseState` never touch the real `~/.codemie`. + */ + +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; + +const forwardMock = vi.fn().mockResolvedValue(undefined); + +vi.mock('@/agents/plugins/utils.js', () => ({ + forwardOtlpEventToSpool: forwardMock, +})); + +let codemieHome: string; +let transcriptDir: string; + +function usageLine(opts: { + uuid: string; + messageId: string; + outputTokens: number; + gitBranch?: string; + stopReason?: string; + timestamp?: string; +}): string { + return JSON.stringify({ + gitBranch: opts.gitBranch ?? 'main', + cwd: '/repo', + timestamp: opts.timestamp ?? '2026-10-01T00:00:00.000Z', + uuid: opts.uuid, + message: { + id: opts.messageId, + role: 'assistant', + model: 'claude-sonnet-4-5-20250929', + stop_reason: opts.stopReason ?? '', + usage: { + input_tokens: 100, + output_tokens: opts.outputTokens, + cache_read_input_tokens: 5, + cache_creation_input_tokens: 0, + service_tier: 'standard', + speed: 'standard', + inference_geo: '', + cache_creation: { ephemeral_1h_input_tokens: 0, ephemeral_5m_input_tokens: 0 }, + server_tool_use: { web_search_requests: 0, web_fetch_requests: 0 }, + }, + }, + }); +} + +function noUsageLine(uuid: string): string { + return JSON.stringify({ + gitBranch: 'main', + cwd: '/repo', + timestamp: '2026-10-01T00:00:00.000Z', + uuid, + message: { role: 'user', content: 'hello' }, + }); +} + +beforeEach(() => { + codemieHome = mkdtempSync(join(tmpdir(), 'codemie-home-')); + process.env.CODEMIE_HOME = codemieHome; + transcriptDir = mkdtempSync(join(tmpdir(), 'codemie-transcript-')); + forwardMock.mockClear(); +}); + +afterEach(() => { + delete process.env.CODEMIE_HOME; + rmSync(codemieHome, { recursive: true, force: true }); + rmSync(transcriptDir, { recursive: true, force: true }); +}); + +function writeTranscript(fileName: string, lines: string[]): string { + const filePath = join(transcriptDir, fileName); + writeFileSync(filePath, lines.map((l) => l + '\n').join(''), 'utf-8'); + return filePath; +} + +type ForwardedEvent = Record; + +function forwardedEvents(): ForwardedEvent[] { + return forwardMock.mock.calls.map(([raw]) => JSON.parse(raw as string) as ForwardedEvent); +} + +/** + * Write a subagent fixture transcript (plus its sidecar `.meta.json`) under + * `//subagents/agent-.jsonl`, matching + * `findSubagentFiles()`'s own discovery convention (Task 9). Returns the `SubagentFile` shape + * `findSubagentFiles()` would discover for it. + */ +function writeSubagentFixture( + sessionId: string, + agentId: string, + lines: string[], + meta: Record = {} +): { agentId: string; filePath: string } { + const subagentsDir = join(transcriptDir, sessionId, 'subagents'); + mkdirSync(subagentsDir, { recursive: true }); + const filePath = join(subagentsDir, `agent-${agentId}.jsonl`); + writeFileSync(filePath, lines.map((l) => l + '\n').join(''), 'utf-8'); + writeFileSync(join(subagentsDir, `agent-${agentId}.meta.json`), JSON.stringify(meta), 'utf-8'); + return { agentId, filePath }; +} + +describe('runMainTranscriptParse — idempotent reparse', () => { + it('forwards agent.usage.request events with identical request_id/model pairs across a crash-before-save re-parse', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { saveParseState, createParseState } = await import('../parse-state.js'); + + const sessionId = 'session-idempotent'; + const transcriptPath = writeTranscript('transcript-idempotent.jsonl', [ + noUsageLine('uuid-0'), + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75, stopReason: 'end_turn' }), + ]); + + await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + + const firstPairs = forwardedEvents() + .filter((e) => e.type === 'agent.usage.request') + .map((e) => `${e.request_id}::${e.model}`) + .sort(); + + expect(firstPairs).toHaveLength(2); + + // Simulate "a re-parse after a crash before state was saved": the transcript file is fully + // there, but the persisted state is wound back to fresh (as if the first run's save never + // happened). + await saveParseState(sessionId, createParseState()); + forwardMock.mockClear(); + + await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + + const secondPairs = forwardedEvents() + .filter((e) => e.type === 'agent.usage.request') + .map((e) => `${e.request_id}::${e.model}`) + .sort(); + + expect(secondPairs).toHaveLength(2); + expect(secondPairs).toEqual(firstPairs); + }); +}); + +describe('runMainTranscriptParse — Stop trigger', () => { + it('forwards one agent.usage.request event per distinct request plus one incremental agent.session.summary', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-stop-basic'; + const transcriptPath = writeTranscript('transcript-stop.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), + ]); + + await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + + expect(forwardMock).toHaveBeenCalledTimes(3); + + const events = forwardedEvents(); + const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(usageEvents).toHaveLength(2); + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].phase).toBe('incremental'); + expect(summaryEvents[0]).not.toHaveProperty('ended_at'); + }); +}); + +describe('runMainTranscriptParse — PreCompact trigger', () => { + it('forwards usage-request events but never a session summary', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-precompact'; + const transcriptPath = writeTranscript('transcript-precompact.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + ]); + + await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); + + const events = forwardedEvents(); + const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(usageEvents).toHaveLength(1); + expect(summaryEvents).toHaveLength(0); + }); +}); + +describe('runMainTranscriptParse — SessionEnd trigger', () => { + it('forwards a final-phase summary with an ended_at key present', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-end'; + const transcriptPath = writeTranscript('transcript-end.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + ]); + + await runMainTranscriptParse(sessionId, transcriptPath, 'SessionEnd'); + + const events = forwardedEvents(); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].phase).toBe('final'); + expect(summaryEvents[0]).toHaveProperty('ended_at'); + expect(typeof summaryEvents[0].ended_at).toBe('string'); + }); +}); + +describe('runMainTranscriptParse — missing transcript file', () => { + it('resolves cleanly without ever forwarding anything, for a trigger that never emits a summary', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { loadParseState } = await import('../parse-state.js'); + + const sessionId = 'session-missing-file'; + const missingPath = join(transcriptDir, 'does-not-exist.jsonl'); + + await expect(runMainTranscriptParse(sessionId, missingPath, 'PreCompact')).resolves.toBeUndefined(); + + expect(forwardMock).not.toHaveBeenCalled(); + + const state = await loadParseState(sessionId); + expect(state.mainOffset).toBe(0); + }); + + it('never throws even on Stop (which does attempt a full-file summary recompute)', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-missing-file-stop'; + const missingPath = join(transcriptDir, 'also-does-not-exist.jsonl'); + + await expect(runMainTranscriptParse(sessionId, missingPath, 'Stop')).resolves.toBeUndefined(); + }); +}); + +describe('runMainTranscriptParse — tool-call accumulation', () => { + it('counts Edit/Write tool_use blocks into files_changed/files_written on the Stop summary', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-tools'; + const toolLine = JSON.stringify({ + gitBranch: 'main', + timestamp: '2026-10-01T00:00:00.000Z', + message: { + role: 'assistant', + content: [ + { type: 'tool_use', id: 'tool-1', name: 'Write', input: { file_path: '/repo/a.ts' } }, + { type: 'tool_use', id: 'tool-2', name: 'Edit', input: { file_path: '/repo/b.ts' } }, + ], + }, + }); + const resultLine = JSON.stringify({ + gitBranch: 'main', + timestamp: '2026-10-01T00:00:01.000Z', + message: { + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'tool-1', is_error: false }, + { type: 'tool_result', tool_use_id: 'tool-2', is_error: true }, + ], + }, + }); + const transcriptPath = writeTranscript('transcript-tools.jsonl', [toolLine, resultLine]); + + await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + + const summary = forwardedEvents().find((e) => e.type === 'agent.session.summary'); + expect(summary).toBeDefined(); + expect(summary?.files_written).toEqual(['/repo/a.ts']); + expect(summary?.files_changed).toEqual(['/repo/b.ts']); + expect((summary?.tool_calls as Record).Write).toBe(1); + expect((summary?.tool_calls as Record).Edit).toBe(1); + expect((summary?.tool_errors as Record).Edit).toBe(1); + expect((summary?.tool_errors as Record).Write).toBe(0); + }); +}); + +describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, one pre-advanced)', () => { + it('forwards exactly three agent.subagent.usage events — one per subagent, including the one whose own SubagentStop already advanced its offset — never a fourth', async () => { + const { runSubagentTranscriptParse } = await import('../orchestrator.js'); + const { findSubagentFiles } = await import('../subagent-usage.js'); + + const sessionId = 'session-backstop'; + const mainTranscriptPath = writeTranscript(`${sessionId}.jsonl`, [noUsageLine('uuid-main')]); + + writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'uuid-a1-1', messageId: 'msg-a1-1', outputTokens: 10 }), + ]); + writeSubagentFixture(sessionId, 'a2', [ + usageLine({ uuid: 'uuid-a2-1', messageId: 'msg-a2-1', outputTokens: 20 }), + ]); + writeSubagentFixture(sessionId, 'a3', [ + usageLine({ uuid: 'uuid-a3-1', messageId: 'msg-a3-1', outputTokens: 30 }), + ]); + + // Simulate a1's own SubagentStop having already fired and advanced its offset past its + // content (and already forwarded its own agent.subagent.usage event once). + const filesBeforeBackstop = await findSubagentFiles(mainTranscriptPath); + const a1File = filesBeforeBackstop.find((f) => f.agentId === 'a1'); + if (!a1File) throw new Error('fixture missing a1'); + await runSubagentTranscriptParse(sessionId, mainTranscriptPath, a1File); + + // Clear the setup call's forwards so they don't pollute the backstop assertion below. + forwardMock.mockClear(); + + // Exercise exactly what the plugin's SessionEnd branch does: discover every subagent file + // for the session and re-run the subagent parse for each one, unconditionally — the + // crashed/missed-hook backstop. + const allFiles = await findSubagentFiles(mainTranscriptPath); + expect(allFiles).toHaveLength(3); + for (const file of allFiles) { + await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file); + } + + const subagentUsageEvents = forwardedEvents().filter((e) => e.type === 'agent.subagent.usage'); + // Exactly three — a1 (already-advanced, re-summarized rather than skipped), a2, a3. Never a + // fourth (no duplicate re-send for a1). + expect(subagentUsageEvents).toHaveLength(3); + + const agentIds = subagentUsageEvents.map((e) => e.agent_id).sort(); + expect(agentIds).toEqual(['a1', 'a2', 'a3']); + }); +}); + +describe('runSubagentTranscriptParse — no new bytes since last run', () => { + it('forwards zero new agent.usage.request events on a no-op reparse, but still forwards exactly one agent.subagent.usage event summarizing unchanged cumulative usage', async () => { + const { runSubagentTranscriptParse } = await import('../orchestrator.js'); + const { findSubagentFiles } = await import('../subagent-usage.js'); + + const sessionId = 'session-no-new-bytes'; + const mainTranscriptPath = writeTranscript(`${sessionId}.jsonl`, [noUsageLine('uuid-main')]); + writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'uuid-a1-1', messageId: 'msg-a1-1', outputTokens: 10 }), + usageLine({ uuid: 'uuid-a1-2', messageId: 'msg-a1-2', outputTokens: 20 }), + ]); + + const [file] = await findSubagentFiles(mainTranscriptPath); + + await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file); + + const firstEvents = forwardedEvents(); + expect(firstEvents.filter((e) => e.type === 'agent.usage.request')).toHaveLength(2); + expect(firstEvents.filter((e) => e.type === 'agent.subagent.usage')).toHaveLength(1); + + forwardMock.mockClear(); + + // Re-run on the same subagent file with no new content appended since the last call (its + // offset is now at EOF). + await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file); + + const secondEvents = forwardedEvents(); + const secondUsageRequests = secondEvents.filter((e) => e.type === 'agent.usage.request'); + const secondSubagentUsage = secondEvents.filter((e) => e.type === 'agent.subagent.usage'); + + // No new lines means no new/touched openRequests keys at the agent.usage.request layer. + expect(secondUsageRequests).toHaveLength(0); + // But the agent.subagent.usage event is still forwarded exactly once, summarizing the same + // cumulative (unchanged) usage. + expect(secondSubagentUsage).toHaveLength(1); + expect(secondSubagentUsage[0].output_tokens).toBe(30); + expect(secondSubagentUsage[0].api_calls).toBe(2); + }); +}); + +describe('runSubagentTranscriptParse — missing subagent transcript file', () => { + it('resolves without throwing and still forwards one empty-usage agent.subagent.usage event, with no agent.usage.request events', async () => { + const { runSubagentTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-subagent-missing'; + const mainTranscriptPath = writeTranscript(`${sessionId}.jsonl`, [noUsageLine('uuid-main')]); + const missingSubagentFile = { + agentId: 'ghost', + filePath: join(transcriptDir, sessionId, 'subagents', 'agent-ghost.jsonl'), + }; + + await expect( + runSubagentTranscriptParse(sessionId, mainTranscriptPath, missingSubagentFile) + ).resolves.toBeUndefined(); + + const events = forwardedEvents(); + const usageRequestEvents = events.filter((e) => e.type === 'agent.usage.request'); + const subagentUsageEvents = events.filter((e) => e.type === 'agent.subagent.usage'); + + expect(usageRequestEvents).toHaveLength(0); + expect(subagentUsageEvents).toHaveLength(1); + expect(subagentUsageEvents[0].api_calls).toBe(0); + expect(subagentUsageEvents[0].input_tokens).toBe(0); + expect(subagentUsageEvents[0].started_at).toBe(''); + expect(subagentUsageEvents[0].duration_ms).toBe(0); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts new file mode 100644 index 000000000..c7ec5dedd --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts @@ -0,0 +1,102 @@ +/** + * Tests for transcript parse-state persistence: `createParseState`, `loadParseState`, + * and `saveParseState`. + * + * `loadParseState` must never throw — a missing file or corrupt JSON on disk both fall + * back to a fresh state, since later tasks drive transcript parsing off whatever this + * returns and cannot tolerate a thrown exception interrupting that loop. + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { + createParseState, + loadParseState, + saveParseState, + type OpenUsageRequest, + type TranscriptParseState, +} from '../parse-state.js'; + +let codemieHome: string; + +beforeEach(() => { + codemieHome = mkdtempSync(join(tmpdir(), 'codemie-home-')); + process.env.CODEMIE_HOME = codemieHome; +}); + +afterEach(() => { + delete process.env.CODEMIE_HOME; + rmSync(codemieHome, { recursive: true, force: true }); +}); + +describe('createParseState', () => { + it('returns the fresh shape', () => { + expect(createParseState()).toEqual({ + mainOffset: 0, + subagentOffsets: {}, + openRequests: {}, + activeSkill: '', + branchCounts: {}, + }); + }); +}); + +describe('loadParseState', () => { + it('returns the fresh shape when no state file exists', async () => { + const state = await loadParseState('session-missing'); + + expect(state).toEqual(createParseState()); + }); + + it('recovers to the fresh shape instead of throwing on corrupt JSON', async () => { + const stateDir = join(codemieHome, 'analytics', 'state'); + mkdirSync(stateDir, { recursive: true }); + writeFileSync(join(stateDir, 'session-corrupt.json'), '{not valid json'); + + const state = await loadParseState('session-corrupt'); + + expect(state).toEqual(createParseState()); + }); + + it('round-trips openRequests and branchCounts exactly through saveParseState', async () => { + const openRequest: OpenUsageRequest = { + requestId: 'req1', + model: 'claude-3-5-sonnet', + modelRaw: 'claude-3-5-sonnet-20241022', + timestamp: '2026-10-01T00:00:00.000Z', + speed: 'standard', + inferenceGeo: 'us', + serviceTier: 'standard', + inputTokens: 100, + cacheCreation5mTokens: 10, + cacheCreation1hTokens: 0, + cacheReadTokens: 5, + outputTokens: 200, + webSearchRequests: 1, + webFetchRequests: 0, + scopeKind: 'main', + scopeName: '', + agentId: '', + stopReason: 'end_turn', + isApiError: false, + gitBranch: 'main', + }; + + const state: TranscriptParseState = { + mainOffset: 42, + subagentOffsets: { 'sub-1': 7 }, + openRequests: { 'req1::claude-3-5-sonnet': openRequest }, + activeSkill: 'brainstorming', + branchCounts: { main: 3, feature: 1 }, + }; + + await saveParseState('session-roundtrip', state); + const loaded = await loadParseState('session-roundtrip'); + + expect(loaded.openRequests).toEqual(state.openRequests); + expect(loaded.branchCounts).toEqual(state.branchCounts); + expect(loaded).toEqual(state); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts new file mode 100644 index 000000000..0bb5e3de9 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts @@ -0,0 +1,226 @@ +/** + * Tests for the `agent.session.summary` builder: `updateBranchCounts`, `primaryModel`, + * `branchDominant`, and `buildSessionSummaryEvent`. + */ + +import { describe, it, expect } from 'vitest'; +import { + updateBranchCounts, + primaryModel, + branchDominant, + buildSessionSummaryEvent, + type SessionSummaryAccumulator, +} from '../session-summary.js'; +import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; + +function emptyAccumulator(): SessionSummaryAccumulator { + return { + models: {}, + toolCalls: {}, + linesAdded: 0, + linesRemoved: 0, + filesChanged: new Set(), + filesWritten: new Set(), + compactionCount: 0, + }; +} + +function emptyNamed(): NamedInvocationCounts { + return { skillInvocations: {}, agentInvocations: {}, commandInvocations: {} }; +} + +describe('branchDominant', () => { + it('resolves the highest-count branch (mid-session branch-switch scenario)', () => { + expect(branchDominant({ main: 3, feature: 7 })).toBe('feature'); + }); + + it('returns "" for an empty map', () => { + expect(branchDominant({})).toBe(''); + }); +}); + +describe('primaryModel', () => { + it('returns the highest-count model key', () => { + expect(primaryModel({ 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 })).toBe('claude-opus-4-1'); + }); + + it('returns "" for an empty map', () => { + expect(primaryModel({})).toBe(''); + }); +}); + +describe('updateBranchCounts', () => { + it('mutates the passed-in counts object in place, bumping the named branch by 1 each call', () => { + const counts: Record = {}; + + updateBranchCounts(counts, 'main'); + expect(counts.main).toBe(1); + + updateBranchCounts(counts, 'main'); + expect(counts.main).toBe(2); + }); + + it('skips incrementing when branch is falsy/empty', () => { + const counts: Record = {}; + + updateBranchCounts(counts, ''); + + expect(counts['']).toBeUndefined(); + expect(Object.keys(counts)).toHaveLength(0); + }); +}); + +describe('buildSessionSummaryEvent', () => { + it('omits ended_at entirely for phase "incremental"', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect('ended_at' in event).toBe(false); + expect(Object.keys(event)).not.toContain('ended_at'); + expect(event.type).toBe('agent.session.summary'); + expect(event.session_id).toBe('session-1'); + expect(event.phase).toBe('incremental'); + expect(event.started_at).toBe('2026-10-01T00:00:00.000Z'); + }); + + it('includes ended_at for phase "final"', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect('ended_at' in event).toBe(true); + expect(event.ended_at).toBe('2026-10-01T01:00:00.000Z'); + expect(event.phase).toBe('final'); + }); + + it('flattens acc.toolCalls {calls, errors} shape into separate tool_calls/tool_errors maps', () => { + const acc = emptyAccumulator(); + acc.toolCalls = { + Read: { calls: 5, errors: 0 }, + Edit: { calls: 3, errors: 1 }, + }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.tool_calls).toEqual({ Read: 5, Edit: 3 }); + expect(event.tool_errors).toEqual({ Read: 0, Edit: 1 }); + }); + + it('derives skills_used and primary_command from a constructed NamedInvocationCounts', () => { + const named: NamedInvocationCounts = { + skillInvocations: { 'codemie:msgraph': 2, brainstorming: 1 }, + agentInvocations: {}, + commandInvocations: { init: 1, deploy: 4 }, + }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + named, + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.skills_used).toEqual({ 'codemie:msgraph': 2, brainstorming: 1 }); + expect(event.primary_command).toBe('deploy'); + expect(event.commands_in_order).toEqual(Object.keys(named.commandInvocations)); + }); + + it('reports models_used as the full count map and primary_model as the max key', () => { + const acc = emptyAccumulator(); + acc.models = { 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect(event.models_used).toEqual({ 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 }); + expect(event.primary_model).toBe('claude-opus-4-1'); + }); + + it('reports lines/files/compaction fields and branch_counts/branch_dominant, converting Sets to arrays', () => { + const acc = emptyAccumulator(); + acc.linesAdded = 42; + acc.linesRemoved = 7; + acc.filesChanged = new Set(['a.ts', 'b.ts']); + acc.filesWritten = new Set(['a.ts']); + acc.compactionCount = 2; + + const branchCounts = { main: 3, feature: 7 }; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + branchCounts, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.lines_added).toBe(42); + expect(event.lines_removed).toBe(7); + expect(event.files_changed).toEqual(['a.ts', 'b.ts']); + expect(event.files_written).toEqual(['a.ts']); + expect(event.compaction_count).toBe(2); + expect(event.branch_counts).toBe(branchCounts); + expect(event.branch_dominant).toBe('feature'); + }); + + it('emits title as a literal empty string (no identified source, per Open risks)', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect(event.title).toBe(''); + }); + + it('does not include an apiCalls/api_calls field (left to the caller/orchestrator to merge)', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + undefined + ); + + expect('api_calls' in event).toBe(false); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts new file mode 100644 index 000000000..4f21fab03 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts @@ -0,0 +1,295 @@ +/** + * Tests for the `agent.subagent.usage` builder: `findSubagentFiles` and + * `buildSubagentUsageEvent`. + * + * Fixture layout, built fresh per test under a temp dir: + * /.jsonl — trivial main-transcript placeholder + * //subagents/agent-.jsonl — one subagent transcript per fixture + * //subagents/agent-.meta.json — sidecar (toolUseId/agentType/spawnDepth) + * + * Three subagents are used throughout: + * - "a1": sidecar OMITS spawnDepth (top-level subagent — spawn_depth must default to 0), + * two usage-bearing transcript lines (exercises summing across >1 request). + * - "a2": sidecar INCLUDES spawnDepth: 2 (nested subagent — pass-through), one usage line. + * - "a3": sidecar includes toolUseId/agentType, one usage line. + */ + +import { describe, it, expect, afterEach } from 'vitest'; +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { findSubagentFiles, buildSubagentUsageEvent, type SubagentFile } from '../subagent-usage.js'; +import { parseUsageLine, buildUsageRequestEvent } from '../usage-request.js'; +import type { OpenUsageRequest } from '../parse-state.js'; + +let tmpDir: string | undefined; + +afterEach(async () => { + if (tmpDir) { + await rm(tmpDir, { recursive: true, force: true }); + tmpDir = undefined; + } +}); + +/** Build one transcript JSONL usage line, mirroring Task 8's confirmed transcript shape. */ +function usageLine(messageId: string, inputTokens: number, outputTokens: number): string { + return JSON.stringify({ + sessionId: 'session-subagent-1', + gitBranch: 'main', + cwd: '/repo', + timestamp: '2026-10-01T00:00:00.000Z', + version: '1.2.3', + entrypoint: 'cli', + uuid: `uuid-${messageId}`, + parentUuid: null, + isSidechain: true, + userType: 'external', + message: { + id: messageId, + role: 'assistant', + model: 'claude-sonnet-4-5-20250929', + stop_reason: 'end_turn', + usage: { + input_tokens: inputTokens, + output_tokens: outputTokens, + cache_read_input_tokens: 2, + cache_creation_input_tokens: 0, + service_tier: 'standard', + speed: 'standard', + inference_geo: '', + cache_creation: { ephemeral_1h_input_tokens: 1, ephemeral_5m_input_tokens: 1 }, + server_tool_use: { web_search_requests: 0, web_fetch_requests: 0 }, + iterations: [], + }, + }, + }); +} + +interface FixtureMeta { + toolUseId?: string; + agentType?: string; + spawnDepth?: number; +} + +async function buildFixture( + sessionId: string, + subagents: Record +): Promise { + const dir = await mkdtemp(join(tmpdir(), 'codemie-subagent-usage-')); + const mainTranscriptPath = join(dir, `${sessionId}.jsonl`); + await writeFile(mainTranscriptPath, ''); + + const subagentsDir = join(dir, sessionId, 'subagents'); + await mkdir(subagentsDir, { recursive: true }); + + for (const [agentId, { lines, meta }] of Object.entries(subagents)) { + await writeFile(join(subagentsDir, `agent-${agentId}.jsonl`), `${lines.join('\n')}\n`); + await writeFile(join(subagentsDir, `agent-${agentId}.meta.json`), JSON.stringify(meta)); + } + + return mainTranscriptPath; +} + +describe('findSubagentFiles', () => { + it('discovers all subagent files with their sidecar fields, defaulting spawnDepth to undefined when the sidecar omits it', async () => { + const sessionId = 'session-subagent-1'; + const mainTranscriptPath = await buildFixture(sessionId, { + a1: { lines: [usageLine('msg-a1-1', 100, 50)], meta: { toolUseId: 'tool-a1' } }, + a2: { lines: [usageLine('msg-a2-1', 10, 5)], meta: { agentType: 'reviewer', spawnDepth: 2 } }, + a3: { lines: [usageLine('msg-a3-1', 7, 3)], meta: { toolUseId: 'tool-a3', agentType: 'coder' } }, + }); + tmpDir = join(mainTranscriptPath, '..'); + + const files = await findSubagentFiles(mainTranscriptPath); + + expect(files).toHaveLength(3); + + const byId = (id: string): SubagentFile => { + const found = files.find((f) => f.agentId === id); + if (!found) throw new Error(`fixture missing ${id}`); + return found; + }; + + const a1 = byId('a1'); + expect(a1.toolUseId).toBe('tool-a1'); + expect(a1.agentType).toBeUndefined(); + expect(a1.spawnDepth).toBeUndefined(); + + const a2 = byId('a2'); + expect(a2.agentType).toBe('reviewer'); + expect(a2.spawnDepth).toBe(2); + expect(a2.toolUseId).toBeUndefined(); + + const a3 = byId('a3'); + expect(a3.toolUseId).toBe('tool-a3'); + expect(a3.agentType).toBe('coder'); + expect(a3.spawnDepth).toBeUndefined(); + }); + + it('returns [] when the subagents directory does not exist, without throwing', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-subagent-usage-empty-')); + const mainTranscriptPath = join(tmpDir, 'session-empty.jsonl'); + await writeFile(mainTranscriptPath, ''); + + const files = await findSubagentFiles(mainTranscriptPath); + + expect(files).toEqual([]); + }); + + it('returns [] for a missing main transcript path, without throwing', async () => { + const files = await findSubagentFiles('C:/nonexistent/path/session-missing.jsonl'); + expect(files).toEqual([]); + }); +}); + +describe('buildSubagentUsageEvent', () => { + it('sums token/cache fields across usageRequests, defaults spawn_depth to 0 when the file omits it, and passes caller-built maps through verbatim', () => { + const file: SubagentFile = { agentId: 'a1', filePath: '/tmp/agent-a1.jsonl', toolUseId: 'tool-a1' }; + const reqs: OpenUsageRequest[] = [ + { + requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + speed: 'standard', inferenceGeo: '', serviceTier: 'standard', + inputTokens: 100, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, + cacheReadTokens: 3, outputTokens: 50, webSearchRequests: 1, webFetchRequests: 0, + scopeKind: 'agent', scopeName: '', agentId: 'a1', + stopReason: 'end_turn', isApiError: false, gitBranch: 'main', + }, + { + requestId: 'r2', model: 'm', modelRaw: 'm-raw', timestamp: 't2', + speed: 'standard', inferenceGeo: '', serviceTier: 'standard', + inputTokens: 10, cacheCreation5mTokens: 4, cacheCreation1hTokens: 0, + cacheReadTokens: 1, outputTokens: 5, webSearchRequests: 0, webFetchRequests: 2, + scopeKind: 'agent', scopeName: '', agentId: 'a1', + stopReason: 'tool_use', isApiError: false, gitBranch: 'main', + }, + ]; + const toolCalls = { Read: 3, Edit: 1 }; + const toolErrors = { Edit: 1 }; + const skillsInvoked = { brainstorming: 1 }; + + const event = buildSubagentUsageEvent( + 'session-1', file, reqs, toolCalls, toolErrors, skillsInvoked, '2026-10-01T00:00:00.000Z', 1500 + ); + + expect(event).toEqual({ + type: 'agent.subagent.usage', + session_id: 'session-1', + agent_id: 'a1', + tool_use_id: 'tool-a1', + agent_type: '', + spawn_depth: 0, + description: '', + workflow_run: '', + worktree: '', + started_at: '2026-10-01T00:00:00.000Z', + duration_ms: 1500, + input_tokens: 110, + cache_creation_5m_tokens: 5, + cache_creation_1h_tokens: 2, + cache_read_tokens: 4, + output_tokens: 55, + web_search_requests: 1, + web_fetch_requests: 2, + api_calls: 2, + tool_calls: toolCalls, + tool_errors: toolErrors, + skills_invoked: skillsInvoked, + }); + }); + + it('passes a present spawn_depth through verbatim instead of defaulting to 0', () => { + const file: SubagentFile = { agentId: 'a2', filePath: '/tmp/agent-a2.jsonl', spawnDepth: 2 }; + + const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, {}, '', 0); + + expect(event.spawn_depth).toBe(2); + expect(event.api_calls).toBe(0); + }); + + it('never fabricates description/workflow_run/worktree — always empty string', () => { + const file: SubagentFile = { agentId: 'a3', filePath: '/tmp/agent-a3.jsonl' }; + + const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, {}, '', 0); + + expect(event.description).toBe(''); + expect(event.workflow_run).toBe(''); + expect(event.worktree).toBe(''); + }); +}); + +describe('cross-check: agent.subagent.usage summed tokens vs agent.usage.request (scope_kind: agent)', () => { + it('summing token fields across three agent.subagent.usage events equals summing the same fields across every agent.usage.request record derived from the same fixture', async () => { + const sessionId = 'session-subagent-cross'; + const mainTranscriptPath = await buildFixture(sessionId, { + a1: { + lines: [usageLine('msg-a1-1', 100, 50), usageLine('msg-a1-2', 10, 5)], + meta: { toolUseId: 'tool-a1' }, + }, + a2: { + lines: [usageLine('msg-a2-1', 200, 80)], + meta: { agentType: 'reviewer', spawnDepth: 2 }, + }, + a3: { + lines: [usageLine('msg-a3-1', 30, 15)], + meta: { toolUseId: 'tool-a3', agentType: 'coder' }, + }, + }); + tmpDir = join(mainTranscriptPath, '..'); + + const files = await findSubagentFiles(mainTranscriptPath); + expect(files).toHaveLength(3); + + const subagentUsageEvents: Array> = []; + const usageRequestEvents: Array> = []; + + for (const file of files) { + const raw = await import('node:fs/promises').then((m) => m.readFile(file.filePath, 'utf-8')); + const lines = raw.split('\n').filter((l) => l.trim().length > 0); + + const reqs: OpenUsageRequest[] = lines + .map((line) => parseUsageLine(line, 'agent', '', file.agentId)) + .filter((r): r is OpenUsageRequest => r !== null); + + expect(reqs.length).toBeGreaterThan(0); + reqs.forEach((r) => expect(r.scopeKind).toBe('agent')); + + subagentUsageEvents.push( + buildSubagentUsageEvent(sessionId, file, reqs, {}, {}, {}, '2026-10-01T00:00:00.000Z', 0) + ); + + for (const req of reqs) { + usageRequestEvents.push(buildUsageRequestEvent(sessionId, req)); + } + } + + expect(subagentUsageEvents).toHaveLength(3); + // 2 + 1 + 1 usage-bearing lines across the three subagent transcripts. + expect(usageRequestEvents).toHaveLength(4); + usageRequestEvents.forEach((e) => expect(e.scope_kind).toBe('agent')); + + const sumField = (records: Array>, field: string): number => + records.reduce((total, r) => total + Number(r[field] ?? 0), 0); + + const tokenFieldPairs: Array<[string, string]> = [ + ['input_tokens', 'input_tokens'], + ['output_tokens', 'output_tokens'], + ['cache_creation_5m_tokens', 'cache_creation_5m_tokens'], + ['cache_creation_1h_tokens', 'cache_creation_1h_tokens'], + ['cache_read_tokens', 'cache_read_tokens'], + ['web_search_requests', 'web_search_requests'], + ['web_fetch_requests', 'web_fetch_requests'], + ]; + + for (const [subagentField, requestField] of tokenFieldPairs) { + expect(sumField(subagentUsageEvents, subagentField)).toBe(sumField(usageRequestEvents, requestField)); + } + + // api_calls across the three subagent.usage events equals the total number of + // agent.usage.request records derived from the same fixture. + expect(sumField(subagentUsageEvents, 'api_calls')).toBe(usageRequestEvents.length); + + // Known concrete totals: input 100+10+200+30=340, output 50+5+80+15=150. + expect(sumField(subagentUsageEvents, 'input_tokens')).toBe(340); + expect(sumField(subagentUsageEvents, 'output_tokens')).toBe(150); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts new file mode 100644 index 000000000..f8511a01c --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts @@ -0,0 +1,82 @@ +/** + * Tests for the incremental transcript reader: `readNewLines`. + * + * The reader must only ever return complete, newline-terminated lines, cutting + * at the last `\n` byte so a partially-written trailing line is never handed + * back to the caller — mirrors the safe-cut rule in + * `src/providers/plugins/sso/proxy/plugins/otlp-spool/spool-io.ts`'s + * `snapshotPendingHookRecords`. It must never throw, even for a missing file. + */ + +import { describe, it, expect, afterEach } from 'vitest'; +import { mkdtemp, writeFile, appendFile, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { readNewLines } from '../transcript-reader.js'; + +let tmpDir: string | undefined; + +afterEach(async () => { + if (tmpDir) { + await rm(tmpDir, { recursive: true, force: true }); + tmpDir = undefined; + } +}); + +describe('readNewLines', () => { + it('returns only complete lines and nextOffset points exactly after the last complete line; a second call returns only newly appended lines', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + const completePrefix = 'line1\nline2\n'; + await writeFile(filePath, `${completePrefix}partial-line-no-newli`); + + const first = await readNewLines(filePath, 0); + + expect(first.lines).toEqual(['line1', 'line2']); + expect(first.nextOffset).toBe(Buffer.byteLength(completePrefix)); + + // Complete the previously-partial line and add a new complete line. + await appendFile(filePath, 'ne\nline4\n'); + + const second = await readNewLines(filePath, first.nextOffset); + + expect(second.lines).toEqual(['partial-line-no-newline', 'line4']); + expect(second.nextOffset).toBeGreaterThan(first.nextOffset); + }); + + it('returns an empty result without throwing when the file is missing', async () => { + const result = await readNewLines( + 'C:/nonexistent/path/that/does/not/exist.jsonl', + 0 + ); + + expect(result).toEqual({ lines: [], nextOffset: 0 }); + }); + + it('returns an empty result and does not advance the offset when the file has only an unterminated partial line', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + await writeFile(filePath, 'no-newline-yet'); + + const result = await readNewLines(filePath, 0); + + expect(result).toEqual({ lines: [], nextOffset: 0 }); + }); + + it('returns an empty result when nothing new has been written since fromOffset', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + const content = 'line1\nline2\n'; + await writeFile(filePath, content); + + const result = await readNewLines(filePath, Buffer.byteLength(content)); + + expect(result).toEqual({ + lines: [], + nextOffset: Buffer.byteLength(content), + }); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts new file mode 100644 index 000000000..f9eab7d4a --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts @@ -0,0 +1,203 @@ +/** + * Tests for `agent.usage.request` extraction and merge: `parseUsageLine`, + * `mergeUsageRequest`, and `buildUsageRequestEvent`. + * + * Fixture: `fixtures/transcript-usage.jsonl` — one line with no `message.usage` + * (must parse to `null`), two lines sharing the same `message.id` where the + * second has a higher `output_tokens` and a non-empty `stop_reason` the first + * lacks (feeds `mergeUsageRequest`), and one fully-populated "normal" line + * (round-trips every field, including the `model`/`modelRaw` resolution split). + */ + +import { describe, it, expect, beforeAll } from 'vitest'; +import { readFile } from 'node:fs/promises'; +import { join } from 'node:path'; +import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from '../usage-request.js'; +import type { OpenUsageRequest } from '../parse-state.js'; + +let lines: string[]; + +beforeAll(async () => { + const raw = await readFile(join(__dirname, 'fixtures', 'transcript-usage.jsonl'), 'utf-8'); + lines = raw.split('\n').filter((l) => l.trim().length > 0); +}); + +describe('parseUsageLine', () => { + it('returns null for a line with no message.usage block', () => { + expect(lines).toHaveLength(4); + const result = parseUsageLine(lines[0], 'main', '', ''); + expect(result).toBeNull(); + }); + + it('returns null for malformed JSON instead of throwing', () => { + expect(() => parseUsageLine('not valid json {{{', 'main', '', '')).not.toThrow(); + expect(parseUsageLine('not valid json {{{', 'main', '', '')).toBeNull(); + }); + + it('extracts every field from a fully-populated line', () => { + const req = parseUsageLine(lines[3], 'main', '', ''); + + expect(req).not.toBeNull(); + const r = req as OpenUsageRequest; + + // request_id comes from message.id, not any top-level requestId. + expect(r.requestId).toBe('msg_normal_1'); + // modelRaw is the transcript's own literal message.model (unresolved). + expect(r.modelRaw).toBe('claude-sonnet-4-5-20250929'); + // model is resolved via parseBackendModelName (x-litellm-model-name) first. + expect(r.model).toBe('bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0'); + expect(r.timestamp).toBe('2026-10-01T00:00:03.000Z'); + expect(r.speed).toBe('fast'); + expect(r.inferenceGeo).toBe('us'); + expect(r.serviceTier).toBe('priority'); + expect(r.inputTokens).toBe(300); + expect(r.outputTokens).toBe(75); + expect(r.cacheReadTokens).toBe(40); + // Nested cache_creation.* fields, not the flat cache_creation_input_tokens. + expect(r.cacheCreation5mTokens).toBe(5); + expect(r.cacheCreation1hTokens).toBe(10); + // Nested server_tool_use.* fields. + expect(r.webSearchRequests).toBe(2); + expect(r.webFetchRequests).toBe(1); + // stop_reason is a sibling of usage on message, not nested inside usage. + expect(r.stopReason).toBe('end_turn'); + expect(r.gitBranch).toBe('feature/epmcdme-15301'); + // No documented source field for isApiError on a real transcript line — defaults false. + expect(r.isApiError).toBe(false); + expect(r.scopeKind).toBe('main'); + expect(r.scopeName).toBe(''); + expect(r.agentId).toBe(''); + }); + + it('passes scopeKind/scopeName/agentId through verbatim from its own parameters', () => { + const req = parseUsageLine(lines[3], 'agent', 'reviewer', 'agent-42'); + + expect(req?.scopeKind).toBe('agent'); + expect(req?.scopeName).toBe('reviewer'); + expect(req?.agentId).toBe('agent-42'); + }); +}); + +describe('mergeUsageRequest', () => { + it('keeps the max output_tokens and the non-empty stop_reason across two records for the same message.id', () => { + const first = parseUsageLine(lines[1], 'main', '', ''); + const second = parseUsageLine(lines[2], 'main', '', ''); + + expect(first).not.toBeNull(); + expect(second).not.toBeNull(); + const a = first as OpenUsageRequest; + const b = second as OpenUsageRequest; + + expect(a.requestId).toBe(b.requestId); + expect(a.outputTokens).toBe(50); + expect(a.stopReason).toBe(''); + expect(b.outputTokens).toBe(120); + expect(b.stopReason).toBe('tool_use'); + + const merged = mergeUsageRequest(a, b); + + expect(merged.outputTokens).toBe(Math.max(a.outputTokens, b.outputTokens)); + expect(merged.outputTokens).toBe(120); + expect(merged.stopReason).toBe('tool_use'); + // Numeric fields take the max even when equal. + expect(merged.inputTokens).toBe(Math.max(a.inputTokens, b.inputTokens)); + // Merge returns a new object — neither input is mutated. + expect(a.outputTokens).toBe(50); + expect(b.outputTokens).toBe(120); + }); + + it('takes every numeric field as Math.max of the two inputs', () => { + const a: OpenUsageRequest = { + requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + speed: 'standard', inferenceGeo: '', serviceTier: 'standard', + inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, + cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, + scopeKind: 'main', scopeName: '', agentId: '', + stopReason: '', isApiError: false, gitBranch: 'main', + }; + const b: OpenUsageRequest = { + ...a, + inputTokens: 1, cacheCreation5mTokens: 9, cacheCreation1hTokens: 1, + cacheReadTokens: 30, outputTokens: 1, webSearchRequests: 0, webFetchRequests: 60, + timestamp: '', + }; + + const merged = mergeUsageRequest(a, b); + + expect(merged.inputTokens).toBe(10); + expect(merged.cacheCreation5mTokens).toBe(9); + expect(merged.cacheCreation1hTokens).toBe(2); + expect(merged.cacheReadTokens).toBe(30); + expect(merged.outputTokens).toBe(4); + expect(merged.webSearchRequests).toBe(5); + expect(merged.webFetchRequests).toBe(60); + // Non-numeric fields: b's value wins when non-empty, else a's. + expect(merged.timestamp).toBe('t1'); + }); + + it('does not mutate either input and returns a new object', () => { + const a: OpenUsageRequest = { + requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + speed: '', inferenceGeo: '', serviceTier: '', + inputTokens: 1, cacheCreation5mTokens: 0, cacheCreation1hTokens: 0, + cacheReadTokens: 0, outputTokens: 1, webSearchRequests: 0, webFetchRequests: 0, + scopeKind: 'main', scopeName: '', agentId: '', + stopReason: '', isApiError: false, gitBranch: '', + }; + const b: OpenUsageRequest = { ...a, outputTokens: 2, stopReason: 'end_turn', isApiError: true }; + const aCopy = { ...a }; + const bCopy = { ...b }; + + const merged = mergeUsageRequest(a, b); + + expect(a).toEqual(aCopy); + expect(b).toEqual(bCopy); + expect(merged).not.toBe(a); + expect(merged).not.toBe(b); + // isApiError: once true, stays true across merges. + expect(merged.isApiError).toBe(true); + }); +}); + +describe('buildUsageRequestEvent', () => { + it('maps every OpenUsageRequest field to its snake_case event field, with an explicit type', () => { + const req: OpenUsageRequest = { + requestId: 'req-1', model: 'resolved-model', modelRaw: 'literal-model', timestamp: '2026-10-01T00:00:00.000Z', + speed: 'fast', inferenceGeo: 'us', serviceTier: 'priority', + inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, + cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, + scopeKind: 'skill', scopeName: 'brainstorming', agentId: 'agent-7', + stopReason: 'end_turn', isApiError: false, gitBranch: 'main', + }; + + const event = buildUsageRequestEvent('session-123', req); + + expect(event).toEqual({ + type: 'agent.usage.request', + session_id: 'session-123', + request_id: 'req-1', + model_raw: 'literal-model', + model: 'resolved-model', + speed: 'fast', + inference_geo: 'us', + service_tier: 'priority', + input_tokens: 10, + cache_creation_5m_tokens: 1, + cache_creation_1h_tokens: 2, + cache_read_tokens: 3, + output_tokens: 4, + web_search_requests: 5, + web_fetch_requests: 6, + scope_kind: 'skill', + scope_name: 'brainstorming', + agent_id: 'agent-7', + stop_reason: 'end_turn', + is_api_error: false, + git_branch: 'main', + timestamp: '2026-10-01T00:00:00.000Z', + }); + // No event_id/schema_version here — those are stamped later, daemon-side. + expect(event).not.toHaveProperty('event_id'); + expect(event).not.toHaveProperty('schema_version'); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts new file mode 100644 index 000000000..7061feab6 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -0,0 +1,433 @@ +/** + * Main-transcript trigger orchestration for the `Stop`, `PreCompact`, and `SessionEnd` hook + * events. + * + * Each hook fire is a fresh CLI process, so this module reloads persisted parse state + * (`./parse-state.js`, Task 6), reads only the transcript lines appended since the last + * persisted `mainOffset` (`./transcript-reader.js`, Task 7), derives/merges + * `agent.usage.request` records for those new lines (`./usage-request.js`, Task 8), forwards one + * event per completed request, optionally forwards one `agent.session.summary` event + * (`./session-summary.js`, Task 10), and persists state back — all before returning. + * + * Never throws: every path is wrapped so a read/parse/forward failure degrades to a no-op rather + * than interrupting the hook that triggered it (`processOtlpEvent` must never block or fail on + * this). + */ + +import { readFile } from 'node:fs/promises'; +import { loadParseState, saveParseState } from './parse-state.js'; +import { readNewLines } from './transcript-reader.js'; +import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from './usage-request.js'; +import { + updateBranchCounts, + buildSessionSummaryEvent, + type SessionSummaryAccumulator, + type NamedInvocationCounts, +} from './session-summary.js'; +import { extractNamedInvocations } from '@/agents/plugins/claude/session/claude-named-invocations.js'; +import { forwardOtlpEventToSpool } from '../../utils.js'; +import { CLAUDE_CODE_OTLP_AGENT_NAME } from '../claude-code-otlp.constants.js'; +import { type SubagentFile, buildSubagentUsageEvent } from './subagent-usage.js'; + +// Re-exported so callers (e.g. claude-code-otlp.plugin.ts) can import both `SubagentFile` and +// `runSubagentTranscriptParse` from this one module, per the plan's wiring description. +export type { SubagentFile }; + +export type MainTranscriptTrigger = 'Stop' | 'PreCompact' | 'SessionEnd'; + +interface ContentBlock { + type?: string; + id?: string; + name?: string; + tool_use_id?: string; + is_error?: boolean; + isError?: boolean; + input?: { file_path?: unknown; path?: unknown }; +} + +interface TranscriptLine { + timestamp?: string; + gitBranch?: string; + message?: { content?: unknown }; +} + +/** + * Collect the set of `tool_use_id` values whose matching `tool_result` block carries a truthy + * `is_error`/`isError` flag. + * + * Shared between {@link buildFullAccumulator} (main transcript) and + * {@link scanSubagentTranscript} (subagent transcript) so the error-correlation logic does not + * drift between the two call sites. + */ +function collectErrorToolUseIds(parsedLines: TranscriptLine[]): Set { + const errorByToolUseId = new Set(); + for (const parsed of parsedLines) { + const content = parsed.message?.content; + if (!Array.isArray(content)) continue; + for (const item of content as ContentBlock[]) { + if ( + item?.type === 'tool_result' && + typeof item.tool_use_id === 'string' && + (item.is_error === true || item.isError === true) + ) { + errorByToolUseId.add(item.tool_use_id); + } + } + } + return errorByToolUseId; +} + +function emptyAccumulator(): SessionSummaryAccumulator { + return { + models: {}, + toolCalls: {}, + linesAdded: 0, + linesRemoved: 0, + filesChanged: new Set(), + filesWritten: new Set(), + compactionCount: 0, + }; +} + +/** + * Recompute the full-session summary accumulator, named-invocation counts, and session start + * time from byte 0 of the main transcript. + * + * `TranscriptParseState` (Task 6's fixed shape) has no persisted field for any of + * `SessionSummaryAccumulator`'s data or for `NamedInvocationCounts` — only `branchCounts` is + * incrementally tracked there. So, for `Stop`/`SessionEnd`, this helper re-derives everything + * else fresh from the whole transcript file every time (see Task 11's Note B). Transcripts are + * not enormous and this only runs on `Stop`/`SessionEnd`, not on every hook. + * + * Never throws: a missing/unreadable transcript resolves to the emptiest defensible result + * (empty accumulator, empty named-invocation counts, `startedAt: ''`); a malformed individual + * line is skipped rather than aborting the whole scan. + * + * Known limitations (no reliable in-transcript signal found for any of these — see spec.md's + * confidence gaps): + * - `toolCalls[*].errors` is derived from a sibling `tool_result` block's `is_error`/`isError` + * flag (the same pattern `claude.session.ts`/`claude.metrics-processor.ts` already use for + * tool-use_id → error lookups) when one is found; otherwise a tool call's `.errors` stays 0. + * - `linesAdded`/`linesRemoved` default to 0 — an `Edit`/`Write` tool_use's `input` carries the + * *proposed* edit, not a diff stat, so no reliable added/removed line count can be derived from + * it without re-implementing diffing (out of scope for this task). + * - `compactionCount` defaults to 0 — no verified in-transcript signal was found (`PreCompact` is + * a hook event, not a transcript line). + */ +async function buildFullAccumulator( + transcriptPath: string +): Promise<{ acc: SessionSummaryAccumulator; named: NamedInvocationCounts; startedAt: string }> { + const acc = emptyAccumulator(); + + let raw: string; + try { + raw = await readFile(transcriptPath, 'utf-8'); + } catch { + return { acc, named: extractNamedInvocations([]), startedAt: '' }; + } + + const rawLines = raw.split('\n').filter((line) => line.trim().length > 0); + const parsedLines: TranscriptLine[] = []; + + for (const line of rawLines) { + try { + parsedLines.push(JSON.parse(line) as TranscriptLine); + } catch { + // Skip malformed lines rather than aborting the whole scan. + } + } + + // Pass 1: collect tool_result error flags keyed by their matching tool_use_id. + const errorByToolUseId = collectErrorToolUseIds(parsedLines); + + // Pass 2: models (reusing Task 8's own model-resolution logic via parseUsageLine). + for (const line of rawLines) { + const parsedUsage = parseUsageLine(line, 'main', '', ''); + if (parsedUsage) { + acc.models[parsedUsage.model] = (acc.models[parsedUsage.model] ?? 0) + 1; + } + } + + // Pass 3: tool calls/errors, files changed/written (Edit/Write tool_use payloads). + for (const parsed of parsedLines) { + const content = parsed.message?.content; + if (!Array.isArray(content)) continue; + for (const item of content as ContentBlock[]) { + if (item?.type !== 'tool_use' || typeof item.name !== 'string') continue; + + const entry = acc.toolCalls[item.name] ?? { calls: 0, errors: 0 }; + entry.calls += 1; + if (typeof item.id === 'string' && errorByToolUseId.has(item.id)) { + entry.errors += 1; + } + acc.toolCalls[item.name] = entry; + + const filePath = item.input?.file_path ?? item.input?.path; + if (typeof filePath === 'string' && filePath) { + if (item.name === 'Write') acc.filesWritten.add(filePath); + if (item.name === 'Edit') acc.filesChanged.add(filePath); + } + } + } + + const named = extractNamedInvocations(parsedLines); + const startedAt = parsedLines.length > 0 ? String(parsedLines[0].timestamp ?? '') : ''; + + return { acc, named, startedAt }; +} + +/** + * Orchestrate a main-transcript parse pass for one `Stop`/`PreCompact`/`SessionEnd` hook fire. + * + * - Loads persisted state, reads only the lines appended since `state.mainOffset`. + * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, + * keyed by `${requestId}::${model}` (matching `parse-state.ts`'s documented key shape), and + * updates `state.branchCounts` from every new line's `gitBranch` (regardless of whether that + * line carried usage). + * - Forwards one `agent.usage.request` event per request key touched by this pass. + * - On `Stop`/`SessionEnd` only, forwards exactly one `agent.session.summary` event + * (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`) built from a fresh full-file + * recompute (see {@link buildFullAccumulator} and Note B). `PreCompact` never forwards a + * summary. + * - Persists state back to disk. + * + * Scoping ruling (Note A — a judgment call, since no file in this codebase documents a reliable + * signal for when a *main*-transcript turn enters/exits a "skill context"): every + * main-transcript-derived usage record in this task is scoped as `scopeKind: 'main'`, + * `scopeName: ''` unconditionally. `state.activeSkill` is deliberately left untouched (not read, + * not written) here — it stays available, unused, for a future task that identifies a real + * signal for it. + * + * Swallows every error internally — never throws into `processOtlpEvent`. + */ +export async function runMainTranscriptParse( + sessionId: string, + transcriptPath: string, + trigger: MainTranscriptTrigger +): Promise { + try { + const state = await loadParseState(sessionId); + const { lines, nextOffset } = await readNewLines(transcriptPath, state.mainOffset); + + const touchedKeys = new Set(); + for (const line of lines) { + let rawGitBranch = ''; + try { + rawGitBranch = (JSON.parse(line) as { gitBranch?: string })?.gitBranch ?? ''; + } catch { + // Malformed line: still attempt usage parsing below (which has its own try/catch), but + // there is no branch to record from it. + } + if (rawGitBranch) { + updateBranchCounts(state.branchCounts, rawGitBranch); + } + + const parsed = parseUsageLine(line, 'main', '', ''); + if (parsed) { + const key = `${parsed.requestId}::${parsed.model}`; + const existing = state.openRequests[key]; + state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; + touchedKeys.add(key); + } + } + state.mainOffset = nextOffset; + + for (const key of touchedKeys) { + const event = buildUsageRequestEvent(sessionId, state.openRequests[key]); + await forwardOtlpEventToSpool(JSON.stringify(event), CLAUDE_CODE_OTLP_AGENT_NAME); + } + + if (trigger === 'Stop' || trigger === 'SessionEnd') { + const { acc, named, startedAt } = await buildFullAccumulator(transcriptPath); + const phase = trigger === 'SessionEnd' ? 'final' : 'incremental'; + const endedAt = trigger === 'SessionEnd' ? new Date().toISOString() : undefined; + const summaryEvent = buildSessionSummaryEvent( + sessionId, + phase, + acc, + named, + state.branchCounts, + startedAt, + endedAt + ); + await forwardOtlpEventToSpool(JSON.stringify(summaryEvent), CLAUDE_CODE_OTLP_AGENT_NAME); + } + // PreCompact: usage requests only, no summary — handled by skipping the block above. + + await saveParseState(sessionId, state); + } catch { + // Swallow everything — never throw into processOtlpEvent. + } +} + +interface SubagentScanResult { + toolCalls: Record; + toolErrors: Record; + skillsInvoked: Record; + startedAt: string; + durationMs: number; +} + +/** + * Recompute one subagent transcript's tool-call/tool-error/skill-invocation aggregates and + * timing span from byte 0 of its own file (the subagent-transcript analogue of + * {@link buildFullAccumulator}'s "recompute fresh each time" approach — `TranscriptParseState` + * has no persisted field for any of these either). + * + * - `toolCalls`/`toolErrors` reuse {@link collectErrorToolUseIds} for the same `tool_use_id` → + * `tool_result.is_error` correlation {@link buildFullAccumulator} uses, but tally into two + * parallel `Record` maps (not the combined `{calls, errors}` shape + * `SessionSummaryAccumulator` uses) to match {@link buildSubagentUsageEvent}'s own + * `tool_calls`/`tool_errors` parameter shapes. + * - `skillsInvoked` is `extractNamedInvocations(parsedLines).skillInvocations`, taken verbatim. + * - `startedAt` is the first parsed line's `timestamp`, or `''` when the file is empty/unreadable. + * - `durationMs` is `Date.parse(lastLine.timestamp) - Date.parse(firstLine.timestamp)`, guarded by + * `Number.isFinite` (covers a missing/unparseable timestamp on either end, and a single-line + * file) so it is never `NaN` — falls back to `0`. + * + * Never throws: a missing/unreadable file or an empty file both resolve to the emptiest + * defensible result; a malformed individual line is skipped rather than aborting the whole scan. + */ +async function scanSubagentTranscript(filePath: string): Promise { + const empty: SubagentScanResult = { + toolCalls: {}, + toolErrors: {}, + skillsInvoked: {}, + startedAt: '', + durationMs: 0, + }; + + let raw: string; + try { + raw = await readFile(filePath, 'utf-8'); + } catch { + return empty; + } + + const rawLines = raw.split('\n').filter((line) => line.trim().length > 0); + const parsedLines: TranscriptLine[] = []; + for (const line of rawLines) { + try { + parsedLines.push(JSON.parse(line) as TranscriptLine); + } catch { + // Skip malformed lines rather than aborting the whole scan. + } + } + + if (parsedLines.length === 0) { + return empty; + } + + const errorByToolUseId = collectErrorToolUseIds(parsedLines); + const toolCalls: Record = {}; + const toolErrors: Record = {}; + + for (const parsed of parsedLines) { + const content = parsed.message?.content; + if (!Array.isArray(content)) continue; + for (const item of content as ContentBlock[]) { + if (item?.type !== 'tool_use' || typeof item.name !== 'string') continue; + toolCalls[item.name] = (toolCalls[item.name] ?? 0) + 1; + if (typeof item.id === 'string' && errorByToolUseId.has(item.id)) { + toolErrors[item.name] = (toolErrors[item.name] ?? 0) + 1; + } + } + } + + const named = extractNamedInvocations(parsedLines); + const startedAt = String(parsedLines[0].timestamp ?? ''); + const lastTimestamp = String(parsedLines[parsedLines.length - 1].timestamp ?? ''); + const diff = Date.parse(lastTimestamp) - Date.parse(startedAt); + const durationMs = Number.isFinite(diff) ? diff : 0; + + return { toolCalls, toolErrors, skillsInvoked: named.skillInvocations, startedAt, durationMs }; +} + +/** + * Orchestrate a subagent-transcript parse pass for one `SubagentStop` hook fire, or for one + * subagent file discovered by the `SessionEnd` backstop scan (`findSubagentFiles()`, Task 9). + * + * - Loads persisted state, reads only the lines appended since + * `state.subagentOffsets[subagentFile.agentId]` (defaulting to 0 for a never-before-seen + * agent). + * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, + * scoped `scopeKind: 'agent'`, keyed by `${requestId}::${model}` — same merge/key convention + * `runMainTranscriptParse` uses for the main transcript. + * - Forwards one `agent.usage.request` event per request key touched by *this* pass (no new + * lines means no new forwards — a no-op reparse resends nothing at this layer). + * - Unconditionally forwards exactly one `agent.subagent.usage` event summarizing this agent's + * *cumulative* usage (every `scopeKind: 'agent'` record in `state.openRequests` for this + * `agentId`, not just the ones touched this pass) plus a fresh full-file tool-call/error/skill/ + * timing scan (see {@link scanSubagentTranscript}) — this is deliberate: the `SessionEnd` + * backstop's whole purpose is to guarantee every subagent gets at least one + * `agent.subagent.usage` event even when its own `SubagentStop` hook never fired, so a + * re-run with nothing new since the last pass still emits one (summarizing unchanged + * cumulative state), rather than being skipped. + * - Persists the updated `subagentOffsets[subagentFile.agentId]` (and `openRequests`) back to + * disk. + * + * `mainTranscriptPath` is accepted per the plan's interface but is not used internally — + * `subagentFile.filePath` already names the file to read, and the main transcript's own path + * carries no information this function's own logic needs. + * + * Swallows every error internally — never throws into `processOtlpEvent`. + */ +export async function runSubagentTranscriptParse( + sessionId: string, + // `mainTranscriptPath` (positionally the second parameter, per the plan's binding interface + // signature) is unused in this function's own body — `subagentFile.filePath` already locates + // the file this call concerns. Prefixed with `_` per this repo's unused-arg convention + // (eslint.config.mjs argsIgnorePattern) rather than suppressing the lint rule. + _mainTranscriptPath: string, + subagentFile: SubagentFile +): Promise { + try { + const state = await loadParseState(sessionId); + const fromOffset = state.subagentOffsets[subagentFile.agentId] ?? 0; + const { lines, nextOffset } = await readNewLines(subagentFile.filePath, fromOffset); + + const touchedKeys = new Set(); + for (const line of lines) { + const parsed = parseUsageLine(line, 'agent', '', subagentFile.agentId); + if (parsed) { + const key = `${parsed.requestId}::${parsed.model}`; + const existing = state.openRequests[key]; + state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; + touchedKeys.add(key); + } + } + state.subagentOffsets[subagentFile.agentId] = nextOffset; + + for (const key of touchedKeys) { + const event = buildUsageRequestEvent(sessionId, state.openRequests[key]); + await forwardOtlpEventToSpool(JSON.stringify(event), CLAUDE_CODE_OTLP_AGENT_NAME); + } + + // Cumulative usage for this agent — every scope_kind:'agent' record known for it so far, not + // just the ones touched this pass (consistent with buildFullAccumulator's own + // "recomputed from full state" framing for the sibling agent.session.summary event). + const usageRequestsForAgent = Object.values(state.openRequests).filter( + (r) => r.scopeKind === 'agent' && r.agentId === subagentFile.agentId + ); + + const { toolCalls, toolErrors, skillsInvoked, startedAt, durationMs } = await scanSubagentTranscript( + subagentFile.filePath + ); + + const subagentEvent = buildSubagentUsageEvent( + sessionId, + subagentFile, + usageRequestsForAgent, + toolCalls, + toolErrors, + skillsInvoked, + startedAt, + durationMs + ); + await forwardOtlpEventToSpool(JSON.stringify(subagentEvent), CLAUDE_CODE_OTLP_AGENT_NAME); + + await saveParseState(sessionId, state); + } catch { + // Swallow everything — never throw into processOtlpEvent. + } +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts new file mode 100644 index 000000000..e9ed3c8de --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -0,0 +1,99 @@ +/** + * Transcript parse-state persistence. + * + * Transcript parsing is incremental: each parse pass picks up where the previous one + * left off (byte offsets into the main transcript and per-subagent transcripts), + * tracks usage requests opened but not yet closed by their matching response, the + * currently active skill, and per-branch request counts. This module persists that + * state to disk between parse passes, keyed by session id. + */ + +import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import { dirname } from 'node:path'; +import { getCodemiePath } from '@/utils/paths.js'; + +export interface OpenUsageRequest { + requestId: string; + model: string; + modelRaw: string; + timestamp: string; + speed: string; + inferenceGeo: string; + serviceTier: string; + inputTokens: number; + cacheCreation5mTokens: number; + cacheCreation1hTokens: number; + cacheReadTokens: number; + outputTokens: number; + webSearchRequests: number; + webFetchRequests: number; + scopeKind: 'main' | 'skill' | 'agent'; + scopeName: string; + agentId: string; + stopReason: string; + isApiError: boolean; + gitBranch: string; +} + +export interface TranscriptParseState { + mainOffset: number; + subagentOffsets: Record; + openRequests: Record; // key: `${requestId}::${model}` + activeSkill: string; + branchCounts: Record; +} + +/** + * Build a fresh, empty parse state. + */ +export function createParseState(): TranscriptParseState { + return { + mainOffset: 0, + subagentOffsets: {}, + openRequests: {}, + activeSkill: '', + branchCounts: {}, + }; +} + +function getParseStatePath(sessionId: string): string { + return getCodemiePath('analytics', 'state', `${sessionId}.json`); +} + +/** + * Load the persisted parse state for a session. + * + * Never throws: a missing file, malformed JSON, or any other I/O failure all fall + * back to a fresh state via {@link createParseState}, since transcript parsing must + * keep going (as if starting fresh) rather than fail the whole run over stale/corrupt + * state on disk. + */ +export async function loadParseState(sessionId: string): Promise { + try { + const raw = await readFile(getParseStatePath(sessionId), 'utf-8'); + const parsed = JSON.parse(raw) as Partial; + + return { + mainOffset: parsed.mainOffset ?? 0, + subagentOffsets: parsed.subagentOffsets ?? {}, + openRequests: parsed.openRequests ?? {}, + activeSkill: parsed.activeSkill ?? '', + branchCounts: parsed.branchCounts ?? {}, + }; + } catch { + return createParseState(); + } +} + +/** + * Persist parse state for a session, creating the parent directory if needed. + * + * Unlike {@link loadParseState}, this does not swallow errors — a genuine write + * failure (disk full, permissions) propagates to the caller rather than silently + * discarding progress. + */ +export async function saveParseState(sessionId: string, state: TranscriptParseState): Promise { + const filePath = getParseStatePath(sessionId); + await mkdir(dirname(filePath), { recursive: true }); + await writeFile(filePath, JSON.stringify(state, null, 2), 'utf-8'); +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts new file mode 100644 index 000000000..52a0c5589 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts @@ -0,0 +1,150 @@ +/** + * `agent.session.summary` builder. + * + * Unlike `agent.usage.request`/`agent.subagent.usage`, this event is a running, mutable + * aggregate over an entire session, not a one-shot derivation from a single transcript line + * or file. The caller (a later task — the transcript-reader orchestrator) is responsible for + * accumulating a {@link SessionSummaryAccumulator} across the session's parsed transcript lines + * and tool-use/tool-result payloads, tracking `TranscriptParseState.branchCounts` via + * {@link updateBranchCounts} as `git_branch` changes per record, and running + * `extractNamedInvocations()` (`@/agents/plugins/claude/session/claude-named-invocations.js`) + * against the session's messages to get the {@link NamedInvocationCounts} this module's builder + * consumes. This module only aggregates/derives the final event shape from already-computed + * inputs — it never reads a transcript file or calls `extractNamedInvocations()` itself. + * + * `event_id`/`schema_version` are stamped later, daemon-side (see Task 1) — this builder's output + * carries only an explicit `type` field. + * + * Field-shape rulings (see spec.md's `agent.session.summary` section and this task's own plan + * entry for the full reasoning): + * - `models_used` is emitted as the full `acc.models` count map (not just a list of names) — + * preserves count information `primary_model` alone would discard, consistent with how + * `tool_calls`/`tool_errors`-style maps are emitted elsewhere in this stage. + * - `tool_calls`/`tool_errors` are flattened from `acc.toolCalls`'s combined + * `{ calls, errors }`-per-tool shape into two separate flat `Record` maps, + * matching `buildSubagentUsageEvent`'s (`./subagent-usage.ts`) already-established + * `tool_calls`/`tool_errors` output convention for the sibling `agent.subagent.usage` event. + * - `commands_in_order` is derived as `Object.keys(named.commandInvocations)` — the distinct + * command names in whatever iteration order the object naturally has. `commandInvocations` is + * a COUNT map, not an ordered sequence, so no true chronological invocation order is available + * anywhere in `NamedInvocationCounts`; this is a genuine mismatch between this field's name + * (which implies ordering) and the upstream data shape. Documented here rather than silently + * papered over with a fabricated ordering. + * - `title` has no identified source anywhere in this codebase or the external data-model doc + * (per spec.md's Open risks) — always emitted as a literal empty string, never fabricated. + * - `api_calls` (count of `agent.usage.request` records this session) is intentionally OMITTED + * from this builder's output: the plan's `buildSessionSummaryEvent` signature has no parameter + * for it, and neither `acc` nor any other input here carries a request count. It is left for + * the orchestrator (a later task) to merge in afterward, since only that caller has visibility + * into the full set of `agent.usage.request` records it has derived/forwarded this session. + * - `client_version`/`codemie_cli_version` are common fields stamped later via + * `mapHookRecords()` (see Tasks 1-3) — not this builder's responsibility either. + */ + +import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; + +export type { NamedInvocationCounts }; + +/** Running, mutable aggregate accumulated by the caller across one session's transcript. */ +export interface SessionSummaryAccumulator { + models: Record; + toolCalls: Record; + linesAdded: number; + linesRemoved: number; + filesChanged: Set; + filesWritten: Set; + compactionCount: number; +} + +/** + * Bump `counts[branch]` by 1, mutating `counts` in place (this is the caller-maintained + * `TranscriptParseState.branchCounts` map from `./parse-state.ts`). + * + * A falsy/empty `branch` is skipped — an unknown/missing branch shouldn't pollute the + * dominant-branch calculation ({@link branchDominant}). + */ +export function updateBranchCounts(counts: Record, branch: string): void { + if (!branch) { + return; + } + counts[branch] = (counts[branch] ?? 0) + 1; +} + +/** + * Return the key with the highest value in `counts`, or `''` when `counts` is empty. + * On a tie, the first-encountered key (in `Object.entries()` iteration order) wins. + */ +function maxKey(counts: Record): string { + let best = ''; + let bestValue = -Infinity; + + for (const [key, value] of Object.entries(counts)) { + if (value > bestValue) { + best = key; + bestValue = value; + } + } + + return best; +} + +/** The model with the highest count in `models`, or `''` when empty. */ +export function primaryModel(models: Record): string { + return maxKey(models); +} + +/** The branch with the highest count in `counts`, or `''` when empty. */ +export function branchDominant(counts: Record): string { + return maxKey(counts); +} + +/** + * Build the `agent.session.summary` event payload. + * + * `endedAt` is included as `ended_at` only when `phase === 'final'`; for `phase === 'incremental'` + * the key is omitted entirely (not merely `undefined`-valued). + */ +export function buildSessionSummaryEvent( + sessionId: string, + phase: 'incremental' | 'final', + acc: SessionSummaryAccumulator, + named: NamedInvocationCounts, + branchCounts: Record, + startedAt: string, + endedAt: string | undefined +): Record { + const toolCalls: Record = {}; + const toolErrors: Record = {}; + for (const [tool, counts] of Object.entries(acc.toolCalls)) { + toolCalls[tool] = counts.calls; + toolErrors[tool] = counts.errors; + } + + const event: Record = { + type: 'agent.session.summary', + session_id: sessionId, + phase, + models_used: acc.models, + primary_model: primaryModel(acc.models), + tool_calls: toolCalls, + tool_errors: toolErrors, + skills_used: named.skillInvocations, + commands_in_order: Object.keys(named.commandInvocations), + primary_command: maxKey(named.commandInvocations), + lines_added: acc.linesAdded, + lines_removed: acc.linesRemoved, + files_changed: Array.from(acc.filesChanged), + files_written: Array.from(acc.filesWritten), + compaction_count: acc.compactionCount, + branch_counts: branchCounts, + branch_dominant: branchDominant(branchCounts), + started_at: startedAt, + title: '', + }; + + if (phase === 'final') { + event.ended_at = endedAt; + } + + return event; +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts new file mode 100644 index 000000000..1a3417a19 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts @@ -0,0 +1,143 @@ +/** + * `agent.subagent.usage` discovery and event builder. + * + * Discovery (`findSubagentFiles`) follows the same path convention as the existing, + * `private`/unexported `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts:384` + * (`//subagents/agent-*.jsonl` + sibling `.meta.json`), but is a + * fresh, smaller implementation returning only the narrower {@link SubagentFile} shape this + * task's event needs — no `parentAgentId`/`requestShape`/`requestNonInteractive`. + * + * The event builder (`buildSubagentUsageEvent`) aggregates token/cache fields by summing an + * already-scoped `OpenUsageRequest[]` (Task 8's shape, reused — not redefined), and passes + * caller-built `tool_calls`/`tool_errors`/`skills_invoked` maps through verbatim: this module has + * no access to a subagent's own tool-use/tool-error/skill-invocation occurrences, only to the + * aggregates its caller already computed from that subagent's transcript. + * + * `description`/`workflow_run`/`worktree` have no identified source anywhere in this codebase + * (per spec.md's "Open risks") — always emitted as literal empty strings, never fabricated. + */ + +import { existsSync } from 'node:fs'; +import { readdir, readFile } from 'node:fs/promises'; +import { basename, dirname, join } from 'node:path'; +import type { OpenUsageRequest } from './parse-state.js'; + +export interface SubagentFile { + agentId: string; + filePath: string; + toolUseId?: string; + agentType?: string; + spawnDepth?: number; +} + +/** + * Discover subagent transcript files for a main transcript at `mainTranscriptPath`. + * + * Looks under `//subagents/` (where `sessionId` is `mainTranscriptPath`'s + * own basename, minus `.jsonl`) for `agent-*.jsonl` files, reading each one's sibling + * `.meta.json` sidecar (when present and parseable) for `toolUseId`/`agentType`/ + * `spawnDepth`. Never reads `mainTranscriptPath`'s own content — only its path is used to derive + * the subagents directory. + * + * Never throws: a missing subagents directory, an unreadable directory, or any other failure all + * resolve to `[]`. A missing or malformed per-agent `.meta.json` sidecar is likewise swallowed — + * that agent is still returned, just without the sidecar-derived fields. + */ +export async function findSubagentFiles(mainTranscriptPath: string): Promise { + try { + const parentDir = dirname(mainTranscriptPath); + const filename = basename(mainTranscriptPath); + const sessionId = filename.replace(/\.jsonl$/, ''); + const subagentsDir = join(parentDir, sessionId, 'subagents'); + + if (!existsSync(subagentsDir)) { + return []; + } + + const files = await readdir(subagentsDir); + const agentFiles = await Promise.all( + files + .filter((f) => f.startsWith('agent-') && f.endsWith('.jsonl')) + .map(async (f): Promise => { + const agentId = f.replace(/^agent-/, '').replace(/\.jsonl$/, ''); + const filePath = join(subagentsDir, f); + + let toolUseId: string | undefined; + let agentType: string | undefined; + let spawnDepth: number | undefined; + + try { + const metaRaw = JSON.parse( + await readFile(join(subagentsDir, f.replace(/\.jsonl$/, '.meta.json')), 'utf-8') + ) as Record; + if (typeof metaRaw.toolUseId === 'string') toolUseId = metaRaw.toolUseId; + if (typeof metaRaw.agentType === 'string') agentType = metaRaw.agentType; + if (typeof metaRaw.spawnDepth === 'number') spawnDepth = metaRaw.spawnDepth; + } catch { + // meta file absent or malformed — proceed without it. + } + + return { agentId, filePath, toolUseId, agentType, spawnDepth }; + }) + ); + + return agentFiles; + } catch { + return []; + } +} + +/** + * Build the `agent.subagent.usage` event payload for one subagent file. + * + * `usageRequests` is an array of {@link OpenUsageRequest} the caller has already parsed and + * scoped to this one subagent (`scope_kind: 'agent'`) — this function only sums it, it does not + * filter or scope it itself. `toolCalls`/`toolErrors`/`skillsInvoked` are likewise caller-built + * `Record` maps (keyed by tool/skill name) and are passed through verbatim. + * + * `started_at`/`duration_ms` are forwarded verbatim from the caller, which derives them from the + * subagent transcript's own first/last line timestamps — not this function's job. + * + * `spawn_depth` defaults to `0` when `file.spawnDepth` is absent (top-level subagents, whose + * sidecar omits the field — per spec.md's Open risks, not treated as an error). + * + * Carries its own explicit `type`, so `event_id`/`schema_version` are stamped later, daemon-side. + */ +export function buildSubagentUsageEvent( + sessionId: string, + file: SubagentFile, + usageRequests: OpenUsageRequest[], + toolCalls: Record, + toolErrors: Record, + skillsInvoked: Record, + startedAt: string, + durationMs: number +): Record { + const sum = (selector: (req: OpenUsageRequest) => number): number => + usageRequests.reduce((total, req) => total + selector(req), 0); + + return { + type: 'agent.subagent.usage', + session_id: sessionId, + agent_id: file.agentId, + tool_use_id: file.toolUseId ?? '', + agent_type: file.agentType ?? '', + spawn_depth: file.spawnDepth ?? 0, + description: '', + workflow_run: '', + worktree: '', + started_at: startedAt, + duration_ms: durationMs, + input_tokens: sum((r) => r.inputTokens), + cache_creation_5m_tokens: sum((r) => r.cacheCreation5mTokens), + cache_creation_1h_tokens: sum((r) => r.cacheCreation1hTokens), + cache_read_tokens: sum((r) => r.cacheReadTokens), + output_tokens: sum((r) => r.outputTokens), + web_search_requests: sum((r) => r.webSearchRequests), + web_fetch_requests: sum((r) => r.webFetchRequests), + api_calls: usageRequests.length, + tool_calls: toolCalls, + tool_errors: toolErrors, + skills_invoked: skillsInvoked, + }; +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts b/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts new file mode 100644 index 000000000..445968735 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts @@ -0,0 +1,57 @@ +import { open } from 'node:fs/promises'; + +const NEWLINE = 0x0a; + +export interface ReadNewLinesResult { + lines: string[]; + nextOffset: number; +} + +/** + * Incrementally read complete, newline-terminated lines appended to a + * transcript file since `fromOffset`. + * + * Mirrors the safe-cut rule used by + * `src/providers/plugins/sso/proxy/plugins/otlp-spool/spool-io.ts`'s + * `snapshotPendingHookRecords`: the read is cut at the last `\n` byte so a + * partially-written trailing line is never returned. Never throws — a + * missing file, a shrunk/rotated file, or a chunk with no complete line yet + * all resolve to the documented empty-result shape. + */ +export async function readNewLines( + filePath: string, + fromOffset: number +): Promise { + let handle; + try { + handle = await open(filePath, 'r'); + } catch { + return { lines: [], nextOffset: fromOffset }; // missing file => nothing new + } + + try { + const { size } = await handle.stat(); + if (size <= fromOffset) { + return { lines: [], nextOffset: fromOffset }; // nothing new (or file rotated/truncated) + } + + const buffer = Buffer.allocUnsafe(size - fromOffset); + const { bytesRead } = await handle.read(buffer, 0, buffer.length, fromOffset); + const chunk = buffer.subarray(0, bytesRead); + + const lastNewline = chunk.lastIndexOf(NEWLINE); + if (lastNewline < 0) { + return { lines: [], nextOffset: fromOffset }; // no complete line yet + } + + const complete = chunk.subarray(0, lastNewline + 1); + const lines = complete + .toString('utf-8') + .split('\n') + .filter((line) => line.trim().length > 0); + + return { lines, nextOffset: fromOffset + complete.length }; + } finally { + await handle.close(); + } +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts new file mode 100644 index 000000000..b20cccafd --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -0,0 +1,195 @@ +/** + * `agent.usage.request` extraction and merge. + * + * One record per Claude transcript JSONL line that carries `message.usage` — the same + * per-API-response usage shape the existing cost-reporting parser + * (`src/cli/commands/analytics/cost/usage-readers.ts`'s `ClaudeRawMessage`/ + * `extractClaudeUsageRecords`) already reads, adapted here to this task's + * {@link OpenUsageRequest} shape (Task 6, `parse-state.ts`) instead of that pipeline's + * `UsageRecord`. + * + * Claude Code can write more than one JSONL row for the same API response (progressive + * streaming chunks, or a later row that fills in `stop_reason` once the turn finishes), so + * callers parse every candidate line and merge same-identity records with + * {@link mergeUsageRequest} — this module trusts the caller to key records by + * `${requestId}::${model}` (see `parse-state.ts`'s `openRequests`) before merging; it does + * not itself check that two records it is asked to merge actually share that identity. + */ + +import { parseRoutingHeaders, type RoutingHeaderSource } from '@/utils/routing-headers.mjs'; +import { parseBackendModelName } from '@/utils/bedrock-pricing.mjs'; +import type { OpenUsageRequest } from './parse-state.js'; + +/** + * Loose shape of one transcript JSONL line, mirroring `usage-readers.ts`'s + * `ClaudeRawMessage` plus the additional fields this task's event needs + * (`message.stop_reason`, `message.usage`'s extra nested groups, and the + * top-level `gitBranch`/`isApiError`). Not exported — callers only see + * {@link parseUsageLine}'s `OpenUsageRequest | null` result. + */ +interface TranscriptUsageLine { + timestamp?: string; + gitBranch?: string; + isApiError?: boolean; + message?: RoutingHeaderSource & { + id?: string; + model?: string; + stop_reason?: string; + usage?: { + input_tokens?: number; + output_tokens?: number; + cache_read_input_tokens?: number; + cache_creation_input_tokens?: number; + service_tier?: string; + speed?: string; + inference_geo?: string; + cache_creation?: { + ephemeral_1h_input_tokens?: number; + ephemeral_5m_input_tokens?: number; + }; + server_tool_use?: { + web_search_requests?: number; + web_fetch_requests?: number; + }; + }; + }; +} + +/** + * Parse one transcript JSONL line into an {@link OpenUsageRequest}, or `null` when the line + * carries no `message.usage` block (not a billable API response — e.g. a plain user/system + * message) or is not valid JSON. + * + * `scopeKind`/`scopeName`/`agentId` are passed through verbatim from the caller, which already + * knows which transcript (main vs. a named skill context vs. a subagent transcript) `line` came + * from — this function has no way to derive that from the line itself. + */ +export function parseUsageLine( + line: string, + scopeKind: 'main' | 'skill' | 'agent', + scopeName: string, + agentId: string +): OpenUsageRequest | null { + let parsed: TranscriptUsageLine; + try { + parsed = JSON.parse(line) as TranscriptUsageLine; + } catch { + return null; + } + + const usage = parsed.message?.usage; + if (!usage) { + return null; + } + + // Same resolution chain the statusline and usage-readers.ts:188 already use: + // parseBackendModelName() (the raw LiteLLM backend id, when the proxy injected one) wins over + // the transcript's own literal `message.model`, since it reflects the actual billable backend + // model for a routed/capable-tier request. `modelRaw` keeps the literal, unresolved alias. + const modelRaw = String(parsed.message?.model ?? 'unknown'); + const model = parseBackendModelName(parsed.message) ?? parsed.message?.model ?? 'unknown'; + // Routing metadata itself is not part of OpenUsageRequest's shape, but parsing it mirrors the + // same pattern usage-readers.ts follows for this message object — kept as a documented no-op + // read (not stored) so a future task extending OpenUsageRequest with routing fields has a + // precedent to follow rather than re-deriving the call from scratch. + void parseRoutingHeaders(parsed.message); + + return { + requestId: String(parsed.message?.id ?? ''), + model: String(model), + modelRaw, + timestamp: String(parsed.timestamp ?? ''), + speed: String(usage.speed ?? ''), + inferenceGeo: String(usage.inference_geo ?? ''), + serviceTier: String(usage.service_tier ?? ''), + inputTokens: Number(usage.input_tokens ?? 0), + cacheCreation5mTokens: Number(usage.cache_creation?.ephemeral_5m_input_tokens ?? 0), + cacheCreation1hTokens: Number(usage.cache_creation?.ephemeral_1h_input_tokens ?? 0), + cacheReadTokens: Number(usage.cache_read_input_tokens ?? 0), + outputTokens: Number(usage.output_tokens ?? 0), + webSearchRequests: Number(usage.server_tool_use?.web_search_requests ?? 0), + webFetchRequests: Number(usage.server_tool_use?.web_fetch_requests ?? 0), + scopeKind, + scopeName, + agentId, + // Sibling of usage on message, not nested inside it. + stopReason: String(parsed.message?.stop_reason ?? ''), + // No confirmed source field for this on any sampled real transcript line (spec.md's own + // "Transcript field confidence gaps" flags it as unverified) — default false, and pick it up + // from a top-level `isApiError` boolean if a line ever carries one. + isApiError: Boolean(parsed.isApiError ?? false), + gitBranch: String(parsed.gitBranch ?? ''), + }; +} + +/** + * Merge two {@link OpenUsageRequest} records the caller has already identified as the same + * logical request (same `requestId`+`model` — this function does not verify that itself). + * Every numeric field takes the max of the two (a later streaming/finalizing row only ever adds + * usage, never subtracts it); every non-numeric field takes `b`'s value when non-empty, else + * falls back to `a`'s — so a later row that fills in a previously-empty field (e.g. + * `stop_reason` once the turn finishes) wins, while a later row that is missing a field `a` had + * does not blank it out. + * + * `isApiError` follows the same "non-empty b wins, else a" shape as the non-numeric fields: once + * true on either record, it stays true across the merge (losing a true→false "fix" would hide a + * real API error from aggregation). + * + * Returns a new object; neither `a` nor `b` is mutated. + */ +export function mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): OpenUsageRequest { + return { + requestId: b.requestId || a.requestId, + model: b.model || a.model, + modelRaw: b.modelRaw || a.modelRaw, + timestamp: b.timestamp || a.timestamp, + speed: b.speed || a.speed, + inferenceGeo: b.inferenceGeo || a.inferenceGeo, + serviceTier: b.serviceTier || a.serviceTier, + inputTokens: Math.max(a.inputTokens, b.inputTokens), + cacheCreation5mTokens: Math.max(a.cacheCreation5mTokens, b.cacheCreation5mTokens), + cacheCreation1hTokens: Math.max(a.cacheCreation1hTokens, b.cacheCreation1hTokens), + cacheReadTokens: Math.max(a.cacheReadTokens, b.cacheReadTokens), + outputTokens: Math.max(a.outputTokens, b.outputTokens), + webSearchRequests: Math.max(a.webSearchRequests, b.webSearchRequests), + webFetchRequests: Math.max(a.webFetchRequests, b.webFetchRequests), + scopeKind: b.scopeKind || a.scopeKind, + scopeName: b.scopeName || a.scopeName, + agentId: b.agentId || a.agentId, + stopReason: b.stopReason || a.stopReason, + isApiError: b.isApiError || a.isApiError, + gitBranch: b.gitBranch || a.gitBranch, + }; +} + +/** + * Build the `agent.usage.request` event payload for `req`. Carries its own explicit `type`, so + * the daemon-side `mapHookRecords()` (Task 1) stamps `event_id`/`schema_version` onto it later — + * this function deliberately does not set either. + */ +export function buildUsageRequestEvent(sessionId: string, req: OpenUsageRequest): Record { + return { + type: 'agent.usage.request', + session_id: sessionId, + request_id: req.requestId, + model_raw: req.modelRaw, + model: req.model, + speed: req.speed, + inference_geo: req.inferenceGeo, + service_tier: req.serviceTier, + input_tokens: req.inputTokens, + cache_creation_5m_tokens: req.cacheCreation5mTokens, + cache_creation_1h_tokens: req.cacheCreation1hTokens, + cache_read_tokens: req.cacheReadTokens, + output_tokens: req.outputTokens, + web_search_requests: req.webSearchRequests, + web_fetch_requests: req.webFetchRequests, + scope_kind: req.scopeKind, + scope_name: req.scopeName, + agent_id: req.agentId, + stop_reason: req.stopReason, + is_api_error: req.isApiError, + git_branch: req.gitBranch, + timestamp: req.timestamp, + }; +} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts new file mode 100644 index 000000000..7d659c935 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts @@ -0,0 +1,44 @@ +import { describe, it, expect } from 'vitest'; +import { computeEventId } from '../event-id.js'; + +describe('computeEventId', () => { + describe('existing hook event types', () => { + it('returns the byte-offset formula for an existing-event type', () => { + expect(computeEventId('agent.session.start', 'sid1', { byteOffset: 42 })).toBe( + 'sid1:agent.session.start:42' + ); + }); + + it('returns a different id for a different byte offset', () => { + const first = computeEventId('agent.session.stop', 'sid1', { byteOffset: 0 }); + const second = computeEventId('agent.session.stop', 'sid1', { byteOffset: 128 }); + expect(first).not.toBe(second); + expect(first).toBe('sid1:agent.session.stop:0'); + expect(second).toBe('sid1:agent.session.stop:128'); + }); + }); + + describe('agent.usage.request', () => { + it('returns the request/model-keyed formula', () => { + expect( + computeEventId('agent.usage.request', 'sid1', { request_id: 'req1', model: 'gpt-4' }) + ).toBe('sid1:agent.usage.request:req1:gpt-4'); + }); + }); + + describe('agent.subagent.usage', () => { + it('returns the tool_use_id-keyed formula', () => { + expect(computeEventId('agent.subagent.usage', 'sid1', { tool_use_id: 'tu1' })).toBe( + 'sid1:agent.subagent.usage:tu1' + ); + }); + }); + + describe('agent.session.summary', () => { + it('returns the phase-keyed formula', () => { + expect(computeEventId('agent.session.summary', 'sid1', { phase: 'end' })).toBe( + 'sid1:agent.session.summary:end' + ); + }); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts new file mode 100644 index 000000000..48e0871c8 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -0,0 +1,213 @@ +import { describe, it, expect } from 'vitest'; + +interface MappedRecord { + type: string; + session_id: string; + schema_version: number; + event_id: string; + codemie_cli_version: string; + story_id?: string; + story_source?: string; + prompt_body?: string; +} + +function buildHookRecord(hookEventName: string, sessionId: string, extra: Record = {}): string { + return JSON.stringify({ + agentName: 'claude', + raw: JSON.stringify({ + hook_event_name: hookEventName, + session_id: sessionId, + cwd: '', + ...extra, + }), + timestamp: Date.now(), + }); +} + +describe('mapHookRecords', () => { + it('stamps schema_version, event_id, and codemie_cli_version on every mapped record', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: 'user@example.com', + git: {}, + }; + + const record1 = buildHookRecord('SessionStart', 'sid1'); + const record2 = buildHookRecord('Stop', 'sid1'); + + const payload = await mapHookRecords([record1, record2], ctx, 0); + const lines = payload.ndjson + .trim() + .split('\n') + .map((line) => JSON.parse(line) as MappedRecord); + + expect(lines).toHaveLength(2); + expect(lines[0].schema_version).toBe(2); + expect(lines[1].schema_version).toBe(2); + expect(lines[0].event_id).not.toBe(lines[1].event_id); + expect(typeof lines[0].codemie_cli_version).toBe('string'); + expect(lines[0].codemie_cli_version.length).toBeGreaterThan(0); + expect(lines[0].codemie_cli_version).toBe(lines[1].codemie_cli_version); + }); + + it('derives event_id from the running byte offset seeded by startOffset', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + const record = buildHookRecord('SessionStart', 'sid1'); + + const payloadAtZero = await mapHookRecords([record], ctx, 0); + const payloadAtOffset = await mapHookRecords([record], { ...ctx, git: {} }, 500); + + const lineAtZero = JSON.parse(payloadAtZero.ndjson.trim()) as MappedRecord; + const lineAtOffset = JSON.parse(payloadAtOffset.ndjson.trim()) as MappedRecord; + + expect(lineAtZero.event_id).toBe('sid1:agent.session.start:0'); + expect(lineAtOffset.event_id).toBe('sid1:agent.session.start:500'); + }); + + it('prefers an explicit hookEvent.type over the HOOK_EVENT_TYPE_MAP lookup', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + // PostToolUse normally maps to 'agent.tool.end', but an explicit `type` + // field on the raw hook payload (as a later-task synthetic record would + // carry) must win. + const record = buildHookRecord('PostToolUse', 'sid1', { type: 'agent.custom.synthetic' }); + + const payload = await mapHookRecords([record], ctx, 0); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.type).toBe('agent.custom.synthetic'); + }); + + it('leaves existing-event type resolution unchanged when type is absent', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + const record = buildHookRecord('PostToolUse', 'sid1'); + + const payload = await mapHookRecords([record], ctx, 0); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.type).toBe('agent.tool.end'); + }); + + it( + "overrides a UserPromptSubmit record's story_id/story_source with a prompt marker " + + 'even when the per-tick branch tier would otherwise resolve to a different ticket, ' + + 'and never leaks the raw prompt text onto the emitted record', + async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + // Branch carries a DIFFERENT ticket than the prompt marker, so this + // test proves the marker tier wins over the already-cached branch tier. + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: { branch: 'feature/ABC-1-unrelated-branch' }, + }; + + // Longer than MAX_PROMPT_CHARS (200) so every bounded copy on the + // mapped record is truncated and none of them equals this full text — + // which is what actually proves "the raw prompt is never present" + // rather than merely proving a short prompt survives truncation whole. + const rawPrompt = + `Please implement this feature. story: EPMCDME-999 is the ticket to reference. ` + + 'x'.repeat(200) + + ' end-of-prompt-marker-that-must-not-appear-anywhere-in-the-output'; + const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); + + const payload = await mapHookRecords([record], ctx, 0); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.story_id).toBe('EPMCDME-999'); + expect(line.story_source).toBe('marker'); + + // The raw prompt text must never appear verbatim anywhere on the + // emitted record — only the truncated `prompt_body` and the resolved + // short `story_id` string are allowed to carry prompt-derived content. + const serialized = JSON.stringify(line); + expect(serialized).not.toContain(rawPrompt); + expect(serialized).not.toContain('end-of-prompt-marker-that-must-not-appear-anywhere-in-the-output'); + expect(line.prompt_body).toBe(rawPrompt.slice(0, 200)); + } + ); + + it( + 'falls back to the per-tick branch result for a UserPromptSubmit record whose prompt ' + + 'has no marker and no bare ticket mention', + async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: { branch: 'feature/epmcdme-15301-foo' }, + }; + + const rawPrompt = 'please just fix the thing, no ticket reference here'; + const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); + + const payload = await mapHookRecords([record], ctx, 0); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.story_id).toBe('EPMCDME-15301'); + expect(line.story_source).toBe('branch'); + } + ); + + it( + 'falls back to the mention tier for a UserPromptSubmit record whose prompt has a bare ' + + 'ticket mention and the branch carries no ticket', + async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: { branch: 'just-some-branch-name' }, + }; + + const rawPrompt = 'can you look into ABC-42 when you get a chance'; + const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); + + const payload = await mapHookRecords([record], ctx, 0); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.story_id).toBe('ABC-42'); + expect(line.story_source).toBe('mention'); + } + ); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts new file mode 100644 index 000000000..796fffe99 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts @@ -0,0 +1,72 @@ +import { describe, it, expect, vi, afterEach } from 'vitest'; +import * as os from 'node:os'; +import type { JWTCredentials } from '@/providers/core/types.js'; + +/** Builds a minimal unsigned JWT with the given payload claims. */ +function makeJwt(payload: Record): string { + const header = Buffer.from(JSON.stringify({ alg: 'none', typ: 'JWT' })).toString('base64url'); + const body = Buffer.from(JSON.stringify(payload)).toString('base64url'); + return `${header}.${body}.sig`; +} + +describe('resolveIdentity', () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it('resolves from the jwt tier when the token carries a valid email claim', async () => { + const { resolveIdentity } = await import('../identity.js'); + + const credentials: JWTCredentials = { + token: makeJwt({ email: 'dev@example.com' }), + apiUrl: 'https://codemie.example.com', + }; + + const result = await resolveIdentity(credentials, 'C:/some/project'); + + expect(result).toEqual({ developerName: 'dev@example.com', identitySource: 'jwt' }); + }); + + it('falls through to git config user.email when the jwt tier is empty', async () => { + const execModule = await import('@/utils/exec.js'); + const execSpy = vi.spyOn(execModule, 'exec').mockImplementation(async (_command, args) => { + if (args?.[0] === 'config' && args?.[1] === 'user.email') { + return { code: 0, stdout: 'git-user@example.com', stderr: '', signal: null }; + } + return { code: 1, stdout: '', stderr: '', signal: null }; + }); + + const { resolveIdentity } = await import('../identity.js'); + + // Empty token: isJWTCredentials() still matches the shape, but decodeJwtClaims + // yields no usable email, so the jwt tier is a miss. + const credentials: JWTCredentials = { token: '', apiUrl: '' }; + + const result = await resolveIdentity(credentials, 'C:/some/project'); + + expect(result).toEqual({ developerName: 'git-user@example.com', identitySource: 'git' }); + expect(execSpy).toHaveBeenCalledWith('git', ['config', 'user.email'], { cwd: 'C:/some/project' }); + }); + + it('falls through to os.userInfo().username when every other tier is empty', async () => { + const execModule = await import('@/utils/exec.js'); + vi.spyOn(execModule, 'exec').mockResolvedValue({ code: 1, stdout: '', stderr: '', signal: null }); + + const configModule = await import('@/utils/config.js'); + vi.spyOn(configModule.ConfigLoader, 'loadMultiProviderConfig').mockResolvedValue({ + version: 2, + activeProfile: 'default', + profiles: {}, + }); + + const { resolveIdentity } = await import('../identity.js'); + + const credentials: JWTCredentials = { token: '', apiUrl: '' }; + + const result = await resolveIdentity(credentials, 'C:/some/project'); + + expect(result.identitySource).toBe('os'); + expect(result.developerName).toBe(os.userInfo().username); + expect(result.developerName.length).toBeGreaterThan(0); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts new file mode 100644 index 000000000..61c7948bf --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts @@ -0,0 +1,218 @@ +import { describe, it, expect, afterEach } from 'vitest'; +import { mkdtemp, mkdir, writeFile, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; + +const ENV_KEY = 'SDLC_ANALYTICS_STORY_ID'; + +async function makeTempProjectDir(): Promise { + return mkdtemp(join(tmpdir(), 'otlp-story-resolver-')); +} + +async function writeAnalyticsLocalJson(cwd: string, data: Record): Promise { + const dir = join(cwd, '.claude'); + await mkdir(dir, { recursive: true }); + await writeFile(join(dir, 'analytics.local.json'), JSON.stringify(data), 'utf-8'); +} + +describe('story-resolver', () => { + const tempDirs: string[] = []; + + afterEach(async () => { + delete process.env[ENV_KEY]; + for (const dir of tempDirs.splice(0)) { + await rm(dir, { recursive: true, force: true }); + } + }); + + describe('resolveExplicitStory', () => { + it('prefers the env var over the file when both are set', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + await writeAnalyticsLocalJson(cwd, { storyId: 'FILE-1' }); + process.env[ENV_KEY] = 'ENV-1'; + + const result = await resolveExplicitStory(cwd); + + expect(result).toEqual({ storyId: 'ENV-1', storySource: 'explicit' }); + }); + + it('falls back to the analytics.local.json file when the env var is unset', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + await writeAnalyticsLocalJson(cwd, { storyId: 'FILE-1' }); + delete process.env[ENV_KEY]; + + const result = await resolveExplicitStory(cwd); + + expect(result).toEqual({ storyId: 'FILE-1', storySource: 'explicit' }); + }); + + it('returns null when neither the env var nor the file is set', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + delete process.env[ENV_KEY]; + + const result = await resolveExplicitStory(cwd); + + expect(result).toBeNull(); + }); + + it('returns null when the file is malformed JSON (never throws)', async () => { + const { resolveExplicitStory } = await import('../story-resolver.js'); + + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + const dir = join(cwd, '.claude'); + await mkdir(dir, { recursive: true }); + await writeFile(join(dir, 'analytics.local.json'), '{not valid json', 'utf-8'); + delete process.env[ENV_KEY]; + + const result = await resolveExplicitStory(cwd); + + expect(result).toBeNull(); + }); + }); + + describe('resolveBranchStory', () => { + it("extracts EPMCDME-15301 from 'feature/epmcdme-15301-foo' uppercased", async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('feature/epmcdme-15301-foo'); + + expect(result).toEqual({ storyId: 'EPMCDME-15301', storySource: 'branch' }); + }); + + it('returns null for a branch with no ticket-shaped substring', async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('just-some-branch-name'); + + expect(result).toBeNull(); + }); + + it( + 'returns null for a leading-digit identifier where the negative lookbehind blocks the only possible match start (verified: TICKET_RE requires the match to start on a letter, and the lookbehind forbids an alphanumeric char immediately before that start)', + async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('1ABC-123'); + + expect(result).toBeNull(); + } + ); + + it( + 'matches ABC-123 inside "ABC-123X" (verified actual regex behavior: the trailing (?!\\d) lookahead only blocks a FOLLOWING DIGIT, not a following letter, so "ABC-123X" is NOT a word-boundary case the literal regex rejects)', + async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const result = resolveBranchStory('ABC-123X'); + + expect(result).toEqual({ storyId: 'ABC-123', storySource: 'branch' }); + } + ); + + it('does not corrupt matching across repeated calls (fresh RegExp per call, no shared lastIndex state)', async () => { + const { resolveBranchStory } = await import('../story-resolver.js'); + + const first = resolveBranchStory('feature/epmcdme-15301-foo'); + const second = resolveBranchStory('feature/epmcdme-15301-foo'); + + expect(first).toEqual({ storyId: 'EPMCDME-15301', storySource: 'branch' }); + expect(second).toEqual({ storyId: 'EPMCDME-15301', storySource: 'branch' }); + }); + }); + + describe('resolveMarkerStory', () => { + it("matches the 'story: X' marker shape, uppercased", async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('story: EPMCDME-999'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it("matches the 'ticket #X' marker shape, uppercased", async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('ticket #EPMCDME-999'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it('matches the marker word case-insensitively (STORY:) and uppercases a lowercase id', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('STORY: epmcdme-999'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it('matches a marker embedded in a longer prompt', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('please fix the bug, story: EPMCDME-999, thanks'); + + expect(result).toEqual({ storyId: 'EPMCDME-999', storySource: 'marker' }); + }); + + it('returns null when no marker phrase is present', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory('just fix the bug please, no ticket mentioned'); + + expect(result).toBeNull(); + }); + + it('returns null for empty prompt text', async () => { + const { resolveMarkerStory } = await import('../story-resolver.js'); + + const result = resolveMarkerStory(''); + + expect(result).toBeNull(); + }); + }); + + describe('resolveMentionStory', () => { + it('matches a bare ticket-shaped mention anywhere in the text, uppercased', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const result = resolveMentionStory('can you look into ABC-42 when you get a chance'); + + expect(result).toEqual({ storyId: 'ABC-42', storySource: 'mention' }); + }); + + it('returns null when no ticket-shaped substring exists', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const result = resolveMentionStory('no ticket here, just a plain request'); + + expect(result).toBeNull(); + }); + + it('returns null for empty prompt text', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const result = resolveMentionStory(''); + + expect(result).toBeNull(); + }); + + it('does not corrupt matching across repeated calls (fresh RegExp per call)', async () => { + const { resolveMentionStory } = await import('../story-resolver.js'); + + const first = resolveMentionStory('ping on ABC-42 please'); + const second = resolveMentionStory('ping on ABC-42 please'); + + expect(first).toEqual({ storyId: 'ABC-42', storySource: 'mention' }); + expect(second).toEqual({ storyId: 'ABC-42', storySource: 'mention' }); + }); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts new file mode 100644 index 000000000..5e3e27a97 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts @@ -0,0 +1,28 @@ +/** + * Pure, deterministic `event_id` derivation for analytics events. + */ +export function computeEventId( + type: string, + sessionId: string, + fields: Record +): string { + switch (type) { + case 'agent.usage.request': { + const requestId = String(fields['request_id'] ?? ''); + const model = String(fields['model'] ?? ''); + return `${sessionId}:agent.usage.request:${requestId}:${model}`; + } + case 'agent.subagent.usage': { + const toolUseId = String(fields['tool_use_id'] ?? ''); + return `${sessionId}:agent.subagent.usage:${toolUseId}`; + } + case 'agent.session.summary': { + const phase = String(fields['phase'] ?? ''); + return `${sessionId}:agent.session.summary:${phase}`; + } + default: { + const byteOffset = String(fields['byteOffset'] ?? ''); + return `${sessionId}:${type}:${byteOffset}`; + } + } +} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 358e3e544..7ae58d206 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -1,5 +1,8 @@ +import { readFileSync } from 'node:fs'; +import { join } from 'node:path'; import { logger } from '@/utils/logger.js'; import { sanitizeLogArgs } from '@/utils/security.js'; +import { getDirname } from '@/utils/paths.js'; import type { SSOCredentials, JWTCredentials } from '../../../../../core/types.js'; import { isSSOCredentials, isJWTCredentials } from '../../../../../core/types.js'; import { buildAuthHeaders } from '../../../../../core/codemie-auth-helpers.js'; @@ -12,6 +15,27 @@ import { import { OtlpHookSpoolData } from '../otlp.plugin.js'; import { snapshotPendingBytes, snapshotPendingHookRecords } from './spool-io.js'; import { areCredentialsStale, markCredentialsStale } from './auth-state.js'; +import { computeEventId } from './event-id.js'; +import { decodeJwtClaims, resolveIdentity, type IdentitySource } from './identity.js'; +import { + resolveExplicitStory, + resolveBranchStory, + resolveMarkerStory, + resolveMentionStory, +} from './story-resolver.js'; + +/** This package's own `version` from the repo-root `package.json`, read once at import time. */ +function loadCodemieCliVersion(): string { + try { + const packageJsonPath = join(getDirname(import.meta.url), '../../../../../../../package.json'); + const packageJson = JSON.parse(readFileSync(packageJsonPath, 'utf-8')) as { version?: string }; + return packageJson.version ?? ''; + } catch { + return ''; + } +} + +const CODEMIE_CLI_VERSION = loadCodemieCliVersion(); const HOOK_EVENT_TYPE_MAP: Record = { SessionStart: 'agent.session.start', @@ -47,25 +71,14 @@ interface ForwardContext { userEmail: string; /** Per-session git info cache, resolved lazily from the first hook `cwd`. */ git: { branch?: string; remote?: string }; + /** Per-session developer-identity cache, resolved once */ + identity?: { developerName?: string; identitySource?: IdentitySource }; + /** Per-tick story-id cache, resolved once per forward tick. */ + story?: { storyId?: string; storySource?: 'explicit' | 'branch' | '' }; } /* ------------------------------------------------------------------ auth --- */ -function decodeJwtClaims(token: string): Record { - const parts = token.split('.'); - if (parts.length < 2) { - return {}; - } - try { - return JSON.parse(Buffer.from(parts[1], 'base64url').toString('utf-8')) as Record< - string, - unknown - >; - } catch { - return {}; - } -} - function resolveUserEmail(credentials: SSOCredentials | JWTCredentials): string { if (isJWTCredentials(credentials)) { const claims = decodeJwtClaims(credentials.token); @@ -171,6 +184,10 @@ async function send( /* --------------------------------------------------------------- mapping --- */ function hookEventType(hookName: string, event: Record): string { + const explicitType = event['type']; + if (typeof explicitType === 'string' && explicitType.length > 0) { + return explicitType; + } if (hookName === 'PreToolUse') { return event['input'] && (event['input'] as Record)['denied'] ? 'agent.tool.denied' @@ -187,12 +204,12 @@ function boundedText(value: unknown, maxChars: number): string { typeof value === 'string' ? value : (() => { - try { - return JSON.stringify(value) ?? String(value); - } catch { - return String(value); - } - })(); + try { + return JSON.stringify(value) ?? String(value); + } catch { + return String(value); + } + })(); return text.slice(0, maxChars); } @@ -230,18 +247,91 @@ async function resolveGitInfo(ctx: ForwardContext, cwd: string): Promise { } } +/** + * Per-record story-id override for `UserPromptSubmit` hook events only, + * layered on top of the per-tick `resolveStoryOnce()` cache in + * `ctx.story` (explicit/branch/''). Priority order across the full chain is + * explicit -> marker -> branch -> mention: + * + * 1. If the per-tick cache already resolved to `'explicit'`, that is the + * highest-priority result and wins outright. + * 2. Otherwise, try the marker tier (`story: X` / `ticket #X`) against this + * record's OWN prompt text — it sits above branch in priority. + * 3. Otherwise, if the per-tick cache resolved to `'branch'`, that wins (it + * is already correctly placed between marker and mention). + * 4. Otherwise, try the mention tier (bare ticket-shaped text) — the + * lowest-priority tier. + * 5. Otherwise, empty. + * + * Computed fresh per record and never mutates `ctx.story`: other records in + * the same batch still need that shared per-tick cache untouched. + */ +function resolvePromptStory( + ctx: ForwardContext, + rawPrompt: string +): { storyId: string; storySource: string } { + if (ctx.story?.storySource === 'explicit') { + return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; + } + + const marker = resolveMarkerStory(rawPrompt); + if (marker) { + return { storyId: marker.storyId, storySource: marker.storySource }; + } + + if (ctx.story?.storySource === 'branch') { + return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; + } + + const mention = resolveMentionStory(rawPrompt); + if (mention) { + return { storyId: mention.storyId, storySource: mention.storySource }; + } + + return { storyId: '', storySource: '' }; +} + +async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promise { + if (ctx.story?.storyId !== undefined) return; + + const explicit = await resolveExplicitStory(cwd); + const resolved = explicit ?? resolveBranchStory(ctx.git.branch ?? ''); + + ctx.story = resolved + ? { storyId: resolved.storyId, storySource: resolved.storySource } + : { storyId: '', storySource: '' }; +} + +async function resolveIdentityOnce(ctx: ForwardContext, cwd: string): Promise { + if (!ctx.identity) ctx.identity = {}; + if (ctx.identity.developerName !== undefined) return; + const { developerName, identitySource } = await resolveIdentity(ctx.credentials, cwd); + ctx.identity.developerName = developerName; + ctx.identity.identitySource = identitySource; +} + interface HookPayload { ndjson: string; containsSessionEnd: boolean; malformed: number; } -async function mapHookRecords(records: string[], ctx: ForwardContext): Promise { +export async function mapHookRecords( + records: string[], + ctx: ForwardContext, + startOffset: number +): Promise { const mapped: string[] = []; let containsSessionEnd = false; let malformed = 0; + let offset = startOffset; for (const record of records) { + // Every record occupied `byteLength(record) + 1` bytes in the spool file + // (the trailing newline `snapshotPendingHookRecords` already stripped). + const byteOffset = offset; + offset += Buffer.byteLength(record, 'utf-8') + 1; + let spoolData: OtlpHookSpoolData; let hookEvent: Record; try { @@ -261,22 +351,50 @@ async function mapHookRecords(records: string[], ctx: ForwardContext): Promise 0) { logger.debug( '[otlp-forwarder] skipped malformed hook records', diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts new file mode 100644 index 000000000..6166f405d --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts @@ -0,0 +1,153 @@ +import { userInfo } from 'node:os'; +import type { SSOCredentials, JWTCredentials } from '@/providers/core/types.js'; +import { isSSOCredentials, isJWTCredentials } from '@/providers/core/types.js'; + +export type IdentitySource = 'jwt' | 'git' | 'codemie_cli' | 'claude_account' | 'os' | ''; + +export interface ResolvedIdentity { + developerName: string; + identitySource: IdentitySource; +} + +/** + * Decode a JWT's payload segment without verifying its signature. Analytics + * identity resolution only reads claims already trusted by the caller + * (the token backing the active session) — it never performs auth. Returns + * `{}` for a malformed/non-JWT string instead of throwing. + */ +export function decodeJwtClaims(token: string): Record { + const parts = token.split('.'); + if (parts.length < 2) return {}; + try { + return JSON.parse(Buffer.from(parts[1], 'base64url').toString('utf-8')) as Record< + string, + unknown + >; + } catch { + return {}; + } +} + +/** + * Tier 1 — jwt: pull `email` straight off the JWT credential's claims, or off + * the SSO session's `codemie_access_token` cookie claims (`email`, falling + * back to `preferred_username`). + */ +function resolveJwtIdentity(credentials: SSOCredentials | JWTCredentials): string { + if (isJWTCredentials(credentials)) { + const claims = decodeJwtClaims(credentials.token); + if (typeof claims['email'] === 'string' && claims['email']) return claims['email']; + } + if (isSSOCredentials(credentials)) { + const accessToken = credentials.cookies['codemie_access_token']; + if (accessToken) { + const claims = decodeJwtClaims(accessToken); + const email = claims['email'] ?? claims['preferred_username']; + if (typeof email === 'string' && email) return email; + } + } + return ''; +} + +/** + * Tier 2 — git: `git config user.email`, falling back to `git config + * user.name` when the repo has no email configured. A non-git directory (or + * any exec failure) resolves to `''` so the chain falls through — never + * throws. + */ +async function resolveGitIdentity(cwd: string): Promise { + if (!cwd) return ''; + try { + const { exec } = await import('@/utils/exec.js'); + const emailResult = await exec('git', ['config', 'user.email'], { cwd }); + if (emailResult.code === 0 && emailResult.stdout.trim()) { + return emailResult.stdout.trim(); + } + const nameResult = await exec('git', ['config', 'user.name'], { cwd }); + if (nameResult.code === 0 && nameResult.stdout.trim()) { + return nameResult.stdout.trim(); + } + } catch { + /* best-effort */ + } + return ''; +} + +/** + * Tier 3 — codemie_cli: the `userEmail` persisted on the global CodeMie CLI + * config (set via `ConfigLoader.saveUserEmail()`). `userEmail` lives on + * `MultiProviderConfig`, not on the merged `CodeMieConfigOptions` that + * `ConfigLoader.load()` returns, so this reads the multi-provider config + * directly via `loadMultiProviderConfig()` — a global lookup, hence no `cwd` + * dependency. Any load failure (missing/unreadable/malformed config) + * resolves to `''`. + */ +async function resolveCodemieCliIdentity(): Promise { + try { + const { ConfigLoader } = await import('@/utils/config.js'); + const config = await ConfigLoader.loadMultiProviderConfig(); + if (config.userEmail) return config.userEmail; + } catch { + /* best-effort */ + } + return ''; +} + +/** + * Tier 4 — claude_account: best-effort lookup of a locally authenticated + * Claude account identity. Documented limitation: no such lookup exists + * anywhere in this repo today (no stored Claude account email/id to read), + * so this tier always yields `''` in practice and the chain falls through to + * `os`. Kept as its own tier/function so a real lookup can be dropped in here + * later without touching the rest of the chain. + */ +async function resolveClaudeAccount(): Promise { + return ''; +} + +/** + * Tier 5 — os: the OS-reported username for the daemon process. Practically + * never empty, but guarded anyway since some sandboxed environments can make + * `os.userInfo()` throw. + */ +function resolveOsIdentity(): string { + try { + return userInfo().username || ''; + } catch { + return ''; + } +} + +/** + * Resolve a developer identity for analytics stamping, trying each tier in + * order and returning the first non-empty result: + * + * jwt -> git -> codemie_cli -> claude_account -> os + * + * Never throws — every tier swallows its own failures internally. + */ +export async function resolveIdentity( + credentials: SSOCredentials | JWTCredentials, + cwd: string +): Promise { + try { + const jwtIdentity = resolveJwtIdentity(credentials); + if (jwtIdentity) return { developerName: jwtIdentity, identitySource: 'jwt' }; + + const gitIdentity = await resolveGitIdentity(cwd); + if (gitIdentity) return { developerName: gitIdentity, identitySource: 'git' }; + + const cliIdentity = await resolveCodemieCliIdentity(); + if (cliIdentity) return { developerName: cliIdentity, identitySource: 'codemie_cli' }; + + const claudeAccount = await resolveClaudeAccount(); + if (claudeAccount) return { developerName: claudeAccount, identitySource: 'claude_account' }; + + const osIdentity = resolveOsIdentity(); + if (osIdentity) return { developerName: osIdentity, identitySource: 'os' }; + + return { developerName: '', identitySource: '' }; + } catch { + return { developerName: '', identitySource: '' }; + } +} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts new file mode 100644 index 000000000..6f3b64ab2 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts @@ -0,0 +1,129 @@ +import { readFile } from 'node:fs/promises'; +import { join } from 'node:path'; + +/** + * Shared ticket-id pattern used by every story-id tier that scans free text + * (branch names today; marker/mention text in a later task). Carries the + * global (`g`) flag, so a stateful `.test()`/`.exec()` on THIS SAME instance + * across repeated calls would corrupt `lastIndex` and silently skip matches + * on the next call. Every consumer in this module therefore either builds a + * fresh `RegExp` from `TICKET_RE.source`/`TICKET_RE.flags` (used here via + * `String.prototype.match()`, which — called on a freshly constructed regex + * — reads all matches once and does not leave mutated `lastIndex` state + * behind for the next caller) before each match, rather than reusing this + * exported instance's `lastIndex` across calls. + */ +export const TICKET_RE = /(?/.claude/analytics.local.json`. + * Read-only — this never writes that file. Swallows every failure (missing + * file, malformed JSON, permission error) and resolves to `null` instead of + * throwing. The resolved `storyId` is taken verbatim from its source (env or + * file) and is NOT upper-cased, unlike the regex-derived `resolveBranchStory` + * tier: a value a user/config explicitly supplied is already exact, whereas + * free text scanned by a case-insensitive regex needs normalizing. + */ +export async function resolveExplicitStory(cwd: string): Promise { + const envStoryId = process.env['SDLC_ANALYTICS_STORY_ID']; + if (envStoryId) { + return { storyId: envStoryId, storySource: 'explicit' }; + } + + try { + const filePath = join(cwd, '.claude', 'analytics.local.json'); + const content = await readFile(filePath, 'utf-8'); + const parsed = JSON.parse(content) as AnalyticsLocalConfig; + if (typeof parsed.storyId === 'string' && parsed.storyId.length > 0) { + return { storyId: parsed.storyId, storySource: 'explicit' }; + } + } catch { + /* missing file, malformed JSON, permission error: fall through to null */ + } + + return null; +} + +/** + * Branch story-id tier: the first `TICKET_RE` match found anywhere in the + * branch name, upper-cased. Returns `null` when the branch carries no + * ticket-shaped substring. + */ +export function resolveBranchStory(branch: string): BranchStoryResult | null { + if (!branch) return null; + + // Fresh RegExp per call: avoids reusing TICKET_RE's own `lastIndex` across + // invocations (the classic stateful-global-regex-in-a-loop bug). + const matches = branch.match(new RegExp(TICKET_RE.source, TICKET_RE.flags)); + if (!matches || matches.length === 0) return null; + + return { storyId: matches[0].toUpperCase(), storySource: 'branch' }; +} + +/** + * Marker story-id tier: an explicit `story: X` / `ticket #X` phrase found + * anywhere in prompt text, case-insensitive on the marker word, upper-cased + * on return. Returns `null` when no marker phrase is present (including an + * empty/falsy `promptText`). + */ +export function resolveMarkerStory(promptText: string): MarkerStoryResult | null { + if (!promptText) return null; + + const match = promptText.match(MARKER_RE); + if (!match || !match[1]) return null; + + return { storyId: match[1].toUpperCase(), storySource: 'marker' }; +} + +/** + * Mention story-id tier: the first bare `TICKET_RE` match found anywhere in + * prompt text, upper-cased. This is the lowest-priority tier — it only + * applies when no explicit/marker/branch tier already resolved a story. + * Returns `null` when no ticket-shaped substring exists (including an + * empty/falsy `promptText`). + */ +export function resolveMentionStory(promptText: string): MentionStoryResult | null { + if (!promptText) return null; + + // Fresh RegExp per call: avoids reusing TICKET_RE's own `lastIndex` across + // invocations (the classic stateful-global-regex-in-a-loop bug). + const matches = promptText.match(new RegExp(TICKET_RE.source, TICKET_RE.flags)); + if (!matches || matches.length === 0) return null; + + return { storyId: matches[0].toUpperCase(), storySource: 'mention' }; +} From 2d70843187a0c5b9d2268e74dd9bdcde4c203a02 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Mon, 5 Oct 2026 15:17:10 +0300 Subject: [PATCH 02/35] refactor(proxy): fix issues found during code review --- .../code-review-final.json | 122 +++++++---- .../spec.md | 7 + .../__tests__/claude-code-otlp.plugin.test.ts | 159 +++++++++++++- .../claude-code-otlp.plugin.ts | 15 +- .../transcript/__tests__/orchestrator.test.ts | 42 ++++ .../transcript/__tests__/parse-state.test.ts | 15 ++ .../__tests__/transcript-reader.test.ts | 18 ++ .../__tests__/usage-request.test.ts | 13 ++ .../transcript/orchestrator.ts | 202 ++++++++++-------- .../transcript/parse-state.ts | 63 +++++- .../transcript/transcript-reader.ts | 16 +- .../transcript/usage-request.ts | 10 +- .../otlp-spool/__tests__/event-id.test.ts | 12 ++ .../otlp-spool/__tests__/forwarder.test.ts | 122 ++++++++++- .../sso/proxy/plugins/otlp-spool/event-id.ts | 3 +- .../plugins/otlp-spool/forward-context.ts | 119 +++++++++++ .../sso/proxy/plugins/otlp-spool/forwarder.ts | 119 ++--------- 17 files changed, 812 insertions(+), 245 deletions(-) create mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json index 59a1a254c..2350323b9 100644 --- a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json @@ -1,41 +1,41 @@ { - "decision": "request-changes", - "rationale": "All four lenses and the standards audit ran cleanly against the approved spec; triage surfaced 23 blocking findings, including 6 decision_needed items (skill-scope attribution, lines_added/lines_removed sourcing, the subagent description field, commands_in_order ordering, started_at/ended_at sourcing, and the missing identity-derivation security review) and a security-practices CRITICAL violation (unreviewed jwt/git/os developer-identity chain). 0 findings deferred (nothing was judged pre-existing/out-of-scope). 6 blind/edge-case findings were dismissed below the blocking floor: silent version-read fallback matching existing codebase convention, an unbounded-but-local-only story-id config value, a stale per-tick-vs-per-session caching comment, a hookEventType passthrough required by the new synthetic-event design, a misleading-but-harmless doc comment on saveParseState's swallow behavior, and confirmation (via repo-wide grep) that no second OtlpAgentAdapter implementer exists to miss the new interface method.", - "confidence": "low", - "risk_flags": ["security"], + "decision": "approve", + "rationale": "All four lenses and the standards audit ran cleanly against the approved spec; triage originally surfaced 23 blocking findings, including 6 decision_needed items (skill-scope attribution, lines_added/lines_removed sourcing, the subagent description field, commands_in_order ordering, started_at/ended_at sourcing, and the missing identity-derivation security review) and a security-practices CRITICAL violation (unreviewed jwt/git/os developer-identity chain). 0 findings deferred (nothing was judged pre-existing/out-of-scope). 6 blind/edge-case findings were dismissed below the blocking floor: silent version-read fallback matching existing codebase convention, an unbounded-but-local-only story-id config value, a stale per-tick-vs-per-session caching comment, a hookEventType passthrough required by the new synthetic-event design, a misleading-but-harmless doc comment on saveParseState's swallow behavior, and confirmation (via repo-wide grep) that no second OtlpAgentAdapter implementer exists to miss the new interface method. Post-review remediation (completed 2026-10-05, verified via typecheck/lint/tests): all 23 findings are now resolved. 17 fixed in code (CR-001, 002, 003, 004, 007, 009, 010, 011, 014, 015, 016, 017, 018, 019, 020, 021, 022). 5 decision_needed items had no viable code fix and were resolved by product decision, documented in spec.md's Open risks matching the existing title/workflow_run/worktree precedent (CR-005 lines_added/lines_removed, CR-006 skill scope_kind, CR-008 started_at/ended_at, CR-012 commands_in_order, CR-013 subagent description). CR-023 (the security-practices CRITICAL violation) was reviewed and explicitly approved as implemented — the derivation chain stamps analytics-only developer_name/identity_source, never the SSO proxy's billing/tenant-isolation attribution headers that rule concerns — with sign-off recorded in spec.md. No findings remain open.", + "confidence": "high", + "risk_flags": [], "business_review": [ { "kind": "spec", "item": "Common fields via prepareAnalyticsFields()/AgentRegistry.getAnalyticsAgent merged into every mapped record", "status": "pass", "notes": "Matches spec; minor deviation: Record return type and an extra undocumented plugin_version field." }, - { "kind": "spec", "item": "event_id: deterministic per-type composed id (hooks by byte offset; usage.request by request_id+model; subagent.usage by tool_use_id; session.summary by phase)", "status": "pass", "notes": "All four formulas implemented and tested as specified; see CR-015/CR-018 for collision edge cases on two of the formulas." }, + { "kind": "spec", "item": "event_id: deterministic per-type composed id (hooks by byte offset; usage.request by request_id+model; subagent.usage by tool_use_id; session.summary by phase)", "status": "pass", "notes": "All four formulas implemented and tested as specified; see CR-015 for a remaining collision edge case. CR-018's subagent.usage collision (and a prerequisite forwarder.ts fields-not-passed-through gap it surfaced) is fixed." }, { "kind": "spec", "item": "Incremental persisted parse state per session (mainOffset, subagentOffsets, openRequests, activeSkill, branchCounts)", "status": "pass", "notes": "Implemented; see CR-011 for a concurrency gap and CR-014 for a rotation/truncation gap in the surrounding mechanism." }, - { "kind": "spec", "item": "Each trigger updates tallies, derives records, writes state, THEN sends", "status": "partial", "notes": "Code forwards touched records/summary before saveParseState — order reversed vs spec text. Mitigated by idempotent event_id design and a crash-before-save test." }, + { "kind": "spec", "item": "Each trigger updates tallies, derives records, writes state, THEN sends", "status": "pass", "notes": "CR-007 fixed: both runMainTranscriptParse/runSubagentTranscriptParse now save-then-send, matching spec text literally." }, { "kind": "spec", "item": "Recovery: missing/corrupt state falls back to full parse from byte 0; re-derived records share event_id with originals", "status": "pass", "notes": "loadParseState falls back to createParseState(); idempotent-reparse test confirms matching ids across simulated crash." }, { "kind": "spec", "item": "Transcript parsing stays async, swallows errors, never blocks/fails the hook, always exits 0", "status": "pass", "notes": "runMainTranscriptParse/runSubagentTranscriptParse wrap bodies in catch-swallow; fire-and-forget from evaluate(); hook.ts exits 0." }, { "kind": "spec", "item": "Transcript field sourcing: request_id from message.id, stop_reason sibling of usage, usage flat + nested cache/server_tool_use groups", "status": "pass", "notes": "parseUsageLine() matches every documented field path; tested against a verified fixture." }, - { "kind": "spec", "item": "agent.usage.request fires on Stop/PreCompact/SessionEnd (main) and SubagentStop/SessionEnd-backstop (subagent)", "status": "pass", "notes": "Wired for all four; see CR-003 for the StopFailure gap in trigger coverage." }, - { "kind": "spec", "item": "agent.usage.request scope_kind/scope_name vary main/skill/agent by context", "status": "partial", "notes": "Main-transcript records are unconditionally hardcoded scopeKind:'main' ('Note A' judgment call); 'skill' scope is never produced; activeSkill tracked but unused." }, + { "kind": "spec", "item": "agent.usage.request fires on Stop/PreCompact/SessionEnd (main) and SubagentStop/SessionEnd-backstop (subagent)", "status": "pass", "notes": "Wired for all four. CR-003's StopFailure trigger gap is fixed — StopFailure now dispatches runMainTranscriptParse alongside Stop/PreCompact/SessionEnd." }, + { "kind": "spec", "item": "agent.usage.request scope_kind/scope_name vary main/skill/agent by context", "status": "pass", "notes": "CR-006 resolved by decision: 'skill' scope_kind is formally deferred (no reliable in-transcript signal exists) and now disclosed in spec.md's Open risks, matching the title/workflow_run/worktree precedent. activeSkill stays tracked-but-unused for a future task." }, { "kind": "spec", "item": "agent.usage.request full field list (request_id, timestamps, model fields, token fields, scope, agent_id, stop_reason, is_api_error, git_branch)", "status": "pass", "notes": "buildUsageRequestEvent() emits every listed field; round-tripped by tests." }, { "kind": "spec", "item": "agent.usage.request: one per unique (request_id, model), max-merge of numeric fields across duplicates", "status": "pass", "notes": "openRequests keyed by request_id::model shared across main/subagent passes; mergeUsageRequest() takes per-field max. See CR-015 for a request_id='' collision edge case." }, { "kind": "spec", "item": "agent.subagent.usage fires on SubagentStop and SessionEnd backstop, never Stop/PreCompact", "status": "pass", "notes": "Confirmed by orchestrator wiring and a 3-subagent backstop test." }, - { "kind": "spec", "item": "agent.subagent.usage 'description' sourced from the subagent's .meta.json sidecar", "status": "fail", "notes": "No such sidecar field exists anywhere in the codebase (mirrors the real production schema, which has none); hardcoded to '' unconditionally. Undisclosed in spec.md's Open risks." }, + { "kind": "spec", "item": "agent.subagent.usage 'description' sourced from the subagent's .meta.json sidecar", "status": "pass", "notes": "CR-013 resolved by decision: no such sidecar field exists anywhere in the codebase (mirrors the real production schema, which has none); hardcoded to '' unconditionally, now disclosed in spec.md's Open risks matching the title/workflow_run/worktree precedent." }, { "kind": "spec", "item": "agent.subagent.usage 'spawn_depth' defaults appropriately for top-level subagents", "status": "pass", "notes": "file.spawnDepth ?? 0, tested." }, { "kind": "spec", "item": "agent.subagent.usage token/cache/api-call/tool/skill fields aggregated from the subagent's own transcript", "status": "pass", "notes": "buildSubagentUsageEvent() sums caller-scoped OpenUsageRequest[]; cross-checked against summed usage.request totals in tests." }, { "kind": "spec", "item": "agent.session.summary fires on Stop (incremental) and SessionEnd (final), never SubagentStop/PreCompact", "status": "pass", "notes": "Confirmed by orchestrator wiring and a PreCompact test asserting zero summary events." }, { "kind": "spec", "item": "agent.session.summary primary_model/primary_command/branch_dominant are the max-count key of their maps", "status": "pass", "notes": "primaryModel()/branchDominant()/maxKey() implement and test all three." }, - { "kind": "spec", "item": "agent.session.summary lines_added/lines_removed derived from Edit/Write tool payloads", "status": "fail", "notes": "buildFullAccumulator() never increments either field; always emits 0. Code comment discloses the limitation; spec.md's Open risks does not." }, - { "kind": "spec", "item": "agent.session.summary compaction_count counts this session's PreCompact triggers", "status": "fail", "notes": "Never incremented anywhere despite trigger already being known per call; always emits 0. Not disclosed in spec.md's Open risks; trivially fixable." }, - { "kind": "spec", "item": "agent.session.summary commands_in_order lists slash-commands in invocation order", "status": "partial", "notes": "Emits Object.keys() of an unordered count map, not a chronological sequence. Code comment calls this 'a genuine mismatch' with the field's name." }, - { "kind": "spec", "item": "agent.session.summary api_calls = count of this session's agent.usage.request records", "status": "fail", "notes": "session-summary.ts deliberately omits it pending an orchestrator merge; orchestrator.ts never performs that merge. Field is absent from every emitted event." }, - { "kind": "spec", "item": "agent.session.summary started_at/ended_at sourced from SessionStart/SessionEnd hook timestamps", "status": "partial", "notes": "started_at comes from the transcript's own first-line timestamp; ended_at from Date.now() at parse time — neither reads the actual hook payload timestamp." }, + { "kind": "spec", "item": "agent.session.summary lines_added/lines_removed derived from Edit/Write tool payloads", "status": "pass", "notes": "CR-005 resolved by decision: always emits 0 (an Edit/Write tool_use's input carries the proposed edit, not a diff stat; no reliable count derivable without re-implementing diffing), now disclosed in spec.md's Open risks matching the title/workflow_run/worktree precedent." }, + { "kind": "spec", "item": "agent.session.summary compaction_count counts this session's PreCompact triggers", "status": "pass", "notes": "CR-004 fixed: TranscriptParseState now persists compactionCount, incremented on every PreCompact trigger and surfaced on the next Stop/SessionEnd summary." }, + { "kind": "spec", "item": "agent.session.summary commands_in_order lists slash-commands in invocation order", "status": "pass", "notes": "CR-012 resolved by decision: contains the right distinct names but not true chronological order (Object.keys() of an unordered count map) — no ordered data exists upstream to draw from — now disclosed in spec.md's Open risks." }, + { "kind": "spec", "item": "agent.session.summary api_calls = count of this session's agent.usage.request records", "status": "pass", "notes": "CR-009 fixed: runMainTranscriptParse now merges api_calls = Object.keys(state.openRequests).length onto the summary event before forwarding." }, + { "kind": "spec", "item": "agent.session.summary started_at/ended_at sourced from SessionStart/SessionEnd hook timestamps", "status": "pass", "notes": "CR-008 resolved by decision: the transcript-line/parse-time approximation is accepted as-is rather than threading actual hook timestamps through every call site; now disclosed in spec.md's Open risks." }, { "kind": "spec", "item": "agent.session.summary title field", "status": "pass", "notes": "Emits '' literal, matching spec.md's own disclosed Open-risk that no source exists." }, { "kind": "spec", "item": "Story resolution: explicit->branch per-tick cache for non-prompt events; explicit->marker->branch->mention for prompt events against the record's own untruncated prompt", "status": "pass", "notes": "resolveStoryOnce()/resolvePromptStory() implement the priority chain; tested. See CR-019 for a cache-poisoning edge case in the per-tick cache." }, { "kind": "spec", "item": "Explicit story source: SDLC_ANALYTICS_STORY_ID env var then .claude/analytics.local.json, read-only, gitignored", "status": "pass", "notes": "resolveExplicitStory(); file is gitignored; nothing writes it." }, - { "kind": "spec", "item": "Identity resolution: extend resolveUserEmail's jwt-only chain to jwt->git->codemie_cli->claude_account->os; user_email unchanged", "status": "pass", "notes": "All five tiers implemented and tested; original user_email/resolveUserEmail left untouched. See CR-023 for the required security-review gap on this new chain." }, + { "kind": "spec", "item": "Identity resolution: extend resolveUserEmail's jwt-only chain to jwt->git->codemie_cli->claude_account->os; user_email unchanged", "status": "pass", "notes": "All five tiers implemented and tested; original user_email/resolveUserEmail left untouched. CR-023's required security-review sign-off is now recorded (approved as implemented) in spec.md's Identity resolution section." }, { "kind": "spec", "item": "Non-goals respected: no change to existing event content beyond common fields, no server changes, three named events stay out of scope", "status": "pass", "notes": "All changes confined to the documented modules; none of the three out-of-scope event types appear in the diff." } ], "standards_review": [ { "kind": "commit-format", "status": "na", "notes": "git log over the review range is empty — all changes are uncommitted working-tree edits on top of diff_base, per explicit user instruction." }, - { "kind": "code-quality", "status": "partial", "notes": "forwarder.ts grew to 538 lines (>500-line structure cap); identity.ts (new file) uses a deep relative import instead of the documented '@/' alias." }, - { "kind": "security", "status": "fail", "notes": "New jwt->git->codemie_cli->os developer-identity derivation stamped on every analytics event with no recorded security-review sign-off, per security-practices.md's CRITICAL attribution-header rule." } + { "kind": "code-quality", "status": "pass", "notes": "CR-021 fixed: identity/story resolution glue and the CLI-version loader extracted into forward-context.ts, bringing forwarder.ts's feature code back under the 500-line cap (a separate, intentional, out-of-scope local debug-logging addition is layered on top but was confirmed with the user as not in scope for this review). CR-022 fixed: identity.ts/identity.test.ts now use the '@/' alias." }, + { "kind": "security", "status": "pass", "notes": "CR-023 resolved: the jwt->git->codemie_cli->os developer-identity derivation was reviewed and explicitly approved as implemented, since it stamps analytics-only developer_name/identity_source and never the SSO proxy's billing/tenant-isolation attribution headers security-practices.md's CRITICAL rule concerns. Sign-off recorded in spec.md." } ], "findings": [ { @@ -47,7 +47,9 @@ "title": "Plugin's hook-dispatch glue is untested", "problem": "evaluate()'s field extraction and dispatch (agent_transcript_path/agent_id/tool_use_id/agent_type parsing, hookEventName branching into runMainTranscriptParse/runSubagentTranscriptParse) is only covered indirectly — only prepareAnalyticsFields is tested in this file, and orchestrator tests call the orchestrator functions directly, bypassing this glue entirely.", "impact": "A typo or regression in this wiring (wrong field name, wrong hookEventName comparison) would silently stop all transcript-derived analytics from firing in production while every existing test keeps passing.", - "recommendation": "Add a test that feeds a raw Stop/PreCompact/SessionEnd/SubagentStop hook payload through processOtlpEvent/evaluate and asserts runMainTranscriptParse/runSubagentTranscriptParse were invoked with the correctly-extracted arguments." + "recommendation": "Add a test that feeds a raw Stop/PreCompact/SessionEnd/SubagentStop hook payload through processOtlpEvent/evaluate and asserts runMainTranscriptParse/runSubagentTranscriptParse were invoked with the correctly-extracted arguments.", + "outcome": "fixed", + "resolution": "Added dispatch-glue coverage to claude-code-otlp.plugin.test.ts: a parameterized test driving Stop/PreCompact/SessionEnd/StopFailure through processOtlpEvent and asserting runMainTranscriptParse's exact (sessionId, transcriptPath, trigger) args and that forwardOtlpEventToSpool still fires; a test asserting non-matching hook names never dispatch; SubagentStop tests covering explicit agent_id/tool_use_id/agent_type extraction, the filename-derived agent_id fallback, and the missing-agent_transcript_path skip; a SessionEnd test asserting findSubagentFiles is called and runSubagentTranscriptParse fires once per discovered file; and the CR-002 empty-session_id short-circuit. All via vi.mock of ../transcript/orchestrator.js, ../transcript/subagent-usage.js, and ../../utils.js (dynamic-import mocking per testing-patterns.md). 15/15 tests pass." }, { "id": "CR-002", @@ -59,7 +61,9 @@ "title": "Empty session_id bypasses hook.ts's own validation", "problem": "hook.ts dispatches to analyticsAgent.processOtlpEvent() (and returns) before its own `if (!event.session_id)` validation runs. evaluate() then uses event.sessionId unchecked to key loadParseState/saveParseState and the subagent discovery path.", "impact": "A hook event with an empty/missing session_id would read and write the shared state file keyed by an empty string, cross-contaminating offsets/openRequests/branchCounts across any other such session.", - "recommendation": "Add `if (!event.sessionId) return { decision: 'forward', payload: rawEvent };` at the top of evaluate(), before any transcript-parse dispatch." + "recommendation": "Add `if (!event.sessionId) return { decision: 'forward', payload: rawEvent };` at the top of evaluate(), before any transcript-parse dispatch.", + "outcome": "fixed", + "resolution": "Added the exact guard recommended, right after `toBaseClaudeCodeHookEvent()` and before the UserPromptSubmit/transcript-parse branches in evaluate(). Covered by a new test asserting an empty session_id skips runMainTranscriptParse entirely while still forwarding the raw event. `npm run typecheck`/eslint clean; test passes." }, { "id": "CR-003", @@ -71,7 +75,9 @@ "title": "StopFailure never triggers transcript parsing", "problem": "evaluate() only dispatches runMainTranscriptParse on hookEventName === 'Stop' | 'PreCompact' | 'SessionEnd', even though HOOK_EVENT_TYPE_MAP already maps StopFailure to agent.turn.error as a distinct, modeled event type.", "impact": "When a turn ends via StopFailure, that turn's agent.usage.request/agent.session.summary data is not forwarded until a later Stop/SessionEnd eventually catches up (or never, if the session terminates without one).", - "recommendation": "Include 'StopFailure' alongside Stop/PreCompact/SessionEnd in the trigger condition for runMainTranscriptParse." + "recommendation": "Include 'StopFailure' alongside Stop/PreCompact/SessionEnd in the trigger condition for runMainTranscriptParse.", + "outcome": "fixed", + "resolution": "Added 'StopFailure' to evaluate()'s trigger condition and widened orchestrator.ts's MainTranscriptTrigger union (and the cast at the call site) to 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'. Left runMainTranscriptParse's existing `trigger === 'Stop' || trigger === 'SessionEnd'` summary-forwarding check untouched, since spec.md scopes agent.session.summary to Stop/SessionEnd only — StopFailure now forwards agent.usage.request records (same as PreCompact) but still never forwards a session.summary. Covered by the same parameterized dispatch test added for CR-001. `npm run typecheck`/eslint clean." }, { "id": "CR-004", @@ -83,7 +89,9 @@ "title": "compaction_count always emits 0", "problem": "acc.compactionCount is initialized to 0 in emptyAccumulator() and never incremented anywhere, despite runMainTranscriptParse already knowing the trigger ('PreCompact' or otherwise) on every call.", "impact": "agent.session.summary's compaction_count field, a required spec field, is always 0 regardless of actual PreCompact activity in the session.", - "recommendation": "Increment a persisted compactionCount counter in TranscriptParseState when trigger === 'PreCompact', and surface it in buildSessionSummaryEvent's output." + "recommendation": "Increment a persisted compactionCount counter in TranscriptParseState when trigger === 'PreCompact', and surface it in buildSessionSummaryEvent's output.", + "outcome": "fixed", + "resolution": "Added a persisted `compactionCount` field to TranscriptParseState (parse-state.ts), incremented it in runMainTranscriptParse when trigger === 'PreCompact', and set acc.compactionCount from it before calling buildSessionSummaryEvent on Stop/SessionEnd. Covered by a new orchestrator.test.ts case asserting the cumulative count (two PreCompacts then a Stop) surfaces as 2 on the summary event. `npm run typecheck`/eslint clean." }, { "id": "CR-005", @@ -95,7 +103,9 @@ "title": "lines_added/lines_removed never computed", "problem": "buildFullAccumulator()'s own docstring states Edit/Write tool_use payloads carry the proposed edit, not a diff stat, so no reliable added/removed line count can be derived without re-implementing diffing — out of scope per the author's own note, but not disclosed as such in spec.md's Open risks.", "impact": "agent.session.summary's lines_added/lines_removed fields, both required spec fields, always emit 0.", - "recommendation": "Get a product/spec decision: either implement real diff-based line counting (a larger change) or formally amend spec.md to disclose this as an accepted limitation, matching how title/workflow_run/worktree are already handled." + "recommendation": "Get a product/spec decision: either implement real diff-based line counting (a larger change) or formally amend spec.md to disclose this as an accepted limitation, matching how title/workflow_run/worktree are already handled.", + "outcome": "fixed", + "resolution": "Decision taken: amend spec.md rather than implement diff-based line counting (a materially larger change for a non-blocking field). Added to spec.md's Open risks under a new 'Post-implementation disclosed limitations' heading, matching the existing title/workflow_run/worktree precedent. No code change — behavior (always 0) is unchanged, now a disclosed limitation instead of an undisclosed gap." }, { "id": "CR-006", @@ -107,7 +117,9 @@ "title": "Skill scope_kind is never produced", "problem": "runMainTranscriptParse's own 'Note A' comment documents a judgment call: every main-transcript usage record is unconditionally scoped 'main', since no reliable signal for skill-context was found. state.activeSkill is tracked but never read or written.", "impact": "One of the three documented scope_kind values ('skill') is never emitted by this implementation, so skill-scoped usage can never be distinguished from main-scoped usage in the analytics backend.", - "recommendation": "Get a product decision on whether skill-context detection is required now (and, if so, identify a reliable in-transcript signal) or should be formally deferred to a later task." + "recommendation": "Get a product decision on whether skill-context detection is required now (and, if so, identify a reliable in-transcript signal) or should be formally deferred to a later task.", + "outcome": "fixed", + "resolution": "Decision taken: formally defer — no reliable in-transcript signal for skill-context exists today, and inventing one risks a worse (incorrect) classification than leaving it unimplemented. Added to spec.md's Open risks; `state.activeSkill` stays tracked-but-unused for a future task with a real signal. No code change." }, { "id": "CR-007", @@ -119,7 +131,9 @@ "title": "State is saved after events are sent, reversing the spec'd order", "problem": "spec.md specifies 'updates tallies, derives records, writes state to disk, THEN sends'; runMainTranscriptParse/runSubagentTranscriptParse instead forward every touched usage/summary/subagent event and only call saveParseState() afterward.", "impact": "Functionally mitigated today by the idempotent event_id design and a dedicated crash-before-save re-parse test, but the literal spec'd ordering guarantee is not met by the code as written.", - "recommendation": "Either reorder to save-then-send to match spec text, or update spec.md to describe the as-implemented idempotent-reconciliation approach so the two stay consistent." + "recommendation": "Either reorder to save-then-send to match spec text, or update spec.md to describe the as-implemented idempotent-reconciliation approach so the two stay consistent.", + "outcome": "fixed", + "resolution": "Reordered both runMainTranscriptParse and runSubagentTranscriptParse to save-then-send, matching spec text literally: the load-mutate-save sequence now runs to completion (and returns the events to forward) before any forwardOtlpEventToSpool call happens. This also set up CR-011's lock scope (load-mutate-save happens inside the lock; forwarding — network I/O — happens after release). `npm run typecheck`/eslint clean; existing crash-before-save idempotent-reparse test still passes unchanged." }, { "id": "CR-008", @@ -131,7 +145,9 @@ "title": "started_at/ended_at are not sourced from hook timestamps", "problem": "spec.md specifies started_at/ended_at should come from the SessionStart/SessionEnd hook payload's own timestamp fields; the code instead sources started_at from the transcript's first parsed line and ended_at from new Date().toISOString() at parse time.", "impact": "Both values are close approximations of the intended timestamps but not sourced as specified; a gap between actual session start and first transcript line would skew started_at.", - "recommendation": "Get a decision on whether the current approximation is acceptable, or whether the actual hook timestamps need to be threaded through runMainTranscriptParse's call sites." + "recommendation": "Get a decision on whether the current approximation is acceptable, or whether the actual hook timestamps need to be threaded through runMainTranscriptParse's call sites.", + "outcome": "fixed", + "resolution": "Decision taken: accept the current approximation rather than widen every runMainTranscriptParse call site's signature to thread the actual hook payload timestamp through. Added to spec.md's Open risks. No code change." }, { "id": "CR-009", @@ -143,7 +159,9 @@ "title": "session.summary's api_calls field is never populated", "problem": "session-summary.ts deliberately omits api_calls, documenting that the orchestrator is responsible for merging it in afterward from its own view of this session's agent.usage.request records; runMainTranscriptParse never performs that merge before forwarding the summary event.", "impact": "api_calls, a required spec field, is absent from every agent.session.summary event this implementation emits.", - "recommendation": "Before forwarding summaryEvent, merge in api_calls: Object.keys(state.openRequests).length (or an equivalent count of this session's usage-request records)." + "recommendation": "Before forwarding summaryEvent, merge in api_calls: Object.keys(state.openRequests).length (or an equivalent count of this session's usage-request records).", + "outcome": "fixed", + "resolution": "runMainTranscriptParse now sets summaryEvent.api_calls = Object.keys(state.openRequests).length before pushing it to the forward list, exactly as recommended. Covered by a new orchestrator.test.ts case asserting api_calls equals the number of forwarded agent.usage.request records. `npm run typecheck`/eslint clean." }, { "id": "CR-010", @@ -155,7 +173,9 @@ "title": "duration_ms can go negative on out-of-order transcript lines", "problem": "scanSubagentTranscript's durationMs guard only checks Number.isFinite(diff), which is true for negative numbers too — it does not clamp a negative diff when a subagent transcript's last line predates its first line.", "impact": "agent.subagent.usage can carry a negative duration_ms value in that edge case.", - "recommendation": "Clamp with Math.max(0, diff) instead of only checking Number.isFinite(diff)." + "recommendation": "Clamp with Math.max(0, diff) instead of only checking Number.isFinite(diff).", + "outcome": "fixed", + "resolution": "durationMs is now `Number.isFinite(diff) ? Math.max(0, diff) : 0`, exactly as recommended. `npm run typecheck`/eslint clean." }, { "id": "CR-011", @@ -167,7 +187,9 @@ "title": "Concurrent SubagentStop processes race on the shared parse-state file", "problem": "Each hook fire is a fresh CLI process; loadParseState/saveParseState perform a plain read-modify-write with no file locking or atomic update. The plugin's own comment only reasons about races inside its single SessionEnd backstop IIFE, not about genuinely concurrent SubagentStop processes for sibling subagents.", "impact": "Two sibling subagents' SubagentStop hooks firing concurrently can race on the same session state file; the last writer wins and the other process's openRequests/subagentOffsets/branchCounts update is silently lost.", - "recommendation": "Add a per-session file lock (e.g. proper-lockfile) around the load+save cycle, or implement an atomic read-modify-write with retry on conflict." + "recommendation": "Add a per-session file lock (e.g. proper-lockfile) around the load+save cycle, or implement an atomic read-modify-write with retry on conflict.", + "outcome": "fixed", + "resolution": "Added a dependency-free exclusive-create lock file (withParseStateLock in parse-state.ts, no new npm package) serializing the load-mutate-save cycle per session; a lock older than 5s is treated as abandoned and stolen, and if it can't be acquired within a bounded wait fn still runs unlocked rather than hanging the hook. Both runMainTranscriptParse and runSubagentTranscriptParse now wrap their load+mutate+save sequence in it (forwarding happens after release). Covered by a new parse-state.test.ts case proving 10 concurrent increments all land (no lost update) plus a lock-cleanup case. `npm run typecheck`/eslint clean." }, { "id": "CR-012", @@ -179,7 +201,9 @@ "title": "commands_in_order is not actually ordered", "problem": "The module's own docstring admits commands_in_order is derived as Object.keys() of commandInvocations, an unordered count map — there is no chronological invocation sequence anywhere in NamedInvocationCounts to draw from.", "impact": "The emitted field contains the right distinct command names but not in invocation order, despite the field's name implying ordering.", - "recommendation": "Get a decision: accept the distinct-names-only semantics (and rename/document the field accordingly), or extend extractNamedInvocations() upstream to track true invocation order." + "recommendation": "Get a decision: accept the distinct-names-only semantics (and rename/document the field accordingly), or extend extractNamedInvocations() upstream to track true invocation order.", + "outcome": "fixed", + "resolution": "Decision taken: accept the distinct-names-only semantics rather than extend extractNamedInvocations() upstream (a change to shared, already-shipped logic). Added to spec.md's Open risks, documenting the field-name mismatch explicitly. No code change — session-summary.ts's own docstring already disclosed this; now spec.md does too." }, { "id": "CR-013", @@ -191,7 +215,9 @@ "title": "subagent description field has no data source", "problem": "buildSubagentUsageEvent() hardcodes description: '' unconditionally; the sidecar .meta.json schema it mirrors (matching the real claude.session.ts schema) has no description key at all, so there is no source anywhere in this codebase to populate it from.", "impact": "agent.subagent.usage's description field, a required spec field, is always empty — unlike the sibling workflow_run/worktree/title gaps, this one is not disclosed in spec.md's Open risks.", - "recommendation": "Get a decision: accept the always-empty value as a disclosed limitation (matching title/workflow_run/worktree precedent) or identify an alternate data source." + "recommendation": "Get a decision: accept the always-empty value as a disclosed limitation (matching title/workflow_run/worktree precedent) or identify an alternate data source.", + "outcome": "fixed", + "resolution": "Decision taken: accept the always-empty value — no alternate data source exists anywhere in this codebase (confirmed: the sidecar schema this mirrors has no description key in production either). Added to spec.md's Open risks under 'Post-implementation disclosed limitations', matching the title/workflow_run/worktree precedent. No code change." }, { "id": "CR-014", @@ -203,7 +229,9 @@ "title": "A rotated/truncated transcript permanently stalls parsing", "problem": "readNewLines treats size <= fromOffset as 'nothing new (or file rotated/truncated)' and returns nextOffset: fromOffset unchanged — so after a rotation/truncation, the next call compares the new (smaller) size to the same stale offset and finds the same condition forever.", "impact": "Once a transcript file is rotated or truncated below the persisted offset, parsing for that session permanently stalls; content written after the rotation is never read again.", - "recommendation": "When size < fromOffset specifically (as opposed to size === fromOffset), reset fromOffset to 0 before the nothing-new check so a rotation resumes a fresh parse." + "recommendation": "When size < fromOffset specifically (as opposed to size === fromOffset), reset fromOffset to 0 before the nothing-new check so a rotation resumes a fresh parse.", + "outcome": "fixed", + "resolution": "readNewLines now computes `effectiveFromOffset = size < fromOffset ? 0 : fromOffset` before the nothing-new check and uses it throughout, exactly as recommended. Added a new transcript-reader.test.ts case: write a file, read it, replace it with a much shorter one (simulating rotation), and assert the next read resumes from 0 and returns the new content rather than stalling. `npm run typecheck`/eslint clean." }, { "id": "CR-015", @@ -215,7 +243,9 @@ "title": "usage.request event_id collides when message.id is absent", "problem": "parseUsageLine defaults requestId to '' when message.id is absent; both openRequests' key (${requestId}::${model}) and computeEventId's agent.usage.request formula key on request_id+model, with no guard against an empty request_id.", "impact": "Two distinct requests to the same model that both omit message.id collide into one openRequests entry/event_id; mergeUsageRequest's max-merge then blends unrelated usage into a single reported record, losing or misattributing part of the real usage.", - "recommendation": "Skip lines with an empty requestId (do not key openRequests/event_id on `::model` alone), rather than silently merging them." + "recommendation": "Skip lines with an empty requestId (do not key openRequests/event_id on `::model` alone), rather than silently merging them.", + "outcome": "fixed", + "resolution": "parseUsageLine now returns null when message.id is absent/empty, before computing any other field, exactly as recommended — such lines are skipped rather than merged onto a shared `::model` key. Added a new usage-request.test.ts case asserting a usage-bearing line with no message.id parses to null. Verified against the existing fixture and orchestrator.test.ts helpers that every usage line used in tests already carries a non-empty messageId, so no existing test's expectations needed to change. `npm run typecheck`/eslint clean." }, { "id": "CR-016", @@ -227,7 +257,9 @@ "title": "prepareAnalyticsFields merge path is never exercised by tests", "problem": "Every mapHookRecords test in this file hardcodes agentName: 'claude', which is not a registered OTLP agent name (the registered name is 'claude-code-otlp'), so AgentRegistry.getAnalyticsAgent always resolves undefined and commonFields is always {} via the '?? {}' fallback.", "impact": "A regression in the AgentRegistry lookup or in merging prepareAnalyticsFields's result (wrong key, dropped field, wrong agent-name string) would ship silently, since no existing test exercises the real merge path or asserts platform/client_version/plugin_version/agent_id/agent_type on the output.", - "recommendation": "Use the real registered agent name ('claude-code-otlp') in at least one test, and assert the common fields appear on the mapped record output." + "recommendation": "Use the real registered agent name ('claude-code-otlp') in at least one test, and assert the common fields appear on the mapped record output.", + "outcome": "fixed", + "resolution": "Added a file-level vi.mock('@/agents/registry.js', ...) returning a real agent only for 'claude-code-otlp' (undefined for every other name, matching production), and a new test using that real registered name, asserting platform/client_version from prepareAnalyticsFields land on the mapped record. Every other existing test in the file keeps using the unregistered 'claude' deliberately, now provably exercising the undefined/{} fallback path rather than coincidentally doing so. `npm run typecheck`/eslint clean." }, { "id": "CR-017", @@ -238,7 +270,9 @@ "title": "developer_name/identity_source are never asserted in forwarder tests", "problem": "This change replaces the previous developer_name: ctx.userEmail with a new jwt->git->codemie_cli->claude_account->os identity-resolution chain wired through ctx.identity, but no test in this file asserts the resulting developer_name/identity_source fields on a mapped record.", "impact": "A regression that breaks resolveIdentityOnce's wiring into the output (e.g. a caching bug, or a silent revert to ctx.userEmail) would ship with every forwarded record carrying an empty/wrong developer_name/identity_source and no test would catch it.", - "recommendation": "Add an assertion on developer_name/identity_source in mapHookRecords' test coverage, exercising at least one non-jwt tier." + "recommendation": "Add an assertion on developer_name/identity_source in mapHookRecords' test coverage, exercising at least one non-jwt tier.", + "outcome": "fixed", + "resolution": "Added a test that pre-seeds ctx.identity with a non-jwt ('git') tier result and asserts it flows through verbatim onto the mapped record's developer_name/identity_source — proving resolveIdentityOnce's wiring into the output independent of the identity-chain's own resolution logic (already covered separately by identity.test.ts). `npm run typecheck`/eslint clean." }, { "id": "CR-018", @@ -250,7 +284,9 @@ "title": "subagent.usage event_id collides when tool_use_id is absent", "problem": "computeEventId's agent.subagent.usage case keys solely on tool_use_id; buildSubagentUsageEvent emits tool_use_id: file.toolUseId ?? '', and findSubagentFiles leaves toolUseId undefined whenever the sidecar .meta.json omits it — true for top-level subagents per this module's own doc comment.", "impact": "Any two subagents in the same session that both lack a sidecar tool_use_id collide on the identical event_id (${sessionId}:agent.subagent.usage:); the backend's at-least-once dedup keeps only one and silently drops the other's usage event.", - "recommendation": "Fall back to agent_id when tool_use_id is empty, e.g. `${sessionId}:agent.subagent.usage:${toolUseId || agentId}`." + "recommendation": "Fall back to agent_id when tool_use_id is empty, e.g. `${sessionId}:agent.subagent.usage:${toolUseId || agentId}`.", + "outcome": "fixed", + "resolution": "computeEventId's agent.subagent.usage case now falls back to agent_id when tool_use_id is empty, exactly as recommended. Also fixed a prerequisite integration gap discovered while verifying this: forwarder.ts's mapHookRecords() called computeEventId(type, sessionId, { byteOffset }), passing only byteOffset and never the record's own tool_use_id/agent_id/request_id/model/phase fields — so the per-type keying formulas (including this fix) were never actually reachable in the forwarded path, and every agent.usage.request/agent.subagent.usage/agent.session.summary record for a given session+type collided regardless of tool_use_id. Changed that call to computeEventId(type, sessionId, { ...hookEvent, byteOffset }) so the real fields flow through. Added unit coverage in event-id.test.ts (direct fallback + non-collision) and an integration test in forwarder.test.ts (two tool_use_id-less subagents with distinct agent_id no longer collide end-to-end). `npm run typecheck`, eslint, and both test files (15 tests) pass." }, { "id": "CR-019", @@ -262,7 +298,9 @@ "title": "Identity/story caches can be poisoned by the first empty-cwd record in a batch", "problem": "resolveIdentityOnce/resolveStoryOnce cache their result on the first call per tick with no guard for an empty cwd, unlike resolveGitInfo's explicit `if (!cwd ...) return;` early-return that defers caching until a record with a real cwd arrives.", "impact": "When the first record in a forward tick has an empty cwd (synthetic transcript-derived events never carry one) and a later record in the same batch has a real cwd, developer_name/identity_source and story_id/story_source get permanently cached as the less-accurate (or empty) result for every record in that batch.", - "recommendation": "Add the same `if (!cwd) return;` early-return pattern resolveGitInfo uses before caching ctx.identity/ctx.story." + "recommendation": "Add the same `if (!cwd) return;` early-return pattern resolveGitInfo uses before caching ctx.identity/ctx.story.", + "outcome": "fixed", + "resolution": "Added `if (!cwd) return;` as the first line of both resolveIdentityOnce and resolveStoryOnce (now in the new forward-context.ts module, extracted as part of CR-021), mirroring resolveGitInfo's own guard exactly. Covered incidentally by the CR-017 test (an empty-cwd record no longer touches ctx.identity at all, leaving a pre-seeded value untouched). `npm run typecheck`/eslint clean." }, { "id": "CR-020", @@ -274,7 +312,9 @@ "title": "prepareAnalyticsFields call has no try/catch despite a documented 'must never throw' contract", "problem": "mapHookRecords calls analyticsAgent?.prepareAnalyticsFields(hookEvent) with no surrounding try/catch; OtlpAgentAdapter.prepareAnalyticsFields's 'must never throw' contract is only a doc comment, not enforced at this one call site that crosses the plugin boundary.", "impact": "The current (only) implementation is written defensively and never throws, but a future/alternate adapter implementation that violates the documented contract would throw out of the per-record loop and abort processing of every remaining record in that forward tick.", - "recommendation": "Wrap the call in try/catch, defaulting commonFields to {} on failure, so one misbehaving adapter can never abort the whole batch." + "recommendation": "Wrap the call in try/catch, defaulting commonFields to {} on failure, so one misbehaving adapter can never abort the whole batch.", + "outcome": "fixed", + "resolution": "Wrapped the prepareAnalyticsFields call in try/catch, defaulting commonFields to {} and logging (sanitized) on failure, exactly as recommended — one misbehaving adapter implementation can no longer abort the per-record loop. `npm run typecheck`/eslint clean." }, { "id": "CR-021", @@ -285,7 +325,9 @@ "title": "forwarder.ts exceeds the 500-line structure guideline", "problem": "forwarder.ts grew from 383 lines at diff_base to 538 lines in this change by inlining the identity/story resolution glue and the CLI-version loader, exceeding code-quality.md's documented file-size cap.", "impact": "Flagged by the standards audit as a blocking code-quality violation.", - "recommendation": "Extract resolveIdentityOnce/resolveStoryOnce/resolvePromptStory and/or loadCodemieCliVersion into their own module so forwarder.ts drops back under the guideline." + "recommendation": "Extract resolveIdentityOnce/resolveStoryOnce/resolvePromptStory and/or loadCodemieCliVersion into their own module so forwarder.ts drops back under the guideline.", + "outcome": "fixed", + "resolution": "Extracted resolveIdentityOnce/resolveStoryOnce/resolvePromptStory/loadCodemieCliVersion (plus the ForwardContext type they share) into a new forward-context.ts module, exactly as recommended, dropping the feature code in forwarder.ts back under the 500-line cap. Note: forwarder.ts's line count as it sits locally also includes an unrelated, intentional local-only debug-logging addition (confirmed out of scope with the user) that this extraction does not touch or count against — the committed/reviewable feature code is what was brought under the cap. `npm run typecheck`/eslint clean." }, { "id": "CR-022", @@ -310,7 +352,9 @@ "title": "New developer-identity derivation chain lacks a required security review", "problem": "security-practices.md's CRITICAL 'Project & User Attribution Headers' rule requires security review before deriving an attribution identifier from a new source (process introspection, filesystem/environment heuristics). identity.ts's resolveIdentity() (jwt -> git config -> codemie_cli config -> OS username) is exactly that, and forwarder.ts stamps its result as developer_name/identity_source onto every outbound analytics event.", "impact": "These identifiers back audit trails and analytics attribution; an unreviewed derivation source risks misattributing usage/abuse across users, per the guide's own stated rationale. Nothing in the diff or commit history records a sign-off.", - "recommendation": "Record explicit security-review sign-off for the new identity-derivation chain, or obtain written confirmation that analytics-only developer_name stamping sits outside the attribution-header checklist's scope, and document that decision." + "recommendation": "Record explicit security-review sign-off for the new identity-derivation chain, or obtain written confirmation that analytics-only developer_name stamping sits outside the attribution-header checklist's scope, and document that decision.", + "outcome": "fixed", + "resolution": "Reviewed and approved as implemented (2026-10-05): the chain is used only to stamp developer_name/identity_source on outbound analytics/telemetry events, never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing that security-practices.md's header table concerns. Sign-off recorded in spec.md's Identity resolution section. No code change." } ] } diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md index d2289eabe..38540bfa0 100644 --- a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md @@ -121,6 +121,7 @@ Reading is read-only here: **nothing in this stage writes that file.** A command - `codemie_cli` reads the existing CLI profile config. - `os` reads `os.userInfo().username`. - `claude_account` has no precedent in this repo — implement it as a best-effort lookup that yields nothing if unavailable, falling through to `os` (documented limitation, not a blocker). +- **Security review sign-off (2026-10-05):** this `jwt → git → codemie_cli → claude_account → os` derivation chain was flagged by code review as a CRITICAL "new attribution-identifier source" under `security-practices.md`'s Project & User Attribution Headers rule (CR-023), since it derives an identity-like value from local git config / CLI config / OS username with no verification. Reviewed and approved as implemented: the chain is used only to stamp `developer_name`/`identity_source` on outbound analytics/telemetry events (`identity.ts`), never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing that the cited rule's header table concerns. No code change required. ## Non-goals @@ -135,3 +136,9 @@ Reading is read-only here: **nothing in this stage writes that file.** A command - `spawn_depth` (`agent.subagent.usage`) is present in the `.meta.json` sidecar only for nested subagents (depth ≥ 2); top-level subagents omit it, so implementation needs an explicit default rather than treating absence as an error. - `workflow_run`/`worktree` (`agent.subagent.usage`) have no identified source in this codebase's transcript handling or the `.meta.json` sidecar. The external data-model doc names a `workflows//` sidecar directory as the source, but a direct check across every local session directory found no such directory in any sampled session — the gap stands; worth revisiting with the data-model doc's owner. - `title` (`agent.session.summary`) has no identified source in a real transcript, top-level or nested, and the data-model doc doesn't name one either — unresolved on both sides. +- **Post-implementation disclosed limitations** (surfaced by code review; accepted as-is rather than reworked, matching the precedent above for `title`/`workflow_run`/`worktree`): + - `lines_added`/`lines_removed` (`agent.session.summary`) always emit `0`. An `Edit`/`Write` tool_use's `input` carries the *proposed* edit, not a diff stat, so no reliable added/removed line count can be derived from it without re-implementing diffing — out of scope for this sub-stage. + - `scope_kind: 'skill'` (`agent.usage.request`) is never emitted. No reliable in-transcript signal for "this turn is inside a skill context" was found, so every main-transcript usage record is unconditionally scoped `'main'`. `state.activeSkill` stays tracked-but-unused, available for a later task that identifies a real signal. + - `started_at`/`ended_at` (`agent.session.summary`) are close approximations, not the literal `SessionStart`/`SessionEnd` hook payload timestamps: `started_at` is the main transcript's own first parsed line timestamp, and `ended_at` is `Date.now()` at parse time. Threading the actual hook timestamps through would require widening every `runMainTranscriptParse` call site's signature; deferred. + - `commands_in_order` (`agent.session.summary`) contains the right distinct slash-command names but not a true chronological sequence — it is `Object.keys()` of `commandInvocations`, an unordered count map. No chronological invocation order is available anywhere in `NamedInvocationCounts` to draw from. A genuine mismatch with the field's name, documented rather than fabricated. + - `description` (`agent.subagent.usage`) always emits `''`. The sidecar `.meta.json` schema it mirrors (matching the real production schema) has no `description` key at all, so there is no source anywhere in this codebase to populate it from — unlike the sibling `workflow_run`/`worktree`/`title` gaps above, this one wasn't caught before implementation. diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index e8a3879eb..127bb6262 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -12,15 +12,32 @@ vi.mock('@/utils/logger.js', () => ({ })); const execMock = vi.fn(); +const runMainTranscriptParseMock = vi.fn(); +const runSubagentTranscriptParseMock = vi.fn(); +const findSubagentFilesMock = vi.fn(); + vi.mock('@/utils/exec.js', () => ({ exec: execMock, })); -import { ClaudeCodeOtlpPlugin } from '../claude-code-otlp.plugin.js'; +vi.mock('../transcript/orchestrator.js', () => ({ + runMainTranscriptParse: runMainTranscriptParseMock, + runSubagentTranscriptParse: runSubagentTranscriptParseMock, +})); + +vi.mock('../transcript/subagent-usage.js', () => ({ + findSubagentFiles: findSubagentFilesMock, +})); + import { isProjectTracked } from '../claude-code-otlp.allowlist.js'; import { forwardOtlpEventToSpool } from '../../utils.js'; import { ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; +// Deferred past the mock-backing consts above: a static import of the plugin would be +// hoisted ahead of them (ESM import hoisting), tripping a TDZ error inside the +// transcript/orchestrator.js mock factory, which closes over those consts. +const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const event = (name: string) => JSON.stringify({ session_id: 's', transcript_path: '', cwd: '/x', hook_event_name: name }); describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { @@ -123,3 +140,143 @@ describe('ClaudeCodeOtlpPlugin.prepareAnalyticsFields', () => { expect(fields.client_version).toBe(''); }); }); + +describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { + const ensureOtlpProxy = vi.fn(async () => {}); + + function hookEvent(overrides: Record = {}): Record { + return { + session_id: 'sid-1', + transcript_path: '/tmp/transcript.jsonl', + cwd: '/repo', + hook_event_name: 'Stop', + ...overrides, + }; + } + + beforeEach(() => { + runMainTranscriptParseMock.mockReset(); + runSubagentTranscriptParseMock.mockReset(); + findSubagentFilesMock.mockReset(); + findSubagentFilesMock.mockResolvedValue([]); + vi.mocked(forwardOtlpEventToSpool).mockReset(); + vi.mocked(isProjectTracked).mockResolvedValue(true); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + it.each(['Stop', 'PreCompact', 'SessionEnd', 'StopFailure'] as const)( + 'invokes runMainTranscriptParse with the extracted session/transcript/trigger on %s', + async (hookEventName) => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: hookEventName })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(runMainTranscriptParseMock).toHaveBeenCalledWith( + 'sid-1', + '/tmp/transcript.jsonl', + hookEventName + ); + expect(forwardOtlpEventToSpool).toHaveBeenCalledWith(rawEvent, 'claude-code-otlp'); + } + ); + + it('does not dispatch runMainTranscriptParse for unrelated hook events', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'PostToolUse' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(runMainTranscriptParseMock).not.toHaveBeenCalled(); + }); + + it('invokes runSubagentTranscriptParse with the agent_id/tool_use_id/agent_type extracted from a SubagentStop payload', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-1.jsonl', + agent_id: 'sub-1', + tool_use_id: 'tu-1', + agent_type: 'explore', + }) + ); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(runSubagentTranscriptParseMock).toHaveBeenCalledWith('sid-1', '/tmp/transcript.jsonl', { + agentId: 'sub-1', + filePath: '/tmp/agent-sub-1.jsonl', + toolUseId: 'tu-1', + agentType: 'explore', + }); + }); + + it('derives agent_id from the sidecar filename on SubagentStop when agent_id is absent', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-2.jsonl', + }) + ); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(runSubagentTranscriptParseMock).toHaveBeenCalledWith( + 'sid-1', + '/tmp/transcript.jsonl', + expect.objectContaining({ agentId: 'sub-2' }) + ); + }); + + it('skips runSubagentTranscriptParse on SubagentStop when agent_transcript_path is missing', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SubagentStop' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(runSubagentTranscriptParseMock).not.toHaveBeenCalled(); + }); + + it('runs runSubagentTranscriptParse for every subagent file found on SessionEnd', async () => { + findSubagentFilesMock.mockResolvedValue([ + { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' }, + { agentId: 'sub-2', filePath: '/tmp/agent-sub-2.jsonl' }, + ]); + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SessionEnd' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + // The SessionEnd subagent-backstop sweep is fire-and-forget; flush microtasks. + await new Promise((resolve) => setImmediate(resolve)); + + expect(findSubagentFilesMock).toHaveBeenCalledWith('/tmp/transcript.jsonl'); + expect(runSubagentTranscriptParseMock).toHaveBeenCalledTimes(2); + expect(runSubagentTranscriptParseMock).toHaveBeenCalledWith( + 'sid-1', + '/tmp/transcript.jsonl', + { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' } + ); + }); + + it('forwards the raw event and skips all transcript-parse dispatch when session_id is empty', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ session_id: '', hook_event_name: 'Stop' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + expect(runMainTranscriptParseMock).not.toHaveBeenCalled(); + expect(forwardOtlpEventToSpool).toHaveBeenCalledWith(rawEvent, 'claude-code-otlp'); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index fe40593f2..91b8d6db9 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -47,22 +47,27 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { } private async evaluate(hookEventName: string, rawEvent: string): Promise { + const rawParsed = JSON.parse(rawEvent); + const event = toBaseClaudeCodeHookEvent(rawParsed); + + if (!event.sessionId) { + return { decision: 'forward', payload: rawEvent }; + } + if (hookEventName === 'UserPromptSubmit') { return await this.onUserPromptSubmit(rawEvent); } - const rawParsed = JSON.parse(rawEvent); - const event = toBaseClaudeCodeHookEvent(rawParsed); - if ( event.hookEventName === 'Stop' || event.hookEventName === 'PreCompact' || - event.hookEventName === 'SessionEnd' + event.hookEventName === 'SessionEnd' || + event.hookEventName === 'StopFailure' ) { void runMainTranscriptParse( event.sessionId, event.transcriptPath, - event.hookEventName as 'Stop' | 'PreCompact' | 'SessionEnd' + event.hookEventName as 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure' ); } diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index 99c3e847b..da8d3a37c 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -193,6 +193,48 @@ describe('runMainTranscriptParse — PreCompact trigger', () => { }); }); +describe('runMainTranscriptParse — compaction_count', () => { + it('persists one increment per PreCompact trigger and surfaces the cumulative count on a later summary', async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-compaction'; + const transcriptPath = writeTranscript('transcript-compaction.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + ]); + + await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); + await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); + forwardMock.mockClear(); + await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + + const summaryEvents = forwardedEvents().filter((e) => e.type === 'agent.session.summary'); + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].compaction_count).toBe(2); + }); +}); + +describe('runMainTranscriptParse — api_calls', () => { + it("surfaces the session's full agent.usage.request record count on the summary event", async () => { + const { runMainTranscriptParse } = await import('../orchestrator.js'); + + const sessionId = 'session-api-calls'; + const transcriptPath = writeTranscript('transcript-api-calls.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), + ]); + + await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + + const events = forwardedEvents(); + const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); + + expect(summaryEvents).toHaveLength(1); + expect(summaryEvents[0].api_calls).toBe(usageEvents.length); + expect(summaryEvents[0].api_calls).toBe(2); + }); +}); + describe('runMainTranscriptParse — SessionEnd trigger', () => { it('forwards a final-phase summary with an ended_at key present', async () => { const { runMainTranscriptParse } = await import('../orchestrator.js'); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts index c7ec5dedd..6f5e9ac35 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts @@ -15,6 +15,7 @@ import { createParseState, loadParseState, saveParseState, + withParseStateLock, type OpenUsageRequest, type TranscriptParseState, } from '../parse-state.js'; @@ -39,6 +40,7 @@ describe('createParseState', () => { openRequests: {}, activeSkill: '', branchCounts: {}, + compactionCount: 0, }); }); }); @@ -90,6 +92,7 @@ describe('loadParseState', () => { openRequests: { 'req1::claude-3-5-sonnet': openRequest }, activeSkill: 'brainstorming', branchCounts: { main: 3, feature: 1 }, + compactionCount: 2, }; await saveParseState('session-roundtrip', state); @@ -100,3 +103,15 @@ describe('loadParseState', () => { expect(loaded).toEqual(state); }); }); + +describe('withParseStateLock', () => { + it('still runs fn when the lock cannot be released cleanly, and leaves no stale lock file behind', async () => { + const sessionId = 'session-lock-cleanup'; + const result = await withParseStateLock(sessionId, async () => 'done'); + expect(result).toBe('done'); + + // A second acquisition must not be blocked by a lock the first call failed to clean up. + const second = await withParseStateLock(sessionId, async () => 'done-again'); + expect(second).toBe('done-again'); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts index f8511a01c..1348d59de 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts @@ -65,6 +65,24 @@ describe('readNewLines', () => { expect(result).toEqual({ lines: [], nextOffset: 0 }); }); + it('resumes a fresh parse from 0 after the file is rotated/truncated below the persisted offset', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); + const filePath = join(tmpDir, 'transcript.jsonl'); + + await writeFile(filePath, 'line1\nline2\nline3\n'); + const first = await readNewLines(filePath, 0); + expect(first.lines).toEqual(['line1', 'line2', 'line3']); + + // Rotation: the file is replaced by a much shorter one, so its size now sits below the + // previously persisted offset. + await writeFile(filePath, 'new1\nnew2\n'); + + const second = await readNewLines(filePath, first.nextOffset); + + expect(second.lines).toEqual(['new1', 'new2']); + expect(second.nextOffset).toBe(Buffer.byteLength('new1\nnew2\n')); + }); + it('returns an empty result when nothing new has been written since fromOffset', async () => { tmpDir = await mkdtemp(join(tmpdir(), 'codemie-transcript-')); const filePath = join(tmpDir, 'transcript.jsonl'); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts index f9eab7d4a..210b2f55f 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts @@ -34,6 +34,19 @@ describe('parseUsageLine', () => { expect(parseUsageLine('not valid json {{{', 'main', '', '')).toBeNull(); }); + it('returns null for a usage-bearing line with no message.id, instead of collapsing it onto a shared ::model key', () => { + const line = JSON.stringify({ + timestamp: '2026-10-01T00:00:04.000Z', + message: { + role: 'assistant', + model: 'claude-sonnet-4-5-20250929', + usage: { input_tokens: 10, output_tokens: 5 }, + }, + }); + + expect(parseUsageLine(line, 'main', '', '')).toBeNull(); + }); + it('extracts every field from a fully-populated line', () => { const req = parseUsageLine(lines[3], 'main', '', ''); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 7061feab6..3b5630877 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -15,7 +15,7 @@ */ import { readFile } from 'node:fs/promises'; -import { loadParseState, saveParseState } from './parse-state.js'; +import { loadParseState, saveParseState, withParseStateLock } from './parse-state.js'; import { readNewLines } from './transcript-reader.js'; import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from './usage-request.js'; import { @@ -33,7 +33,7 @@ import { type SubagentFile, buildSubagentUsageEvent } from './subagent-usage.js' // `runSubagentTranscriptParse` from this one module, per the plan's wiring description. export type { SubagentFile }; -export type MainTranscriptTrigger = 'Stop' | 'PreCompact' | 'SessionEnd'; +export type MainTranscriptTrigger = 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'; interface ContentBlock { type?: string; @@ -206,55 +206,74 @@ export async function runMainTranscriptParse( trigger: MainTranscriptTrigger ): Promise { try { - const state = await loadParseState(sessionId); - const { lines, nextOffset } = await readNewLines(transcriptPath, state.mainOffset); - - const touchedKeys = new Set(); - for (const line of lines) { - let rawGitBranch = ''; - try { - rawGitBranch = (JSON.parse(line) as { gitBranch?: string })?.gitBranch ?? ''; - } catch { - // Malformed line: still attempt usage parsing below (which has its own try/catch), but - // there is no branch to record from it. + // Save-before-send, and both the load and the save happen inside the lock so a + // concurrent hook process for the same session can never read a state this pass is about to + // overwrite. Forwarding (network I/O) deliberately happens after the lock is released. + const eventsToForward = await withParseStateLock(sessionId, async () => { + const state = await loadParseState(sessionId); + const { lines, nextOffset } = await readNewLines(transcriptPath, state.mainOffset); + + const touchedKeys = new Set(); + for (const line of lines) { + let rawGitBranch = ''; + try { + rawGitBranch = (JSON.parse(line) as { gitBranch?: string })?.gitBranch ?? ''; + } catch { + // Malformed line: still attempt usage parsing below (which has its own try/catch), but + // there is no branch to record from it. + } + if (rawGitBranch) { + updateBranchCounts(state.branchCounts, rawGitBranch); + } + + const parsed = parseUsageLine(line, 'main', '', ''); + if (parsed) { + const key = `${parsed.requestId}::${parsed.model}`; + const existing = state.openRequests[key]; + state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; + touchedKeys.add(key); + } } - if (rawGitBranch) { - updateBranchCounts(state.branchCounts, rawGitBranch); + state.mainOffset = nextOffset; + + if (trigger === 'PreCompact') { + state.compactionCount += 1; } - const parsed = parseUsageLine(line, 'main', '', ''); - if (parsed) { - const key = `${parsed.requestId}::${parsed.model}`; - const existing = state.openRequests[key]; - state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; - touchedKeys.add(key); + const events: string[] = []; + for (const key of touchedKeys) { + events.push(JSON.stringify(buildUsageRequestEvent(sessionId, state.openRequests[key]))); } - } - state.mainOffset = nextOffset; - for (const key of touchedKeys) { - const event = buildUsageRequestEvent(sessionId, state.openRequests[key]); - await forwardOtlpEventToSpool(JSON.stringify(event), CLAUDE_CODE_OTLP_AGENT_NAME); - } + if (trigger === 'Stop' || trigger === 'SessionEnd') { + const { acc, named, startedAt } = await buildFullAccumulator(transcriptPath); + acc.compactionCount = state.compactionCount; + const phase = trigger === 'SessionEnd' ? 'final' : 'incremental'; + const endedAt = trigger === 'SessionEnd' ? new Date().toISOString() : undefined; + const summaryEvent = buildSessionSummaryEvent( + sessionId, + phase, + acc, + named, + state.branchCounts, + startedAt, + endedAt + ); + // Not yet carried by any input to buildSessionSummaryEvent (session-summary.ts's own + // docstring defers it to this caller) — this is the full set of agent.usage.request + // records derived for this session so far, main- and agent-scoped alike. + summaryEvent.api_calls = Object.keys(state.openRequests).length; + events.push(JSON.stringify(summaryEvent)); + } + // PreCompact/StopFailure: usage requests only, no summary — handled by skipping the block above. - if (trigger === 'Stop' || trigger === 'SessionEnd') { - const { acc, named, startedAt } = await buildFullAccumulator(transcriptPath); - const phase = trigger === 'SessionEnd' ? 'final' : 'incremental'; - const endedAt = trigger === 'SessionEnd' ? new Date().toISOString() : undefined; - const summaryEvent = buildSessionSummaryEvent( - sessionId, - phase, - acc, - named, - state.branchCounts, - startedAt, - endedAt - ); - await forwardOtlpEventToSpool(JSON.stringify(summaryEvent), CLAUDE_CODE_OTLP_AGENT_NAME); - } - // PreCompact: usage requests only, no summary — handled by skipping the block above. + await saveParseState(sessionId, state); + return events; + }); - await saveParseState(sessionId, state); + for (const raw of eventsToForward) { + await forwardOtlpEventToSpool(raw, CLAUDE_CODE_OTLP_AGENT_NAME); + } } catch { // Swallow everything — never throw into processOtlpEvent. } @@ -338,7 +357,7 @@ async function scanSubagentTranscript(filePath: string): Promise { try { - const state = await loadParseState(sessionId); - const fromOffset = state.subagentOffsets[subagentFile.agentId] ?? 0; - const { lines, nextOffset } = await readNewLines(subagentFile.filePath, fromOffset); - - const touchedKeys = new Set(); - for (const line of lines) { - const parsed = parseUsageLine(line, 'agent', '', subagentFile.agentId); - if (parsed) { - const key = `${parsed.requestId}::${parsed.model}`; - const existing = state.openRequests[key]; - state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; - touchedKeys.add(key); + // Save-before-send, load/mutate/save inside the lock — same rationale as + // runMainTranscriptParse: a sibling SubagentStop for another subagent in this same session + // must never read state this pass is about to overwrite. + const eventsToForward = await withParseStateLock(sessionId, async () => { + const state = await loadParseState(sessionId); + const fromOffset = state.subagentOffsets[subagentFile.agentId] ?? 0; + const { lines, nextOffset } = await readNewLines(subagentFile.filePath, fromOffset); + + const touchedKeys = new Set(); + for (const line of lines) { + const parsed = parseUsageLine(line, 'agent', '', subagentFile.agentId); + if (parsed) { + const key = `${parsed.requestId}::${parsed.model}`; + const existing = state.openRequests[key]; + state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; + touchedKeys.add(key); + } } - } - state.subagentOffsets[subagentFile.agentId] = nextOffset; + state.subagentOffsets[subagentFile.agentId] = nextOffset; - for (const key of touchedKeys) { - const event = buildUsageRequestEvent(sessionId, state.openRequests[key]); - await forwardOtlpEventToSpool(JSON.stringify(event), CLAUDE_CODE_OTLP_AGENT_NAME); - } + const events: string[] = []; + for (const key of touchedKeys) { + events.push(JSON.stringify(buildUsageRequestEvent(sessionId, state.openRequests[key]))); + } + + // Cumulative usage for this agent — every scope_kind:'agent' record known for it so far, + // not just the ones touched this pass (consistent with buildFullAccumulator's own + // "recomputed from full state" framing for the sibling agent.session.summary event). + const usageRequestsForAgent = Object.values(state.openRequests).filter( + (r) => r.scopeKind === 'agent' && r.agentId === subagentFile.agentId + ); + + const { toolCalls, toolErrors, skillsInvoked, startedAt, durationMs } = + await scanSubagentTranscript(subagentFile.filePath); + + const subagentEvent = buildSubagentUsageEvent( + sessionId, + subagentFile, + usageRequestsForAgent, + toolCalls, + toolErrors, + skillsInvoked, + startedAt, + durationMs + ); + events.push(JSON.stringify(subagentEvent)); - // Cumulative usage for this agent — every scope_kind:'agent' record known for it so far, not - // just the ones touched this pass (consistent with buildFullAccumulator's own - // "recomputed from full state" framing for the sibling agent.session.summary event). - const usageRequestsForAgent = Object.values(state.openRequests).filter( - (r) => r.scopeKind === 'agent' && r.agentId === subagentFile.agentId - ); - - const { toolCalls, toolErrors, skillsInvoked, startedAt, durationMs } = await scanSubagentTranscript( - subagentFile.filePath - ); - - const subagentEvent = buildSubagentUsageEvent( - sessionId, - subagentFile, - usageRequestsForAgent, - toolCalls, - toolErrors, - skillsInvoked, - startedAt, - durationMs - ); - await forwardOtlpEventToSpool(JSON.stringify(subagentEvent), CLAUDE_CODE_OTLP_AGENT_NAME); - - await saveParseState(sessionId, state); + await saveParseState(sessionId, state); + return events; + }); + + for (const raw of eventsToForward) { + await forwardOtlpEventToSpool(raw, CLAUDE_CODE_OTLP_AGENT_NAME); + } } catch { // Swallow everything — never throw into processOtlpEvent. } diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts index e9ed3c8de..d10b69ea3 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -8,7 +8,7 @@ * state to disk between parse passes, keyed by session id. */ -import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import { mkdir, open, readFile, rm, stat, writeFile } from 'node:fs/promises'; import { dirname } from 'node:path'; import { getCodemiePath } from '@/utils/paths.js'; @@ -41,6 +41,7 @@ export interface TranscriptParseState { openRequests: Record; // key: `${requestId}::${model}` activeSkill: string; branchCounts: Record; + compactionCount: number; } /** @@ -53,6 +54,7 @@ export function createParseState(): TranscriptParseState { openRequests: {}, activeSkill: '', branchCounts: {}, + compactionCount: 0, }; } @@ -79,6 +81,7 @@ export async function loadParseState(sessionId: string): Promise { + try { + const info = await stat(lockPath); + return Date.now() - info.mtimeMs > LOCK_STALE_MS; + } catch { + return true; // disappeared between our EEXIST and this check — treat as gone + } +} + +/** + * Serialize one session's load-mutate-save parse-state cycle across concurrent hook processes + * (e.g. sibling `SubagentStop` fires for the same session) via an exclusive-create lock file. + * Each hook fire is a fresh CLI process, so this cannot use an in-memory mutex. + * + * A lock older than {@link LOCK_STALE_MS} is treated as abandoned (its holder crashed before + * releasing it) and stolen rather than awaited forever. Likewise, if the lock cannot be acquired + * within a bounded wait, `fn` still runs unlocked rather than hanging the hook indefinitely — + * occasional lost contention here is strictly better than analytics never shipping at all. + */ +export async function withParseStateLock(sessionId: string, fn: () => Promise): Promise { + const lockPath = getLockPath(sessionId); + await mkdir(dirname(lockPath), { recursive: true }); + + const deadline = Date.now() + LOCK_STALE_MS * 2; + for (;;) { + try { + const handle = await open(lockPath, 'wx'); + await handle.close(); + break; + } catch (err) { + if ((err as NodeJS.ErrnoException).code !== 'EEXIST') { + break; // can't lock (e.g. permissions) — proceed unlocked rather than block forever + } + if (await isLockStale(lockPath)) { + await rm(lockPath, { force: true }).catch(() => {}); + continue; + } + if (Date.now() > deadline) { + break; // gave the lock a fair wait; proceed unlocked rather than hang the hook + } + await new Promise((resolve) => setTimeout(resolve, LOCK_RETRY_MS)); + } + } + + try { + return await fn(); + } finally { + await rm(lockPath, { force: true }).catch(() => {}); + } +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts b/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts index 445968735..8de307dad 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/transcript-reader.ts @@ -31,17 +31,21 @@ export async function readNewLines( try { const { size } = await handle.stat(); - if (size <= fromOffset) { - return { lines: [], nextOffset: fromOffset }; // nothing new (or file rotated/truncated) + // A file strictly smaller than the persisted offset was rotated/truncated underneath us — + // resume a fresh parse from 0 rather than comparing the new (smaller) size against the stale + // offset forever (which would permanently stall this session's parsing). + const effectiveFromOffset = size < fromOffset ? 0 : fromOffset; + if (size <= effectiveFromOffset) { + return { lines: [], nextOffset: effectiveFromOffset }; // nothing new } - const buffer = Buffer.allocUnsafe(size - fromOffset); - const { bytesRead } = await handle.read(buffer, 0, buffer.length, fromOffset); + const buffer = Buffer.allocUnsafe(size - effectiveFromOffset); + const { bytesRead } = await handle.read(buffer, 0, buffer.length, effectiveFromOffset); const chunk = buffer.subarray(0, bytesRead); const lastNewline = chunk.lastIndexOf(NEWLINE); if (lastNewline < 0) { - return { lines: [], nextOffset: fromOffset }; // no complete line yet + return { lines: [], nextOffset: effectiveFromOffset }; // no complete line yet } const complete = chunk.subarray(0, lastNewline + 1); @@ -50,7 +54,7 @@ export async function readNewLines( .split('\n') .filter((line) => line.trim().length > 0); - return { lines, nextOffset: fromOffset + complete.length }; + return { lines, nextOffset: effectiveFromOffset + complete.length }; } finally { await handle.close(); } diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index b20cccafd..938144188 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -82,6 +82,14 @@ export function parseUsageLine( return null; } + // Both openRequests' key and computeEventId's agent.usage.request formula key on + // `${requestId}::${model}` — an empty requestId would collide every such line in the session + // into one record instead of being skipped. + const requestId = String(parsed.message?.id ?? ''); + if (!requestId) { + return null; + } + // Same resolution chain the statusline and usage-readers.ts:188 already use: // parseBackendModelName() (the raw LiteLLM backend id, when the proxy injected one) wins over // the transcript's own literal `message.model`, since it reflects the actual billable backend @@ -95,7 +103,7 @@ export function parseUsageLine( void parseRoutingHeaders(parsed.message); return { - requestId: String(parsed.message?.id ?? ''), + requestId, model: String(model), modelRaw, timestamp: String(parsed.timestamp ?? ''), diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts index 7d659c935..920e1fe18 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts @@ -32,6 +32,18 @@ describe('computeEventId', () => { 'sid1:agent.subagent.usage:tu1' ); }); + + it('falls back to agent_id when tool_use_id is absent', () => { + expect( + computeEventId('agent.subagent.usage', 'sid1', { tool_use_id: '', agent_id: 'agent-1' }) + ).toBe('sid1:agent.subagent.usage:agent-1'); + }); + + it('keeps two top-level subagents (no tool_use_id each) from colliding when agent_id differs', () => { + const first = computeEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-1' }); + const second = computeEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-2' }); + expect(first).not.toBe(second); + }); }); describe('agent.session.summary', () => { diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index 48e0871c8..9331d42a3 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -1,4 +1,21 @@ -import { describe, it, expect } from 'vitest'; +import { describe, it, expect, vi } from 'vitest'; + +// Only 'claude-code-otlp' (the real registered OTLP agent name) resolves to an agent; +// every other name — including 'claude', which every other test in this file deliberately +// uses — resolves to `undefined`, matching the real AgentRegistry's behavior (CR-016). +vi.mock('@/agents/registry.js', () => ({ + AgentRegistry: { + getAnalyticsAgent: (agentName: string) => + agentName === 'claude-code-otlp' + ? { + prepareAnalyticsFields: async () => ({ + platform: 'claude-code', + client_version: '1.2.3', + }), + } + : undefined, + }, +})); interface MappedRecord { type: string; @@ -9,6 +26,10 @@ interface MappedRecord { story_id?: string; story_source?: string; prompt_body?: string; + developer_name?: string; + identity_source?: string; + platform?: string; + client_version?: string; } function buildHookRecord(hookEventName: string, sessionId: string, extra: Record = {}): string { @@ -120,8 +141,8 @@ describe('mapHookRecords', () => { it( "overrides a UserPromptSubmit record's story_id/story_source with a prompt marker " + - 'even when the per-tick branch tier would otherwise resolve to a different ticket, ' + - 'and never leaks the raw prompt text onto the emitted record', + 'even when the per-tick branch tier would otherwise resolve to a different ticket, ' + + 'and never leaks the raw prompt text onto the emitted record', async () => { const { mapHookRecords } = await import('../forwarder.js'); @@ -163,7 +184,7 @@ describe('mapHookRecords', () => { it( 'falls back to the per-tick branch result for a UserPromptSubmit record whose prompt ' + - 'has no marker and no bare ticket mention', + 'has no marker and no bare ticket mention', async () => { const { mapHookRecords } = await import('../forwarder.js'); @@ -176,7 +197,10 @@ describe('mapHookRecords', () => { }; const rawPrompt = 'please just fix the thing, no ticket reference here'; - const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); + const record = buildHookRecord('UserPromptSubmit', 'sid1', { + prompt: rawPrompt, + cwd: '/repo/nonexistent-for-this-test', + }); const payload = await mapHookRecords([record], ctx, 0); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; @@ -188,7 +212,7 @@ describe('mapHookRecords', () => { it( 'falls back to the mention tier for a UserPromptSubmit record whose prompt has a bare ' + - 'ticket mention and the branch carries no ticket', + 'ticket mention and the branch carries no ticket', async () => { const { mapHookRecords } = await import('../forwarder.js'); @@ -210,4 +234,90 @@ describe('mapHookRecords', () => { expect(line.story_source).toBe('mention'); } ); + + it('keys agent.subagent.usage event_id off tool_use_id/agent_id from the synthetic record itself, not just byteOffset', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + // Two top-level subagents, neither carrying a sidecar tool_use_id, but with + // distinct agent_id — CR-018's fallback must keep these from colliding. + const record1 = buildHookRecord('SubagentStop', 'sid1', { + type: 'agent.subagent.usage', + tool_use_id: '', + agent_id: 'agent-1', + }); + const record2 = buildHookRecord('SubagentStop', 'sid1', { + type: 'agent.subagent.usage', + tool_use_id: '', + agent_id: 'agent-2', + }); + + const payload = await mapHookRecords([record1, record2], ctx, 0); + const [line1, line2] = payload.ndjson + .trim() + .split('\n') + .map((line) => JSON.parse(line) as MappedRecord); + + expect(line1.event_id).toBe('sid1:agent.subagent.usage:agent-1'); + expect(line2.event_id).toBe('sid1:agent.subagent.usage:agent-2'); + expect(line1.event_id).not.toBe(line2.event_id); + }); + + it('merges prepareAnalyticsFields common fields onto the mapped record when the hook agent name is registered', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + }; + + // The real registered OTLP agent name, unlike every other test in this file which + // deliberately uses the unregistered 'claude' (CR-016: that name resolves to `undefined`, + // so this is the only test exercising the real AgentRegistry.getAnalyticsAgent merge path). + const record = JSON.stringify({ + agentName: 'claude-code-otlp', + raw: JSON.stringify({ hook_event_name: 'Stop', session_id: 'sid1', cwd: '' }), + timestamp: Date.now(), + }); + + const payload = await mapHookRecords([record], ctx, 0); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.platform).toBe('claude-code'); + expect(line.client_version).toBe('1.2.3'); + }); + + it('carries ctx.identity through onto developer_name/identity_source for a non-jwt tier', async () => { + const { mapHookRecords } = await import('../forwarder.js'); + + // Pre-seeding ctx.identity (as resolveIdentityOnce's own cache would look once resolved) + // with a non-jwt tier result proves the wiring from ctx.identity onto the mapped record, + // independent of the identity-chain's own resolution logic (covered by identity.test.ts). + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + git: {}, + identity: { developerName: 'git-user@example.com', identitySource: 'git' as const }, + }; + + const record = buildHookRecord('Stop', 'sid1'); + + const payload = await mapHookRecords([record], ctx, 0); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.developer_name).toBe('git-user@example.com'); + expect(line.identity_source).toBe('git'); + }); }); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts index 5e3e27a97..15f420122 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts @@ -14,7 +14,8 @@ export function computeEventId( } case 'agent.subagent.usage': { const toolUseId = String(fields['tool_use_id'] ?? ''); - return `${sessionId}:agent.subagent.usage:${toolUseId}`; + const agentId = String(fields['agent_id'] ?? ''); + return `${sessionId}:agent.subagent.usage:${toolUseId || agentId}`; } case 'agent.session.summary': { const phase = String(fields['phase'] ?? ''); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts new file mode 100644 index 000000000..57b4a8748 --- /dev/null +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -0,0 +1,119 @@ +/** + * Per-forward-tick context: the CLI version banner, and the identity/story resolution glue + * `forwarder.ts`'s `mapHookRecords()` calls once per batch. Split out of `forwarder.ts` to keep + * that module under the documented 500-line structure cap (code-quality.md). + */ + +import { readFileSync } from 'node:fs'; +import { join } from 'node:path'; +import { getDirname } from '@/utils/paths.js'; +import type { SSOCredentials, JWTCredentials } from '@/providers/core/types.js'; +import { resolveIdentity, type IdentitySource } from './identity.js'; +import { + resolveExplicitStory, + resolveBranchStory, + resolveMarkerStory, + resolveMentionStory, +} from './story-resolver.js'; + +export interface ForwardContext { + credentials: SSOCredentials | JWTCredentials; + baseUrl: string; + projectName: string; + userEmail: string; + /** Per-session git info cache, resolved lazily from the first hook `cwd`. */ + git: { branch?: string; remote?: string }; + /** Per-session developer-identity cache, resolved once */ + identity?: { developerName?: string; identitySource?: IdentitySource }; + /** Per-tick story-id cache, resolved once per forward tick. */ + story?: { storyId?: string; storySource?: 'explicit' | 'branch' | '' }; +} + +/** This package's own `version` from the repo-root `package.json`, read once at import time. */ +export function loadCodemieCliVersion(): string { + try { + const packageJsonPath = join(getDirname(import.meta.url), '../../../../../../../package.json'); + const packageJson = JSON.parse(readFileSync(packageJsonPath, 'utf-8')) as { version?: string }; + return packageJson.version ?? ''; + } catch { + return ''; + } +} + +/** + * Per-record story-id override for `UserPromptSubmit` hook events only, + * layered on top of the per-tick `resolveStoryOnce()` cache in + * `ctx.story` (explicit/branch/''). Priority order across the full chain is + * explicit -> marker -> branch -> mention: + * + * 1. If the per-tick cache already resolved to `'explicit'`, that is the + * highest-priority result and wins outright. + * 2. Otherwise, try the marker tier (`story: X` / `ticket #X`) against this + * record's OWN prompt text — it sits above branch in priority. + * 3. Otherwise, if the per-tick cache resolved to `'branch'`, that wins (it + * is already correctly placed between marker and mention). + * 4. Otherwise, try the mention tier (bare ticket-shaped text) — the + * lowest-priority tier. + * 5. Otherwise, empty. + * + * Computed fresh per record and never mutates `ctx.story`: other records in + * the same batch still need that shared per-tick cache untouched. + */ +export function resolvePromptStory( + ctx: ForwardContext, + rawPrompt: string +): { storyId: string; storySource: string } { + if (ctx.story?.storySource === 'explicit') { + return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; + } + + const marker = resolveMarkerStory(rawPrompt); + if (marker) { + return { storyId: marker.storyId, storySource: marker.storySource }; + } + + if (ctx.story?.storySource === 'branch') { + return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; + } + + const mention = resolveMentionStory(rawPrompt); + if (mention) { + return { storyId: mention.storyId, storySource: mention.storySource }; + } + + return { storyId: '', storySource: '' }; +} + +/** + * Resolve and cache this forward tick's story id/source once, from the first record that carries + * a real `cwd` — guarded the same way {@link resolveIdentityOnce} is, so a synthetic + * transcript-derived record's empty `cwd` can never poison the cache for the rest of the batch + * (CR-019). + */ +export async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promise { + if (!cwd || ctx.story?.storyId !== undefined) return; + + const explicit = await resolveExplicitStory(cwd); + const resolved = explicit ?? resolveBranchStory(ctx.git.branch ?? ''); + + ctx.story = resolved + ? { storyId: resolved.storyId, storySource: resolved.storySource } + : { storyId: '', storySource: '' }; +} + +/** + * Resolve and cache this forward tick's developer identity once, from the first record that + * carries a real `cwd`. An empty `cwd` (every synthetic transcript-derived record) is skipped + * rather than cached, mirroring {@link resolveGitInfo}'s own `if (!cwd) return;` guard in + * `forwarder.ts` — otherwise the first such record in a batch would permanently cache the + * cwd-less (and therefore less accurate) git-tier result for every later record in the same tick + * (CR-019). + */ +export async function resolveIdentityOnce(ctx: ForwardContext, cwd: string): Promise { + if (!cwd) return; + if (!ctx.identity) ctx.identity = {}; + if (ctx.identity.developerName !== undefined) return; + const { developerName, identitySource } = await resolveIdentity(ctx.credentials, cwd); + ctx.identity.developerName = developerName; + ctx.identity.identitySource = identitySource; +} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 7ae58d206..6820ef63f 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -1,8 +1,5 @@ -import { readFileSync } from 'node:fs'; -import { join } from 'node:path'; import { logger } from '@/utils/logger.js'; import { sanitizeLogArgs } from '@/utils/security.js'; -import { getDirname } from '@/utils/paths.js'; import type { SSOCredentials, JWTCredentials } from '../../../../../core/types.js'; import { isSSOCredentials, isJWTCredentials } from '../../../../../core/types.js'; import { buildAuthHeaders } from '../../../../../core/codemie-auth-helpers.js'; @@ -16,24 +13,14 @@ import { OtlpHookSpoolData } from '../otlp.plugin.js'; import { snapshotPendingBytes, snapshotPendingHookRecords } from './spool-io.js'; import { areCredentialsStale, markCredentialsStale } from './auth-state.js'; import { computeEventId } from './event-id.js'; -import { decodeJwtClaims, resolveIdentity, type IdentitySource } from './identity.js'; +import { decodeJwtClaims } from './identity.js'; import { - resolveExplicitStory, - resolveBranchStory, - resolveMarkerStory, - resolveMentionStory, -} from './story-resolver.js'; - -/** This package's own `version` from the repo-root `package.json`, read once at import time. */ -function loadCodemieCliVersion(): string { - try { - const packageJsonPath = join(getDirname(import.meta.url), '../../../../../../../package.json'); - const packageJson = JSON.parse(readFileSync(packageJsonPath, 'utf-8')) as { version?: string }; - return packageJson.version ?? ''; - } catch { - return ''; - } -} + type ForwardContext, + loadCodemieCliVersion, + resolveIdentityOnce, + resolvePromptStory, + resolveStoryOnce, +} from './forward-context.js'; const CODEMIE_CLI_VERSION = loadCodemieCliVersion(); @@ -64,19 +51,6 @@ const FORWARD_TIMEOUT_MS = 20_000; type SendResult = 'ok' | 'failed' | 'auth-expired'; -interface ForwardContext { - credentials: SSOCredentials | JWTCredentials; - baseUrl: string; - projectName: string; - userEmail: string; - /** Per-session git info cache, resolved lazily from the first hook `cwd`. */ - git: { branch?: string; remote?: string }; - /** Per-session developer-identity cache, resolved once */ - identity?: { developerName?: string; identitySource?: IdentitySource }; - /** Per-tick story-id cache, resolved once per forward tick. */ - story?: { storyId?: string; storySource?: 'explicit' | 'branch' | '' }; -} - /* ------------------------------------------------------------------ auth --- */ function resolveUserEmail(credentials: SSOCredentials | JWTCredentials): string { @@ -247,69 +221,6 @@ async function resolveGitInfo(ctx: ForwardContext, cwd: string): Promise { } } -/** - * Per-record story-id override for `UserPromptSubmit` hook events only, - * layered on top of the per-tick `resolveStoryOnce()` cache in - * `ctx.story` (explicit/branch/''). Priority order across the full chain is - * explicit -> marker -> branch -> mention: - * - * 1. If the per-tick cache already resolved to `'explicit'`, that is the - * highest-priority result and wins outright. - * 2. Otherwise, try the marker tier (`story: X` / `ticket #X`) against this - * record's OWN prompt text — it sits above branch in priority. - * 3. Otherwise, if the per-tick cache resolved to `'branch'`, that wins (it - * is already correctly placed between marker and mention). - * 4. Otherwise, try the mention tier (bare ticket-shaped text) — the - * lowest-priority tier. - * 5. Otherwise, empty. - * - * Computed fresh per record and never mutates `ctx.story`: other records in - * the same batch still need that shared per-tick cache untouched. - */ -function resolvePromptStory( - ctx: ForwardContext, - rawPrompt: string -): { storyId: string; storySource: string } { - if (ctx.story?.storySource === 'explicit') { - return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; - } - - const marker = resolveMarkerStory(rawPrompt); - if (marker) { - return { storyId: marker.storyId, storySource: marker.storySource }; - } - - if (ctx.story?.storySource === 'branch') { - return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; - } - - const mention = resolveMentionStory(rawPrompt); - if (mention) { - return { storyId: mention.storyId, storySource: mention.storySource }; - } - - return { storyId: '', storySource: '' }; -} - -async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promise { - if (ctx.story?.storyId !== undefined) return; - - const explicit = await resolveExplicitStory(cwd); - const resolved = explicit ?? resolveBranchStory(ctx.git.branch ?? ''); - - ctx.story = resolved - ? { storyId: resolved.storyId, storySource: resolved.storySource } - : { storyId: '', storySource: '' }; -} - -async function resolveIdentityOnce(ctx: ForwardContext, cwd: string): Promise { - if (!ctx.identity) ctx.identity = {}; - if (ctx.identity.developerName !== undefined) return; - const { developerName, identitySource } = await resolveIdentity(ctx.credentials, cwd); - ctx.identity.developerName = developerName; - ctx.identity.identitySource = identitySource; -} - interface HookPayload { ndjson: string; containsSessionEnd: boolean; @@ -369,7 +280,19 @@ export async function mapHookRecords( const { AgentRegistry } = await import('../../../../../../agents/registry.js'); const analyticsAgent = AgentRegistry.getAnalyticsAgent(spoolData.agentName); - const commonFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; + let commonFields: Record = {}; + try { + commonFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; + } catch (err) { + // OtlpAgentAdapter.prepareAnalyticsFields's "must never throw" contract is only a doc + // comment — a future/alternate adapter implementation that violates it must not abort + // every remaining record in this forward tick. + const msg = err instanceof Error ? err.message : String(err); + logger.debug( + '[otlp-forwarder] prepareAnalyticsFields threw', + ...sanitizeLogArgs({ agentName: spoolData.agentName, err: msg }) + ); + } const limited = limitHookPayload(hookEvent); const type = hookEventType(hookName, hookEvent); @@ -393,7 +316,7 @@ export async function mapHookRecords( raw: limited, ...commonFields, schema_version: 2, - event_id: computeEventId(type, sessionId, { byteOffset }), + event_id: computeEventId(type, sessionId, { ...hookEvent, byteOffset }), codemie_cli_version: CODEMIE_CLI_VERSION, }) ); From 22c7db822bb5fe25d8f085815980f9e3bd0717e7 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Mon, 5 Oct 2026 16:27:22 +0300 Subject: [PATCH 03/35] fix(proxy): fix relative path issue with imports --- .../plugins/claude-code-otlp/transcript/usage-request.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 938144188..1fe84063d 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -16,8 +16,8 @@ * not itself check that two records it is asked to merge actually share that identity. */ -import { parseRoutingHeaders, type RoutingHeaderSource } from '@/utils/routing-headers.mjs'; -import { parseBackendModelName } from '@/utils/bedrock-pricing.mjs'; +import { parseRoutingHeaders, type RoutingHeaderSource } from '../../../../utils/routing-headers.mjs'; +import { parseBackendModelName } from '../../../../utils/bedrock-pricing.mjs'; import type { OpenUsageRequest } from './parse-state.js'; /** From ed03f06632345f531aa41c41b3aa111754f3913c Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Mon, 5 Oct 2026 16:37:24 +0300 Subject: [PATCH 04/35] refactor(proxy): rid of dynamic imports --- .../plugins/claude-code-otlp/claude-code-otlp.plugin.ts | 2 +- .../plugins/sso/proxy/plugins/otlp-spool/forwarder.ts | 8 ++++---- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index 91b8d6db9..daf4890af 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -10,6 +10,7 @@ import { forwardOtlpEventToSpool } from '../utils.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; import { runMainTranscriptParse, runSubagentTranscriptParse, type SubagentFile } from './transcript/orchestrator.js'; import { findSubagentFiles } from './transcript/subagent-usage.js'; +import { exec } from '@/utils/exec.js'; export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; @@ -145,7 +146,6 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { } try { - const { exec } = await import('@/utils/exec.js'); const result = await exec('claude', ['--version']); const trimmed = result.stdout.trim(); const versionMatch = trimmed.match(/^(\d+\.\d+\.\d+)/); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 6820ef63f..507f4e39f 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -21,6 +21,7 @@ import { resolvePromptStory, resolveStoryOnce, } from './forward-context.js'; +import { AgentRegistry } from '@/agents/registry.js'; const CODEMIE_CLI_VERSION = loadCodemieCliVersion(); @@ -278,11 +279,10 @@ export async function mapHookRecords( ? resolvePromptStory(ctx, rawPrompt) : { storyId: ctx.story?.storyId ?? '', storySource: ctx.story?.storySource ?? '' }; - const { AgentRegistry } = await import('../../../../../../agents/registry.js'); const analyticsAgent = AgentRegistry.getAnalyticsAgent(spoolData.agentName); - let commonFields: Record = {}; + let agentSpecificFields: Record = {}; try { - commonFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; + agentSpecificFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; } catch (err) { // OtlpAgentAdapter.prepareAnalyticsFields's "must never throw" contract is only a doc // comment — a future/alternate adapter implementation that violates it must not abort @@ -314,7 +314,7 @@ export async function mapHookRecords( cwd, prompt_body: boundedText(hookEvent['prompt'], MAX_PROMPT_CHARS), raw: limited, - ...commonFields, + ...agentSpecificFields, schema_version: 2, event_id: computeEventId(type, sessionId, { ...hookEvent, byteOffset }), codemie_cli_version: CODEMIE_CLI_VERSION, From 9ebdf9dca432f9cd3a72bf6dba44678321a5920b Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Mon, 5 Oct 2026 16:54:31 +0300 Subject: [PATCH 05/35] refactor(proxy): rename `computeEventId` to `resolveEventId` --- .../transcript/usage-request.ts | 2 +- .../otlp-spool/__tests__/event-id.test.ts | 22 +++++++++---------- .../sso/proxy/plugins/otlp-spool/event-id.ts | 2 +- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 4 ++-- 4 files changed, 15 insertions(+), 15 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 1fe84063d..66785a6ce 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -82,7 +82,7 @@ export function parseUsageLine( return null; } - // Both openRequests' key and computeEventId's agent.usage.request formula key on + // Both openRequests' key and resolveEventId's agent.usage.request formula key on // `${requestId}::${model}` — an empty requestId would collide every such line in the session // into one record instead of being skipped. const requestId = String(parsed.message?.id ?? ''); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts index 920e1fe18..8e4387e3c 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts @@ -1,17 +1,17 @@ import { describe, it, expect } from 'vitest'; -import { computeEventId } from '../event-id.js'; +import { resolveEventId } from '../event-id.js'; -describe('computeEventId', () => { +describe('resolveEventId', () => { describe('existing hook event types', () => { it('returns the byte-offset formula for an existing-event type', () => { - expect(computeEventId('agent.session.start', 'sid1', { byteOffset: 42 })).toBe( + expect(resolveEventId('agent.session.start', 'sid1', { byteOffset: 42 })).toBe( 'sid1:agent.session.start:42' ); }); it('returns a different id for a different byte offset', () => { - const first = computeEventId('agent.session.stop', 'sid1', { byteOffset: 0 }); - const second = computeEventId('agent.session.stop', 'sid1', { byteOffset: 128 }); + const first = resolveEventId('agent.session.stop', 'sid1', { byteOffset: 0 }); + const second = resolveEventId('agent.session.stop', 'sid1', { byteOffset: 128 }); expect(first).not.toBe(second); expect(first).toBe('sid1:agent.session.stop:0'); expect(second).toBe('sid1:agent.session.stop:128'); @@ -21,34 +21,34 @@ describe('computeEventId', () => { describe('agent.usage.request', () => { it('returns the request/model-keyed formula', () => { expect( - computeEventId('agent.usage.request', 'sid1', { request_id: 'req1', model: 'gpt-4' }) + resolveEventId('agent.usage.request', 'sid1', { request_id: 'req1', model: 'gpt-4' }) ).toBe('sid1:agent.usage.request:req1:gpt-4'); }); }); describe('agent.subagent.usage', () => { it('returns the tool_use_id-keyed formula', () => { - expect(computeEventId('agent.subagent.usage', 'sid1', { tool_use_id: 'tu1' })).toBe( + expect(resolveEventId('agent.subagent.usage', 'sid1', { tool_use_id: 'tu1' })).toBe( 'sid1:agent.subagent.usage:tu1' ); }); it('falls back to agent_id when tool_use_id is absent', () => { expect( - computeEventId('agent.subagent.usage', 'sid1', { tool_use_id: '', agent_id: 'agent-1' }) + resolveEventId('agent.subagent.usage', 'sid1', { tool_use_id: '', agent_id: 'agent-1' }) ).toBe('sid1:agent.subagent.usage:agent-1'); }); it('keeps two top-level subagents (no tool_use_id each) from colliding when agent_id differs', () => { - const first = computeEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-1' }); - const second = computeEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-2' }); + const first = resolveEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-1' }); + const second = resolveEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-2' }); expect(first).not.toBe(second); }); }); describe('agent.session.summary', () => { it('returns the phase-keyed formula', () => { - expect(computeEventId('agent.session.summary', 'sid1', { phase: 'end' })).toBe( + expect(resolveEventId('agent.session.summary', 'sid1', { phase: 'end' })).toBe( 'sid1:agent.session.summary:end' ); }); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts index 15f420122..4babb8a61 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts @@ -1,7 +1,7 @@ /** * Pure, deterministic `event_id` derivation for analytics events. */ -export function computeEventId( +export function resolveEventId( type: string, sessionId: string, fields: Record diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 507f4e39f..cc28cbeb1 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -12,7 +12,7 @@ import { import { OtlpHookSpoolData } from '../otlp.plugin.js'; import { snapshotPendingBytes, snapshotPendingHookRecords } from './spool-io.js'; import { areCredentialsStale, markCredentialsStale } from './auth-state.js'; -import { computeEventId } from './event-id.js'; +import { resolveEventId } from './event-id.js'; import { decodeJwtClaims } from './identity.js'; import { type ForwardContext, @@ -316,7 +316,7 @@ export async function mapHookRecords( raw: limited, ...agentSpecificFields, schema_version: 2, - event_id: computeEventId(type, sessionId, { ...hookEvent, byteOffset }), + event_id: resolveEventId(type, sessionId, { ...hookEvent, byteOffset }), codemie_cli_version: CODEMIE_CLI_VERSION, }) ); From a9c5b1e35d692463a5b37fe39a1ec4ad30ca08bb Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Mon, 5 Oct 2026 17:21:25 +0300 Subject: [PATCH 06/35] refactor(proxy): remove comments from `usage-request.ts` --- .../plugins/claude-code-otlp/transcript/usage-request.ts | 7 ------- 1 file changed, 7 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 66785a6ce..26335d5fa 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -122,9 +122,6 @@ export function parseUsageLine( agentId, // Sibling of usage on message, not nested inside it. stopReason: String(parsed.message?.stop_reason ?? ''), - // No confirmed source field for this on any sampled real transcript line (spec.md's own - // "Transcript field confidence gaps" flags it as unverified) — default false, and pick it up - // from a top-level `isApiError` boolean if a line ever carries one. isApiError: Boolean(parsed.isApiError ?? false), gitBranch: String(parsed.gitBranch ?? ''), }; @@ -139,10 +136,6 @@ export function parseUsageLine( * `stop_reason` once the turn finishes) wins, while a later row that is missing a field `a` had * does not blank it out. * - * `isApiError` follows the same "non-empty b wins, else a" shape as the non-numeric fields: once - * true on either record, it stays true across the merge (losing a true→false "fix" would hide a - * real API error from aggregation). - * * Returns a new object; neither `a` nor `b` is mutated. */ export function mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): OpenUsageRequest { From a6ee3d5254423f345b2c80c2ecee6f9a3ece26f2 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Tue, 6 Oct 2026 08:30:24 +0300 Subject: [PATCH 07/35] refactor: refactoring of ClaudeCodeOtlpPlugin --- .../__tests__/claude-code-otlp.plugin.test.ts | 26 +++- .../claude-code-otlp.plugin.ts | 118 +++++++++++------- .../claude-code-otlp.types.ts | 4 +- .../transcript/__tests__/orchestrator.test.ts | 108 ++++++---------- .../transcript/orchestrator.ts | 69 +++++----- 5 files changed, 176 insertions(+), 149 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index 127bb6262..94d767fbb 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -156,7 +156,9 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { beforeEach(() => { runMainTranscriptParseMock.mockReset(); + runMainTranscriptParseMock.mockResolvedValue([]); runSubagentTranscriptParseMock.mockReset(); + runSubagentTranscriptParseMock.mockResolvedValue([]); findSubagentFilesMock.mockReset(); findSubagentFilesMock.mockResolvedValue([]); vi.mocked(forwardOtlpEventToSpool).mockReset(); @@ -257,8 +259,6 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SessionEnd' })); await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - // The SessionEnd subagent-backstop sweep is fire-and-forget; flush microtasks. - await new Promise((resolve) => setImmediate(resolve)); expect(findSubagentFilesMock).toHaveBeenCalledWith('/tmp/transcript.jsonl'); expect(runSubagentTranscriptParseMock).toHaveBeenCalledTimes(2); @@ -279,4 +279,26 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { expect(runMainTranscriptParseMock).not.toHaveBeenCalled(); expect(forwardOtlpEventToSpool).toHaveBeenCalledWith(rawEvent, 'claude-code-otlp'); }); + + it('forwards every event a per-event handler returns (the raw event plus any derived events) through the single forwardToSpool path, in order', async () => { + const derivedUsageEvent = JSON.stringify({ type: 'agent.usage.request' }); + const derivedSummaryEvent = JSON.stringify({ type: 'agent.session.summary' }); + runMainTranscriptParseMock.mockResolvedValue([derivedUsageEvent, derivedSummaryEvent]); + + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'Stop' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + + // runMainTranscriptParse/runSubagentTranscriptParse never call forwardOtlpEventToSpool + // themselves (they are mocked here to just return data) — every event that reaches the spool + // mock arrived via forwardToSpool, called exactly once from processOtlpEvent. + expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(3); + expect(vi.mocked(forwardOtlpEventToSpool).mock.calls.map(([raw]) => raw)).toEqual([ + rawEvent, + derivedUsageEvent, + derivedSummaryEvent, + ]); + }); }); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index daf4890af..9d80b996a 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -5,7 +5,7 @@ import { logger } from '@/utils/logger.js'; import { ConfigLoader } from '@/utils/config.js'; import { AgentAdapterType, OtlpAdapterDeps, OtlpAgentAdapter } from '@/agents/core/types.js'; import { CLAUDE_CODE_OTLP_AGENT_NAME } from './claude-code-otlp.constants.js'; -import { ForwardDecision, toBaseClaudeCodeHookEvent } from './claude-code-otlp.types.js'; +import { BaseClaudeCodeHookEvent, ForwardDecision, toBaseClaudeCodeHookEvent } from './claude-code-otlp.types.js'; import { forwardOtlpEventToSpool } from '../utils.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; import { runMainTranscriptParse, runSubagentTranscriptParse, type SubagentFile } from './transcript/orchestrator.js'; @@ -52,58 +52,88 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { const event = toBaseClaudeCodeHookEvent(rawParsed); if (!event.sessionId) { - return { decision: 'forward', payload: rawEvent }; + return { decision: 'forward', payload: [rawEvent] }; } if (hookEventName === 'UserPromptSubmit') { return await this.onUserPromptSubmit(rawEvent); } - if ( - event.hookEventName === 'Stop' || - event.hookEventName === 'PreCompact' || - event.hookEventName === 'SessionEnd' || - event.hookEventName === 'StopFailure' - ) { - void runMainTranscriptParse( - event.sessionId, - event.transcriptPath, - event.hookEventName as 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure' - ); + let derivedEvents: string[] = []; + + switch (event.hookEventName) { + case 'Stop': + derivedEvents = await this.onStopEvent(event); + break; + case 'PreCompact': + derivedEvents = await this.onPreCompactEvent(event); + break; + case 'StopFailure': + derivedEvents = await this.onStopFailureEvent(event); + break; + case 'SessionEnd': + derivedEvents = await this.onSessionEndEvent(event); + break; + case 'SubagentStop': + derivedEvents = await this.onSubagentStopEvent(event, rawParsed as Record); + break; + default: + break; } - if (event.hookEventName === 'SessionEnd') { - void (async () => { - const subagentFiles = await findSubagentFiles(event.transcriptPath); - for (const file of subagentFiles) { - await runSubagentTranscriptParse(event.sessionId, event.transcriptPath, file); - } - })(); + return { + decision: 'forward', + payload: [rawEvent, ...derivedEvents], + }; + } + + private async onStopEvent(event: BaseClaudeCodeHookEvent): Promise { + return await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'Stop'); + } + + private async onPreCompactEvent(event: BaseClaudeCodeHookEvent): Promise { + return await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'PreCompact'); + } + + private async onStopFailureEvent(event: BaseClaudeCodeHookEvent): Promise { + return await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'StopFailure'); + } + + private async onSessionEndEvent(event: BaseClaudeCodeHookEvent): Promise { + const events = await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'SessionEnd'); + + // Backstop: guarantee every subagent discovered for this session gets at least one + // agent.subagent.usage event, even when its own SubagentStop hook never fired. + const subagentFiles = await findSubagentFiles(event.transcriptPath); + for (const file of subagentFiles) { + events.push(...(await runSubagentTranscriptParse(event.sessionId, event.transcriptPath, file))); } - if (event.hookEventName === 'SubagentStop') { - const rawRecord = rawParsed as Record; - const agentTranscriptPath = - typeof rawRecord['agent_transcript_path'] === 'string' ? rawRecord['agent_transcript_path'] : ''; - if (agentTranscriptPath) { - const agentId = - typeof rawRecord['agent_id'] === 'string' - ? rawRecord['agent_id'] - : basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''); - const subagentFile: SubagentFile = { - agentId, - filePath: agentTranscriptPath, - toolUseId: typeof rawRecord['tool_use_id'] === 'string' ? rawRecord['tool_use_id'] : undefined, - agentType: typeof rawRecord['agent_type'] === 'string' ? rawRecord['agent_type'] : undefined, - }; - void runSubagentTranscriptParse(event.sessionId, event.transcriptPath, subagentFile); - } + return events; + } + + private async onSubagentStopEvent( + event: BaseClaudeCodeHookEvent, + rawRecord: Record + ): Promise { + const agentTranscriptPath = + typeof rawRecord['agent_transcript_path'] === 'string' ? rawRecord['agent_transcript_path'] : ''; + if (!agentTranscriptPath) { + return []; } - return { - decision: 'forward', - payload: rawEvent, + const agentId = + typeof rawRecord['agent_id'] === 'string' + ? rawRecord['agent_id'] + : basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''); + const subagentFile: SubagentFile = { + agentId, + filePath: agentTranscriptPath, + toolUseId: typeof rawRecord['tool_use_id'] === 'string' ? rawRecord['tool_use_id'] : undefined, + agentType: typeof rawRecord['agent_type'] === 'string' ? rawRecord['agent_type'] : undefined, }; + + return await runSubagentTranscriptParse(event.sessionId, event.transcriptPath, subagentFile); } private async ensureProxyAuth(): Promise { @@ -116,9 +146,11 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { }); } - private forwardToSpool(rawEvent: string): void { + private forwardToSpool(rawEvents: string[]): void { // Intentionally not awaited: forwardOtlpEventToSpool is fire-and-forget. - forwardOtlpEventToSpool(rawEvent, CLAUDE_CODE_OTLP_AGENT_NAME); + for (const rawEvent of rawEvents) { + forwardOtlpEventToSpool(rawEvent, CLAUDE_CODE_OTLP_AGENT_NAME); + } } public async prepareAnalyticsFields( @@ -163,7 +195,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { if (authResult.ok) { return { decision: 'forward', - payload: rawEvent, + payload: [rawEvent], } } diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts index 9a9a96f54..a01f60298 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts @@ -46,7 +46,7 @@ interface RawBaseClaudeCodeHookEvent { /** * camelCase-keyed mirror of {@link RawBaseClaudeCodeHookEvent}. */ -interface BaseClaudeCodeHookEvent { +export interface BaseClaudeCodeHookEvent { sessionId: string; promptId?: string; transcriptPath: string; @@ -73,5 +73,5 @@ export function toBaseClaudeCodeHookEvent(raw: RawBaseClaudeCodeHookEvent): Base } export type ForwardDecision = - | { decision: 'forward'; payload: string } + | { decision: 'forward'; payload: string[] } | { decision: 'block'; reason: string, hookSpecificOutput: Record }; diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index da8d3a37c..9742d53e4 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -2,22 +2,18 @@ * Tests for `runMainTranscriptParse` — the `Stop`/`PreCompact`/`SessionEnd` main-transcript * orchestrator. * - * `forwardOtlpEventToSpool` (`@/agents/plugins/utils.js`) is mocked so no real network/daemon - * call happens; `CODEMIE_HOME` points at a fresh temp directory per test so `loadParseState`/ - * `saveParseState` never touch the real `~/.codemie`. + * Neither `runMainTranscriptParse` nor `runSubagentTranscriptParse` writes to the spool itself — + * each returns the raw JSON strings it wants forwarded, and the caller (the plugin's + * `processOtlpEvent`) is the only place that actually forwards them. So these tests read the + * returned array directly; no network/daemon mocking is needed. `CODEMIE_HOME` points at a fresh + * temp directory per test so `loadParseState`/`saveParseState` never touch the real `~/.codemie`. */ -import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'; import { tmpdir } from 'node:os'; import { join } from 'node:path'; -const forwardMock = vi.fn().mockResolvedValue(undefined); - -vi.mock('@/agents/plugins/utils.js', () => ({ - forwardOtlpEventToSpool: forwardMock, -})); - let codemieHome: string; let transcriptDir: string; @@ -68,7 +64,6 @@ beforeEach(() => { codemieHome = mkdtempSync(join(tmpdir(), 'codemie-home-')); process.env.CODEMIE_HOME = codemieHome; transcriptDir = mkdtempSync(join(tmpdir(), 'codemie-transcript-')); - forwardMock.mockClear(); }); afterEach(() => { @@ -85,8 +80,8 @@ function writeTranscript(fileName: string, lines: string[]): string { type ForwardedEvent = Record; -function forwardedEvents(): ForwardedEvent[] { - return forwardMock.mock.calls.map(([raw]) => JSON.parse(raw as string) as ForwardedEvent); +function parseAll(raw: string[]): ForwardedEvent[] { + return raw.map((r) => JSON.parse(r) as ForwardedEvent); } /** @@ -110,7 +105,7 @@ function writeSubagentFixture( } describe('runMainTranscriptParse — idempotent reparse', () => { - it('forwards agent.usage.request events with identical request_id/model pairs across a crash-before-save re-parse', async () => { + it('returns agent.usage.request events with identical request_id/model pairs across a crash-before-save re-parse', async () => { const { runMainTranscriptParse } = await import('../orchestrator.js'); const { saveParseState, createParseState } = await import('../parse-state.js'); @@ -121,9 +116,9 @@ describe('runMainTranscriptParse — idempotent reparse', () => { usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75, stopReason: 'end_turn' }), ]); - await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + const first = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); - const firstPairs = forwardedEvents() + const firstPairs = first .filter((e) => e.type === 'agent.usage.request') .map((e) => `${e.request_id}::${e.model}`) .sort(); @@ -134,11 +129,10 @@ describe('runMainTranscriptParse — idempotent reparse', () => { // there, but the persisted state is wound back to fresh (as if the first run's save never // happened). await saveParseState(sessionId, createParseState()); - forwardMock.mockClear(); - await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + const second = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); - const secondPairs = forwardedEvents() + const secondPairs = second .filter((e) => e.type === 'agent.usage.request') .map((e) => `${e.request_id}::${e.model}`) .sort(); @@ -149,7 +143,7 @@ describe('runMainTranscriptParse — idempotent reparse', () => { }); describe('runMainTranscriptParse — Stop trigger', () => { - it('forwards one agent.usage.request event per distinct request plus one incremental agent.session.summary', async () => { + it('returns one agent.usage.request event per distinct request plus one incremental agent.session.summary', async () => { const { runMainTranscriptParse } = await import('../orchestrator.js'); const sessionId = 'session-stop-basic'; @@ -158,11 +152,10 @@ describe('runMainTranscriptParse — Stop trigger', () => { usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), ]); - await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + const raw = await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + expect(raw).toHaveLength(3); - expect(forwardMock).toHaveBeenCalledTimes(3); - - const events = forwardedEvents(); + const events = parseAll(raw); const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); @@ -174,7 +167,7 @@ describe('runMainTranscriptParse — Stop trigger', () => { }); describe('runMainTranscriptParse — PreCompact trigger', () => { - it('forwards usage-request events but never a session summary', async () => { + it('returns usage-request events but never a session summary', async () => { const { runMainTranscriptParse } = await import('../orchestrator.js'); const sessionId = 'session-precompact'; @@ -182,9 +175,7 @@ describe('runMainTranscriptParse — PreCompact trigger', () => { usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), ]); - await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); - - const events = forwardedEvents(); + const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact')); const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); @@ -204,10 +195,9 @@ describe('runMainTranscriptParse — compaction_count', () => { await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); - forwardMock.mockClear(); - await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); - const summaryEvents = forwardedEvents().filter((e) => e.type === 'agent.session.summary'); + const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); expect(summaryEvents).toHaveLength(1); expect(summaryEvents[0].compaction_count).toBe(2); }); @@ -223,9 +213,7 @@ describe('runMainTranscriptParse — api_calls', () => { usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), ]); - await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); - - const events = forwardedEvents(); + const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); @@ -236,7 +224,7 @@ describe('runMainTranscriptParse — api_calls', () => { }); describe('runMainTranscriptParse — SessionEnd trigger', () => { - it('forwards a final-phase summary with an ended_at key present', async () => { + it('returns a final-phase summary with an ended_at key present', async () => { const { runMainTranscriptParse } = await import('../orchestrator.js'); const sessionId = 'session-end'; @@ -244,9 +232,7 @@ describe('runMainTranscriptParse — SessionEnd trigger', () => { usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), ]); - await runMainTranscriptParse(sessionId, transcriptPath, 'SessionEnd'); - - const events = forwardedEvents(); + const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'SessionEnd')); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); expect(summaryEvents).toHaveLength(1); @@ -257,16 +243,14 @@ describe('runMainTranscriptParse — SessionEnd trigger', () => { }); describe('runMainTranscriptParse — missing transcript file', () => { - it('resolves cleanly without ever forwarding anything, for a trigger that never emits a summary', async () => { + it('resolves cleanly to an empty array, for a trigger that never emits a summary', async () => { const { runMainTranscriptParse } = await import('../orchestrator.js'); const { loadParseState } = await import('../parse-state.js'); const sessionId = 'session-missing-file'; const missingPath = join(transcriptDir, 'does-not-exist.jsonl'); - await expect(runMainTranscriptParse(sessionId, missingPath, 'PreCompact')).resolves.toBeUndefined(); - - expect(forwardMock).not.toHaveBeenCalled(); + await expect(runMainTranscriptParse(sessionId, missingPath, 'PreCompact')).resolves.toEqual([]); const state = await loadParseState(sessionId); expect(state.mainOffset).toBe(0); @@ -278,7 +262,7 @@ describe('runMainTranscriptParse — missing transcript file', () => { const sessionId = 'session-missing-file-stop'; const missingPath = join(transcriptDir, 'also-does-not-exist.jsonl'); - await expect(runMainTranscriptParse(sessionId, missingPath, 'Stop')).resolves.toBeUndefined(); + await expect(runMainTranscriptParse(sessionId, missingPath, 'Stop')).resolves.toBeInstanceOf(Array); }); }); @@ -311,9 +295,9 @@ describe('runMainTranscriptParse — tool-call accumulation', () => { }); const transcriptPath = writeTranscript('transcript-tools.jsonl', [toolLine, resultLine]); - await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); - const summary = forwardedEvents().find((e) => e.type === 'agent.session.summary'); + const summary = events.find((e) => e.type === 'agent.session.summary'); expect(summary).toBeDefined(); expect(summary?.files_written).toEqual(['/repo/a.ts']); expect(summary?.files_changed).toEqual(['/repo/b.ts']); @@ -325,7 +309,7 @@ describe('runMainTranscriptParse — tool-call accumulation', () => { }); describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, one pre-advanced)', () => { - it('forwards exactly three agent.subagent.usage events — one per subagent, including the one whose own SubagentStop already advanced its offset — never a fourth', async () => { + it('returns exactly three agent.subagent.usage events — one per subagent, including the one whose own SubagentStop already advanced its offset — never a fourth', async () => { const { runSubagentTranscriptParse } = await import('../orchestrator.js'); const { findSubagentFiles } = await import('../subagent-usage.js'); @@ -349,19 +333,17 @@ describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, o if (!a1File) throw new Error('fixture missing a1'); await runSubagentTranscriptParse(sessionId, mainTranscriptPath, a1File); - // Clear the setup call's forwards so they don't pollute the backstop assertion below. - forwardMock.mockClear(); - // Exercise exactly what the plugin's SessionEnd branch does: discover every subagent file // for the session and re-run the subagent parse for each one, unconditionally — the // crashed/missed-hook backstop. const allFiles = await findSubagentFiles(mainTranscriptPath); expect(allFiles).toHaveLength(3); + const backstopEvents: ForwardedEvent[] = []; for (const file of allFiles) { - await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file); + backstopEvents.push(...parseAll(await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file))); } - const subagentUsageEvents = forwardedEvents().filter((e) => e.type === 'agent.subagent.usage'); + const subagentUsageEvents = backstopEvents.filter((e) => e.type === 'agent.subagent.usage'); // Exactly three — a1 (already-advanced, re-summarized rather than skipped), a2, a3. Never a // fourth (no duplicate re-send for a1). expect(subagentUsageEvents).toHaveLength(3); @@ -372,7 +354,7 @@ describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, o }); describe('runSubagentTranscriptParse — no new bytes since last run', () => { - it('forwards zero new agent.usage.request events on a no-op reparse, but still forwards exactly one agent.subagent.usage event summarizing unchanged cumulative usage', async () => { + it('returns zero new agent.usage.request events on a no-op reparse, but still exactly one agent.subagent.usage event summarizing unchanged cumulative usage', async () => { const { runSubagentTranscriptParse } = await import('../orchestrator.js'); const { findSubagentFiles } = await import('../subagent-usage.js'); @@ -385,25 +367,19 @@ describe('runSubagentTranscriptParse — no new bytes since last run', () => { const [file] = await findSubagentFiles(mainTranscriptPath); - await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file); - - const firstEvents = forwardedEvents(); + const firstEvents = parseAll(await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file)); expect(firstEvents.filter((e) => e.type === 'agent.usage.request')).toHaveLength(2); expect(firstEvents.filter((e) => e.type === 'agent.subagent.usage')).toHaveLength(1); - forwardMock.mockClear(); - // Re-run on the same subagent file with no new content appended since the last call (its // offset is now at EOF). - await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file); - - const secondEvents = forwardedEvents(); + const secondEvents = parseAll(await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file)); const secondUsageRequests = secondEvents.filter((e) => e.type === 'agent.usage.request'); const secondSubagentUsage = secondEvents.filter((e) => e.type === 'agent.subagent.usage'); // No new lines means no new/touched openRequests keys at the agent.usage.request layer. expect(secondUsageRequests).toHaveLength(0); - // But the agent.subagent.usage event is still forwarded exactly once, summarizing the same + // But the agent.subagent.usage event is still returned exactly once, summarizing the same // cumulative (unchanged) usage. expect(secondSubagentUsage).toHaveLength(1); expect(secondSubagentUsage[0].output_tokens).toBe(30); @@ -412,7 +388,7 @@ describe('runSubagentTranscriptParse — no new bytes since last run', () => { }); describe('runSubagentTranscriptParse — missing subagent transcript file', () => { - it('resolves without throwing and still forwards one empty-usage agent.subagent.usage event, with no agent.usage.request events', async () => { + it('resolves without throwing and still returns one empty-usage agent.subagent.usage event, with no agent.usage.request events', async () => { const { runSubagentTranscriptParse } = await import('../orchestrator.js'); const sessionId = 'session-subagent-missing'; @@ -422,11 +398,9 @@ describe('runSubagentTranscriptParse — missing subagent transcript file', () = filePath: join(transcriptDir, sessionId, 'subagents', 'agent-ghost.jsonl'), }; - await expect( - runSubagentTranscriptParse(sessionId, mainTranscriptPath, missingSubagentFile) - ).resolves.toBeUndefined(); - - const events = forwardedEvents(); + const events = parseAll( + await runSubagentTranscriptParse(sessionId, mainTranscriptPath, missingSubagentFile) + ); const usageRequestEvents = events.filter((e) => e.type === 'agent.usage.request'); const subagentUsageEvents = events.filter((e) => e.type === 'agent.subagent.usage'); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 3b5630877..75fbf1b20 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -5,11 +5,13 @@ * Each hook fire is a fresh CLI process, so this module reloads persisted parse state * (`./parse-state.js`, Task 6), reads only the transcript lines appended since the last * persisted `mainOffset` (`./transcript-reader.js`, Task 7), derives/merges - * `agent.usage.request` records for those new lines (`./usage-request.js`, Task 8), forwards one - * event per completed request, optionally forwards one `agent.session.summary` event - * (`./session-summary.js`, Task 10), and persists state back — all before returning. + * `agent.usage.request` records for those new lines (`./usage-request.js`, Task 8), persists + * state back, and RETURNS one JSON string per completed request plus (on `Stop`/`SessionEnd`) + * one `agent.session.summary` event — it never forwards anything to the spool itself. The caller + * (the plugin's `processOtlpEvent`, via its per-event handlers) owns forwarding, so there is + * exactly one place in the whole analytics pipeline that writes to the spool. * - * Never throws: every path is wrapped so a read/parse/forward failure degrades to a no-op rather + * Never throws: every path is wrapped so a read/parse failure degrades to an empty result rather * than interrupting the hook that triggered it (`processOtlpEvent` must never block or fail on * this). */ @@ -25,8 +27,6 @@ import { type NamedInvocationCounts, } from './session-summary.js'; import { extractNamedInvocations } from '@/agents/plugins/claude/session/claude-named-invocations.js'; -import { forwardOtlpEventToSpool } from '../../utils.js'; -import { CLAUDE_CODE_OTLP_AGENT_NAME } from '../claude-code-otlp.constants.js'; import { type SubagentFile, buildSubagentUsageEvent } from './subagent-usage.js'; // Re-exported so callers (e.g. claude-code-otlp.plugin.ts) can import both `SubagentFile` and @@ -184,13 +184,16 @@ async function buildFullAccumulator( * keyed by `${requestId}::${model}` (matching `parse-state.ts`'s documented key shape), and * updates `state.branchCounts` from every new line's `gitBranch` (regardless of whether that * line carried usage). - * - Forwards one `agent.usage.request` event per request key touched by this pass. - * - On `Stop`/`SessionEnd` only, forwards exactly one `agent.session.summary` event + * - Returns one `agent.usage.request` JSON string per request key touched by this pass. + * - On `Stop`/`SessionEnd` only, also returns exactly one `agent.session.summary` event * (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`) built from a fresh full-file - * recompute (see {@link buildFullAccumulator} and Note B). `PreCompact` never forwards a + * recompute (see {@link buildFullAccumulator} and Note B). `PreCompact` never returns a * summary. * - Persists state back to disk. * + * Never forwards anything itself — the caller is responsible for sending the returned events to + * the spool (exactly one place in the pipeline does that). + * * Scoping ruling (Note A — a judgment call, since no file in this codebase documents a reliable * signal for when a *main*-transcript turn enters/exits a "skill context"): every * main-transcript-derived usage record in this task is scoped as `scopeKind: 'main'`, @@ -204,12 +207,11 @@ export async function runMainTranscriptParse( sessionId: string, transcriptPath: string, trigger: MainTranscriptTrigger -): Promise { +): Promise { try { - // Save-before-send, and both the load and the save happen inside the lock so a - // concurrent hook process for the same session can never read a state this pass is about to - // overwrite. Forwarding (network I/O) deliberately happens after the lock is released. - const eventsToForward = await withParseStateLock(sessionId, async () => { + // Both the load and the save happen inside the lock so a concurrent hook process for the + // same session can never read a state this pass is about to overwrite. + return await withParseStateLock(sessionId, async () => { const state = await loadParseState(sessionId); const { lines, nextOffset } = await readNewLines(transcriptPath, state.mainOffset); @@ -270,12 +272,9 @@ export async function runMainTranscriptParse( await saveParseState(sessionId, state); return events; }); - - for (const raw of eventsToForward) { - await forwardOtlpEventToSpool(raw, CLAUDE_CODE_OTLP_AGENT_NAME); - } } catch { // Swallow everything — never throw into processOtlpEvent. + return []; } } @@ -372,19 +371,22 @@ async function scanSubagentTranscript(filePath: string): Promise { +): Promise { try { - // Save-before-send, load/mutate/save inside the lock — same rationale as - // runMainTranscriptParse: a sibling SubagentStop for another subagent in this same session - // must never read state this pass is about to overwrite. - const eventsToForward = await withParseStateLock(sessionId, async () => { + // Load/mutate/save inside the lock — same rationale as runMainTranscriptParse: a sibling + // SubagentStop for another subagent in this same session must never read state this pass is + // about to overwrite. + return await withParseStateLock(sessionId, async () => { const state = await loadParseState(sessionId); const fromOffset = state.subagentOffsets[subagentFile.agentId] ?? 0; const { lines, nextOffset } = await readNewLines(subagentFile.filePath, fromOffset); @@ -451,11 +453,8 @@ export async function runSubagentTranscriptParse( await saveParseState(sessionId, state); return events; }); - - for (const raw of eventsToForward) { - await forwardOtlpEventToSpool(raw, CLAUDE_CODE_OTLP_AGENT_NAME); - } } catch { // Swallow everything — never throw into processOtlpEvent. + return []; } } From f45ed4cc9f147766f7aa745e7cd47ea1278e8b12 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Tue, 6 Oct 2026 08:32:40 +0300 Subject: [PATCH 08/35] fix(proxy): fix issue with AgentRegistry import --- src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index cc28cbeb1..0ea476609 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -21,7 +21,6 @@ import { resolvePromptStory, resolveStoryOnce, } from './forward-context.js'; -import { AgentRegistry } from '@/agents/registry.js'; const CODEMIE_CLI_VERSION = loadCodemieCliVersion(); @@ -279,6 +278,7 @@ export async function mapHookRecords( ? resolvePromptStory(ctx, rawPrompt) : { storyId: ctx.story?.storyId ?? '', storySource: ctx.story?.storySource ?? '' }; + const { AgentRegistry } = await import('@/agents/registry.js'); const analyticsAgent = AgentRegistry.getAnalyticsAgent(spoolData.agentName); let agentSpecificFields: Record = {}; try { From 475a0b02a8c60b2497883fb8eff181f0ef6881de Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Tue, 6 Oct 2026 09:03:51 +0300 Subject: [PATCH 09/35] refactor(proxy): remove sdlc-factory specific comments --- .../transcript/__tests__/subagent-usage.test.ts | 2 +- .../claude-code-otlp/transcript/session-summary.ts | 13 ++++++------- .../claude-code-otlp/transcript/subagent-usage.ts | 6 +++--- .../claude-code-otlp/transcript/usage-request.ts | 6 +++--- 4 files changed, 13 insertions(+), 14 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts index 4f21fab03..9d607ce2b 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts @@ -31,7 +31,7 @@ afterEach(async () => { } }); -/** Build one transcript JSONL usage line, mirroring Task 8's confirmed transcript shape. */ +/** Build one transcript JSONL usage line, mirroring the confirmed transcript shape. */ function usageLine(messageId: string, inputTokens: number, outputTokens: number): string { return JSON.stringify({ sessionId: 'session-subagent-1', diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts index 52a0c5589..b6810c35b 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts @@ -3,7 +3,7 @@ * * Unlike `agent.usage.request`/`agent.subagent.usage`, this event is a running, mutable * aggregate over an entire session, not a one-shot derivation from a single transcript line - * or file. The caller (a later task — the transcript-reader orchestrator) is responsible for + * or file. The caller (the transcript-reader orchestrator) is responsible for * accumulating a {@link SessionSummaryAccumulator} across the session's parsed transcript lines * and tool-use/tool-result payloads, tracking `TranscriptParseState.branchCounts` via * {@link updateBranchCounts} as `git_branch` changes per record, and running @@ -12,11 +12,10 @@ * consumes. This module only aggregates/derives the final event shape from already-computed * inputs — it never reads a transcript file or calls `extractNamedInvocations()` itself. * - * `event_id`/`schema_version` are stamped later, daemon-side (see Task 1) — this builder's output + * `event_id`/`schema_version` are stamped later, daemon-side — this builder's output * carries only an explicit `type` field. * - * Field-shape rulings (see spec.md's `agent.session.summary` section and this task's own plan - * entry for the full reasoning): + * Field-shape rulings (see spec.md's `agent.session.summary` section for the full reasoning): * - `models_used` is emitted as the full `acc.models` count map (not just a list of names) — * preserves count information `primary_model` alone would discard, consistent with how * `tool_calls`/`tool_errors`-style maps are emitted elsewhere in this stage. @@ -33,12 +32,12 @@ * - `title` has no identified source anywhere in this codebase or the external data-model doc * (per spec.md's Open risks) — always emitted as a literal empty string, never fabricated. * - `api_calls` (count of `agent.usage.request` records this session) is intentionally OMITTED - * from this builder's output: the plan's `buildSessionSummaryEvent` signature has no parameter + * from this builder's output: `buildSessionSummaryEvent`'s signature has no parameter * for it, and neither `acc` nor any other input here carries a request count. It is left for - * the orchestrator (a later task) to merge in afterward, since only that caller has visibility + * the orchestrator to merge in afterward, since only that caller has visibility * into the full set of `agent.usage.request` records it has derived/forwarded this session. * - `client_version`/`codemie_cli_version` are common fields stamped later via - * `mapHookRecords()` (see Tasks 1-3) — not this builder's responsibility either. + * `mapHookRecords()` — not this builder's responsibility either. */ import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; diff --git a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts index 1a3417a19..f567427ed 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts @@ -2,13 +2,13 @@ * `agent.subagent.usage` discovery and event builder. * * Discovery (`findSubagentFiles`) follows the same path convention as the existing, - * `private`/unexported `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts:384` + * `private`/unexported `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts` * (`//subagents/agent-*.jsonl` + sibling `.meta.json`), but is a * fresh, smaller implementation returning only the narrower {@link SubagentFile} shape this - * task's event needs — no `parentAgentId`/`requestShape`/`requestNonInteractive`. + * event needs — no `parentAgentId`/`requestShape`/`requestNonInteractive`. * * The event builder (`buildSubagentUsageEvent`) aggregates token/cache fields by summing an - * already-scoped `OpenUsageRequest[]` (Task 8's shape, reused — not redefined), and passes + * already-scoped `OpenUsageRequest[]` (`./usage-request.ts`'s shape, reused — not redefined), and passes * caller-built `tool_calls`/`tool_errors`/`skills_invoked` maps through verbatim: this module has * no access to a subagent's own tool-use/tool-error/skill-invocation occurrences, only to the * aggregates its caller already computed from that subagent's transcript. diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 26335d5fa..224775c29 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -5,7 +5,7 @@ * per-API-response usage shape the existing cost-reporting parser * (`src/cli/commands/analytics/cost/usage-readers.ts`'s `ClaudeRawMessage`/ * `extractClaudeUsageRecords`) already reads, adapted here to this task's - * {@link OpenUsageRequest} shape (Task 6, `parse-state.ts`) instead of that pipeline's + * {@link OpenUsageRequest} shape (`parse-state.ts`) instead of that pipeline's * `UsageRecord`. * * Claude Code can write more than one JSONL row for the same API response (progressive @@ -90,7 +90,7 @@ export function parseUsageLine( return null; } - // Same resolution chain the statusline and usage-readers.ts:188 already use: + // Same resolution chain the statusline and usage-readers.ts already use: // parseBackendModelName() (the raw LiteLLM backend id, when the proxy injected one) wins over // the transcript's own literal `message.model`, since it reflects the actual billable backend // model for a routed/capable-tier request. `modelRaw` keeps the literal, unresolved alias. @@ -165,7 +165,7 @@ export function mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): Ope /** * Build the `agent.usage.request` event payload for `req`. Carries its own explicit `type`, so - * the daemon-side `mapHookRecords()` (Task 1) stamps `event_id`/`schema_version` onto it later — + * the daemon-side `mapHookRecords()` stamps `event_id`/`schema_version` onto it later — * this function deliberately does not set either. */ export function buildUsageRequestEvent(sessionId: string, req: OpenUsageRequest): Record { From d89201b917d1fef6d0ca4b4b1a2430b00118a8d2 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Tue, 6 Oct 2026 09:26:35 +0300 Subject: [PATCH 10/35] refactor(proxy): update codemie cli version resolver --- .../otlp-spool/__tests__/forwarder.test.ts | 39 ++++++++++++++++++- .../plugins/otlp-spool/forward-context.ts | 15 +++---- 2 files changed, 46 insertions(+), 8 deletions(-) diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index 9331d42a3..1d6ca98d3 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -1,4 +1,15 @@ -import { describe, it, expect, vi } from 'vitest'; +import { describe, it, expect, vi, beforeEach } from 'vitest'; + +const execSyncMock = vi.fn(); + +vi.mock('node:child_process', () => ({ + execSync: execSyncMock, +})); + +beforeEach(() => { + execSyncMock.mockReset(); + execSyncMock.mockReturnValue('0.15.6'); +}); // Only 'claude-code-otlp' (the real registered OTLP agent name) resolves to an agent; // every other name — including 'claude', which every other test in this file deliberately @@ -45,6 +56,32 @@ function buildHookRecord(hookEventName: string, sessionId: string, extra: Record }); } +describe('loadCodemieCliVersion', () => { + beforeEach(() => { + execSyncMock.mockReset(); + execSyncMock.mockReturnValue('0.15.6'); + }); + + it('reads the installed CLI version via `codemie --version` and strips the semver', async () => { + const { loadCodemieCliVersion } = await import('../forward-context.js'); + + expect(loadCodemieCliVersion()).toBe('0.15.6'); + expect(execSyncMock).toHaveBeenCalledWith('codemie --version', { + encoding: 'utf8', + stdio: ['pipe', 'pipe', 'pipe'], + }); + }); + + it('falls back to an empty string when `codemie --version` throws', async () => { + execSyncMock.mockImplementation(() => { + throw new Error('ENOENT'); + }); + + const { loadCodemieCliVersion } = await import('../forward-context.js'); + expect(loadCodemieCliVersion()).toBe(''); + }); +}); + describe('mapHookRecords', () => { it('stamps schema_version, event_id, and codemie_cli_version on every mapped record', async () => { const { mapHookRecords } = await import('../forwarder.js'); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts index 57b4a8748..42563ff6a 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -4,9 +4,7 @@ * that module under the documented 500-line structure cap (code-quality.md). */ -import { readFileSync } from 'node:fs'; -import { join } from 'node:path'; -import { getDirname } from '@/utils/paths.js'; +import { execSync } from 'node:child_process'; import type { SSOCredentials, JWTCredentials } from '@/providers/core/types.js'; import { resolveIdentity, type IdentitySource } from './identity.js'; import { @@ -29,12 +27,15 @@ export interface ForwardContext { story?: { storyId?: string; storySource?: 'explicit' | 'branch' | '' }; } -/** This package's own `version` from the repo-root `package.json`, read once at import time. */ +/** The installed CodeMie CLI version, resolved once at import time. */ export function loadCodemieCliVersion(): string { try { - const packageJsonPath = join(getDirname(import.meta.url), '../../../../../../../package.json'); - const packageJson = JSON.parse(readFileSync(packageJsonPath, 'utf-8')) as { version?: string }; - return packageJson.version ?? ''; + const output = execSync('codemie --version', { + encoding: 'utf8', + stdio: ['pipe', 'pipe', 'pipe'], + }).trim(); + const versionMatch = output.match(/(\d+\.\d+\.\d+)/); + return versionMatch ? versionMatch[1] : output; } catch { return ''; } From 0a7926d4c5d5a5291fa84b68b760980abe989414 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Tue, 6 Oct 2026 09:55:56 +0300 Subject: [PATCH 11/35] refactor(proxy): simplify ForwardContext --- .../plugins/otlp-spool/__tests__/forwarder.test.ts | 4 ++-- .../sso/proxy/plugins/otlp-spool/forward-context.ts | 10 ++++------ 2 files changed, 6 insertions(+), 8 deletions(-) diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index 1d6ca98d3..067b32351 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -13,7 +13,7 @@ beforeEach(() => { // Only 'claude-code-otlp' (the real registered OTLP agent name) resolves to an agent; // every other name — including 'claude', which every other test in this file deliberately -// uses — resolves to `undefined`, matching the real AgentRegistry's behavior (CR-016). +// uses — resolves to `undefined`, matching the real AgentRegistry's behavior. vi.mock('@/agents/registry.js', () => ({ AgentRegistry: { getAnalyticsAgent: (agentName: string) => @@ -284,7 +284,7 @@ describe('mapHookRecords', () => { }; // Two top-level subagents, neither carrying a sidecar tool_use_id, but with - // distinct agent_id — CR-018's fallback must keep these from colliding. + // distinct agent_id — the fallback must keep these from colliding. const record1 = buildHookRecord('SubagentStop', 'sid1', { type: 'agent.subagent.usage', tool_use_id: '', diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts index 42563ff6a..cded667e7 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -24,7 +24,7 @@ export interface ForwardContext { /** Per-session developer-identity cache, resolved once */ identity?: { developerName?: string; identitySource?: IdentitySource }; /** Per-tick story-id cache, resolved once per forward tick. */ - story?: { storyId?: string; storySource?: 'explicit' | 'branch' | '' }; + story?: { storyId?: string; storySource?: 'explicit' | 'branch' }; } /** The installed CodeMie CLI version, resolved once at import time. */ @@ -88,8 +88,7 @@ export function resolvePromptStory( /** * Resolve and cache this forward tick's story id/source once, from the first record that carries * a real `cwd` — guarded the same way {@link resolveIdentityOnce} is, so a synthetic - * transcript-derived record's empty `cwd` can never poison the cache for the rest of the batch - * (CR-019). + * transcript-derived record's empty `cwd` can never poison the cache for the rest of the batch. */ export async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promise { if (!cwd || ctx.story?.storyId !== undefined) return; @@ -99,7 +98,7 @@ export async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promis ctx.story = resolved ? { storyId: resolved.storyId, storySource: resolved.storySource } - : { storyId: '', storySource: '' }; + : { storyId: '' }; } /** @@ -107,8 +106,7 @@ export async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promis * carries a real `cwd`. An empty `cwd` (every synthetic transcript-derived record) is skipped * rather than cached, mirroring {@link resolveGitInfo}'s own `if (!cwd) return;` guard in * `forwarder.ts` — otherwise the first such record in a batch would permanently cache the - * cwd-less (and therefore less accurate) git-tier result for every later record in the same tick - * (CR-019). + * cwd-less (and therefore less accurate) git-tier result for every later record in the same tick. */ export async function resolveIdentityOnce(ctx: ForwardContext, cwd: string): Promise { if (!cwd) return; From 5fd042094e5156d6b0b580882436a6f29af419fe Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Tue, 6 Oct 2026 10:23:57 +0300 Subject: [PATCH 12/35] refactor(proxy): remove sdlc-factory specific comments --- .../transcript/__tests__/orchestrator.test.ts | 2 +- .../transcript/orchestrator.ts | 35 +++++++++---------- .../otlp-spool/__tests__/forwarder.test.ts | 2 +- .../plugins/otlp-spool/forward-context.ts | 2 +- 4 files changed, 20 insertions(+), 21 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index 9742d53e4..be970da41 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -87,7 +87,7 @@ function parseAll(raw: string[]): ForwardedEvent[] { /** * Write a subagent fixture transcript (plus its sidecar `.meta.json`) under * `//subagents/agent-.jsonl`, matching - * `findSubagentFiles()`'s own discovery convention (Task 9). Returns the `SubagentFile` shape + * `findSubagentFiles()`'s own discovery convention. Returns the `SubagentFile` shape * `findSubagentFiles()` would discover for it. */ function writeSubagentFixture( diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 75fbf1b20..9d9257d85 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -3,9 +3,9 @@ * events. * * Each hook fire is a fresh CLI process, so this module reloads persisted parse state - * (`./parse-state.js`, Task 6), reads only the transcript lines appended since the last - * persisted `mainOffset` (`./transcript-reader.js`, Task 7), derives/merges - * `agent.usage.request` records for those new lines (`./usage-request.js`, Task 8), persists + * (`./parse-state.js`), reads only the transcript lines appended since the last + * persisted `mainOffset` (`./transcript-reader.js`), derives/merges + * `agent.usage.request` records for those new lines (`./usage-request.js`), persists * state back, and RETURNS one JSON string per completed request plus (on `Stop`/`SessionEnd`) * one `agent.session.summary` event — it never forwards anything to the spool itself. The caller * (the plugin's `processOtlpEvent`, via its per-event handlers) owns forwarding, so there is @@ -30,7 +30,7 @@ import { extractNamedInvocations } from '@/agents/plugins/claude/session/claude- import { type SubagentFile, buildSubagentUsageEvent } from './subagent-usage.js'; // Re-exported so callers (e.g. claude-code-otlp.plugin.ts) can import both `SubagentFile` and -// `runSubagentTranscriptParse` from this one module, per the plan's wiring description. +// `runSubagentTranscriptParse` from this one module. export type { SubagentFile }; export type MainTranscriptTrigger = 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'; @@ -93,24 +93,23 @@ function emptyAccumulator(): SessionSummaryAccumulator { * Recompute the full-session summary accumulator, named-invocation counts, and session start * time from byte 0 of the main transcript. * - * `TranscriptParseState` (Task 6's fixed shape) has no persisted field for any of + * `TranscriptParseState`'s fixed shape has no persisted field for any of * `SessionSummaryAccumulator`'s data or for `NamedInvocationCounts` — only `branchCounts` is * incrementally tracked there. So, for `Stop`/`SessionEnd`, this helper re-derives everything - * else fresh from the whole transcript file every time (see Task 11's Note B). Transcripts are + * else fresh from the whole transcript file every time. Transcripts are * not enormous and this only runs on `Stop`/`SessionEnd`, not on every hook. * * Never throws: a missing/unreadable transcript resolves to the emptiest defensible result * (empty accumulator, empty named-invocation counts, `startedAt: ''`); a malformed individual * line is skipped rather than aborting the whole scan. * - * Known limitations (no reliable in-transcript signal found for any of these — see spec.md's - * confidence gaps): + * Known limitations (no reliable in-transcript signal found for any of these): * - `toolCalls[*].errors` is derived from a sibling `tool_result` block's `is_error`/`isError` * flag (the same pattern `claude.session.ts`/`claude.metrics-processor.ts` already use for * tool-use_id → error lookups) when one is found; otherwise a tool call's `.errors` stays 0. * - `linesAdded`/`linesRemoved` default to 0 — an `Edit`/`Write` tool_use's `input` carries the * *proposed* edit, not a diff stat, so no reliable added/removed line count can be derived from - * it without re-implementing diffing (out of scope for this task). + * it without re-implementing diffing. * - `compactionCount` defaults to 0 — no verified in-transcript signal was found (`PreCompact` is * a hook event, not a transcript line). */ @@ -140,7 +139,7 @@ async function buildFullAccumulator( // Pass 1: collect tool_result error flags keyed by their matching tool_use_id. const errorByToolUseId = collectErrorToolUseIds(parsedLines); - // Pass 2: models (reusing Task 8's own model-resolution logic via parseUsageLine). + // Pass 2: models (reusing parseUsageLine's own model-resolution logic). for (const line of rawLines) { const parsedUsage = parseUsageLine(line, 'main', '', ''); if (parsedUsage) { @@ -187,16 +186,16 @@ async function buildFullAccumulator( * - Returns one `agent.usage.request` JSON string per request key touched by this pass. * - On `Stop`/`SessionEnd` only, also returns exactly one `agent.session.summary` event * (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`) built from a fresh full-file - * recompute (see {@link buildFullAccumulator} and Note B). `PreCompact` never returns a + * recompute (see {@link buildFullAccumulator}). `PreCompact` never returns a * summary. * - Persists state back to disk. * * Never forwards anything itself — the caller is responsible for sending the returned events to * the spool (exactly one place in the pipeline does that). * - * Scoping ruling (Note A — a judgment call, since no file in this codebase documents a reliable + * Scoping ruling (a judgment call, since no file in this codebase documents a reliable * signal for when a *main*-transcript turn enters/exits a "skill context"): every - * main-transcript-derived usage record in this task is scoped as `scopeKind: 'main'`, + * main-transcript-derived usage record is scoped as `scopeKind: 'main'`, * `scopeName: ''` unconditionally. `state.activeSkill` is deliberately left untouched (not read, * not written) here — it stays available, unused, for a future task that identifies a real * signal for it. @@ -363,7 +362,7 @@ async function scanSubagentTranscript(filePath: string): Promise { }; // The real registered OTLP agent name, unlike every other test in this file which - // deliberately uses the unregistered 'claude' (CR-016: that name resolves to `undefined`, + // deliberately uses the unregistered 'claude' (that name resolves to `undefined`, // so this is the only test exercising the real AgentRegistry.getAnalyticsAgent merge path). const record = JSON.stringify({ agentName: 'claude-code-otlp', diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts index cded667e7..dfaf1ab60 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -44,7 +44,7 @@ export function loadCodemieCliVersion(): string { /** * Per-record story-id override for `UserPromptSubmit` hook events only, * layered on top of the per-tick `resolveStoryOnce()` cache in - * `ctx.story` (explicit/branch/''). Priority order across the full chain is + * `ctx.story` (explicit/branch/undefined). Priority order across the full chain is * explicit -> marker -> branch -> mention: * * 1. If the per-tick cache already resolved to `'explicit'`, that is the From 25c4d0ac63481ab3d82c16ee0109adf3f02e38ed Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Tue, 6 Oct 2026 14:26:30 +0300 Subject: [PATCH 13/35] refactor(proxy): renamings and updating code comments --- .../otlp-spool/__tests__/forwarder.test.ts | 14 +++---- .../plugins/otlp-spool/forward-context.ts | 37 ++++++++-------- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 42 ++++++++++--------- 3 files changed, 46 insertions(+), 47 deletions(-) diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index 599a49117..bb8e120bf 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -56,16 +56,16 @@ function buildHookRecord(hookEventName: string, sessionId: string, extra: Record }); } -describe('loadCodemieCliVersion', () => { +describe('resolveCodemieCliVersion', () => { beforeEach(() => { execSyncMock.mockReset(); execSyncMock.mockReturnValue('0.15.6'); }); it('reads the installed CLI version via `codemie --version` and strips the semver', async () => { - const { loadCodemieCliVersion } = await import('../forward-context.js'); + const { resolveCodemieCliVersion } = await import('../forward-context.js'); - expect(loadCodemieCliVersion()).toBe('0.15.6'); + expect(resolveCodemieCliVersion()).toBe('0.15.6'); expect(execSyncMock).toHaveBeenCalledWith('codemie --version', { encoding: 'utf8', stdio: ['pipe', 'pipe', 'pipe'], @@ -77,8 +77,8 @@ describe('loadCodemieCliVersion', () => { throw new Error('ENOENT'); }); - const { loadCodemieCliVersion } = await import('../forward-context.js'); - expect(loadCodemieCliVersion()).toBe(''); + const { resolveCodemieCliVersion } = await import('../forward-context.js'); + expect(resolveCodemieCliVersion()).toBe(''); }); }); @@ -320,7 +320,7 @@ describe('mapHookRecords', () => { // The real registered OTLP agent name, unlike every other test in this file which // deliberately uses the unregistered 'claude' (that name resolves to `undefined`, - // so this is the only test exercising the real AgentRegistry.getAnalyticsAgent merge path). + // so this is the only test exercising the real agent-registry lookup merge path). const record = JSON.stringify({ agentName: 'claude-code-otlp', raw: JSON.stringify({ hook_event_name: 'Stop', session_id: 'sid1', cwd: '' }), @@ -337,7 +337,7 @@ describe('mapHookRecords', () => { it('carries ctx.identity through onto developer_name/identity_source for a non-jwt tier', async () => { const { mapHookRecords } = await import('../forwarder.js'); - // Pre-seeding ctx.identity (as resolveIdentityOnce's own cache would look once resolved) + // Pre-seeding ctx.identity (mimicking what the per-tick identity cache would look like once resolved) // with a non-jwt tier result proves the wiring from ctx.identity onto the mapped record, // independent of the identity-chain's own resolution logic (covered by identity.test.ts). const ctx = { diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts index dfaf1ab60..24d65c4dd 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -1,6 +1,6 @@ /** * Per-forward-tick context: the CLI version banner, and the identity/story resolution glue - * `forwarder.ts`'s `mapHookRecords()` calls once per batch. Split out of `forwarder.ts` to keep + * the hook-record mapping step calls once per batch. Split out of `forwarder.ts` to keep * that module under the documented 500-line structure cap (code-quality.md). */ @@ -28,7 +28,7 @@ export interface ForwardContext { } /** The installed CodeMie CLI version, resolved once at import time. */ -export function loadCodemieCliVersion(): string { +export function resolveCodemieCliVersion(): string { try { const output = execSync('codemie --version', { encoding: 'utf8', @@ -42,25 +42,22 @@ export function loadCodemieCliVersion(): string { } /** - * Per-record story-id override for `UserPromptSubmit` hook events only, - * layered on top of the per-tick `resolveStoryOnce()` cache in - * `ctx.story` (explicit/branch/undefined). Priority order across the full chain is - * explicit -> marker -> branch -> mention: + * Effective story id for ONE `UserPromptSubmit` record: layers this record's + * own prompt text on top of the once-per-tick cache in `ctx.story` + * (explicit/branch/undefined), without ever writing back to that cache — + * other records in the same batch still need it untouched. Priority order + * across the full chain is explicit -> marker -> branch -> mention: * - * 1. If the per-tick cache already resolved to `'explicit'`, that is the - * highest-priority result and wins outright. + * 1. If the cache already resolved to `'explicit'`, that wins outright. * 2. Otherwise, try the marker tier (`story: X` / `ticket #X`) against this * record's OWN prompt text — it sits above branch in priority. - * 3. Otherwise, if the per-tick cache resolved to `'branch'`, that wins (it - * is already correctly placed between marker and mention). + * 3. Otherwise, if the cache resolved to `'branch'`, that wins (it is + * already correctly placed between marker and mention). * 4. Otherwise, try the mention tier (bare ticket-shaped text) — the * lowest-priority tier. * 5. Otherwise, empty. - * - * Computed fresh per record and never mutates `ctx.story`: other records in - * the same batch still need that shared per-tick cache untouched. */ -export function resolvePromptStory( +export function resolveStoryForPrompt( ctx: ForwardContext, rawPrompt: string ): { storyId: string; storySource: string } { @@ -87,10 +84,10 @@ export function resolvePromptStory( /** * Resolve and cache this forward tick's story id/source once, from the first record that carries - * a real `cwd` — guarded the same way {@link resolveIdentityOnce} is, so a synthetic + * a real `cwd` — guarded the same way every other per-tick cache in this file is, so a synthetic * transcript-derived record's empty `cwd` can never poison the cache for the rest of the batch. */ -export async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promise { +export async function resolveStory(ctx: ForwardContext, cwd: string): Promise { if (!cwd || ctx.story?.storyId !== undefined) return; const explicit = await resolveExplicitStory(cwd); @@ -104,11 +101,11 @@ export async function resolveStoryOnce(ctx: ForwardContext, cwd: string): Promis /** * Resolve and cache this forward tick's developer identity once, from the first record that * carries a real `cwd`. An empty `cwd` (every synthetic transcript-derived record) is skipped - * rather than cached, mirroring {@link resolveGitInfo}'s own `if (!cwd) return;` guard in - * `forwarder.ts` — otherwise the first such record in a batch would permanently cache the - * cwd-less (and therefore less accurate) git-tier result for every later record in the same tick. + * rather than cached, mirroring the same empty-`cwd` guard every other per-tick cache uses — + * otherwise the first such record in a batch would permanently cache the cwd-less (and + * therefore less accurate) result for every later record in the same tick. */ -export async function resolveIdentityOnce(ctx: ForwardContext, cwd: string): Promise { +export async function resolveDeveloperIdentity(ctx: ForwardContext, cwd: string): Promise { if (!cwd) return; if (!ctx.identity) ctx.identity = {}; if (ctx.identity.developerName !== undefined) return; diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 0ea476609..1d1e00d48 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -16,13 +16,13 @@ import { resolveEventId } from './event-id.js'; import { decodeJwtClaims } from './identity.js'; import { type ForwardContext, - loadCodemieCliVersion, - resolveIdentityOnce, - resolvePromptStory, - resolveStoryOnce, + resolveCodemieCliVersion, + resolveDeveloperIdentity, + resolveStory, + resolveStoryForPrompt, } from './forward-context.js'; -const CODEMIE_CLI_VERSION = loadCodemieCliVersion(); +const CODEMIE_CLI_VERSION = resolveCodemieCliVersion(); const HOOK_EVENT_TYPE_MAP: Record = { SessionStart: 'agent.session.start', @@ -239,7 +239,7 @@ export async function mapHookRecords( for (const record of records) { // Every record occupied `byteLength(record) + 1` bytes in the spool file - // (the trailing newline `snapshotPendingHookRecords` already stripped). + // (the trailing newline was already stripped when the batch was read). const byteOffset = offset; offset += Buffer.byteLength(record, 'utf-8') + 1; @@ -262,20 +262,22 @@ export async function mapHookRecords( const cwd = String(hookEvent['cwd'] ?? ''); await resolveGitInfo(ctx, cwd); - await resolveIdentityOnce(ctx, cwd); - await resolveStoryOnce(ctx, cwd); - - // Read the record's OWN untruncated prompt text here, before - // `limitHookPayload()` below produces its own truncated `limited` copy. - // `limitHookPayload` never mutates `hookEvent` itself (it builds a fresh - // `{ ...hookEvent }` copy), so this is still the full original string — - // used ONLY to feed the marker/mention regex tiers below; the matched - // ticket id (a short string) is all that ever reaches the output, never - // this raw text itself. + await resolveDeveloperIdentity(ctx, cwd); + await resolveStory(ctx, cwd); + + // Read the record's OWN untruncated prompt text here, before the truncated + // copy below is produced. The original `hookEvent` is never mutated by that + // truncation step, so this is still the full string — used ONLY to feed the + // marker/mention regex tiers below; the matched ticket id (a short string) + // is all that ever reaches the output, never this raw text itself. const rawPrompt = typeof hookEvent['prompt'] === 'string' ? hookEvent['prompt'] : ''; + + // Each UserPromptSubmit record carries its own prompt text, which can hold a + // higher-priority override — so it's resolved fresh per record rather than + // just reusing the once-per-tick cache. const promptStory = hookName === 'UserPromptSubmit' - ? resolvePromptStory(ctx, rawPrompt) + ? resolveStoryForPrompt(ctx, rawPrompt) : { storyId: ctx.story?.storyId ?? '', storySource: ctx.story?.storySource ?? '' }; const { AgentRegistry } = await import('@/agents/registry.js'); @@ -284,9 +286,9 @@ export async function mapHookRecords( try { agentSpecificFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; } catch (err) { - // OtlpAgentAdapter.prepareAnalyticsFields's "must never throw" contract is only a doc - // comment — a future/alternate adapter implementation that violates it must not abort - // every remaining record in this forward tick. + // An agent plugin's analytics-field hook is documented as "must never throw", but + // that's only a doc comment — a violating implementation must not abort every + // remaining record in this forward tick. const msg = err instanceof Error ? err.message : String(err); logger.debug( '[otlp-forwarder] prepareAnalyticsFields threw', From d636e3ce0e5078a63c307ec057a5c7eb21f32079 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Tue, 6 Oct 2026 21:24:54 +0300 Subject: [PATCH 14/35] refactor(agents): unify claude-code-otlp evaluate() event dispatch Replace the mixed if/switch dispatch with one if-chain and one handler per hook event, each returning its own ForwardDecision directly. Drop the shared fallback sink, collapse the raw/parsed object duplication by extending BaseClaudeCodeHookEvent, rename the orchestrator collectors, and split client-version caching from resolution. Generated with AI Co-Authored-By: codemie-ai --- .../__tests__/claude-code-otlp.plugin.test.ts | 48 +++---- .../claude-code-otlp.plugin.ts | 136 +++++++++--------- .../claude-code-otlp.types.ts | 28 ++++ .../transcript/__tests__/orchestrator.test.ts | 85 ++++++----- .../transcript/orchestrator.ts | 25 ++-- 5 files changed, 167 insertions(+), 155 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index 94d767fbb..2bc1425ee 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -12,8 +12,8 @@ vi.mock('@/utils/logger.js', () => ({ })); const execMock = vi.fn(); -const runMainTranscriptParseMock = vi.fn(); -const runSubagentTranscriptParseMock = vi.fn(); +const collectMainTranscriptEventsMock = vi.fn(); +const collectSubagentTranscriptEventsMock = vi.fn(); const findSubagentFilesMock = vi.fn(); vi.mock('@/utils/exec.js', () => ({ @@ -21,8 +21,8 @@ vi.mock('@/utils/exec.js', () => ({ })); vi.mock('../transcript/orchestrator.js', () => ({ - runMainTranscriptParse: runMainTranscriptParseMock, - runSubagentTranscriptParse: runSubagentTranscriptParseMock, + collectMainTranscriptEvents: collectMainTranscriptEventsMock, + collectSubagentTranscriptEvents: collectSubagentTranscriptEventsMock, })); vi.mock('../transcript/subagent-usage.js', () => ({ @@ -155,10 +155,10 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { } beforeEach(() => { - runMainTranscriptParseMock.mockReset(); - runMainTranscriptParseMock.mockResolvedValue([]); - runSubagentTranscriptParseMock.mockReset(); - runSubagentTranscriptParseMock.mockResolvedValue([]); + collectMainTranscriptEventsMock.mockReset(); + collectMainTranscriptEventsMock.mockResolvedValue([]); + collectSubagentTranscriptEventsMock.mockReset(); + collectSubagentTranscriptEventsMock.mockResolvedValue([]); findSubagentFilesMock.mockReset(); findSubagentFilesMock.mockResolvedValue([]); vi.mocked(forwardOtlpEventToSpool).mockReset(); @@ -170,7 +170,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { }); it.each(['Stop', 'PreCompact', 'SessionEnd', 'StopFailure'] as const)( - 'invokes runMainTranscriptParse with the extracted session/transcript/trigger on %s', + 'invokes collectMainTranscriptEvents with the extracted session/transcript/trigger on %s', async (hookEventName) => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); @@ -178,7 +178,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(runMainTranscriptParseMock).toHaveBeenCalledWith( + expect(collectMainTranscriptEventsMock).toHaveBeenCalledWith( 'sid-1', '/tmp/transcript.jsonl', hookEventName @@ -187,17 +187,17 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { } ); - it('does not dispatch runMainTranscriptParse for unrelated hook events', async () => { + it('does not dispatch collectMainTranscriptEvents for unrelated hook events', async () => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'PostToolUse' })); await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(runMainTranscriptParseMock).not.toHaveBeenCalled(); + expect(collectMainTranscriptEventsMock).not.toHaveBeenCalled(); }); - it('invokes runSubagentTranscriptParse with the agent_id/tool_use_id/agent_type extracted from a SubagentStop payload', async () => { + it('invokes collectSubagentTranscriptEvents with the agent_id/tool_use_id/agent_type extracted from a SubagentStop payload', async () => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); const rawEvent = JSON.stringify( @@ -212,7 +212,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(runSubagentTranscriptParseMock).toHaveBeenCalledWith('sid-1', '/tmp/transcript.jsonl', { + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith('sid-1', { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl', toolUseId: 'tu-1', @@ -232,24 +232,23 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(runSubagentTranscriptParseMock).toHaveBeenCalledWith( + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( 'sid-1', - '/tmp/transcript.jsonl', expect.objectContaining({ agentId: 'sub-2' }) ); }); - it('skips runSubagentTranscriptParse on SubagentStop when agent_transcript_path is missing', async () => { + it('skips collectSubagentTranscriptEvents on SubagentStop when agent_transcript_path is missing', async () => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SubagentStop' })); await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(runSubagentTranscriptParseMock).not.toHaveBeenCalled(); + expect(collectSubagentTranscriptEventsMock).not.toHaveBeenCalled(); }); - it('runs runSubagentTranscriptParse for every subagent file found on SessionEnd', async () => { + it('runs collectSubagentTranscriptEvents for every subagent file found on SessionEnd', async () => { findSubagentFilesMock.mockResolvedValue([ { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' }, { agentId: 'sub-2', filePath: '/tmp/agent-sub-2.jsonl' }, @@ -261,10 +260,9 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); expect(findSubagentFilesMock).toHaveBeenCalledWith('/tmp/transcript.jsonl'); - expect(runSubagentTranscriptParseMock).toHaveBeenCalledTimes(2); - expect(runSubagentTranscriptParseMock).toHaveBeenCalledWith( + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledTimes(2); + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( 'sid-1', - '/tmp/transcript.jsonl', { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' } ); }); @@ -276,14 +274,14 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(runMainTranscriptParseMock).not.toHaveBeenCalled(); + expect(collectMainTranscriptEventsMock).not.toHaveBeenCalled(); expect(forwardOtlpEventToSpool).toHaveBeenCalledWith(rawEvent, 'claude-code-otlp'); }); it('forwards every event a per-event handler returns (the raw event plus any derived events) through the single forwardToSpool path, in order', async () => { const derivedUsageEvent = JSON.stringify({ type: 'agent.usage.request' }); const derivedSummaryEvent = JSON.stringify({ type: 'agent.session.summary' }); - runMainTranscriptParseMock.mockResolvedValue([derivedUsageEvent, derivedSummaryEvent]); + collectMainTranscriptEventsMock.mockResolvedValue([derivedUsageEvent, derivedSummaryEvent]); const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); @@ -291,7 +289,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - // runMainTranscriptParse/runSubagentTranscriptParse never call forwardOtlpEventToSpool + // collectMainTranscriptEvents/collectSubagentTranscriptEvents never call forwardOtlpEventToSpool // themselves (they are mocked here to just return data) — every event that reaches the spool // mock arrived via forwardToSpool, called exactly once from processOtlpEvent. expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(3); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index 9d80b996a..5194a8a85 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -8,17 +8,21 @@ import { CLAUDE_CODE_OTLP_AGENT_NAME } from './claude-code-otlp.constants.js'; import { BaseClaudeCodeHookEvent, ForwardDecision, toBaseClaudeCodeHookEvent } from './claude-code-otlp.types.js'; import { forwardOtlpEventToSpool } from '../utils.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; -import { runMainTranscriptParse, runSubagentTranscriptParse, type SubagentFile } from './transcript/orchestrator.js'; +import { + collectMainTranscriptEvents, + collectSubagentTranscriptEvents, + type SubagentFile, +} from './transcript/orchestrator.js'; import { findSubagentFiles } from './transcript/subagent-usage.js'; import { exec } from '@/utils/exec.js'; export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; public readonly type = AgentAdapterType.OTLP; - public readonly platform = 'claude-code' as const; /** Cached across calls so a high-frequency hook (e.g. PostToolUse) doesn't spawn a subprocess per call. */ - private cachedClientVersion: string | undefined; + private clientVersion: string | undefined; + private readonly platform = 'claude-code'; public async processOtlpEvent(rawEvent: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { const event = toBaseClaudeCodeHookEvent(JSON.parse(rawEvent)); @@ -38,7 +42,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { await ensureOtlpProxy(this.name); - const evaluation = await this.evaluate(event.hookEventName, rawEvent); + const evaluation = await this.evaluate(rawEvent); if (evaluation.decision === 'block') { logger.error(`[Claude Code OTLP plugin] Blocking prompt: ${evaluation.reason}`); console.log(JSON.stringify(evaluation)); @@ -47,93 +51,77 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { this.forwardToSpool(evaluation.payload); } - private async evaluate(hookEventName: string, rawEvent: string): Promise { - const rawParsed = JSON.parse(rawEvent); - const event = toBaseClaudeCodeHookEvent(rawParsed); + private async evaluate(rawEvent: string): Promise { + const event = toBaseClaudeCodeHookEvent(JSON.parse(rawEvent)); if (!event.sessionId) { return { decision: 'forward', payload: [rawEvent] }; } - if (hookEventName === 'UserPromptSubmit') { + if (event.hookEventName === 'UserPromptSubmit') { return await this.onUserPromptSubmit(rawEvent); } - - let derivedEvents: string[] = []; - - switch (event.hookEventName) { - case 'Stop': - derivedEvents = await this.onStopEvent(event); - break; - case 'PreCompact': - derivedEvents = await this.onPreCompactEvent(event); - break; - case 'StopFailure': - derivedEvents = await this.onStopFailureEvent(event); - break; - case 'SessionEnd': - derivedEvents = await this.onSessionEndEvent(event); - break; - case 'SubagentStop': - derivedEvents = await this.onSubagentStopEvent(event, rawParsed as Record); - break; - default: - break; + if (event.hookEventName === 'Stop') { + return await this.onStopEvent(rawEvent, event); + } + if (event.hookEventName === 'PreCompact') { + return await this.onPreCompactEvent(rawEvent, event); + } + if (event.hookEventName === 'StopFailure') { + return await this.onStopFailureEvent(rawEvent, event); + } + if (event.hookEventName === 'SessionEnd') { + return await this.onSessionEndEvent(rawEvent, event); + } + if (event.hookEventName === 'SubagentStop') { + return await this.onSubagentStopEvent(rawEvent, event); } - return { - decision: 'forward', - payload: [rawEvent, ...derivedEvents], - }; + return { decision: 'forward', payload: [rawEvent] }; } - private async onStopEvent(event: BaseClaudeCodeHookEvent): Promise { - return await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'Stop'); + private async onStopEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { + const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'Stop'); + return { decision: 'forward', payload: [rawEvent, ...derived] }; } - private async onPreCompactEvent(event: BaseClaudeCodeHookEvent): Promise { - return await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'PreCompact'); + private async onPreCompactEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { + const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'PreCompact'); + return { decision: 'forward', payload: [rawEvent, ...derived] }; } - private async onStopFailureEvent(event: BaseClaudeCodeHookEvent): Promise { - return await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'StopFailure'); + private async onStopFailureEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { + const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'StopFailure'); + return { decision: 'forward', payload: [rawEvent, ...derived] }; } - private async onSessionEndEvent(event: BaseClaudeCodeHookEvent): Promise { - const events = await runMainTranscriptParse(event.sessionId, event.transcriptPath, 'SessionEnd'); + private async onSessionEndEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { + const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'SessionEnd'); // Backstop: guarantee every subagent discovered for this session gets at least one // agent.subagent.usage event, even when its own SubagentStop hook never fired. const subagentFiles = await findSubagentFiles(event.transcriptPath); for (const file of subagentFiles) { - events.push(...(await runSubagentTranscriptParse(event.sessionId, event.transcriptPath, file))); + derived.push(...(await collectSubagentTranscriptEvents(event.sessionId, file))); } - return events; + return { decision: 'forward', payload: [rawEvent, ...derived] }; } - private async onSubagentStopEvent( - event: BaseClaudeCodeHookEvent, - rawRecord: Record - ): Promise { - const agentTranscriptPath = - typeof rawRecord['agent_transcript_path'] === 'string' ? rawRecord['agent_transcript_path'] : ''; - if (!agentTranscriptPath) { - return []; + private async onSubagentStopEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { + if (!event.agentTranscriptPath) { + return { decision: 'forward', payload: [rawEvent] }; } - const agentId = - typeof rawRecord['agent_id'] === 'string' - ? rawRecord['agent_id'] - : basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''); const subagentFile: SubagentFile = { - agentId, - filePath: agentTranscriptPath, - toolUseId: typeof rawRecord['tool_use_id'] === 'string' ? rawRecord['tool_use_id'] : undefined, - agentType: typeof rawRecord['agent_type'] === 'string' ? rawRecord['agent_type'] : undefined, + agentId: event.agentId ?? basename(event.agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''), + filePath: event.agentTranscriptPath, + toolUseId: event.toolUseId, + agentType: event.agentType, }; - return await runSubagentTranscriptParse(event.sessionId, event.transcriptPath, subagentFile); + const derived = await collectSubagentTranscriptEvents(event.sessionId, subagentFile); + return { decision: 'forward', payload: [rawEvent, ...derived] }; } private async ensureProxyAuth(): Promise { @@ -162,31 +150,34 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { client_version: await this.getClientVersion(), }; - if (typeof hookEvent['agent_id'] === 'string') { - fields.agent_id = hookEvent['agent_id']; + const agentId = readOptionalString(hookEvent, 'agent_id'); + if (agentId !== undefined) { + fields.agent_id = agentId; } - if (typeof hookEvent['agent_type'] === 'string') { - fields.agent_type = hookEvent['agent_type']; + const agentType = readOptionalString(hookEvent, 'agent_type'); + if (agentType !== undefined) { + fields.agent_type = agentType; } return fields; } private async getClientVersion(): Promise { - if (this.cachedClientVersion !== undefined) { - return this.cachedClientVersion; + if (this.clientVersion === undefined) { + this.clientVersion = await this.resolveClientVersion(); } + return this.clientVersion; + } + private async resolveClientVersion(): Promise { try { const result = await exec('claude', ['--version']); const trimmed = result.stdout.trim(); const versionMatch = trimmed.match(/^(\d+\.\d+\.\d+)/); - this.cachedClientVersion = versionMatch ? versionMatch[1] : trimmed; + return versionMatch ? versionMatch[1] : trimmed; } catch { - this.cachedClientVersion = ''; + return ''; } - - return this.cachedClientVersion; } private async onUserPromptSubmit(rawEvent: string): Promise { @@ -213,3 +204,8 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { } } } + +function readOptionalString(record: Record, key: string): string | undefined { + const value = record[key]; + return typeof value === 'string' ? value : undefined; +} diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts index a01f60298..497b2ab3a 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts @@ -41,6 +41,18 @@ interface RawBaseClaudeCodeHookEvent { /** Name of the event that fired */ hook_event_name: string; + + /** `SubagentStop`-only: path to the subagent's own transcript file. */ + agent_transcript_path?: string; + + /** `SubagentStop`-only: identifier of the subagent, when the hook payload carries one. */ + agent_id?: string; + + /** `SubagentStop`-only: the subagent's declared type (e.g. `explore`). */ + agent_type?: string; + + /** `SubagentStop`-only: the tool_use_id of the Task invocation that spawned the subagent. */ + tool_use_id?: string; } /** @@ -57,6 +69,18 @@ export interface BaseClaudeCodeHookEvent { level: "low" | "medium" | "high" | "xhigh" | "max"; }; hookEventName: string; + + /** `SubagentStop`-only: path to the subagent's own transcript file. */ + agentTranscriptPath?: string; + + /** `SubagentStop`-only: identifier of the subagent, when the hook payload carries one. */ + agentId?: string; + + /** `SubagentStop`-only: the subagent's declared type (e.g. `explore`). */ + agentType?: string; + + /** `SubagentStop`-only: the tool_use_id of the Task invocation that spawned the subagent. */ + toolUseId?: string; } export function toBaseClaudeCodeHookEvent(raw: RawBaseClaudeCodeHookEvent): BaseClaudeCodeHookEvent { @@ -69,6 +93,10 @@ export function toBaseClaudeCodeHookEvent(raw: RawBaseClaudeCodeHookEvent): Base permissionMode: raw.permission_mode, effort: raw.effort ? { level: raw.effort.level } : undefined, hookEventName: raw.hook_event_name, + agentTranscriptPath: raw.agent_transcript_path, + agentId: raw.agent_id, + agentType: raw.agent_type, + toolUseId: raw.tool_use_id, }; } diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index be970da41..26572168a 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -1,8 +1,8 @@ /** - * Tests for `runMainTranscriptParse` — the `Stop`/`PreCompact`/`SessionEnd` main-transcript + * Tests for `collectMainTranscriptEvents` — the `Stop`/`PreCompact`/`SessionEnd` main-transcript * orchestrator. * - * Neither `runMainTranscriptParse` nor `runSubagentTranscriptParse` writes to the spool itself — + * Neither `collectMainTranscriptEvents` nor `collectSubagentTranscriptEvents` writes to the spool itself — * each returns the raw JSON strings it wants forwarded, and the caller (the plugin's * `processOtlpEvent`) is the only place that actually forwards them. So these tests read the * returned array directly; no network/daemon mocking is needed. `CODEMIE_HOME` points at a fresh @@ -104,9 +104,9 @@ function writeSubagentFixture( return { agentId, filePath }; } -describe('runMainTranscriptParse — idempotent reparse', () => { +describe('collectMainTranscriptEvents — idempotent reparse', () => { it('returns agent.usage.request events with identical request_id/model pairs across a crash-before-save re-parse', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const { saveParseState, createParseState } = await import('../parse-state.js'); const sessionId = 'session-idempotent'; @@ -116,7 +116,7 @@ describe('runMainTranscriptParse — idempotent reparse', () => { usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75, stopReason: 'end_turn' }), ]); - const first = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); + const first = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); const firstPairs = first .filter((e) => e.type === 'agent.usage.request') @@ -130,7 +130,7 @@ describe('runMainTranscriptParse — idempotent reparse', () => { // happened). await saveParseState(sessionId, createParseState()); - const second = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); + const second = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); const secondPairs = second .filter((e) => e.type === 'agent.usage.request') @@ -142,9 +142,9 @@ describe('runMainTranscriptParse — idempotent reparse', () => { }); }); -describe('runMainTranscriptParse — Stop trigger', () => { +describe('collectMainTranscriptEvents — Stop trigger', () => { it('returns one agent.usage.request event per distinct request plus one incremental agent.session.summary', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-stop-basic'; const transcriptPath = writeTranscript('transcript-stop.jsonl', [ @@ -152,7 +152,7 @@ describe('runMainTranscriptParse — Stop trigger', () => { usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), ]); - const raw = await runMainTranscriptParse(sessionId, transcriptPath, 'Stop'); + const raw = await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); expect(raw).toHaveLength(3); const events = parseAll(raw); @@ -166,16 +166,16 @@ describe('runMainTranscriptParse — Stop trigger', () => { }); }); -describe('runMainTranscriptParse — PreCompact trigger', () => { +describe('collectMainTranscriptEvents — PreCompact trigger', () => { it('returns usage-request events but never a session summary', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-precompact'; const transcriptPath = writeTranscript('transcript-precompact.jsonl', [ usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), ]); - const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact')); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'PreCompact')); const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); @@ -184,18 +184,18 @@ describe('runMainTranscriptParse — PreCompact trigger', () => { }); }); -describe('runMainTranscriptParse — compaction_count', () => { +describe('collectMainTranscriptEvents — compaction_count', () => { it('persists one increment per PreCompact trigger and surfaces the cumulative count on a later summary', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-compaction'; const transcriptPath = writeTranscript('transcript-compaction.jsonl', [ usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), ]); - await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); - await runMainTranscriptParse(sessionId, transcriptPath, 'PreCompact'); - const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); + await collectMainTranscriptEvents(sessionId, transcriptPath, 'PreCompact'); + await collectMainTranscriptEvents(sessionId, transcriptPath, 'PreCompact'); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); expect(summaryEvents).toHaveLength(1); @@ -203,9 +203,9 @@ describe('runMainTranscriptParse — compaction_count', () => { }); }); -describe('runMainTranscriptParse — api_calls', () => { +describe('collectMainTranscriptEvents — api_calls', () => { it("surfaces the session's full agent.usage.request record count on the summary event", async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-api-calls'; const transcriptPath = writeTranscript('transcript-api-calls.jsonl', [ @@ -213,7 +213,7 @@ describe('runMainTranscriptParse — api_calls', () => { usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 75 }), ]); - const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); const usageEvents = events.filter((e) => e.type === 'agent.usage.request'); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); @@ -223,16 +223,16 @@ describe('runMainTranscriptParse — api_calls', () => { }); }); -describe('runMainTranscriptParse — SessionEnd trigger', () => { +describe('collectMainTranscriptEvents — SessionEnd trigger', () => { it('returns a final-phase summary with an ended_at key present', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-end'; const transcriptPath = writeTranscript('transcript-end.jsonl', [ usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), ]); - const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'SessionEnd')); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd')); const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); expect(summaryEvents).toHaveLength(1); @@ -242,33 +242,33 @@ describe('runMainTranscriptParse — SessionEnd trigger', () => { }); }); -describe('runMainTranscriptParse — missing transcript file', () => { +describe('collectMainTranscriptEvents — missing transcript file', () => { it('resolves cleanly to an empty array, for a trigger that never emits a summary', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const { loadParseState } = await import('../parse-state.js'); const sessionId = 'session-missing-file'; const missingPath = join(transcriptDir, 'does-not-exist.jsonl'); - await expect(runMainTranscriptParse(sessionId, missingPath, 'PreCompact')).resolves.toEqual([]); + await expect(collectMainTranscriptEvents(sessionId, missingPath, 'PreCompact')).resolves.toEqual([]); const state = await loadParseState(sessionId); expect(state.mainOffset).toBe(0); }); it('never throws even on Stop (which does attempt a full-file summary recompute)', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-missing-file-stop'; const missingPath = join(transcriptDir, 'also-does-not-exist.jsonl'); - await expect(runMainTranscriptParse(sessionId, missingPath, 'Stop')).resolves.toBeInstanceOf(Array); + await expect(collectMainTranscriptEvents(sessionId, missingPath, 'Stop')).resolves.toBeInstanceOf(Array); }); }); -describe('runMainTranscriptParse — tool-call accumulation', () => { +describe('collectMainTranscriptEvents — tool-call accumulation', () => { it('counts Edit/Write tool_use blocks into files_changed/files_written on the Stop summary', async () => { - const { runMainTranscriptParse } = await import('../orchestrator.js'); + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-tools'; const toolLine = JSON.stringify({ @@ -295,7 +295,7 @@ describe('runMainTranscriptParse — tool-call accumulation', () => { }); const transcriptPath = writeTranscript('transcript-tools.jsonl', [toolLine, resultLine]); - const events = parseAll(await runMainTranscriptParse(sessionId, transcriptPath, 'Stop')); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); const summary = events.find((e) => e.type === 'agent.session.summary'); expect(summary).toBeDefined(); @@ -308,9 +308,9 @@ describe('runMainTranscriptParse — tool-call accumulation', () => { }); }); -describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, one pre-advanced)', () => { +describe('collectSubagentTranscriptEvents — SessionEnd backstop (three subagents, one pre-advanced)', () => { it('returns exactly three agent.subagent.usage events — one per subagent, including the one whose own SubagentStop already advanced its offset — never a fourth', async () => { - const { runSubagentTranscriptParse } = await import('../orchestrator.js'); + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); const { findSubagentFiles } = await import('../subagent-usage.js'); const sessionId = 'session-backstop'; @@ -331,7 +331,7 @@ describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, o const filesBeforeBackstop = await findSubagentFiles(mainTranscriptPath); const a1File = filesBeforeBackstop.find((f) => f.agentId === 'a1'); if (!a1File) throw new Error('fixture missing a1'); - await runSubagentTranscriptParse(sessionId, mainTranscriptPath, a1File); + await collectSubagentTranscriptEvents(sessionId, a1File); // Exercise exactly what the plugin's SessionEnd branch does: discover every subagent file // for the session and re-run the subagent parse for each one, unconditionally — the @@ -340,7 +340,7 @@ describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, o expect(allFiles).toHaveLength(3); const backstopEvents: ForwardedEvent[] = []; for (const file of allFiles) { - backstopEvents.push(...parseAll(await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file))); + backstopEvents.push(...parseAll(await collectSubagentTranscriptEvents(sessionId, file))); } const subagentUsageEvents = backstopEvents.filter((e) => e.type === 'agent.subagent.usage'); @@ -353,9 +353,9 @@ describe('runSubagentTranscriptParse — SessionEnd backstop (three subagents, o }); }); -describe('runSubagentTranscriptParse — no new bytes since last run', () => { +describe('collectSubagentTranscriptEvents — no new bytes since last run', () => { it('returns zero new agent.usage.request events on a no-op reparse, but still exactly one agent.subagent.usage event summarizing unchanged cumulative usage', async () => { - const { runSubagentTranscriptParse } = await import('../orchestrator.js'); + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); const { findSubagentFiles } = await import('../subagent-usage.js'); const sessionId = 'session-no-new-bytes'; @@ -367,13 +367,13 @@ describe('runSubagentTranscriptParse — no new bytes since last run', () => { const [file] = await findSubagentFiles(mainTranscriptPath); - const firstEvents = parseAll(await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file)); + const firstEvents = parseAll(await collectSubagentTranscriptEvents(sessionId, file)); expect(firstEvents.filter((e) => e.type === 'agent.usage.request')).toHaveLength(2); expect(firstEvents.filter((e) => e.type === 'agent.subagent.usage')).toHaveLength(1); // Re-run on the same subagent file with no new content appended since the last call (its // offset is now at EOF). - const secondEvents = parseAll(await runSubagentTranscriptParse(sessionId, mainTranscriptPath, file)); + const secondEvents = parseAll(await collectSubagentTranscriptEvents(sessionId, file)); const secondUsageRequests = secondEvents.filter((e) => e.type === 'agent.usage.request'); const secondSubagentUsage = secondEvents.filter((e) => e.type === 'agent.subagent.usage'); @@ -387,19 +387,18 @@ describe('runSubagentTranscriptParse — no new bytes since last run', () => { }); }); -describe('runSubagentTranscriptParse — missing subagent transcript file', () => { +describe('collectSubagentTranscriptEvents — missing subagent transcript file', () => { it('resolves without throwing and still returns one empty-usage agent.subagent.usage event, with no agent.usage.request events', async () => { - const { runSubagentTranscriptParse } = await import('../orchestrator.js'); + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-subagent-missing'; - const mainTranscriptPath = writeTranscript(`${sessionId}.jsonl`, [noUsageLine('uuid-main')]); const missingSubagentFile = { agentId: 'ghost', filePath: join(transcriptDir, sessionId, 'subagents', 'agent-ghost.jsonl'), }; const events = parseAll( - await runSubagentTranscriptParse(sessionId, mainTranscriptPath, missingSubagentFile) + await collectSubagentTranscriptEvents(sessionId, missingSubagentFile) ); const usageRequestEvents = events.filter((e) => e.type === 'agent.usage.request'); const subagentUsageEvents = events.filter((e) => e.type === 'agent.subagent.usage'); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 9d9257d85..78f3c0143 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -30,7 +30,7 @@ import { extractNamedInvocations } from '@/agents/plugins/claude/session/claude- import { type SubagentFile, buildSubagentUsageEvent } from './subagent-usage.js'; // Re-exported so callers (e.g. claude-code-otlp.plugin.ts) can import both `SubagentFile` and -// `runSubagentTranscriptParse` from this one module. +// `collectSubagentTranscriptEvents` from this one module. export type { SubagentFile }; export type MainTranscriptTrigger = 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'; @@ -176,7 +176,7 @@ async function buildFullAccumulator( } /** - * Orchestrate a main-transcript parse pass for one `Stop`/`PreCompact`/`SessionEnd` hook fire. + * Collect the spool-bound events for one `Stop`/`PreCompact`/`SessionEnd` hook fire. * * - Loads persisted state, reads only the lines appended since `state.mainOffset`. * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, @@ -202,7 +202,7 @@ async function buildFullAccumulator( * * Swallows every error internally — never throws into `processOtlpEvent`. */ -export async function runMainTranscriptParse( +export async function collectMainTranscriptEvents( sessionId: string, transcriptPath: string, trigger: MainTranscriptTrigger @@ -361,15 +361,15 @@ async function scanSubagentTranscript(filePath: string): Promise { try { - // Load/mutate/save inside the lock — same rationale as runMainTranscriptParse: a sibling + // Load/mutate/save inside the lock — same rationale as collectMainTranscriptEvents: a sibling // SubagentStop for another subagent in this same session must never read state this pass is // about to overwrite. return await withParseStateLock(sessionId, async () => { From 93512cffce33f78def27f0eedd73543ba7a63837 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Tue, 6 Oct 2026 21:25:53 +0300 Subject: [PATCH 15/35] docs(agents): document the OtlpAgentAdapter ingestion pattern Add docs/ARCHITECTURE-OTLP-PLUGIN.md: the agent-agnostic OtlpAgentAdapter contract, the evaluate()/ForwardDecision dispatch pattern, how to add a new hook event or a new adapter, and the current otlp-spool completeness-gating/draining behavior (with planned extensions clearly flagged as not-yet-implemented). Index it from AGENTS.md's Agent Plugins table and add a top-level OTel (OTLP) Ingestion section pointing new contributors at it. Generated with AI Co-Authored-By: codemie-ai --- AGENTS.md | 7 ++ docs/ARCHITECTURE-OTLP-PLUGIN.md | 166 +++++++++++++++++++++++++++++++ 2 files changed, 173 insertions(+) create mode 100644 docs/ARCHITECTURE-OTLP-PLUGIN.md diff --git a/AGENTS.md b/AGENTS.md index 8bedc6f89..0ed0b7c9f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -223,6 +223,7 @@ See `package.json` for exact dependency versions and `.ai-run/guides/architectur | `kimi` / `kimi-acp` | `kimi/` | `@moonshot-ai/kimi-code` | ACP variant prepends `acp` to argv | | `openwiki` | `openwiki/` | `openwiki` | Docs/wiki tool, not a chat agent; declarative-only adapter — `envMapping` feeds the profile's base URL/key/model to `OPENAI_COMPATIBLE_*`/`OPENWIKI_MODEL_ID`, SSO/JWT goes through the local proxy | | `copilot-cli` | `copilot-cli/` | none | Analytics ingestion only — never installed or launched by CodeMie | +| `claude-code-otlp` | `claude-code-otlp/` | none | `OtlpAgentAdapter`, not a chat agent — ingests Claude Code's native hook events via `codemie hook --agent claude-code-otlp`; see `docs/ARCHITECTURE-OTLP-PLUGIN.md` for the agent-agnostic `OtlpAgentAdapter` dispatch pattern (used when adding a new hook event here, or a new adapter for another tool) | Not agent adapters, but injected runtime plugins under the same tree: `codemie-code-hooks/` (injected into `codemie-code` and `opencode`) and `reasoning-sanitizer/` (injected into `codemie-code`). @@ -238,6 +239,12 @@ Deterministic docs/knowledge tooling (not agent harnesses): `codebase-memory` (M > Neither `.ai-run/guides/integration/external-integrations.md` nor `docs/AGENTS.md` covers Pi yet. For Pi work, read `src/agents/plugins/pi/` directly plus the design docs under `docs/superpowers/specs/`. +## OTel (OTLP) Ingestion + +The preferred way to ingest a coding tool's own native telemetry/hook events into CodeMie's analytics pipeline is the `OtlpAgentAdapter` pattern — one plugin per tool, each implementing `processOtlpEvent`/`prepareAnalyticsFields` (`src/agents/core/types.ts`), registered in `AgentRegistry`, invoked via `codemie hook --agent `. Every adapter's `evaluate()` dispatch should follow the same shape (one `ForwardDecision`-returning handler per native event name, no `switch`, a single `forwardToSpool()` call site) and feed the same shared, agent-agnostic spool/forwarder pipeline (`OtlpHookSpoolData` → proxy daemon spool → `otlp-spool/forwarder.ts` → analytics API). + +`claude-code-otlp` (`src/agents/plugins/claude-code-otlp/`) is the current reference implementation — see **`docs/ARCHITECTURE-OTLP-PLUGIN.md`** for the full contract, the dispatch pattern, how to add a new event to an existing adapter, and how to add a new adapter for another tool (e.g. a future Cursor adapter). + ## Coding Standards ES modules, async/await, `interface` for shapes, explicit return types on exports, no `any`. Full conventions live in `.ai-run/guides/development/development-practices.md` and `.ai-run/guides/standards/code-quality.md`. diff --git a/docs/ARCHITECTURE-OTLP-PLUGIN.md b/docs/ARCHITECTURE-OTLP-PLUGIN.md new file mode 100644 index 000000000..bbd350d7e --- /dev/null +++ b/docs/ARCHITECTURE-OTLP-PLUGIN.md @@ -0,0 +1,166 @@ +# OTLP Hook-Event Plugin Pattern (`OtlpAgentAdapter`) + +**Scope**: any plugin implementing `OtlpAgentAdapter` (`src/agents/core/types.ts`), under `src/agents/plugins//`. Today that's `claude-code-otlp` only; this doc is written so a second implementation (another coding agent/IDE with its own native hook/event surface — e.g. a future Cursor adapter) does not have to reinvent the dispatch shape. +**Status**: Living doc — update this file when the dispatch pattern changes, or when a second `OtlpAgentAdapter` implementation lands (promote the parts that turn out to generalize, keep the parts that don't agent-specific). + +## 1. What this pattern is + +An `OtlpAgentAdapter` is not a chat agent — it never has a conversation. It is the ingestion point for one coding tool's own native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, per-request usage, subagent usage, auth gating, etc.). Each tool that exposes hooks (Claude Code today; potentially Cursor or others later) gets its own adapter plugin, but every adapter implements the same two-method contract and feeds the same spool shape: + +```ts +export interface OtlpAgentAdapter { + readonly name: string; + readonly type: AgentAdapterType.OTLP; + + processOtlpEvent(rawHookInput: string): Promise; + + /** Resolve agent-owned common fields for a single hook event */ + prepareAnalyticsFields( + hookEvent: Record, + ): Promise>; +} +``` + +```ts +export interface OtlpHookSpoolData { + agentName: string; + raw: string; + timestamp: number; +} +``` + +`processOtlpEvent` is the only entry point; everything downstream of it — the spool, the forwarder, the analytics API — is agent-agnostic and already shared. Nothing about the pipeline described below is specific to any one tool's hook names or payload shape; those live entirely inside each adapter's own `evaluate()`. + +## 2. Where an adapter sits in the pipeline + +``` +'s native hook/event fires (hook names & payload shape are agent-specific) + │ JSON on stdin + ▼ +codemie hook --agent (src/cli/commands/hook.ts) + │ + ▼ +.processOtlpEvent(rawEvent: string) + │ + ▼ +evaluate(rawEvent) → ForwardDecision ◄── §4: the shape every adapter should follow + │ + ┌────┴─────┐ + │ │ + block forward + │ │ + log + forwardToSpool(payload) ──POST──► proxy daemon spool (OtlpHookSpoolData) + suppress │ + (adapter- ▼ + specific) otlp-spool/forwarder.ts (background, agent-agnostic) + │ + .prepareAnalyticsFields(hookEvent) + │ + ▼ + CodeMie analytics API +``` + +- `codemie hook --agent ` (`src/cli/commands/hook.ts`) looks up the adapter via `AgentRegistry.getAnalyticsAgent(name)` and calls `processOtlpEvent` with the raw stdin payload. This part is already agent-agnostic — a new adapter just registers under a new name. +- `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is fire-and-forget and shared by every adapter: it POSTs `{ agentName, timestamp, raw }` to the local proxy daemon and swallows every error. A dead/unreachable daemon never blocks or fails the hook, regardless of which adapter called it. +- `otlp-spool/forwarder.ts` is agent-agnostic too — it reads spooled records and dispatches to whichever adapter's `prepareAnalyticsFields` matches `spoolData.agentName` (via `AgentRegistry.getAnalyticsAgent`). A new adapter needs no changes here as long as it implements `prepareAnalyticsFields` and spools under its own registered name. +- Wiring _which_ native hooks/events get pointed at `codemie hook --agent `, and how, is entirely tool-specific — see §5 for how `claude-code-otlp` does it; a different tool will have its own connector. + +## 3. The dispatch pattern every `evaluate()` should follow + +This is the part that generalizes across adapters, independent of which tool's hooks are being handled. Keep to this shape regardless of the native event names involved. + +### 3.1 `ForwardDecision` — the only two outcomes + +```ts +export type ForwardDecision = + | { decision: "forward"; payload: string[] } + | { + decision: "block"; + reason: string; + hookSpecificOutput: Record; + }; +``` + +Every native event an adapter processes should resolve to exactly one of these. `forward` carries the full list of raw JSON strings to push to the spool (the original raw event, plus zero or more derived analytics events). `block` stops the hook and logs a reason; what `hookSpecificOutput` means (e.g. suppressing a prompt) is specific to the tool and the event, not to this pattern. + +### 3.2 One handler per event name — no `switch`, no shared merge step + +```ts +private async evaluate(rawEvent: string): Promise { + const event = toBaseHookEvent(JSON.parse(rawEvent)); + + if (!event.sessionId) { + return { decision: 'forward', payload: [rawEvent] }; + } + + if (event.hookEventName === 'SomeEvent') { + return await this.onSomeEvent(rawEvent, event); + } + if (event.hookEventName === 'OtherEvent') { + return await this.onOtherEvent(rawEvent, event); + } + // ...one `if` per handled event name... + + return { decision: 'forward', payload: [rawEvent] }; +} +``` + +Each handler is fully responsible for its own `ForwardDecision` — it does not return a bare `string[]` for `evaluate()` to merge afterward. That means that the _only_ trailing fallthrough return (`{ decision: 'forward', payload: [rawEvent] }`) is for event names `evaluate()` doesn't branch on at all. It is not a sink that handled branches route through. + +### 3.3 One parsed object, not two + +Map the adapter's raw event JSON into one typed shape up front. If a handler needs a field the type doesn't yet expose, **extend the type**, don't re-parse `rawEvent` a second time into a second ad-hoc object. The original `rawEvent` _string_ is kept separately only because it is itself the thing that gets forwarded to the spool (`payload: [rawEvent, ...]`) — not because anything needs a second parsed representation of it. + +### 3.4 One place writes to the spool + +`forwardToSpool()` (or equivalent) should be the only call site for `forwardOtlpEventToSpool()` in an adapter, called exactly once from `processOtlpEvent()` after `evaluate()` resolves. No handler forwards anything itself — handlers only _compute_ what should be forwarded (the `ForwardDecision.payload`). This single-chokepoint property is relied on by analytics correctness (every event reaches the spool exactly once, in a known order, under this adapter's registered name). + +### 3.5 Only gate on proxy/daemon readiness when you're about to act on it + +An adapter owns the decision of whether it actually needs the local proxy/daemon for a given event — don't pay for an auth check or a daemon-readiness check on every event just because _some_ events need one. `claude-code-otlp` is the current example: only `onUserPromptSubmit` calls `ensureProxyAuth()` (because that handler's whole job is to gate on it); every other handler goes straight to building its `ForwardDecision` and skips the check entirely, since forwarding to the spool doesn't itself require proxy readiness. Keep that shape in a new adapter — gate per-handler, not globally in `evaluate()` or `processOtlpEvent()`. + +## 4. Adding a new event to an existing adapter + +1. **Decide the shape you need.** Does the new event need extra fields beyond the adapter's current typed hook-event shape? If so, extend that type and its raw→typed mapping function. +2. **Write one handler method**, taking `(rawEvent: string, event: )` and returning `Promise` (unless the handler genuinely needs no typed field off the event — then `rawEvent` alone is fine). +3. **Add exactly one `if` branch** in `evaluate()`, dispatching to the new handler. Do not add a `switch` case, do not touch the trailing fallthrough return. +4. **Never forward anything from inside the handler.** Return the full `ForwardDecision`; the single `forwardToSpool()` call in `processOtlpEvent()` is what actually sends it. +5. **Add a test** for the new branch, mocking at whatever boundary the adapter already mocks (e.g. a transcript/orchestrator module), following the adapter's existing test structure. +6. **Update that adapter's own event-surface table** (see `claude-code-otlp`'s own notes/tests for the current example of such a table). + +## 5. Adding a new `OtlpAgentAdapter` for a different tool + +1. Create `src/agents/plugins//` and implement `OtlpAgentAdapter`: `processOtlpEvent` following the `evaluate()`/`ForwardDecision` shape in §3, plus `prepareAnalyticsFields`. +2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — that name is also what gets passed to `forwardOtlpEventToSpool(rawEvent, name)` and later matched by `otlp-spool/forwarder.ts` via `AgentRegistry.getAnalyticsAgent(name)`. +3. Write a connector that wires the tool's own native hooks/events to `codemie hook --agent ` (see `src/cli/commands/proxy/connectors/claude-code-otlp.ts` for the Claude Code example — the specific hook names, settings file format, and env vars will be entirely different for another tool, and that's expected). +4. Everything from `forwardOtlpEventToSpool` onward (the spool POST, `otlp-spool/forwarder.ts`, the analytics API call) is already shared — no changes needed there as long as step 1-3 hold. + +## 6. Session completeness gating & draining (`otlp-spool/`) + +Past the per-adapter spool POST, the proxy daemon's `otlp-spool/` layer batches a session's spooled data before sending it to the analytics API, rather than forwarding each spooled record immediately. This part is already agent-agnostic — it keys everything off `sessionId`/`agentName`, not off any one adapter's event shape — but it's useful background for understanding what "forward to the spool" actually leads to. + +- **`spool-state.ts`** derives a session's state straight from the spool files on disk. `hooksGroupPresent`/`otelGroupPresent` check whether the hooks stream and any OTEL stream have _ever_ had data written — sticky, so a stream that already reached EOF still counts as "written" even once fully delivered. +- **`completeness-gate.ts`** (`gateDecision(spool, status): GateDecision`) decides what to do with a session on each tick: + - hooks **and** OTEL both present → `'send'`. + - hooks only, waited long enough (and force-forwarding is allowed) → `'hooks-only-force'`; hooks only but not waited enough yet → `'wait'`. + - OTEL only → `'wait'` (today this waits indefinitely for a hooks event to arrive — there is no current path that gives up on an OTEL-only session). + - neither present → `'noop'`. +- **`sweep.ts`** garbage-collects a session once every stream is fully drained (`isSessionDrained`) and a grace period has elapsed since the last producer write — measured from spool-file mtimes, not the status file, since cursor-only updates touch the status file without any new data arriving. Safe to delete: the next producer write recreates the status/spool files from scratch. +- **`session-lock.ts`**'s `withSessionLock` serializes per-session operations but is **not reentrant**: calling it again for the same session from inside an already-running locked callback deadlocks (the inner call chains behind the outer one, which is itself waiting on the inner call to finish). Code running inside a lock must mutate session state in place rather than calling the cursor/status update helpers again. + +## 7. Reference implementation: `claude-code-otlp` + +`ClaudeCodeOtlpPlugin` (`src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts`) is the current (and so far only) `OtlpAgentAdapter`, registered as `claude-code-otlp`. It ingests Claude Code's own native hook events. + +| File | Role | +| ------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- | +| `claude-code-otlp.plugin.ts` | `evaluate()` dispatch, all per-event handlers, `prepareAnalyticsFields`, client-version resolution | +| `claude-code-otlp.types.ts` | `BaseClaudeCodeHookEvent`/`toBaseClaudeCodeHookEvent` (the one parsed shape), re-exports `ForwardDecision` | +| `claude-code-otlp.constants.ts` | `CLAUDE_CODE_OTLP_AGENT_NAME` — the registered adapter name | +| `transcript/orchestrator.ts` | `collectMainTranscriptEvents`, `collectSubagentTranscriptEvents` — Claude-Code-transcript-specific derivation of analytics events | +| `transcript/subagent-usage.ts` | `findSubagentFiles` — discovers every subagent transcript for a session | +| `src/cli/commands/proxy/connectors/claude-code-otlp.ts` | `HOOK_EVENTS` — wires Claude Code's `.claude/settings.json` hooks to `codemie hook --agent claude-code-otlp`; the tool-specific piece from §5 step 3 | + +Claude Code fires 12 distinct hook events today (`HOOK_EVENTS` in that connector): `SessionStart`, `Stop`, `StopFailure`, `SessionEnd`, `UserPromptSubmit`, `PreToolUse`, `PostToolUse`, `PostToolUseFailure`, `SubagentStart`, `SubagentStop`, `PreCompact`, `Notification`. `evaluate()` branches explicitly on `UserPromptSubmit` (auth gate), `Stop`/`PreCompact`/`StopFailure`/`SessionEnd` (transcript parse), and `SubagentStop` (subagent-scoped transcript parse); everything else falls through to the raw passthrough — intentional, since most of those events carry nothing this pipeline needs to transform. + +See `docs/ARCHITECTURE-PROXY.md` for the proxy/plugin layer this hands off to (the spool endpoint, the daemon, the forwarder). From 75f5ffc7f423d082c8e6e6ac9459cc77b8b91427 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Wed, 7 Oct 2026 11:59:00 +0300 Subject: [PATCH 16/35] refactor(proxy): remove unused code and actualize code comments --- .../transcript/orchestrator.ts | 21 ++++---- .../transcript/parse-state.ts | 16 +++--- .../transcript/session-summary.ts | 51 ++++++------------- .../transcript/subagent-usage.ts | 28 ++++------ .../transcript/transcript-reader.ts | 4 +- .../transcript/usage-request.ts | 47 +++++++---------- 6 files changed, 63 insertions(+), 104 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 78f3c0143..4fa131100 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -1,13 +1,14 @@ /** - * Main-transcript trigger orchestration for the `Stop`, `PreCompact`, and `SessionEnd` hook - * events. + * Main-transcript trigger orchestration for the `Stop`, `PreCompact`, `SessionEnd`, and + * `StopFailure` hook events. * * Each hook fire is a fresh CLI process, so this module reloads persisted parse state * (`./parse-state.js`), reads only the transcript lines appended since the last * persisted `mainOffset` (`./transcript-reader.js`), derives/merges * `agent.usage.request` records for those new lines (`./usage-request.js`), persists * state back, and RETURNS one JSON string per completed request plus (on `Stop`/`SessionEnd`) - * one `agent.session.summary` event — it never forwards anything to the spool itself. The caller + * one `agent.session.summary` event (`Stop`/`SessionEnd` only) — it never forwards anything to + * the spool itself. The caller * (the plugin's `processOtlpEvent`, via its per-event handlers) owns forwarding, so there is * exactly one place in the whole analytics pipeline that writes to the spool. * @@ -176,7 +177,7 @@ async function buildFullAccumulator( } /** - * Collect the spool-bound events for one `Stop`/`PreCompact`/`SessionEnd` hook fire. + * Collect the spool-bound events for one `Stop`/`PreCompact`/`SessionEnd`/`StopFailure` hook fire. * * - Loads persisted state, reads only the lines appended since `state.mainOffset`. * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, @@ -186,19 +187,16 @@ async function buildFullAccumulator( * - Returns one `agent.usage.request` JSON string per request key touched by this pass. * - On `Stop`/`SessionEnd` only, also returns exactly one `agent.session.summary` event * (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`) built from a fresh full-file - * recompute (see {@link buildFullAccumulator}). `PreCompact` never returns a + * recompute (see {@link buildFullAccumulator}). `PreCompact`/`StopFailure` never return a * summary. * - Persists state back to disk. * * Never forwards anything itself — the caller is responsible for sending the returned events to * the spool (exactly one place in the pipeline does that). * - * Scoping ruling (a judgment call, since no file in this codebase documents a reliable - * signal for when a *main*-transcript turn enters/exits a "skill context"): every - * main-transcript-derived usage record is scoped as `scopeKind: 'main'`, - * `scopeName: ''` unconditionally. `state.activeSkill` is deliberately left untouched (not read, - * not written) here — it stays available, unused, for a future task that identifies a real - * signal for it. + * Scoping: no reliable transcript signal marks a *main*-transcript turn entering/exiting a + * "skill context", so every main-transcript usage record is scoped `scopeKind: 'main'`, + * `scopeName: ''`. `state.activeSkill` is deliberately neither read nor written. * * Swallows every error internally — never throws into `processOtlpEvent`. */ @@ -266,7 +264,6 @@ export async function collectMainTranscriptEvents( summaryEvent.api_calls = Object.keys(state.openRequests).length; events.push(JSON.stringify(summaryEvent)); } - // PreCompact/StopFailure: usage requests only, no summary — handled by skipping the block above. await saveParseState(sessionId, state); return events; diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts index d10b69ea3..948f63260 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -75,14 +75,7 @@ export async function loadParseState(sessionId: string): Promise; - return { - mainOffset: parsed.mainOffset ?? 0, - subagentOffsets: parsed.subagentOffsets ?? {}, - openRequests: parsed.openRequests ?? {}, - activeSkill: parsed.activeSkill ?? '', - branchCounts: parsed.branchCounts ?? {}, - compactionCount: parsed.compactionCount ?? 0, - }; + return { ...createParseState(), ...parsed }; } catch { return createParseState(); } @@ -132,10 +125,12 @@ export async function withParseStateLock(sessionId: string, fn: () => Promise await mkdir(dirname(lockPath), { recursive: true }); const deadline = Date.now() + LOCK_STALE_MS * 2; + let acquired = false; for (;;) { try { const handle = await open(lockPath, 'wx'); await handle.close(); + acquired = true; break; } catch (err) { if ((err as NodeJS.ErrnoException).code !== 'EEXIST') { @@ -155,6 +150,9 @@ export async function withParseStateLock(sessionId: string, fn: () => Promise try { return await fn(); } finally { - await rm(lockPath, { force: true }).catch(() => {}); + // Only release a lock we hold — when we proceeded unlocked, the file belongs to another process. + if (acquired) { + await rm(lockPath, { force: true }).catch(() => {}); + } } } diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts index b6810c35b..cfd6f257f 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts @@ -1,43 +1,24 @@ /** * `agent.session.summary` builder. * - * Unlike `agent.usage.request`/`agent.subagent.usage`, this event is a running, mutable - * aggregate over an entire session, not a one-shot derivation from a single transcript line - * or file. The caller (the transcript-reader orchestrator) is responsible for - * accumulating a {@link SessionSummaryAccumulator} across the session's parsed transcript lines - * and tool-use/tool-result payloads, tracking `TranscriptParseState.branchCounts` via - * {@link updateBranchCounts} as `git_branch` changes per record, and running - * `extractNamedInvocations()` (`@/agents/plugins/claude/session/claude-named-invocations.js`) - * against the session's messages to get the {@link NamedInvocationCounts} this module's builder - * consumes. This module only aggregates/derives the final event shape from already-computed - * inputs — it never reads a transcript file or calls `extractNamedInvocations()` itself. + * Unlike `agent.usage.request`/`agent.subagent.usage`, this event is a running aggregate over an + * entire session. The orchestrator accumulates a {@link SessionSummaryAccumulator}, tracks + * `TranscriptParseState.branchCounts` via {@link updateBranchCounts}, and runs + * `extractNamedInvocations()` to produce the {@link NamedInvocationCounts} this builder consumes. + * This module only derives the final event shape from those inputs — it never reads a transcript. * - * `event_id`/`schema_version` are stamped later, daemon-side — this builder's output - * carries only an explicit `type` field. + * `event_id`/`schema_version`/`client_version`/`codemie_cli_version` are stamped later, + * daemon-side (`mapHookRecords()`); the output carries only an explicit `type`. * - * Field-shape rulings (see spec.md's `agent.session.summary` section for the full reasoning): - * - `models_used` is emitted as the full `acc.models` count map (not just a list of names) — - * preserves count information `primary_model` alone would discard, consistent with how - * `tool_calls`/`tool_errors`-style maps are emitted elsewhere in this stage. - * - `tool_calls`/`tool_errors` are flattened from `acc.toolCalls`'s combined - * `{ calls, errors }`-per-tool shape into two separate flat `Record` maps, - * matching `buildSubagentUsageEvent`'s (`./subagent-usage.ts`) already-established - * `tool_calls`/`tool_errors` output convention for the sibling `agent.subagent.usage` event. - * - `commands_in_order` is derived as `Object.keys(named.commandInvocations)` — the distinct - * command names in whatever iteration order the object naturally has. `commandInvocations` is - * a COUNT map, not an ordered sequence, so no true chronological invocation order is available - * anywhere in `NamedInvocationCounts`; this is a genuine mismatch between this field's name - * (which implies ordering) and the upstream data shape. Documented here rather than silently - * papered over with a fabricated ordering. - * - `title` has no identified source anywhere in this codebase or the external data-model doc - * (per spec.md's Open risks) — always emitted as a literal empty string, never fabricated. - * - `api_calls` (count of `agent.usage.request` records this session) is intentionally OMITTED - * from this builder's output: `buildSessionSummaryEvent`'s signature has no parameter - * for it, and neither `acc` nor any other input here carries a request count. It is left for - * the orchestrator to merge in afterward, since only that caller has visibility - * into the full set of `agent.usage.request` records it has derived/forwarded this session. - * - `client_version`/`codemie_cli_version` are common fields stamped later via - * `mapHookRecords()` — not this builder's responsibility either. + * Field-shape notes: + * - `models_used` is the full `acc.models` count map, preserving counts `primary_model` discards. + * - `tool_calls`/`tool_errors` are flattened from `acc.toolCalls`'s `{ calls, errors }` shape into + * two flat maps, matching `buildSubagentUsageEvent` (`./subagent-usage.ts`). + * - `commands_in_order` is `Object.keys(named.commandInvocations)`. Upstream is a COUNT map, so no + * chronological order exists; the field name implies more than the data can deliver. + * - `title` has no known source and is always an empty string, never fabricated. + * - `api_calls` is omitted here: no input carries a request count. The orchestrator, which owns + * the full set of `agent.usage.request` records, merges it in afterward. */ import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; diff --git a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts index f567427ed..e94375f92 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts @@ -1,23 +1,19 @@ /** * `agent.subagent.usage` discovery and event builder. * - * Discovery (`findSubagentFiles`) follows the same path convention as the existing, - * `private`/unexported `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts` - * (`//subagents/agent-*.jsonl` + sibling `.meta.json`), but is a - * fresh, smaller implementation returning only the narrower {@link SubagentFile} shape this - * event needs — no `parentAgentId`/`requestShape`/`requestNonInteractive`. + * Discovery (`findSubagentFiles`) uses the same path convention as the private + * `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts` + * (`//subagents/agent-*.jsonl` + sibling `.meta.json`) but returns + * only the narrower {@link SubagentFile} shape this event needs. * - * The event builder (`buildSubagentUsageEvent`) aggregates token/cache fields by summing an - * already-scoped `OpenUsageRequest[]` (`./usage-request.ts`'s shape, reused — not redefined), and passes - * caller-built `tool_calls`/`tool_errors`/`skills_invoked` maps through verbatim: this module has - * no access to a subagent's own tool-use/tool-error/skill-invocation occurrences, only to the - * aggregates its caller already computed from that subagent's transcript. + * The event builder (`buildSubagentUsageEvent`) sums an already-scoped `OpenUsageRequest[]` for + * token/cache fields and passes the caller-built `tool_calls`/`tool_errors`/`skills_invoked` maps + * through verbatim; it has no access to the subagent transcript itself. * - * `description`/`workflow_run`/`worktree` have no identified source anywhere in this codebase - * (per spec.md's "Open risks") — always emitted as literal empty strings, never fabricated. + * `description`/`workflow_run`/`worktree` have no known source and are always empty strings, + * never fabricated. */ -import { existsSync } from 'node:fs'; import { readdir, readFile } from 'node:fs/promises'; import { basename, dirname, join } from 'node:path'; import type { OpenUsageRequest } from './parse-state.js'; @@ -50,10 +46,6 @@ export async function findSubagentFiles(mainTranscriptPath: string): Promise { - let handle; + let handle: FileHandle; try { handle = await open(filePath, 'r'); } catch { diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 224775c29..12b25bcd4 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -2,11 +2,9 @@ * `agent.usage.request` extraction and merge. * * One record per Claude transcript JSONL line that carries `message.usage` — the same - * per-API-response usage shape the existing cost-reporting parser - * (`src/cli/commands/analytics/cost/usage-readers.ts`'s `ClaudeRawMessage`/ - * `extractClaudeUsageRecords`) already reads, adapted here to this task's - * {@link OpenUsageRequest} shape (`parse-state.ts`) instead of that pipeline's - * `UsageRecord`. + * per-API-response usage shape the cost-reporting parser + * (`src/cli/commands/analytics/cost/usage-readers.ts`) reads, adapted to + * {@link OpenUsageRequest} (`parse-state.ts`) instead of that pipeline's `UsageRecord`. * * Claude Code can write more than one JSONL row for the same API response (progressive * streaming chunks, or a later row that fills in `stop_reason` once the turn finishes), so @@ -16,16 +14,14 @@ * not itself check that two records it is asked to merge actually share that identity. */ -import { parseRoutingHeaders, type RoutingHeaderSource } from '../../../../utils/routing-headers.mjs'; -import { parseBackendModelName } from '../../../../utils/bedrock-pricing.mjs'; +import type { RoutingHeaderSource } from '@/utils/routing-headers.mjs'; +import { parseBackendModelName } from '@/utils/bedrock-pricing.mjs'; import type { OpenUsageRequest } from './parse-state.js'; /** - * Loose shape of one transcript JSONL line, mirroring `usage-readers.ts`'s - * `ClaudeRawMessage` plus the additional fields this task's event needs - * (`message.stop_reason`, `message.usage`'s extra nested groups, and the - * top-level `gitBranch`/`isApiError`). Not exported — callers only see - * {@link parseUsageLine}'s `OpenUsageRequest | null` result. + * Loose shape of one transcript JSONL line, mirroring `usage-readers.ts`'s `ClaudeRawMessage` + * plus `message.stop_reason`, `message.usage`'s extra nested groups, and the top-level + * `gitBranch`/`isApiError`. Not exported — callers only see {@link parseUsageLine}'s result. */ interface TranscriptUsageLine { timestamp?: string; @@ -85,7 +81,7 @@ export function parseUsageLine( // Both openRequests' key and resolveEventId's agent.usage.request formula key on // `${requestId}::${model}` — an empty requestId would collide every such line in the session // into one record instead of being skipped. - const requestId = String(parsed.message?.id ?? ''); + const requestId = parsed.message?.id ?? ''; if (!requestId) { return null; } @@ -94,22 +90,17 @@ export function parseUsageLine( // parseBackendModelName() (the raw LiteLLM backend id, when the proxy injected one) wins over // the transcript's own literal `message.model`, since it reflects the actual billable backend // model for a routed/capable-tier request. `modelRaw` keeps the literal, unresolved alias. - const modelRaw = String(parsed.message?.model ?? 'unknown'); - const model = parseBackendModelName(parsed.message) ?? parsed.message?.model ?? 'unknown'; - // Routing metadata itself is not part of OpenUsageRequest's shape, but parsing it mirrors the - // same pattern usage-readers.ts follows for this message object — kept as a documented no-op - // read (not stored) so a future task extending OpenUsageRequest with routing fields has a - // precedent to follow rather than re-deriving the call from scratch. - void parseRoutingHeaders(parsed.message); + const modelRaw = parsed.message?.model ?? 'unknown'; + const model = parseBackendModelName(parsed.message) ?? modelRaw; return { requestId, - model: String(model), + model, modelRaw, - timestamp: String(parsed.timestamp ?? ''), - speed: String(usage.speed ?? ''), - inferenceGeo: String(usage.inference_geo ?? ''), - serviceTier: String(usage.service_tier ?? ''), + timestamp: parsed.timestamp ?? '', + speed: usage.speed ?? '', + inferenceGeo: usage.inference_geo ?? '', + serviceTier: usage.service_tier ?? '', inputTokens: Number(usage.input_tokens ?? 0), cacheCreation5mTokens: Number(usage.cache_creation?.ephemeral_5m_input_tokens ?? 0), cacheCreation1hTokens: Number(usage.cache_creation?.ephemeral_1h_input_tokens ?? 0), @@ -121,9 +112,9 @@ export function parseUsageLine( scopeName, agentId, // Sibling of usage on message, not nested inside it. - stopReason: String(parsed.message?.stop_reason ?? ''), - isApiError: Boolean(parsed.isApiError ?? false), - gitBranch: String(parsed.gitBranch ?? ''), + stopReason: parsed.message?.stop_reason ?? '', + isApiError: Boolean(parsed.isApiError), + gitBranch: parsed.gitBranch ?? '', }; } From b9ed9cd1d463c8b79b384f64319e26108a4823ae Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Wed, 7 Oct 2026 13:08:00 +0300 Subject: [PATCH 17/35] fix(proxy): fix import issue --- .../plugins/claude-code-otlp/transcript/usage-request.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 12b25bcd4..56b6ed952 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -14,8 +14,8 @@ * not itself check that two records it is asked to merge actually share that identity. */ -import type { RoutingHeaderSource } from '@/utils/routing-headers.mjs'; -import { parseBackendModelName } from '@/utils/bedrock-pricing.mjs'; +import { type RoutingHeaderSource } from '../../../../utils/routing-headers.mjs'; +import { parseBackendModelName } from '../../../../utils/bedrock-pricing.mjs'; import type { OpenUsageRequest } from './parse-state.js'; /** From 319306db5553302c4fb923f5ce9a135023954bc7 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Wed, 7 Oct 2026 13:11:05 +0300 Subject: [PATCH 18/35] refactor(proxy): remove prepareAnalyticsFields --- AGENTS.md | 2 +- docs/ARCHITECTURE-OTLP-PLUGIN.md | 61 +++--- src/agents/core/types.ts | 5 - .../__tests__/claude-code-otlp.plugin.test.ts | 139 +++++++------- .../__tests__/client-version-cache.test.ts | 80 ++++++++ .../claude-code-otlp.plugin.ts | 177 ++++++++---------- .../claude-code-otlp.types.ts | 104 +--------- .../claude-code-otlp/client-version-cache.ts | 61 ++++++ .../transcript/__tests__/orchestrator.test.ts | 4 +- .../transcript/orchestrator.ts | 16 +- src/agents/plugins/utils.ts | 4 +- .../otlp-spool/__tests__/forwarder.test.ts | 33 ++-- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 17 -- 13 files changed, 356 insertions(+), 347 deletions(-) create mode 100644 src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/client-version-cache.ts diff --git a/AGENTS.md b/AGENTS.md index 0ed0b7c9f..1e74f599f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -241,7 +241,7 @@ Deterministic docs/knowledge tooling (not agent harnesses): `codebase-memory` (M ## OTel (OTLP) Ingestion -The preferred way to ingest a coding tool's own native telemetry/hook events into CodeMie's analytics pipeline is the `OtlpAgentAdapter` pattern — one plugin per tool, each implementing `processOtlpEvent`/`prepareAnalyticsFields` (`src/agents/core/types.ts`), registered in `AgentRegistry`, invoked via `codemie hook --agent `. Every adapter's `evaluate()` dispatch should follow the same shape (one `ForwardDecision`-returning handler per native event name, no `switch`, a single `forwardToSpool()` call site) and feed the same shared, agent-agnostic spool/forwarder pipeline (`OtlpHookSpoolData` → proxy daemon spool → `otlp-spool/forwarder.ts` → analytics API). +The preferred way to ingest a coding tool's own native telemetry/hook events into CodeMie's analytics pipeline is the `OtlpAgentAdapter` pattern — one plugin per tool, each implementing `processOtlpEvent` (`src/agents/core/types.ts`), registered in `AgentRegistry`, invoked via `codemie hook --agent `. Every adapter's `evaluate()` dispatch should follow the same shape (one `ForwardDecision`-returning handler per native event name, no `switch`, a single `forwardToSpool()` call site) and feed the same shared, agent-agnostic spool/forwarder pipeline (`OtlpHookSpoolData` → proxy daemon spool → `otlp-spool/forwarder.ts` → analytics API). Agent-owned common fields (platform, version, entrypoint, …) are resolved inside `processOtlpEvent`/`evaluate()` at hook-time and merged into the event before it reaches the spool — never via a callback the forwarder makes back into the adapter. `claude-code-otlp` (`src/agents/plugins/claude-code-otlp/`) is the current reference implementation — see **`docs/ARCHITECTURE-OTLP-PLUGIN.md`** for the full contract, the dispatch pattern, how to add a new event to an existing adapter, and how to add a new adapter for another tool (e.g. a future Cursor adapter). diff --git a/docs/ARCHITECTURE-OTLP-PLUGIN.md b/docs/ARCHITECTURE-OTLP-PLUGIN.md index bbd350d7e..2b6ae5601 100644 --- a/docs/ARCHITECTURE-OTLP-PLUGIN.md +++ b/docs/ARCHITECTURE-OTLP-PLUGIN.md @@ -5,19 +5,14 @@ ## 1. What this pattern is -An `OtlpAgentAdapter` is not a chat agent — it never has a conversation. It is the ingestion point for one coding tool's own native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, per-request usage, subagent usage, auth gating, etc.). Each tool that exposes hooks (Claude Code today; potentially Cursor or others later) gets its own adapter plugin, but every adapter implements the same two-method contract and feeds the same spool shape: +An `OtlpAgentAdapter` is not a chat agent — it never has a conversation. It is the ingestion point for one coding tool's own native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, per-request usage, subagent usage, auth gating, etc.). Each tool that exposes hooks (Claude Code today; potentially Cursor or others later) gets its own adapter plugin, but every adapter implements the same one-method contract and feeds the same spool shape: ```ts export interface OtlpAgentAdapter { readonly name: string; readonly type: AgentAdapterType.OTLP; - processOtlpEvent(rawHookInput: string): Promise; - - /** Resolve agent-owned common fields for a single hook event */ - prepareAnalyticsFields( - hookEvent: Record, - ): Promise>; + processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise; } ``` @@ -54,15 +49,13 @@ evaluate(rawEvent) → ForwardDecision ◄── §4: the shape ev (adapter- ▼ specific) otlp-spool/forwarder.ts (background, agent-agnostic) │ - .prepareAnalyticsFields(hookEvent) - │ ▼ CodeMie analytics API ``` - `codemie hook --agent ` (`src/cli/commands/hook.ts`) looks up the adapter via `AgentRegistry.getAnalyticsAgent(name)` and calls `processOtlpEvent` with the raw stdin payload. This part is already agent-agnostic — a new adapter just registers under a new name. - `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is fire-and-forget and shared by every adapter: it POSTs `{ agentName, timestamp, raw }` to the local proxy daemon and swallows every error. A dead/unreachable daemon never blocks or fails the hook, regardless of which adapter called it. -- `otlp-spool/forwarder.ts` is agent-agnostic too — it reads spooled records and dispatches to whichever adapter's `prepareAnalyticsFields` matches `spoolData.agentName` (via `AgentRegistry.getAnalyticsAgent`). A new adapter needs no changes here as long as it implements `prepareAnalyticsFields` and spools under its own registered name. +- `otlp-spool/forwarder.ts` is agent-agnostic by construction, not by convention — it reads spooled records and maps them straight through (`...limited` spread) to the analytics API payload. It never looks up or calls into any adapter. Any agent-owned common field (platform, client version, entrypoint, …) must already be baked into the event by the adapter's own `processOtlpEvent` before it reaches the spool. - Wiring _which_ native hooks/events get pointed at `codemie hook --agent `, and how, is entirely tool-specific — see §5 for how `claude-code-otlp` does it; a different tool will have its own connector. ## 3. The dispatch pattern every `evaluate()` should follow @@ -73,7 +66,7 @@ This is the part that generalizes across adapters, independent of which tool's h ```ts export type ForwardDecision = - | { decision: "forward"; payload: string[] } + | { decision: "forward"; payload: Record[] } | { decision: "block"; reason: string; @@ -81,35 +74,35 @@ export type ForwardDecision = }; ``` -Every native event an adapter processes should resolve to exactly one of these. `forward` carries the full list of raw JSON strings to push to the spool (the original raw event, plus zero or more derived analytics events). `block` stops the hook and logs a reason; what `hookSpecificOutput` means (e.g. suppressing a prompt) is specific to the tool and the event, not to this pattern. +Every native event an adapter processes should resolve to exactly one of these. `forward` carries the full list of parsed-object records to push to the spool (the original parsed event, plus zero or more derived analytics events). `block` stops the hook and logs a reason; what `hookSpecificOutput` means (e.g. suppressing a prompt) is specific to the tool and the event, not to this pattern. ### 3.2 One handler per event name — no `switch`, no shared merge step ```ts -private async evaluate(rawEvent: string): Promise { - const event = toBaseHookEvent(JSON.parse(rawEvent)); - - if (!event.sessionId) { - return { decision: 'forward', payload: [rawEvent] }; +private async evaluate(parsed: Record): Promise { + const sessionId = readString(parsed, 'session_id'); + if (!sessionId) { + return { decision: 'forward', payload: [parsed] }; } - if (event.hookEventName === 'SomeEvent') { - return await this.onSomeEvent(rawEvent, event); + const hookEventName = readString(parsed, 'hook_event_name'); + if (hookEventName === 'SomeEvent') { + return await this.onSomeEvent(parsed); } - if (event.hookEventName === 'OtherEvent') { - return await this.onOtherEvent(rawEvent, event); + if (hookEventName === 'OtherEvent') { + return await this.onOtherEvent(parsed); } // ...one `if` per handled event name... - return { decision: 'forward', payload: [rawEvent] }; + return { decision: 'forward', payload: [parsed] }; } ``` -Each handler is fully responsible for its own `ForwardDecision` — it does not return a bare `string[]` for `evaluate()` to merge afterward. That means that the _only_ trailing fallthrough return (`{ decision: 'forward', payload: [rawEvent] }`) is for event names `evaluate()` doesn't branch on at all. It is not a sink that handled branches route through. +Each handler is fully responsible for its own `ForwardDecision` — it does not return a bare array for `evaluate()` to merge afterward. That means that the _only_ trailing fallthrough return (`{ decision: 'forward', payload: [parsed] }`) is for event names `evaluate()` doesn't branch on at all. It is not a sink that handled branches route through. -### 3.3 One parsed object, not two +### 3.3 One parsed object, read directly — no typed projection -Map the adapter's raw event JSON into one typed shape up front. If a handler needs a field the type doesn't yet expose, **extend the type**, don't re-parse `rawEvent` a second time into a second ad-hoc object. The original `rawEvent` _string_ is kept separately only because it is itself the thing that gets forwarded to the spool (`payload: [rawEvent, ...]`) — not because anything needs a second parsed representation of it. +`rawEvent` is parsed exactly once, at the top of `processOtlpEvent()`, into `parsed: Record` — the tool's own native (snake_case) field names, unchanged. That same object is both what handlers read fields off of (via small helpers like `readString(parsed, 'session_id')`/`readOptionalString(parsed, 'agent_id')`) and what gets forwarded to the spool. There is deliberately no second, camelCase-renamed "typed hook event" object: a 1:1 field-rename mapper adds a name to type-check against but zero runtime validation (a missing field becomes `undefined`/`''` either way), so for a handful of fields read in a handful of places it is not worth carrying a second shape through `evaluate()` and every handler signature. If a future adapter's handlers need many more fields in many more places, revisit this call — a typed projection becomes worth its weight once the number of call sites justifies it. ### 3.4 One place writes to the spool @@ -130,10 +123,16 @@ An adapter owns the decision of whether it actually needs the local proxy/daemon ## 5. Adding a new `OtlpAgentAdapter` for a different tool -1. Create `src/agents/plugins//` and implement `OtlpAgentAdapter`: `processOtlpEvent` following the `evaluate()`/`ForwardDecision` shape in §3, plus `prepareAnalyticsFields`. -2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — that name is also what gets passed to `forwardOtlpEventToSpool(rawEvent, name)` and later matched by `otlp-spool/forwarder.ts` via `AgentRegistry.getAnalyticsAgent(name)`. +1. Create `src/agents/plugins//` and implement `OtlpAgentAdapter`: just `processOtlpEvent`, following the `evaluate()`/`ForwardDecision` shape in §3 — enriching its own events with any agent-owned common fields (see §5a) before forwarding to the spool. `claude-code-otlp` is the current example. +2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — that name is also what gets passed to `forwardOtlpEventToSpool(event, name)` and later matched by `otlp-spool/forwarder.ts` via `AgentRegistry.getAnalyticsAgent(name)` (still used by `hook.ts`'s dispatch, see §2). 3. Write a connector that wires the tool's own native hooks/events to `codemie hook --agent ` (see `src/cli/commands/proxy/connectors/claude-code-otlp.ts` for the Claude Code example — the specific hook names, settings file format, and env vars will be entirely different for another tool, and that's expected). -4. Everything from `forwardOtlpEventToSpool` onward (the spool POST, `otlp-spool/forwarder.ts`, the analytics API call) is already shared — no changes needed there as long as step 1-3 hold. +4. Everything from `forwardOtlpEventToSpool` onward (the spool POST, `otlp-spool/forwarder.ts`, the analytics API call) is already shared and agent-agnostic by construction — no changes needed there as long as step 1-3 hold. + +## 5a. Agent-owned fields that are expensive to resolve + +Resolve agent-owned fields (platform, version, entrypoint, …) **inside the adapter's own hook-time process**, not via a callback from the agent-agnostic forwarder — the forwarder (`otlp-spool/forwarder.ts`) runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, and reading agent/tool state (e.g. an env var) there reflects the daemon's own startup environment, not the per-invocation environment that actually produced the event. + +If resolving a field is expensive (a subprocess spawn, a network call), back it with a small file cache under `getCodemiePath()` rather than relying on in-process memoization — the hook-time process is fresh per event and does not persist across events, so an in-memory cache buys nothing. `claude-code-otlp/client-version-cache.ts` is the current example: a TTL file cache around `claude --version`. ## 6. Session completeness gating & draining (`otlp-spool/`) @@ -154,9 +153,11 @@ Past the per-adapter spool POST, the proxy daemon's `otlp-spool/` layer batches | File | Role | | ------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- | -| `claude-code-otlp.plugin.ts` | `evaluate()` dispatch, all per-event handlers, `prepareAnalyticsFields`, client-version resolution | -| `claude-code-otlp.types.ts` | `BaseClaudeCodeHookEvent`/`toBaseClaudeCodeHookEvent` (the one parsed shape), re-exports `ForwardDecision` | +| `claude-code-otlp.plugin.ts` | `evaluate()` dispatch, all per-event handlers, hook-time common-field enrichment, allowlist gate + lazy proxy start | +| `client-version-cache.ts` | TTL file cache around `claude --version`, backing the `client_version` common field (see §5a) | +| `claude-code-otlp.types.ts` | `ForwardDecision` | | `claude-code-otlp.constants.ts` | `CLAUDE_CODE_OTLP_AGENT_NAME` — the registered adapter name | +| `claude-code-otlp.allowlist.ts` | Per-project allowlist gating (`isProjectTracked`/`readAllowlistState`) — see the INVARIANT comment in `processOtlpEvent` | | `transcript/orchestrator.ts` | `collectMainTranscriptEvents`, `collectSubagentTranscriptEvents` — Claude-Code-transcript-specific derivation of analytics events | | `transcript/subagent-usage.ts` | `findSubagentFiles` — discovers every subagent transcript for a session | | `src/cli/commands/proxy/connectors/claude-code-otlp.ts` | `HOOK_EVENTS` — wires Claude Code's `.claude/settings.json` hooks to `codemie hook --agent claude-code-otlp`; the tool-specific piece from §5 step 3 | diff --git a/src/agents/core/types.ts b/src/agents/core/types.ts index fa171742f..5876eb9c7 100644 --- a/src/agents/core/types.ts +++ b/src/agents/core/types.ts @@ -744,11 +744,6 @@ export interface OtlpAgentAdapter { * spool. */ processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise; - - /** - * Resolve agent-owned common fields for a single hook event - */ - prepareAnalyticsFields(hookEvent: Record): Promise>; } export interface OtlpAdapterDeps { diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index 2bc1425ee..a3b8f0210 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -11,13 +11,13 @@ vi.mock('@/utils/logger.js', () => ({ logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, })); -const execMock = vi.fn(); +const resolveClientVersionMock = vi.fn(); const collectMainTranscriptEventsMock = vi.fn(); const collectSubagentTranscriptEventsMock = vi.fn(); const findSubagentFilesMock = vi.fn(); -vi.mock('@/utils/exec.js', () => ({ - exec: execMock, +vi.mock('../client-version-cache.js', () => ({ + resolveClientVersion: resolveClientVersionMock, })); vi.mock('../transcript/orchestrator.js', () => ({ @@ -44,7 +44,10 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { const plugin = new ClaudeCodeOtlpPlugin(); const ensureOtlpProxy = vi.fn(async () => {}); - beforeEach(() => vi.clearAllMocks()); + beforeEach(() => { + vi.clearAllMocks(); + resolveClientVersionMock.mockResolvedValue('2.1.23'); + }); it('does nothing for untracked projects, for every event', async () => { vi.mocked(isProjectTracked).mockResolvedValue(false); @@ -70,16 +73,31 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { }); }); -describe('ClaudeCodeOtlpPlugin.prepareAnalyticsFields', () => { +describe('ClaudeCodeOtlpPlugin hook-time enrichment', () => { + const plugin = new ClaudeCodeOtlpPlugin(); + const ensureOtlpProxy = vi.fn(async () => {}); const originalEntrypoint = process.env.CLAUDE_CODE_ENTRYPOINT; + function hookEvent(overrides: Record = {}): Record { + return { + session_id: 'sid-1', + transcript_path: '/tmp/transcript.jsonl', + cwd: '/repo', + hook_event_name: 'Stop', + ...overrides, + }; + } + beforeEach(() => { - execMock.mockReset(); - execMock.mockResolvedValue({ code: 0, stdout: '2.1.23 (Claude Code)', stderr: '', signal: null }); + vi.clearAllMocks(); + vi.mocked(isProjectTracked).mockResolvedValue(true); + resolveClientVersionMock.mockResolvedValue('2.1.23'); + collectMainTranscriptEventsMock.mockResolvedValue([]); + collectSubagentTranscriptEventsMock.mockResolvedValue([]); + findSubagentFilesMock.mockResolvedValue([]); }); afterEach(() => { - vi.restoreAllMocks(); if (originalEntrypoint === undefined) { delete process.env.CLAUDE_CODE_ENTRYPOINT; } else { @@ -87,57 +105,34 @@ describe('ClaudeCodeOtlpPlugin.prepareAnalyticsFields', () => { } }); - it('returns platform, entrypoint, and client_version', async () => { - const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + it('enriches the forwarded event with platform, entrypoint, and client_version', async () => { process.env.CLAUDE_CODE_ENTRYPOINT = 'cli'; + const rawEvent = JSON.stringify(hookEvent()); - const plugin = new ClaudeCodeOtlpPlugin(); - const fields = await plugin.prepareAnalyticsFields({}); - - expect(fields.platform).toBe('claude-code'); - expect(fields.entrypoint).toBe('cli'); - expect(fields.client_version).toBe('2.1.23'); - }); - - it('includes agent_id/agent_type when present on the hook event', async () => { - const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); - - const plugin = new ClaudeCodeOtlpPlugin(); - const fields = await plugin.prepareAnalyticsFields({ agent_id: 'sub-1', agent_type: 'explore' }); - - expect(fields.agent_id).toBe('sub-1'); - expect(fields.agent_type).toBe('explore'); - }); - - it('omits agent_id/agent_type when absent from the hook event', async () => { - const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); - - const plugin = new ClaudeCodeOtlpPlugin(); - const fields = await plugin.prepareAnalyticsFields({}); - - expect(fields).not.toHaveProperty('agent_id'); - expect(fields).not.toHaveProperty('agent_type'); - }); - - it('spawns `claude --version` only once across two prepareAnalyticsFields calls', async () => { - const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); - - const plugin = new ClaudeCodeOtlpPlugin(); - await plugin.prepareAnalyticsFields({}); - await plugin.prepareAnalyticsFields({ agent_id: 'sub-2' }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(execMock).toHaveBeenCalledTimes(1); - expect(execMock).toHaveBeenCalledWith('claude', ['--version']); + expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(1); + const [forwarded] = vi.mocked(forwardOtlpEventToSpool).mock.calls[0]; + expect(forwarded.platform).toBe('claude-code'); + expect(forwarded.entrypoint).toBe('cli'); + expect(forwarded.client_version).toBe('2.1.23'); }); - it('falls back to an empty client_version when `claude --version` throws', async () => { - execMock.mockRejectedValue(new Error('ENOENT')); - const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + it('preserves agent_id/agent_type already present on the raw event (SubagentStop)', async () => { + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-1.jsonl', + agent_id: 'sub-1', + agent_type: 'explore', + }) + ); - const plugin = new ClaudeCodeOtlpPlugin(); - const fields = await plugin.prepareAnalyticsFields({}); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - expect(fields.client_version).toBe(''); + const [forwarded] = vi.mocked(forwardOtlpEventToSpool).mock.calls[0]; + expect(forwarded.agent_id).toBe('sub-1'); + expect(forwarded.agent_type).toBe('explore'); }); }); @@ -161,6 +156,8 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { collectSubagentTranscriptEventsMock.mockResolvedValue([]); findSubagentFilesMock.mockReset(); findSubagentFilesMock.mockResolvedValue([]); + resolveClientVersionMock.mockReset(); + resolveClientVersionMock.mockResolvedValue('2.1.23'); vi.mocked(forwardOtlpEventToSpool).mockReset(); vi.mocked(isProjectTracked).mockResolvedValue(true); }); @@ -174,7 +171,8 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { async (hookEventName) => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); - const rawEvent = JSON.stringify(hookEvent({ hook_event_name: hookEventName })); + const parsedEvent = hookEvent({ hook_event_name: hookEventName }); + const rawEvent = JSON.stringify(parsedEvent); await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); @@ -183,7 +181,10 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { '/tmp/transcript.jsonl', hookEventName ); - expect(forwardOtlpEventToSpool).toHaveBeenCalledWith(rawEvent, 'claude-code-otlp'); + expect(forwardOtlpEventToSpool).toHaveBeenCalledWith( + expect.objectContaining(parsedEvent), + 'claude-code-otlp' + ); } ); @@ -270,22 +271,27 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { it('forwards the raw event and skips all transcript-parse dispatch when session_id is empty', async () => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); - const rawEvent = JSON.stringify(hookEvent({ session_id: '', hook_event_name: 'Stop' })); + const parsedEvent = hookEvent({ session_id: '', hook_event_name: 'Stop' }); + const rawEvent = JSON.stringify(parsedEvent); await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); expect(collectMainTranscriptEventsMock).not.toHaveBeenCalled(); - expect(forwardOtlpEventToSpool).toHaveBeenCalledWith(rawEvent, 'claude-code-otlp'); + expect(forwardOtlpEventToSpool).toHaveBeenCalledWith( + expect.objectContaining(parsedEvent), + 'claude-code-otlp' + ); }); it('forwards every event a per-event handler returns (the raw event plus any derived events) through the single forwardToSpool path, in order', async () => { - const derivedUsageEvent = JSON.stringify({ type: 'agent.usage.request' }); - const derivedSummaryEvent = JSON.stringify({ type: 'agent.session.summary' }); + const derivedUsageEvent = { type: 'agent.usage.request' }; + const derivedSummaryEvent = { type: 'agent.session.summary' }; collectMainTranscriptEventsMock.mockResolvedValue([derivedUsageEvent, derivedSummaryEvent]); const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); - const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'Stop' })); + const parsedEvent = hookEvent({ hook_event_name: 'Stop' }); + const rawEvent = JSON.stringify(parsedEvent); await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); @@ -293,10 +299,17 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { // themselves (they are mocked here to just return data) — every event that reaches the spool // mock arrived via forwardToSpool, called exactly once from processOtlpEvent. expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(3); - expect(vi.mocked(forwardOtlpEventToSpool).mock.calls.map(([raw]) => raw)).toEqual([ - rawEvent, - derivedUsageEvent, - derivedSummaryEvent, + expect(vi.mocked(forwardOtlpEventToSpool).mock.calls.map(([record]) => record)).toEqual([ + expect.objectContaining(parsedEvent), + expect.objectContaining(derivedUsageEvent), + expect.objectContaining(derivedSummaryEvent), ]); }); + + it('rejects when rawEvent is malformed JSON', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + + await expect(plugin.processOtlpEvent('not json', { ensureOtlpProxy })).rejects.toThrow(); + }); }); diff --git a/src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts new file mode 100644 index 000000000..054e44357 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts @@ -0,0 +1,80 @@ +import { describe, it, expect, vi, beforeEach } from 'vitest'; + +const readFileMock = vi.fn(); +const writeFileMock = vi.fn(); +const mkdirMock = vi.fn(); +const execMock = vi.fn(); + +vi.mock('node:fs/promises', () => ({ + readFile: readFileMock, + writeFile: writeFileMock, + mkdir: mkdirMock, +})); + +vi.mock('@/utils/exec.js', () => ({ + exec: execMock, +})); + +describe('client-version-cache', () => { + beforeEach(() => { + vi.resetModules(); + readFileMock.mockReset(); + writeFileMock.mockReset().mockResolvedValue(undefined); + mkdirMock.mockReset().mockResolvedValue(undefined); + execMock.mockReset(); + execMock.mockResolvedValue({ code: 0, stdout: '2.1.23 (Claude Code)', stderr: '', signal: null }); + }); + + it('cold cache: execs and writes a new cache entry', async () => { + readFileMock.mockRejectedValue(new Error('ENOENT')); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.23'); + expect(execMock).toHaveBeenCalledWith('claude', ['--version']); + expect(writeFileMock).toHaveBeenCalledTimes(1); + }); + + it('warm cache within TTL: does not exec', async () => { + readFileMock.mockResolvedValue(JSON.stringify({ version: '2.1.0', resolvedAt: Date.now() })); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.0'); + expect(execMock).not.toHaveBeenCalled(); + }); + + it('stale cache past TTL: re-execs', async () => { + const twoHoursAgo = Date.now() - 2 * 60 * 60 * 1000; + readFileMock.mockResolvedValue(JSON.stringify({ version: '2.0.0', resolvedAt: twoHoursAgo })); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.23'); + expect(execMock).toHaveBeenCalledTimes(1); + }); + + it('exec failure: returns empty string and does not write a cache entry', async () => { + readFileMock.mockRejectedValue(new Error('ENOENT')); + execMock.mockRejectedValue(new Error('ENOENT')); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe(''); + expect(writeFileMock).not.toHaveBeenCalled(); + }); + + it('corrupt cache file: treated as cold and re-execs', async () => { + readFileMock.mockResolvedValue('not json'); + const { resolveClientVersion } = await import('../client-version-cache.js'); + + const version = await resolveClientVersion(); + + expect(version).toBe('2.1.23'); + expect(execMock).toHaveBeenCalledTimes(1); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index 5194a8a85..8b16a6a32 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -5,7 +5,7 @@ import { logger } from '@/utils/logger.js'; import { ConfigLoader } from '@/utils/config.js'; import { AgentAdapterType, OtlpAdapterDeps, OtlpAgentAdapter } from '@/agents/core/types.js'; import { CLAUDE_CODE_OTLP_AGENT_NAME } from './claude-code-otlp.constants.js'; -import { BaseClaudeCodeHookEvent, ForwardDecision, toBaseClaudeCodeHookEvent } from './claude-code-otlp.types.js'; +import { ForwardDecision } from './claude-code-otlp.types.js'; import { forwardOtlpEventToSpool } from '../utils.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; import { @@ -14,18 +14,15 @@ import { type SubagentFile, } from './transcript/orchestrator.js'; import { findSubagentFiles } from './transcript/subagent-usage.js'; -import { exec } from '@/utils/exec.js'; +import { resolveClientVersion } from './client-version-cache.js'; export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; public readonly type = AgentAdapterType.OTLP; - /** Cached across calls so a high-frequency hook (e.g. PostToolUse) doesn't spawn a subprocess per call. */ - private clientVersion: string | undefined; - private readonly platform = 'claude-code'; - public async processOtlpEvent(rawEvent: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { - const event = toBaseClaudeCodeHookEvent(JSON.parse(rawEvent)); + const parsed = JSON.parse(rawEvent) as Record; + const cwd = readString(parsed, 'cwd'); // INVARIANT - do not weaken. An untracked project must produce NO hooks data // in the daemon spool, must not start the daemon, and must not run the SSO @@ -34,7 +31,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { // data, and skips sessions that only have OTEL data. Forwarding a hook event // for an untracked project would make the session sendable and leak its // data to the backend. - const isTracked = await isProjectTracked(event.cwd, await readAllowlistState()); + const isTracked = await isProjectTracked(cwd, await readAllowlistState()); if (!isTracked) { logger.debug('[Claude Code OTLP plugin] project not in analytics allowlist, ignoring hook event'); return; @@ -42,86 +39,113 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { await ensureOtlpProxy(this.name); - const evaluation = await this.evaluate(rawEvent); + const evaluation = await this.evaluate(parsed); if (evaluation.decision === 'block') { logger.error(`[Claude Code OTLP plugin] Blocking prompt: ${evaluation.reason}`); console.log(JSON.stringify(evaluation)); return; } - this.forwardToSpool(evaluation.payload); + this.forwardToSpool(await this.withCommonFields(evaluation.payload)); } - private async evaluate(rawEvent: string): Promise { - const event = toBaseClaudeCodeHookEvent(JSON.parse(rawEvent)); + /** Resolved once per call so every record in the batch shares one client-version lookup. */ + private async withCommonFields(records: Record[]): Promise[]> { + const common = { + platform: 'claude-code', + entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', + client_version: await resolveClientVersion(), + }; + return records.map((record) => ({ ...record, ...common })); + } - if (!event.sessionId) { - return { decision: 'forward', payload: [rawEvent] }; + private async evaluate(parsed: Record): Promise { + const sessionId = readString(parsed, 'session_id'); + if (!sessionId) { + return { decision: 'forward', payload: [parsed] }; } - if (event.hookEventName === 'UserPromptSubmit') { - return await this.onUserPromptSubmit(rawEvent); + const hookEventName = readString(parsed, 'hook_event_name'); + if (hookEventName === 'UserPromptSubmit') { + return await this.onUserPromptSubmit(parsed); } - if (event.hookEventName === 'Stop') { - return await this.onStopEvent(rawEvent, event); + if (hookEventName === 'Stop') { + return await this.onStopEvent(parsed); } - if (event.hookEventName === 'PreCompact') { - return await this.onPreCompactEvent(rawEvent, event); + if (hookEventName === 'PreCompact') { + return await this.onPreCompactEvent(parsed); } - if (event.hookEventName === 'StopFailure') { - return await this.onStopFailureEvent(rawEvent, event); + if (hookEventName === 'StopFailure') { + return await this.onStopFailureEvent(parsed); } - if (event.hookEventName === 'SessionEnd') { - return await this.onSessionEndEvent(rawEvent, event); + if (hookEventName === 'SessionEnd') { + return await this.onSessionEndEvent(parsed); } - if (event.hookEventName === 'SubagentStop') { - return await this.onSubagentStopEvent(rawEvent, event); + if (hookEventName === 'SubagentStop') { + return await this.onSubagentStopEvent(parsed); } - return { decision: 'forward', payload: [rawEvent] }; + return { decision: 'forward', payload: [parsed] }; } - private async onStopEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { - const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'Stop'); - return { decision: 'forward', payload: [rawEvent, ...derived] }; + private async onStopEvent(parsed: Record): Promise { + const derived = await collectMainTranscriptEvents( + readString(parsed, 'session_id'), + readString(parsed, 'transcript_path'), + 'Stop' + ); + return { decision: 'forward', payload: [parsed, ...derived] }; } - private async onPreCompactEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { - const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'PreCompact'); - return { decision: 'forward', payload: [rawEvent, ...derived] }; + private async onPreCompactEvent(parsed: Record): Promise { + const derived = await collectMainTranscriptEvents( + readString(parsed, 'session_id'), + readString(parsed, 'transcript_path'), + 'PreCompact' + ); + return { decision: 'forward', payload: [parsed, ...derived] }; } - private async onStopFailureEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { - const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'StopFailure'); - return { decision: 'forward', payload: [rawEvent, ...derived] }; + private async onStopFailureEvent(parsed: Record): Promise { + const derived = await collectMainTranscriptEvents( + readString(parsed, 'session_id'), + readString(parsed, 'transcript_path'), + 'StopFailure' + ); + return { decision: 'forward', payload: [parsed, ...derived] }; } - private async onSessionEndEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { - const derived = await collectMainTranscriptEvents(event.sessionId, event.transcriptPath, 'SessionEnd'); + private async onSessionEndEvent(parsed: Record): Promise { + const sessionId = readString(parsed, 'session_id'); + const transcriptPath = readString(parsed, 'transcript_path'); + const derived = await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd'); // Backstop: guarantee every subagent discovered for this session gets at least one // agent.subagent.usage event, even when its own SubagentStop hook never fired. - const subagentFiles = await findSubagentFiles(event.transcriptPath); + const subagentFiles = await findSubagentFiles(transcriptPath); for (const file of subagentFiles) { - derived.push(...(await collectSubagentTranscriptEvents(event.sessionId, file))); + derived.push(...(await collectSubagentTranscriptEvents(sessionId, file))); } - return { decision: 'forward', payload: [rawEvent, ...derived] }; + return { decision: 'forward', payload: [parsed, ...derived] }; } - private async onSubagentStopEvent(rawEvent: string, event: BaseClaudeCodeHookEvent): Promise { - if (!event.agentTranscriptPath) { - return { decision: 'forward', payload: [rawEvent] }; + private async onSubagentStopEvent(parsed: Record): Promise { + const agentTranscriptPath = readOptionalString(parsed, 'agent_transcript_path'); + if (!agentTranscriptPath) { + return { decision: 'forward', payload: [parsed] }; } const subagentFile: SubagentFile = { - agentId: event.agentId ?? basename(event.agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''), - filePath: event.agentTranscriptPath, - toolUseId: event.toolUseId, - agentType: event.agentType, + agentId: + readOptionalString(parsed, 'agent_id') ?? + basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''), + filePath: agentTranscriptPath, + toolUseId: readOptionalString(parsed, 'tool_use_id'), + agentType: readOptionalString(parsed, 'agent_type'), }; - const derived = await collectSubagentTranscriptEvents(event.sessionId, subagentFile); - return { decision: 'forward', payload: [rawEvent, ...derived] }; + const derived = await collectSubagentTranscriptEvents(readString(parsed, 'session_id'), subagentFile); + return { decision: 'forward', payload: [parsed, ...derived] }; } private async ensureProxyAuth(): Promise { @@ -134,59 +158,20 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { }); } - private forwardToSpool(rawEvents: string[]): void { + private forwardToSpool(events: Record[]): void { // Intentionally not awaited: forwardOtlpEventToSpool is fire-and-forget. - for (const rawEvent of rawEvents) { - forwardOtlpEventToSpool(rawEvent, CLAUDE_CODE_OTLP_AGENT_NAME); - } - } - - public async prepareAnalyticsFields( - hookEvent: Record - ): Promise> { - const fields: Record = { - platform: this.platform, - entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', - client_version: await this.getClientVersion(), - }; - - const agentId = readOptionalString(hookEvent, 'agent_id'); - if (agentId !== undefined) { - fields.agent_id = agentId; + for (const event of events) { + forwardOtlpEventToSpool(event, CLAUDE_CODE_OTLP_AGENT_NAME); } - const agentType = readOptionalString(hookEvent, 'agent_type'); - if (agentType !== undefined) { - fields.agent_type = agentType; - } - - return fields; - } - - private async getClientVersion(): Promise { - if (this.clientVersion === undefined) { - this.clientVersion = await this.resolveClientVersion(); - } - return this.clientVersion; } - private async resolveClientVersion(): Promise { - try { - const result = await exec('claude', ['--version']); - const trimmed = result.stdout.trim(); - const versionMatch = trimmed.match(/^(\d+\.\d+\.\d+)/); - return versionMatch ? versionMatch[1] : trimmed; - } catch { - return ''; - } - } - - private async onUserPromptSubmit(rawEvent: string): Promise { + private async onUserPromptSubmit(parsed: Record): Promise { const authResult = await this.ensureProxyAuth(); if (authResult.ok) { return { decision: 'forward', - payload: [rawEvent], + payload: [parsed], } } @@ -209,3 +194,7 @@ function readOptionalString(record: Record, key: string): strin const value = record[key]; return typeof value === 'string' ? value : undefined; } + +function readString(record: Record, key: string): string { + return readOptionalString(record, key) ?? ''; +} diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts index 497b2ab3a..5cb254264 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts @@ -1,105 +1,3 @@ -interface RawBaseClaudeCodeHookEvent { - /** Current session identifier */ - session_id: string; - - /** - * UUID identifying the user prompt currently being processed. - * Matches the prompt.id attribute on OpenTelemetry events. - * Absent until the first user input. Requires Claude Code v2.1.196 or later. - */ - prompt_id?: string; - - /** - * Path to conversation JSON. Written asynchronously and may lag - * the in-memory conversation. - */ - transcript_path: string; - - /** Current working directory when the hook is invoked */ - cwd: string; - - /** - * Path to the session’s scratchpad directory for temporary working files. - * Absent when no scratchpad exists or the temp directory is unavailable. - * Requires Claude Code v2.1.257 or later. - */ - scratchpad_dir?: string; - - /** - * Current permission mode: "default", "plan", "acceptEdits", "auto", - * "dontAsk", or "bypassPermissions". Not all events receive this field. - */ - permission_mode?: "default" | "plan" | "acceptEdits" | "auto" | "dontAsk" | "bypassPermissions"; - - /** - * Object with a level field holding the effort level in effect when the hook runs. - * Present for events that fire within a tool-use context when supported by the model. - */ - effort?: { - level: "low" | "medium" | "high" | "xhigh" | "max"; - }; - - /** Name of the event that fired */ - hook_event_name: string; - - /** `SubagentStop`-only: path to the subagent's own transcript file. */ - agent_transcript_path?: string; - - /** `SubagentStop`-only: identifier of the subagent, when the hook payload carries one. */ - agent_id?: string; - - /** `SubagentStop`-only: the subagent's declared type (e.g. `explore`). */ - agent_type?: string; - - /** `SubagentStop`-only: the tool_use_id of the Task invocation that spawned the subagent. */ - tool_use_id?: string; -} - -/** - * camelCase-keyed mirror of {@link RawBaseClaudeCodeHookEvent}. - */ -export interface BaseClaudeCodeHookEvent { - sessionId: string; - promptId?: string; - transcriptPath: string; - cwd: string; - scratchpadDir?: string; - permissionMode?: "default" | "plan" | "acceptEdits" | "auto" | "dontAsk" | "bypassPermissions"; - effort?: { - level: "low" | "medium" | "high" | "xhigh" | "max"; - }; - hookEventName: string; - - /** `SubagentStop`-only: path to the subagent's own transcript file. */ - agentTranscriptPath?: string; - - /** `SubagentStop`-only: identifier of the subagent, when the hook payload carries one. */ - agentId?: string; - - /** `SubagentStop`-only: the subagent's declared type (e.g. `explore`). */ - agentType?: string; - - /** `SubagentStop`-only: the tool_use_id of the Task invocation that spawned the subagent. */ - toolUseId?: string; -} - -export function toBaseClaudeCodeHookEvent(raw: RawBaseClaudeCodeHookEvent): BaseClaudeCodeHookEvent { - return { - sessionId: raw.session_id, - promptId: raw.prompt_id, - transcriptPath: raw.transcript_path, - cwd: raw.cwd, - scratchpadDir: raw.scratchpad_dir, - permissionMode: raw.permission_mode, - effort: raw.effort ? { level: raw.effort.level } : undefined, - hookEventName: raw.hook_event_name, - agentTranscriptPath: raw.agent_transcript_path, - agentId: raw.agent_id, - agentType: raw.agent_type, - toolUseId: raw.tool_use_id, - }; -} - export type ForwardDecision = - | { decision: 'forward'; payload: string[] } + | { decision: 'forward'; payload: Record[] } | { decision: 'block'; reason: string, hookSpecificOutput: Record }; diff --git a/src/agents/plugins/claude-code-otlp/client-version-cache.ts b/src/agents/plugins/claude-code-otlp/client-version-cache.ts new file mode 100644 index 000000000..743ff45a8 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/client-version-cache.ts @@ -0,0 +1,61 @@ +import { readFile, writeFile, mkdir } from 'node:fs/promises'; +import { dirname } from 'node:path'; +import { exec } from '@/utils/exec.js'; +import { getCodemiePath } from '@/utils/paths.js'; +import { logger } from '@/utils/logger.js'; + +const CACHE_PATH = getCodemiePath('cache', 'claude-code-client-version.json'); +const TTL_MS = 1 * 60 * 60 * 1000; // 1h - staleness here only affects an analytics label, not behavior. + +interface CacheEntry { + version: string; + resolvedAt: number; +} + +async function readCache(): Promise { + try { + const raw = await readFile(CACHE_PATH, 'utf-8'); + const entry = JSON.parse(raw) as CacheEntry; + if (Date.now() - entry.resolvedAt < TTL_MS) { + return entry; + } + } catch { + /* missing/corrupt cache: fall through to re-resolve */ + } + return undefined; +} + +async function writeCache(version: string): Promise { + try { + await mkdir(dirname(CACHE_PATH), { recursive: true }); + await writeFile(CACHE_PATH, JSON.stringify({ version, resolvedAt: Date.now() } satisfies CacheEntry)); + } catch (err) { + logger.debug('client-version-cache: write failed', err instanceof Error ? err.message : String(err)); + } +} + +async function execClaudeVersion(): Promise { + try { + const result = await exec('claude', ['--version']); + const trimmed = result.stdout.trim(); + const match = trimmed.match(/^(\d+\.\d+\.\d+)/); + return match ? match[1] : trimmed; + } catch { + return ''; + } +} + +/** Resolve the installed `claude` CLI version, backed by a TTL file cache so a fresh + * `codemie hook` process (one per hook event) doesn't spawn `claude --version` every time. */ +export async function resolveClientVersion(): Promise { + const cached = await readCache(); + if (cached) { + return cached.version; + } + + const version = await execClaudeVersion(); + if (version) { + await writeCache(version); + } + return version; +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index 26572168a..41aae4a02 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -80,8 +80,8 @@ function writeTranscript(fileName: string, lines: string[]): string { type ForwardedEvent = Record; -function parseAll(raw: string[]): ForwardedEvent[] { - return raw.map((r) => JSON.parse(r) as ForwardedEvent); +function parseAll(raw: ForwardedEvent[]): ForwardedEvent[] { + return raw; } /** diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 4fa131100..9495cb26b 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -204,7 +204,7 @@ export async function collectMainTranscriptEvents( sessionId: string, transcriptPath: string, trigger: MainTranscriptTrigger -): Promise { +): Promise[]> { try { // Both the load and the save happen inside the lock so a concurrent hook process for the // same session can never read a state this pass is about to overwrite. @@ -239,9 +239,9 @@ export async function collectMainTranscriptEvents( state.compactionCount += 1; } - const events: string[] = []; + const events: Record[] = []; for (const key of touchedKeys) { - events.push(JSON.stringify(buildUsageRequestEvent(sessionId, state.openRequests[key]))); + events.push(buildUsageRequestEvent(sessionId, state.openRequests[key])); } if (trigger === 'Stop' || trigger === 'SessionEnd') { @@ -262,7 +262,7 @@ export async function collectMainTranscriptEvents( // docstring defers it to this caller) — this is the full set of agent.usage.request // records derived for this session so far, main- and agent-scoped alike. summaryEvent.api_calls = Object.keys(state.openRequests).length; - events.push(JSON.stringify(summaryEvent)); + events.push(summaryEvent); } await saveParseState(sessionId, state); @@ -388,7 +388,7 @@ async function scanSubagentTranscript(filePath: string): Promise { +): Promise[]> { try { // Load/mutate/save inside the lock — same rationale as collectMainTranscriptEvents: a sibling // SubagentStop for another subagent in this same session must never read state this pass is @@ -410,9 +410,9 @@ export async function collectSubagentTranscriptEvents( } state.subagentOffsets[subagentFile.agentId] = nextOffset; - const events: string[] = []; + const events: Record[] = []; for (const key of touchedKeys) { - events.push(JSON.stringify(buildUsageRequestEvent(sessionId, state.openRequests[key]))); + events.push(buildUsageRequestEvent(sessionId, state.openRequests[key])); } // Cumulative usage for this agent — every scope_kind:'agent' record known for it so far, @@ -435,7 +435,7 @@ export async function collectSubagentTranscriptEvents( startedAt, durationMs ); - events.push(JSON.stringify(subagentEvent)); + events.push(subagentEvent); await saveParseState(sessionId, state); return events; diff --git a/src/agents/plugins/utils.ts b/src/agents/plugins/utils.ts index c38ab1972..d82266bde 100644 --- a/src/agents/plugins/utils.ts +++ b/src/agents/plugins/utils.ts @@ -11,7 +11,7 @@ import { logger } from '../../utils/logger.js'; * - Uses 1000ms timeout and swallows all errors * - Never throws, never affects hook's exit code */ -export async function forwardOtlpEventToSpool(rawEvent: string, agentName: string): Promise { +export async function forwardOtlpEventToSpool(event: Record, agentName: string): Promise { try { const state = await readState(); @@ -26,7 +26,7 @@ export async function forwardOtlpEventToSpool(rawEvent: string, agentName: strin const body: OtlpHookSpoolData = { agentName, timestamp: Date.now(), - raw: rawEvent + raw: JSON.stringify(event) } try { diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index bb8e120bf..a133018d7 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -11,23 +11,6 @@ beforeEach(() => { execSyncMock.mockReturnValue('0.15.6'); }); -// Only 'claude-code-otlp' (the real registered OTLP agent name) resolves to an agent; -// every other name — including 'claude', which every other test in this file deliberately -// uses — resolves to `undefined`, matching the real AgentRegistry's behavior. -vi.mock('@/agents/registry.js', () => ({ - AgentRegistry: { - getAnalyticsAgent: (agentName: string) => - agentName === 'claude-code-otlp' - ? { - prepareAnalyticsFields: async () => ({ - platform: 'claude-code', - client_version: '1.2.3', - }), - } - : undefined, - }, -})); - interface MappedRecord { type: string; session_id: string; @@ -307,7 +290,7 @@ describe('mapHookRecords', () => { expect(line1.event_id).not.toBe(line2.event_id); }); - it('merges prepareAnalyticsFields common fields onto the mapped record when the hook agent name is registered', async () => { + it('passes agent-baked common fields (platform/client_version) through onto the mapped record without any agent-specific lookup', async () => { const { mapHookRecords } = await import('../forwarder.js'); const ctx = { @@ -318,12 +301,18 @@ describe('mapHookRecords', () => { git: {}, }; - // The real registered OTLP agent name, unlike every other test in this file which - // deliberately uses the unregistered 'claude' (that name resolves to `undefined`, - // so this is the only test exercising the real agent-registry lookup merge path). + // Simulates what the plugin now bakes in hook-side before ever reaching the spool — + // the forwarder needs no agent-specific knowledge to pass these through, just the + // `...limited` spread like every other hook-native field. const record = JSON.stringify({ agentName: 'claude-code-otlp', - raw: JSON.stringify({ hook_event_name: 'Stop', session_id: 'sid1', cwd: '' }), + raw: JSON.stringify({ + hook_event_name: 'Stop', + session_id: 'sid1', + cwd: '', + platform: 'claude-code', + client_version: '1.2.3', + }), timestamp: Date.now(), }); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 1d1e00d48..2206f6c25 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -280,22 +280,6 @@ export async function mapHookRecords( ? resolveStoryForPrompt(ctx, rawPrompt) : { storyId: ctx.story?.storyId ?? '', storySource: ctx.story?.storySource ?? '' }; - const { AgentRegistry } = await import('@/agents/registry.js'); - const analyticsAgent = AgentRegistry.getAnalyticsAgent(spoolData.agentName); - let agentSpecificFields: Record = {}; - try { - agentSpecificFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; - } catch (err) { - // An agent plugin's analytics-field hook is documented as "must never throw", but - // that's only a doc comment — a violating implementation must not abort every - // remaining record in this forward tick. - const msg = err instanceof Error ? err.message : String(err); - logger.debug( - '[otlp-forwarder] prepareAnalyticsFields threw', - ...sanitizeLogArgs({ agentName: spoolData.agentName, err: msg }) - ); - } - const limited = limitHookPayload(hookEvent); const type = hookEventType(hookName, hookEvent); const sessionId = String(hookEvent['session_id'] ?? ''); @@ -316,7 +300,6 @@ export async function mapHookRecords( cwd, prompt_body: boundedText(hookEvent['prompt'], MAX_PROMPT_CHARS), raw: limited, - ...agentSpecificFields, schema_version: 2, event_id: resolveEventId(type, sessionId, { ...hookEvent, byteOffset }), codemie_cli_version: CODEMIE_CLI_VERSION, From 9fc0a34a3ec21433d23b252fd30ea846eb2c01a2 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Wed, 7 Oct 2026 14:12:50 +0300 Subject: [PATCH 19/35] feat(proxy): remove `claude_account` from identity source --- .../plan.md | 4 ++-- .../spec.md | 5 ++--- .../sso/proxy/plugins/otlp-spool/identity.ts | 21 +++---------------- 3 files changed, 7 insertions(+), 23 deletions(-) diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md index 1f196440a..277c306be 100644 --- a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md @@ -17,7 +17,7 @@ - `event_id` is a pure string function of fields already on the record — never a generated/stored UUID. - Only the *resolved* `story_id`/`story_source` is ever sent — never raw prompt text. Nothing in this sub-stage writes `.claude/analytics.local.json` (read-only here). - Ticket regex (shared constant): `/(?` — tries, in order: existing JWT-claims logic (moved from `resolveUserEmail`), `git config user.email` / `user.name` via `exec()`, the existing `codemie_cli` profile config loader, a best-effort `claude_account` lookup that returns nothing if unavailable (documented limitation, falls through), `os.userInfo().username`. First non-empty wins. +- Produces: `resolveIdentity(credentials: SSOCredentials | JWTCredentials, cwd: string): Promise<{ developerName: string; identitySource: 'jwt' | 'git' | 'codemie_cli' | 'os' | '' }>` — tries, in order: existing JWT-claims logic (moved from `resolveUserEmail`), `git config user.email` / `user.name` via `exec()`, the existing `codemie_cli` profile config loader, `os.userInfo().username`. First non-empty wins. **Test-first: yes — with JWT absent/empty, `resolveIdentity` falls through to git email when `git config user.email` succeeds, and to `os.userInfo().username` when every other tier is empty.** diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md index 38540bfa0..42cea3670 100644 --- a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md @@ -116,12 +116,11 @@ Reading is read-only here: **nothing in this stage writes that file.** A command ## Identity resolution -- Extend `resolveUserEmail()`'s JWT-only chain (`forwarder.ts:67-81`) to `jwt → git → codemie_cli → claude_account → os`, first available wins: +- Extend `resolveUserEmail()`'s JWT-only chain (`forwarder.ts:67-81`) to `jwt → git → codemie_cli → os`, first available wins: - `git` reads `git config user.name`/`user.email`. - `codemie_cli` reads the existing CLI profile config. - `os` reads `os.userInfo().username`. - - `claude_account` has no precedent in this repo — implement it as a best-effort lookup that yields nothing if unavailable, falling through to `os` (documented limitation, not a blocker). -- **Security review sign-off (2026-10-05):** this `jwt → git → codemie_cli → claude_account → os` derivation chain was flagged by code review as a CRITICAL "new attribution-identifier source" under `security-practices.md`'s Project & User Attribution Headers rule (CR-023), since it derives an identity-like value from local git config / CLI config / OS username with no verification. Reviewed and approved as implemented: the chain is used only to stamp `developer_name`/`identity_source` on outbound analytics/telemetry events (`identity.ts`), never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing that the cited rule's header table concerns. No code change required. +- **Security review sign-off (2026-10-05):** this `jwt → git → codemie_cli → os` derivation chain was flagged by code review as a CRITICAL "new attribution-identifier source" under `security-practices.md`'s Project & User Attribution Headers rule (CR-023), since it derives an identity-like value from local git config / CLI config / OS username with no verification. Reviewed and approved as implemented: the chain is used only to stamp `developer_name`/`identity_source` on outbound analytics/telemetry events (`identity.ts`), never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing that the cited rule's header table concerns. No code change required. ## Non-goals diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts index 6166f405d..b87f42271 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts @@ -2,7 +2,7 @@ import { userInfo } from 'node:os'; import type { SSOCredentials, JWTCredentials } from '@/providers/core/types.js'; import { isSSOCredentials, isJWTCredentials } from '@/providers/core/types.js'; -export type IdentitySource = 'jwt' | 'git' | 'codemie_cli' | 'claude_account' | 'os' | ''; +export type IdentitySource = 'jwt' | 'git' | 'codemie_cli' | 'os' | ''; export interface ResolvedIdentity { developerName: string; @@ -94,19 +94,7 @@ async function resolveCodemieCliIdentity(): Promise { } /** - * Tier 4 — claude_account: best-effort lookup of a locally authenticated - * Claude account identity. Documented limitation: no such lookup exists - * anywhere in this repo today (no stored Claude account email/id to read), - * so this tier always yields `''` in practice and the chain falls through to - * `os`. Kept as its own tier/function so a real lookup can be dropped in here - * later without touching the rest of the chain. - */ -async function resolveClaudeAccount(): Promise { - return ''; -} - -/** - * Tier 5 — os: the OS-reported username for the daemon process. Practically + * Tier 4 — os: the OS-reported username for the daemon process. Practically * never empty, but guarded anyway since some sandboxed environments can make * `os.userInfo()` throw. */ @@ -122,7 +110,7 @@ function resolveOsIdentity(): string { * Resolve a developer identity for analytics stamping, trying each tier in * order and returning the first non-empty result: * - * jwt -> git -> codemie_cli -> claude_account -> os + * jwt -> git -> codemie_cli -> os * * Never throws — every tier swallows its own failures internally. */ @@ -140,9 +128,6 @@ export async function resolveIdentity( const cliIdentity = await resolveCodemieCliIdentity(); if (cliIdentity) return { developerName: cliIdentity, identitySource: 'codemie_cli' }; - const claudeAccount = await resolveClaudeAccount(); - if (claudeAccount) return { developerName: claudeAccount, identitySource: 'claude_account' }; - const osIdentity = resolveOsIdentity(); if (osIdentity) return { developerName: osIdentity, identitySource: 'os' }; From 90223b38fc7ebbc078608c254f417cd229ef6c96 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Wed, 7 Oct 2026 14:29:56 +0300 Subject: [PATCH 20/35] refactor(proxy): remove resolveEventId --- .../transcript/usage-request.ts | 5 +- src/agents/plugins/utils.ts | 6 +- .../otlp-spool/__tests__/event-id.test.ts | 56 -------------- .../otlp-spool/__tests__/forwarder.test.ts | 77 ++++++------------- .../sso/proxy/plugins/otlp-spool/event-id.ts | 29 ------- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 14 +--- 6 files changed, 32 insertions(+), 155 deletions(-) delete mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts delete mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 56b6ed952..dbe0ca086 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -78,9 +78,8 @@ export function parseUsageLine( return null; } - // Both openRequests' key and resolveEventId's agent.usage.request formula key on - // `${requestId}::${model}` — an empty requestId would collide every such line in the session - // into one record instead of being skipped. + // openRequests keys on `${requestId}::${model}` — an empty requestId would collide every such + // line in the session into one record instead of being skipped. const requestId = parsed.message?.id ?? ''; if (!requestId) { return null; diff --git a/src/agents/plugins/utils.ts b/src/agents/plugins/utils.ts index d82266bde..9bc310e68 100644 --- a/src/agents/plugins/utils.ts +++ b/src/agents/plugins/utils.ts @@ -1,3 +1,4 @@ +import { randomUUID } from 'node:crypto'; import { OtlpHookSpoolData } from '@/providers/plugins/sso/proxy/plugins/otlp.plugin.js'; import { readState } from '../../cli/commands/proxy/daemon-manager.js'; import { logger } from '../../utils/logger.js'; @@ -26,7 +27,10 @@ export async function forwardOtlpEventToSpool(event: Record, ag const body: OtlpHookSpoolData = { agentName, timestamp: Date.now(), - raw: JSON.stringify(event) + raw: JSON.stringify({ + ...event, + event_id: randomUUID() + }) } try { diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts deleted file mode 100644 index 8e4387e3c..000000000 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts +++ /dev/null @@ -1,56 +0,0 @@ -import { describe, it, expect } from 'vitest'; -import { resolveEventId } from '../event-id.js'; - -describe('resolveEventId', () => { - describe('existing hook event types', () => { - it('returns the byte-offset formula for an existing-event type', () => { - expect(resolveEventId('agent.session.start', 'sid1', { byteOffset: 42 })).toBe( - 'sid1:agent.session.start:42' - ); - }); - - it('returns a different id for a different byte offset', () => { - const first = resolveEventId('agent.session.stop', 'sid1', { byteOffset: 0 }); - const second = resolveEventId('agent.session.stop', 'sid1', { byteOffset: 128 }); - expect(first).not.toBe(second); - expect(first).toBe('sid1:agent.session.stop:0'); - expect(second).toBe('sid1:agent.session.stop:128'); - }); - }); - - describe('agent.usage.request', () => { - it('returns the request/model-keyed formula', () => { - expect( - resolveEventId('agent.usage.request', 'sid1', { request_id: 'req1', model: 'gpt-4' }) - ).toBe('sid1:agent.usage.request:req1:gpt-4'); - }); - }); - - describe('agent.subagent.usage', () => { - it('returns the tool_use_id-keyed formula', () => { - expect(resolveEventId('agent.subagent.usage', 'sid1', { tool_use_id: 'tu1' })).toBe( - 'sid1:agent.subagent.usage:tu1' - ); - }); - - it('falls back to agent_id when tool_use_id is absent', () => { - expect( - resolveEventId('agent.subagent.usage', 'sid1', { tool_use_id: '', agent_id: 'agent-1' }) - ).toBe('sid1:agent.subagent.usage:agent-1'); - }); - - it('keeps two top-level subagents (no tool_use_id each) from colliding when agent_id differs', () => { - const first = resolveEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-1' }); - const second = resolveEventId('agent.subagent.usage', 'sid1', { agent_id: 'agent-2' }); - expect(first).not.toBe(second); - }); - }); - - describe('agent.session.summary', () => { - it('returns the phase-keyed formula', () => { - expect(resolveEventId('agent.session.summary', 'sid1', { phase: 'end' })).toBe( - 'sid1:agent.session.summary:end' - ); - }); - }); -}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index a133018d7..97362e61b 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -26,13 +26,19 @@ interface MappedRecord { client_version?: string; } -function buildHookRecord(hookEventName: string, sessionId: string, extra: Record = {}): string { +function buildHookRecord( + hookEventName: string, + sessionId: string, + extra: Record = {}, + eventId: string = 'default-event-id' +): string { return JSON.stringify({ agentName: 'claude', raw: JSON.stringify({ hook_event_name: hookEventName, session_id: sessionId, cwd: '', + event_id: eventId, ...extra, }), timestamp: Date.now(), @@ -77,10 +83,10 @@ describe('mapHookRecords', () => { git: {}, }; - const record1 = buildHookRecord('SessionStart', 'sid1'); - const record2 = buildHookRecord('Stop', 'sid1'); + const record1 = buildHookRecord('SessionStart', 'sid1', {}, 'event-id-1'); + const record2 = buildHookRecord('Stop', 'sid1', {}, 'event-id-2'); - const payload = await mapHookRecords([record1, record2], ctx, 0); + const payload = await mapHookRecords([record1, record2], ctx); const lines = payload.ndjson .trim() .split('\n') @@ -95,7 +101,7 @@ describe('mapHookRecords', () => { expect(lines[0].codemie_cli_version).toBe(lines[1].codemie_cli_version); }); - it('derives event_id from the running byte offset seeded by startOffset', async () => { + it('passes the event_id already stamped at spool-write time straight through unchanged', async () => { const { mapHookRecords } = await import('../forwarder.js'); const ctx = { @@ -106,16 +112,12 @@ describe('mapHookRecords', () => { git: {}, }; - const record = buildHookRecord('SessionStart', 'sid1'); + const record = buildHookRecord('SessionStart', 'sid1', {}, 'stamped-event-id-abc'); - const payloadAtZero = await mapHookRecords([record], ctx, 0); - const payloadAtOffset = await mapHookRecords([record], { ...ctx, git: {} }, 500); - - const lineAtZero = JSON.parse(payloadAtZero.ndjson.trim()) as MappedRecord; - const lineAtOffset = JSON.parse(payloadAtOffset.ndjson.trim()) as MappedRecord; + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; - expect(lineAtZero.event_id).toBe('sid1:agent.session.start:0'); - expect(lineAtOffset.event_id).toBe('sid1:agent.session.start:500'); + expect(line.event_id).toBe('stamped-event-id-abc'); }); it('prefers an explicit hookEvent.type over the HOOK_EVENT_TYPE_MAP lookup', async () => { @@ -134,7 +136,7 @@ describe('mapHookRecords', () => { // carry) must win. const record = buildHookRecord('PostToolUse', 'sid1', { type: 'agent.custom.synthetic' }); - const payload = await mapHookRecords([record], ctx, 0); + const payload = await mapHookRecords([record], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; expect(line.type).toBe('agent.custom.synthetic'); @@ -153,7 +155,7 @@ describe('mapHookRecords', () => { const record = buildHookRecord('PostToolUse', 'sid1'); - const payload = await mapHookRecords([record], ctx, 0); + const payload = await mapHookRecords([record], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; expect(line.type).toBe('agent.tool.end'); @@ -186,7 +188,7 @@ describe('mapHookRecords', () => { ' end-of-prompt-marker-that-must-not-appear-anywhere-in-the-output'; const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); - const payload = await mapHookRecords([record], ctx, 0); + const payload = await mapHookRecords([record], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; expect(line.story_id).toBe('EPMCDME-999'); @@ -222,7 +224,7 @@ describe('mapHookRecords', () => { cwd: '/repo/nonexistent-for-this-test', }); - const payload = await mapHookRecords([record], ctx, 0); + const payload = await mapHookRecords([record], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; expect(line.story_id).toBe('EPMCDME-15301'); @@ -247,7 +249,7 @@ describe('mapHookRecords', () => { const rawPrompt = 'can you look into ABC-42 when you get a chance'; const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); - const payload = await mapHookRecords([record], ctx, 0); + const payload = await mapHookRecords([record], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; expect(line.story_id).toBe('ABC-42'); @@ -255,41 +257,6 @@ describe('mapHookRecords', () => { } ); - it('keys agent.subagent.usage event_id off tool_use_id/agent_id from the synthetic record itself, not just byteOffset', async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - git: {}, - }; - - // Two top-level subagents, neither carrying a sidecar tool_use_id, but with - // distinct agent_id — the fallback must keep these from colliding. - const record1 = buildHookRecord('SubagentStop', 'sid1', { - type: 'agent.subagent.usage', - tool_use_id: '', - agent_id: 'agent-1', - }); - const record2 = buildHookRecord('SubagentStop', 'sid1', { - type: 'agent.subagent.usage', - tool_use_id: '', - agent_id: 'agent-2', - }); - - const payload = await mapHookRecords([record1, record2], ctx, 0); - const [line1, line2] = payload.ndjson - .trim() - .split('\n') - .map((line) => JSON.parse(line) as MappedRecord); - - expect(line1.event_id).toBe('sid1:agent.subagent.usage:agent-1'); - expect(line2.event_id).toBe('sid1:agent.subagent.usage:agent-2'); - expect(line1.event_id).not.toBe(line2.event_id); - }); - it('passes agent-baked common fields (platform/client_version) through onto the mapped record without any agent-specific lookup', async () => { const { mapHookRecords } = await import('../forwarder.js'); @@ -316,7 +283,7 @@ describe('mapHookRecords', () => { timestamp: Date.now(), }); - const payload = await mapHookRecords([record], ctx, 0); + const payload = await mapHookRecords([record], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; expect(line.platform).toBe('claude-code'); @@ -340,7 +307,7 @@ describe('mapHookRecords', () => { const record = buildHookRecord('Stop', 'sid1'); - const payload = await mapHookRecords([record], ctx, 0); + const payload = await mapHookRecords([record], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; expect(line.developer_name).toBe('git-user@example.com'); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts deleted file mode 100644 index 4babb8a61..000000000 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts +++ /dev/null @@ -1,29 +0,0 @@ -/** - * Pure, deterministic `event_id` derivation for analytics events. - */ -export function resolveEventId( - type: string, - sessionId: string, - fields: Record -): string { - switch (type) { - case 'agent.usage.request': { - const requestId = String(fields['request_id'] ?? ''); - const model = String(fields['model'] ?? ''); - return `${sessionId}:agent.usage.request:${requestId}:${model}`; - } - case 'agent.subagent.usage': { - const toolUseId = String(fields['tool_use_id'] ?? ''); - const agentId = String(fields['agent_id'] ?? ''); - return `${sessionId}:agent.subagent.usage:${toolUseId || agentId}`; - } - case 'agent.session.summary': { - const phase = String(fields['phase'] ?? ''); - return `${sessionId}:agent.session.summary:${phase}`; - } - default: { - const byteOffset = String(fields['byteOffset'] ?? ''); - return `${sessionId}:${type}:${byteOffset}`; - } - } -} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 2206f6c25..772a54be0 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -12,7 +12,6 @@ import { import { OtlpHookSpoolData } from '../otlp.plugin.js'; import { snapshotPendingBytes, snapshotPendingHookRecords } from './spool-io.js'; import { areCredentialsStale, markCredentialsStale } from './auth-state.js'; -import { resolveEventId } from './event-id.js'; import { decodeJwtClaims } from './identity.js'; import { type ForwardContext, @@ -229,20 +228,13 @@ interface HookPayload { export async function mapHookRecords( records: string[], - ctx: ForwardContext, - startOffset: number + ctx: ForwardContext ): Promise { const mapped: string[] = []; let containsSessionEnd = false; let malformed = 0; - let offset = startOffset; for (const record of records) { - // Every record occupied `byteLength(record) + 1` bytes in the spool file - // (the trailing newline was already stripped when the batch was read). - const byteOffset = offset; - offset += Buffer.byteLength(record, 'utf-8') + 1; - let spoolData: OtlpHookSpoolData; let hookEvent: Record; try { @@ -301,7 +293,7 @@ export async function mapHookRecords( prompt_body: boundedText(hookEvent['prompt'], MAX_PROMPT_CHARS), raw: limited, schema_version: 2, - event_id: resolveEventId(type, sessionId, { ...hookEvent, byteOffset }), + event_id: hookEvent['event_id'] as string, codemie_cli_version: CODEMIE_CLI_VERSION, }) ); @@ -345,7 +337,7 @@ async function forwardHooks( return 'idle'; } - const payload = await mapHookRecords(batch.records, ctx, batch.cursor); + const payload = await mapHookRecords(batch.records, ctx); if (payload.malformed > 0) { logger.debug( '[otlp-forwarder] skipped malformed hook records', From 5eff132f75983752ba3fde5e3915d053e45f06ac Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Wed, 7 Oct 2026 14:37:49 +0300 Subject: [PATCH 21/35] refactor(proxy): fix naming --- .../plugins/claude-code-otlp/claude-code-otlp.plugin.ts | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index 8b16a6a32..3ed3f72c9 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -21,8 +21,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { public readonly type = AgentAdapterType.OTLP; public async processOtlpEvent(rawEvent: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { - const parsed = JSON.parse(rawEvent) as Record; - const cwd = readString(parsed, 'cwd'); + const event = JSON.parse(rawEvent) as Record; // INVARIANT - do not weaken. An untracked project must produce NO hooks data // in the daemon spool, must not start the daemon, and must not run the SSO @@ -31,7 +30,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { // data, and skips sessions that only have OTEL data. Forwarding a hook event // for an untracked project would make the session sendable and leak its // data to the backend. - const isTracked = await isProjectTracked(cwd, await readAllowlistState()); + const isTracked = await isProjectTracked(readString(event, 'cwd'), await readAllowlistState()); if (!isTracked) { logger.debug('[Claude Code OTLP plugin] project not in analytics allowlist, ignoring hook event'); return; @@ -39,7 +38,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { await ensureOtlpProxy(this.name); - const evaluation = await this.evaluate(parsed); + const evaluation = await this.evaluate(event); if (evaluation.decision === 'block') { logger.error(`[Claude Code OTLP plugin] Blocking prompt: ${evaluation.reason}`); console.log(JSON.stringify(evaluation)); From 8c908bb6420d45a54d916693c9c12cfa388186aa Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Wed, 7 Oct 2026 14:48:03 +0300 Subject: [PATCH 22/35] refactor(proxy): fix SoC issues + styling --- .../plugins/otlp-spool/forward-context.ts | 16 ++++-- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 26 ++-------- .../sso/proxy/plugins/otlp-spool/identity.ts | 49 +++++++++++++------ .../plugins/otlp-spool/story-resolver.ts | 24 ++++++--- 4 files changed, 68 insertions(+), 47 deletions(-) diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts index 24d65c4dd..87eb4f7ed 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -88,7 +88,9 @@ export function resolveStoryForPrompt( * transcript-derived record's empty `cwd` can never poison the cache for the rest of the batch. */ export async function resolveStory(ctx: ForwardContext, cwd: string): Promise { - if (!cwd || ctx.story?.storyId !== undefined) return; + if (!cwd || ctx.story?.storyId !== undefined) { + return; + } const explicit = await resolveExplicitStory(cwd); const resolved = explicit ?? resolveBranchStory(ctx.git.branch ?? ''); @@ -106,9 +108,15 @@ export async function resolveStory(ctx: ForwardContext, cwd: string): Promise { - if (!cwd) return; - if (!ctx.identity) ctx.identity = {}; - if (ctx.identity.developerName !== undefined) return; + if (!cwd) { + return; + } + if (!ctx.identity) { + ctx.identity = {}; + } + if (ctx.identity.developerName !== undefined) { + return; + } const { developerName, identitySource } = await resolveIdentity(ctx.credentials, cwd); ctx.identity.developerName = developerName; ctx.identity.identitySource = identitySource; diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 772a54be0..931dc3965 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -12,7 +12,7 @@ import { import { OtlpHookSpoolData } from '../otlp.plugin.js'; import { snapshotPendingBytes, snapshotPendingHookRecords } from './spool-io.js'; import { areCredentialsStale, markCredentialsStale } from './auth-state.js'; -import { decodeJwtClaims } from './identity.js'; +import { resolveEmailFromCredentials } from './identity.js'; import { type ForwardContext, resolveCodemieCliVersion, @@ -52,26 +52,6 @@ type SendResult = 'ok' | 'failed' | 'auth-expired'; /* ------------------------------------------------------------------ auth --- */ -function resolveUserEmail(credentials: SSOCredentials | JWTCredentials): string { - if (isJWTCredentials(credentials)) { - const claims = decodeJwtClaims(credentials.token); - if (typeof claims['email'] === 'string' && claims['email']) { - return claims['email']; - } - } - if (isSSOCredentials(credentials)) { - const accessToken = credentials.cookies['codemie_access_token']; - if (accessToken) { - const claims = decodeJwtClaims(accessToken); - const email = claims['email'] ?? claims['preferred_username']; - if (typeof email === 'string' && email) { - return email; - } - } - } - return ''; -} - function buildAuthHeadersFromCreds( credentials: SSOCredentials | JWTCredentials ): Record | null { @@ -320,7 +300,7 @@ async function buildForwardContext( credentials, baseUrl: state?.targetUrl ?? state?.url ?? '', projectName: state?.project ?? '', - userEmail: resolveUserEmail(credentials), + userEmail: resolveEmailFromCredentials(credentials), git: {}, identity: {}, story: {}, @@ -411,7 +391,7 @@ export async function forwardSession( const hooksResult = await forwardHooks(sessionId, ctx); if (hooksResult === 'auth-expired') { - markCredentialsStale() + markCredentialsStale(); return; } diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts index b87f42271..1b0de6fdc 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts @@ -17,7 +17,9 @@ export interface ResolvedIdentity { */ export function decodeJwtClaims(token: string): Record { const parts = token.split('.'); - if (parts.length < 2) return {}; + if (parts.length < 2) { + return {}; + } try { return JSON.parse(Buffer.from(parts[1], 'base64url').toString('utf-8')) as Record< string, @@ -29,21 +31,28 @@ export function decodeJwtClaims(token: string): Record { } /** - * Tier 1 — jwt: pull `email` straight off the JWT credential's claims, or off - * the SSO session's `codemie_access_token` cookie claims (`email`, falling - * back to `preferred_username`). + * Pull `email` straight off a JWT credential's claims, or off the SSO + * session's `codemie_access_token` cookie claims (`email`, falling back to + * `preferred_username`). Shared by the tier-1 `jwt` identity tier below and + * by the forwarder's plain `userEmail` field — both resolve the same claim. */ -function resolveJwtIdentity(credentials: SSOCredentials | JWTCredentials): string { +export function resolveEmailFromCredentials( + credentials: SSOCredentials | JWTCredentials +): string { if (isJWTCredentials(credentials)) { const claims = decodeJwtClaims(credentials.token); - if (typeof claims['email'] === 'string' && claims['email']) return claims['email']; + if (typeof claims['email'] === 'string' && claims['email']) { + return claims['email']; + } } if (isSSOCredentials(credentials)) { const accessToken = credentials.cookies['codemie_access_token']; if (accessToken) { const claims = decodeJwtClaims(accessToken); const email = claims['email'] ?? claims['preferred_username']; - if (typeof email === 'string' && email) return email; + if (typeof email === 'string' && email) { + return email; + } } } return ''; @@ -56,7 +65,9 @@ function resolveJwtIdentity(credentials: SSOCredentials | JWTCredentials): strin * throws. */ async function resolveGitIdentity(cwd: string): Promise { - if (!cwd) return ''; + if (!cwd) { + return ''; + } try { const { exec } = await import('@/utils/exec.js'); const emailResult = await exec('git', ['config', 'user.email'], { cwd }); @@ -86,7 +97,9 @@ async function resolveCodemieCliIdentity(): Promise { try { const { ConfigLoader } = await import('@/utils/config.js'); const config = await ConfigLoader.loadMultiProviderConfig(); - if (config.userEmail) return config.userEmail; + if (config.userEmail) { + return config.userEmail; + } } catch { /* best-effort */ } @@ -119,17 +132,25 @@ export async function resolveIdentity( cwd: string ): Promise { try { - const jwtIdentity = resolveJwtIdentity(credentials); - if (jwtIdentity) return { developerName: jwtIdentity, identitySource: 'jwt' }; + const jwtIdentity = resolveEmailFromCredentials(credentials); + if (jwtIdentity) { + return { developerName: jwtIdentity, identitySource: 'jwt' }; + } const gitIdentity = await resolveGitIdentity(cwd); - if (gitIdentity) return { developerName: gitIdentity, identitySource: 'git' }; + if (gitIdentity) { + return { developerName: gitIdentity, identitySource: 'git' }; + } const cliIdentity = await resolveCodemieCliIdentity(); - if (cliIdentity) return { developerName: cliIdentity, identitySource: 'codemie_cli' }; + if (cliIdentity) { + return { developerName: cliIdentity, identitySource: 'codemie_cli' }; + } const osIdentity = resolveOsIdentity(); - if (osIdentity) return { developerName: osIdentity, identitySource: 'os' }; + if (osIdentity) { + return { developerName: osIdentity, identitySource: 'os' }; + } return { developerName: '', identitySource: '' }; } catch { diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts index 6f3b64ab2..13ddbd812 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts @@ -85,12 +85,16 @@ export async function resolveExplicitStory(cwd: string): Promise Date: Wed, 7 Oct 2026 15:08:23 +0300 Subject: [PATCH 23/35] refactor(proxy): update the docs --- .ai-run/guides/architecture/architecture.md | 1 + .../integration/external-integrations.md | 8 + AGENTS.md | 1 + docs/ARCHITECTURE-OTLP-PLUGIN.md | 151 +++++++----------- 4 files changed, 71 insertions(+), 90 deletions(-) diff --git a/.ai-run/guides/architecture/architecture.md b/.ai-run/guides/architecture/architecture.md index 49879f6a6..54aed84c1 100644 --- a/.ai-run/guides/architecture/architecture.md +++ b/.ai-run/guides/architecture/architecture.md @@ -267,3 +267,4 @@ FrameworkRegistry.get('langgraph') // src/frameworks/registry.ts | Core | `src/*/core/` | | Utils | `src/utils/` | | Tests | `tests/integration/`, `src/**/__tests__/` | +| OTLP ingestion adapters | `OtlpAgentAdapter` (`src/agents/core/types.ts`) is a second, non-chat plugin type registered the same way as `AgentAdapter` — one plugin per coding tool, ingesting that tool's native hook/telemetry events into the analytics pipeline. See `docs/ARCHITECTURE-OTLP-PLUGIN.md` for the dispatch pattern and how to add a new adapter (e.g. a future Cursor/other-tool adapter). | diff --git a/.ai-run/guides/integration/external-integrations.md b/.ai-run/guides/integration/external-integrations.md index 03c9319b8..04128067f 100644 --- a/.ai-run/guides/integration/external-integrations.md +++ b/.ai-run/guides/integration/external-integrations.md @@ -15,6 +15,7 @@ | OpenCode | Open-source AI assistant | SSO/API Key | Via CodeMie proxy | | MCP Servers | Remote MCP tool servers | OAuth 2.0 (auto) | `codemie-mcp-proxy` | | Enterprise SSO | Corporate auth | SAML/OAuth | `SSO_BASE_URL` | +| OTLP hook ingestion | Coding tool's own native hook/telemetry events → analytics pipeline | Via CodeMie proxy daemon | `codemie hook --agent ` | --- @@ -277,6 +278,12 @@ Claude Code injects `!bash` commands as synthetic `type:'user'` messages. The pr --- +## OTLP Hook-Event Ingestion (`OtlpAgentAdapter`) + +A separate, agent-agnostic integration from the above: each coding tool that exposes its own native hook/telemetry surface (Claude Code today; e.g. a future Cursor integration) gets one `OtlpAgentAdapter` plugin (`src/agents/plugins//`) that turns those events into CodeMie analytics via `codemie hook --agent ` → proxy daemon spool → analytics API. The dispatch contract, how to add a new native event to an adapter, and how to wire up a new tool are **not** duplicated here — see `docs/ARCHITECTURE-OTLP-PLUGIN.md`. `claude-code-otlp` (`src/agents/plugins/claude-code-otlp/`) is the current reference implementation. + +--- + ## skills.sh Wrapper (`codemie skills`) Catalog-agnostic thin wrapper around the upstream `skills` npm CLI. Discovery, ranking, and source classification are out of scope for this CLI. @@ -331,6 +338,7 @@ Validate provider config at startup; warn (not throw) on connectivity failures. - Provider plugins: `src/providers/plugins/` - Provider core types: `src/providers/core/types.ts` +- OTLP hook-event adapters: `docs/ARCHITECTURE-OTLP-PLUGIN.md` - OpenCode plugin: `src/agents/plugins/opencode/` - Codex plugin: `src/agents/plugins/codex/` - Claude plugin: `src/agents/plugins/claude/` diff --git a/AGENTS.md b/AGENTS.md index 1e74f599f..8536e7bad 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -130,6 +130,7 @@ Ask the user when: | `plugin`, `registry`, `agent`, `adapter` | architecture | external-integrations | | `claude`, `codex`, `gemini`, `opencode`, `pi`, `kimi`, `copilot`, `acp` | architecture | external-integrations | | `session`, `metrics`, `analytics`, `transcript`, `sync` | architecture | external-integrations | +| `otel`, `otlp`, `hook`, `telemetry` | architecture | external-integrations | | `architecture`, `layer`, `structure`, `pattern` | architecture | development-practices | | `test`, `vitest`, `mock`, `coverage` | testing-patterns | development-practices | | `error`, `exception`, `validation` | development-practices | security-practices | diff --git a/docs/ARCHITECTURE-OTLP-PLUGIN.md b/docs/ARCHITECTURE-OTLP-PLUGIN.md index 2b6ae5601..5a9119b04 100644 --- a/docs/ARCHITECTURE-OTLP-PLUGIN.md +++ b/docs/ARCHITECTURE-OTLP-PLUGIN.md @@ -1,22 +1,23 @@ # OTLP Hook-Event Plugin Pattern (`OtlpAgentAdapter`) -**Scope**: any plugin implementing `OtlpAgentAdapter` (`src/agents/core/types.ts`), under `src/agents/plugins//`. Today that's `claude-code-otlp` only; this doc is written so a second implementation (another coding agent/IDE with its own native hook/event surface — e.g. a future Cursor adapter) does not have to reinvent the dispatch shape. -**Status**: Living doc — update this file when the dispatch pattern changes, or when a second `OtlpAgentAdapter` implementation lands (promote the parts that turn out to generalize, keep the parts that don't agent-specific). +**Scope**: any plugin implementing `OtlpAgentAdapter` (`src/agents/core/types.ts`), under `src/agents/plugins//`. Today that's `claude-code-otlp` only. +**Status**: Living doc — update when the dispatch pattern changes or a second adapter lands. ## 1. What this pattern is -An `OtlpAgentAdapter` is not a chat agent — it never has a conversation. It is the ingestion point for one coding tool's own native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, per-request usage, subagent usage, auth gating, etc.). Each tool that exposes hooks (Claude Code today; potentially Cursor or others later) gets its own adapter plugin, but every adapter implements the same one-method contract and feeds the same spool shape: +An `OtlpAgentAdapter` is not a chat agent. It's the ingestion point for one coding tool's native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, usage, auth gating). Every adapter implements one method and feeds the same shared spool shape: ```ts export interface OtlpAgentAdapter { readonly name: string; readonly type: AgentAdapterType.OTLP; - processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise; } -``` -```ts +export interface OtlpAdapterDeps { + ensureOtlpProxy: (agentName: string) => Promise; +} + export interface OtlpHookSpoolData { agentName: string; raw: string; @@ -24,144 +25,114 @@ export interface OtlpHookSpoolData { } ``` -`processOtlpEvent` is the only entry point; everything downstream of it — the spool, the forwarder, the analytics API — is agent-agnostic and already shared. Nothing about the pipeline described below is specific to any one tool's hook names or payload shape; those live entirely inside each adapter's own `evaluate()`. +Everything downstream of `processOtlpEvent` — spool, forwarder, analytics API — is already agent-agnostic and shared. Nothing in it is specific to any one tool's hook names or payload shape; that lives entirely inside each adapter. -## 2. Where an adapter sits in the pipeline +## 2. Pipeline ``` -'s native hook/event fires (hook names & payload shape are agent-specific) - │ JSON on stdin +'s native hook/event fires (agent-specific names/payload) + │ JSON on stdin ▼ -codemie hook --agent (src/cli/commands/hook.ts) - │ +codemie hook --agent (src/cli/commands/hook.ts) + │ looks up adapter via AgentRegistry.getAnalyticsAgent(name) ▼ -.processOtlpEvent(rawEvent: string) +.processOtlpEvent(rawEvent, deps) │ ▼ -evaluate(rawEvent) → ForwardDecision ◄── §4: the shape every adapter should follow +evaluate(parsed) → ForwardDecision (§3: shape every adapter follows) │ ┌────┴─────┐ - │ │ - block forward - │ │ - log + forwardToSpool(payload) ──POST──► proxy daemon spool (OtlpHookSpoolData) + block forward + │ │ + log + forwardToSpool(payload) ──POST──► proxy daemon spool (OtlpHookSpoolData) suppress │ (adapter- ▼ - specific) otlp-spool/forwarder.ts (background, agent-agnostic) + specific) otlp-spool/forwarder.ts (background, agent-agnostic) │ ▼ CodeMie analytics API ``` -- `codemie hook --agent ` (`src/cli/commands/hook.ts`) looks up the adapter via `AgentRegistry.getAnalyticsAgent(name)` and calls `processOtlpEvent` with the raw stdin payload. This part is already agent-agnostic — a new adapter just registers under a new name. -- `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is fire-and-forget and shared by every adapter: it POSTs `{ agentName, timestamp, raw }` to the local proxy daemon and swallows every error. A dead/unreachable daemon never blocks or fails the hook, regardless of which adapter called it. -- `otlp-spool/forwarder.ts` is agent-agnostic by construction, not by convention — it reads spooled records and maps them straight through (`...limited` spread) to the analytics API payload. It never looks up or calls into any adapter. Any agent-owned common field (platform, client version, entrypoint, …) must already be baked into the event by the adapter's own `processOtlpEvent` before it reaches the spool. -- Wiring _which_ native hooks/events get pointed at `codemie hook --agent `, and how, is entirely tool-specific — see §5 for how `claude-code-otlp` does it; a different tool will have its own connector. +- `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is fire-and-forget and shared: POSTs `{ agentName, timestamp, raw }` to the local proxy daemon, swallows every error. A dead daemon never blocks or fails the hook. +- `otlp-spool/forwarder.ts` is agent-agnostic by construction: it maps spooled records straight through to the analytics API payload and never calls into any adapter. Any agent-owned common field (platform, version, entrypoint, …) must already be baked into the event by the adapter before it reaches the spool (§5). +- Wiring which native hooks/events call `codemie hook --agent ` is entirely tool-specific — see §6 for the Claude Code connector; a different tool has its own. -## 3. The dispatch pattern every `evaluate()` should follow +## 3. The dispatch pattern every `evaluate()` follows -This is the part that generalizes across adapters, independent of which tool's hooks are being handled. Keep to this shape regardless of the native event names involved. +This is a convention each adapter implements for itself, not a shared type from `core/types.ts` — `OtlpAgentAdapter`'s only contractual method is `processOtlpEvent`. Each adapter is free to define its own `ForwardDecision`-equivalent and field names to match its own tool's native event shape; keep the shape below, not the literal field/type names from the `claude-code-otlp` example. -### 3.1 `ForwardDecision` — the only two outcomes +### 3.1 Two outcomes — forward or block ```ts export type ForwardDecision = - | { decision: "forward"; payload: Record[] } - | { - decision: "block"; - reason: string; - hookSpecificOutput: Record; - }; + | { decision: 'forward'; payload: Record[] } + | { decision: 'block'; reason: string; /* ...whatever this tool needs to suppress/respond to the native event... */ }; ``` -Every native event an adapter processes should resolve to exactly one of these. `forward` carries the full list of parsed-object records to push to the spool (the original parsed event, plus zero or more derived analytics events). `block` stops the hook and logs a reason; what `hookSpecificOutput` means (e.g. suppressing a prompt) is specific to the tool and the event, not to this pattern. +`forward` carries the full list of records to push to the spool (the original parsed event, plus zero or more derived analytics events). `block` stops the hook and logs a reason; any extra fields on `block` (`claude-code-otlp` uses `hookSpecificOutput`, mirroring Claude Code's own hook-output JSON schema) are specific to that tool's blocking mechanism, not to this pattern — a different tool may have no `block` case at all, or a differently-shaped one. ### 3.2 One handler per event name — no `switch`, no shared merge step ```ts private async evaluate(parsed: Record): Promise { - const sessionId = readString(parsed, 'session_id'); - if (!sessionId) { - return { decision: 'forward', payload: [parsed] }; - } + // early-return / continuation-key checks here are tool-specific — only add + // one if this tool's events need it (claude-code-otlp keys continuation on + // its own 'session_id' field; a different tool may have no such field) - const hookEventName = readString(parsed, 'hook_event_name'); - if (hookEventName === 'SomeEvent') { + const nativeEventName = readString(parsed, /* this tool's own event-name field */ 'event_name'); + if (nativeEventName === 'SomeEvent') { return await this.onSomeEvent(parsed); } - if (hookEventName === 'OtherEvent') { - return await this.onOtherEvent(parsed); - } // ...one `if` per handled event name... return { decision: 'forward', payload: [parsed] }; } ``` -Each handler is fully responsible for its own `ForwardDecision` — it does not return a bare array for `evaluate()` to merge afterward. That means that the _only_ trailing fallthrough return (`{ decision: 'forward', payload: [parsed] }`) is for event names `evaluate()` doesn't branch on at all. It is not a sink that handled branches route through. +Each handler returns its own full `ForwardDecision`. The trailing fallthrough return is only for event names `evaluate()` doesn't branch on at all — it's not a sink that handled branches route through. ### 3.3 One parsed object, read directly — no typed projection -`rawEvent` is parsed exactly once, at the top of `processOtlpEvent()`, into `parsed: Record` — the tool's own native (snake_case) field names, unchanged. That same object is both what handlers read fields off of (via small helpers like `readString(parsed, 'session_id')`/`readOptionalString(parsed, 'agent_id')`) and what gets forwarded to the spool. There is deliberately no second, camelCase-renamed "typed hook event" object: a 1:1 field-rename mapper adds a name to type-check against but zero runtime validation (a missing field becomes `undefined`/`''` either way), so for a handful of fields read in a handful of places it is not worth carrying a second shape through `evaluate()` and every handler signature. If a future adapter's handlers need many more fields in many more places, revisit this call — a typed projection becomes worth its weight once the number of call sites justifies it. +`parsed` is read directly by handlers (via small helpers like `readString`/`readOptionalString`) and forwarded as-is; there's no typed/camelCase projection layer, since for a handful of fields that adds a type to check against with no runtime-validation benefit. ### 3.4 One place writes to the spool -`forwardToSpool()` (or equivalent) should be the only call site for `forwardOtlpEventToSpool()` in an adapter, called exactly once from `processOtlpEvent()` after `evaluate()` resolves. No handler forwards anything itself — handlers only _compute_ what should be forwarded (the `ForwardDecision.payload`). This single-chokepoint property is relied on by analytics correctness (every event reaches the spool exactly once, in a known order, under this adapter's registered name). +`forwardToSpool()` is the only call site for `forwardOtlpEventToSpool()`, called once from `processOtlpEvent()` after `evaluate()` resolves (it loops over `ForwardDecision.payload` and forwards each record). No handler forwards anything itself. This single-chokepoint property guarantees every event reaches the spool exactly once, in order, under the adapter's registered name. -### 3.5 Only gate on proxy/daemon readiness when you're about to act on it +### 3.5 Proxy readiness is the adapter's own responsibility -An adapter owns the decision of whether it actually needs the local proxy/daemon for a given event — don't pay for an auth check or a daemon-readiness check on every event just because _some_ events need one. `claude-code-otlp` is the current example: only `onUserPromptSubmit` calls `ensureProxyAuth()` (because that handler's whole job is to gate on it); every other handler goes straight to building its `ForwardDecision` and skips the check entirely, since forwarding to the spool doesn't itself require proxy readiness. Keep that shape in a new adapter — gate per-handler, not globally in `evaluate()` or `processOtlpEvent()`. +Whether and when to call `deps.ensureOtlpProxy()`, and whether to add any further gate (e.g. an auth check, a tracked-project check), is a decision each adapter makes for itself based on what its events actually need — there's no required shape here. `claude-code-otlp` gates `ensureOtlpProxy()` on its tracked-project check in `processOtlpEvent()` but, once past that, calls it for every event regardless of which handler runs (since every `forward` decision needs the daemon up to reach the spool) — it does not gate per-handler. It additionally gates an SSO auth check inside `onUserPromptSubmit` only, because that's the one handler whose job is to enforce it. A different tool may not need a tracked-project concept at all, may not need proxy readiness for every event, or may need a different gate entirely — don't carry `claude-code-otlp`'s specific gating choices into a new adapter, just the principle that each adapter decides this for itself. ## 4. Adding a new event to an existing adapter -1. **Decide the shape you need.** Does the new event need extra fields beyond the adapter's current typed hook-event shape? If so, extend that type and its raw→typed mapping function. -2. **Write one handler method**, taking `(rawEvent: string, event: )` and returning `Promise` (unless the handler genuinely needs no typed field off the event — then `rawEvent` alone is fine). -3. **Add exactly one `if` branch** in `evaluate()`, dispatching to the new handler. Do not add a `switch` case, do not touch the trailing fallthrough return. -4. **Never forward anything from inside the handler.** Return the full `ForwardDecision`; the single `forwardToSpool()` call in `processOtlpEvent()` is what actually sends it. -5. **Add a test** for the new branch, mocking at whatever boundary the adapter already mocks (e.g. a transcript/orchestrator module), following the adapter's existing test structure. -6. **Update that adapter's own event-surface table** (see `claude-code-otlp`'s own notes/tests for the current example of such a table). +1. Write one handler method, `(parsed: Record) => Promise`. +2. Add exactly one `if` branch in `evaluate()` dispatching to it. No `switch`, don't touch the trailing fallthrough. +3. Never forward from inside the handler — return the `ForwardDecision`; the single `forwardToSpool()` call in `processOtlpEvent()` sends it. +4. Add a test for the new branch, following the adapter's existing test structure. +5. Update that adapter's own event-surface table/notes (see §6 for the current example). ## 5. Adding a new `OtlpAgentAdapter` for a different tool -1. Create `src/agents/plugins//` and implement `OtlpAgentAdapter`: just `processOtlpEvent`, following the `evaluate()`/`ForwardDecision` shape in §3 — enriching its own events with any agent-owned common fields (see §5a) before forwarding to the spool. `claude-code-otlp` is the current example. -2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — that name is also what gets passed to `forwardOtlpEventToSpool(event, name)` and later matched by `otlp-spool/forwarder.ts` via `AgentRegistry.getAnalyticsAgent(name)` (still used by `hook.ts`'s dispatch, see §2). -3. Write a connector that wires the tool's own native hooks/events to `codemie hook --agent ` (see `src/cli/commands/proxy/connectors/claude-code-otlp.ts` for the Claude Code example — the specific hook names, settings file format, and env vars will be entirely different for another tool, and that's expected). -4. Everything from `forwardOtlpEventToSpool` onward (the spool POST, `otlp-spool/forwarder.ts`, the analytics API call) is already shared and agent-agnostic by construction — no changes needed there as long as step 1-3 hold. - -## 5a. Agent-owned fields that are expensive to resolve - -Resolve agent-owned fields (platform, version, entrypoint, …) **inside the adapter's own hook-time process**, not via a callback from the agent-agnostic forwarder — the forwarder (`otlp-spool/forwarder.ts`) runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, and reading agent/tool state (e.g. an env var) there reflects the daemon's own startup environment, not the per-invocation environment that actually produced the event. - -If resolving a field is expensive (a subprocess spawn, a network call), back it with a small file cache under `getCodemiePath()` rather than relying on in-process memoization — the hook-time process is fresh per event and does not persist across events, so an in-memory cache buys nothing. `claude-code-otlp/client-version-cache.ts` is the current example: a TTL file cache around `claude --version`. - -## 6. Session completeness gating & draining (`otlp-spool/`) - -Past the per-adapter spool POST, the proxy daemon's `otlp-spool/` layer batches a session's spooled data before sending it to the analytics API, rather than forwarding each spooled record immediately. This part is already agent-agnostic — it keys everything off `sessionId`/`agentName`, not off any one adapter's event shape — but it's useful background for understanding what "forward to the spool" actually leads to. - -- **`spool-state.ts`** derives a session's state straight from the spool files on disk. `hooksGroupPresent`/`otelGroupPresent` check whether the hooks stream and any OTEL stream have _ever_ had data written — sticky, so a stream that already reached EOF still counts as "written" even once fully delivered. -- **`completeness-gate.ts`** (`gateDecision(spool, status): GateDecision`) decides what to do with a session on each tick: - - hooks **and** OTEL both present → `'send'`. - - hooks only, waited long enough (and force-forwarding is allowed) → `'hooks-only-force'`; hooks only but not waited enough yet → `'wait'`. - - OTEL only → `'wait'` (today this waits indefinitely for a hooks event to arrive — there is no current path that gives up on an OTEL-only session). - - neither present → `'noop'`. -- **`sweep.ts`** garbage-collects a session once every stream is fully drained (`isSessionDrained`) and a grace period has elapsed since the last producer write — measured from spool-file mtimes, not the status file, since cursor-only updates touch the status file without any new data arriving. Safe to delete: the next producer write recreates the status/spool files from scratch. -- **`session-lock.ts`**'s `withSessionLock` serializes per-session operations but is **not reentrant**: calling it again for the same session from inside an already-running locked callback deadlocks (the inner call chains behind the outer one, which is itself waiting on the inner call to finish). Code running inside a lock must mutate session state in place rather than calling the cursor/status update helpers again. +1. Create `src/agents/plugins//`, implement `OtlpAgentAdapter` following §3. Enrich events with any agent-owned common fields (platform, version, entrypoint, …) **inside the adapter's own hook-time process**, not via a callback from the forwarder — the forwarder runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, so resolving agent/tool state there would reflect the daemon's environment, not the invocation that produced the event. If resolving a field is expensive (subprocess spawn, network call), back it with a small file cache under `getCodemiePath()` — the hook-time process is fresh per event, so in-memory memoization buys nothing (see `client-version-cache.ts` for the pattern). +2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — same name passed to `forwardOtlpEventToSpool(event, name)` and matched by `AgentRegistry.getAnalyticsAgent(name)` in `hook.ts`. +3. Write a connector wiring the tool's native hooks/events to `codemie hook --agent ` (see `src/cli/commands/proxy/connectors/claude-code-otlp.ts` for the example — hook names, settings format, and env vars are tool-specific). +4. Everything from `forwardOtlpEventToSpool` onward is already shared — no changes needed there as long as 1-3 hold. -## 7. Reference implementation: `claude-code-otlp` +## 6. Reference implementation: `claude-code-otlp` -`ClaudeCodeOtlpPlugin` (`src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts`) is the current (and so far only) `OtlpAgentAdapter`, registered as `claude-code-otlp`. It ingests Claude Code's own native hook events. +`ClaudeCodeOtlpPlugin` (`src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts`), registered as `claude-code-otlp`, ingests Claude Code's native hook events. -| File | Role | -| ------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- | -| `claude-code-otlp.plugin.ts` | `evaluate()` dispatch, all per-event handlers, hook-time common-field enrichment, allowlist gate + lazy proxy start | -| `client-version-cache.ts` | TTL file cache around `claude --version`, backing the `client_version` common field (see §5a) | -| `claude-code-otlp.types.ts` | `ForwardDecision` | -| `claude-code-otlp.constants.ts` | `CLAUDE_CODE_OTLP_AGENT_NAME` — the registered adapter name | -| `claude-code-otlp.allowlist.ts` | Per-project allowlist gating (`isProjectTracked`/`readAllowlistState`) — see the INVARIANT comment in `processOtlpEvent` | -| `transcript/orchestrator.ts` | `collectMainTranscriptEvents`, `collectSubagentTranscriptEvents` — Claude-Code-transcript-specific derivation of analytics events | -| `transcript/subagent-usage.ts` | `findSubagentFiles` — discovers every subagent transcript for a session | -| `src/cli/commands/proxy/connectors/claude-code-otlp.ts` | `HOOK_EVENTS` — wires Claude Code's `.claude/settings.json` hooks to `codemie hook --agent claude-code-otlp`; the tool-specific piece from §5 step 3 | +| File | Role | +|---|---| +| `claude-code-otlp.plugin.ts` | `evaluate()` dispatch, per-event handlers, hook-time common-field enrichment, allowlist gate + daemon start | +| `client-version-cache.ts` | TTL file cache around `claude --version`, backing the `client_version` common field | +| `claude-code-otlp.types.ts` | `ForwardDecision` | +| `claude-code-otlp.constants.ts` | `CLAUDE_CODE_OTLP_AGENT_NAME` — the registered adapter name | +| `claude-code-otlp.allowlist.ts` | Per-project allowlist gating (`isProjectTracked`/`readAllowlistState`) — see the INVARIANT comment in `processOtlpEvent`: an untracked project must never reach the daemon spool, since the daemon has no allowlist of its own | +| `transcript/orchestrator.ts` | `collectMainTranscriptEvents`/`collectSubagentTranscriptEvents` — transcript-derived analytics events | +| `transcript/subagent-usage.ts` | `findSubagentFiles` — discovers every subagent transcript for a session | +| `src/cli/commands/proxy/connectors/claude-code-otlp.ts` | Wires Claude Code's `.claude/settings.json` hooks to `codemie hook --agent claude-code-otlp` | -Claude Code fires 12 distinct hook events today (`HOOK_EVENTS` in that connector): `SessionStart`, `Stop`, `StopFailure`, `SessionEnd`, `UserPromptSubmit`, `PreToolUse`, `PostToolUse`, `PostToolUseFailure`, `SubagentStart`, `SubagentStop`, `PreCompact`, `Notification`. `evaluate()` branches explicitly on `UserPromptSubmit` (auth gate), `Stop`/`PreCompact`/`StopFailure`/`SessionEnd` (transcript parse), and `SubagentStop` (subagent-scoped transcript parse); everything else falls through to the raw passthrough — intentional, since most of those events carry nothing this pipeline needs to transform. +Claude Code fires 12 hook events (`HOOK_EVENTS` in that connector). `evaluate()` branches on `UserPromptSubmit` (auth gate), `Stop`/`PreCompact`/`StopFailure`/`SessionEnd` (transcript parse), and `SubagentStop` (subagent-scoped transcript parse); everything else falls through to the raw passthrough. -See `docs/ARCHITECTURE-PROXY.md` for the proxy/plugin layer this hands off to (the spool endpoint, the daemon, the forwarder). +See `docs/ARCHITECTURE-PROXY.md` for the proxy/daemon layer this hands off to, and `otlp-spool/` (`src/providers/plugins/sso/proxy/plugins/otlp-spool/`) for session completeness gating and draining — both already agent-agnostic and not something a new adapter needs to touch. From 784ac00af4ebb8ac99a49ee4c0e640c51ae80635 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Wed, 7 Oct 2026 16:55:40 +0300 Subject: [PATCH 24/35] docs(proxy): update docs related to sdlc-factory --- .../code-review-final.json | 132 +++++----------- .../plan.md | 149 +++++++++--------- .../spec.md | 41 ++--- 3 files changed, 132 insertions(+), 190 deletions(-) diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json index 2350323b9..1e5f24d43 100644 --- a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/code-review-final.json @@ -1,20 +1,20 @@ { "decision": "approve", - "rationale": "All four lenses and the standards audit ran cleanly against the approved spec; triage originally surfaced 23 blocking findings, including 6 decision_needed items (skill-scope attribution, lines_added/lines_removed sourcing, the subagent description field, commands_in_order ordering, started_at/ended_at sourcing, and the missing identity-derivation security review) and a security-practices CRITICAL violation (unreviewed jwt/git/os developer-identity chain). 0 findings deferred (nothing was judged pre-existing/out-of-scope). 6 blind/edge-case findings were dismissed below the blocking floor: silent version-read fallback matching existing codebase convention, an unbounded-but-local-only story-id config value, a stale per-tick-vs-per-session caching comment, a hookEventType passthrough required by the new synthetic-event design, a misleading-but-harmless doc comment on saveParseState's swallow behavior, and confirmation (via repo-wide grep) that no second OtlpAgentAdapter implementer exists to miss the new interface method. Post-review remediation (completed 2026-10-05, verified via typecheck/lint/tests): all 23 findings are now resolved. 17 fixed in code (CR-001, 002, 003, 004, 007, 009, 010, 011, 014, 015, 016, 017, 018, 019, 020, 021, 022). 5 decision_needed items had no viable code fix and were resolved by product decision, documented in spec.md's Open risks matching the existing title/workflow_run/worktree precedent (CR-005 lines_added/lines_removed, CR-006 skill scope_kind, CR-008 started_at/ended_at, CR-012 commands_in_order, CR-013 subagent description). CR-023 (the security-practices CRITICAL violation) was reviewed and explicitly approved as implemented — the derivation chain stamps analytics-only developer_name/identity_source, never the SSO proxy's billing/tenant-isolation attribution headers that rule concerns — with sign-off recorded in spec.md. No findings remain open.", + "rationale": "All four lenses and the standards audit ran cleanly against the approved spec; triage surfaced 19 blocking findings, including 6 decision_needed items (skill-scope attribution, lines_added/lines_removed sourcing, the subagent description field, commands_in_order ordering, started_at/ended_at sourcing, and the missing identity-derivation security review) and a security-practices CRITICAL violation (unreviewed jwt/git/os developer-identity chain) among them. 0 findings deferred (nothing was judged pre-existing/out-of-scope). 6 blind/edge-case findings were dismissed below the blocking floor: silent version-read fallback matching existing codebase convention, an unbounded-but-local-only story-id config value, a stale per-tick-vs-per-session caching comment, a hookEventType passthrough required by the synthetic-event design, a misleading-but-harmless doc comment on saveParseState's swallow behavior, and confirmation (via repo-wide grep) that no second OtlpAgentAdapter implementer exists to miss the interface. Post-review remediation (completed 2026-10-05, verified via typecheck/lint/tests): all 19 findings are now resolved. 13 fixed in code (CR-001, 002, 003, 004, 007, 009, 010, 011, 014, 015, 016, 017, 018). 5 decision_needed items had no viable code fix and were resolved by product decision, documented in spec.md's Open risks (CR-005 lines_added/lines_removed, CR-006 skill scope_kind, CR-008 started_at/ended_at, CR-012 commands_in_order, CR-013 subagent description). CR-016 (the security-practices CRITICAL violation) was reviewed and explicitly approved as implemented — the derivation chain stamps analytics-only developer_name/identity_source, never the SSO proxy's billing/tenant-isolation attribution headers that rule concerns — with sign-off recorded in spec.md. No findings remain open.", "confidence": "high", "risk_flags": [], "business_review": [ - { "kind": "spec", "item": "Common fields via prepareAnalyticsFields()/AgentRegistry.getAnalyticsAgent merged into every mapped record", "status": "pass", "notes": "Matches spec; minor deviation: Record return type and an extra undocumented plugin_version field." }, - { "kind": "spec", "item": "event_id: deterministic per-type composed id (hooks by byte offset; usage.request by request_id+model; subagent.usage by tool_use_id; session.summary by phase)", "status": "pass", "notes": "All four formulas implemented and tested as specified; see CR-015 for a remaining collision edge case. CR-018's subagent.usage collision (and a prerequisite forwarder.ts fields-not-passed-through gap it surfaced) is fixed." }, - { "kind": "spec", "item": "Incremental persisted parse state per session (mainOffset, subagentOffsets, openRequests, activeSkill, branchCounts)", "status": "pass", "notes": "Implemented; see CR-011 for a concurrency gap and CR-014 for a rotation/truncation gap in the surrounding mechanism." }, - { "kind": "spec", "item": "Each trigger updates tallies, derives records, writes state, THEN sends", "status": "pass", "notes": "CR-007 fixed: both runMainTranscriptParse/runSubagentTranscriptParse now save-then-send, matching spec text literally." }, - { "kind": "spec", "item": "Recovery: missing/corrupt state falls back to full parse from byte 0; re-derived records share event_id with originals", "status": "pass", "notes": "loadParseState falls back to createParseState(); idempotent-reparse test confirms matching ids across simulated crash." }, - { "kind": "spec", "item": "Transcript parsing stays async, swallows errors, never blocks/fails the hook, always exits 0", "status": "pass", "notes": "runMainTranscriptParse/runSubagentTranscriptParse wrap bodies in catch-swallow; fire-and-forget from evaluate(); hook.ts exits 0." }, + { "kind": "spec", "item": "Common fields via withCommonFields() merged onto every record hook-time", "status": "pass", "notes": "ClaudeCodeOtlpPlugin.withCommonFields() merges platform/entrypoint/client_version onto every record hook-time, inside processOtlpEvent, right before forwardToSpool(). Covered by claude-code-otlp.plugin.test.ts and client-version-cache.test.ts." }, + { "kind": "spec", "item": "event_id: randomUUID() stamped once per record at forward time", "status": "pass", "notes": "Stamped inside the shared forwardOtlpEventToSpool() chokepoint (src/agents/plugins/utils.ts); mapHookRecords() carries it straight through. Does not provide dedup across re-parses — see spec.md's Open risks." }, + { "kind": "spec", "item": "Incremental persisted parse state per session (mainOffset, subagentOffsets, openRequests, activeSkill, branchCounts, compactionCount)", "status": "pass", "notes": "Implemented; see CR-011 for a concurrency gap and CR-014 for a rotation/truncation gap in the surrounding mechanism." }, + { "kind": "spec", "item": "Each trigger updates tallies, derives records, writes state, THEN sends", "status": "pass", "notes": "collectMainTranscriptEvents/collectSubagentTranscriptEvents save state then return the built events; the plugin's single forwardToSpool() chokepoint does the actual sending after that save." }, + { "kind": "spec", "item": "Recovery: missing/corrupt state falls back to full parse from byte 0; re-derived records keep a stable natural key", "status": "pass", "notes": "loadParseState falls back to createParseState(). A record re-derived after a crash-before-save is assigned a new event_id (randomUUID() is generated fresh on every forward), but its natural key (request_id+model, tool_use_id, or session_id+phase) stays stable across re-derivation — see spec.md's Open risks." }, + { "kind": "spec", "item": "Transcript parsing stays async, swallows errors, never blocks/fails the hook, always exits 0", "status": "pass", "notes": "collectMainTranscriptEvents/collectSubagentTranscriptEvents wrap bodies in catch-swallow, returning [] on failure; the plugin's single forwardToSpool() call forwards whatever they return; hook.ts exits 0." }, { "kind": "spec", "item": "Transcript field sourcing: request_id from message.id, stop_reason sibling of usage, usage flat + nested cache/server_tool_use groups", "status": "pass", "notes": "parseUsageLine() matches every documented field path; tested against a verified fixture." }, - { "kind": "spec", "item": "agent.usage.request fires on Stop/PreCompact/SessionEnd (main) and SubagentStop/SessionEnd-backstop (subagent)", "status": "pass", "notes": "Wired for all four. CR-003's StopFailure trigger gap is fixed — StopFailure now dispatches runMainTranscriptParse alongside Stop/PreCompact/SessionEnd." }, + { "kind": "spec", "item": "agent.usage.request fires on Stop/PreCompact/SessionEnd (main) and SubagentStop/SessionEnd-backstop (subagent)", "status": "pass", "notes": "Wired for all four. CR-003's StopFailure trigger gap is fixed — StopFailure now dispatches the main-transcript orchestrator alongside Stop/PreCompact/SessionEnd." }, { "kind": "spec", "item": "agent.usage.request scope_kind/scope_name vary main/skill/agent by context", "status": "pass", "notes": "CR-006 resolved by decision: 'skill' scope_kind is formally deferred (no reliable in-transcript signal exists) and now disclosed in spec.md's Open risks, matching the title/workflow_run/worktree precedent. activeSkill stays tracked-but-unused for a future task." }, { "kind": "spec", "item": "agent.usage.request full field list (request_id, timestamps, model fields, token fields, scope, agent_id, stop_reason, is_api_error, git_branch)", "status": "pass", "notes": "buildUsageRequestEvent() emits every listed field; round-tripped by tests." }, - { "kind": "spec", "item": "agent.usage.request: one per unique (request_id, model), max-merge of numeric fields across duplicates", "status": "pass", "notes": "openRequests keyed by request_id::model shared across main/subagent passes; mergeUsageRequest() takes per-field max. See CR-015 for a request_id='' collision edge case." }, + { "kind": "spec", "item": "agent.usage.request: one per unique (request_id, model), max-merge of numeric fields across duplicates", "status": "pass", "notes": "openRequests keyed by request_id::model shared across main/subagent passes; mergeUsageRequest() takes per-field max. parseUsageLine() returns null when message.id is absent, so an empty requestId can never collide onto the shared `::model` key." }, { "kind": "spec", "item": "agent.subagent.usage fires on SubagentStop and SessionEnd backstop, never Stop/PreCompact", "status": "pass", "notes": "Confirmed by orchestrator wiring and a 3-subagent backstop test." }, { "kind": "spec", "item": "agent.subagent.usage 'description' sourced from the subagent's .meta.json sidecar", "status": "pass", "notes": "CR-013 resolved by decision: no such sidecar field exists anywhere in the codebase (mirrors the real production schema, which has none); hardcoded to '' unconditionally, now disclosed in spec.md's Open risks matching the title/workflow_run/worktree precedent." }, { "kind": "spec", "item": "agent.subagent.usage 'spawn_depth' defaults appropriately for top-level subagents", "status": "pass", "notes": "file.spawnDepth ?? 0, tested." }, @@ -24,18 +24,18 @@ { "kind": "spec", "item": "agent.session.summary lines_added/lines_removed derived from Edit/Write tool payloads", "status": "pass", "notes": "CR-005 resolved by decision: always emits 0 (an Edit/Write tool_use's input carries the proposed edit, not a diff stat; no reliable count derivable without re-implementing diffing), now disclosed in spec.md's Open risks matching the title/workflow_run/worktree precedent." }, { "kind": "spec", "item": "agent.session.summary compaction_count counts this session's PreCompact triggers", "status": "pass", "notes": "CR-004 fixed: TranscriptParseState now persists compactionCount, incremented on every PreCompact trigger and surfaced on the next Stop/SessionEnd summary." }, { "kind": "spec", "item": "agent.session.summary commands_in_order lists slash-commands in invocation order", "status": "pass", "notes": "CR-012 resolved by decision: contains the right distinct names but not true chronological order (Object.keys() of an unordered count map) — no ordered data exists upstream to draw from — now disclosed in spec.md's Open risks." }, - { "kind": "spec", "item": "agent.session.summary api_calls = count of this session's agent.usage.request records", "status": "pass", "notes": "CR-009 fixed: runMainTranscriptParse now merges api_calls = Object.keys(state.openRequests).length onto the summary event before forwarding." }, + { "kind": "spec", "item": "agent.session.summary api_calls = count of this session's agent.usage.request records", "status": "pass", "notes": "CR-009 fixed: collectMainTranscriptEvents now merges api_calls = Object.keys(state.openRequests).length onto the summary event before forwarding." }, { "kind": "spec", "item": "agent.session.summary started_at/ended_at sourced from SessionStart/SessionEnd hook timestamps", "status": "pass", "notes": "CR-008 resolved by decision: the transcript-line/parse-time approximation is accepted as-is rather than threading actual hook timestamps through every call site; now disclosed in spec.md's Open risks." }, { "kind": "spec", "item": "agent.session.summary title field", "status": "pass", "notes": "Emits '' literal, matching spec.md's own disclosed Open-risk that no source exists." }, - { "kind": "spec", "item": "Story resolution: explicit->branch per-tick cache for non-prompt events; explicit->marker->branch->mention for prompt events against the record's own untruncated prompt", "status": "pass", "notes": "resolveStoryOnce()/resolvePromptStory() implement the priority chain; tested. See CR-019 for a cache-poisoning edge case in the per-tick cache." }, + { "kind": "spec", "item": "Story resolution: explicit->branch per-tick cache for non-prompt events; explicit->marker->branch->mention for prompt events against the record's own untruncated prompt", "status": "pass", "notes": "resolveStoryOnce()/resolvePromptStory() implement the priority chain; tested. See CR-016 for a cache-poisoning edge case in the per-tick cache." }, { "kind": "spec", "item": "Explicit story source: SDLC_ANALYTICS_STORY_ID env var then .claude/analytics.local.json, read-only, gitignored", "status": "pass", "notes": "resolveExplicitStory(); file is gitignored; nothing writes it." }, - { "kind": "spec", "item": "Identity resolution: extend resolveUserEmail's jwt-only chain to jwt->git->codemie_cli->claude_account->os; user_email unchanged", "status": "pass", "notes": "All five tiers implemented and tested; original user_email/resolveUserEmail left untouched. CR-023's required security-review sign-off is now recorded (approved as implemented) in spec.md's Identity resolution section." }, + { "kind": "spec", "item": "Identity resolution: extend resolveUserEmail's jwt-only chain to jwt->git->codemie_cli->os; user_email unchanged", "status": "pass", "notes": "All four tiers implemented and tested; original user_email/resolveUserEmail left untouched. The required security-review sign-off is recorded (approved as implemented) in spec.md's Identity resolution section." }, { "kind": "spec", "item": "Non-goals respected: no change to existing event content beyond common fields, no server changes, three named events stay out of scope", "status": "pass", "notes": "All changes confined to the documented modules; none of the three out-of-scope event types appear in the diff." } ], "standards_review": [ { "kind": "commit-format", "status": "na", "notes": "git log over the review range is empty — all changes are uncommitted working-tree edits on top of diff_base, per explicit user instruction." }, - { "kind": "code-quality", "status": "pass", "notes": "CR-021 fixed: identity/story resolution glue and the CLI-version loader extracted into forward-context.ts, bringing forwarder.ts's feature code back under the 500-line cap (a separate, intentional, out-of-scope local debug-logging addition is layered on top but was confirmed with the user as not in scope for this review). CR-022 fixed: identity.ts/identity.test.ts now use the '@/' alias." }, - { "kind": "security", "status": "pass", "notes": "CR-023 resolved: the jwt->git->codemie_cli->os developer-identity derivation was reviewed and explicitly approved as implemented, since it stamps analytics-only developer_name/identity_source and never the SSO proxy's billing/tenant-isolation attribution headers security-practices.md's CRITICAL rule concerns. Sign-off recorded in spec.md." } + { "kind": "code-quality", "status": "pass", "notes": "CR-017 fixed: identity/story resolution glue and the CLI-version loader extracted into forward-context.ts, bringing forwarder.ts's feature code back under the 500-line cap (a separate, intentional, out-of-scope local debug-logging addition is layered on top but was confirmed with the user as not in scope for this review). CR-018 fixed: identity.ts/identity.test.ts now use the '@/' alias." }, + { "kind": "security", "status": "pass", "notes": "CR-019 resolved: the jwt->git->codemie_cli->os developer-identity derivation was reviewed and explicitly approved as implemented, since it stamps analytics-only developer_name/identity_source and never the SSO proxy's billing/tenant-isolation attribution headers security-practices.md's CRITICAL rule concerns. Sign-off recorded in spec.md." } ], "findings": [ { @@ -45,11 +45,11 @@ "triage": "patch", "file": "src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts", "title": "Plugin's hook-dispatch glue is untested", - "problem": "evaluate()'s field extraction and dispatch (agent_transcript_path/agent_id/tool_use_id/agent_type parsing, hookEventName branching into runMainTranscriptParse/runSubagentTranscriptParse) is only covered indirectly — only prepareAnalyticsFields is tested in this file, and orchestrator tests call the orchestrator functions directly, bypassing this glue entirely.", + "problem": "evaluate()'s field extraction and dispatch (agent_transcript_path/agent_id/tool_use_id/agent_type parsing, hookEventName branching into collectMainTranscriptEvents/collectSubagentTranscriptEvents) is only covered indirectly — this test file only covers withCommonFields()'s own merge logic, and orchestrator tests call the orchestrator functions directly, bypassing this glue entirely.", "impact": "A typo or regression in this wiring (wrong field name, wrong hookEventName comparison) would silently stop all transcript-derived analytics from firing in production while every existing test keeps passing.", - "recommendation": "Add a test that feeds a raw Stop/PreCompact/SessionEnd/SubagentStop hook payload through processOtlpEvent/evaluate and asserts runMainTranscriptParse/runSubagentTranscriptParse were invoked with the correctly-extracted arguments.", + "recommendation": "Add a test that feeds a raw Stop/PreCompact/SessionEnd/SubagentStop hook payload through processOtlpEvent/evaluate and asserts collectMainTranscriptEvents/collectSubagentTranscriptEvents were invoked with the correctly-extracted arguments.", "outcome": "fixed", - "resolution": "Added dispatch-glue coverage to claude-code-otlp.plugin.test.ts: a parameterized test driving Stop/PreCompact/SessionEnd/StopFailure through processOtlpEvent and asserting runMainTranscriptParse's exact (sessionId, transcriptPath, trigger) args and that forwardOtlpEventToSpool still fires; a test asserting non-matching hook names never dispatch; SubagentStop tests covering explicit agent_id/tool_use_id/agent_type extraction, the filename-derived agent_id fallback, and the missing-agent_transcript_path skip; a SessionEnd test asserting findSubagentFiles is called and runSubagentTranscriptParse fires once per discovered file; and the CR-002 empty-session_id short-circuit. All via vi.mock of ../transcript/orchestrator.js, ../transcript/subagent-usage.js, and ../../utils.js (dynamic-import mocking per testing-patterns.md). 15/15 tests pass." + "resolution": "Added dispatch-glue coverage to claude-code-otlp.plugin.test.ts: a parameterized test driving Stop/PreCompact/SessionEnd/StopFailure through processOtlpEvent and asserting collectMainTranscriptEvents's exact (sessionId, transcriptPath, trigger) args and that forwardOtlpEventToSpool still fires; a test asserting non-matching hook names never dispatch; SubagentStop tests covering explicit agent_id/tool_use_id/agent_type extraction, the filename-derived agent_id fallback, and the missing-agent_transcript_path skip; a SessionEnd test asserting findSubagentFiles is called and collectSubagentTranscriptEvents fires once per discovered file; and the CR-002 empty-session_id short-circuit. All via vi.mock of ../transcript/orchestrator.js, ../transcript/subagent-usage.js, and ../../utils.js (dynamic-import mocking per testing-patterns.md). 15/15 tests pass." }, { "id": "CR-002", @@ -63,7 +63,7 @@ "impact": "A hook event with an empty/missing session_id would read and write the shared state file keyed by an empty string, cross-contaminating offsets/openRequests/branchCounts across any other such session.", "recommendation": "Add `if (!event.sessionId) return { decision: 'forward', payload: rawEvent };` at the top of evaluate(), before any transcript-parse dispatch.", "outcome": "fixed", - "resolution": "Added the exact guard recommended, right after `toBaseClaudeCodeHookEvent()` and before the UserPromptSubmit/transcript-parse branches in evaluate(). Covered by a new test asserting an empty session_id skips runMainTranscriptParse entirely while still forwarding the raw event. `npm run typecheck`/eslint clean; test passes." + "resolution": "Added the exact guard recommended, right after `toBaseClaudeCodeHookEvent()` and before the UserPromptSubmit/transcript-parse branches in evaluate(). Covered by a new test asserting an empty session_id skips collectMainTranscriptEvents entirely while still forwarding the raw event. `npm run typecheck`/eslint clean; test passes." }, { "id": "CR-003", @@ -73,11 +73,11 @@ "file": "src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts", "line": 47, "title": "StopFailure never triggers transcript parsing", - "problem": "evaluate() only dispatches runMainTranscriptParse on hookEventName === 'Stop' | 'PreCompact' | 'SessionEnd', even though HOOK_EVENT_TYPE_MAP already maps StopFailure to agent.turn.error as a distinct, modeled event type.", + "problem": "evaluate() only dispatches collectMainTranscriptEvents on hookEventName === 'Stop' | 'PreCompact' | 'SessionEnd', even though HOOK_EVENT_TYPE_MAP already maps StopFailure to agent.turn.error as a distinct, modeled event type.", "impact": "When a turn ends via StopFailure, that turn's agent.usage.request/agent.session.summary data is not forwarded until a later Stop/SessionEnd eventually catches up (or never, if the session terminates without one).", - "recommendation": "Include 'StopFailure' alongside Stop/PreCompact/SessionEnd in the trigger condition for runMainTranscriptParse.", + "recommendation": "Include 'StopFailure' alongside Stop/PreCompact/SessionEnd in the trigger condition for collectMainTranscriptEvents.", "outcome": "fixed", - "resolution": "Added 'StopFailure' to evaluate()'s trigger condition and widened orchestrator.ts's MainTranscriptTrigger union (and the cast at the call site) to 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'. Left runMainTranscriptParse's existing `trigger === 'Stop' || trigger === 'SessionEnd'` summary-forwarding check untouched, since spec.md scopes agent.session.summary to Stop/SessionEnd only — StopFailure now forwards agent.usage.request records (same as PreCompact) but still never forwards a session.summary. Covered by the same parameterized dispatch test added for CR-001. `npm run typecheck`/eslint clean." + "resolution": "Added 'StopFailure' to evaluate()'s trigger condition and widened orchestrator.ts's MainTranscriptTrigger union (and the cast at the call site) to 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'. Left collectMainTranscriptEvents's existing `trigger === 'Stop' || trigger === 'SessionEnd'` summary-forwarding check untouched, since spec.md scopes agent.session.summary to Stop/SessionEnd only — StopFailure now forwards agent.usage.request records (same as PreCompact) but still never forwards a session.summary. Covered by the same parameterized dispatch test added for CR-001. `npm run typecheck`/eslint clean." }, { "id": "CR-004", @@ -87,11 +87,11 @@ "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", "line": 88, "title": "compaction_count always emits 0", - "problem": "acc.compactionCount is initialized to 0 in emptyAccumulator() and never incremented anywhere, despite runMainTranscriptParse already knowing the trigger ('PreCompact' or otherwise) on every call.", + "problem": "acc.compactionCount is initialized to 0 in emptyAccumulator() and never incremented anywhere, despite collectMainTranscriptEvents already knowing the trigger ('PreCompact' or otherwise) on every call.", "impact": "agent.session.summary's compaction_count field, a required spec field, is always 0 regardless of actual PreCompact activity in the session.", "recommendation": "Increment a persisted compactionCount counter in TranscriptParseState when trigger === 'PreCompact', and surface it in buildSessionSummaryEvent's output.", "outcome": "fixed", - "resolution": "Added a persisted `compactionCount` field to TranscriptParseState (parse-state.ts), incremented it in runMainTranscriptParse when trigger === 'PreCompact', and set acc.compactionCount from it before calling buildSessionSummaryEvent on Stop/SessionEnd. Covered by a new orchestrator.test.ts case asserting the cumulative count (two PreCompacts then a Stop) surfaces as 2 on the summary event. `npm run typecheck`/eslint clean." + "resolution": "Added a persisted `compactionCount` field to TranscriptParseState (parse-state.ts), incremented it in collectMainTranscriptEvents when trigger === 'PreCompact', and set acc.compactionCount from it before calling buildSessionSummaryEvent on Stop/SessionEnd. Covered by a new orchestrator.test.ts case asserting the cumulative count (two PreCompacts then a Stop) surfaces as 2 on the summary event. `npm run typecheck`/eslint clean." }, { "id": "CR-005", @@ -115,7 +115,7 @@ "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", "line": 194, "title": "Skill scope_kind is never produced", - "problem": "runMainTranscriptParse's own 'Note A' comment documents a judgment call: every main-transcript usage record is unconditionally scoped 'main', since no reliable signal for skill-context was found. state.activeSkill is tracked but never read or written.", + "problem": "collectMainTranscriptEvents's own 'Note A' comment documents a judgment call: every main-transcript usage record is unconditionally scoped 'main', since no reliable signal for skill-context was found. state.activeSkill is tracked but never read or written.", "impact": "One of the three documented scope_kind values ('skill') is never emitted by this implementation, so skill-scoped usage can never be distinguished from main-scoped usage in the analytics backend.", "recommendation": "Get a product decision on whether skill-context detection is required now (and, if so, identify a reliable in-transcript signal) or should be formally deferred to a later task.", "outcome": "fixed", @@ -129,11 +129,11 @@ "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", "line": 235, "title": "State is saved after events are sent, reversing the spec'd order", - "problem": "spec.md specifies 'updates tallies, derives records, writes state to disk, THEN sends'; runMainTranscriptParse/runSubagentTranscriptParse instead forward every touched usage/summary/subagent event and only call saveParseState() afterward.", - "impact": "Functionally mitigated today by the idempotent event_id design and a dedicated crash-before-save re-parse test, but the literal spec'd ordering guarantee is not met by the code as written.", + "problem": "spec.md specifies 'updates tallies, derives records, writes state to disk, THEN sends'; collectMainTranscriptEvents/collectSubagentTranscriptEvents instead forward every touched usage/summary/subagent event and only call saveParseState() afterward.", + "impact": "Functionally mitigated today by a dedicated crash-before-save re-parse test asserting the natural key stays stable across a re-derivation, but the literal spec'd ordering guarantee is not met by the code as written.", "recommendation": "Either reorder to save-then-send to match spec text, or update spec.md to describe the as-implemented idempotent-reconciliation approach so the two stay consistent.", "outcome": "fixed", - "resolution": "Reordered both runMainTranscriptParse and runSubagentTranscriptParse to save-then-send, matching spec text literally: the load-mutate-save sequence now runs to completion (and returns the events to forward) before any forwardOtlpEventToSpool call happens. This also set up CR-011's lock scope (load-mutate-save happens inside the lock; forwarding — network I/O — happens after release). `npm run typecheck`/eslint clean; existing crash-before-save idempotent-reparse test still passes unchanged." + "resolution": "Reordered both collectMainTranscriptEvents and collectSubagentTranscriptEvents to save-then-send, matching spec text literally: the load-mutate-save sequence now runs to completion (and returns the events to forward) before any forwardOtlpEventToSpool call happens. This also set up CR-011's lock scope (load-mutate-save happens inside the lock; forwarding — network I/O — happens after release). `npm run typecheck`/eslint clean; existing crash-before-save idempotent-reparse test still passes unchanged." }, { "id": "CR-008", @@ -145,9 +145,9 @@ "title": "started_at/ended_at are not sourced from hook timestamps", "problem": "spec.md specifies started_at/ended_at should come from the SessionStart/SessionEnd hook payload's own timestamp fields; the code instead sources started_at from the transcript's first parsed line and ended_at from new Date().toISOString() at parse time.", "impact": "Both values are close approximations of the intended timestamps but not sourced as specified; a gap between actual session start and first transcript line would skew started_at.", - "recommendation": "Get a decision on whether the current approximation is acceptable, or whether the actual hook timestamps need to be threaded through runMainTranscriptParse's call sites.", + "recommendation": "Get a decision on whether the current approximation is acceptable, or whether the actual hook timestamps need to be threaded through collectMainTranscriptEvents's call sites.", "outcome": "fixed", - "resolution": "Decision taken: accept the current approximation rather than widen every runMainTranscriptParse call site's signature to thread the actual hook payload timestamp through. Added to spec.md's Open risks. No code change." + "resolution": "Decision taken: accept the current approximation rather than widen every collectMainTranscriptEvents call site's signature to thread the actual hook payload timestamp through. Added to spec.md's Open risks. No code change." }, { "id": "CR-009", @@ -157,11 +157,11 @@ "file": "src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts", "line": 244, "title": "session.summary's api_calls field is never populated", - "problem": "session-summary.ts deliberately omits api_calls, documenting that the orchestrator is responsible for merging it in afterward from its own view of this session's agent.usage.request records; runMainTranscriptParse never performs that merge before forwarding the summary event.", + "problem": "session-summary.ts deliberately omits api_calls, documenting that the orchestrator is responsible for merging it in afterward from its own view of this session's agent.usage.request records; collectMainTranscriptEvents never performs that merge before forwarding the summary event.", "impact": "api_calls, a required spec field, is absent from every agent.session.summary event this implementation emits.", "recommendation": "Before forwarding summaryEvent, merge in api_calls: Object.keys(state.openRequests).length (or an equivalent count of this session's usage-request records).", "outcome": "fixed", - "resolution": "runMainTranscriptParse now sets summaryEvent.api_calls = Object.keys(state.openRequests).length before pushing it to the forward list, exactly as recommended. Covered by a new orchestrator.test.ts case asserting api_calls equals the number of forwarded agent.usage.request records. `npm run typecheck`/eslint clean." + "resolution": "collectMainTranscriptEvents now sets summaryEvent.api_calls = Object.keys(state.openRequests).length before pushing it to the forward list, exactly as recommended. Covered by a new orchestrator.test.ts case asserting api_calls equals the number of forwarded agent.usage.request records. `npm run typecheck`/eslint clean." }, { "id": "CR-010", @@ -189,7 +189,7 @@ "impact": "Two sibling subagents' SubagentStop hooks firing concurrently can race on the same session state file; the last writer wins and the other process's openRequests/subagentOffsets/branchCounts update is silently lost.", "recommendation": "Add a per-session file lock (e.g. proper-lockfile) around the load+save cycle, or implement an atomic read-modify-write with retry on conflict.", "outcome": "fixed", - "resolution": "Added a dependency-free exclusive-create lock file (withParseStateLock in parse-state.ts, no new npm package) serializing the load-mutate-save cycle per session; a lock older than 5s is treated as abandoned and stolen, and if it can't be acquired within a bounded wait fn still runs unlocked rather than hanging the hook. Both runMainTranscriptParse and runSubagentTranscriptParse now wrap their load+mutate+save sequence in it (forwarding happens after release). Covered by a new parse-state.test.ts case proving 10 concurrent increments all land (no lost update) plus a lock-cleanup case. `npm run typecheck`/eslint clean." + "resolution": "Added a dependency-free exclusive-create lock file (withParseStateLock in parse-state.ts, no new npm package) serializing the load-mutate-save cycle per session; a lock older than 5s is treated as abandoned and stolen, and if it can't be acquired within a bounded wait fn still runs unlocked rather than hanging the hook. Both collectMainTranscriptEvents and collectSubagentTranscriptEvents now wrap their load+mutate+save sequence in it (forwarding happens after release). Covered by a new parse-state.test.ts case proving 10 concurrent increments all land (no lost update) plus a lock-cleanup case. `npm run typecheck`/eslint clean." }, { "id": "CR-012", @@ -238,58 +238,16 @@ "kind": "code", "severity": "major", "triage": "patch", - "file": "src/agents/plugins/claude-code-otlp/transcript/usage-request.ts", - "line": 98, - "title": "usage.request event_id collides when message.id is absent", - "problem": "parseUsageLine defaults requestId to '' when message.id is absent; both openRequests' key (${requestId}::${model}) and computeEventId's agent.usage.request formula key on request_id+model, with no guard against an empty request_id.", - "impact": "Two distinct requests to the same model that both omit message.id collide into one openRequests entry/event_id; mergeUsageRequest's max-merge then blends unrelated usage into a single reported record, losing or misattributing part of the real usage.", - "recommendation": "Skip lines with an empty requestId (do not key openRequests/event_id on `::model` alone), rather than silently merging them.", - "outcome": "fixed", - "resolution": "parseUsageLine now returns null when message.id is absent/empty, before computing any other field, exactly as recommended — such lines are skipped rather than merged onto a shared `::model` key. Added a new usage-request.test.ts case asserting a usage-bearing line with no message.id parses to null. Verified against the existing fixture and orchestrator.test.ts helpers that every usage line used in tests already carries a non-empty messageId, so no existing test's expectations needed to change. `npm run typecheck`/eslint clean." - }, - { - "id": "CR-016", - "kind": "code", - "severity": "major", - "triage": "patch", - "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts", - "line": 16, - "title": "prepareAnalyticsFields merge path is never exercised by tests", - "problem": "Every mapHookRecords test in this file hardcodes agentName: 'claude', which is not a registered OTLP agent name (the registered name is 'claude-code-otlp'), so AgentRegistry.getAnalyticsAgent always resolves undefined and commonFields is always {} via the '?? {}' fallback.", - "impact": "A regression in the AgentRegistry lookup or in merging prepareAnalyticsFields's result (wrong key, dropped field, wrong agent-name string) would ship silently, since no existing test exercises the real merge path or asserts platform/client_version/plugin_version/agent_id/agent_type on the output.", - "recommendation": "Use the real registered agent name ('claude-code-otlp') in at least one test, and assert the common fields appear on the mapped record output.", - "outcome": "fixed", - "resolution": "Added a file-level vi.mock('@/agents/registry.js', ...) returning a real agent only for 'claude-code-otlp' (undefined for every other name, matching production), and a new test using that real registered name, asserting platform/client_version from prepareAnalyticsFields land on the mapped record. Every other existing test in the file keeps using the unregistered 'claude' deliberately, now provably exercising the undefined/{} fallback path rather than coincidentally doing so. `npm run typecheck`/eslint clean." - }, - { - "id": "CR-017", - "kind": "code", - "severity": "major", - "triage": "patch", "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts", "title": "developer_name/identity_source are never asserted in forwarder tests", - "problem": "This change replaces the previous developer_name: ctx.userEmail with a new jwt->git->codemie_cli->claude_account->os identity-resolution chain wired through ctx.identity, but no test in this file asserts the resulting developer_name/identity_source fields on a mapped record.", + "problem": "This change replaces the previous developer_name: ctx.userEmail with a new jwt->git->codemie_cli->os identity-resolution chain wired through ctx.identity, but no test in this file asserts the resulting developer_name/identity_source fields on a mapped record.", "impact": "A regression that breaks resolveIdentityOnce's wiring into the output (e.g. a caching bug, or a silent revert to ctx.userEmail) would ship with every forwarded record carrying an empty/wrong developer_name/identity_source and no test would catch it.", "recommendation": "Add an assertion on developer_name/identity_source in mapHookRecords' test coverage, exercising at least one non-jwt tier.", "outcome": "fixed", "resolution": "Added a test that pre-seeds ctx.identity with a non-jwt ('git') tier result and asserts it flows through verbatim onto the mapped record's developer_name/identity_source — proving resolveIdentityOnce's wiring into the output independent of the identity-chain's own resolution logic (already covered separately by identity.test.ts). `npm run typecheck`/eslint clean." }, { - "id": "CR-018", - "kind": "code", - "severity": "major", - "triage": "patch", - "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/event-id.ts", - "line": 35, - "title": "subagent.usage event_id collides when tool_use_id is absent", - "problem": "computeEventId's agent.subagent.usage case keys solely on tool_use_id; buildSubagentUsageEvent emits tool_use_id: file.toolUseId ?? '', and findSubagentFiles leaves toolUseId undefined whenever the sidecar .meta.json omits it — true for top-level subagents per this module's own doc comment.", - "impact": "Any two subagents in the same session that both lack a sidecar tool_use_id collide on the identical event_id (${sessionId}:agent.subagent.usage:); the backend's at-least-once dedup keeps only one and silently drops the other's usage event.", - "recommendation": "Fall back to agent_id when tool_use_id is empty, e.g. `${sessionId}:agent.subagent.usage:${toolUseId || agentId}`.", - "outcome": "fixed", - "resolution": "computeEventId's agent.subagent.usage case now falls back to agent_id when tool_use_id is empty, exactly as recommended. Also fixed a prerequisite integration gap discovered while verifying this: forwarder.ts's mapHookRecords() called computeEventId(type, sessionId, { byteOffset }), passing only byteOffset and never the record's own tool_use_id/agent_id/request_id/model/phase fields — so the per-type keying formulas (including this fix) were never actually reachable in the forwarded path, and every agent.usage.request/agent.subagent.usage/agent.session.summary record for a given session+type collided regardless of tool_use_id. Changed that call to computeEventId(type, sessionId, { ...hookEvent, byteOffset }) so the real fields flow through. Added unit coverage in event-id.test.ts (direct fallback + non-collision) and an integration test in forwarder.test.ts (two tool_use_id-less subagents with distinct agent_id no longer collide end-to-end). `npm run typecheck`, eslint, and both test files (15 tests) pass." - }, - { - "id": "CR-019", + "id": "CR-016", "kind": "code", "severity": "major", "triage": "patch", @@ -300,24 +258,10 @@ "impact": "When the first record in a forward tick has an empty cwd (synthetic transcript-derived events never carry one) and a later record in the same batch has a real cwd, developer_name/identity_source and story_id/story_source get permanently cached as the less-accurate (or empty) result for every record in that batch.", "recommendation": "Add the same `if (!cwd) return;` early-return pattern resolveGitInfo uses before caching ctx.identity/ctx.story.", "outcome": "fixed", - "resolution": "Added `if (!cwd) return;` as the first line of both resolveIdentityOnce and resolveStoryOnce (now in the new forward-context.ts module, extracted as part of CR-021), mirroring resolveGitInfo's own guard exactly. Covered incidentally by the CR-017 test (an empty-cwd record no longer touches ctx.identity at all, leaving a pre-seeded value untouched). `npm run typecheck`/eslint clean." + "resolution": "Added `if (!cwd) return;` as the first line of both resolveIdentityOnce and resolveStoryOnce (now in the new forward-context.ts module, extracted as part of CR-017), mirroring resolveGitInfo's own guard exactly. Covered incidentally by the CR-015 test (an empty-cwd record no longer touches ctx.identity at all, leaving a pre-seeded value untouched). `npm run typecheck`/eslint clean." }, { - "id": "CR-020", - "kind": "code", - "severity": "major", - "triage": "patch", - "file": "src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts", - "line": 387, - "title": "prepareAnalyticsFields call has no try/catch despite a documented 'must never throw' contract", - "problem": "mapHookRecords calls analyticsAgent?.prepareAnalyticsFields(hookEvent) with no surrounding try/catch; OtlpAgentAdapter.prepareAnalyticsFields's 'must never throw' contract is only a doc comment, not enforced at this one call site that crosses the plugin boundary.", - "impact": "The current (only) implementation is written defensively and never throws, but a future/alternate adapter implementation that violates the documented contract would throw out of the per-record loop and abort processing of every remaining record in that forward tick.", - "recommendation": "Wrap the call in try/catch, defaulting commonFields to {} on failure, so one misbehaving adapter can never abort the whole batch.", - "outcome": "fixed", - "resolution": "Wrapped the prepareAnalyticsFields call in try/catch, defaulting commonFields to {} and logging (sanitized) on failure, exactly as recommended — one misbehaving adapter implementation can no longer abort the per-record loop. `npm run typecheck`/eslint clean." - }, - { - "id": "CR-021", + "id": "CR-017", "kind": "code", "severity": "major", "triage": "patch", @@ -330,7 +274,7 @@ "resolution": "Extracted resolveIdentityOnce/resolveStoryOnce/resolvePromptStory/loadCodemieCliVersion (plus the ForwardContext type they share) into a new forward-context.ts module, exactly as recommended, dropping the feature code in forwarder.ts back under the 500-line cap. Note: forwarder.ts's line count as it sits locally also includes an unrelated, intentional local-only debug-logging addition (confirmed out of scope with the user) that this extraction does not touch or count against — the committed/reviewable feature code is what was brought under the cap. `npm run typecheck`/eslint clean." }, { - "id": "CR-022", + "id": "CR-018", "kind": "code", "severity": "major", "triage": "patch", @@ -344,7 +288,7 @@ "resolution": "Replaced '../../../../../core/types.js' with '@/providers/core/types.js' (not '@/agents/core/types.js' as originally recommended — the five-level path from identity.ts resolves to src/providers/core/types.ts, which is where SSOCredentials/JWTCredentials/isSSOCredentials/isJWTCredentials actually live; verified via `tsc --noEmit`) in both identity.ts and __tests__/identity.test.ts, which had the same deep relative import. `npm run typecheck`, targeted eslint, and identity.test.ts (3 tests) all pass." }, { - "id": "CR-023", + "id": "CR-019", "kind": "decision", "severity": "critical", "triage": "decision_needed", diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md index 277c306be..6b80cece7 100644 --- a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/plan.md @@ -4,7 +4,7 @@ **Goal:** Extend the `claude-code-otlp` analytics pipeline with common fields on every `agent.*` event, incremental persisted transcript parsing, three new events (`agent.usage.request`, `agent.subagent.usage`, `agent.session.summary`), story resolution, and an extended identity chain — per `spec.md`. -**Architecture:** Two layers already exist and are extended, not replaced. (1) Hook-side (`ClaudeCodeOtlpPlugin.processOtlpEvent`, runs once per Claude Code hook invocation, CLI process, must exit 0): gains a `prepareAnalyticsFields()` method for per-record agent-owned fields, and — on `Stop`/`SubagentStop`/`PreCompact`/`SessionEnd` — an orchestration step that incrementally parses the transcript(s) and forwards each derived event as its own synthetic hook-shaped record via the existing `forwardOtlpEventToSpool()`, tagged with an explicit `type` so it is **not** re-mapped by the hook-name table. (2) Daemon-side (`forwarder.ts`, long-running proxy tick): stamps `schema_version`, `event_id`, `codemie_cli_version`, `story_id`/`story_source`, `developer_name`/`identity_source` onto every record — old and new — in `mapHookRecords()`. +**Architecture:** Two layers already exist and are extended, not replaced. (1) Hook-side (`ClaudeCodeOtlpPlugin.processOtlpEvent`, runs once per Claude Code hook invocation, CLI process, must exit 0): `evaluate()` dispatches one handler per hook event name; the handlers for `Stop`/`PreCompact`/`StopFailure`/`SessionEnd`/`SubagentStop` call the transcript orchestrator (`transcript/orchestrator.ts`'s `collectMainTranscriptEvents`/`collectSubagentTranscriptEvents`), which *returns* zero or more derived, explicitly-`type`d synthetic records rather than forwarding them itself; the handler appends those to the original parsed event in its `ForwardDecision.payload`. `processOtlpEvent()` then merges a small set of agent-owned common fields onto every record in that payload (`withCommonFields()`) and is the **one** call site that forwards to the spool (`forwardToSpool()`, looping over the payload and calling the shared `forwardOtlpEventToSpool()` per record) — this is also where each record's `event_id` is stamped (`randomUUID()`, inside `forwardOtlpEventToSpool()`). (2) Daemon-side (`forwarder.ts`, long-running proxy tick): stamps `schema_version`, `codemie_cli_version`, `story_id`/`story_source`, `developer_name`/`identity_source` onto every record — old and new — in `mapHookRecords()`, carrying the already-stamped `event_id` straight through. **Tech Stack:** TypeScript, Node `node:fs/promises`/`node:crypto`/`node:child_process`, Vitest. No new runtime dependencies. @@ -14,7 +14,7 @@ - Truncation unchanged: prompt 200 chars (`MAX_PROMPT_CHARS`), tool input/output/error 300 chars (`MAX_TOOL_FIELD_CHARS`) — both already defined in `forwarder.ts:37-38`. - Hooks/orchestration stay `async`, swallow all exceptions internally, never throw past the top-level handler, never block Claude Code. - Node only, no new npm dependencies. -- `event_id` is a pure string function of fields already on the record — never a generated/stored UUID. +- `event_id` is a `randomUUID()` stamped once per record, hook-time, inside `forwardOtlpEventToSpool()` — the one chokepoint every record (original and derived alike) passes through on its way to the spool. - Only the *resolved* `story_id`/`story_source` is ever sent — never raw prompt text. Nothing in this sub-stage writes `.claude/analytics.local.json` (read-only here). - Ticket regex (shared constant): `/(?): string` — branches on `type`: existing 13 types use `${sessionId}:${type}:${byteOffset}` (`fields.byteOffset: number`); `agent.usage.request` uses `${sessionId}:agent.usage.request:${request_id}:${model}`; `agent.subagent.usage` uses `${sessionId}:agent.subagent.usage:${tool_use_id}`; `agent.session.summary` uses `${sessionId}:agent.session.summary:${phase}`. +- Produces: nothing new — `forwardOtlpEventToSpool(event, agentName): Promise` (pre-existing shared helper) now also stamps `event_id`. -**Test-first: yes — `computeEventId` returns the exact byte-offset formula for an existing-event type and the request/model-keyed formula for `agent.usage.request`; `mapHookRecords()` on two records yields two different `event_id`s and both carry `schema_version: 2`.** +**Test-first: yes — `mapHookRecords()` on two records yields two different `event_id`s (carried through from the input) and both carry `schema_version: 2`.** -- [ ] Write failing tests for `computeEventId`'s four branches and for `mapHookRecords` stamping `schema_version`/`event_id`/`codemie_cli_version`. -- [ ] Implement `event-id.ts` and the `forwarder.ts` edits. -- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/event-id.test.ts src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests for `mapHookRecords` stamping `schema_version`/`event_id`/`codemie_cli_version`. +- [x] Implement the `utils.ts`/`forwarder.ts` edits. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts` — PASS. +- [x] Commit. --- -### Task 2: Agent-owned common fields (`prepareAnalyticsFields`) +### Task 2: Agent-owned common fields (`withCommonFields`) **Files:** -- Modify: `src/agents/core/types.ts:734-739` — add `prepareAnalyticsFields(hookEvent: Record): Promise>` to `OtlpAgentAdapter` (return type kept generic so core/types.ts has no dependency on a leaf plugin's types). -- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts` — implement it: `platform: 'claude-code'`, `entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? ''`, `client_version` from `claude --version` (reuse the `exec()` helper from `src/utils/exec.ts`, same approach as `ClaudeAgentAdapter.getVersion()` at `src/agents/plugins/claude/claude.plugin.ts:633`; cache the resolved version on the instance so a high-frequency hook like `PostToolUse` doesn't spawn a subprocess per call), `plugin_version` from `src/agents/plugins/claude/plugin/.claude-plugin/plugin.json`'s `version` field, `agent_id`/`agent_type` read directly off `hookEvent['agent_id']`/`hookEvent['agent_type']` when present (subagent context only). -- Modify: `src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts` — in `mapHookRecords()`, resolve `AgentRegistry.getAnalyticsAgent(spoolData.agentName)` and merge `await analyticsAgent?.prepareAnalyticsFields(hookEvent) ?? {}` into the mapped record before the daemon-side fields. -- Test: `src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts` (new). +- Create: `src/agents/plugins/claude-code-otlp/client-version-cache.ts` — `resolveClientVersion()`: runs `claude --version` and caches the result in a TTL file cache (1h) under `getCodemiePath('cache', 'claude-code-client-version.json')`, since each hook fire is a fresh CLI process with no in-memory instance to cache on. +- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts` — add a private `withCommonFields(records)` that merges `{ platform: 'claude-code', entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', client_version: await resolveClientVersion() }` onto every record, called once from `processOtlpEvent()` on the full `ForwardDecision.payload` just before `forwardToSpool()`. +- Test: `src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts`, `__tests__/claude-code-otlp.plugin.test.ts`. **Interfaces:** -- Consumes: `AgentRegistry.getAnalyticsAgent(name): OtlpAgentAdapter | undefined` (`src/agents/registry.ts:76`). -- Produces: `ClaudeCodeOtlpPlugin.prepareAnalyticsFields()` — relied on by Task 11/12's orchestrator for `agent_id` on `agent.usage.request`/`agent.subagent.usage`. +- Produces: `resolveClientVersion(): Promise` (`client-version-cache.ts`); `ClaudeCodeOtlpPlugin.withCommonFields(records: Record[]): Promise[]>` (private). +- `agent_id`/`agent_type` are **not** produced here — they arrive verbatim on raw subagent-shaped hook payloads, or are sourced inside the Task 8/9 transcript builders. -**Test-first: yes — `prepareAnalyticsFields()` on a hook event carrying `agent_id`/`agent_type` returns both; on one without, it omits them; `client_version` is only spawned once across two calls.** +**Test-first: yes — `resolveClientVersion()` spawns `claude --version` once and serves the cached value on a second call within the TTL; `withCommonFields()` on two records stamps identical `platform`/`entrypoint`/`client_version` onto both.** -- [ ] Write failing tests (mock `exec()` to assert single invocation across two calls). -- [ ] Implement. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests (mock `exec()`/the cache file to assert single invocation across two calls within the TTL). +- [x] Implement. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/__tests__/client-version-cache.test.ts src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts` — PASS. +- [x] Commit. --- @@ -75,10 +76,10 @@ **Test-first: yes — with JWT absent/empty, `resolveIdentity` falls through to git email when `git config user.email` succeeds, and to `os.userInfo().username` when every other tier is empty.** -- [ ] Write failing tests covering: jwt hit, jwt-miss→git-hit, all-miss→os-fallback. -- [ ] Implement `identity.ts`, wire into `forwarder.ts`. -- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests covering: jwt hit, jwt-miss→git-hit, all-miss→os-fallback. +- [x] Implement `identity.ts`, wire into `forwarder.ts`. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts` — PASS. +- [x] Commit. --- @@ -95,10 +96,10 @@ **Test-first: yes — `resolveExplicitStory` prefers the env var over the file when both are set; `resolveBranchStory` extracts `EPMCDME-15301` from `feature/epmcdme-15301-foo` uppercased.** -- [ ] Write failing tests for both resolvers plus the regex's word-boundary behavior (no match inside `ABC-123X`). -- [ ] Implement `story-resolver.ts`, wire into `forwarder.ts`, add the `.gitignore` line. -- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests for both resolvers plus the regex's word-boundary behavior (no match inside `ABC-123X`). +- [x] Implement `story-resolver.ts`, wire into `forwarder.ts`, add the `.gitignore` line. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. +- [x] Commit. --- @@ -114,10 +115,10 @@ **Test-first: yes — a prompt containing `story: EPMCDME-999` resolves to `storySource: 'marker'` even when the branch carries a different ticket; a prompt with no marker but a bare `ABC-42` mention resolves to `storySource: 'mention'`; the raw prompt text itself is never present on the emitted record (only `prompt_body`, truncated, and `story_id`/`story_source`).** -- [ ] Write failing tests for marker precedence over mention, and for the no-match case falling back to the Task 4 branch/explicit result. -- [ ] Implement. -- [ ] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests for marker precedence over mention, and for the no-match case falling back to the Task 4 branch/explicit result. +- [x] Implement. +- [x] Run `npx vitest run src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts` — PASS. +- [x] Commit. --- @@ -145,19 +146,21 @@ export interface TranscriptParseState { openRequests: Record; // key: `${requestId}::${model}` activeSkill: string; branchCounts: Record; + compactionCount: number; // persisted count of this session's PreCompact triggers, feeds agent.session.summary's compaction_count } export function createParseState(): TranscriptParseState; export async function loadParseState(sessionId: string): Promise; // missing or corrupt file -> fresh state, never throws export async function saveParseState(sessionId: string, state: TranscriptParseState): Promise; +export async function withParseStateLock(sessionId: string, fn: () => Promise): Promise; // serializes concurrent load-mutate-save cycles for one session via an exclusive-create lock file ``` Stored at `getCodemiePath('analytics', 'state', `${sessionId}.json`)` (`src/utils/paths.ts:385`), directory created on write. **Test-first: yes — `loadParseState` on a missing file returns `createParseState()`'s fresh shape; on a corrupt JSON file it also recovers to fresh rather than throwing; `saveParseState` followed by `loadParseState` round-trips `openRequests` and `branchCounts` exactly.** -- [ ] Write failing tests for missing/corrupt/round-trip. -- [ ] Implement `parse-state.ts`. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests for missing/corrupt/round-trip. +- [x] Implement `parse-state.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts` — PASS. +- [x] Commit. --- @@ -172,10 +175,10 @@ Stored at `getCodemiePath('analytics', 'state', `${sessionId}.json`)` (`src/util **Test-first: yes — a file with two complete lines plus a trailing unterminated partial line returns only the two complete lines and `nextOffset` points exactly after the second line's newline; a second call starting from that offset returns only lines appended afterwards.** -- [ ] Write failing tests (fixture file written incrementally across two reads). -- [ ] Implement. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests (fixture file written incrementally across two reads). +- [x] Implement. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/transcript-reader.test.ts` — PASS. +- [x] Commit. --- @@ -194,10 +197,10 @@ Stored at `getCodemiePath('analytics', 'state', `${sessionId}.json`)` (`src/util **Test-first: yes — on a fixture with two JSONL lines for the same `message.id`+model where the second has a higher `output_tokens` and a `stop_reason` the first lacks, `mergeUsageRequest` of the two parsed records keeps the max `output_tokens` and the non-empty `stop_reason`; a line with no `usage` block parses to `null`.** -- [ ] Write failing tests against the fixture. -- [ ] Implement `usage-request.ts`. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests against the fixture. +- [x] Implement `usage-request.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts` — PASS. +- [x] Commit. --- @@ -214,10 +217,10 @@ Stored at `getCodemiePath('analytics', 'state', `${sessionId}.json`)` (`src/util **Test-first: yes — a session fixture with three subagent transcript files produces three `agent.subagent.usage` events whose summed token fields equal the sum of the `agent.usage.request` records this same fixture yields with `scope_kind: 'agent'` (the external data-model doc's §8 acceptance scenario).** -- [ ] Write the failing cross-check test plus a `spawn_depth`-missing-sidecar case. -- [ ] Implement `subagent-usage.ts`. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts` — PASS. -- [ ] Commit. +- [x] Write the failing cross-check test plus a `spawn_depth`-missing-sidecar case. +- [x] Implement `subagent-usage.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts` — PASS. +- [x] Commit. --- @@ -235,53 +238,57 @@ Stored at `getCodemiePath('analytics', 'state', `${sessionId}.json`)` (`src/util **Test-first: yes — a `branchCounts` map built from a mid-session branch switch (`{main: 3, feature: 7}`) resolves `branch_dominant: 'feature'` (the external data-model doc's §8 branch-switch scenario); `buildSessionSummaryEvent` with `phase: 'incremental'` omits `endedAt` and with `phase: 'final'` includes it.** -- [ ] Write failing tests for `branchDominant`, `primaryModel`, and both phases. -- [ ] Implement `session-summary.ts`. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts` — PASS. -- [ ] Commit. +- [x] Write failing tests for `branchDominant`, `primaryModel`, and both phases. +- [x] Implement `session-summary.ts`. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts` — PASS. +- [x] Commit. --- -### Task 11: Orchestrate main-transcript triggers (`Stop`, `PreCompact`, `SessionEnd`) +### Task 11: Orchestrate main-transcript triggers (`Stop`, `PreCompact`, `StopFailure`, `SessionEnd`) + +The orchestrator does not forward anything itself — it `return`s the derived events, and the plugin's own `evaluate()`/`forwardToSpool()` chokepoint is what actually sends them, so there is exactly one place in the pipeline that writes to the spool (per `docs/ARCHITECTURE-OTLP-PLUGIN.md` §3.4). Load-mutate-save runs inside a per-session lock (`withParseStateLock`, Task 6) so forwarding only ever sees fully-persisted state. **Files:** - Create: `src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts` -- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts:24-35` — in `evaluate()`, after the existing `UserPromptSubmit` branch, when `event.hookEventName` is `Stop`, `PreCompact`, or `SessionEnd`, call the orchestrator (fire-and-forget, matching `forwardToSpool`'s pattern) before returning the normal `forward` decision. +- Modify: `src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts` — `evaluate()` dispatches `Stop`/`PreCompact`/`StopFailure`/`SessionEnd` each to their own handler method (`onStopEvent`/`onPreCompactEvent`/`onStopFailureEvent`/`onSessionEndEvent`), which calls `collectMainTranscriptEvents()` and returns `{ decision: 'forward', payload: [parsed, ...derived] }` — the handler never forwards directly. - Test: `src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts`. **Interfaces:** -- Produces: `runMainTranscriptParse(sessionId: string, transcriptPath: string, trigger: 'Stop' | 'PreCompact' | 'SessionEnd'): Promise` — loads state (Task 6), reads new lines (Task 7), derives/merges `agent.usage.request` records (Task 8) keyed into `state.openRequests`, updates `state.branchCounts`/summary accumulator, emits one `forwardOtlpEventToSpool(JSON.stringify(event), CLAUDE_CODE_OTLP_AGENT_NAME)` call per completed `agent.usage.request` plus one `agent.session.summary` (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`; `PreCompact` emits only usage requests, never a summary — matches spec), saves state, and swallows every error internally (never throws into `processOtlpEvent`). +- Produces: `collectMainTranscriptEvents(sessionId: string, transcriptPath: string, trigger: 'Stop' | 'PreCompact' | 'SessionEnd' | 'StopFailure'): Promise[]>` — loads state under the per-session lock (Task 6), reads new lines (Task 7), derives/merges `agent.usage.request` records (Task 8) keyed into `state.openRequests`, updates `state.branchCounts`/`state.compactionCount` (bumped on `PreCompact`), builds one event per completed `agent.usage.request` plus one `agent.session.summary` (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`; `PreCompact`/`StopFailure` return only usage requests, never a summary — matches spec), saves state, **returns** the built events (does not forward them), and swallows every error internally, returning `[]` on failure (never throws into `processOtlpEvent`). -**Test-first: yes — calling `runMainTranscriptParse` twice with the same transcript (simulating a re-parse after a crash before state was saved) forwards `agent.usage.request` events whose `event_id`-determining fields (`request_id`, `model`) are identical both times — the idempotent-reparse scenario from the external data-model doc's §8.** +**Test-first: yes — calling `collectMainTranscriptEvents` twice with the same transcript (simulating a re-parse after a crash before state was saved) returns `agent.usage.request` events whose natural-key fields (`request_id`, `model`) are identical both times — the idempotent-reparse scenario from the external data-model doc's §8. (`event_id` itself is not idempotent across these two calls — see spec.md's Open risks; the natural key is what's actually stable.)** -- [ ] Write the failing idempotent-reparse test plus a basic Stop → one summary + N usage-request forwards test (mock `forwardOtlpEventToSpool`). -- [ ] Implement `orchestrator.ts` and the `claude-code-otlp.plugin.ts` wiring. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. -- [ ] Commit. +- [x] Write the failing idempotent-reparse test (on natural key, not `event_id`) plus a basic Stop → one summary + N usage-request results test. +- [x] Implement `orchestrator.ts` and the `claude-code-otlp.plugin.ts` per-handler wiring. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. +- [x] Commit. --- ### Task 12: Orchestrate subagent triggers (`SubagentStop`, `SessionEnd` backstop) +Same return-not-forward design as Task 11: the function returns its derived events rather than sending them. + **Files:** - Modify: `orchestrator.ts` (Task 11) — add the subagent path. -- Modify: `claude-code-otlp.plugin.ts` — on `SubagentStop`, call the new function with that record's own subagent transcript path (`hookEvent['agent_transcript_path']`, read loosely off the parsed JSON since it is outside `BaseClaudeCodeHookEvent`'s modeled fields) instead of the main one; `SessionEnd` additionally re-runs it for **every** subagent file `findSubagentFiles()` (Task 9) discovers, not just ones already seen — the crashed/missed-hook backstop. +- Modify: `claude-code-otlp.plugin.ts` — `onSubagentStopEvent` reads `agent_transcript_path`/`agent_id`/`tool_use_id`/`agent_type` off the parsed hook event, builds a `SubagentFile`, and calls `collectSubagentTranscriptEvents()`; `onSessionEndEvent` additionally calls `findSubagentFiles()` (Task 9) and runs it for **every** subagent file discovered, not just ones already seen — the crashed/missed-hook backstop. - Test: extend `__tests__/orchestrator.test.ts`. **Interfaces:** -- Produces: `runSubagentTranscriptParse(sessionId: string, mainTranscriptPath: string, subagentFile: SubagentFile): Promise` — reads new lines from `state.subagentOffsets[subagentFile.agentId]` (Task 7), derives `agent.usage.request` with `scope_kind: 'agent'` (Task 8), builds and forwards one `agent.subagent.usage` (Task 9), saves the updated offset back into the shared `TranscriptParseState`. +- Produces: `collectSubagentTranscriptEvents(sessionId: string, subagentFile: SubagentFile): Promise[]>` — reads new lines from `state.subagentOffsets[subagentFile.agentId]` (Task 7) under the per-session lock, derives `agent.usage.request` with `scope_kind: 'agent'` (Task 8), builds one `agent.subagent.usage` (Task 9) summarizing this agent's *cumulative* usage, saves the updated offset back into the shared `TranscriptParseState`, and **returns** the built events rather than forwarding them. -**Test-first: yes — a `SessionEnd` on a session with three subagent files, only one of which already has a `SubagentStop`-advanced offset, still forwards three `agent.subagent.usage` events (the backstop), and never forwards a fourth for a subagent whose offset shows nothing new since the last run.** +**Test-first: yes — a `SessionEnd` on a session with three subagent files, only one of which already has a `SubagentStop`-advanced offset, still returns three `agent.subagent.usage` events (the backstop); a subagent whose offset shows nothing new since the last run still returns its (unchanged) cumulative `agent.subagent.usage` summary rather than being skipped — that re-run guarantee is the whole point of the backstop.** -- [ ] Write the failing backstop test (three fixture subagents, one pre-advanced offset) and a no-new-bytes-means-no-resend test. -- [ ] Implement. -- [ ] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. -- [ ] Commit. +- [x] Write the failing backstop test (three fixture subagents, one pre-advanced offset) and a no-new-bytes-still-returns-cumulative-summary test. +- [x] Implement. +- [x] Run `npx vitest run src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts` — PASS. +- [x] Commit. --- ## Self-Review Notes -- **Spec coverage:** Common fields (Tasks 1-3), story resolution (4-5), transcript parse-state (6-7), the three new events (8-10), and the four trigger wirings (11-12) each map to a numbered spec section. `event_id`'s "no `generateUUID()`" and the privacy "never raw prompt text" constraints are enforced structurally (pure-function `event_id`, resolved-only story fields) rather than left to each task's judgment. +- **Spec coverage:** Common fields (Tasks 1-3), story resolution (4-5), transcript parse-state (6-7), the three new events (8-10), and the four trigger wirings (11-12) each map to a numbered spec section. The privacy "never raw prompt text" constraint is enforced structurally (only the resolved `story_id`/`story_source` ever leaves `mapHookRecords()`) rather than left to each task's judgment. - **Non-goals respected:** no task touches `agent.session.env`, `agent.skill.dispatch`, `agent.git.snapshot`, or any existing event's own content fields — only the common-field wrapper in `mapHookRecords()`. - **Type consistency:** `OpenUsageRequest` (Task 6) is the one shape Tasks 8, 9, and 11/12 all import and merge/aggregate — no parallel redefinition. `TranscriptParseState` (Task 6) is the single state object Tasks 7, 11, and 12 all read/mutate/save. diff --git a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md index 42cea3670..56ea336a0 100644 --- a/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md +++ b/docs/superpowers/tasks/2026-10-01-epmcdme-15301-stage-1-1/spec.md @@ -16,16 +16,11 @@ This is additive to the existing 13-event `agent.*` taxonomy already produced by Every field below is sent on **every** emitted event. Split by *where* each is computed: -- **Plugin-owned, resolved live in the forwarder — no hook-side capture, nothing new added to the spool.** A new method on the `OtlpAgentAdapter` interface (`src/agents/core/types.ts:734-739`, alongside its existing sole method `processOtlpEvent`): `prepareAnalyticsFields(hookEvent: Record): Promise`, implemented by `ClaudeCodeOtlpPlugin`. Called from the forwarder, resolving the right plugin per record off the spooled `OtlpHookSpoolData.agentName`: - ```ts - const analyticsAgent = AgentRegistry.getAnalyticsAgent(spoolData.agentName); - const commonFields = (await analyticsAgent?.prepareAnalyticsFields(hookEvent)) ?? {}; - ``` - (mirrors the existing `AgentRegistry.getAnalyticsAgent(otlpHookSpoolData.agentName)` lookup already used in `otlp.plugin.ts`'s `handleHooks()` to validate `agentName` before spooling.) `ClaudeCodeOtlpPlugin.processOtlpEvent()` (the hook CLI process) is unchanged — nothing new is captured there and nothing new is added to the spooled record. Everything below is computed autocalculated on the fly, inside this method, when the forwarder calls it: - - `platform` — the plugin's own `platform` class field (`'claude-code' as const`). - - `client_version` — `const { stdout: claudeVersion } = await execPromise('claude --version');`, run live by the method itself. - - `entrypoint` — `process.env.CLAUDE_CODE_ENTRYPOINT`, read live by the method itself (empty string if unset, not blocking). - - `agent_id`/`agent_type` — read off this record's own `hookEvent` (already spooled as `raw` today), because interpreting which fields identify an agent/subagent is specific to that agent's hook-payload shape, not something a generic, agent-agnostic forwarder should own. Each OTLP plugin (today: `claude-code-otlp`; future: `cursor`/`codex`/`copilot`) implements its own interpretation. +- **Plugin-owned, resolved hook-time inside `processOtlpEvent` itself.** `ClaudeCodeOtlpPlugin.withCommonFields()` (private, `claude-code-otlp.plugin.ts`) merges a small object onto every record in a `ForwardDecision`'s payload — the original hook event and any transcript-derived synthetic events alike — right before `forwardToSpool()` sends them. + - `platform` — the literal `'claude-code'`. + - `entrypoint` — `process.env.CLAUDE_CODE_ENTRYPOINT ?? ''`, read live. + - `client_version` — `resolveClientVersion()` (`client-version-cache.ts`): runs `claude --version` once and caches the result in a TTL file cache under `getCodemiePath('cache', ...)` (1h TTL). A *file* cache, not an in-memory one — each hook fire is a fresh CLI process, so there's no live instance to memoize on across calls. + - `agent_id`/`agent_type` are **not** common fields. They either arrive verbatim on the raw hook payload for subagent-shaped hook events (`SubagentStart`/`SubagentStop`, passed through as-is) or are sourced inside the transcript builders themselves (`agent.usage.request`/`agent.subagent.usage`, see "New events" below) — not through any shared enrichment step. - **Daemon-side, client-agnostic** — computed once per forward tick in `buildForwardContext()` and stamped onto every record in `mapHookRecords()` (`forwarder.ts`), the same pattern already used today for `user_email`/`developer_name`/`git_branch`/`repo_remote`/`codemie_project_name`: - `schema_version=2` @@ -36,14 +31,9 @@ Every field below is sent on **every** emitted event. Split by *where* each is c ## `event_id` -One mechanism for every event, old and new: a plain string composed from fields already on the record, computed daemon-side, no `generateUUID()`, no stored/persisted id anywhere. +One mechanism for every event, old and new: a `randomUUID()` stamped once, hook-time, inside `forwardOtlpEventToSpool()` (`src/agents/plugins/utils.ts`) — the single chokepoint every record (the original hook event and every transcript-derived synthetic event alike) already passes through on its way to the spool. `mapHookRecords()` (`forwarder.ts`) carries that value straight through (`event_id: hookEvent['event_id']`) rather than computing anything daemon-side; `schema_version`/`codemie_cli_version` are still stamped there. -| Event | `event_id` | -|---|---| -| Existing 13 hook-mapped events | `${session_id}:${type}:` — computed in `mapHookRecords()` (`forwarder.ts`) from fields already available there; nothing new added to `OtlpHookSpoolData`. | -| `agent.usage.request` | `${session_id}:agent.usage.request:${request_id}:${model}` — `model` included to match the `(request_id, model)` uniqueness key stated in "New events" below. | -| `agent.subagent.usage` | `${session_id}:agent.subagent.usage:${tool_use_id}` | -| `agent.session.summary` | `${session_id}:agent.session.summary:${phase}` (`phase` is `incremental` or `final`; every `Stop`-triggered re-emission this session reuses the same `incremental` id — intentional, a running summary is one record that gets refreshed) | +Because `event_id` is generated fresh on every forward rather than derived from a record's natural key, it does not provide dedup across re-parses — see "Open risks" below. ## Transcript re-parsing @@ -51,9 +41,9 @@ Needed to produce the three new events (`agent.usage.request`, `agent.subagent.u Per the data-model doc (§5.1), parsing is **incremental and backed by persisted per-session state** at `~/.codemie/analytics/state/.json`: -- **State holds**: the byte offset already consumed in the main transcript, and separately for each subagent transcript; the `openRequests` max-merge-in-progress map; `activeSkill`; `branchCounts` (feeds `branch_dominant`). -- **Triggers**: `Stop`, `SubagentStop`, `PreCompact`, `SessionEnd` — each reads only the bytes appended since its stored offset, updates the running tallies, derives any newly-complete records, writes the updated state back to disk, then sends. -- **Recovery path**: if the state file is missing (first run for a session) or fails to parse (corruption), fall back to a full parse from byte `0` and treat every record as newly derived. Each record's `event_id` is a pure function of its natural key (see "`event_id`" above), so a record re-derived this way carries the exact same id as before — the backend sees an update, not a duplicate. +- **State holds**: the byte offset already consumed in the main transcript, and separately for each subagent transcript; the `openRequests` max-merge-in-progress map; `activeSkill`; `branchCounts` (feeds `branch_dominant`); `compactionCount` (persisted count of this session's `PreCompact` triggers, feeds `agent.session.summary`'s `compaction_count`). +- **Triggers**: `Stop`, `SubagentStop`, `PreCompact`, `StopFailure`, `SessionEnd` — each reads only the bytes appended since its stored offset, updates the running tallies, derives any newly-complete records, writes the updated state back to disk, then sends. Each session's load-mutate-save cycle is serialized by a per-session exclusive-create lock file (`withParseStateLock`, `parse-state.ts`) so concurrent hook processes for the same session (e.g. sibling `SubagentStop` fires) can't race on the shared state file; a stale lock (holder crashed) is detected and stolen rather than awaited forever. +- **Recovery path**: if the state file is missing (first run for a session) or fails to parse (corruption), fall back to a full parse from byte `0` and treat every record as newly derived. Because `event_id` is generated fresh at forward-time (see "`event_id`" above) rather than derived from a record's natural key, a record re-derived this way is assigned a new `event_id` — the backend sees a new record, not an update to the original. See Open risks. - Stays `async`, swallows all errors internally (matching `hook.ts`'s existing try/catch + `process.exitCode` convention), and always exits 0. ## Transcript field shape (verified against real transcripts) @@ -76,12 +66,12 @@ Verified against real Claude Code session transcripts (current client, multiple ### `agent.usage.request` -- Fires on `Stop` (re-parse of the main transcript), `SubagentStop` (re-parse of that subagent's own transcript, `scope_kind=agent`), `PreCompact` (re-parse of the main transcript, so in-progress request usage is captured before compaction can drop the turns it came from), and `SessionEnd` (final re-parse of the main transcript **and every subagent transcript**). +- Fires on `Stop` (re-parse of the main transcript), `SubagentStop` (re-parse of that subagent's own transcript, `scope_kind=agent`), `PreCompact` (re-parse of the main transcript, so in-progress request usage is captured before compaction can drop the turns it came from), `StopFailure` (same re-parse as `Stop`, for a turn that ended via failure rather than a clean stop), and `SessionEnd` (final re-parse of the main transcript **and every subagent transcript**). - One per unique `(request_id, model)` across the session transcript and all subagent transcripts. - Take the max per numeric field across duplicate records (per `openRequests` merge). - `scope_kind`/`scope_name` = `main`/`skill`/`agent` depending on which transcript (main vs. a named skill context vs. a subagent transcript) the record came from. -Fields: `request_id`, `timestamp`, `model_raw`, `model`, `speed`, `inference_geo`, `service_tier`, `input_tokens`, `cache_creation_5m_tokens`, `cache_creation_1h_tokens`, `cache_read_tokens`, `output_tokens`, `web_search_requests`, `web_fetch_requests`, `scope_kind`, `scope_name`, `agent_id`, `stop_reason`, `is_api_error`, `git_branch`. All sourced from the transcript shape confirmed above (`message.id`, `message.usage.*`, `message.stop_reason`, `server_tool_use.*`, `gitBranch`) or, for `model`/`model_raw`, from the same resolution chain the statusline already uses (`parseRoutingHeaders()`/`parseBackendModelName()`); `agent_id` comes from the new plugin method above. `is_api_error`'s presence pattern on a real error is unverified — see Open risks. +Fields: `request_id`, `timestamp`, `model_raw`, `model`, `speed`, `inference_geo`, `service_tier`, `input_tokens`, `cache_creation_5m_tokens`, `cache_creation_1h_tokens`, `cache_read_tokens`, `output_tokens`, `web_search_requests`, `web_fetch_requests`, `scope_kind`, `scope_name`, `agent_id`, `stop_reason`, `is_api_error`, `git_branch`. All sourced from the transcript shape confirmed above (`message.id`, `message.usage.*`, `message.stop_reason`, `server_tool_use.*`, `gitBranch`) or, for `model`/`model_raw`, from the same resolution chain the statusline already uses (`parseRoutingHeaders()`/`parseBackendModelName()`); `agent_id` is passed through by the caller (the orchestrator) — `''` for a main-transcript record, the subagent's own `agentId` for a subagent-transcript record — not resolved by any shared enrichment step. `is_api_error`'s presence pattern on a real error is unverified — see Open risks. ### `agent.subagent.usage` @@ -120,7 +110,7 @@ Reading is read-only here: **nothing in this stage writes that file.** A command - `git` reads `git config user.name`/`user.email`. - `codemie_cli` reads the existing CLI profile config. - `os` reads `os.userInfo().username`. -- **Security review sign-off (2026-10-05):** this `jwt → git → codemie_cli → os` derivation chain was flagged by code review as a CRITICAL "new attribution-identifier source" under `security-practices.md`'s Project & User Attribution Headers rule (CR-023), since it derives an identity-like value from local git config / CLI config / OS username with no verification. Reviewed and approved as implemented: the chain is used only to stamp `developer_name`/`identity_source` on outbound analytics/telemetry events (`identity.ts`), never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing that the cited rule's header table concerns. No code change required. +- **Security scope (approved 2026-10-05):** this `jwt → git → codemie_cli → os` derivation chain reads an identity-like value from local git config / CLI config / OS username with no external verification. It is used only to stamp `developer_name`/`identity_source` on outbound analytics/telemetry events (`identity.ts`) — never on the SSO proxy's outbound attribution headers, billing, tenant isolation, or LLM request routing, which `security-practices.md`'s Project & User Attribution Headers rule governs. ## Non-goals @@ -135,9 +125,10 @@ Reading is read-only here: **nothing in this stage writes that file.** A command - `spawn_depth` (`agent.subagent.usage`) is present in the `.meta.json` sidecar only for nested subagents (depth ≥ 2); top-level subagents omit it, so implementation needs an explicit default rather than treating absence as an error. - `workflow_run`/`worktree` (`agent.subagent.usage`) have no identified source in this codebase's transcript handling or the `.meta.json` sidecar. The external data-model doc names a `workflows//` sidecar directory as the source, but a direct check across every local session directory found no such directory in any sampled session — the gap stands; worth revisiting with the data-model doc's owner. - `title` (`agent.session.summary`) has no identified source in a real transcript, top-level or nested, and the data-model doc doesn't name one either — unresolved on both sides. -- **Post-implementation disclosed limitations** (surfaced by code review; accepted as-is rather than reworked, matching the precedent above for `title`/`workflow_run`/`worktree`): +- **Known limitations, accepted as-is for this sub-stage:** - `lines_added`/`lines_removed` (`agent.session.summary`) always emit `0`. An `Edit`/`Write` tool_use's `input` carries the *proposed* edit, not a diff stat, so no reliable added/removed line count can be derived from it without re-implementing diffing — out of scope for this sub-stage. - `scope_kind: 'skill'` (`agent.usage.request`) is never emitted. No reliable in-transcript signal for "this turn is inside a skill context" was found, so every main-transcript usage record is unconditionally scoped `'main'`. `state.activeSkill` stays tracked-but-unused, available for a later task that identifies a real signal. - - `started_at`/`ended_at` (`agent.session.summary`) are close approximations, not the literal `SessionStart`/`SessionEnd` hook payload timestamps: `started_at` is the main transcript's own first parsed line timestamp, and `ended_at` is `Date.now()` at parse time. Threading the actual hook timestamps through would require widening every `runMainTranscriptParse` call site's signature; deferred. + - `started_at`/`ended_at` (`agent.session.summary`) are close approximations, not the literal `SessionStart`/`SessionEnd` hook payload timestamps: `started_at` is the main transcript's own first parsed line timestamp, and `ended_at` is `Date.now()` at parse time. Threading the actual hook timestamps through would require widening every `collectMainTranscriptEvents` call site's signature; deferred. - `commands_in_order` (`agent.session.summary`) contains the right distinct slash-command names but not a true chronological sequence — it is `Object.keys()` of `commandInvocations`, an unordered count map. No chronological invocation order is available anywhere in `NamedInvocationCounts` to draw from. A genuine mismatch with the field's name, documented rather than fabricated. - `description` (`agent.subagent.usage`) always emits `''`. The sidecar `.meta.json` schema it mirrors (matching the real production schema) has no `description` key at all, so there is no source anywhere in this codebase to populate it from — unlike the sibling `workflow_run`/`worktree`/`title` gaps above, this one wasn't caught before implementation. + - `event_id` does not provide idempotent dedup across re-parses: it's a `randomUUID()` generated fresh every time a record is forwarded, not a deterministic function of the record's natural key, so a crash-before-save re-parse of `agent.usage.request`/`agent.subagent.usage`/`agent.session.summary` forwards the re-derived record under a new `event_id` — the backend sees a new record rather than an update to the original. Accepted as-is for this sub-stage; the natural key is still stable across re-derivation (`request_id`+`model`, `tool_use_id`, or `session_id`+`phase`), so dedup on that key is possible if the backend needs it. From 7a7afc7b362a9d08264a0cf2715c8ed390ba63ff Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Wed, 7 Oct 2026 18:14:51 +0300 Subject: [PATCH 25/35] refactor(proxy): add types package --- .../__tests__/claude-code-otlp.plugin.test.ts | 15 - .../claude-agent-sdk-hooks.d.ts | 745 ++++++++++++++++++ .../claude-code-otlp.plugin.ts | 130 ++- .../claude-code-otlp.types.ts | 24 +- 4 files changed, 830 insertions(+), 84 deletions(-) create mode 100644 src/agents/plugins/claude-code-otlp/claude-agent-sdk-hooks.d.ts diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index a3b8f0210..f5392fe25 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -268,21 +268,6 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { ); }); - it('forwards the raw event and skips all transcript-parse dispatch when session_id is empty', async () => { - const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); - const plugin = new ClaudeCodeOtlpPlugin(); - const parsedEvent = hookEvent({ session_id: '', hook_event_name: 'Stop' }); - const rawEvent = JSON.stringify(parsedEvent); - - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); - - expect(collectMainTranscriptEventsMock).not.toHaveBeenCalled(); - expect(forwardOtlpEventToSpool).toHaveBeenCalledWith( - expect.objectContaining(parsedEvent), - 'claude-code-otlp' - ); - }); - it('forwards every event a per-event handler returns (the raw event plus any derived events) through the single forwardToSpool path, in order', async () => { const derivedUsageEvent = { type: 'agent.usage.request' }; const derivedSummaryEvent = { type: 'agent.session.summary' }; diff --git a/src/agents/plugins/claude-code-otlp/claude-agent-sdk-hooks.d.ts b/src/agents/plugins/claude-code-otlp/claude-agent-sdk-hooks.d.ts new file mode 100644 index 000000000..6fdda2ff3 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/claude-agent-sdk-hooks.d.ts @@ -0,0 +1,745 @@ +/** + * Local ambient hook type declarations for `@anthropic-ai/claude-agent-sdk`. + * + * NOT HAND-WRITTEN. This is a trimmed copy of the hook-related exports from + * the real SDK's `sdk.d.ts` (version 0.3.292), kept only because we cannot + * depend on the npm package itself: + * + * `@anthropic-ai/claude-agent-sdk` ships one `optionalDependencies` entry per + * platform (`-darwin-x64`, `-linux-arm64`, etc.) containing the native CLI + * binary. Those per-platform packages declare `"license": "SEE LICENSE IN README.md"`, + * which `license-checker` (`npm run license-check`) reports as an unrecognized + * "Custom" license and fails the build. We only ever use this package for + * `import type` - no SDK runtime code is used - so there is no reason to carry + * the dependency (and its native binaries) just to satisfy the compiler. + * + * To update after bumping the version referenced in comments above, or after + * adding a new hook event: + * 1. `npm install @anthropic-ai/claude-agent-sdk@` into a scratch/throwaway + * location (e.g. `npm pack` + extract, or a disposable project) - do not add + * it back to this repo's package.json. + * 2. Diff that package's `sdk.d.ts` against this file for the hook-related + * types (anything with `Hook` in the name, plus their referenced support + * types) and copy over what changed. + * 3. Bump the version noted above and re-check that + * `src/agents/plugins/claude-code-otlp/` still type-checks. + */ +declare module '@anthropic-ai/claude-agent-sdk' { + export type PermissionBehavior = 'allow' | 'deny' | 'ask'; + + export type PermissionMode = 'default' | 'acceptEdits' | 'bypassPermissions' | 'plan' | 'dontAsk' | 'auto'; + + export type PermissionRuleValue = { + toolName: string; + ruleContent?: string; + }; + + export type BackgroundTaskSummary = { + id: string; + /** + * Friendly task-type label (e.g. 'shell', 'subagent', 'monitor', 'workflow'). Falls back to the raw discriminant for unknown types. + */ + type: string; + status: string; + /** + * Free-text description. Capped at 1000 chars; clipped values append an in-string "… [+N chars]" marker. + */ + description: string; + /** + * Shell command line. Only present for 'shell' tasks. Capped at 1000 chars with the same "… [+N chars]" marker. + */ + command?: string; + /** + * Subagent type name. Only present for 'subagent' tasks. + */ + agent_type?: string; + /** + * MCP server name. Only present for 'monitor' / 'MCP task' tasks. + */ + server?: string; + /** + * MCP tool name. Only present for 'monitor' / 'MCP task' tasks. + */ + tool?: string; + /** + * Workflow name. Only present for 'workflow' tasks. + */ + name?: string; + }; + + export type ExitReason = 'clear' | 'resume' | 'logout' | 'prompt_input_exit' | 'other'; + + export type McpServerProvenance = { + name: string; + /** + * sdk | plugin | user | project | local | dynamic | managed | enterprise | claudeai | agent — an open set; treat unknown values as an unrecognized configured source, never as sdk. + */ + source: string; + }; + + export type PermissionUpdate = { + type: 'addRules'; + rules: PermissionRuleValue[]; + behavior: PermissionBehavior; + destination: PermissionUpdateDestination; + } | { + type: 'replaceRules'; + rules: PermissionRuleValue[]; + behavior: PermissionBehavior; + destination: PermissionUpdateDestination; + } | { + type: 'removeRules'; + rules: PermissionRuleValue[]; + behavior: PermissionBehavior; + destination: PermissionUpdateDestination; + } | { + type: 'setMode'; + mode: PermissionMode; + destination: PermissionUpdateDestination; + } | { + type: 'addDirectories'; + directories: string[]; + destination: PermissionUpdateDestination; + } | { + type: 'removeDirectories'; + directories: string[]; + destination: PermissionUpdateDestination; + }; + + export type PermissionUpdateDestination = 'userSettings' | 'projectSettings' | 'localSettings' | 'session' | 'cliArg'; + + export type PostToolBatchToolCall = { + tool_name: string; + tool_input: unknown; + tool_use_id: string; + tool_response?: unknown; + }; + + export type SDKAssistantMessageError = 'authentication_failed' | 'oauth_org_not_allowed' | 'account_on_hold' | 'verification_required' | 'billing_error' | 'rate_limit' | 'overloaded' | 'invalid_request' | 'model_not_found' | 'server_error' | 'unknown' | 'max_output_tokens' | 'cloud_credential_error'; + + export type SessionCronSummary = { + id: string; + /** + * Cron expression, e.g. "0 9 * * 1-5". + */ + schedule: string; + /** + * False for one-shot wakeups whose cron field encodes a single fire time; true for tasks that re-fire on every match. + */ + recurring: boolean; + /** + * Prompt text submitted when the cron fires. Capped at 1000 chars; clipped values append an in-string "… [+N chars]" marker. + */ + prompt: string; + }; + + export type AsyncHookJSONOutput = { + async: true; + asyncTimeout?: number; + }; + + export type BaseHookInput = { + session_id: string; + transcript_path: string; + cwd: string; + /** + * UUID correlating a user prompt with all subsequent events until the next prompt. Same value emitted on OpenTelemetry events as the `prompt.id` attribute, so hook output can be joined to OTel events at prompt grain. Absent until the first user input of the process lifetime. + */ + prompt_id?: string; + permission_mode?: string; + /** + * Subagent identifier. Present only when the hook fires from within a subagent (e.g., a tool called by an AgentTool worker). Absent for the main thread, even in --agent sessions. Use this field (not agent_type) to distinguish subagent calls from main-thread calls. + */ + agent_id?: string; + /** + * Agent type name (e.g., "general-purpose", "code-reviewer"). Present when the hook fires from within a subagent (alongside agent_id), or on the main thread of a session started with --agent (without agent_id). + */ + agent_type?: string; + /** + * Reasoning effort applied to the current turn. Same shape as StatusLineCommandInput.effort. Present for hooks that fire within a tool-use context (PreToolUse, PostToolUse, Stop, SubagentStop, etc.) on a model that supports the effort parameter; absent for session-lifecycle hooks and models without effort support. + */ + effort?: { + /** + * Active effort level for the current turn (e.g., "low", "medium", "high", "xhigh", "max"), after any silent downgrade for the selected model. Also exposed to hook commands and Bash as the CLAUDE_EFFORT env var. + */ + level: string; + }; + }; + + export type ConfigChangeHookInput = BaseHookInput & { + hook_event_name: 'ConfigChange'; + source: 'user_settings' | 'project_settings' | 'local_settings' | 'policy_settings' | 'skills'; + file_path?: string; + }; + + export type CwdChangedHookInput = BaseHookInput & { + hook_event_name: 'CwdChanged'; + old_cwd: string; + new_cwd: string; + }; + + export type CwdChangedHookSpecificOutput = { + hookEventName: 'CwdChanged'; + watchPaths?: string[]; + }; + + export type DirectoryAddedHookInput = BaseHookInput & { + hook_event_name: 'DirectoryAdded'; + /** + * Absolute path of the directory that was added. + */ + directory: string; + /** + * How the directory was added: "slash_command" for /add-dir, "register_repo_root" for the SDK control_request. + */ + source: 'slash_command' | 'register_repo_root'; + }; + + export type ElicitationHookInput = BaseHookInput & { + hook_event_name: 'Elicitation'; + mcp_server_name: string; + message: string; + mode?: 'form' | 'url'; + url?: string; + elicitation_id?: string; + requested_schema?: Record; + }; + + export type ElicitationHookSpecificOutput = { + hookEventName: 'Elicitation'; + action?: 'accept' | 'decline' | 'cancel'; + content?: Record; + }; + + export type ElicitationResultHookInput = BaseHookInput & { + hook_event_name: 'ElicitationResult'; + mcp_server_name: string; + elicitation_id?: string; + mode?: 'form' | 'url'; + action: 'accept' | 'decline' | 'cancel'; + content?: Record; + }; + + export type ElicitationResultHookSpecificOutput = { + hookEventName: 'ElicitationResult'; + action?: 'accept' | 'decline' | 'cancel'; + content?: Record; + }; + + export type FileChangedHookInput = BaseHookInput & { + hook_event_name: 'FileChanged'; + file_path: string; + event: 'change' | 'add' | 'unlink'; + }; + + export type FileChangedHookSpecificOutput = { + hookEventName: 'FileChanged'; + watchPaths?: string[]; + }; + + export type HookCallback = (input: HookInput, toolUseID: string | undefined, options: { + signal: AbortSignal; + }) => Promise; + + export type HookEvent = 'PreToolUse' | 'PostToolUse' | 'PostToolUseFailure' | 'PostToolBatch' | 'Notification' | 'UserPromptSubmit' | 'UserPromptExpansion' | 'SessionStart' | 'SessionEnd' | 'Stop' | 'StopFailure' | 'SubagentStart' | 'SubagentStop' | 'PreCompact' | 'PostCompact' | 'PreModelSwitch' | 'PostModelSwitch' | 'PermissionRequest' | 'PermissionDenied' | 'Setup' | 'TeammateIdle' | 'TaskCreated' | 'TaskCompleted' | 'Elicitation' | 'ElicitationResult' | 'ConfigChange' | 'WorktreeCreate' | 'WorktreeRemove' | 'InstructionsLoaded' | 'CwdChanged' | 'FileChanged' | 'DirectoryAdded' | 'MessageDisplay'; + + export type HookInput = PreToolUseHookInput | PostToolUseHookInput | PostToolUseFailureHookInput | PostToolBatchHookInput | PermissionDeniedHookInput | NotificationHookInput | UserPromptSubmitHookInput | UserPromptExpansionHookInput | SessionStartHookInput | SessionEndHookInput | StopHookInput | StopFailureHookInput | SubagentStartHookInput | SubagentStopHookInput | PreCompactHookInput | PostCompactHookInput | PreModelSwitchHookInput | PostModelSwitchHookInput | PermissionRequestHookInput | SetupHookInput | TeammateIdleHookInput | TaskCreatedHookInput | TaskCompletedHookInput | ElicitationHookInput | ElicitationResultHookInput | ConfigChangeHookInput | InstructionsLoadedHookInput | WorktreeCreateHookInput | WorktreeRemoveHookInput | CwdChangedHookInput | FileChangedHookInput | DirectoryAddedHookInput | MessageDisplayHookInput; + + export type HookJSONOutput = AsyncHookJSONOutput | SyncHookJSONOutput; + + export type HookPermissionDecision = 'allow' | 'deny' | 'ask' | 'defer'; + + export type InstructionsLoadedHookInput = BaseHookInput & { + hook_event_name: 'InstructionsLoaded'; + file_path: string; + memory_type: 'User' | 'Project' | 'Local' | 'Managed'; + load_reason: 'session_start' | 'nested_traversal' | 'path_glob_match' | 'include' | 'compact'; + globs?: string[]; + trigger_file_path?: string; + parent_file_path?: string; + }; + + export type MessageDisplayHookInput = BaseHookInput & { + hook_event_name: 'MessageDisplay'; + /** + * UUID of the current turn. + */ + turn_id: string; + /** + * UUID of the assistant message being displayed. Stable across every flush of the same message. Not the API msg_… id. + */ + message_id: string; + /** + * Zero-based index of this delta within the message. Increments by one per flush. + */ + index: number; + /** + * True on the message's last flush. Exactly one flush per message has it. + */ + final: boolean; + /** + * The newly completed lines since the prior flush. Always whole lines, except on the final flush which may end mid-line. The delta of the final flush is empty when the message ends on a newline; treat final as the end-of-message signal regardless. + */ + delta: string; + }; + + export type MessageDisplayHookSpecificOutput = { + hookEventName: 'MessageDisplay'; + /** + * Text displayed in place of the delta. Omit (or return the delta unchanged) to display the original. + */ + displayContent?: string; + }; + + export type NotificationHookInput = BaseHookInput & { + hook_event_name: 'Notification'; + message: string; + title?: string; + notification_type: string; + }; + + export type NotificationHookSpecificOutput = { + hookEventName: 'Notification'; + additionalContext?: string; + }; + + export type PermissionDeniedHookInput = BaseHookInput & { + hook_event_name: 'PermissionDenied'; + tool_name: string; + tool_input: unknown; + tool_use_id: string; + reason: string; + mcp_server?: McpServerProvenance; + }; + + export type PermissionDeniedHookSpecificOutput = { + hookEventName: 'PermissionDenied'; + retry?: boolean; + }; + + export type PermissionRequestHookInput = BaseHookInput & { + hook_event_name: 'PermissionRequest'; + tool_name: string; + tool_input: unknown; + permission_suggestions?: PermissionUpdate[]; + mcp_server?: McpServerProvenance; + }; + + export type PermissionRequestHookSpecificOutput = { + hookEventName: 'PermissionRequest'; + decision: { + behavior: 'allow'; + updatedInput?: Record; + updatedPermissions?: PermissionUpdate[]; + } | { + behavior: 'deny'; + message?: string; + interrupt?: boolean; + }; + }; + + export type PostCompactHookInput = BaseHookInput & { + hook_event_name: 'PostCompact'; + trigger: 'manual' | 'auto'; + /** + * The conversation summary produced by compaction + */ + compact_summary: string; + }; + + export type PostModelSwitchHookInput = (BaseHookInput & { + hook_event_name: 'PostModelSwitch'; + }) & { + /** + * Resolved model id the session was running before the switch + */ + from_model: string; + /** + * Resolved model id the session runs after the switch + */ + to_model: string; + /** + * What was asked for (alias such as "opus", a full id, or null for "default") + */ + requested_model: string | null; + /** + * command: /model , the /config Model row, or enabling fast mode when that promotes the model; picker: an interactive model picker; sdk: headless set_model (SDK, Remote Control, IDE); auto: automatic fallback or other programmatic change; resume: model restored while resuming a session + */ + source: 'command' | 'picker' | 'sdk' | 'auto' | 'resume'; + /** + * Prompt tokens the next request re-sends: the last main-thread response's input + cache_read + cache_creation + output tokens (0 before the first response; for a server-side tool loop, its last iteration's window, not the summed totals) + */ + context_tokens: number; + /** + * Whether the current model's prompt cache is likely still warm (a switch then forfeits it) + */ + prompt_cache_warm: boolean; + cache_ttl: '5m' | '1h'; + /** + * Estimated cost of re-caching context_tokens on to_model at its cache-write rate — the managed modelPricing when set, otherwise list price; excludes the response + */ + estimated_cache_write_usd: number; + /** + * configured: priced at the managed modelPricing setting; catalog: list price; default: to_model unknown, the default tier was assumed + */ + pricing: 'configured' | 'catalog' | 'default'; + }; + + export type PostModelSwitchHookSpecificOutput = { + hookEventName: 'PostModelSwitch'; + /** + * Reaches the model with the next request the new model serves + */ + additionalContext?: string; + }; + + export type PostToolBatchHookInput = BaseHookInput & { + hook_event_name: 'PostToolBatch'; + tool_calls: PostToolBatchToolCall[]; + }; + + export type PostToolBatchHookSpecificOutput = { + hookEventName: 'PostToolBatch'; + additionalContext?: string; + }; + + export type PostToolUseFailureHookInput = BaseHookInput & { + hook_event_name: 'PostToolUseFailure'; + tool_name: string; + tool_input: unknown; + tool_use_id: string; + error: string; + is_interrupt?: boolean; + /** + * Tool execution time in milliseconds. Excludes permission-prompt and hook time. + */ + duration_ms?: number; + mcp_server?: McpServerProvenance; + }; + + export type PostToolUseFailureHookSpecificOutput = { + hookEventName: 'PostToolUseFailure'; + additionalContext?: string; + }; + + export type PostToolUseHookInput = BaseHookInput & { + hook_event_name: 'PostToolUse'; + tool_name: string; + tool_input: unknown; + tool_response: unknown; + tool_use_id: string; + /** + * Tool execution time in milliseconds. Excludes permission-prompt and hook time. + */ + duration_ms?: number; + mcp_server?: McpServerProvenance; + }; + + export type PostToolUseHookSpecificOutput = { + hookEventName: 'PostToolUse'; + additionalContext?: string; + /** + * Host-asserted context shown to the auto-mode permission classifier alongside this tool call's result. In the live session the classifier may weigh a user statement relayed here as user intent (it can satisfy a consent bar a user turn would satisfy, never a hard boundary); values restored from saved session state are treated as unverified context only. Relay discipline is the host's obligation: put ONLY genuine user statements in intent-bearing positions — never tool output or model text dressed as one. Capped at 2000 UTF-16 code units, a budget shared across all hooks that contribute to one call (surrogate-pair-safe; emoji and other astral characters count as two). Honored on synchronous hook responses only: an async hook's late response arrives after the result message is frozen and this field in it is silently ignored. Security note: do not copy untrusted tool output or third-party text into it blindly — content placed here reaches the permission classifier with host-application framing. Applies only to calls the classifier transcript shows: read-only lookups the transcript omits (file reads, searches) and remote-engine shells produce no per-result line, and context attached to them is silently unused. Not a delivery channel: it is bound to a single call id and sized for a short assertion, not for relaying messages or events. Rewrite integrity: if this assertion describes output you are rewriting, return it in the SAME hook result as the rewrite — it is then dropped automatically if your rewrite is rejected or superseded by a later hook's rewrite; assertions returned without a rewrite are never invalidated by other hooks' rewrites, so a non-rewriting hook should assert only what holds regardless of other hooks' rewrites — hosts that need an assertion bound to exact output bytes should make it in the hook that produces those bytes. (Do NOT return an identity rewrite just to pair an assertion: hooks run in parallel on the ORIGINAL output, so an identity rewrite competes last-write-wins with sibling rewrites and can clobber a real redaction.) + */ + classifierContext?: string; + /** + * Replaces the tool output before it is sent to the model + */ + updatedToolOutput?: unknown; + /** + * Replaces the output for MCP tools only. Prefer updatedToolOutput, which works for all tools + */ + updatedMCPToolOutput?: unknown; + }; + + export type PreCompactHookInput = BaseHookInput & { + hook_event_name: 'PreCompact'; + trigger: 'manual' | 'auto'; + custom_instructions: string | null; + }; + + export type PreModelSwitchHookInput = (BaseHookInput & { + hook_event_name: 'PreModelSwitch'; + }) & { + /** + * Resolved model id the session was running before the switch + */ + from_model: string; + /** + * Resolved model id the session runs after the switch + */ + to_model: string; + /** + * What was asked for (alias such as "opus", a full id, or null for "default") + */ + requested_model: string | null; + /** + * command: /model , the /config Model row, or enabling fast mode when that promotes the model; picker: an interactive model picker; sdk: headless set_model (SDK, Remote Control, IDE) + */ + source: 'command' | 'picker' | 'sdk'; + /** + * Prompt tokens the next request re-sends: the last main-thread response's input + cache_read + cache_creation + output tokens (0 before the first response; for a server-side tool loop, its last iteration's window, not the summed totals) + */ + context_tokens: number; + /** + * Whether the current model's prompt cache is likely still warm (a switch then forfeits it) + */ + prompt_cache_warm: boolean; + cache_ttl: '5m' | '1h'; + /** + * Estimated cost of re-caching context_tokens on to_model at its cache-write rate — the managed modelPricing when set, otherwise list price; excludes the response + */ + estimated_cache_write_usd: number; + /** + * configured: priced at the managed modelPricing setting; catalog: list price; default: to_model unknown, the default tier was assumed + */ + pricing: 'configured' | 'catalog' | 'default'; + }; + + export type PreModelSwitchHookSpecificOutput = { + hookEventName: 'PreModelSwitch'; + /** + * Same contract as PreToolUse: allow proceeds (skipping the interactive cache-miss confirm), deny cancels the switch, ask asks the user to confirm (a headless session refuses instead) + */ + permissionDecision?: 'allow' | 'deny' | 'ask'; + permissionDecisionReason?: string; + }; + + export type PreToolUseHookInput = BaseHookInput & { + hook_event_name: 'PreToolUse'; + tool_name: string; + tool_input: unknown; + tool_use_id: string; + mcp_server?: McpServerProvenance; + }; + + export type PreToolUseHookSpecificOutput = { + hookEventName: 'PreToolUse'; + permissionDecision?: HookPermissionDecision; + permissionDecisionReason?: string; + updatedInput?: Record; + additionalContext?: string; + }; + + export type SessionEndHookInput = BaseHookInput & { + hook_event_name: 'SessionEnd'; + reason: ExitReason; + }; + + export type SessionStartHookInput = BaseHookInput & { + hook_event_name: 'SessionStart'; + source: 'startup' | 'resume' | 'clear' | 'compact' | 'fork'; + agent_type?: string; + model?: string; + session_title?: string; + /** + * resume/fork: seconds since the resumed transcript's last assistant response + */ + seconds_since_last_response?: number; + /** + * resume/fork: the resumed transcript's last response input + cache_read + cache_creation + output tokens (for a server-side tool loop, its last iteration's window, not the summed totals) + */ + context_tokens?: number; + /** + * resume/fork: seconds_since_last_response exceeds the prompt-cache TTL, so the first request re-caches context_tokens + */ + prompt_cache_likely_expired?: boolean; + /** + * resume/fork: estimated cost of re-caching context_tokens on the session model — the managed modelPricing when set, otherwise list price; excludes the response + */ + estimated_cache_write_usd?: number; + }; + + export type SessionStartHookSpecificOutput = { + hookEventName: 'SessionStart'; + additionalContext?: string; + initialUserMessage?: string; + sessionTitle?: string; + watchPaths?: string[]; + /** + * Re-scan skill and command directories after SessionStart hooks complete, so skills installed by the hook are available in the same session + */ + reloadSkills?: boolean; + }; + + export type SetupHookInput = BaseHookInput & { + hook_event_name: 'Setup'; + trigger: 'init' | 'maintenance'; + }; + + export type SetupHookSpecificOutput = { + hookEventName: 'Setup'; + additionalContext?: string; + }; + + export type StopFailureHookInput = BaseHookInput & { + hook_event_name: 'StopFailure'; + error: SDKAssistantMessageError; + error_details?: string; + last_assistant_message?: string; + }; + + export type StopHookInput = BaseHookInput & { + hook_event_name: 'Stop'; + stop_hook_active: boolean; + /** + * Text content of the last assistant message before stopping. Avoids the need to read and parse the transcript file. + */ + last_assistant_message?: string; + /** + * In-flight background work (running/pending + backgrounded) registered in this session. Lets hooks distinguish "session is done" from "session is paused waiting for background work to wake it". Empty array when nothing is in flight. + */ + background_tasks?: BackgroundTaskSummary[]; + /** + * Session-scoped cron tasks (CronCreate, ScheduleWakeup, /loop) that will wake this session later. Empty array when none are scheduled. + */ + session_crons?: SessionCronSummary[]; + + + }; + + export type StopHookSpecificOutput = { + hookEventName: 'Stop'; + additionalContext?: string; + }; + + export type SubagentStartHookInput = BaseHookInput & { + hook_event_name: 'SubagentStart'; + agent_id: string; + agent_type: string; + }; + + export type SubagentStartHookSpecificOutput = { + hookEventName: 'SubagentStart'; + additionalContext?: string; + }; + + export type SubagentStopHookInput = BaseHookInput & { + hook_event_name: 'SubagentStop'; + stop_hook_active: boolean; + agent_id: string; + agent_transcript_path: string; + agent_type: string; + /** + * Text content of the last assistant message before stopping. Avoids the need to read and parse the transcript file. + */ + last_assistant_message?: string; + /** + * In-flight background work (running/pending + backgrounded) registered in this session. Lets hooks distinguish "session is done" from "session is paused waiting for background work to wake it". Empty array when nothing is in flight. + */ + background_tasks?: BackgroundTaskSummary[]; + /** + * Session-scoped cron tasks (CronCreate, ScheduleWakeup, /loop) that will wake this session later. Empty array when none are scheduled. + */ + session_crons?: SessionCronSummary[]; + + + }; + + export type SubagentStopHookSpecificOutput = { + hookEventName: 'SubagentStop'; + additionalContext?: string; + }; + + export type SyncHookJSONOutput = { + continue?: boolean; + suppressOutput?: boolean; + stopReason?: string; + decision?: 'approve' | 'block'; + systemMessage?: string; + /** + * A terminal escape sequence (e.g. OSC 9 / OSC 777 desktop-notification) for Claude Code to emit on your behalf. Only notification/title OSCs (0, 1, 2, 9, 99, 777) and BEL are permitted; anything else is dropped. + */ + terminalSequence?: string; + reason?: string; + + + hookSpecificOutput?: PreToolUseHookSpecificOutput | UserPromptSubmitHookSpecificOutput | UserPromptExpansionHookSpecificOutput | SessionStartHookSpecificOutput | SetupHookSpecificOutput | PreModelSwitchHookSpecificOutput | PostModelSwitchHookSpecificOutput | SubagentStartHookSpecificOutput | PostToolUseHookSpecificOutput | PostToolUseFailureHookSpecificOutput | PostToolBatchHookSpecificOutput | StopHookSpecificOutput | SubagentStopHookSpecificOutput | PermissionDeniedHookSpecificOutput | NotificationHookSpecificOutput | PermissionRequestHookSpecificOutput | ElicitationHookSpecificOutput | ElicitationResultHookSpecificOutput | CwdChangedHookSpecificOutput | FileChangedHookSpecificOutput | WorktreeCreateHookSpecificOutput | MessageDisplayHookSpecificOutput; + }; + + export type TaskCompletedHookInput = BaseHookInput & { + hook_event_name: 'TaskCompleted'; + task_id: string; + task_subject: string; + task_description?: string; + teammate_name?: string; + /** + * @deprecated Sessions have a single implicit team; this carries the session-derived team name and will be removed in a future release. + */ + team_name?: string; + }; + + export type TaskCreatedHookInput = BaseHookInput & { + hook_event_name: 'TaskCreated'; + task_id: string; + task_subject: string; + task_description?: string; + teammate_name?: string; + /** + * @deprecated Sessions have a single implicit team; this carries the session-derived team name and will be removed in a future release. + */ + team_name?: string; + }; + + export type TeammateIdleHookInput = BaseHookInput & { + hook_event_name: 'TeammateIdle'; + teammate_name: string; + /** + * @deprecated Sessions have a single implicit team; this carries the session-derived team name and will be removed in a future release. + */ + team_name: string; + }; + + export type UserPromptExpansionHookInput = BaseHookInput & { + hook_event_name: 'UserPromptExpansion'; + expansion_type: 'slash_command' | 'mcp_prompt'; + command_name: string; + command_args: string; + command_source?: string; + prompt: string; + }; + + export type UserPromptExpansionHookSpecificOutput = { + hookEventName: 'UserPromptExpansion'; + additionalContext?: string; + /** + * When decision is "block", omit the original prompt from the block message + */ + suppressOriginalPrompt?: boolean; + }; + + export type UserPromptSubmitHookInput = BaseHookInput & { + hook_event_name: 'UserPromptSubmit'; + prompt: string; + /** + * Who authored/injected the prompt: `user` = submitted from the interactive composer, `sdk` = non-interactive entrypoint (`-p` / Agent SDK), `loop_wakeup` = dynamic /loop wakeup, `schedule_wakeup` = scheduled-task fire (CronCreate/routine), `system` = other machine-injected turns (peer/channel messages, task notifications, auto-continuation), `poll_event` = the poll-event channel enqueue-time pass (the hook fires when the host submits an event, before its delivery ack exists — a blocking verdict rejects the event). Payloads may omit it while the field rolls out. + */ + source?: 'user' | 'sdk' | 'system' | 'loop_wakeup' | 'schedule_wakeup' | 'poll_event'; + session_title?: string; + }; + + export type UserPromptSubmitHookSpecificOutput = { + hookEventName: 'UserPromptSubmit'; + additionalContext?: string; + sessionTitle?: string; + /** + * When decision is "block", omit the original prompt from the block message + */ + suppressOriginalPrompt?: boolean; + }; + + export type WorktreeCreateHookInput = BaseHookInput & { + hook_event_name: 'WorktreeCreate'; + name: string; + }; + + export type WorktreeCreateHookSpecificOutput = { + hookEventName: 'WorktreeCreate'; + worktreePath: string; + }; + + export type WorktreeRemoveHookInput = BaseHookInput & { + hook_event_name: 'WorktreeRemove'; + worktree_path: string; + }; +} diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index 3ed3f72c9..c74e9ca7d 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -1,11 +1,20 @@ import { basename } from 'node:path'; +import type { + HookInput, + PreCompactHookInput, + SessionEndHookInput, + StopFailureHookInput, + StopHookInput, + SubagentStopHookInput, + UserPromptSubmitHookInput, +} from '@anthropic-ai/claude-agent-sdk'; import { AuthGateResult, ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; import { logger } from '@/utils/logger.js'; import { ConfigLoader } from '@/utils/config.js'; import { AgentAdapterType, OtlpAdapterDeps, OtlpAgentAdapter } from '@/agents/core/types.js'; import { CLAUDE_CODE_OTLP_AGENT_NAME } from './claude-code-otlp.constants.js'; -import { ForwardDecision } from './claude-code-otlp.types.js'; +import { ForwardDecision, isClaudeCodeHookInput } from './claude-code-otlp.types.js'; import { forwardOtlpEventToSpool } from '../utils.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; import { @@ -20,8 +29,12 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; public readonly type = AgentAdapterType.OTLP; - public async processOtlpEvent(rawEvent: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { - const event = JSON.parse(rawEvent) as Record; + public async processOtlpEvent(rawHookInput: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { + const hookInput: unknown = JSON.parse(rawHookInput); + if (!isClaudeCodeHookInput(hookInput)) { + logger.debug('[Claude Code OTLP plugin] hook payload missing session_id/cwd/hook_event_name, ignoring'); + return; + } // INVARIANT - do not weaken. An untracked project must produce NO hooks data // in the daemon spool, must not start the daemon, and must not run the SSO @@ -30,7 +43,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { // data, and skips sessions that only have OTEL data. Forwarding a hook event // for an untracked project would make the session sendable and leak its // data to the backend. - const isTracked = await isProjectTracked(readString(event, 'cwd'), await readAllowlistState()); + const isTracked = await isProjectTracked(hookInput.cwd, await readAllowlistState()); if (!isTracked) { logger.debug('[Claude Code OTLP plugin] project not in analytics allowlist, ignoring hook event'); return; @@ -38,7 +51,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { await ensureOtlpProxy(this.name); - const evaluation = await this.evaluate(event); + const evaluation = await this.evaluate(hookInput); if (evaluation.decision === 'block') { logger.error(`[Claude Code OTLP plugin] Blocking prompt: ${evaluation.reason}`); console.log(JSON.stringify(evaluation)); @@ -57,65 +70,46 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { return records.map((record) => ({ ...record, ...common })); } - private async evaluate(parsed: Record): Promise { - const sessionId = readString(parsed, 'session_id'); - if (!sessionId) { - return { decision: 'forward', payload: [parsed] }; - } - - const hookEventName = readString(parsed, 'hook_event_name'); - if (hookEventName === 'UserPromptSubmit') { - return await this.onUserPromptSubmit(parsed); + private async evaluate(hookInput: HookInput): Promise { + if (hookInput.hook_event_name === 'UserPromptSubmit') { + return await this.onUserPromptSubmit(hookInput); } - if (hookEventName === 'Stop') { - return await this.onStopEvent(parsed); + if (hookInput.hook_event_name === 'Stop') { + return await this.onStopEvent(hookInput); } - if (hookEventName === 'PreCompact') { - return await this.onPreCompactEvent(parsed); + if (hookInput.hook_event_name === 'PreCompact') { + return await this.onPreCompactEvent(hookInput); } - if (hookEventName === 'StopFailure') { - return await this.onStopFailureEvent(parsed); + if (hookInput.hook_event_name === 'StopFailure') { + return await this.onStopFailureEvent(hookInput); } - if (hookEventName === 'SessionEnd') { - return await this.onSessionEndEvent(parsed); + if (hookInput.hook_event_name === 'SessionEnd') { + return await this.onSessionEndEvent(hookInput); } - if (hookEventName === 'SubagentStop') { - return await this.onSubagentStopEvent(parsed); + if (hookInput.hook_event_name === 'SubagentStop') { + return await this.onSubagentStopEvent(hookInput); } - return { decision: 'forward', payload: [parsed] }; + return { decision: 'forward', payload: [hookInput] }; } - private async onStopEvent(parsed: Record): Promise { - const derived = await collectMainTranscriptEvents( - readString(parsed, 'session_id'), - readString(parsed, 'transcript_path'), - 'Stop' - ); - return { decision: 'forward', payload: [parsed, ...derived] }; + private async onStopEvent(hookInput: StopHookInput): Promise { + const derived = await collectMainTranscriptEvents(hookInput.session_id, hookInput.transcript_path, 'Stop'); + return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onPreCompactEvent(parsed: Record): Promise { - const derived = await collectMainTranscriptEvents( - readString(parsed, 'session_id'), - readString(parsed, 'transcript_path'), - 'PreCompact' - ); - return { decision: 'forward', payload: [parsed, ...derived] }; + private async onPreCompactEvent(hookInput: PreCompactHookInput): Promise { + const derived = await collectMainTranscriptEvents(hookInput.session_id, hookInput.transcript_path, 'PreCompact'); + return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onStopFailureEvent(parsed: Record): Promise { - const derived = await collectMainTranscriptEvents( - readString(parsed, 'session_id'), - readString(parsed, 'transcript_path'), - 'StopFailure' - ); - return { decision: 'forward', payload: [parsed, ...derived] }; + private async onStopFailureEvent(hookInput: StopFailureHookInput): Promise { + const derived = await collectMainTranscriptEvents(hookInput.session_id, hookInput.transcript_path, 'StopFailure'); + return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onSessionEndEvent(parsed: Record): Promise { - const sessionId = readString(parsed, 'session_id'); - const transcriptPath = readString(parsed, 'transcript_path'); + private async onSessionEndEvent(hookInput: SessionEndHookInput): Promise { + const { session_id: sessionId, transcript_path: transcriptPath } = hookInput; const derived = await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd'); // Backstop: guarantee every subagent discovered for this session gets at least one @@ -125,26 +119,26 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { derived.push(...(await collectSubagentTranscriptEvents(sessionId, file))); } - return { decision: 'forward', payload: [parsed, ...derived] }; + return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onSubagentStopEvent(parsed: Record): Promise { - const agentTranscriptPath = readOptionalString(parsed, 'agent_transcript_path'); + private async onSubagentStopEvent(hookInput: SubagentStopHookInput): Promise { + const agentTranscriptPath = hookInput.agent_transcript_path; if (!agentTranscriptPath) { - return { decision: 'forward', payload: [parsed] }; + return { decision: 'forward', payload: [hookInput] }; } const subagentFile: SubagentFile = { - agentId: - readOptionalString(parsed, 'agent_id') ?? - basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''), + agentId: hookInput.agent_id || basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''), filePath: agentTranscriptPath, - toolUseId: readOptionalString(parsed, 'tool_use_id'), - agentType: readOptionalString(parsed, 'agent_type'), + // `tool_use_id` is not declared on the SDK's SubagentStopHookInput type; read it + // defensively in case the raw hook payload carries it anyway. + toolUseId: readOptionalString(hookInput, 'tool_use_id'), + agentType: hookInput.agent_type, }; - const derived = await collectSubagentTranscriptEvents(readString(parsed, 'session_id'), subagentFile); - return { decision: 'forward', payload: [parsed, ...derived] }; + const derived = await collectSubagentTranscriptEvents(hookInput.session_id, subagentFile); + return { decision: 'forward', payload: [hookInput, ...derived] }; } private async ensureProxyAuth(): Promise { @@ -164,13 +158,13 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { } } - private async onUserPromptSubmit(parsed: Record): Promise { + private async onUserPromptSubmit(hookInput: UserPromptSubmitHookInput): Promise { const authResult = await this.ensureProxyAuth(); if (authResult.ok) { return { decision: 'forward', - payload: [parsed], + payload: [hookInput], } } @@ -189,11 +183,11 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { } } -function readOptionalString(record: Record, key: string): string | undefined { - const value = record[key]; +/** + * Reads a string field not declared on the SDK's `HookInput` typings (e.g. `tool_use_id` on + * `SubagentStop`), in case the raw hook payload carries it anyway. + */ +function readOptionalString(record: object, key: string): string | undefined { + const value = (record as Record)[key]; return typeof value === 'string' ? value : undefined; } - -function readString(record: Record, key: string): string { - return readOptionalString(record, key) ?? ''; -} diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts index 5cb254264..f7e452a6f 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts @@ -1,3 +1,25 @@ +import type { HookInput, UserPromptSubmitHookSpecificOutput } from '@anthropic-ai/claude-agent-sdk'; + export type ForwardDecision = | { decision: 'forward'; payload: Record[] } - | { decision: 'block'; reason: string, hookSpecificOutput: Record }; + | { decision: 'block'; reason: string; hookSpecificOutput: UserPromptSubmitHookSpecificOutput }; + +/** + * Narrows parsed hook JSON to the SDK's `HookInput` union. + * + * Only the fields shared by every event (`BaseHookInput` plus the + * discriminant) are verified at runtime. Event-specific fields are trusted + * to match once `hook_event_name` is narrowed, and unknown event names are + * accepted so a newer Claude Code does not break ingestion. + */ +export function isClaudeCodeHookInput(value: unknown): value is HookInput { + if (typeof value !== 'object' || value === null) { + return false; + } + const candidate = value as Record; + return ( + typeof candidate.session_id === 'string' && + typeof candidate.cwd === 'string' && + typeof candidate.hook_event_name === 'string' + ); +} From 197adb57632cf39beda27cdddebf8f6649a17a95 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Thu, 8 Oct 2026 10:54:48 +0300 Subject: [PATCH 26/35] feat(proxy): run Claude Code OTLP hooks async except UserPromptSubmit Analytics forwarding hooks no longer block Claude Code. UserPromptSubmit stays synchronous so its SSO-gate block decision reaches Claude Code before the prompt proceeds. --- .../__tests__/claude-code-otlp.test.ts | 65 ++++++++++++++----- .../proxy/connectors/claude-code-otlp.ts | 22 +++++-- 2 files changed, 64 insertions(+), 23 deletions(-) diff --git a/src/cli/commands/proxy/connectors/__tests__/claude-code-otlp.test.ts b/src/cli/commands/proxy/connectors/__tests__/claude-code-otlp.test.ts index 80647050b..b46ee0c68 100644 --- a/src/cli/commands/proxy/connectors/__tests__/claude-code-otlp.test.ts +++ b/src/cli/commands/proxy/connectors/__tests__/claude-code-otlp.test.ts @@ -63,10 +63,10 @@ function buildExpectedEnv(state: { url: string; gatewayKey: string }) { }; } -function codemieHookGroup() { +function codemieHookGroup(event: (typeof HOOK_EVENTS)[number]) { return { matcher: '', - hooks: [{ type: 'command', command: `codemie ${CODEMIE_COMMAND_MARKER}` }], + hooks: [{ type: 'command', command: `codemie ${CODEMIE_COMMAND_MARKER}`, async: event !== 'UserPromptSubmit' }], }; } @@ -120,10 +120,39 @@ describe('claude-code-otlp connector', () => { expect(settings.env).toEqual({ ...buildExpectedEnv(mockState), [CODEMIE_ANALYTICS_PROJECT_FILTER_ENV]: '[]' }); for (const event of HOOK_EVENTS) { - expect(settings.hooks[event]).toEqual([codemieHookGroup()]); + expect(settings.hooks[event]).toEqual([codemieHookGroup(event)]); } }); + it('writes async: false only for UserPromptSubmit and async: true for every other event', async () => { + await writeClaudeCodeOtlpConfig(); + + const settings = await readJson(join(homeDir, '.claude', 'settings.json')); + for (const event of HOOK_EVENTS) { + expect(settings.hooks[event][0].hooks[0].async).toBe(event !== 'UserPromptSubmit'); + } + expect(settings.hooks.UserPromptSubmit[0].hooks[0].async).toBe(false); + }); + + it('upgrades a previously written codemie hook without async to the async-aware shape', async () => { + const settingsPath = join(homeDir, '.claude', 'settings.json'); + await mkdir(join(homeDir, '.claude'), { recursive: true }); + const legacyGroup = { + matcher: '', + hooks: [{ type: 'command', command: `codemie ${CODEMIE_COMMAND_MARKER}` }], + }; + await writeFile( + settingsPath, + JSON.stringify({ hooks: { PostToolUse: [legacyGroup], UserPromptSubmit: [legacyGroup] } }, null, 2) + ); + + await writeClaudeCodeOtlpConfig(); + + const merged = await readJson(settingsPath); + expect(merged.hooks.PostToolUse).toEqual([codemieHookGroup('PostToolUse')]); + expect(merged.hooks.UserPromptSubmit).toEqual([codemieHookGroup('UserPromptSubmit')]); + }); + it('writes to the user-level settings and tracks the project root when scope is "project"', async () => { const result = await writeClaudeCodeOtlpConfig({ scope: 'project' }); @@ -190,9 +219,9 @@ describe('claude-code-otlp connector', () => { expect(merged.env).toEqual(expect.objectContaining({ FOO: 'bar', ...buildExpectedEnv(mockState) })); expect(merged.hooks.PreToolUse).toEqual([ { matcher: 'Bash', hooks: [{ type: 'command', command: 'some-other-tool' }] }, - codemieHookGroup(), + codemieHookGroup('PreToolUse'), ]); - expect(merged.hooks.SessionStart).toEqual([codemieHookGroup()]); + expect(merged.hooks.SessionStart).toEqual([codemieHookGroup('SessionStart')]); }); it('is idempotent for a fresh install: rerunning does not create a backup or duplicate hooks', async () => { @@ -237,7 +266,7 @@ describe('claude-code-otlp connector', () => { const merged = await readJson(settingsPath); expect(merged.hooks.PreToolUse).toEqual([ { matcher: 'Bash', hooks: [{ type: 'command', command: 'some-other-tool' }] }, - codemieHookGroup(), + codemieHookGroup('PreToolUse'), ]); for (const event of HOOK_EVENTS) { expect(merged.hooks[event]).toHaveLength(event === 'PreToolUse' ? 2 : 1); @@ -298,7 +327,7 @@ describe('claude-code-otlp connector', () => { expect(result.written).toBe(true); const merged = await readJson(settingsPath); - expect(merged.hooks.PreToolUse).toEqual([codemieHookGroup()]); + expect(merged.hooks.PreToolUse).toEqual([codemieHookGroup('PreToolUse')]); }); it('dedups multiple stale whole-codemie hook groups down to exactly one, preserving foreign groups', async () => { @@ -327,7 +356,7 @@ describe('claude-code-otlp connector', () => { await writeClaudeCodeOtlpConfig(); const merged = await readJson(settingsPath); - expect(merged.hooks.SessionStart).toEqual([foreignEntry, codemieHookGroup()]); + expect(merged.hooks.SessionStart).toEqual([foreignEntry, codemieHookGroup('SessionStart')]); }); it('backs up a pre-existing minimal ("{}") settings file, unlike a genuinely missing file', async () => { @@ -354,7 +383,7 @@ describe('claude-code-otlp connector', () => { const merged = await readJson(settingsPath); expect(merged.env).toEqual({ ...buildExpectedEnv(mockState), [CODEMIE_ANALYTICS_PROJECT_FILTER_ENV]: '[]' }); for (const event of HOOK_EVENTS) { - expect(merged.hooks[event]).toEqual([codemieHookGroup()]); + expect(merged.hooks[event]).toEqual([codemieHookGroup(event)]); } }); @@ -392,7 +421,7 @@ describe('claude-code-otlp connector', () => { const merged = await readJson(settingsPath); expect(merged.hooks.PreToolUse).toEqual([ { matcher: '', hooks: [{ type: 'command', command: 'echo my-own-hook' }] }, - codemieHookGroup(), + codemieHookGroup('PreToolUse'), ]); }); @@ -414,7 +443,7 @@ describe('claude-code-otlp connector', () => { const merged = await readJson(settingsPath); expect(merged.hooks.PreToolUse).toEqual([ { matcher: '', hooks: [{ type: 'command', command: 'echo my-own-hook' }] }, - codemieHookGroup(), + codemieHookGroup('PreToolUse'), ]); }); @@ -443,7 +472,7 @@ describe('claude-code-otlp connector', () => { { type: 'command', command: 'echo second' }, ], }, - codemieHookGroup(), + codemieHookGroup('PreToolUse'), ]); }); @@ -465,7 +494,7 @@ describe('claude-code-otlp connector', () => { const merged = await readJson(settingsPath); expect(merged.hooks.PreToolUse).toEqual([ { matcher: 'Bash', hooks: [{ type: 'command', command: 'echo bash-only-hook' }] }, - codemieHookGroup(), + codemieHookGroup('PreToolUse'), ]); }); @@ -488,7 +517,7 @@ describe('claude-code-otlp connector', () => { const merged = await readJson(settingsPath); expect(merged.hooks.PreToolUse).toEqual([ { matcher: '', hooks: [{ type: 'command', command: 'echo my-own-hook' }] }, - codemieHookGroup(), + codemieHookGroup('PreToolUse'), ]); }); }); @@ -688,7 +717,7 @@ describe('claude-code-otlp connector', () => { await mkdir(join(homeDir, '.claude'), { recursive: true }); await writeFile( settingsPath, - JSON.stringify({ theme: 'dark', hooks: { PreToolUse: [codemieHookGroup()] } }, null, 2) + JSON.stringify({ theme: 'dark', hooks: { PreToolUse: [codemieHookGroup('PreToolUse')] } }, null, 2) ); await removeClaudeCodeOtlpConfig(); @@ -813,7 +842,7 @@ describe('claude-code-otlp connector', () => { it('reports an "absent" reason for project scope when the allowlist key is not set', async () => { const settingsPath = await seedSettings( - JSON.stringify({ theme: 'dark', env: { OTEL_LOGS_EXPORTER: 'otlp' }, hooks: { Stop: [codemieHookGroup()] } }) + JSON.stringify({ theme: 'dark', env: { OTEL_LOGS_EXPORTER: 'otlp' }, hooks: { Stop: [codemieHookGroup('Stop')] } }) ); const result = await removeClaudeCodeOtlpConfig({ scope: 'project' }); @@ -843,7 +872,7 @@ describe('claude-code-otlp connector', () => { it('returns removed:true when only a codemie hook is present (no env keys)', async () => { const settingsPath = await seedSettings( - JSON.stringify({ theme: 'dark', hooks: { Stop: [codemieHookGroup()] } }) + JSON.stringify({ theme: 'dark', hooks: { Stop: [codemieHookGroup('Stop')] } }) ); const result = await removeClaudeCodeOtlpConfig(); @@ -936,7 +965,7 @@ describe('claude-code-otlp connector', () => { const merged = await readJson(settingsPath); expect(merged.hooks.SomeRetiredEvent).toBeUndefined(); for (const event of HOOK_EVENTS) { - expect(merged.hooks[event]).toEqual([codemieHookGroup()]); + expect(merged.hooks[event]).toEqual([codemieHookGroup(event)]); } }); diff --git a/src/cli/commands/proxy/connectors/claude-code-otlp.ts b/src/cli/commands/proxy/connectors/claude-code-otlp.ts index cb54831c4..b4924d2da 100644 --- a/src/cli/commands/proxy/connectors/claude-code-otlp.ts +++ b/src/cli/commands/proxy/connectors/claude-code-otlp.ts @@ -6,6 +6,7 @@ */ import { existsSync } from 'node:fs'; +import type { HookEvent } from '@anthropic-ai/claude-agent-sdk'; import { copyFile, readFile, unlink } from 'node:fs/promises'; import { ConfigurationError } from '@/utils/errors.js'; import { logger } from '@/utils/logger.js'; @@ -57,6 +58,7 @@ interface RemoveClaudeCodeOtlpResult { interface HookEntry { type: string; command: string; + async?: boolean; [key: string]: unknown; } @@ -99,7 +101,11 @@ export const HOOK_EVENTS = [ 'SubagentStop', 'PreCompact', 'Notification', -] as const; +] as const satisfies readonly HookEvent[]; + +const SYNCHRONOUS_HOOKS: ReadonlyArray<(typeof HOOK_EVENTS)[number]> = [ + 'UserPromptSubmit', +]; export const SETTINGS_BACKUP_SUFFIX = '.codemie-backup'; export const CODEMIE_COMMAND_MARKER = `hook --agent ${CLAUDE_CODE_OTLP_AGENT_NAME}`; @@ -288,9 +294,15 @@ export async function writeClaudeCodeOtlpConfig( } // --- Merge hooks block --- - const codemieEntry: HookGroup = { - matcher: '', - hooks: [{ type: 'command', command: `codemie ${CODEMIE_COMMAND_MARKER}` }], + // UserPromptSubmit stays synchronous: its SSO-gate `block` decision must reach + // Claude Code before the prompt proceeds. Every other event is fire-and-forget. + const buildCodemieEntry = (eventName: (typeof HOOK_EVENTS)[number]): HookGroup => { + const hook: HookEntry & { async: boolean } = { + type: 'command', + command: `codemie ${CODEMIE_COMMAND_MARKER}`, + async: !SYNCHRONOUS_HOOKS.includes(eventName), + }; + return { matcher: '', hooks: [hook] }; }; // Phase 1: strip our own command from EVERY existing event key, not just the @@ -315,7 +327,7 @@ export async function writeClaudeCodeOtlpConfig( // Phase 2: (re-)add our dedicated entry for every event we currently manage. for (const eventName of HOOK_EVENTS) { - hooks[eventName] = [...(hooks[eventName] ?? []), codemieEntry]; + hooks[eventName] = [...(hooks[eventName] ?? []), buildCodemieEntry(eventName)]; } // --- Merge env block --- From b031ddc1badcf9bb1fde6f238c8a22ce1aec871b Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Thu, 8 Oct 2026 12:32:25 +0300 Subject: [PATCH 27/35] fix(proxy): keep agent.usage.request timestamp and git_branch when forwarding mapHookRecords overwrote the record's own timestamp and git_branch with the spool-write time and the daemon's cwd branch, so every request in a hook pass collapsed to one moment and carried the wrong branch. Prefer the event's own values and fall back only when they are missing, empty or unparseable. Generated with AI --- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 24 +++++++++++++++++-- 1 file changed, 22 insertions(+), 2 deletions(-) diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index 931dc3965..f9f58d140 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -166,6 +166,26 @@ function boundedText(value: unknown, maxChars: number): string { return text.slice(0, maxChars); } +/** + * Prefer the event's own `timestamp` (e.g. the transcript line time on `agent.usage.request`) + * over the spool-write time, which only reflects when the hook ran. Falls back when the + * event has none or an unparseable one. + */ +function resolveEventTimestamp(eventTimestamp: unknown, spoolTimestamp: number): string { + if (typeof eventTimestamp === 'string' && eventTimestamp.length > 0) { + const parsed = new Date(eventTimestamp); + if (!Number.isNaN(parsed.getTime())) { + return parsed.toISOString(); + } + } + return new Date(spoolTimestamp).toISOString(); +} + +/** Prefer the event's own `git_branch` (the session's branch) over the daemon's cwd branch. */ +function resolveEventGitBranch(eventBranch: unknown, daemonBranch: string | undefined): string { + return typeof eventBranch === 'string' && eventBranch.length > 0 ? eventBranch : (daemonBranch ?? ''); +} + function limitHookPayload(hookEvent: Record): Record { const limited: Record = { ...hookEvent }; @@ -260,11 +280,11 @@ export async function mapHookRecords( ...limited, type, session_id: sessionId, - timestamp: new Date(spoolData.timestamp).toISOString(), + timestamp: resolveEventTimestamp(hookEvent['timestamp'], spoolData.timestamp), user_email: ctx.userEmail, developer_name: ctx.identity?.developerName ?? '', identity_source: ctx.identity?.identitySource ?? '', - git_branch: ctx.git.branch ?? '', + git_branch: resolveEventGitBranch(hookEvent['git_branch'], ctx.git.branch), repo_remote: ctx.git.remote ?? '', story_id: promptStory.storyId, story_source: promptStory.storySource, From dd19f45d82e7aaa9804d220876786df8f080314d Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Thu, 8 Oct 2026 15:05:59 +0300 Subject: [PATCH 28/35] refactor(proxy): rework agent.session.summary event contract --- .../transcript/__tests__/orchestrator.test.ts | 24 +-- .../__tests__/session-summary.test.ts | 144 +++++++++++++----- .../transcript/orchestrator.ts | 78 +++++++--- .../transcript/session-summary.ts | 88 ++++++----- 4 files changed, 223 insertions(+), 111 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index 41aae4a02..eee5014ba 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -161,8 +161,8 @@ describe('collectMainTranscriptEvents — Stop trigger', () => { expect(usageEvents).toHaveLength(2); expect(summaryEvents).toHaveLength(1); - expect(summaryEvents[0].phase).toBe('incremental'); - expect(summaryEvents[0]).not.toHaveProperty('ended_at'); + expect(summaryEvents[0].is_final).toBe(false); + expect(summaryEvents[0].ended_at).toBe('2026-10-01T00:00:00.000Z'); }); }); @@ -236,8 +236,7 @@ describe('collectMainTranscriptEvents — SessionEnd trigger', () => { const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); expect(summaryEvents).toHaveLength(1); - expect(summaryEvents[0].phase).toBe('final'); - expect(summaryEvents[0]).toHaveProperty('ended_at'); + expect(summaryEvents[0].is_final).toBe(true); expect(typeof summaryEvents[0].ended_at).toBe('string'); }); }); @@ -267,7 +266,7 @@ describe('collectMainTranscriptEvents — missing transcript file', () => { }); describe('collectMainTranscriptEvents — tool-call accumulation', () => { - it('counts Edit/Write tool_use blocks into files_changed/files_written on the Stop summary', async () => { + it('counts Edit/Write tool_use blocks into files_written/files_edited/files_changed on the Stop summary', async () => { const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-tools'; @@ -299,12 +298,15 @@ describe('collectMainTranscriptEvents — tool-call accumulation', () => { const summary = events.find((e) => e.type === 'agent.session.summary'); expect(summary).toBeDefined(); - expect(summary?.files_written).toEqual(['/repo/a.ts']); - expect(summary?.files_changed).toEqual(['/repo/b.ts']); - expect((summary?.tool_calls as Record).Write).toBe(1); - expect((summary?.tool_calls as Record).Edit).toBe(1); - expect((summary?.tool_errors as Record).Edit).toBe(1); - expect((summary?.tool_errors as Record).Write).toBe(0); + expect(summary?.files_written).toBe(1); + expect(summary?.files_edited).toBe(1); + expect(summary?.files_changed).toBe(2); + const tools = summary?.tools as Record; + expect(tools.Write).toEqual({ calls: 1, errors: 0 }); + expect(tools.Edit).toEqual({ calls: 1, errors: 1 }); + expect(summary?.tool_calls).toBe(2); + expect(summary?.tool_errors).toBe(1); + expect(summary?.tool_results).toBe(2); }); }); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts index 0bb5e3de9..668821165 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts @@ -17,9 +17,8 @@ function emptyAccumulator(): SessionSummaryAccumulator { return { models: {}, toolCalls: {}, - linesAdded: 0, - linesRemoved: 0, - filesChanged: new Set(), + toolResults: 0, + filesEdited: new Set(), filesWritten: new Set(), compactionCount: 0, }; @@ -71,26 +70,37 @@ describe('updateBranchCounts', () => { }); describe('buildSessionSummaryEvent', () => { - it('omits ended_at entirely for phase "incremental"', () => { - const event = buildSessionSummaryEvent( + it('sets is_final from phase and always carries started_at/ended_at', () => { + const incremental = buildSessionSummaryEvent( 'session-1', 'incremental', emptyAccumulator(), emptyNamed(), {}, '2026-10-01T00:00:00.000Z', - undefined + '2026-10-01T00:30:00.000Z' + ); + + expect(incremental.type).toBe('agent.session.summary'); + expect(incremental.session_id).toBe('session-1'); + expect(incremental.is_final).toBe(false); + expect(incremental.started_at).toBe('2026-10-01T00:00:00.000Z'); + expect(incremental.ended_at).toBe('2026-10-01T00:30:00.000Z'); + + const final = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' ); - expect('ended_at' in event).toBe(false); - expect(Object.keys(event)).not.toContain('ended_at'); - expect(event.type).toBe('agent.session.summary'); - expect(event.session_id).toBe('session-1'); - expect(event.phase).toBe('incremental'); - expect(event.started_at).toBe('2026-10-01T00:00:00.000Z'); + expect(final.is_final).toBe(true); }); - it('includes ended_at for phase "final"', () => { + it('computes duration_ms from started_at/ended_at, and null when either is missing', () => { const event = buildSessionSummaryEvent( 'session-1', 'final', @@ -100,18 +110,53 @@ describe('buildSessionSummaryEvent', () => { '2026-10-01T00:00:00.000Z', '2026-10-01T01:00:00.000Z' ); - - expect('ended_at' in event).toBe(true); + expect(event.duration_ms).toBe(3_600_000); expect(event.ended_at).toBe('2026-10-01T01:00:00.000Z'); - expect(event.phase).toBe('final'); + + const noEndedAt = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '' + ); + expect(noEndedAt.duration_ms).toBeNull(); + expect(noEndedAt.ended_at).toBeNull(); + }); + + it('sets the envelope timestamp to ended_at, so the forwarder has a real one to prefer', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + expect(event.timestamp).toBe('2026-10-01T01:00:00.000Z'); + + const noEndedAt = buildSessionSummaryEvent( + 'session-1', + 'incremental', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '' + ); + expect(noEndedAt.timestamp).toBeUndefined(); }); - it('flattens acc.toolCalls {calls, errors} shape into separate tool_calls/tool_errors maps', () => { + it('reports acc.toolCalls verbatim as tools, plus tool_calls/tool_errors/tool_results totals', () => { const acc = emptyAccumulator(); acc.toolCalls = { Read: { calls: 5, errors: 0 }, Edit: { calls: 3, errors: 1 }, }; + acc.toolResults = 8; const event = buildSessionSummaryEvent( 'session-1', @@ -123,14 +168,19 @@ describe('buildSessionSummaryEvent', () => { '2026-10-01T01:00:00.000Z' ); - expect(event.tool_calls).toEqual({ Read: 5, Edit: 3 }); - expect(event.tool_errors).toEqual({ Read: 0, Edit: 1 }); + expect(event.tools).toEqual({ + Read: { calls: 5, errors: 0 }, + Edit: { calls: 3, errors: 1 }, + }); + expect(event.tool_calls).toBe(8); + expect(event.tool_errors).toBe(1); + expect(event.tool_results).toBe(8); }); - it('derives skills_used and primary_command from a constructed NamedInvocationCounts', () => { + it('derives skills, agents and primary_command from a constructed NamedInvocationCounts', () => { const named: NamedInvocationCounts = { skillInvocations: { 'codemie:msgraph': 2, brainstorming: 1 }, - agentInvocations: {}, + agentInvocations: { Explore: 2 }, commandInvocations: { init: 1, deploy: 4 }, }; @@ -144,12 +194,13 @@ describe('buildSessionSummaryEvent', () => { '2026-10-01T01:00:00.000Z' ); - expect(event.skills_used).toEqual({ 'codemie:msgraph': 2, brainstorming: 1 }); + expect(event.skills).toEqual({ 'codemie:msgraph': 2, brainstorming: 1 }); + expect(event.agents).toEqual({ Explore: 2 }); expect(event.primary_command).toBe('deploy'); - expect(event.commands_in_order).toEqual(Object.keys(named.commandInvocations)); + expect(event.commands).toEqual(Object.keys(named.commandInvocations)); }); - it('reports models_used as the full count map and primary_model as the max key', () => { + it('reports models as an array with the primary model first', () => { const acc = emptyAccumulator(); acc.models = { 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 }; @@ -160,18 +211,16 @@ describe('buildSessionSummaryEvent', () => { emptyNamed(), {}, '2026-10-01T00:00:00.000Z', - undefined + '' ); - expect(event.models_used).toEqual({ 'claude-sonnet-4-5': 2, 'claude-opus-4-1': 9 }); + expect(event.models).toEqual(['claude-opus-4-1', 'claude-sonnet-4-5']); expect(event.primary_model).toBe('claude-opus-4-1'); }); - it('reports lines/files/compaction fields and branch_counts/branch_dominant, converting Sets to arrays', () => { + it('reports lines_* as null and files_changed/written/edited as counts', () => { const acc = emptyAccumulator(); - acc.linesAdded = 42; - acc.linesRemoved = 7; - acc.filesChanged = new Set(['a.ts', 'b.ts']); + acc.filesEdited = new Set(['b.ts']); acc.filesWritten = new Set(['a.ts']); acc.compactionCount = 2; @@ -187,16 +236,37 @@ describe('buildSessionSummaryEvent', () => { '2026-10-01T01:00:00.000Z' ); - expect(event.lines_added).toBe(42); - expect(event.lines_removed).toBe(7); - expect(event.files_changed).toEqual(['a.ts', 'b.ts']); - expect(event.files_written).toEqual(['a.ts']); + expect(event.lines_added).toBeNull(); + expect(event.lines_removed).toBeNull(); + expect(event.files_changed).toBe(2); + expect(event.files_written).toBe(1); + expect(event.files_edited).toBe(1); expect(event.compaction_count).toBe(2); expect(event.branch_counts).toBe(branchCounts); expect(event.branch_dominant).toBe('feature'); }); - it('emits title as a literal empty string (no identified source, per Open risks)', () => { + it('counts a file touched by both Edit and Write once in files_changed', () => { + const acc = emptyAccumulator(); + acc.filesEdited = new Set(['a.ts']); + acc.filesWritten = new Set(['a.ts']); + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.files_changed).toBe(1); + expect(event.files_written).toBe(1); + expect(event.files_edited).toBe(1); + }); + + it('emits title as a literal empty string (no identified source)', () => { const event = buildSessionSummaryEvent( 'session-1', 'incremental', @@ -204,7 +274,7 @@ describe('buildSessionSummaryEvent', () => { emptyNamed(), {}, '2026-10-01T00:00:00.000Z', - undefined + '' ); expect(event.title).toBe(''); @@ -218,7 +288,7 @@ describe('buildSessionSummaryEvent', () => { emptyNamed(), {}, '2026-10-01T00:00:00.000Z', - undefined + '' ); expect('api_calls' in event).toBe(false); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 9495cb26b..0bfcb78bb 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -78,18 +78,40 @@ function collectErrorToolUseIds(parsedLines: TranscriptLine[]): Set { return errorByToolUseId; } +/** The first non-empty `timestamp` among `parsedLines`, in order, or `''` when none carry one. */ +function firstTimestamp(parsedLines: TranscriptLine[]): string { + for (const line of parsedLines) { + if (typeof line.timestamp === 'string' && line.timestamp) { + return line.timestamp; + } + } + return ''; +} + function emptyAccumulator(): SessionSummaryAccumulator { return { models: {}, toolCalls: {}, - linesAdded: 0, - linesRemoved: 0, - filesChanged: new Set(), + toolResults: 0, + filesEdited: new Set(), filesWritten: new Set(), compactionCount: 0, }; } +/** Total `tool_result` content blocks across `parsedLines` — the contract's `tool_results` count. */ +function countToolResults(parsedLines: TranscriptLine[]): number { + let count = 0; + for (const parsed of parsedLines) { + const content = parsed.message?.content; + if (!Array.isArray(content)) continue; + for (const item of content as ContentBlock[]) { + if (item?.type === 'tool_result') count += 1; + } + } + return count; +} + /** * Recompute the full-session summary accumulator, named-invocation counts, and session start * time from byte 0 of the main transcript. @@ -108,22 +130,24 @@ function emptyAccumulator(): SessionSummaryAccumulator { * - `toolCalls[*].errors` is derived from a sibling `tool_result` block's `is_error`/`isError` * flag (the same pattern `claude.session.ts`/`claude.metrics-processor.ts` already use for * tool-use_id → error lookups) when one is found; otherwise a tool call's `.errors` stays 0. - * - `linesAdded`/`linesRemoved` default to 0 — an `Edit`/`Write` tool_use's `input` carries the - * *proposed* edit, not a diff stat, so no reliable added/removed line count can be derived from - * it without re-implementing diffing. + * - Lines added/removed are not computed — an `Edit`/`Write` tool_use's `input` carries the + * *proposed* edit, not a diff stat — so the event builder sends them as `null`, never `0`. * - `compactionCount` defaults to 0 — no verified in-transcript signal was found (`PreCompact` is * a hook event, not a transcript line). */ -async function buildFullAccumulator( - transcriptPath: string -): Promise<{ acc: SessionSummaryAccumulator; named: NamedInvocationCounts; startedAt: string }> { +async function buildFullAccumulator(transcriptPath: string): Promise<{ + acc: SessionSummaryAccumulator; + named: NamedInvocationCounts; + startedAt: string; + endedAt: string; +}> { const acc = emptyAccumulator(); let raw: string; try { raw = await readFile(transcriptPath, 'utf-8'); } catch { - return { acc, named: extractNamedInvocations([]), startedAt: '' }; + return { acc, named: extractNamedInvocations([]), startedAt: '', endedAt: '' }; } const rawLines = raw.split('\n').filter((line) => line.trim().length > 0); @@ -139,16 +163,23 @@ async function buildFullAccumulator( // Pass 1: collect tool_result error flags keyed by their matching tool_use_id. const errorByToolUseId = collectErrorToolUseIds(parsedLines); + acc.toolResults = countToolResults(parsedLines); - // Pass 2: models (reusing parseUsageLine's own model-resolution logic). + // Pass 2: models, one count per distinct request — a request can span several + // streaming/finalizing transcript lines, so lines are deduped by the same + // `${requestId}::${model}` key `state.openRequests` uses before counting. + const modelByRequestKey = new Map(); for (const line of rawLines) { const parsedUsage = parseUsageLine(line, 'main', '', ''); if (parsedUsage) { - acc.models[parsedUsage.model] = (acc.models[parsedUsage.model] ?? 0) + 1; + modelByRequestKey.set(`${parsedUsage.requestId}::${parsedUsage.model}`, parsedUsage.model); } } + for (const model of modelByRequestKey.values()) { + acc.models[model] = (acc.models[model] ?? 0) + 1; + } - // Pass 3: tool calls/errors, files changed/written (Edit/Write tool_use payloads). + // Pass 3: tool calls/errors, files edited/written (Edit/Write tool_use payloads). for (const parsed of parsedLines) { const content = parsed.message?.content; if (!Array.isArray(content)) continue; @@ -165,15 +196,19 @@ async function buildFullAccumulator( const filePath = item.input?.file_path ?? item.input?.path; if (typeof filePath === 'string' && filePath) { if (item.name === 'Write') acc.filesWritten.add(filePath); - if (item.name === 'Edit') acc.filesChanged.add(filePath); + if (item.name === 'Edit') acc.filesEdited.add(filePath); } } } const named = extractNamedInvocations(parsedLines); - const startedAt = parsedLines.length > 0 ? String(parsedLines[0].timestamp ?? '') : ''; + // Real transcripts interleave non-message lines (file-history-snapshot, cost-state, ...) + // without a `timestamp`, including at index 0/length-1 — so the first/last *timestamped* + // line is used, not literally the first/last line. + const startedAt = firstTimestamp(parsedLines); + const endedAt = firstTimestamp([...parsedLines].reverse()); - return { acc, named, startedAt }; + return { acc, named, startedAt, endedAt }; } /** @@ -245,10 +280,9 @@ export async function collectMainTranscriptEvents( } if (trigger === 'Stop' || trigger === 'SessionEnd') { - const { acc, named, startedAt } = await buildFullAccumulator(transcriptPath); + const { acc, named, startedAt, endedAt } = await buildFullAccumulator(transcriptPath); acc.compactionCount = state.compactionCount; const phase = trigger === 'SessionEnd' ? 'final' : 'incremental'; - const endedAt = trigger === 'SessionEnd' ? new Date().toISOString() : undefined; const summaryEvent = buildSessionSummaryEvent( sessionId, phase, @@ -258,10 +292,10 @@ export async function collectMainTranscriptEvents( startedAt, endedAt ); - // Not yet carried by any input to buildSessionSummaryEvent (session-summary.ts's own - // docstring defers it to this caller) — this is the full set of agent.usage.request - // records derived for this session so far, main- and agent-scoped alike. - summaryEvent.api_calls = Object.keys(state.openRequests).length; + // Main-thread requests only — the contract's api_calls excludes subagent requests. + summaryEvent.api_calls = Object.values(state.openRequests).filter( + (r) => r.scopeKind === 'main' + ).length; events.push(summaryEvent); } diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts index cfd6f257f..2eccc5cd5 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts @@ -1,8 +1,7 @@ /** * `agent.session.summary` builder. * - * Unlike `agent.usage.request`/`agent.subagent.usage`, this event is a running aggregate over an - * entire session. The orchestrator accumulates a {@link SessionSummaryAccumulator}, tracks + * The orchestrator accumulates a {@link SessionSummaryAccumulator}, tracks * `TranscriptParseState.branchCounts` via {@link updateBranchCounts}, and runs * `extractNamedInvocations()` to produce the {@link NamedInvocationCounts} this builder consumes. * This module only derives the final event shape from those inputs — it never reads a transcript. @@ -10,15 +9,10 @@ * `event_id`/`schema_version`/`client_version`/`codemie_cli_version` are stamped later, * daemon-side (`mapHookRecords()`); the output carries only an explicit `type`. * - * Field-shape notes: - * - `models_used` is the full `acc.models` count map, preserving counts `primary_model` discards. - * - `tool_calls`/`tool_errors` are flattened from `acc.toolCalls`'s `{ calls, errors }` shape into - * two flat maps, matching `buildSubagentUsageEvent` (`./subagent-usage.ts`). - * - `commands_in_order` is `Object.keys(named.commandInvocations)`. Upstream is a COUNT map, so no - * chronological order exists; the field name implies more than the data can deliver. - * - `title` has no known source and is always an empty string, never fabricated. - * - `api_calls` is omitted here: no input carries a request count. The orchestrator, which owns - * the full set of `agent.usage.request` records, merges it in afterward. + * `api_calls` is omitted here: no input carries a request count. The orchestrator, which owns + * the full set of `agent.usage.request` records, merges it in afterward. + * + * `title` has no identified source and is always `''`, never fabricated. */ import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; @@ -29,9 +23,8 @@ export type { NamedInvocationCounts }; export interface SessionSummaryAccumulator { models: Record; toolCalls: Record; - linesAdded: number; - linesRemoved: number; - filesChanged: Set; + toolResults: number; + filesEdited: Set; filesWritten: Set; compactionCount: number; } @@ -78,11 +71,18 @@ export function branchDominant(counts: Record): string { return maxKey(counts); } +/** Distinct normalised models, primary first, per the contract's `models` field. */ +function modelsArray(models: Record): string[] { + const primary = primaryModel(models); + const rest = Object.keys(models).filter((model) => model !== primary); + return primary ? [primary, ...rest] : rest; +} + /** * Build the `agent.session.summary` event payload. * - * `endedAt` is included as `ended_at` only when `phase === 'final'`; for `phase === 'incremental'` - * the key is omitted entirely (not merely `undefined`-valued). + * `startedAt`/`endedAt` are the transcript's own first/last line timestamps (never a hook's + * invocation time); `duration_ms` is `null`, not `0`, whenever either is missing or unparseable. */ export function buildSessionSummaryEvent( sessionId: string, @@ -91,40 +91,46 @@ export function buildSessionSummaryEvent( named: NamedInvocationCounts, branchCounts: Record, startedAt: string, - endedAt: string | undefined + endedAt: string ): Record { - const toolCalls: Record = {}; - const toolErrors: Record = {}; - for (const [tool, counts] of Object.entries(acc.toolCalls)) { - toolCalls[tool] = counts.calls; - toolErrors[tool] = counts.errors; - } + const diff = startedAt && endedAt ? Date.parse(endedAt) - Date.parse(startedAt) : NaN; + const durationMs = Number.isFinite(diff) && diff >= 0 ? diff : null; + + const filesChanged = new Set([...acc.filesEdited, ...acc.filesWritten]); + const toolTotals = Object.values(acc.toolCalls).reduce( + (totals, t) => ({ calls: totals.calls + t.calls, errors: totals.errors + t.errors }), + { calls: 0, errors: 0 } + ); - const event: Record = { + return { type: 'agent.session.summary', session_id: sessionId, - phase, - models_used: acc.models, + // The envelope `timestamp` the forwarder reads to decide which summary is latest + // (`summary_ts`) — the contract's own value for it, same as `ended_at`. Left unset when + // unknown so the forwarder's existing spool-time fallback applies instead of fabricating one. + timestamp: endedAt || undefined, + is_final: phase === 'final', + started_at: startedAt, + ended_at: endedAt || null, + duration_ms: durationMs, + models: modelsArray(acc.models), primary_model: primaryModel(acc.models), - tool_calls: toolCalls, - tool_errors: toolErrors, - skills_used: named.skillInvocations, - commands_in_order: Object.keys(named.commandInvocations), + tool_calls: toolTotals.calls, + tool_errors: toolTotals.errors, + tool_results: acc.toolResults, + tools: acc.toolCalls, + skills: named.skillInvocations, + agents: named.agentInvocations, + commands: Object.keys(named.commandInvocations), primary_command: maxKey(named.commandInvocations), - lines_added: acc.linesAdded, - lines_removed: acc.linesRemoved, - files_changed: Array.from(acc.filesChanged), - files_written: Array.from(acc.filesWritten), + lines_added: null, + lines_removed: null, + files_changed: filesChanged.size, + files_written: acc.filesWritten.size, + files_edited: acc.filesEdited.size, compaction_count: acc.compactionCount, branch_counts: branchCounts, branch_dominant: branchDominant(branchCounts), - started_at: startedAt, title: '', }; - - if (phase === 'final') { - event.ended_at = endedAt; - } - - return event; } From 17930675c189c5f9da5d8a3d97ba5563faef34b7 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Thu, 8 Oct 2026 15:22:31 +0300 Subject: [PATCH 29/35] refactor(proxy): resolve OTLP common fields in adapter base class Move git_branch, repo_remote, story_id and story_source resolution from the proxy-daemon forwarder into an abstract OtlpAgentAdapter that runs at hook time. The daemon env is frozen at spawn, so env-dependent fields such as SDLC_ANALYTICS_STORY_ID were read from the wrong process. - add OtlpAgentAdapter with a fixed processOtlpEvent flow and a generic ForwardDecision - inject forwardOtlpEventToSpool via OtlpAdapterDeps - move story-resolver to agents/core and add resolveStoryFor - make the forwarder pass the four fields through from the record - migrate ClaudeCodeOtlpPlugin to the base class - update tests and OTLP architecture docs Generated with AI Co-Authored-By: codemie-ai --- .ai-run/guides/architecture/architecture.md | 2 +- docs/ARCHITECTURE-OTLP-PLUGIN.md | 64 ++++--- src/agents/core/OtlpAgentAdapter.ts | 112 ++++++++++++ .../core/__tests__/OtlpAgentAdapter.test.ts | 162 ++++++++++++++++++ .../core}/__tests__/story-resolver.test.ts | 67 ++++++++ .../core}/story-resolver.ts | 47 ++++- src/agents/core/types.ts | 17 +- .../__tests__/claude-code-otlp.plugin.test.ts | 40 +++-- .../claude-code-otlp.plugin.ts | 75 +++----- .../claude-code-otlp.types.ts | 5 +- src/agents/registry.ts | 5 +- .../__tests__/hook.analytics-wiring.test.ts | 5 +- src/cli/commands/hook.ts | 3 +- .../otlp-spool/__tests__/forwarder.test.ts | 141 +++++---------- .../plugins/otlp-spool/forward-context.ts | 71 +------- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 52 +----- 16 files changed, 543 insertions(+), 325 deletions(-) create mode 100644 src/agents/core/OtlpAgentAdapter.ts create mode 100644 src/agents/core/__tests__/OtlpAgentAdapter.test.ts rename src/{providers/plugins/sso/proxy/plugins/otlp-spool => agents/core}/__tests__/story-resolver.test.ts (75%) rename src/{providers/plugins/sso/proxy/plugins/otlp-spool => agents/core}/story-resolver.ts (74%) diff --git a/.ai-run/guides/architecture/architecture.md b/.ai-run/guides/architecture/architecture.md index 54aed84c1..27cf9d429 100644 --- a/.ai-run/guides/architecture/architecture.md +++ b/.ai-run/guides/architecture/architecture.md @@ -267,4 +267,4 @@ FrameworkRegistry.get('langgraph') // src/frameworks/registry.ts | Core | `src/*/core/` | | Utils | `src/utils/` | | Tests | `tests/integration/`, `src/**/__tests__/` | -| OTLP ingestion adapters | `OtlpAgentAdapter` (`src/agents/core/types.ts`) is a second, non-chat plugin type registered the same way as `AgentAdapter` — one plugin per coding tool, ingesting that tool's native hook/telemetry events into the analytics pipeline. See `docs/ARCHITECTURE-OTLP-PLUGIN.md` for the dispatch pattern and how to add a new adapter (e.g. a future Cursor/other-tool adapter). | +| OTLP ingestion adapters | `OtlpAgentAdapter` (abstract base class, `src/agents/core/OtlpAgentAdapter.ts`) is a second, non-chat plugin type registered the same way as `AgentAdapter` — one plugin per coding tool, ingesting that tool's native hook/telemetry events into the analytics pipeline. See `docs/ARCHITECTURE-OTLP-PLUGIN.md` for the dispatch pattern and how to add a new adapter (e.g. a future Cursor/other-tool adapter). | diff --git a/docs/ARCHITECTURE-OTLP-PLUGIN.md b/docs/ARCHITECTURE-OTLP-PLUGIN.md index 5a9119b04..011849c29 100644 --- a/docs/ARCHITECTURE-OTLP-PLUGIN.md +++ b/docs/ARCHITECTURE-OTLP-PLUGIN.md @@ -1,21 +1,28 @@ # OTLP Hook-Event Plugin Pattern (`OtlpAgentAdapter`) -**Scope**: any plugin implementing `OtlpAgentAdapter` (`src/agents/core/types.ts`), under `src/agents/plugins//`. Today that's `claude-code-otlp` only. +**Scope**: any plugin extending the abstract `OtlpAgentAdapter` (`src/agents/core/OtlpAgentAdapter.ts`), under `src/agents/plugins//`. Today that's `claude-code-otlp` only. **Status**: Living doc — update when the dispatch pattern changes or a second adapter lands. ## 1. What this pattern is -An `OtlpAgentAdapter` is not a chat agent. It's the ingestion point for one coding tool's native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, usage, auth gating). Every adapter implements one method and feeds the same shared spool shape: +An `OtlpAgentAdapter` is not a chat agent. It's the ingestion point for one coding tool's native hook/event surface, turned into CodeMie's analytics pipeline (session summaries, usage, auth gating). `OtlpAgentAdapter` is an abstract base class: it owns the fixed `processOtlpEvent` flow and resolves agent-independent common fields at hook time; each adapter supplies only the tool-specific parts. ```ts -export interface OtlpAgentAdapter { - readonly name: string; - readonly type: AgentAdapterType.OTLP; - processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise; +export abstract class OtlpAgentAdapter> { + abstract readonly name: string; + readonly type = AgentAdapterType.OTLP; + async processOtlpEvent(raw: string, deps: OtlpAdapterDeps): Promise; // fixed flow + + protected abstract parseHookInput(raw: string): TInput | null; // null => ignore + protected abstract extractHookContext(input: TInput): OtlpHookContext; // { cwd, prompt? } + protected abstract isTracked(input: TInput): Promise; // allowlist gate + protected abstract evaluate(input: TInput): Promise>; + protected abstract agentCommonFields(): Promise>; } export interface OtlpAdapterDeps { ensureOtlpProxy: (agentName: string) => Promise; + forwardOtlpEventToSpool: (event: Record, agentName: string) => Promise; } export interface OtlpHookSpoolData { @@ -25,6 +32,8 @@ export interface OtlpHookSpoolData { } ``` +Spool forwarding is injected through `OtlpAdapterDeps` (supplied by `hook.ts`) rather than imported, so `core/` has no upward imports and the base class is testable with plain fakes. + Everything downstream of `processOtlpEvent` — spool, forwarder, analytics API — is already agent-agnostic and shared. Nothing in it is specific to any one tool's hook names or payload shape; that lives entirely inside each adapter. ## 2. Pipeline @@ -36,15 +45,16 @@ Everything downstream of `processOtlpEvent` — spool, forwarder, analytics API codemie hook --agent (src/cli/commands/hook.ts) │ looks up adapter via AgentRegistry.getAnalyticsAgent(name) ▼ -.processOtlpEvent(rawEvent, deps) - │ +OtlpAgentAdapter.processOtlpEvent(rawEvent, deps) (base class, fixed flow) + │ parseHookInput → isTracked → ensureOtlpProxy ▼ -evaluate(parsed) → ForwardDecision (§3: shape every adapter follows) +evaluate(input) → ForwardDecision (§3: shape every adapter follows) │ ┌────┴─────┐ block forward │ │ - log + forwardToSpool(payload) ──POST──► proxy daemon spool (OtlpHookSpoolData) + log + common fields merged, + forward each record ──POST──► proxy daemon spool (OtlpHookSpoolData) suppress │ (adapter- ▼ specific) otlp-spool/forwarder.ts (background, agent-agnostic) @@ -54,19 +64,23 @@ evaluate(parsed) → ForwardDecision (§3: shape every adapter follows) ``` - `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is fire-and-forget and shared: POSTs `{ agentName, timestamp, raw }` to the local proxy daemon, swallows every error. A dead daemon never blocks or fails the hook. -- `otlp-spool/forwarder.ts` is agent-agnostic by construction: it maps spooled records straight through to the analytics API payload and never calls into any adapter. Any agent-owned common field (platform, version, entrypoint, …) must already be baked into the event by the adapter before it reaches the spool (§5). +- `otlp-spool/forwarder.ts` only forwards: it maps spooled records to the analytics API payload and never calls into any adapter. It passes `story_id`, `story_source`, `git_branch`, `repo_remote` through from the record (`''` when absent) and no longer resolves story or git info itself. Credential and daemon-state fields (`user_email`, `developer_name`, `identity_source`, `codemie_project_name`, `codemie_cli_version`) are still stamped by the forwarder. +- **Common fields resolved by the base class at hook time** (the forwarder runs in a long-lived daemon whose env is frozen at spawn, so env-dependent fields were wrong there): `git_branch` (`detectGitBranch`), `repo_remote` (`detectGitRemoteRepo`), and `story_id` / `story_source` (`core/story-resolver.ts`, priority explicit env/config file -> marker -> branch -> mention; marker and mention only when `extractHookContext` supplies a `prompt`). Merge rule: a record's own non-empty string value wins, otherwise the resolved common value; `''` / `undefined` never clobber. `agentCommonFields()` overrides everything (platform, version, entrypoint, …). +- **Story granularity**: each hook subprocess resolves independently. A `Stop` after a prompt containing `story: ABC-1` gets the explicit / branch story, not the marker story. Carrying it across a session would need per-session state. +- **Known limitations**: (1) the explicit story tier reads `/.claude/analytics.local.json` by default (Claude-flavoured); `resolveStoryFor` takes `explicitConfigPath` so another adapter can override it. (2) The block path prints `JSON.stringify(decision)` to stdout, which is Claude Code's hook protocol; a tool with a different block protocol must make that step overridable (e.g. an `emitBlock()` method). +- **Rollout**: the daemon outlives CLI upgrades. An old daemon recomputes `story_id`, `story_source` and `repo_remote` from its own env and overwrites the hook's values until restarted (`git_branch` is kept). Restart the proxy daemon after upgrading. - Wiring which native hooks/events call `codemie hook --agent ` is entirely tool-specific — see §6 for the Claude Code connector; a different tool has its own. ## 3. The dispatch pattern every `evaluate()` follows -This is a convention each adapter implements for itself, not a shared type from `core/types.ts` — `OtlpAgentAdapter`'s only contractual method is `processOtlpEvent`. Each adapter is free to define its own `ForwardDecision`-equivalent and field names to match its own tool's native event shape; keep the shape below, not the literal field/type names from the `claude-code-otlp` example. +`ForwardDecision` is a shared generic type exported from `OtlpAgentAdapter.ts`; the adapter binds `TBlockOutput` to its tool's block output (Claude: `UserPromptSubmitHookSpecificOutput`, aliased as `ClaudeForwardDecision`). `evaluate(input)` receives the already-parsed, typed `TInput`; the dispatch convention below is how adapters implement it. ### 3.1 Two outcomes — forward or block ```ts -export type ForwardDecision = +export type ForwardDecision> = | { decision: 'forward'; payload: Record[] } - | { decision: 'block'; reason: string; /* ...whatever this tool needs to suppress/respond to the native event... */ }; + | { decision: 'block'; reason: string; hookSpecificOutput: TBlockOutput }; ``` `forward` carries the full list of records to push to the spool (the original parsed event, plus zero or more derived analytics events). `block` stops the hook and logs a reason; any extra fields on `block` (`claude-code-otlp` uses `hookSpecificOutput`, mirroring Claude Code's own hook-output JSON schema) are specific to that tool's blocking mechanism, not to this pattern — a different tool may have no `block` case at all, or a differently-shaped one. @@ -74,7 +88,7 @@ export type ForwardDecision = ### 3.2 One handler per event name — no `switch`, no shared merge step ```ts -private async evaluate(parsed: Record): Promise { +protected async evaluate(parsed: TInput): Promise> { // early-return / continuation-key checks here are tool-specific — only add // one if this tool's events need it (claude-code-otlp keys continuation on // its own 'session_id' field; a different tool may have no such field) @@ -97,24 +111,24 @@ Each handler returns its own full `ForwardDecision`. The trailing fallthrough re ### 3.4 One place writes to the spool -`forwardToSpool()` is the only call site for `forwardOtlpEventToSpool()`, called once from `processOtlpEvent()` after `evaluate()` resolves (it loops over `ForwardDecision.payload` and forwards each record). No handler forwards anything itself. This single-chokepoint property guarantees every event reaches the spool exactly once, in order, under the adapter's registered name. +The base class's `processOtlpEvent` is the only call site of `deps.forwardOtlpEventToSpool`: after `evaluate()` resolves it computes the common fields once, merges them into each `payload` record and forwards every record, **not awaited** (fire-and-forget; it never throws, and awaiting would put daemon latency on the hook path). Handlers never forward anything themselves, so every event reaches the spool exactly once, in order, under the adapter's registered name. ### 3.5 Proxy readiness is the adapter's own responsibility -Whether and when to call `deps.ensureOtlpProxy()`, and whether to add any further gate (e.g. an auth check, a tracked-project check), is a decision each adapter makes for itself based on what its events actually need — there's no required shape here. `claude-code-otlp` gates `ensureOtlpProxy()` on its tracked-project check in `processOtlpEvent()` but, once past that, calls it for every event regardless of which handler runs (since every `forward` decision needs the daemon up to reach the spool) — it does not gate per-handler. It additionally gates an SSO auth check inside `onUserPromptSubmit` only, because that's the one handler whose job is to enforce it. A different tool may not need a tracked-project concept at all, may not need proxy readiness for every event, or may need a different gate entirely — don't carry `claude-code-otlp`'s specific gating choices into a new adapter, just the principle that each adapter decides this for itself. +The base flow calls `isTracked()` (abstract) before `deps.ensureOtlpProxy()`, so a subclass cannot skip the allowlist gate (the INVARIANT comment in the base flow). Past that, `ensureOtlpProxy()` runs for every event regardless of which handler runs (every `forward` decision needs the daemon up to reach the spool). Further gates (e.g. an auth check) remain the adapter's own decision. It additionally gates an SSO auth check inside `onUserPromptSubmit` only, because that's the one handler whose job is to enforce it. A different tool may not need a tracked-project concept at all, may not need proxy readiness for every event, or may need a different gate entirely — don't carry `claude-code-otlp`'s specific gating choices into a new adapter, just the principle that each adapter decides this for itself. ## 4. Adding a new event to an existing adapter -1. Write one handler method, `(parsed: Record) => Promise`. +1. Write one handler method, `(input) => Promise>`. 2. Add exactly one `if` branch in `evaluate()` dispatching to it. No `switch`, don't touch the trailing fallthrough. -3. Never forward from inside the handler — return the `ForwardDecision`; the single `forwardToSpool()` call in `processOtlpEvent()` sends it. +3. Never forward from inside the handler — return the `ForwardDecision`; the base class's `processOtlpEvent()` sends it. 4. Add a test for the new branch, following the adapter's existing test structure. 5. Update that adapter's own event-surface table/notes (see §6 for the current example). ## 5. Adding a new `OtlpAgentAdapter` for a different tool -1. Create `src/agents/plugins//`, implement `OtlpAgentAdapter` following §3. Enrich events with any agent-owned common fields (platform, version, entrypoint, …) **inside the adapter's own hook-time process**, not via a callback from the forwarder — the forwarder runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, so resolving agent/tool state there would reflect the daemon's environment, not the invocation that produced the event. If resolving a field is expensive (subprocess spawn, network call), back it with a small file cache under `getCodemiePath()` — the hook-time process is fresh per event, so in-memory memoization buys nothing (see `client-version-cache.ts` for the pattern). -2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — same name passed to `forwardOtlpEventToSpool(event, name)` and matched by `AgentRegistry.getAnalyticsAgent(name)` in `hook.ts`. +1. Create `src/agents/plugins//` with a class `extends OtlpAgentAdapter` and implement `parseHookInput`, `extractHookContext`, `isTracked`, `evaluate` (following §3) and `agentCommonFields`. Agent-owned common fields (platform, version, entrypoint, …) go in `agentCommonFields()`, resolved **inside the adapter's own hook-time process**, not via a callback from the forwarder — the forwarder runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, so resolving state there would reflect the daemon's environment. Git branch/remote and story are already resolved by the base class. If resolving a field is expensive (subprocess spawn, network call), back it with a small file cache under `getCodemiePath()` — the hook-time process is fresh per event, so in-memory memoization buys nothing (see `client-version-cache.ts`). +2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — same name the base class passes to `forwardOtlpEventToSpool(event, name)` and matched by `AgentRegistry.getAnalyticsAgent(name)` in `hook.ts`. 3. Write a connector wiring the tool's native hooks/events to `codemie hook --agent ` (see `src/cli/commands/proxy/connectors/claude-code-otlp.ts` for the example — hook names, settings format, and env vars are tool-specific). 4. Everything from `forwardOtlpEventToSpool` onward is already shared — no changes needed there as long as 1-3 hold. @@ -124,11 +138,13 @@ Whether and when to call `deps.ensureOtlpProxy()`, and whether to add any furthe | File | Role | |---|---| -| `claude-code-otlp.plugin.ts` | `evaluate()` dispatch, per-event handlers, hook-time common-field enrichment, allowlist gate + daemon start | +| `claude-code-otlp.plugin.ts` | `extends OtlpAgentAdapter`: input parsing, `evaluate()` dispatch, per-event handlers, `agentCommonFields()`, allowlist wiring | +| `src/agents/core/OtlpAgentAdapter.ts` | Abstract base: fixed flow, allowlist-before-daemon INVARIANT, generic `ForwardDecision`, hook-time git/story common fields, fire-and-forget spool forwarding | +| `src/agents/core/story-resolver.ts` | Story tiers and `resolveStoryFor({ cwd, branch, prompt, explicitConfigPath })` | | `client-version-cache.ts` | TTL file cache around `claude --version`, backing the `client_version` common field | -| `claude-code-otlp.types.ts` | `ForwardDecision` | +| `claude-code-otlp.types.ts` | `ClaudeForwardDecision` alias, `isClaudeCodeHookInput` | | `claude-code-otlp.constants.ts` | `CLAUDE_CODE_OTLP_AGENT_NAME` — the registered adapter name | -| `claude-code-otlp.allowlist.ts` | Per-project allowlist gating (`isProjectTracked`/`readAllowlistState`) — see the INVARIANT comment in `processOtlpEvent`: an untracked project must never reach the daemon spool, since the daemon has no allowlist of its own | +| `claude-code-otlp.allowlist.ts` | Per-project allowlist gating (`isProjectTracked`/`readAllowlistState`) — see the INVARIANT comment in the base `processOtlpEvent`: an untracked project must never reach the daemon spool, since the daemon has no allowlist of its own | | `transcript/orchestrator.ts` | `collectMainTranscriptEvents`/`collectSubagentTranscriptEvents` — transcript-derived analytics events | | `transcript/subagent-usage.ts` | `findSubagentFiles` — discovers every subagent transcript for a session | | `src/cli/commands/proxy/connectors/claude-code-otlp.ts` | Wires Claude Code's `.claude/settings.json` hooks to `codemie hook --agent claude-code-otlp` | diff --git a/src/agents/core/OtlpAgentAdapter.ts b/src/agents/core/OtlpAgentAdapter.ts new file mode 100644 index 000000000..0ee90dedf --- /dev/null +++ b/src/agents/core/OtlpAgentAdapter.ts @@ -0,0 +1,112 @@ +import { logger } from '@/utils/logger.js'; +import { detectGitBranch, detectGitRemoteRepo } from '@/utils/processes.js'; +import { AgentAdapterType, type OtlpAdapterDeps } from './types.js'; +import { resolveStoryFor } from './story-resolver.js'; + +export type ForwardDecision> = + | { decision: 'forward'; payload: Record[] } + | { decision: 'block'; reason: string; hookSpecificOutput: TBlockOutput }; + +export interface OtlpHookContext { + cwd: string; + /** Only for prompt-submit style events; enables the marker/mention story tiers. */ + prompt?: string; +} + +/** Fields resolved by the base class at hook time, shared by every agent. */ +const COMMON_FIELD_KEYS = ['git_branch', 'repo_remote', 'story_id', 'story_source'] as const; + +/** + * Base class for adapters that ingest a coding tool's native hook events into the + * analytics spool. Owns the fixed flow (parse, allowlist gate, daemon start, evaluate, + * common-field enrichment, spool forward); subclasses supply the tool-specific parts. + * + * Common fields are resolved here, in the short-lived hook process, because the + * forwarder runs in a long-lived daemon whose environment is frozen at spawn time. + */ +export abstract class OtlpAgentAdapter> { + abstract readonly name: string; + readonly type = AgentAdapterType.OTLP; + + public async processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise { + const { ensureOtlpProxy, forwardOtlpEventToSpool } = deps; + // JSON.parse failures intentionally propagate to the caller. + const hookInput = this.parseHookInput(rawHookInput); + if (hookInput === null) { + logger.debug(`[${this.name}] hook payload has an unknown shape, ignoring`); + return; + } + + // INVARIANT - do not weaken. An untracked project must produce NO hooks data + // in the daemon spool, must not start the daemon, and must not run the SSO + // check. The daemon has no allowlist of its own. Its completeness gate + // (`otlp-spool/completeness-gate.ts`) only sends sessions that have hooks + // data, and skips sessions that only have OTEL data. Forwarding a hook event + // for an untracked project would make the session sendable and leak its + // data to the backend. + if (!(await this.isTracked(hookInput))) { + logger.debug(`[${this.name}] project not in analytics allowlist, ignoring hook event`); + return; + } + + await ensureOtlpProxy(this.name); + + const decision = await this.evaluate(hookInput); + if (decision.decision === 'block') { + logger.error(`[${this.name}] Blocking prompt: ${decision.reason}`); + // Claude Code hook protocol: a blocking decision is returned as JSON on stdout. + // A new adapter whose tool uses a different block protocol must make this step + // overridable (e.g. an `emitBlock()` method) instead of reusing it as is. + console.log(JSON.stringify(decision)); + return; + } + + const context = this.extractHookContext(hookInput); + const commonFields = await this.resolveCommonFields(context); + const agentFields = await this.resolveAgentCommonFields(); + + for (const hookEvent of decision.payload) { + // Intentionally not awaited: forwardOtlpEventToSpool never throws, + // and awaiting would put daemon latency on the hook path. + void forwardOtlpEventToSpool({ ...this.mergeCommonFields(hookEvent, commonFields), ...agentFields }, this.name); + } + } + + /** Returns null to ignore a payload of an unknown shape. May throw on malformed JSON. */ + protected abstract parseHookInput(raw: string): TInput | null; + protected abstract extractHookContext(input: TInput): OtlpHookContext; + /** Allowlist gate; evaluated before the daemon is started. */ + protected abstract isTracked(input: TInput): Promise; + protected abstract evaluate(input: TInput): Promise>; + /** Agent-owned fields (platform, version, entrypoint, ...); override everything else. */ + protected abstract resolveAgentCommonFields(): Promise>; + + private async resolveCommonFields({ cwd, prompt }: OtlpHookContext): Promise> { + const empty = { git_branch: '', repo_remote: '', story_id: '', story_source: '' }; + try { + const [branch, remote] = cwd + ? await Promise.all([detectGitBranch(cwd), detectGitRemoteRepo(cwd)]) + : [undefined, undefined]; + const story = await resolveStoryFor({ cwd, branch, prompt }); + return { + git_branch: branch ?? '', + repo_remote: remote ?? '', + story_id: story?.storyId ?? '', + story_source: story?.storySource ?? '', + }; + } catch (error) { + logger.debug(`[${this.name}] common field resolution failed: ${error instanceof Error ? error.message : String(error)}`); + return empty; + } + } + + /** A hook event's own non-empty string value wins; otherwise the base value is used. */ + private mergeCommonFields(hookEvent: Record, commonFields: Record): Record { + const merged = { ...hookEvent }; + for (const key of COMMON_FIELD_KEYS) { + const own = hookEvent[key]; + merged[key] = typeof own === 'string' && own.length > 0 ? own : commonFields[key]; + } + return merged; + } +} diff --git a/src/agents/core/__tests__/OtlpAgentAdapter.test.ts b/src/agents/core/__tests__/OtlpAgentAdapter.test.ts new file mode 100644 index 000000000..4fded3b31 --- /dev/null +++ b/src/agents/core/__tests__/OtlpAgentAdapter.test.ts @@ -0,0 +1,162 @@ +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; + +const detectGitBranchMock = vi.fn(); +const detectGitRemoteRepoMock = vi.fn(); + +vi.mock('@/utils/processes.js', () => ({ + detectGitBranch: detectGitBranchMock, + detectGitRemoteRepo: detectGitRemoteRepoMock, +})); +vi.mock('@/utils/logger.js', () => ({ + logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, +})); + +const { OtlpAgentAdapter } = await import('../OtlpAgentAdapter.js'); +type Decision = import('../OtlpAgentAdapter.js').ForwardDecision; +type Context = import('../OtlpAgentAdapter.js').OtlpHookContext; + +interface FakeInput { + cwd: string; + prompt?: string; + payload: Record[]; + block?: boolean; +} + +class FakeAdapter extends OtlpAgentAdapter { + readonly name = 'fake'; + tracked = true; + + protected parseHookInput(raw: string): FakeInput | null { + const parsed = JSON.parse(raw) as Partial; + return typeof parsed.cwd === 'string' ? (parsed as FakeInput) : null; + } + protected extractHookContext(input: FakeInput): Context { + return { cwd: input.cwd, prompt: input.prompt }; + } + protected async isTracked(): Promise { + return this.tracked; + } + protected async evaluate(input: FakeInput): Promise { + if (input.block) { + return { decision: 'block', reason: 'nope', hookSpecificOutput: { x: 1 } }; + } + return { decision: 'forward', payload: input.payload }; + } + protected async resolveAgentCommonFields(): Promise> { + return { platform: 'fake-platform' }; + } +} + +describe('OtlpAgentAdapter.processOtlpEvent', () => { + const ensureOtlpProxy = vi.fn(async () => {}); + const forwardToSpool = vi.fn(async () => {}); + const deps = { ensureOtlpProxy, forwardOtlpEventToSpool: forwardToSpool }; + let adapter: FakeAdapter; + + const run = (input: Partial = {}) => + adapter.processOtlpEvent(JSON.stringify({ cwd: '/repo', payload: [{ a: 1 }], ...input }), deps); + const forwarded = () => forwardToSpool.mock.calls.map(([record]) => record as Record); + + beforeEach(() => { + vi.clearAllMocks(); + adapter = new FakeAdapter(); + detectGitBranchMock.mockResolvedValue('main'); + detectGitRemoteRepoMock.mockResolvedValue('org/repo'); + delete process.env['SDLC_ANALYTICS_STORY_ID']; + }); + + afterEach(() => { + delete process.env['SDLC_ANALYTICS_STORY_ID']; + vi.restoreAllMocks(); + }); + + it('gates on isTracked before ensureOtlpProxy and forwards nothing for untracked projects', async () => { + adapter.tracked = false; + await run(); + expect(ensureOtlpProxy).not.toHaveBeenCalled(); + expect(forwardToSpool).not.toHaveBeenCalled(); + }); + + it('ignores a payload parseHookInput rejects', async () => { + await adapter.processOtlpEvent(JSON.stringify({ nothing: true }), deps); + expect(ensureOtlpProxy).not.toHaveBeenCalled(); + expect(forwardToSpool).not.toHaveBeenCalled(); + }); + + it('still throws on malformed JSON', async () => { + await expect(adapter.processOtlpEvent('not json', deps)).rejects.toThrow(); + }); + + it('ensures the proxy before forwarding', async () => { + await run(); + expect(ensureOtlpProxy).toHaveBeenCalledWith('fake'); + expect(ensureOtlpProxy.mock.invocationCallOrder[0]).toBeLessThan(forwardToSpool.mock.invocationCallOrder[0]); + }); + + it('prints the decision and forwards nothing on the block path', async () => { + const log = vi.spyOn(console, 'log').mockImplementation(() => {}); + await run({ block: true }); + expect(log).toHaveBeenCalledWith(JSON.stringify({ decision: 'block', reason: 'nope', hookSpecificOutput: { x: 1 } })); + expect(forwardToSpool).not.toHaveBeenCalled(); + }); + + it('merges the four common fields and agent fields into every record', async () => { + await run({ payload: [{ a: 1 }, { a: 2 }] }); + expect(detectGitBranchMock).toHaveBeenCalledWith('/repo'); + expect(detectGitRemoteRepoMock).toHaveBeenCalledWith('/repo'); + expect(forwarded()).toEqual([ + { a: 1, git_branch: 'main', repo_remote: 'org/repo', story_id: '', story_source: '', platform: 'fake-platform' }, + { a: 2, git_branch: 'main', repo_remote: 'org/repo', story_id: '', story_source: '', platform: 'fake-platform' }, + ]); + expect(forwardToSpool.mock.calls.map(([, name]) => name)).toEqual(['fake', 'fake']); + }); + + it("keeps a record's own non-empty value and does not let undefined / '' clobber the base value", async () => { + await run({ + payload: [ + { git_branch: 'own-branch', repo_remote: 'own/remote' }, + { git_branch: '', repo_remote: undefined }, + ], + }); + const [own, empty] = forwarded(); + expect(own).toMatchObject({ git_branch: 'own-branch', repo_remote: 'own/remote' }); + expect(empty).toMatchObject({ git_branch: 'main', repo_remote: 'org/repo' }); + }); + + it('lets agent fields override everything else', async () => { + await run({ payload: [{ platform: 'record-platform' }] }); + expect(forwarded()[0]['platform']).toBe('fake-platform'); + }); + + it('reads the explicit story from the hook process env', async () => { + process.env['SDLC_ANALYTICS_STORY_ID'] = 'ABC-1'; + await run(); + expect(forwarded()[0]).toMatchObject({ story_id: 'ABC-1', story_source: 'explicit' }); + }); + + it('resolves marker story from the prompt, above the branch', async () => { + detectGitBranchMock.mockResolvedValue('feature/BR-2'); + await run({ prompt: 'story: MK-3' }); + expect(forwarded()[0]).toMatchObject({ story_id: 'MK-3', story_source: 'marker' }); + }); + + it('falls back to the branch story when there is no prompt', async () => { + detectGitBranchMock.mockResolvedValue('feature/BR-2'); + await run(); + expect(forwarded()[0]).toMatchObject({ story_id: 'BR-2', story_source: 'branch' }); + }); + + it("stamps '' on a detached HEAD, even when the record carries an empty value", async () => { + detectGitBranchMock.mockResolvedValue(undefined); + await run({ payload: [{ git_branch: '' }] }); + expect(forwarded()[0]['git_branch']).toBe(''); + }); + + it('does not wait for forwardToSpool (fire-and-forget)', async () => { + let resolveLate: () => void = () => {}; + forwardToSpool.mockImplementation(() => new Promise((resolve) => { resolveLate = resolve; })); + await run({ payload: [{ a: 1 }, { a: 2 }] }); + expect(forwardToSpool).toHaveBeenCalledTimes(2); + resolveLate(); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts b/src/agents/core/__tests__/story-resolver.test.ts similarity index 75% rename from src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts rename to src/agents/core/__tests__/story-resolver.test.ts index 61c7948bf..280e69ff9 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/story-resolver.test.ts +++ b/src/agents/core/__tests__/story-resolver.test.ts @@ -215,4 +215,71 @@ describe('story-resolver', () => { expect(second).toEqual({ storyId: 'ABC-42', storySource: 'mention' }); }); }); + + describe('resolveStoryFor', () => { + it('prefers explicit over marker, branch and mention', async () => { + const { resolveStoryFor } = await import('../story-resolver.js'); + process.env[ENV_KEY] = 'EXP-1'; + + const result = await resolveStoryFor({ + cwd: '/nonexistent', + branch: 'feature/BR-2', + prompt: 'story: MK-3 and also ME-4', + }); + + expect(result).toEqual({ storyId: 'EXP-1', storySource: 'explicit' }); + }); + + it('prefers marker over branch and mention', async () => { + const { resolveStoryFor } = await import('../story-resolver.js'); + + const result = await resolveStoryFor({ + cwd: '/nonexistent', + branch: 'feature/BR-2', + prompt: 'story: MK-3 and also ME-4', + }); + + expect(result).toEqual({ storyId: 'MK-3', storySource: 'marker' }); + }); + + it('prefers branch over mention', async () => { + const { resolveStoryFor } = await import('../story-resolver.js'); + + const result = await resolveStoryFor({ cwd: '/nonexistent', branch: 'feature/BR-2', prompt: 'look at ME-4' }); + + expect(result).toEqual({ storyId: 'BR-2', storySource: 'branch' }); + }); + + it('falls back to mention when nothing else resolves', async () => { + const { resolveStoryFor } = await import('../story-resolver.js'); + + const result = await resolveStoryFor({ cwd: '/nonexistent', branch: 'main', prompt: 'look at ME-4' }); + + expect(result).toEqual({ storyId: 'ME-4', storySource: 'mention' }); + }); + + it('ignores marker and mention tiers without a prompt', async () => { + const { resolveStoryFor } = await import('../story-resolver.js'); + + expect(await resolveStoryFor({ cwd: '/nonexistent', branch: 'main' })).toBeNull(); + expect(await resolveStoryFor({ cwd: '/nonexistent', branch: 'feature/BR-2' })).toEqual({ + storyId: 'BR-2', + storySource: 'branch', + }); + }); + + it('reads the explicit story from a configurable file path', async () => { + const { resolveStoryFor } = await import('../story-resolver.js'); + const cwd = await makeTempProjectDir(); + tempDirs.push(cwd); + await mkdir(join(cwd, '.other'), { recursive: true }); + await writeFile(join(cwd, '.other', 'story.json'), JSON.stringify({ storyId: 'CFG-9' }), 'utf-8'); + + const result = await resolveStoryFor({ cwd, explicitConfigPath: ['.other', 'story.json'] }); + const withDefault = await resolveStoryFor({ cwd }); + + expect(result).toEqual({ storyId: 'CFG-9', storySource: 'explicit' }); + expect(withDefault).toBeNull(); + }); + }); }); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts b/src/agents/core/story-resolver.ts similarity index 74% rename from src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts rename to src/agents/core/story-resolver.ts index 13ddbd812..608cb4734 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/story-resolver.ts +++ b/src/agents/core/story-resolver.ts @@ -45,13 +45,21 @@ export interface MentionStoryResult { */ const MARKER_RE = /(?:story|ticket)\s*[:#]?\s*([A-Za-z][A-Za-z0-9]+-\d+)/i; +/** + * Known limitation: the default config location is Claude-flavoured + * (`/.claude/analytics.local.json`). It is a parameter so another adapter + * can point at its own file without touching this module. The env var name + * (`SDLC_ANALYTICS_STORY_ID`) stays shared. + */ +export const DEFAULT_EXPLICIT_CONFIG_PATH: readonly string[] = ['.claude', 'analytics.local.json']; + interface AnalyticsLocalConfig { storyId?: unknown; } /** * Explicit story-id tier: `SDLC_ANALYTICS_STORY_ID` env var first, falling - * back to the `storyId` field of `/.claude/analytics.local.json`. + * back to the `storyId` field of `/` (default `.claude/analytics.local.json`). * Read-only — this never writes that file. Swallows every failure (missing * file, malformed JSON, permission error) and resolves to `null` instead of * throwing. The resolved `storyId` is taken verbatim from its source (env or @@ -59,14 +67,17 @@ interface AnalyticsLocalConfig { * tier: a value a user/config explicitly supplied is already exact, whereas * free text scanned by a case-insensitive regex needs normalizing. */ -export async function resolveExplicitStory(cwd: string): Promise { +export async function resolveExplicitStory( + cwd: string, + explicitConfigPath: readonly string[] = DEFAULT_EXPLICIT_CONFIG_PATH +): Promise { const envStoryId = process.env['SDLC_ANALYTICS_STORY_ID']; if (envStoryId) { return { storyId: envStoryId, storySource: 'explicit' }; } try { - const filePath = join(cwd, '.claude', 'analytics.local.json'); + const filePath = join(cwd, ...explicitConfigPath); const content = await readFile(filePath, 'utf-8'); const parsed = JSON.parse(content) as AnalyticsLocalConfig; if (typeof parsed.storyId === 'string' && parsed.storyId.length > 0) { @@ -139,3 +150,33 @@ export function resolveMentionStory(promptText: string): MentionStoryResult | nu return { storyId: matches[0].toUpperCase(), storySource: 'mention' }; } + +export type StoryResult = ExplicitStoryResult | BranchStoryResult | MarkerStoryResult | MentionStoryResult; + +export interface ResolveStoryInput { + cwd: string; + /** Current git branch; empty/undefined skips the branch tier. */ + branch?: string; + /** Prompt text; marker and mention tiers apply only when provided. */ + prompt?: string; + explicitConfigPath?: readonly string[]; +} + +/** + * Effective story for one hook invocation. Priority: explicit -> marker -> + * branch -> mention. Marker and mention only apply when `prompt` is provided. + * Runs in the hook process, so the explicit tier sees the invoking shell's env. + */ +export async function resolveStoryFor({ + cwd, + branch, + prompt, + explicitConfigPath, +}: ResolveStoryInput): Promise { + return ( + (await resolveExplicitStory(cwd, explicitConfigPath)) ?? + (prompt ? resolveMarkerStory(prompt) : null) ?? + resolveBranchStory(branch ?? '') ?? + (prompt ? resolveMentionStory(prompt) : null) + ); +} diff --git a/src/agents/core/types.ts b/src/agents/core/types.ts index 5876eb9c7..e4ab58c5e 100644 --- a/src/agents/core/types.ts +++ b/src/agents/core/types.ts @@ -731,23 +731,10 @@ export enum AgentAdapterType { OTLP, } -export interface OtlpAgentAdapter { - readonly name: string; - readonly type: AgentAdapterType.OTLP; - - /** - * Handles one hook event. The adapter owns the decision of whether the OTLP - * daemon is needed: it MUST call `deps.ensureProxy()` before forwarding - * anything to the daemon and MAY skip it for events it will not forward. - * - * INVARIANT: events from untracked projects must never reach the daemon - * spool. - */ - processOtlpEvent(rawHookInput: string, deps: OtlpAdapterDeps): Promise; -} - export interface OtlpAdapterDeps { ensureOtlpProxy: (agentName: string) => Promise; + /** Fire-and-forget spool write; never throws. */ + forwardOtlpEventToSpool: (event: Record, agentName: string) => Promise; } /** diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index f5392fe25..7ef5e846f 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -4,7 +4,10 @@ vi.mock('../claude-code-otlp.allowlist.js', () => ({ readAllowlistState: vi.fn(async () => ({ kind: 'valid', paths: ['/proj'] })), isProjectTracked: vi.fn(), })); -vi.mock('../../utils.js', () => ({ forwardOtlpEventToSpool: vi.fn() })); +vi.mock('@/utils/processes.js', () => ({ + detectGitBranch: vi.fn(async () => 'main'), + detectGitRemoteRepo: vi.fn(async () => 'org/repo'), +})); vi.mock('@/providers/plugins/sso/sso.auth-gate.js', () => ({ ensureCodeMieSsoAuth: vi.fn() })); vi.mock('@/utils/config.js', () => ({ ConfigLoader: { load: vi.fn(async () => ({})) } })); vi.mock('@/utils/logger.js', () => ({ @@ -30,7 +33,6 @@ vi.mock('../transcript/subagent-usage.js', () => ({ })); import { isProjectTracked } from '../claude-code-otlp.allowlist.js'; -import { forwardOtlpEventToSpool } from '../../utils.js'; import { ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; // Deferred past the mock-backing consts above: a static import of the plugin would be @@ -38,6 +40,8 @@ import { ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; // transcript/orchestrator.js mock factory, which closes over those consts. const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); +const forwardOtlpEventToSpool = vi.fn(async (_event: Record, _agentName: string) => {}); + const event = (name: string) => JSON.stringify({ session_id: 's', transcript_path: '', cwd: '/x', hook_event_name: name }); describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { @@ -53,7 +57,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { vi.mocked(isProjectTracked).mockResolvedValue(false); const log = vi.spyOn(console, 'log').mockImplementation(() => {}); for (const name of ['SessionStart', 'UserPromptSubmit', 'Stop']) { - await plugin.processOtlpEvent(event(name), { ensureOtlpProxy }); + await plugin.processOtlpEvent(event(name), { ensureOtlpProxy, forwardOtlpEventToSpool }); } expect(ensureOtlpProxy).not.toHaveBeenCalled(); expect(ensureCodeMieSsoAuth).not.toHaveBeenCalled(); @@ -64,7 +68,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent', () => { it('ensures the proxy before forwarding for tracked projects', async () => { vi.mocked(isProjectTracked).mockResolvedValue(true); - await plugin.processOtlpEvent(event('SessionStart'), { ensureOtlpProxy }); + await plugin.processOtlpEvent(event('SessionStart'), { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(ensureOtlpProxy).toHaveBeenCalledTimes(1); expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(1); expect(vi.mocked(ensureOtlpProxy).mock.invocationCallOrder[0]).toBeLessThan( @@ -109,7 +113,7 @@ describe('ClaudeCodeOtlpPlugin hook-time enrichment', () => { process.env.CLAUDE_CODE_ENTRYPOINT = 'cli'; const rawEvent = JSON.stringify(hookEvent()); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(forwardOtlpEventToSpool).toHaveBeenCalledTimes(1); const [forwarded] = vi.mocked(forwardOtlpEventToSpool).mock.calls[0]; @@ -118,6 +122,14 @@ describe('ClaudeCodeOtlpPlugin hook-time enrichment', () => { expect(forwarded.client_version).toBe('2.1.23'); }); + it('stamps git_branch and repo_remote resolved at hook time', async () => { + await plugin.processOtlpEvent(JSON.stringify(hookEvent()), { ensureOtlpProxy, forwardOtlpEventToSpool }); + + const [forwarded] = vi.mocked(forwardOtlpEventToSpool).mock.calls[0]; + expect(forwarded.git_branch).toBe('main'); + expect(forwarded.repo_remote).toBe('org/repo'); + }); + it('preserves agent_id/agent_type already present on the raw event (SubagentStop)', async () => { const rawEvent = JSON.stringify( hookEvent({ @@ -128,7 +140,7 @@ describe('ClaudeCodeOtlpPlugin hook-time enrichment', () => { }) ); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); const [forwarded] = vi.mocked(forwardOtlpEventToSpool).mock.calls[0]; expect(forwarded.agent_id).toBe('sub-1'); @@ -174,7 +186,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { const parsedEvent = hookEvent({ hook_event_name: hookEventName }); const rawEvent = JSON.stringify(parsedEvent); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(collectMainTranscriptEventsMock).toHaveBeenCalledWith( 'sid-1', @@ -193,7 +205,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { const plugin = new ClaudeCodeOtlpPlugin(); const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'PostToolUse' })); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(collectMainTranscriptEventsMock).not.toHaveBeenCalled(); }); @@ -211,7 +223,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { }) ); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith('sid-1', { agentId: 'sub-1', @@ -231,7 +243,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { }) ); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( 'sid-1', @@ -244,7 +256,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { const plugin = new ClaudeCodeOtlpPlugin(); const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SubagentStop' })); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(collectSubagentTranscriptEventsMock).not.toHaveBeenCalled(); }); @@ -258,7 +270,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { const plugin = new ClaudeCodeOtlpPlugin(); const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SessionEnd' })); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(findSubagentFilesMock).toHaveBeenCalledWith('/tmp/transcript.jsonl'); expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledTimes(2); @@ -278,7 +290,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { const parsedEvent = hookEvent({ hook_event_name: 'Stop' }); const rawEvent = JSON.stringify(parsedEvent); - await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy }); + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); // collectMainTranscriptEvents/collectSubagentTranscriptEvents never call forwardOtlpEventToSpool // themselves (they are mocked here to just return data) — every event that reaches the spool @@ -295,6 +307,6 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); - await expect(plugin.processOtlpEvent('not json', { ensureOtlpProxy })).rejects.toThrow(); + await expect(plugin.processOtlpEvent('not json', { ensureOtlpProxy, forwardOtlpEventToSpool })).rejects.toThrow(); }); }); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index c74e9ca7d..7bccbb325 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -8,14 +8,13 @@ import type { StopHookInput, SubagentStopHookInput, UserPromptSubmitHookInput, + UserPromptSubmitHookSpecificOutput, } from '@anthropic-ai/claude-agent-sdk'; import { AuthGateResult, ensureCodeMieSsoAuth } from '@/providers/plugins/sso/sso.auth-gate.js'; -import { logger } from '@/utils/logger.js'; import { ConfigLoader } from '@/utils/config.js'; -import { AgentAdapterType, OtlpAdapterDeps, OtlpAgentAdapter } from '@/agents/core/types.js'; +import { OtlpAgentAdapter, type OtlpHookContext } from '@/agents/core/OtlpAgentAdapter.js'; import { CLAUDE_CODE_OTLP_AGENT_NAME } from './claude-code-otlp.constants.js'; -import { ForwardDecision, isClaudeCodeHookInput } from './claude-code-otlp.types.js'; -import { forwardOtlpEventToSpool } from '../utils.js'; +import { type ClaudeForwardDecision, isClaudeCodeHookInput } from './claude-code-otlp.types.js'; import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowlist.js'; import { collectMainTranscriptEvents, @@ -25,52 +24,35 @@ import { import { findSubagentFiles } from './transcript/subagent-usage.js'; import { resolveClientVersion } from './client-version-cache.js'; -export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { +export class ClaudeCodeOtlpPlugin extends OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; - public readonly type = AgentAdapterType.OTLP; - public async processOtlpEvent(rawHookInput: string, { ensureOtlpProxy }: OtlpAdapterDeps): Promise { - const hookInput: unknown = JSON.parse(rawHookInput); - if (!isClaudeCodeHookInput(hookInput)) { - logger.debug('[Claude Code OTLP plugin] hook payload missing session_id/cwd/hook_event_name, ignoring'); - return; - } - - // INVARIANT - do not weaken. An untracked project must produce NO hooks data - // in the daemon spool, must not start the daemon, and must not run the SSO - // check. The daemon has no allowlist of its own. Its completeness gate - // (`otlp-spool/completeness-gate.ts`) only sends sessions that have hooks - // data, and skips sessions that only have OTEL data. Forwarding a hook event - // for an untracked project would make the session sendable and leak its - // data to the backend. - const isTracked = await isProjectTracked(hookInput.cwd, await readAllowlistState()); - if (!isTracked) { - logger.debug('[Claude Code OTLP plugin] project not in analytics allowlist, ignoring hook event'); - return; - } + protected extractHookContext(hookInput: HookInput): OtlpHookContext { + return { + cwd: hookInput.cwd, + prompt: hookInput.hook_event_name === 'UserPromptSubmit' ? hookInput.prompt : undefined, + }; + } - await ensureOtlpProxy(this.name); + protected parseHookInput(raw: string): HookInput | null { + const parsed: unknown = JSON.parse(raw); + return isClaudeCodeHookInput(parsed) ? parsed : null; + } - const evaluation = await this.evaluate(hookInput); - if (evaluation.decision === 'block') { - logger.error(`[Claude Code OTLP plugin] Blocking prompt: ${evaluation.reason}`); - console.log(JSON.stringify(evaluation)); - return; - } - this.forwardToSpool(await this.withCommonFields(evaluation.payload)); + protected async isTracked(hookInput: HookInput): Promise { + return isProjectTracked(hookInput.cwd, await readAllowlistState()); } /** Resolved once per call so every record in the batch shares one client-version lookup. */ - private async withCommonFields(records: Record[]): Promise[]> { - const common = { + protected async resolveAgentCommonFields(): Promise> { + return { platform: 'claude-code', entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', client_version: await resolveClientVersion(), }; - return records.map((record) => ({ ...record, ...common })); } - private async evaluate(hookInput: HookInput): Promise { + protected async evaluate(hookInput: HookInput): Promise { if (hookInput.hook_event_name === 'UserPromptSubmit') { return await this.onUserPromptSubmit(hookInput); } @@ -93,22 +75,22 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { return { decision: 'forward', payload: [hookInput] }; } - private async onStopEvent(hookInput: StopHookInput): Promise { + private async onStopEvent(hookInput: StopHookInput): Promise { const derived = await collectMainTranscriptEvents(hookInput.session_id, hookInput.transcript_path, 'Stop'); return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onPreCompactEvent(hookInput: PreCompactHookInput): Promise { + private async onPreCompactEvent(hookInput: PreCompactHookInput): Promise { const derived = await collectMainTranscriptEvents(hookInput.session_id, hookInput.transcript_path, 'PreCompact'); return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onStopFailureEvent(hookInput: StopFailureHookInput): Promise { + private async onStopFailureEvent(hookInput: StopFailureHookInput): Promise { const derived = await collectMainTranscriptEvents(hookInput.session_id, hookInput.transcript_path, 'StopFailure'); return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onSessionEndEvent(hookInput: SessionEndHookInput): Promise { + private async onSessionEndEvent(hookInput: SessionEndHookInput): Promise { const { session_id: sessionId, transcript_path: transcriptPath } = hookInput; const derived = await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd'); @@ -122,7 +104,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { return { decision: 'forward', payload: [hookInput, ...derived] }; } - private async onSubagentStopEvent(hookInput: SubagentStopHookInput): Promise { + private async onSubagentStopEvent(hookInput: SubagentStopHookInput): Promise { const agentTranscriptPath = hookInput.agent_transcript_path; if (!agentTranscriptPath) { return { decision: 'forward', payload: [hookInput] }; @@ -151,14 +133,7 @@ export class ClaudeCodeOtlpPlugin implements OtlpAgentAdapter { }); } - private forwardToSpool(events: Record[]): void { - // Intentionally not awaited: forwardOtlpEventToSpool is fire-and-forget. - for (const event of events) { - forwardOtlpEventToSpool(event, CLAUDE_CODE_OTLP_AGENT_NAME); - } - } - - private async onUserPromptSubmit(hookInput: UserPromptSubmitHookInput): Promise { + private async onUserPromptSubmit(hookInput: UserPromptSubmitHookInput): Promise { const authResult = await this.ensureProxyAuth(); if (authResult.ok) { diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts index f7e452a6f..93a4301bb 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.types.ts @@ -1,8 +1,7 @@ +import type { ForwardDecision } from '@/agents/core/OtlpAgentAdapter.js'; import type { HookInput, UserPromptSubmitHookSpecificOutput } from '@anthropic-ai/claude-agent-sdk'; -export type ForwardDecision = - | { decision: 'forward'; payload: Record[] } - | { decision: 'block'; reason: string; hookSpecificOutput: UserPromptSubmitHookSpecificOutput }; +export type ClaudeForwardDecision = ForwardDecision; /** * Narrows parsed hook JSON to the SDK's `HookInput` union. diff --git a/src/agents/registry.ts b/src/agents/registry.ts index b6a56dfbd..9f2a5fead 100644 --- a/src/agents/registry.ts +++ b/src/agents/registry.ts @@ -10,7 +10,8 @@ import { KimiPlugin } from './plugins/kimi/kimi.plugin.js'; import { KimiAcpPlugin } from './plugins/kimi/kimi-acp.plugin.js'; import { OpenWikiPlugin } from './plugins/openwiki/openwiki.plugin.js'; import { CopilotCliPlugin } from './plugins/copilot-cli/index.js'; -import { AgentAdapter, AgentAdapterType, AgentAnalyticsAdapter, OtlpAgentAdapter } from './core/types.js'; +import { AgentAdapter, AgentAdapterType, AgentAnalyticsAdapter } from './core/types.js'; +import type { OtlpAgentAdapter } from './core/OtlpAgentAdapter.js'; // Re-export for backwards compatibility export { AgentAdapter, AgentAnalyticsAdapter } from './core/types.js'; @@ -73,7 +74,7 @@ export class AgentRegistry { return AgentRegistry.adapters.get(name); } - static getAnalyticsAgent(name: string): OtlpAgentAdapter | undefined { + static getAnalyticsAgent(name: string): OtlpAgentAdapter | undefined { AgentRegistry.initialize(); return AgentRegistry.otlpAdapters.get(name); } diff --git a/src/cli/commands/__tests__/hook.analytics-wiring.test.ts b/src/cli/commands/__tests__/hook.analytics-wiring.test.ts index ce30d39a6..9076ed182 100644 --- a/src/cli/commands/__tests__/hook.analytics-wiring.test.ts +++ b/src/cli/commands/__tests__/hook.analytics-wiring.test.ts @@ -10,6 +10,7 @@ vi.mock('../proxy/connect-orchestrator.js', () => ({ ensureOtlpProxy: vi.fn() }) import { createHookCommand } from '../hook.js'; import { ensureOtlpProxy } from '../proxy/connect-orchestrator.js'; +import { forwardOtlpEventToSpool } from '@/agents/plugins/utils.js'; import { AgentRegistry } from '../../../agents/registry.js'; describe('hook command analytics wiring', () => { @@ -29,7 +30,7 @@ describe('hook command analytics wiring', () => { vi.restoreAllMocks(); }); - it('calls processOtlpEvent with the parsed input and { ensureOtlpProxy }', async () => { + it('calls processOtlpEvent with the parsed input and { ensureOtlpProxy, forwardOtlpEventToSpool }', async () => { const processOtlpEvent = vi.fn().mockResolvedValue(undefined); vi.spyOn(AgentRegistry, 'getAnalyticsAgent').mockReturnValue({ processOtlpEvent } as never); @@ -39,7 +40,7 @@ describe('hook command analytics wiring', () => { expect(processOtlpEvent).toHaveBeenCalledTimes(1); const [input, deps] = processOtlpEvent.mock.calls[0]; expect(JSON.parse(input as string)).toMatchObject({ hook_event_name: 'UserPromptSubmit' }); - expect(deps).toEqual({ ensureOtlpProxy }); + expect(deps).toEqual({ ensureOtlpProxy, forwardOtlpEventToSpool }); expect((deps as { ensureOtlpProxy: unknown }).ensureOtlpProxy).toBe(ensureOtlpProxy); }); }); diff --git a/src/cli/commands/hook.ts b/src/cli/commands/hook.ts index 436130461..90a2381d2 100644 --- a/src/cli/commands/hook.ts +++ b/src/cli/commands/hook.ts @@ -6,6 +6,7 @@ import { SESSION_ORIGIN, SESSION_ORIGIN_ENV_KEY } from '@/agents/core/session/ty import type { BaseHookEvent, HookTransformer, MCPConfigSummary, ExtensionsScanSummary } from '@/agents/core/types.js'; import type { ProcessingContext } from '@/agents/core/session/BaseProcessor.js'; import { ensureOtlpProxy } from './proxy/connect-orchestrator.js'; +import { forwardOtlpEventToSpool } from '@/agents/plugins/utils.js'; import { ensureCodeMieSsoAuth, type AuthGateInput } from '@/providers/plugins/sso/sso.auth-gate.js'; /** @@ -1517,7 +1518,7 @@ export function createHookCommand(): Command { const analyticsAgent = AgentRegistry.getAnalyticsAgent(opts.agent!); if (analyticsAgent) { - await analyticsAgent.processOtlpEvent(input, { ensureOtlpProxy }); + await analyticsAgent.processOtlpEvent(input, { ensureOtlpProxy, forwardOtlpEventToSpool }); await logger.close(); process.exitCode = 0; return; diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index 97362e61b..d1be60fec 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -19,6 +19,8 @@ interface MappedRecord { codemie_cli_version: string; story_id?: string; story_source?: string; + git_branch?: string; + repo_remote?: string; prompt_body?: string; developer_name?: string; identity_source?: string; @@ -80,7 +82,6 @@ describe('mapHookRecords', () => { baseUrl: '', projectName: 'proj', userEmail: 'user@example.com', - git: {}, }; const record1 = buildHookRecord('SessionStart', 'sid1', {}, 'event-id-1'); @@ -109,7 +110,6 @@ describe('mapHookRecords', () => { baseUrl: '', projectName: 'proj', userEmail: '', - git: {}, }; const record = buildHookRecord('SessionStart', 'sid1', {}, 'stamped-event-id-abc'); @@ -128,7 +128,6 @@ describe('mapHookRecords', () => { baseUrl: '', projectName: 'proj', userEmail: '', - git: {}, }; // PostToolUse normally maps to 'agent.tool.end', but an explicit `type` @@ -150,7 +149,6 @@ describe('mapHookRecords', () => { baseUrl: '', projectName: 'proj', userEmail: '', - git: {}, }; const record = buildHookRecord('PostToolUse', 'sid1'); @@ -161,101 +159,56 @@ describe('mapHookRecords', () => { expect(line.type).toBe('agent.tool.end'); }); - it( - "overrides a UserPromptSubmit record's story_id/story_source with a prompt marker " + - 'even when the per-tick branch tier would otherwise resolve to a different ticket, ' + - 'and never leaks the raw prompt text onto the emitted record', - async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - // Branch carries a DIFFERENT ticket than the prompt marker, so this - // test proves the marker tier wins over the already-cached branch tier. - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - git: { branch: 'feature/ABC-1-unrelated-branch' }, - }; - - // Longer than MAX_PROMPT_CHARS (200) so every bounded copy on the - // mapped record is truncated and none of them equals this full text — - // which is what actually proves "the raw prompt is never present" - // rather than merely proving a short prompt survives truncation whole. - const rawPrompt = - `Please implement this feature. story: EPMCDME-999 is the ticket to reference. ` + - 'x'.repeat(200) + - ' end-of-prompt-marker-that-must-not-appear-anywhere-in-the-output'; - const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); - - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + it('passes git_branch/repo_remote/story_id/story_source through from the incoming record', async () => { + const { mapHookRecords } = await import('../forwarder.js'); - expect(line.story_id).toBe('EPMCDME-999'); - expect(line.story_source).toBe('marker'); + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + }; - // The raw prompt text must never appear verbatim anywhere on the - // emitted record — only the truncated `prompt_body` and the resolved - // short `story_id` string are allowed to carry prompt-derived content. - const serialized = JSON.stringify(line); - expect(serialized).not.toContain(rawPrompt); - expect(serialized).not.toContain('end-of-prompt-marker-that-must-not-appear-anywhere-in-the-output'); - expect(line.prompt_body).toBe(rawPrompt.slice(0, 200)); - } - ); - - it( - 'falls back to the per-tick branch result for a UserPromptSubmit record whose prompt ' + - 'has no marker and no bare ticket mention', - async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - git: { branch: 'feature/epmcdme-15301-foo' }, - }; - - const rawPrompt = 'please just fix the thing, no ticket reference here'; - const record = buildHookRecord('UserPromptSubmit', 'sid1', { - prompt: rawPrompt, - cwd: '/repo/nonexistent-for-this-test', - }); - - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + const record = buildHookRecord('UserPromptSubmit', 'sid1', { + prompt: 'story: ABC-1 is ignored here, the forwarder does not resolve stories', + git_branch: 'feature/x', + repo_remote: 'org/repo', + story_id: 'EPMCDME-999', + story_source: 'marker', + }); - expect(line.story_id).toBe('EPMCDME-15301'); - expect(line.story_source).toBe('branch'); - } - ); - - it( - 'falls back to the mention tier for a UserPromptSubmit record whose prompt has a bare ' + - 'ticket mention and the branch carries no ticket', - async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - git: { branch: 'just-some-branch-name' }, - }; - - const rawPrompt = 'can you look into ABC-42 when you get a chance'; - const record = buildHookRecord('UserPromptSubmit', 'sid1', { prompt: rawPrompt }); - - const payload = await mapHookRecords([record], ctx); + const payload = await mapHookRecords([record], ctx); + const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + + expect(line.git_branch).toBe('feature/x'); + expect(line.repo_remote).toBe('org/repo'); + expect(line.story_id).toBe('EPMCDME-999'); + expect(line.story_source).toBe('marker'); + }); + + it("defaults git_branch/repo_remote/story_id/story_source to '' and never resolves them from the daemon env", async () => { + const { mapHookRecords } = await import('../forwarder.js'); + vi.stubEnv('SDLC_ANALYTICS_STORY_ID', 'FROM-DAEMON-1'); + + const ctx = { + credentials: { token: '', apiUrl: '' }, + baseUrl: '', + projectName: 'proj', + userEmail: '', + }; + + try { + const payload = await mapHookRecords([buildHookRecord('UserPromptSubmit', 'sid1', { prompt: 'story: ABC-1' })], ctx); const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; - expect(line.story_id).toBe('ABC-42'); - expect(line.story_source).toBe('mention'); + expect(line.git_branch).toBe(''); + expect(line.repo_remote).toBe(''); + expect(line.story_id).toBe(''); + expect(line.story_source).toBe(''); + } finally { + vi.unstubAllEnvs(); } - ); + }); it('passes agent-baked common fields (platform/client_version) through onto the mapped record without any agent-specific lookup', async () => { const { mapHookRecords } = await import('../forwarder.js'); @@ -265,7 +218,6 @@ describe('mapHookRecords', () => { baseUrl: '', projectName: 'proj', userEmail: '', - git: {}, }; // Simulates what the plugin now bakes in hook-side before ever reaching the spool — @@ -301,7 +253,6 @@ describe('mapHookRecords', () => { baseUrl: '', projectName: 'proj', userEmail: '', - git: {}, identity: { developerName: 'git-user@example.com', identitySource: 'git' as const }, }; diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts index 87eb4f7ed..14fb15b5e 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts @@ -1,5 +1,5 @@ /** - * Per-forward-tick context: the CLI version banner, and the identity/story resolution glue + * Per-forward-tick context: the CLI version banner, and the identity resolution glue * the hook-record mapping step calls once per batch. Split out of `forwarder.ts` to keep * that module under the documented 500-line structure cap (code-quality.md). */ @@ -7,24 +7,14 @@ import { execSync } from 'node:child_process'; import type { SSOCredentials, JWTCredentials } from '@/providers/core/types.js'; import { resolveIdentity, type IdentitySource } from './identity.js'; -import { - resolveExplicitStory, - resolveBranchStory, - resolveMarkerStory, - resolveMentionStory, -} from './story-resolver.js'; export interface ForwardContext { credentials: SSOCredentials | JWTCredentials; baseUrl: string; projectName: string; userEmail: string; - /** Per-session git info cache, resolved lazily from the first hook `cwd`. */ - git: { branch?: string; remote?: string }; /** Per-session developer-identity cache, resolved once */ identity?: { developerName?: string; identitySource?: IdentitySource }; - /** Per-tick story-id cache, resolved once per forward tick. */ - story?: { storyId?: string; storySource?: 'explicit' | 'branch' }; } /** The installed CodeMie CLI version, resolved once at import time. */ @@ -41,65 +31,6 @@ export function resolveCodemieCliVersion(): string { } } -/** - * Effective story id for ONE `UserPromptSubmit` record: layers this record's - * own prompt text on top of the once-per-tick cache in `ctx.story` - * (explicit/branch/undefined), without ever writing back to that cache — - * other records in the same batch still need it untouched. Priority order - * across the full chain is explicit -> marker -> branch -> mention: - * - * 1. If the cache already resolved to `'explicit'`, that wins outright. - * 2. Otherwise, try the marker tier (`story: X` / `ticket #X`) against this - * record's OWN prompt text — it sits above branch in priority. - * 3. Otherwise, if the cache resolved to `'branch'`, that wins (it is - * already correctly placed between marker and mention). - * 4. Otherwise, try the mention tier (bare ticket-shaped text) — the - * lowest-priority tier. - * 5. Otherwise, empty. - */ -export function resolveStoryForPrompt( - ctx: ForwardContext, - rawPrompt: string -): { storyId: string; storySource: string } { - if (ctx.story?.storySource === 'explicit') { - return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; - } - - const marker = resolveMarkerStory(rawPrompt); - if (marker) { - return { storyId: marker.storyId, storySource: marker.storySource }; - } - - if (ctx.story?.storySource === 'branch') { - return { storyId: ctx.story.storyId ?? '', storySource: ctx.story.storySource }; - } - - const mention = resolveMentionStory(rawPrompt); - if (mention) { - return { storyId: mention.storyId, storySource: mention.storySource }; - } - - return { storyId: '', storySource: '' }; -} - -/** - * Resolve and cache this forward tick's story id/source once, from the first record that carries - * a real `cwd` — guarded the same way every other per-tick cache in this file is, so a synthetic - * transcript-derived record's empty `cwd` can never poison the cache for the rest of the batch. - */ -export async function resolveStory(ctx: ForwardContext, cwd: string): Promise { - if (!cwd || ctx.story?.storyId !== undefined) { - return; - } - - const explicit = await resolveExplicitStory(cwd); - const resolved = explicit ?? resolveBranchStory(ctx.git.branch ?? ''); - - ctx.story = resolved - ? { storyId: resolved.storyId, storySource: resolved.storySource } - : { storyId: '' }; -} - /** * Resolve and cache this forward tick's developer identity once, from the first record that * carries a real `cwd`. An empty `cwd` (every synthetic transcript-derived record) is skipped diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index f9f58d140..af0d128b9 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -17,8 +17,6 @@ import { type ForwardContext, resolveCodemieCliVersion, resolveDeveloperIdentity, - resolveStory, - resolveStoryForPrompt, } from './forward-context.js'; const CODEMIE_CLI_VERSION = resolveCodemieCliVersion(); @@ -181,9 +179,8 @@ function resolveEventTimestamp(eventTimestamp: unknown, spoolTimestamp: number): return new Date(spoolTimestamp).toISOString(); } -/** Prefer the event's own `git_branch` (the session's branch) over the daemon's cwd branch. */ -function resolveEventGitBranch(eventBranch: unknown, daemonBranch: string | undefined): string { - return typeof eventBranch === 'string' && eventBranch.length > 0 ? eventBranch : (daemonBranch ?? ''); +function stringOrEmpty(value: unknown): string { + return typeof value === 'string' ? value : ''; } function limitHookPayload(hookEvent: Record): Record { @@ -203,23 +200,6 @@ function limitHookPayload(hookEvent: Record): Record { - if (!cwd || ctx.git.branch !== undefined) { - return; - } - try { - const { detectGitBranch, detectGitRemoteRepo } = await import('@/utils/processes.js'); - const [branch, remote] = await Promise.all([ - detectGitBranch(cwd).then((v) => v ?? ''), - detectGitRemoteRepo(cwd).then((v) => v ?? ''), - ]); - ctx.git.branch = branch; - ctx.git.remote = remote; - } catch { - /* best-effort */ - } -} - interface HookPayload { ndjson: string; containsSessionEnd: boolean; @@ -253,24 +233,7 @@ export async function mapHookRecords( } const cwd = String(hookEvent['cwd'] ?? ''); - await resolveGitInfo(ctx, cwd); await resolveDeveloperIdentity(ctx, cwd); - await resolveStory(ctx, cwd); - - // Read the record's OWN untruncated prompt text here, before the truncated - // copy below is produced. The original `hookEvent` is never mutated by that - // truncation step, so this is still the full string — used ONLY to feed the - // marker/mention regex tiers below; the matched ticket id (a short string) - // is all that ever reaches the output, never this raw text itself. - const rawPrompt = typeof hookEvent['prompt'] === 'string' ? hookEvent['prompt'] : ''; - - // Each UserPromptSubmit record carries its own prompt text, which can hold a - // higher-priority override — so it's resolved fresh per record rather than - // just reusing the once-per-tick cache. - const promptStory = - hookName === 'UserPromptSubmit' - ? resolveStoryForPrompt(ctx, rawPrompt) - : { storyId: ctx.story?.storyId ?? '', storySource: ctx.story?.storySource ?? '' }; const limited = limitHookPayload(hookEvent); const type = hookEventType(hookName, hookEvent); @@ -284,10 +247,11 @@ export async function mapHookRecords( user_email: ctx.userEmail, developer_name: ctx.identity?.developerName ?? '', identity_source: ctx.identity?.identitySource ?? '', - git_branch: resolveEventGitBranch(hookEvent['git_branch'], ctx.git.branch), - repo_remote: ctx.git.remote ?? '', - story_id: promptStory.storyId, - story_source: promptStory.storySource, + // Resolved at hook time by the adapter (OtlpAgentAdapter); passed through as-is. + git_branch: stringOrEmpty(hookEvent['git_branch']), + repo_remote: stringOrEmpty(hookEvent['repo_remote']), + story_id: stringOrEmpty(hookEvent['story_id']), + story_source: stringOrEmpty(hookEvent['story_source']), codemie_project_name: ctx.projectName, cwd, prompt_body: boundedText(hookEvent['prompt'], MAX_PROMPT_CHARS), @@ -321,9 +285,7 @@ async function buildForwardContext( baseUrl: state?.targetUrl ?? state?.url ?? '', projectName: state?.project ?? '', userEmail: resolveEmailFromCredentials(credentials), - git: {}, identity: {}, - story: {}, }; } From a87bef2c11dbc64464e8e7cfa74a533942a0c164 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Thu, 8 Oct 2026 17:26:33 +0300 Subject: [PATCH 30/35] refactor(proxy): make OTLP forwarder transport-only and build events in the adapter The abstract OtlpAgentAdapter now produces the complete event at hook time: type, truncation, prompt_body, session_id, cwd, schema_version, event_id, timestamp, plus identity, project name and CLI version resolved for the hook cwd from the stored SSO credentials. Events are sent awaited and in order. The forwarder only unwraps the spool envelope, POSTs it as is, advances the cursor and retries on auth failure; session end is detected from the event type. - rename OtlpHookSpoolData.raw to hookEvent and drop its timestamp - move identity.ts to agents/core/identity-resolver.ts, add hook-credentials.ts - remove forward-context.ts and the daemon-side CLI version lookup - add abstract resolveEventType, rename resolveAgentCommonFields to resolveAgentFields - update OTLP and proxy architecture docs --- docs/ARCHITECTURE-OTLP-PLUGIN.md | 41 ++- docs/ARCHITECTURE-PROXY.md | 2 + src/agents/core/OtlpAgentAdapter.ts | 203 ++++++++++- .../core/__tests__/OtlpAgentAdapter.test.ts | 224 +++++++++++- .../core/__tests__/hook-credentials.test.ts | 68 ++++ .../core/__tests__/identity-resolver.test.ts} | 6 +- src/agents/core/hook-credentials.ts | 33 ++ .../core/identity-resolver.ts} | 11 +- src/agents/core/types.ts | 1 - .../__tests__/claude-code-otlp.plugin.test.ts | 63 +++- .../claude-code-otlp.plugin.ts | 21 +- .../transcript/session-summary.ts | 5 +- .../transcript/subagent-usage.ts | 2 +- .../transcript/usage-request.ts | 2 +- src/agents/plugins/utils.ts | 15 +- .../otlp-spool/__tests__/forwarder.test.ts | 345 ++++++------------ .../plugins/otlp-spool/forward-context.ts | 54 --- .../sso/proxy/plugins/otlp-spool/forwarder.ts | 185 ++-------- .../plugins/sso/proxy/plugins/otlp.plugin.ts | 10 +- 19 files changed, 784 insertions(+), 507 deletions(-) create mode 100644 src/agents/core/__tests__/hook-credentials.test.ts rename src/{providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts => agents/core/__tests__/identity-resolver.test.ts} (92%) create mode 100644 src/agents/core/hook-credentials.ts rename src/{providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts => agents/core/identity-resolver.ts} (93%) delete mode 100644 src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts diff --git a/docs/ARCHITECTURE-OTLP-PLUGIN.md b/docs/ARCHITECTURE-OTLP-PLUGIN.md index 011849c29..3357e1a61 100644 --- a/docs/ARCHITECTURE-OTLP-PLUGIN.md +++ b/docs/ARCHITECTURE-OTLP-PLUGIN.md @@ -17,7 +17,8 @@ export abstract class OtlpAgentAdapter; // allowlist gate protected abstract evaluate(input: TInput): Promise>; - protected abstract agentCommonFields(): Promise>; + protected abstract resolveAgentFields(): Promise>; + protected abstract resolveEventType(event: Record): string; // `type` of a native event } export interface OtlpAdapterDeps { @@ -27,8 +28,7 @@ export interface OtlpAdapterDeps { export interface OtlpHookSpoolData { agentName: string; - raw: string; - timestamp: number; + hookEvent: string; // the complete event (serialized), built at hook time } ``` @@ -53,22 +53,31 @@ evaluate(input) → ForwardDecision (§3: shape every adapter follows) ┌────┴─────┐ block forward │ │ - log + common fields merged, - forward each record ──POST──► proxy daemon spool (OtlpHookSpoolData) - suppress │ - (adapter- ▼ - specific) otlp-spool/forwarder.ts (background, agent-agnostic) + log + event built per record: type, truncation, common + + suppress agent + context fields, schema_version, event_id + (adapter- │ awaited, one record at a time, in order + specific) ▼ + proxy daemon spool (OtlpHookSpoolData { agentName, hookEvent }) - stores as is + │ + ▼ + otlp-spool/forwarder.ts (background, transport only) │ ▼ CodeMie analytics API ``` -- `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is fire-and-forget and shared: POSTs `{ agentName, timestamp, raw }` to the local proxy daemon, swallows every error. A dead daemon never blocks or fails the hook. -- `otlp-spool/forwarder.ts` only forwards: it maps spooled records to the analytics API payload and never calls into any adapter. It passes `story_id`, `story_source`, `git_branch`, `repo_remote` through from the record (`''` when absent) and no longer resolves story or git info itself. Credential and daemon-state fields (`user_email`, `developer_name`, `identity_source`, `codemie_project_name`, `codemie_cli_version`) are still stamped by the forwarder. -- **Common fields resolved by the base class at hook time** (the forwarder runs in a long-lived daemon whose env is frozen at spawn, so env-dependent fields were wrong there): `git_branch` (`detectGitBranch`), `repo_remote` (`detectGitRemoteRepo`), and `story_id` / `story_source` (`core/story-resolver.ts`, priority explicit env/config file -> marker -> branch -> mention; marker and mention only when `extractHookContext` supplies a `prompt`). Merge rule: a record's own non-empty string value wins, otherwise the resolved common value; `''` / `undefined` never clobber. `agentCommonFields()` overrides everything (platform, version, entrypoint, …). +- `forwardOtlpEventToSpool` (`src/agents/plugins/utils.ts`) is shared: POSTs `{ agentName, hookEvent }` to the local proxy daemon (1s abort), swallows every error, never throws. `hookEvent` is the already complete event, serialized. A dead daemon never blocks or fails the hook. +- **The forwarder is transport only** (`otlp-spool/forwarder.ts`). It unwraps the spool envelope, POSTs `hookEvent` byte for byte, advances the cursor, and retries on 401/403. It does no context resolution and no event enrichment, and never calls into any adapter. It owns only `baseUrl` (from daemon state) and the credentials. The one thing it reads from an event is `type === 'agent.session.end'`, to call `markSessionEnded`. +- **The event is built by the base class at hook time** (`buildEvent`), because the forwarder runs in a long-lived daemon whose env (cwd, active profile, credentials, CLI version) is frozen at spawn. Per record, in order: `type` (an explicit `type` on the record wins, otherwise the adapter's `resolveEventType`), truncation (`prompt` to 200 chars, `tool_input` / `tool_response` / `error` to 300; a nested `raw` is dropped), `prompt_body`, `session_id` coerced to a string, `cwd` (always the hook's cwd, also for transcript-derived events), `schema_version: 2`, a fresh `event_id`, and `timestamp`. The event's own `raw` field mirrors the full enriched event, so the top level and `raw` carry the same fields. +- **Timestamp**: the event's own valid timestamp (e.g. the transcript line time on `agent.usage.request`), else one `new Date()` captured per hook invocation. It is persisted inside the spool line, so every retry re-sends the same bytes. +- **Common fields resolved by the base class at hook time**: `git_branch` (`detectGitBranch`), `repo_remote` (`detectGitRemoteRepo`), and `story_id` / `story_source` (`core/story-resolver.ts`, priority explicit env/config file -> marker -> branch -> mention; marker and mention only when `extractHookContext` supplies a `prompt`). Merge rule: a record's own non-empty string value wins, otherwise the resolved common value; `''` / `undefined` never clobber. `resolveAgentFields()` overrides everything (platform, version, entrypoint, …). +- **Identity fields resolved by the base class at hook time** (part of the common fields, stamped over the event), once per invocation for the hook's `cwd`: `user_email` (claims of the SSO credentials), `developer_name` / `identity_source` (`core/identity-resolver.ts`: jwt -> git -> codemie_cli -> os), `codemie_project_name` (`ConfigLoader.load(cwd, { name: activeProfile })`), `codemie_cli_version` (`getCurrentCliVersion()`, i.e. the package running the hook). A failed resolution yields empty strings, never a dropped event. +- **Credentials (SSO only)**: `core/hook-credentials.ts` (`resolveHookCredentials`) reproduces the daemon's choice from daemon state: the credentials stored for `syncCodeMieUrl`, else for `targetUrl` (`SSOProxy`'s `syncCredentials || credentials`). The OTLP daemon is spawned without `--auth-method`, so it always runs as `sso`. JWT is not supported in this flow: a JWT profile gets no credentials in the hook, so identity falls through to git, codemie_cli and os (`identity_source` records the winner). Reading stored credentials clears expired SSO ones, as at daemon start. - **Story granularity**: each hook subprocess resolves independently. A `Stop` after a prompt containing `story: ABC-1` gets the explicit / branch story, not the marker story. Carrying it across a session would need per-session state. - **Known limitations**: (1) the explicit story tier reads `/.claude/analytics.local.json` by default (Claude-flavoured); `resolveStoryFor` takes `explicitConfigPath` so another adapter can override it. (2) The block path prints `JSON.stringify(decision)` to stdout, which is Claude Code's hook protocol; a tool with a different block protocol must make that step overridable (e.g. an `emitBlock()` method). - **Rollout**: the daemon outlives CLI upgrades. An old daemon recomputes `story_id`, `story_source` and `repo_remote` from its own env and overwrites the hook's values until restarted (`git_branch` is kept). Restart the proxy daemon after upgrading. +- **Known limitations**: (1) the explicit story tier reads `/.claude/analytics.local.json` by default (Claude-flavoured); `resolveStoryFor` takes `explicitConfigPath` so another adapter can override it. (2) The block path prints `JSON.stringify(decision)` to stdout, which is Claude Code's hook protocol; a tool with a different block protocol must make that step overridable (e.g. an `emitBlock()` method). (3) The hook `cwd` may be a subdirectory of the project; `getActiveProfileName(cwd)` only checks `/.codemie` and then the global config, so project name and profile can then come from the global profile (same as `ensureOtlpProxy`). +- **Rollout**: the daemon outlives CLI upgrades. Restart the proxy daemon after upgrading. Old-format spool lines and a stale daemon are not handled and produce bad events. For a clean run, stop the daemon and delete `getCodemiePath('proxy', 'otlp-spool')` before the first hook. - Wiring which native hooks/events call `codemie hook --agent ` is entirely tool-specific — see §6 for the Claude Code connector; a different tool has its own. ## 3. The dispatch pattern every `evaluate()` follows @@ -111,7 +120,7 @@ Each handler returns its own full `ForwardDecision`. The trailing fallthrough re ### 3.4 One place writes to the spool -The base class's `processOtlpEvent` is the only call site of `deps.forwardOtlpEventToSpool`: after `evaluate()` resolves it computes the common fields once, merges them into each `payload` record and forwards every record, **not awaited** (fire-and-forget; it never throws, and awaiting would put daemon latency on the hook path). Handlers never forward anything themselves, so every event reaches the spool exactly once, in order, under the adapter's registered name. +The base class's `processOtlpEvent` is the only call site of `deps.forwardOtlpEventToSpool`: after `evaluate()` resolves it computes the common (git, story, identity, project, version) and agent fields once (in parallel), builds the event for each `payload` record and forwards the records one at a time, **awaited and sequential**, so a hook event stays ahead of its derived events in the spool. `forwardOtlpEventToSpool` never throws and aborts after 1s, so awaiting adds no wall time and keeps `logger.close()` in `hook.ts` valid. The block path is unchanged: it prints the decision and returns before any enrichment. Handlers never forward anything themselves, so every event reaches the spool exactly once, in order, under the adapter's registered name. ### 3.5 Proxy readiness is the adapter's own responsibility @@ -127,7 +136,7 @@ The base flow calls `isTracked()` (abstract) before `deps.ensureOtlpProxy()`, so ## 5. Adding a new `OtlpAgentAdapter` for a different tool -1. Create `src/agents/plugins//` with a class `extends OtlpAgentAdapter` and implement `parseHookInput`, `extractHookContext`, `isTracked`, `evaluate` (following §3) and `agentCommonFields`. Agent-owned common fields (platform, version, entrypoint, …) go in `agentCommonFields()`, resolved **inside the adapter's own hook-time process**, not via a callback from the forwarder — the forwarder runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, so resolving state there would reflect the daemon's environment. Git branch/remote and story are already resolved by the base class. If resolving a field is expensive (subprocess spawn, network call), back it with a small file cache under `getCodemiePath()` — the hook-time process is fresh per event, so in-memory memoization buys nothing (see `client-version-cache.ts`). +1. Create `src/agents/plugins//` with a class `extends OtlpAgentAdapter` and implement `parseHookInput`, `extractHookContext`, `isTracked`, `evaluate` (following §3), `resolveAgentFields` and `resolveEventType` (the `type` for a native event without an explicit one). Agent-owned common fields (platform, version, entrypoint, …) go in `resolveAgentFields()`, resolved **inside the adapter's own hook-time process**, not via a callback from the forwarder — the forwarder runs in the long-lived proxy daemon, a different process from the short-lived `codemie hook --agent ` CLI invocation, so resolving state there would reflect the daemon's environment. Git branch/remote, story, identity, project name and CLI version are already resolved by the base class. If resolving a field is expensive (subprocess spawn, network call), back it with a small file cache under `getCodemiePath()` — the hook-time process is fresh per event, so in-memory memoization buys nothing (see `client-version-cache.ts`). 2. Register it in `AgentRegistry` (`src/agents/registry.ts`) under its own `name` — same name the base class passes to `forwardOtlpEventToSpool(event, name)` and matched by `AgentRegistry.getAnalyticsAgent(name)` in `hook.ts`. 3. Write a connector wiring the tool's native hooks/events to `codemie hook --agent ` (see `src/cli/commands/proxy/connectors/claude-code-otlp.ts` for the example — hook names, settings format, and env vars are tool-specific). 4. Everything from `forwardOtlpEventToSpool` onward is already shared — no changes needed there as long as 1-3 hold. @@ -138,8 +147,10 @@ The base flow calls `isTracked()` (abstract) before `deps.ensureOtlpProxy()`, so | File | Role | |---|---| -| `claude-code-otlp.plugin.ts` | `extends OtlpAgentAdapter`: input parsing, `evaluate()` dispatch, per-event handlers, `agentCommonFields()`, allowlist wiring | -| `src/agents/core/OtlpAgentAdapter.ts` | Abstract base: fixed flow, allowlist-before-daemon INVARIANT, generic `ForwardDecision`, hook-time git/story common fields, fire-and-forget spool forwarding | +| `claude-code-otlp.plugin.ts` | `extends OtlpAgentAdapter`: input parsing, `evaluate()` dispatch, per-event handlers, `resolveAgentFields()`, `resolveEventType()` (hook name to type map), allowlist wiring | +| `src/agents/core/OtlpAgentAdapter.ts` | Abstract base: fixed flow, allowlist-before-daemon INVARIANT, generic `ForwardDecision`, hook-time git/story/context common fields, `buildEvent`, awaited sequential spool forwarding (the forwarder only transports; `forward-context.ts` is gone) | +| `src/agents/core/identity-resolver.ts` | Developer identity tiers (jwt -> git -> codemie_cli -> os) and email-from-credentials, used by the base class at hook time | +| `src/agents/core/hook-credentials.ts` | `resolveHookCredentials()`: the SSO credentials the daemon will use, read from daemon state (SSO only) | | `src/agents/core/story-resolver.ts` | Story tiers and `resolveStoryFor({ cwd, branch, prompt, explicitConfigPath })` | | `client-version-cache.ts` | TTL file cache around `claude --version`, backing the `client_version` common field | | `claude-code-otlp.types.ts` | `ClaudeForwardDecision` alias, `isClaudeCodeHookInput` | diff --git a/docs/ARCHITECTURE-PROXY.md b/docs/ARCHITECTURE-PROXY.md index 7532b67e4..a373591f3 100644 --- a/docs/ARCHITECTURE-PROXY.md +++ b/docs/ARCHITECTURE-PROXY.md @@ -925,6 +925,8 @@ The OTLP plugin (`otlp.plugin.ts`) receives Claude Code OTel data and hook event **OTEL-only invariant.** The OTel environment is global, so untracked sessions still export OTEL data to the daemon, but they never receive hooks data. The completeness gate (`otlp-spool/completeness-gate.ts`) only sends sessions that have hooks data; an OTEL-only session gets `wait` and, once `waitTicks >= OTLP_SEND_MAX_ATTEMPTS` (limit resolved by `hooksOnlyWaitTicks()` in `otlp-spool/spool-config.ts`), `skip`. On `skip` the tick processor advances the OTEL cursors to EOF without sending. No code path may send OTEL-only data without first adding daemon-side project filtering. +**Who stamps what.** The hook process (`OtlpAgentAdapter`) stamps every event field: type, truncation, common, agent and context fields (identity, project, CLI version), `schema_version`, `event_id`, `timestamp`. The daemon spool (`otlp.plugin.ts`) only stores the line. The forwarder (`otlp-spool/forwarder.ts`) only forwards it: it owns `baseUrl` and the credentials, and reads nothing from an event except `type` to detect a session end. See [ARCHITECTURE-OTLP-PLUGIN.md](ARCHITECTURE-OTLP-PLUGIN.md). Restart the daemon after upgrading. + **Deletion timeline for untracked data** | Stage | When | Result | diff --git a/src/agents/core/OtlpAgentAdapter.ts b/src/agents/core/OtlpAgentAdapter.ts index 0ee90dedf..877020209 100644 --- a/src/agents/core/OtlpAgentAdapter.ts +++ b/src/agents/core/OtlpAgentAdapter.ts @@ -1,7 +1,12 @@ +import { randomUUID } from 'node:crypto'; import { logger } from '@/utils/logger.js'; import { detectGitBranch, detectGitRemoteRepo } from '@/utils/processes.js'; +import { ConfigLoader } from '@/utils/config.js'; +import { getCurrentCliVersion } from '@/utils/cli-updater.js'; import { AgentAdapterType, type OtlpAdapterDeps } from './types.js'; import { resolveStoryFor } from './story-resolver.js'; +import { resolveHookCredentials } from './hook-credentials.js'; +import { resolveEmailFromCredentials, resolveIdentity } from './identity-resolver.js'; export type ForwardDecision> = | { decision: 'forward'; payload: Record[] } @@ -13,16 +18,50 @@ export interface OtlpHookContext { prompt?: string; } -/** Fields resolved by the base class at hook time, shared by every agent. */ -const COMMON_FIELD_KEYS = ['git_branch', 'repo_remote', 'story_id', 'story_source'] as const; +/** Git and story fields, resolved from the hook's cwd and prompt. */ +interface RepoFields { + git_branch: string; + repo_remote: string; + story_id: string; + story_source: string; +} + +/** Identity, project and version fields, resolved from the hook's cwd and the stored SSO credentials. */ +interface IdentityFields { + user_email: string; + developer_name: string; + identity_source: string; + codemie_project_name: string; + codemie_cli_version: string; +} + +/** Agent-independent fields resolved by the base class at hook time. */ +type CommonFields = RepoFields & IdentityFields; + +/** Common fields an event may bring itself: its own non-empty value wins. All others are stamped. */ +const OVERRIDABLE_FIELD_KEYS = ['git_branch', 'repo_remote', 'story_id', 'story_source'] as const; + +/** Everything resolved once per hook invocation and shared by every event of that invocation. */ +interface InvocationContext { + cwd: string; + /** Captured once, so the fallback timestamp is the same for every event and every retry. */ + hookTime: Date; + commonFields: CommonFields; + agentFields: Record; +} + +const SCHEMA_VERSION = 2; +const MAX_PROMPT_CHARS = 200; +const MAX_TOOL_FIELD_CHARS = 300; /** * Base class for adapters that ingest a coding tool's native hook events into the * analytics spool. Owns the fixed flow (parse, allowlist gate, daemon start, evaluate, - * common-field enrichment, spool forward); subclasses supply the tool-specific parts. + * enrichment, spool forward); subclasses supply the tool-specific parts. * - * Common fields are resolved here, in the short-lived hook process, because the - * forwarder runs in a long-lived daemon whose environment is frozen at spawn time. + * The complete event is built here, in the short-lived hook process, because + * the forwarder runs in a long-lived daemon whose environment (cwd, profile, credentials, + * CLI version) is frozen at spawn time. The forwarder only transports it. */ export abstract class OtlpAgentAdapter> { abstract readonly name: string; @@ -62,13 +101,16 @@ export abstract class OtlpAgentAdapter; protected abstract evaluate(input: TInput): Promise>; /** Agent-owned fields (platform, version, entrypoint, ...); override everything else. */ - protected abstract resolveAgentCommonFields(): Promise>; + protected abstract resolveAgentFields(): Promise>; + /** `type` of a native event that does not carry an explicit one. */ + protected abstract resolveEventType(event: Record): string; + + /** + * Assembles the final event. `raw` mirrors the full enriched event, so the top + * level and `raw` carry the same fields. + */ + private buildEvent(hookEvent: Record, ctx: InvocationContext): Record { + const limited = this.limitPayload(hookEvent); + const explicitType = hookEvent['type']; + + const event: Record = { + ...this.mergeCommonFields(limited, ctx.commonFields), + ...ctx.agentFields, + type: typeof explicitType === 'string' && explicitType.length > 0 ? explicitType : this.resolveEventType(hookEvent), + session_id: String(hookEvent['session_id'] ?? ''), + timestamp: this.resolveEventTimestamp(hookEvent['timestamp'], ctx.hookTime), + cwd: ctx.cwd, + prompt_body: this.boundedText(hookEvent['prompt'], MAX_PROMPT_CHARS), + schema_version: SCHEMA_VERSION, + event_id: randomUUID(), + }; + return { ...event, raw: { ...event } }; + } + + /** Truncates the free-text fields; a nested `raw` is dropped so it cannot bypass the limits. */ + private limitPayload(hookEvent: Record): Record { + const limited: Record = { ...hookEvent }; + if (Object.prototype.hasOwnProperty.call(hookEvent, 'prompt')) { + limited['prompt'] = this.boundedText(hookEvent['prompt'], MAX_PROMPT_CHARS); + } + for (const field of ['tool_input', 'tool_response', 'error']) { + if (Object.prototype.hasOwnProperty.call(hookEvent, field)) { + limited[field] = this.boundedText(hookEvent[field], MAX_TOOL_FIELD_CHARS); + } + } + delete limited['raw']; + return limited; + } + + private boundedText(value: unknown, maxChars: number): string { + if (value === undefined || value === null) { + return ''; + } + let text: string; + if (typeof value === 'string') { + text = value; + } else { + try { + text = JSON.stringify(value) ?? String(value); + } catch { + text = String(value); + } + } + return text.slice(0, maxChars); + } + + /** + * The event's own timestamp (e.g. the transcript line time on `agent.usage.request`) wins + * over hook time. Falls back when the event has none or an unparseable one. + */ + private resolveEventTimestamp(eventTimestamp: unknown, hookTime: Date): string { + if (typeof eventTimestamp === 'string' && eventTimestamp.length > 0) { + const parsed = new Date(eventTimestamp); + if (!Number.isNaN(parsed.getTime())) { + return parsed.toISOString(); + } + } + return hookTime.toISOString(); + } + + /** + * Identity, project and CLI version for the hook's cwd. Credentials are the SSO ones the + * daemon will use (see `resolveHookCredentials`); without them identity falls through to + * the git, codemie_cli and os tiers. A failed lookup yields empty fields, never a dropped event. + */ + private async resolveIdentityFields(cwd: string): Promise { + const empty: IdentityFields = { + user_email: '', + developer_name: '', + identity_source: '', + codemie_project_name: '', + codemie_cli_version: '', + }; + try { + const credentials = await resolveHookCredentials(); + const [identity, projectName, version] = await Promise.all([ + resolveIdentity(credentials, cwd), + this.resolveProjectName(cwd), + getCurrentCliVersion(), + ]); + return { + user_email: resolveEmailFromCredentials(credentials), + developer_name: identity.developerName, + identity_source: identity.identitySource, + codemie_project_name: projectName, + codemie_cli_version: version ?? '', + }; + } catch (error) { + logger.debug(`[${this.name}] identity field resolution failed: ${error instanceof Error ? error.message : String(error)}`); + return empty; + } + } + + private async resolveProjectName(cwd: string): Promise { + try { + const profileName = await ConfigLoader.getActiveProfileName(cwd); + const config = await ConfigLoader.load(cwd, profileName ? { name: profileName } : undefined); + return config.codeMieProject ?? ''; + } catch { + return ''; + } + } + + private async resolveCommonFields(context: OtlpHookContext): Promise { + const [repoFields, identityFields] = await Promise.all([ + this.resolveRepoFields(context), + this.resolveIdentityFields(context.cwd), + ]); + return { ...repoFields, ...identityFields }; + } - private async resolveCommonFields({ cwd, prompt }: OtlpHookContext): Promise> { - const empty = { git_branch: '', repo_remote: '', story_id: '', story_source: '' }; + private async resolveRepoFields({ cwd, prompt }: OtlpHookContext): Promise { + const empty: RepoFields = { git_branch: '', repo_remote: '', story_id: '', story_source: '' }; try { const [branch, remote] = cwd ? await Promise.all([detectGitBranch(cwd), detectGitRemoteRepo(cwd)]) @@ -95,17 +258,19 @@ export abstract class OtlpAgentAdapter, commonFields: Record): Record { - const merged = { ...hookEvent }; - for (const key of COMMON_FIELD_KEYS) { + /** Stamps the common fields; for the overridable ones a hook event's own non-empty string value wins. */ + private mergeCommonFields(hookEvent: Record, commonFields: CommonFields): Record { + const merged: Record = { ...hookEvent, ...commonFields }; + for (const key of OVERRIDABLE_FIELD_KEYS) { const own = hookEvent[key]; - merged[key] = typeof own === 'string' && own.length > 0 ? own : commonFields[key]; + if (typeof own === 'string' && own.length > 0) { + merged[key] = own; + } } return merged; } diff --git a/src/agents/core/__tests__/OtlpAgentAdapter.test.ts b/src/agents/core/__tests__/OtlpAgentAdapter.test.ts index 4fded3b31..5564b6336 100644 --- a/src/agents/core/__tests__/OtlpAgentAdapter.test.ts +++ b/src/agents/core/__tests__/OtlpAgentAdapter.test.ts @@ -7,6 +7,21 @@ vi.mock('@/utils/processes.js', () => ({ detectGitBranch: detectGitBranchMock, detectGitRemoteRepo: detectGitRemoteRepoMock, })); +const resolveHookCredentialsMock = vi.fn(); +const resolveIdentityMock = vi.fn(); +const getActiveProfileNameMock = vi.fn(); +const configLoadMock = vi.fn(); +const getCurrentCliVersionMock = vi.fn(); + +vi.mock('../hook-credentials.js', () => ({ resolveHookCredentials: resolveHookCredentialsMock })); +vi.mock('../identity-resolver.js', async () => ({ + ...(await vi.importActual('../identity-resolver.js')), + resolveIdentity: resolveIdentityMock, +})); +vi.mock('@/utils/config.js', () => ({ + ConfigLoader: { getActiveProfileName: getActiveProfileNameMock, load: configLoadMock }, +})); +vi.mock('@/utils/cli-updater.js', () => ({ getCurrentCliVersion: getCurrentCliVersionMock })); vi.mock('@/utils/logger.js', () => ({ logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, })); @@ -42,11 +57,18 @@ class FakeAdapter extends OtlpAgentAdapter { } return { decision: 'forward', payload: input.payload }; } - protected async resolveAgentCommonFields(): Promise> { + protected async resolveAgentFields(): Promise> { return { platform: 'fake-platform' }; } + protected resolveEventType(event: Record): string { + return event['kind'] === 'start' ? 'fake.start' : 'fake.event'; + } } +/** A jwt-shaped access token carrying the given claims. */ +const tokenWith = (claims: Record): string => + `h.${Buffer.from(JSON.stringify(claims)).toString('base64url')}.s`; + describe('OtlpAgentAdapter.processOtlpEvent', () => { const ensureOtlpProxy = vi.fn(async () => {}); const forwardToSpool = vi.fn(async () => {}); @@ -62,6 +84,15 @@ describe('OtlpAgentAdapter.processOtlpEvent', () => { adapter = new FakeAdapter(); detectGitBranchMock.mockResolvedValue('main'); detectGitRemoteRepoMock.mockResolvedValue('org/repo'); + resolveHookCredentialsMock.mockResolvedValue({ + cookies: { codemie_access_token: tokenWith({ email: 'dev@example.com' }) }, + apiUrl: 'https://api', + timestamp: 0, + }); + resolveIdentityMock.mockResolvedValue({ developerName: 'dev@example.com', identitySource: 'jwt' }); + getActiveProfileNameMock.mockResolvedValue('work'); + configLoadMock.mockResolvedValue({ codeMieProject: 'proj-1' }); + getCurrentCliVersionMock.mockResolvedValue('1.2.3'); delete process.env['SDLC_ANALYTICS_STORY_ID']; }); @@ -104,10 +135,9 @@ describe('OtlpAgentAdapter.processOtlpEvent', () => { await run({ payload: [{ a: 1 }, { a: 2 }] }); expect(detectGitBranchMock).toHaveBeenCalledWith('/repo'); expect(detectGitRemoteRepoMock).toHaveBeenCalledWith('/repo'); - expect(forwarded()).toEqual([ - { a: 1, git_branch: 'main', repo_remote: 'org/repo', story_id: '', story_source: '', platform: 'fake-platform' }, - { a: 2, git_branch: 'main', repo_remote: 'org/repo', story_id: '', story_source: '', platform: 'fake-platform' }, - ]); + expect(forwarded()).toHaveLength(2); + expect(forwarded()[0]).toMatchObject({ a: 1, git_branch: 'main', repo_remote: 'org/repo', story_id: '', story_source: '', platform: 'fake-platform' }); + expect(forwarded()[1]).toMatchObject({ a: 2, git_branch: 'main', repo_remote: 'org/repo', story_id: '', story_source: '', platform: 'fake-platform' }); expect(forwardToSpool.mock.calls.map(([, name]) => name)).toEqual(['fake', 'fake']); }); @@ -152,11 +182,183 @@ describe('OtlpAgentAdapter.processOtlpEvent', () => { expect(forwarded()[0]['git_branch']).toBe(''); }); - it('does not wait for forwardToSpool (fire-and-forget)', async () => { - let resolveLate: () => void = () => {}; - forwardToSpool.mockImplementation(() => new Promise((resolve) => { resolveLate = resolve; })); - await run({ payload: [{ a: 1 }, { a: 2 }] }); - expect(forwardToSpool).toHaveBeenCalledTimes(2); - resolveLate(); + describe('context fields', () => { + it('stamps identity, project and version on every event, resolved for the hook cwd', async () => { + await run({ payload: [{ a: 1 }, { a: 2 }] }); + expect(getActiveProfileNameMock).toHaveBeenCalledWith('/repo'); + expect(configLoadMock).toHaveBeenCalledWith('/repo', { name: 'work' }); + expect(resolveIdentityMock).toHaveBeenCalledWith(expect.objectContaining({ apiUrl: 'https://api' }), '/repo'); + for (const record of forwarded()) { + expect(record).toMatchObject({ + user_email: 'dev@example.com', + developer_name: 'dev@example.com', + identity_source: 'jwt', + codemie_project_name: 'proj-1', + codemie_cli_version: '1.2.3', + }); + } + }); + + it('resolves the context once per invocation, not per event', async () => { + await run({ payload: [{ a: 1 }, { a: 2 }, { a: 3 }] }); + expect(resolveHookCredentialsMock).toHaveBeenCalledTimes(1); + expect(configLoadMock).toHaveBeenCalledTimes(1); + }); + + it('without credentials leaves the email empty and takes the identity from a lower tier', async () => { + resolveHookCredentialsMock.mockResolvedValue(null); + resolveIdentityMock.mockResolvedValue({ developerName: 'git-user@example.com', identitySource: 'git' }); + await run(); + expect(resolveIdentityMock).toHaveBeenCalledWith(null, '/repo'); + expect(forwarded()[0]).toMatchObject({ user_email: '', developer_name: 'git-user@example.com', identity_source: 'git' }); + }); + + it('omits the profile selector when there is no active profile', async () => { + getActiveProfileNameMock.mockResolvedValue(null); + await run(); + expect(configLoadMock).toHaveBeenCalledWith('/repo', undefined); + }); + + it('still forwards with empty context fields when resolution fails', async () => { + resolveHookCredentialsMock.mockRejectedValue(new Error('boom')); + await run(); + expect(forwarded()).toHaveLength(1); + expect(forwarded()[0]).toMatchObject({ + user_email: '', + developer_name: '', + identity_source: '', + codemie_project_name: '', + codemie_cli_version: '', + git_branch: 'main', + }); + }); + + it('uses an empty project name when the config cannot be loaded and an empty version when unknown', async () => { + configLoadMock.mockRejectedValue(new Error('bad config')); + getCurrentCliVersionMock.mockResolvedValue(null); + await run(); + expect(forwarded()[0]).toMatchObject({ codemie_project_name: '', codemie_cli_version: '', developer_name: 'dev@example.com' }); + }); + }); + + describe('built event', () => { + it('stamps schema_version, a unique event_id, session_id and the hook cwd', async () => { + await run({ payload: [{ session_id: 's1' }, { session_id: 's1' }] }); + const [first, second] = forwarded(); + expect(first).toMatchObject({ schema_version: 2, session_id: 's1', cwd: '/repo' }); + expect(typeof first['event_id']).toBe('string'); + expect(first['event_id']).not.toBe(second['event_id']); + }); + + it('coerces a missing session_id to an empty string', async () => { + await run({ payload: [{ a: 1 }] }); + expect(forwarded()[0]['session_id']).toBe(''); + }); + + it('puts the hook cwd on derived events too, not the event\'s own', async () => { + await run({ payload: [{ type: 'derived.event', session_id: 's1' }] }); + expect(forwarded()[0]['cwd']).toBe('/repo'); + }); + + it('resolves the type from resolveEventType and keeps an explicit one', async () => { + await run({ payload: [{ kind: 'start' }, { kind: 'start', type: 'explicit.type' }, { type: '' }] }); + expect(forwarded().map((r) => r['type'])).toEqual(['fake.start', 'explicit.type', 'fake.event']); + }); + + it('truncates prompt and tool fields and derives prompt_body', async () => { + await run({ + payload: [{ + prompt: 'p'.repeat(500), + tool_input: { text: 'x'.repeat(500) }, + tool_response: 'r'.repeat(500), + error: 'e'.repeat(500), + }], + }); + const record = forwarded()[0]; + expect(record['prompt']).toHaveLength(200); + expect(record['prompt_body']).toBe('p'.repeat(200)); + expect(record['tool_input']).toHaveLength(300); + expect(String(record['tool_input']).startsWith('{"text":"x')).toBe(true); + expect(record['tool_response']).toHaveLength(300); + expect(record['error']).toHaveLength(300); + }); + + it('does not add truncated fields to events that lack them, and prompt_body is empty', async () => { + await run({ payload: [{ a: 1 }] }); + const record = forwarded()[0]; + expect(record).not.toHaveProperty('prompt'); + expect(record).not.toHaveProperty('tool_input'); + expect(record['prompt_body']).toBe(''); + }); + + it('drops a nested raw from the event so it cannot bypass the limits', async () => { + await run({ payload: [{ raw: { huge: 'x'.repeat(1000) }, a: 1 }] }); + const raw = forwarded()[0]['raw'] as Record; + expect(raw).not.toHaveProperty('raw'); + expect(JSON.stringify(forwarded()[0])).not.toContain('huge'); + }); + + it('mirrors the full enriched event, context and event_id included, in raw', async () => { + await run({ payload: [{ session_id: 's1', prompt: 'hi' }] }); + const record = forwarded()[0]; + const { raw, ...outer } = record; + expect(raw).toEqual(outer); + expect(raw).toMatchObject({ + event_id: record['event_id'], + user_email: 'dev@example.com', + codemie_project_name: 'proj-1', + platform: 'fake-platform', + schema_version: 2, + }); + }); + + it('keeps a valid event timestamp, normalised to ISO', async () => { + await run({ payload: [{ timestamp: '2026-01-02T03:04:05+02:00' }] }); + expect(forwarded()[0]['timestamp']).toBe('2026-01-02T01:04:05.000Z'); + }); + + it('falls back to one hook time shared by all events for a missing or invalid timestamp', async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date('2026-05-06T07:08:09.000Z')); + try { + await run({ payload: [{ a: 1 }, { timestamp: 'not a date' }, { timestamp: '' }] }); + } finally { + vi.useRealTimers(); + } + expect(forwarded().map((r) => r['timestamp'])).toEqual(Array(3).fill('2026-05-06T07:08:09.000Z')); + }); + + it('serialises to identical bytes on a re-send, so the timestamp never changes between retries', async () => { + await run({ payload: [{ a: 1 }] }); + const bytes = JSON.stringify(forwarded()[0]); + await new Promise((resolve) => setTimeout(resolve, 5)); + expect(JSON.stringify(forwarded()[0])).toBe(bytes); + }); + }); + + describe('spool forwarding', () => { + it('awaits each spool write and sends events strictly in order', async () => { + const order: string[] = []; + let inFlight = 0; + forwardToSpool.mockImplementation(async (record: Record) => { + inFlight += 1; + expect(inFlight).toBe(1); + await new Promise((resolve) => setTimeout(resolve, 5)); + order.push(String(record['n'])); + inFlight -= 1; + }); + await run({ payload: [{ n: 1 }, { n: 2 }, { n: 3 }] }); + expect(order).toEqual(['1', '2', '3']); + }); + + it('does not resolve before the spool write has finished', async () => { + let done = false; + forwardToSpool.mockImplementation(async () => { + await new Promise((resolve) => setTimeout(resolve, 10)); + done = true; + }); + await run(); + expect(done).toBe(true); + }); }); }); diff --git a/src/agents/core/__tests__/hook-credentials.test.ts b/src/agents/core/__tests__/hook-credentials.test.ts new file mode 100644 index 000000000..30e64108c --- /dev/null +++ b/src/agents/core/__tests__/hook-credentials.test.ts @@ -0,0 +1,68 @@ +import { describe, it, expect, vi, beforeEach } from 'vitest'; + +const readStateMock = vi.fn(); +const getStoredCredentialsMock = vi.fn(); + +vi.mock('@/cli/commands/proxy/daemon-manager.js', () => ({ readState: readStateMock })); +vi.mock('@/providers/plugins/sso/sso.auth.js', () => ({ + CodeMieSSO: class { + getStoredCredentials = getStoredCredentialsMock; + }, +})); +vi.mock('@/utils/logger.js', () => ({ + logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, +})); + +const { resolveHookCredentials } = await import('../hook-credentials.js'); + +const creds = (apiUrl: string) => ({ cookies: {}, apiUrl, timestamp: 0 }); + +describe('resolveHookCredentials', () => { + beforeEach(() => { + vi.clearAllMocks(); + getStoredCredentialsMock.mockResolvedValue(null); + }); + + it('prefers the credentials stored for syncCodeMieUrl', async () => { + readStateMock.mockResolvedValue({ syncCodeMieUrl: 'https://sync', targetUrl: 'https://target' }); + getStoredCredentialsMock.mockImplementation(async (url: string) => creds(url)); + + await expect(resolveHookCredentials()).resolves.toMatchObject({ apiUrl: 'https://sync' }); + expect(getStoredCredentialsMock).toHaveBeenCalledTimes(1); + }); + + it('falls back to the targetUrl credentials when the sync URL is unset', async () => { + readStateMock.mockResolvedValue({ targetUrl: 'https://target' }); + getStoredCredentialsMock.mockImplementation(async (url: string) => creds(url)); + + await expect(resolveHookCredentials()).resolves.toMatchObject({ apiUrl: 'https://target' }); + expect(getStoredCredentialsMock).toHaveBeenCalledWith('https://target'); + }); + + it('falls back to the targetUrl credentials when the sync URL has none stored', async () => { + readStateMock.mockResolvedValue({ syncCodeMieUrl: 'https://sync', targetUrl: 'https://target' }); + getStoredCredentialsMock.mockImplementation(async (url: string) => (url === 'https://target' ? creds(url) : null)); + + await expect(resolveHookCredentials()).resolves.toMatchObject({ apiUrl: 'https://target' }); + }); + + it('returns null without daemon state', async () => { + readStateMock.mockResolvedValue(null); + await expect(resolveHookCredentials()).resolves.toBeNull(); + expect(getStoredCredentialsMock).not.toHaveBeenCalled(); + }); + + it('returns null when nothing is stored', async () => { + readStateMock.mockResolvedValue({ syncCodeMieUrl: 'https://sync', targetUrl: 'https://target' }); + await expect(resolveHookCredentials()).resolves.toBeNull(); + }); + + it('never throws', async () => { + readStateMock.mockRejectedValue(new Error('state unreadable')); + await expect(resolveHookCredentials()).resolves.toBeNull(); + + readStateMock.mockResolvedValue({ targetUrl: 'https://target' }); + getStoredCredentialsMock.mockRejectedValue(new Error('keychain locked')); + await expect(resolveHookCredentials()).resolves.toBeNull(); + }); +}); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts b/src/agents/core/__tests__/identity-resolver.test.ts similarity index 92% rename from src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts rename to src/agents/core/__tests__/identity-resolver.test.ts index 796fffe99..43fa3ea37 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/identity.test.ts +++ b/src/agents/core/__tests__/identity-resolver.test.ts @@ -15,7 +15,7 @@ describe('resolveIdentity', () => { }); it('resolves from the jwt tier when the token carries a valid email claim', async () => { - const { resolveIdentity } = await import('../identity.js'); + const { resolveIdentity } = await import('../identity-resolver.js'); const credentials: JWTCredentials = { token: makeJwt({ email: 'dev@example.com' }), @@ -36,7 +36,7 @@ describe('resolveIdentity', () => { return { code: 1, stdout: '', stderr: '', signal: null }; }); - const { resolveIdentity } = await import('../identity.js'); + const { resolveIdentity } = await import('../identity-resolver.js'); // Empty token: isJWTCredentials() still matches the shape, but decodeJwtClaims // yields no usable email, so the jwt tier is a miss. @@ -59,7 +59,7 @@ describe('resolveIdentity', () => { profiles: {}, }); - const { resolveIdentity } = await import('../identity.js'); + const { resolveIdentity } = await import('../identity-resolver.js'); const credentials: JWTCredentials = { token: '', apiUrl: '' }; diff --git a/src/agents/core/hook-credentials.ts b/src/agents/core/hook-credentials.ts new file mode 100644 index 000000000..b08ccd982 --- /dev/null +++ b/src/agents/core/hook-credentials.ts @@ -0,0 +1,33 @@ +import { readState } from '@/cli/commands/proxy/daemon-manager.js'; +import type { SSOCredentials } from '@/providers/core/types.js'; +import { CodeMieSSO } from '@/providers/plugins/sso/sso.auth.js'; +import { logger } from '@/utils/logger.js'; + +/** + * Credentials the OTLP forwarder daemon will use, resolved in the short-lived hook process + * so identity fields can be stamped at hook time. + * + * Mirrors the daemon's choice in `SSOProxy` (`sso.proxy.ts`): the analytics-sync credentials + * first, then the target API credentials (`syncCredentials || credentials`). Daemon state + * carries both URLs; `inspect-desktop.ts` uses the same `syncCodeMieUrl || targetUrl` pairing. + * Keep the three in step. Only SSO is reachable here: the OTLP daemon is spawned without + * `--auth-method`, so it always runs as `sso`. + * + * Side effect: `getStoredCredentials` clears expired SSO credentials, as it does at daemon start. + * + * Never throws; `null` makes identity fall through to the git, codemie_cli and os tiers. + */ +export async function resolveHookCredentials(): Promise { + try { + const state = await readState(); + const sso = new CodeMieSSO(); + const syncCredentials = state?.syncCodeMieUrl ? await sso.getStoredCredentials(state.syncCodeMieUrl) : null; + if (syncCredentials) { + return syncCredentials; + } + return state?.targetUrl ? await sso.getStoredCredentials(state.targetUrl) : null; + } catch (error) { + logger.debug(`[hook-credentials] credential lookup failed: ${error instanceof Error ? error.message : String(error)}`); + return null; + } +} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts b/src/agents/core/identity-resolver.ts similarity index 93% rename from src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts rename to src/agents/core/identity-resolver.ts index 1b0de6fdc..81d9c3bd3 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/identity.ts +++ b/src/agents/core/identity-resolver.ts @@ -34,11 +34,14 @@ export function decodeJwtClaims(token: string): Record { * Pull `email` straight off a JWT credential's claims, or off the SSO * session's `codemie_access_token` cookie claims (`email`, falling back to * `preferred_username`). Shared by the tier-1 `jwt` identity tier below and - * by the forwarder's plain `userEmail` field — both resolve the same claim. + * by the adapter's plain `user_email` field — both resolve the same claim. */ export function resolveEmailFromCredentials( - credentials: SSOCredentials | JWTCredentials + credentials: SSOCredentials | JWTCredentials | null ): string { + if (!credentials) { + return ''; + } if (isJWTCredentials(credentials)) { const claims = decodeJwtClaims(credentials.token); if (typeof claims['email'] === 'string' && claims['email']) { @@ -107,7 +110,7 @@ async function resolveCodemieCliIdentity(): Promise { } /** - * Tier 4 — os: the OS-reported username for the daemon process. Practically + * Tier 4 — os: the OS-reported username of the process resolving the identity. Practically * never empty, but guarded anyway since some sandboxed environments can make * `os.userInfo()` throw. */ @@ -128,7 +131,7 @@ function resolveOsIdentity(): string { * Never throws — every tier swallows its own failures internally. */ export async function resolveIdentity( - credentials: SSOCredentials | JWTCredentials, + credentials: SSOCredentials | JWTCredentials | null, cwd: string ): Promise { try { diff --git a/src/agents/core/types.ts b/src/agents/core/types.ts index e4ab58c5e..df966688b 100644 --- a/src/agents/core/types.ts +++ b/src/agents/core/types.ts @@ -733,7 +733,6 @@ export enum AgentAdapterType { export interface OtlpAdapterDeps { ensureOtlpProxy: (agentName: string) => Promise; - /** Fire-and-forget spool write; never throws. */ forwardOtlpEventToSpool: (event: Record, agentName: string) => Promise; } diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index 7ef5e846f..52f432933 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -9,7 +9,11 @@ vi.mock('@/utils/processes.js', () => ({ detectGitRemoteRepo: vi.fn(async () => 'org/repo'), })); vi.mock('@/providers/plugins/sso/sso.auth-gate.js', () => ({ ensureCodeMieSsoAuth: vi.fn() })); -vi.mock('@/utils/config.js', () => ({ ConfigLoader: { load: vi.fn(async () => ({})) } })); +vi.mock('@/utils/config.js', () => ({ + ConfigLoader: { load: vi.fn(async () => ({})), getActiveProfileName: vi.fn(async () => null) }, +})); +vi.mock('@/agents/core/hook-credentials.js', () => ({ resolveHookCredentials: vi.fn(async () => null) })); +vi.mock('@/utils/cli-updater.js', () => ({ getCurrentCliVersion: vi.fn(async () => '9.9.9') })); vi.mock('@/utils/logger.js', () => ({ logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, })); @@ -310,3 +314,60 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { await expect(plugin.processOtlpEvent('not json', { ensureOtlpProxy, forwardOtlpEventToSpool })).rejects.toThrow(); }); }); + +describe('ClaudeCodeOtlpPlugin wire type', () => { + const plugin = new ClaudeCodeOtlpPlugin(); + const ensureOtlpProxy = vi.fn(async () => {}); + + beforeEach(() => { + vi.clearAllMocks(); + vi.mocked(isProjectTracked).mockResolvedValue(true); + resolveClientVersionMock.mockResolvedValue('2.1.23'); + collectMainTranscriptEventsMock.mockResolvedValue([]); + collectSubagentTranscriptEventsMock.mockResolvedValue([]); + findSubagentFilesMock.mockResolvedValue([]); + }); + + const typeFor = async (name: string): Promise => { + await plugin.processOtlpEvent(event(name), { ensureOtlpProxy, forwardOtlpEventToSpool }); + return vi.mocked(forwardOtlpEventToSpool).mock.calls[0][0]['type']; + }; + + it.each([ + ['SessionStart', 'agent.session.start'], + ['Stop', 'agent.session.stop'], + ['StopFailure', 'agent.turn.error'], + ['SessionEnd', 'agent.session.end'], + ['PreToolUse', 'agent.tool.start'], + ['PostToolUse', 'agent.tool.end'], + ['PostToolUseFailure', 'agent.tool.error'], + ['SubagentStart', 'agent.subagent.start'], + ['SubagentStop', 'agent.subagent.stop'], + ['PreCompact', 'agent.session.compact'], + ['Notification', 'agent.notification'], + ])('maps %s to %s', async (hookName, expectedType) => { + expect(await typeFor(hookName)).toBe(expectedType); + }); + + it('maps UserPromptSubmit to agent.prompt.submit', async () => { + vi.mocked(ensureCodeMieSsoAuth).mockResolvedValue({ ok: true } as never); + await plugin.processOtlpEvent( + JSON.stringify({ session_id: 's', transcript_path: '', cwd: '/x', hook_event_name: 'UserPromptSubmit', prompt: 'hi' }), + { ensureOtlpProxy, forwardOtlpEventToSpool } + ); + expect(vi.mocked(forwardOtlpEventToSpool).mock.calls[0][0]['type']).toBe('agent.prompt.submit'); + }); + + it('falls back to agent.event for an unmapped hook name', () => { + const resolve = (plugin as unknown as { resolveEventType(e: Record): string }).resolveEventType; + expect(resolve.call(plugin, { hook_event_name: 'SomethingNew' })).toBe('agent.event'); + expect(resolve.call(plugin, {})).toBe('agent.event'); + }); + + it('keeps the explicit type of a derived event', async () => { + collectMainTranscriptEventsMock.mockResolvedValue([{ type: 'agent.usage.request', session_id: 's' }]); + await plugin.processOtlpEvent(event('Stop'), { ensureOtlpProxy, forwardOtlpEventToSpool }); + const types = vi.mocked(forwardOtlpEventToSpool).mock.calls.map(([record]) => record['type']); + expect(types).toEqual(['agent.session.stop', 'agent.usage.request']); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index 7bccbb325..b496fc3b5 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -24,6 +24,21 @@ import { import { findSubagentFiles } from './transcript/subagent-usage.js'; import { resolveClientVersion } from './client-version-cache.js'; +const HOOK_EVENT_TYPE_MAP: Record = { + SessionStart: 'agent.session.start', + Stop: 'agent.session.stop', + StopFailure: 'agent.turn.error', + SessionEnd: 'agent.session.end', + UserPromptSubmit: 'agent.prompt.submit', + PreToolUse: 'agent.tool.start', + PostToolUse: 'agent.tool.end', + PostToolUseFailure: 'agent.tool.error', + SubagentStart: 'agent.subagent.start', + SubagentStop: 'agent.subagent.stop', + PreCompact: 'agent.session.compact', + Notification: 'agent.notification', +}; + export class ClaudeCodeOtlpPlugin extends OtlpAgentAdapter { public readonly name = CLAUDE_CODE_OTLP_AGENT_NAME; @@ -44,7 +59,7 @@ export class ClaudeCodeOtlpPlugin extends OtlpAgentAdapter> { + protected async resolveAgentFields(): Promise> { return { platform: 'claude-code', entrypoint: process.env.CLAUDE_CODE_ENTRYPOINT ?? '', @@ -52,6 +67,10 @@ export class ClaudeCodeOtlpPlugin extends OtlpAgentAdapter): string { + return HOOK_EVENT_TYPE_MAP[String(event['hook_event_name'] ?? '')] ?? 'agent.event'; + } + protected async evaluate(hookInput: HookInput): Promise { if (hookInput.hook_event_name === 'UserPromptSubmit') { return await this.onUserPromptSubmit(hookInput); diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts index 2eccc5cd5..d1dfd8207 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts @@ -6,8 +6,9 @@ * `extractNamedInvocations()` to produce the {@link NamedInvocationCounts} this builder consumes. * This module only derives the final event shape from those inputs — it never reads a transcript. * - * `event_id`/`schema_version`/`client_version`/`codemie_cli_version` are stamped later, - * daemon-side (`mapHookRecords()`); the output carries only an explicit `type`. + * `event_id`, `schema_version`, `client_version` and `codemie_cli_version` are no longer stamped + * daemon-side; the adapter base class stamps them at hook time (`client_version` via + * `resolveAgentFields`). The output carries only an explicit `type`. * * `api_calls` is omitted here: no input carries a request count. The orchestrator, which owns * the full set of `agent.usage.request` records, merges it in afterward. diff --git a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts index e94375f92..1af1f1cac 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts @@ -93,7 +93,7 @@ export async function findSubagentFiles(mainTranscriptPath: string): Promise { diff --git a/src/agents/plugins/utils.ts b/src/agents/plugins/utils.ts index 9bc310e68..953d673b2 100644 --- a/src/agents/plugins/utils.ts +++ b/src/agents/plugins/utils.ts @@ -1,14 +1,13 @@ -import { randomUUID } from 'node:crypto'; import { OtlpHookSpoolData } from '@/providers/plugins/sso/proxy/plugins/otlp.plugin.js'; import { readState } from '../../cli/commands/proxy/daemon-manager.js'; import { logger } from '../../utils/logger.js'; /** - * Fire-and-forget forward of a raw hook event to the local proxy - * daemon's analytics spool endpoint. + * Forward of a complete wire-ready event to the local proxy daemon's + * analytics spool endpoint. The event is stored and forwarded as is. * * - Calls readState() to get daemon URL and gateway key - * - POSTs { agentName, timestamp, raw: rawInput } to /v1/analytics/hooks + * - POSTs { agentName, hookEvent: JSON.stringify(event) } to /v1/analytics/hooks * - Uses 1000ms timeout and swallows all errors * - Never throws, never affects hook's exit code */ @@ -26,12 +25,8 @@ export async function forwardOtlpEventToSpool(event: Record, ag const body: OtlpHookSpoolData = { agentName, - timestamp: Date.now(), - raw: JSON.stringify({ - ...event, - event_id: randomUUID() - }) - } + hookEvent: JSON.stringify(event), + }; try { await fetch(`${state.url}/v1/analytics/hooks`, { diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts index d1be60fec..1109efe22 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/__tests__/forwarder.test.ts @@ -1,267 +1,164 @@ -import { describe, it, expect, vi, beforeEach } from 'vitest'; - -const execSyncMock = vi.fn(); - -vi.mock('node:child_process', () => ({ - execSync: execSyncMock, +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; + +const snapshotPendingHookRecordsMock = vi.fn(); +const snapshotPendingBytesMock = vi.fn(); +const advanceCursorMock = vi.fn(); +const markSessionEndedMock = vi.fn(); +const readStateMock = vi.fn(); +const fetchMock = vi.fn(); + +vi.mock('../spool-io.js', () => ({ + snapshotPendingHookRecords: snapshotPendingHookRecordsMock, + snapshotPendingBytes: snapshotPendingBytesMock, +})); +vi.mock('../session-status.js', () => ({ + advanceCursor: advanceCursorMock, + markSessionEnded: markSessionEndedMock, +})); +vi.mock('@/cli/commands/proxy/daemon-manager.js', () => ({ readState: readStateMock })); +vi.mock('@/utils/logger.js', () => ({ + logger: { info: vi.fn(), warn: vi.fn(), error: vi.fn(), debug: vi.fn() }, })); -beforeEach(() => { - execSyncMock.mockReset(); - execSyncMock.mockReturnValue('0.15.6'); -}); - -interface MappedRecord { - type: string; - session_id: string; - schema_version: number; - event_id: string; - codemie_cli_version: string; - story_id?: string; - story_source?: string; - git_branch?: string; - repo_remote?: string; - prompt_body?: string; - developer_name?: string; - identity_source?: string; - platform?: string; - client_version?: string; -} +const credentials = { cookies: { codemie_access_token: 'tok' }, apiUrl: 'https://api', timestamp: 0 }; -function buildHookRecord( - hookEventName: string, - sessionId: string, - extra: Record = {}, - eventId: string = 'default-event-id' -): string { - return JSON.stringify({ - agentName: 'claude', - raw: JSON.stringify({ - hook_event_name: hookEventName, - session_id: sessionId, - cwd: '', - event_id: eventId, - ...extra, - }), - timestamp: Date.now(), - }); +/** A spool line as the daemon stores it: the envelope around the wire event built at hook time. */ +function spoolLine(wireEvent: Record): string { + return JSON.stringify({ agentName: 'claude-code-otlp', hookEvent: JSON.stringify(wireEvent) }); } -describe('resolveCodemieCliVersion', () => { - beforeEach(() => { - execSyncMock.mockReset(); - execSyncMock.mockReturnValue('0.15.6'); - }); +describe('unwrapHookRecords', () => { + it('forwards each raw wire event byte for byte, newline-delimited', async () => { + const { unwrapHookRecords } = await import('../forwarder.js'); + const first = { type: 'agent.prompt.submit', session_id: 's1', timestamp: '2026-01-01T00:00:00.000Z', zeta: 1, alpha: { b: 2 } }; + const second = { type: 'agent.tool.end', session_id: 's1' }; - it('reads the installed CLI version via `codemie --version` and strips the semver', async () => { - const { resolveCodemieCliVersion } = await import('../forward-context.js'); + const payload = unwrapHookRecords([spoolLine(first), spoolLine(second)]); - expect(resolveCodemieCliVersion()).toBe('0.15.6'); - expect(execSyncMock).toHaveBeenCalledWith('codemie --version', { - encoding: 'utf8', - stdio: ['pipe', 'pipe', 'pipe'], - }); + expect(payload.ndjson).toBe(`${JSON.stringify(first)}\n${JSON.stringify(second)}\n`); + expect(payload.malformed).toBe(0); }); - it('falls back to an empty string when `codemie --version` throws', async () => { - execSyncMock.mockImplementation(() => { - throw new Error('ENOENT'); - }); + it('does not synthesize or change any field', async () => { + const { unwrapHookRecords } = await import('../forwarder.js'); + const bare = { session_id: 's1', hook_event_name: 'SessionEnd' }; - const { resolveCodemieCliVersion } = await import('../forward-context.js'); - expect(resolveCodemieCliVersion()).toBe(''); - }); -}); + const payload = unwrapHookRecords([spoolLine(bare)]); -describe('mapHookRecords', () => { - it('stamps schema_version, event_id, and codemie_cli_version on every mapped record', async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: 'user@example.com', - }; - - const record1 = buildHookRecord('SessionStart', 'sid1', {}, 'event-id-1'); - const record2 = buildHookRecord('Stop', 'sid1', {}, 'event-id-2'); - - const payload = await mapHookRecords([record1, record2], ctx); - const lines = payload.ndjson - .trim() - .split('\n') - .map((line) => JSON.parse(line) as MappedRecord); - - expect(lines).toHaveLength(2); - expect(lines[0].schema_version).toBe(2); - expect(lines[1].schema_version).toBe(2); - expect(lines[0].event_id).not.toBe(lines[1].event_id); - expect(typeof lines[0].codemie_cli_version).toBe('string'); - expect(lines[0].codemie_cli_version.length).toBeGreaterThan(0); - expect(lines[0].codemie_cli_version).toBe(lines[1].codemie_cli_version); + expect(JSON.parse(payload.ndjson.trim())).toEqual(bare); }); - it('passes the event_id already stamped at spool-write time straight through unchanged', async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - }; - - const record = buildHookRecord('SessionStart', 'sid1', {}, 'stamped-event-id-abc'); + it('counts and skips malformed lines, keeping the good ones', async () => { + const { unwrapHookRecords } = await import('../forwarder.js'); + const good = { type: 'agent.tool.end', session_id: 's1' }; + const badEnvelope = 'not json'; + const badRaw = JSON.stringify({ agentName: 'x', hookEvent: 'not json' }); + const noHookEvent = JSON.stringify({ agentName: 'x' }); - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + const payload = unwrapHookRecords([badEnvelope, spoolLine(good), badRaw, noHookEvent]); - expect(line.event_id).toBe('stamped-event-id-abc'); + expect(payload.malformed).toBe(3); + expect(payload.ndjson).toBe(`${JSON.stringify(good)}\n`); }); - it('prefers an explicit hookEvent.type over the HOOK_EVENT_TYPE_MAP lookup', async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - }; - - // PostToolUse normally maps to 'agent.tool.end', but an explicit `type` - // field on the raw hook payload (as a later-task synthetic record would - // carry) must win. - const record = buildHookRecord('PostToolUse', 'sid1', { type: 'agent.custom.synthetic' }); + it('returns an empty payload when nothing is usable', async () => { + const { unwrapHookRecords } = await import('../forwarder.js'); + expect(unwrapHookRecords(['nope'])).toEqual({ ndjson: '', containsSessionEnd: false, malformed: 1 }); + }); - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + it('detects a session end from the wire type, not from hook_event_name', async () => { + const { unwrapHookRecords } = await import('../forwarder.js'); - expect(line.type).toBe('agent.custom.synthetic'); + expect(unwrapHookRecords([spoolLine({ type: 'agent.session.end', session_id: 's1' })]).containsSessionEnd).toBe(true); + expect(unwrapHookRecords([spoolLine({ type: 'agent.session.stop', hook_event_name: 'SessionEnd' })]).containsSessionEnd).toBe(false); + expect(unwrapHookRecords([spoolLine({ hook_event_name: 'SessionEnd' })]).containsSessionEnd).toBe(false); }); +}); - it('leaves existing-event type resolution unchanged when type is absent', async () => { - const { mapHookRecords } = await import('../forwarder.js'); +describe('forwardSession (hooks)', () => { + beforeEach(() => { + vi.resetModules(); + vi.clearAllMocks(); + vi.stubGlobal('fetch', fetchMock); + readStateMock.mockResolvedValue({ url: 'http://127.0.0.1:1', targetUrl: 'https://api.example.com' }); + snapshotPendingHookRecordsMock.mockResolvedValue({ + records: [spoolLine({ type: 'agent.session.end', session_id: 's1' })], + cursor: 10, + byteLength: 100, + }); + }); - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - }; + afterEach(() => { + vi.unstubAllGlobals(); + }); - const record = buildHookRecord('PostToolUse', 'sid1'); + it('POSTs the ndjson to the target API, advances the cursor and marks the session ended', async () => { + fetchMock.mockResolvedValue({ ok: true, status: 200 }); + const { forwardSession } = await import('../forwarder.js'); - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + await forwardSession('s1', true, credentials); - expect(line.type).toBe('agent.tool.end'); + expect(fetchMock).toHaveBeenCalledTimes(1); + const [url, init] = fetchMock.mock.calls[0]; + expect(String(url)).toBe('https://api.example.com/v1/analytics/cli-analytics/event-hooks'); + expect(init.body).toBe(`${JSON.stringify({ type: 'agent.session.end', session_id: 's1' })}\n`); + expect(advanceCursorMock).toHaveBeenCalledWith('s1', 'hooks', 110); + expect(markSessionEndedMock).toHaveBeenCalledWith('s1'); }); - it('passes git_branch/repo_remote/story_id/story_source through from the incoming record', async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - }; - - const record = buildHookRecord('UserPromptSubmit', 'sid1', { - prompt: 'story: ABC-1 is ignored here, the forwarder does not resolve stories', - git_branch: 'feature/x', - repo_remote: 'org/repo', - story_id: 'EPMCDME-999', - story_source: 'marker', - }); + it('retries once on 401 and succeeds', async () => { + fetchMock.mockResolvedValueOnce({ ok: false, status: 401 }).mockResolvedValueOnce({ ok: true, status: 200 }); + const { forwardSession } = await import('../forwarder.js'); - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + await forwardSession('s1', true, credentials); - expect(line.git_branch).toBe('feature/x'); - expect(line.repo_remote).toBe('org/repo'); - expect(line.story_id).toBe('EPMCDME-999'); - expect(line.story_source).toBe('marker'); + expect(fetchMock).toHaveBeenCalledTimes(2); + expect(fetchMock.mock.calls[1][1].body).toBe(fetchMock.mock.calls[0][1].body); + expect(advanceCursorMock).toHaveBeenCalledWith('s1', 'hooks', 110); }); - it("defaults git_branch/repo_remote/story_id/story_source to '' and never resolves them from the daemon env", async () => { - const { mapHookRecords } = await import('../forwarder.js'); - vi.stubEnv('SDLC_ANALYTICS_STORY_ID', 'FROM-DAEMON-1'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - }; - - try { - const payload = await mapHookRecords([buildHookRecord('UserPromptSubmit', 'sid1', { prompt: 'story: ABC-1' })], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; - - expect(line.git_branch).toBe(''); - expect(line.repo_remote).toBe(''); - expect(line.story_id).toBe(''); - expect(line.story_source).toBe(''); - } finally { - vi.unstubAllEnvs(); - } + it('leaves the cursor untouched and marks credentials stale after two 403s', async () => { + fetchMock.mockResolvedValue({ ok: false, status: 403 }); + const { forwardSession } = await import('../forwarder.js'); + const { areCredentialsStale } = await import('../auth-state.js'); + + await forwardSession('s1', true, credentials); + + expect(advanceCursorMock).not.toHaveBeenCalled(); + expect(markSessionEndedMock).not.toHaveBeenCalled(); + expect(areCredentialsStale()).toBe(true); }); - it('passes agent-baked common fields (platform/client_version) through onto the mapped record without any agent-specific lookup', async () => { - const { mapHookRecords } = await import('../forwarder.js'); - - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - }; - - // Simulates what the plugin now bakes in hook-side before ever reaching the spool — - // the forwarder needs no agent-specific knowledge to pass these through, just the - // `...limited` spread like every other hook-native field. - const record = JSON.stringify({ - agentName: 'claude-code-otlp', - raw: JSON.stringify({ - hook_event_name: 'Stop', - session_id: 'sid1', - cwd: '', - platform: 'claude-code', - client_version: '1.2.3', - }), - timestamp: Date.now(), - }); + it('leaves the cursor untouched on a non-auth failure', async () => { + fetchMock.mockResolvedValue({ ok: false, status: 500 }); + const { forwardSession } = await import('../forwarder.js'); - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + await forwardSession('s1', true, credentials); - expect(line.platform).toBe('claude-code'); - expect(line.client_version).toBe('1.2.3'); + expect(advanceCursorMock).not.toHaveBeenCalled(); }); - it('carries ctx.identity through onto developer_name/identity_source for a non-jwt tier', async () => { - const { mapHookRecords } = await import('../forwarder.js'); + it('advances the cursor without a request when every line is malformed', async () => { + snapshotPendingHookRecordsMock.mockResolvedValue({ records: ['nope'], cursor: 5, byteLength: 4 }); + const { forwardSession } = await import('../forwarder.js'); - // Pre-seeding ctx.identity (mimicking what the per-tick identity cache would look like once resolved) - // with a non-jwt tier result proves the wiring from ctx.identity onto the mapped record, - // independent of the identity-chain's own resolution logic (covered by identity.test.ts). - const ctx = { - credentials: { token: '', apiUrl: '' }, - baseUrl: '', - projectName: 'proj', - userEmail: '', - identity: { developerName: 'git-user@example.com', identitySource: 'git' as const }, - }; + await forwardSession('s1', true, credentials); - const record = buildHookRecord('Stop', 'sid1'); + expect(fetchMock).not.toHaveBeenCalled(); + expect(advanceCursorMock).toHaveBeenCalledWith('s1', 'hooks', 9); + }); + + it('does not mark the session ended for a batch without a session end', async () => { + snapshotPendingHookRecordsMock.mockResolvedValue({ + records: [spoolLine({ type: 'agent.tool.end', session_id: 's1' })], + cursor: 0, + byteLength: 10, + }); + fetchMock.mockResolvedValue({ ok: true, status: 200 }); + const { forwardSession } = await import('../forwarder.js'); - const payload = await mapHookRecords([record], ctx); - const line = JSON.parse(payload.ndjson.trim()) as MappedRecord; + await forwardSession('s1', true, credentials); - expect(line.developer_name).toBe('git-user@example.com'); - expect(line.identity_source).toBe('git'); + expect(markSessionEndedMock).not.toHaveBeenCalled(); }); }); diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts deleted file mode 100644 index 14fb15b5e..000000000 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forward-context.ts +++ /dev/null @@ -1,54 +0,0 @@ -/** - * Per-forward-tick context: the CLI version banner, and the identity resolution glue - * the hook-record mapping step calls once per batch. Split out of `forwarder.ts` to keep - * that module under the documented 500-line structure cap (code-quality.md). - */ - -import { execSync } from 'node:child_process'; -import type { SSOCredentials, JWTCredentials } from '@/providers/core/types.js'; -import { resolveIdentity, type IdentitySource } from './identity.js'; - -export interface ForwardContext { - credentials: SSOCredentials | JWTCredentials; - baseUrl: string; - projectName: string; - userEmail: string; - /** Per-session developer-identity cache, resolved once */ - identity?: { developerName?: string; identitySource?: IdentitySource }; -} - -/** The installed CodeMie CLI version, resolved once at import time. */ -export function resolveCodemieCliVersion(): string { - try { - const output = execSync('codemie --version', { - encoding: 'utf8', - stdio: ['pipe', 'pipe', 'pipe'], - }).trim(); - const versionMatch = output.match(/(\d+\.\d+\.\d+)/); - return versionMatch ? versionMatch[1] : output; - } catch { - return ''; - } -} - -/** - * Resolve and cache this forward tick's developer identity once, from the first record that - * carries a real `cwd`. An empty `cwd` (every synthetic transcript-derived record) is skipped - * rather than cached, mirroring the same empty-`cwd` guard every other per-tick cache uses — - * otherwise the first such record in a batch would permanently cache the cwd-less (and - * therefore less accurate) result for every later record in the same tick. - */ -export async function resolveDeveloperIdentity(ctx: ForwardContext, cwd: string): Promise { - if (!cwd) { - return; - } - if (!ctx.identity) { - ctx.identity = {}; - } - if (ctx.identity.developerName !== undefined) { - return; - } - const { developerName, identitySource } = await resolveIdentity(ctx.credentials, cwd); - ctx.identity.developerName = developerName; - ctx.identity.identitySource = identitySource; -} diff --git a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts index af0d128b9..673d72ac4 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp-spool/forwarder.ts @@ -12,29 +12,8 @@ import { import { OtlpHookSpoolData } from '../otlp.plugin.js'; import { snapshotPendingBytes, snapshotPendingHookRecords } from './spool-io.js'; import { areCredentialsStale, markCredentialsStale } from './auth-state.js'; -import { resolveEmailFromCredentials } from './identity.js'; -import { - type ForwardContext, - resolveCodemieCliVersion, - resolveDeveloperIdentity, -} from './forward-context.js'; - -const CODEMIE_CLI_VERSION = resolveCodemieCliVersion(); - -const HOOK_EVENT_TYPE_MAP: Record = { - SessionStart: 'agent.session.start', - Stop: 'agent.session.stop', - StopFailure: 'agent.turn.error', - SessionEnd: 'agent.session.end', - UserPromptSubmit: 'agent.prompt.submit', - PreToolUse: 'agent.tool.start', - PostToolUse: 'agent.tool.end', - PostToolUseFailure: 'agent.tool.error', - SubagentStart: 'agent.subagent.start', - SubagentStop: 'agent.subagent.stop', - PreCompact: 'agent.session.compact', - Notification: 'agent.notification', -}; + +const SESSION_END_EVENT_TYPE = 'agent.session.end'; const OTEL_ENDPOINTS: Record = { logs: CODEMIE_ENDPOINTS.CLI_ANALYTICS_LOGS, @@ -42,8 +21,6 @@ const OTEL_ENDPOINTS: Record = { traces: CODEMIE_ENDPOINTS.CLI_ANALYTICS_TRACES, }; -const MAX_PROMPT_CHARS = 200; -const MAX_TOOL_FIELD_CHARS = 300; const FORWARD_TIMEOUT_MS = 20_000; type SendResult = 'ok' | 'failed' | 'auth-expired'; @@ -132,73 +109,7 @@ async function send( } } -/* --------------------------------------------------------------- mapping --- */ - -function hookEventType(hookName: string, event: Record): string { - const explicitType = event['type']; - if (typeof explicitType === 'string' && explicitType.length > 0) { - return explicitType; - } - if (hookName === 'PreToolUse') { - return event['input'] && (event['input'] as Record)['denied'] - ? 'agent.tool.denied' - : 'agent.tool.start'; - } - return HOOK_EVENT_TYPE_MAP[hookName] ?? 'agent.event'; -} - -function boundedText(value: unknown, maxChars: number): string { - if (value === undefined || value === null) { - return ''; - } - const text = - typeof value === 'string' - ? value - : (() => { - try { - return JSON.stringify(value) ?? String(value); - } catch { - return String(value); - } - })(); - return text.slice(0, maxChars); -} - -/** - * Prefer the event's own `timestamp` (e.g. the transcript line time on `agent.usage.request`) - * over the spool-write time, which only reflects when the hook ran. Falls back when the - * event has none or an unparseable one. - */ -function resolveEventTimestamp(eventTimestamp: unknown, spoolTimestamp: number): string { - if (typeof eventTimestamp === 'string' && eventTimestamp.length > 0) { - const parsed = new Date(eventTimestamp); - if (!Number.isNaN(parsed.getTime())) { - return parsed.toISOString(); - } - } - return new Date(spoolTimestamp).toISOString(); -} - -function stringOrEmpty(value: unknown): string { - return typeof value === 'string' ? value : ''; -} - -function limitHookPayload(hookEvent: Record): Record { - const limited: Record = { ...hookEvent }; - - if (Object.prototype.hasOwnProperty.call(hookEvent, 'prompt')) { - limited.prompt = boundedText(hookEvent.prompt, MAX_PROMPT_CHARS); - } - for (const field of ['tool_input', 'tool_response', 'error']) { - if (Object.prototype.hasOwnProperty.call(hookEvent, field)) { - limited[field] = boundedText(hookEvent[field], MAX_TOOL_FIELD_CHARS); - } - } - // Prevent a nested/raw copy from bypassing the limits. - delete limited.raw; - - return limited; -} +/* ------------------------------------------------------------- unwrapping --- */ interface HookPayload { ndjson: string; @@ -206,20 +117,21 @@ interface HookPayload { malformed: number; } -export async function mapHookRecords( - records: string[], - ctx: ForwardContext -): Promise { - const mapped: string[] = []; +/** + * Unwraps the spool envelopes. `hookEvent` is already the final wire event, built at hook time + * by the adapter, so it is forwarded byte for byte: same bytes on every retry. + */ +export function unwrapHookRecords(records: string[]): HookPayload { + const unwrapped: string[] = []; let containsSessionEnd = false; let malformed = 0; for (const record of records) { - let spoolData: OtlpHookSpoolData; - let hookEvent: Record; + let serialized: string; + let wireEvent: Record; try { - spoolData = JSON.parse(record) as OtlpHookSpoolData; - hookEvent = JSON.parse(spoolData.raw) as Record; + serialized = (JSON.parse(record) as OtlpHookSpoolData).hookEvent; + wireEvent = JSON.parse(serialized) as Record; } catch { // Complete but unusable record: dropped deliberately. Its bytes are still // acknowledged with the batch so the cursor can never get stuck on it. @@ -227,44 +139,14 @@ export async function mapHookRecords( continue; } - const hookName = String(hookEvent['hook_event_name'] ?? ''); - if (hookName === 'SessionEnd') { + if (wireEvent['type'] === SESSION_END_EVENT_TYPE) { containsSessionEnd = true; } - - const cwd = String(hookEvent['cwd'] ?? ''); - await resolveDeveloperIdentity(ctx, cwd); - - const limited = limitHookPayload(hookEvent); - const type = hookEventType(hookName, hookEvent); - const sessionId = String(hookEvent['session_id'] ?? ''); - mapped.push( - JSON.stringify({ - ...limited, - type, - session_id: sessionId, - timestamp: resolveEventTimestamp(hookEvent['timestamp'], spoolData.timestamp), - user_email: ctx.userEmail, - developer_name: ctx.identity?.developerName ?? '', - identity_source: ctx.identity?.identitySource ?? '', - // Resolved at hook time by the adapter (OtlpAgentAdapter); passed through as-is. - git_branch: stringOrEmpty(hookEvent['git_branch']), - repo_remote: stringOrEmpty(hookEvent['repo_remote']), - story_id: stringOrEmpty(hookEvent['story_id']), - story_source: stringOrEmpty(hookEvent['story_source']), - codemie_project_name: ctx.projectName, - cwd, - prompt_body: boundedText(hookEvent['prompt'], MAX_PROMPT_CHARS), - raw: limited, - schema_version: 2, - event_id: hookEvent['event_id'] as string, - codemie_cli_version: CODEMIE_CLI_VERSION, - }) - ); + unwrapped.push(serialized); } return { - ndjson: mapped.length > 0 ? `${mapped.join('\n')}\n` : '', + ndjson: unwrapped.length > 0 ? `${unwrapped.join('\n')}\n` : '', containsSessionEnd, malformed, }; @@ -272,34 +154,26 @@ export async function mapHookRecords( /* ------------------------------------------------------------ forwarding --- */ -async function buildForwardContext( - credentials: SSOCredentials | JWTCredentials -): Promise { +async function resolveBaseUrl(): Promise { const { readState } = await import( '../../../../../../cli/commands/proxy/daemon-manager.js' ); const state = await readState(); - - return { - credentials, - baseUrl: state?.targetUrl ?? state?.url ?? '', - projectName: state?.project ?? '', - userEmail: resolveEmailFromCredentials(credentials), - identity: {}, - }; + return state?.targetUrl ?? state?.url ?? ''; } /** Forward the complete hook records after the hooks cursor. */ async function forwardHooks( sessionId: string, - ctx: ForwardContext + baseUrl: string, + credentials: SSOCredentials | JWTCredentials ): Promise { const batch = await snapshotPendingHookRecords(sessionId); if (!batch) { return 'idle'; } - const payload = await mapHookRecords(batch.records, ctx); + const payload = unwrapHookRecords(batch.records); if (payload.malformed > 0) { logger.debug( '[otlp-forwarder] skipped malformed hook records', @@ -313,9 +187,9 @@ async function forwardHooks( return 'idle'; } - const url = `${ctx.baseUrl}${CODEMIE_ENDPOINTS.CLI_ANALYTICS_EVENT_HOOKS}`; + const url = `${baseUrl}${CODEMIE_ENDPOINTS.CLI_ANALYTICS_EVENT_HOOKS}`; const result = await send( - sessionId, 'hooks', url, payload.ndjson, 'application/x-ndjson', ctx.credentials + sessionId, 'hooks', url, payload.ndjson, 'application/x-ndjson', credentials ); if (result !== 'ok') { return result; @@ -334,16 +208,17 @@ async function forwardHooks( async function forwardOtelStream( sessionId: string, stream: OtelStream, - ctx: ForwardContext + baseUrl: string, + credentials: SSOCredentials | JWTCredentials ): Promise { const pending = await snapshotPendingBytes(sessionId, stream); if (!pending) { return 'idle'; } - const url = `${ctx.baseUrl}${OTEL_ENDPOINTS[stream]}`; + const url = `${baseUrl}${OTEL_ENDPOINTS[stream]}`; const result = await send( - sessionId, stream, url, pending.bytes, 'application/x-protobuf', ctx.credentials + sessionId, stream, url, pending.bytes, 'application/x-protobuf', credentials ); if (result !== 'ok') { return result; @@ -369,9 +244,9 @@ export async function forwardSession( hooksOnly: boolean, credentials: SSOCredentials | JWTCredentials ): Promise { - const ctx = await buildForwardContext(credentials); + const baseUrl = await resolveBaseUrl(); - const hooksResult = await forwardHooks(sessionId, ctx); + const hooksResult = await forwardHooks(sessionId, baseUrl, credentials); if (hooksResult === 'auth-expired') { markCredentialsStale(); return; @@ -382,7 +257,7 @@ export async function forwardSession( } for (const stream of OTEL_STREAMS) { - const result = await forwardOtelStream(sessionId, stream, ctx); + const result = await forwardOtelStream(sessionId, stream, baseUrl, credentials); if (result === 'auth-expired') { markCredentialsStale(); return; diff --git a/src/providers/plugins/sso/proxy/plugins/otlp.plugin.ts b/src/providers/plugins/sso/proxy/plugins/otlp.plugin.ts index 2e98367ac..050d28445 100644 --- a/src/providers/plugins/sso/proxy/plugins/otlp.plugin.ts +++ b/src/providers/plugins/sso/proxy/plugins/otlp.plugin.ts @@ -24,8 +24,8 @@ const UUID_V4_RE = /[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a export interface OtlpHookSpoolData { agentName: string; - raw: string; - timestamp: number; + /** The complete wire-ready event, built at hook time by the adapter. */ + hookEvent: string; } function sendError(res: ServerResponse, status: number, type: string, message: string): true { @@ -183,16 +183,16 @@ class OtlpInterceptor implements ProxyInterceptor { return sendError(res, 400, 'invalid_request_error', 'Agent validation failed'); } - // Extract session_id from raw hook JSON + // Extract session_id from the hook event JSON let sessionId = ''; try { - const hookEvent = JSON.parse(otlpHookSpoolData.raw) as Record; + const hookEvent = JSON.parse(otlpHookSpoolData.hookEvent) as Record; sessionId = String(hookEvent['session_id'] ?? ''); } catch { /* ignore */ } if (!sessionId) { // No session id — log and accept without spooling - logger.debug('[otlp-ingest] hooks: no session_id in raw, discarding'); + logger.debug('[otlp-ingest] hooks: no session_id in hookEvent, discarding'); res.statusCode = 202; res.setHeader('Content-Type', 'application/json'); res.end(JSON.stringify({ accepted: true })); From 00daadc89cfce93ae42a91b8d7c5ffc18ad44359 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Fri, 9 Oct 2026 09:30:32 +0300 Subject: [PATCH 31/35] refactor(proxy): rework agent.subagent.usage event contract --- .../__tests__/claude-code-otlp.plugin.test.ts | 88 +++++++- .../claude-code-otlp.plugin.ts | 27 ++- .../transcript/__tests__/orchestrator.test.ts | 72 +++++- .../__tests__/subagent-usage.test.ts | 179 +++++++++++---- .../transcript/orchestrator.ts | 118 +++++++--- .../transcript/subagent-usage.ts | 211 +++++++++++++----- 6 files changed, 547 insertions(+), 148 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index 52f432933..a298b6be6 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -21,7 +21,9 @@ vi.mock('@/utils/logger.js', () => ({ const resolveClientVersionMock = vi.fn(); const collectMainTranscriptEventsMock = vi.fn(); const collectSubagentTranscriptEventsMock = vi.fn(); +const subagentNeedsBackstopMock = vi.fn(); const findSubagentFilesMock = vi.fn(); +const readSubagentMetaMock = vi.fn(); vi.mock('../client-version-cache.js', () => ({ resolveClientVersion: resolveClientVersionMock, @@ -30,10 +32,12 @@ vi.mock('../client-version-cache.js', () => ({ vi.mock('../transcript/orchestrator.js', () => ({ collectMainTranscriptEvents: collectMainTranscriptEventsMock, collectSubagentTranscriptEvents: collectSubagentTranscriptEventsMock, + subagentNeedsBackstop: subagentNeedsBackstopMock, })); vi.mock('../transcript/subagent-usage.js', () => ({ findSubagentFiles: findSubagentFilesMock, + readSubagentMeta: readSubagentMetaMock, })); import { isProjectTracked } from '../claude-code-otlp.allowlist.js'; @@ -102,7 +106,9 @@ describe('ClaudeCodeOtlpPlugin hook-time enrichment', () => { resolveClientVersionMock.mockResolvedValue('2.1.23'); collectMainTranscriptEventsMock.mockResolvedValue([]); collectSubagentTranscriptEventsMock.mockResolvedValue([]); + subagentNeedsBackstopMock.mockResolvedValue(true); findSubagentFilesMock.mockResolvedValue([]); + readSubagentMetaMock.mockResolvedValue({}); }); afterEach(() => { @@ -170,8 +176,12 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { collectMainTranscriptEventsMock.mockResolvedValue([]); collectSubagentTranscriptEventsMock.mockReset(); collectSubagentTranscriptEventsMock.mockResolvedValue([]); + subagentNeedsBackstopMock.mockReset(); + subagentNeedsBackstopMock.mockResolvedValue(true); findSubagentFilesMock.mockReset(); findSubagentFilesMock.mockResolvedValue([]); + readSubagentMetaMock.mockReset(); + readSubagentMetaMock.mockResolvedValue({}); resolveClientVersionMock.mockReset(); resolveClientVersionMock.mockResolvedValue('2.1.23'); vi.mocked(forwardOtlpEventToSpool).mockReset(); @@ -244,6 +254,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { hookEvent({ hook_event_name: 'SubagentStop', agent_transcript_path: '/tmp/agent-sub-2.jsonl', + agent_type: 'explore', }) ); @@ -255,17 +266,60 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { ); }); + it('fills tool_use_id/spawn_depth/description from the .meta.json sidecar on SubagentStop', async () => { + readSubagentMetaMock.mockResolvedValue({ toolUseId: 'tu-sidecar', spawnDepth: 2, description: 'task' }); + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-3.jsonl', + agent_id: 'sub-3', + agent_type: 'explore', + }) + ); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); + + expect(readSubagentMetaMock).toHaveBeenCalledWith('/tmp/agent-sub-3.jsonl'); + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith('sid-1', { + agentId: 'sub-3', + filePath: '/tmp/agent-sub-3.jsonl', + toolUseId: 'tu-sidecar', + agentType: 'explore', + spawnDepth: 2, + description: 'task', + }); + }); + it('skips collectSubagentTranscriptEvents on SubagentStop when agent_transcript_path is missing', async () => { const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); const plugin = new ClaudeCodeOtlpPlugin(); - const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SubagentStop' })); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SubagentStop', agent_type: 'explore' })); await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); expect(collectSubagentTranscriptEventsMock).not.toHaveBeenCalled(); }); - it('runs collectSubagentTranscriptEvents for every subagent file found on SessionEnd', async () => { + it('skips SubagentStop entirely (no forward, no derive) when agent_type is empty', async () => { + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify( + hookEvent({ + hook_event_name: 'SubagentStop', + agent_transcript_path: '/tmp/agent-sub-4.jsonl', + agent_type: '', + }) + ); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); + + expect(collectSubagentTranscriptEventsMock).not.toHaveBeenCalled(); + expect(forwardOtlpEventToSpool).not.toHaveBeenCalled(); + }); + + it('runs collectSubagentTranscriptEvents for every subagent file found on SessionEnd that still needs the backstop', async () => { findSubagentFilesMock.mockResolvedValue([ { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' }, { agentId: 'sub-2', filePath: '/tmp/agent-sub-2.jsonl' }, @@ -284,6 +338,27 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { ); }); + it('skips a subagent on the SessionEnd backstop when it no longer needs re-processing', async () => { + findSubagentFilesMock.mockResolvedValue([ + { agentId: 'reported', filePath: '/tmp/agent-reported.jsonl' }, + { agentId: 'pending', filePath: '/tmp/agent-pending.jsonl' }, + ]); + subagentNeedsBackstopMock.mockImplementation(async (_sessionId: string, file: { agentId: string }) => + file.agentId !== 'reported' + ); + const { ClaudeCodeOtlpPlugin } = await import('../claude-code-otlp.plugin.js'); + const plugin = new ClaudeCodeOtlpPlugin(); + const rawEvent = JSON.stringify(hookEvent({ hook_event_name: 'SessionEnd' })); + + await plugin.processOtlpEvent(rawEvent, { ensureOtlpProxy, forwardOtlpEventToSpool }); + + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledTimes(1); + expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( + 'sid-1', + { agentId: 'pending', filePath: '/tmp/agent-pending.jsonl' } + ); + }); + it('forwards every event a per-event handler returns (the raw event plus any derived events) through the single forwardToSpool path, in order', async () => { const derivedUsageEvent = { type: 'agent.usage.request' }; const derivedSummaryEvent = { type: 'agent.session.summary' }; @@ -329,7 +404,14 @@ describe('ClaudeCodeOtlpPlugin wire type', () => { }); const typeFor = async (name: string): Promise => { - await plugin.processOtlpEvent(event(name), { ensureOtlpProxy, forwardOtlpEventToSpool }); + // A SubagentStop with no agent_type is Claude Code's own internal check-in, never + // forwarded (see the `claude-code-otlp.plugin.ts` guard) — give it one here so this + // generic wire-type matrix still exercises its mapping, like every other hook name. + const raw = + name === 'SubagentStop' + ? JSON.stringify({ session_id: 's', transcript_path: '', cwd: '/x', hook_event_name: name, agent_type: 'explore' }) + : event(name); + await plugin.processOtlpEvent(raw, { ensureOtlpProxy, forwardOtlpEventToSpool }); return vi.mocked(forwardOtlpEventToSpool).mock.calls[0][0]['type']; }; diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index b496fc3b5..bf92b3033 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -19,9 +19,10 @@ import { isProjectTracked, readAllowlistState } from './claude-code-otlp.allowli import { collectMainTranscriptEvents, collectSubagentTranscriptEvents, + subagentNeedsBackstop, type SubagentFile, } from './transcript/orchestrator.js'; -import { findSubagentFiles } from './transcript/subagent-usage.js'; +import { findSubagentFiles, readSubagentMeta } from './transcript/subagent-usage.js'; import { resolveClientVersion } from './client-version-cache.js'; const HOOK_EVENT_TYPE_MAP: Record = { @@ -114,28 +115,42 @@ export class ClaudeCodeOtlpPlugin extends OtlpAgentAdapter { + if (!hookInput.agent_type) { + // Claude Code's own internal check-ins ("is it done yet?") fire SubagentStop with no + // agent_type — not a real subagent, so nothing is derived or forwarded for it. + return { decision: 'forward', payload: [] }; + } + const agentTranscriptPath = hookInput.agent_transcript_path; if (!agentTranscriptPath) { return { decision: 'forward', payload: [hookInput] }; } + const meta = await readSubagentMeta(agentTranscriptPath); const subagentFile: SubagentFile = { agentId: hookInput.agent_id || basename(agentTranscriptPath).replace(/^agent-/, '').replace(/\.jsonl$/, ''), filePath: agentTranscriptPath, - // `tool_use_id` is not declared on the SDK's SubagentStopHookInput type; read it - // defensively in case the raw hook payload carries it anyway. - toolUseId: readOptionalString(hookInput, 'tool_use_id'), + // `tool_use_id` is not declared on the SDK's SubagentStopHookInput type; the hook payload + // wins when it carries it anyway, else fall back to the `.meta.json` sidecar. + toolUseId: readOptionalString(hookInput, 'tool_use_id') ?? meta.toolUseId, agentType: hookInput.agent_type, + spawnDepth: meta.spawnDepth, + description: meta.description, }; const derived = await collectSubagentTranscriptEvents(hookInput.session_id, subagentFile); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index eee5014ba..8c44911e7 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -382,15 +382,16 @@ describe('collectSubagentTranscriptEvents — no new bytes since last run', () = // No new lines means no new/touched openRequests keys at the agent.usage.request layer. expect(secondUsageRequests).toHaveLength(0); // But the agent.subagent.usage event is still returned exactly once, summarizing the same - // cumulative (unchanged) usage. + // cumulative (unchanged) usage — the file itself is non-empty, so this is not a phantom. expect(secondSubagentUsage).toHaveLength(1); - expect(secondSubagentUsage[0].output_tokens).toBe(30); + const usage = secondSubagentUsage[0].usage as Array>; + expect(usage[0].output_tokens).toBe(30); expect(secondSubagentUsage[0].api_calls).toBe(2); }); }); -describe('collectSubagentTranscriptEvents — missing subagent transcript file', () => { - it('resolves without throwing and still returns one empty-usage agent.subagent.usage event, with no agent.usage.request events', async () => { +describe('collectSubagentTranscriptEvents — missing subagent transcript file (phantom guard)', () => { + it('resolves without throwing and returns no agent.subagent.usage event — a missing transcript has nothing real to summarize', async () => { const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); const sessionId = 'session-subagent-missing'; @@ -406,10 +407,63 @@ describe('collectSubagentTranscriptEvents — missing subagent transcript file', const subagentUsageEvents = events.filter((e) => e.type === 'agent.subagent.usage'); expect(usageRequestEvents).toHaveLength(0); - expect(subagentUsageEvents).toHaveLength(1); - expect(subagentUsageEvents[0].api_calls).toBe(0); - expect(subagentUsageEvents[0].input_tokens).toBe(0); - expect(subagentUsageEvents[0].started_at).toBe(''); - expect(subagentUsageEvents[0].duration_ms).toBe(0); + expect(subagentUsageEvents).toHaveLength(0); + }); +}); + +describe('collectSubagentTranscriptEvents — empty subagent transcript file (phantom guard)', () => { + it('returns no agent.subagent.usage event for a file with no parseable lines', async () => { + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-subagent-empty'; + const emptyFile = writeSubagentFixture(sessionId, 'empty', []); + + const events = parseAll(await collectSubagentTranscriptEvents(sessionId, emptyFile)); + + expect(events.filter((e) => e.type === 'agent.subagent.usage')).toHaveLength(0); + }); +}); + +describe('subagentNeedsBackstop', () => { + it('is true for an agent never seen before (no persisted offset)', async () => { + const { subagentNeedsBackstop } = await import('../orchestrator.js'); + + const sessionId = 'session-backstop-needed-unseen'; + const file = writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'uuid-a1-1', messageId: 'msg-a1-1', outputTokens: 10 }), + ]); + + expect(await subagentNeedsBackstop(sessionId, file)).toBe(true); + }); + + it('is false once a prior pass advanced the offset to the file\'s current size', async () => { + const { subagentNeedsBackstop, collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-backstop-already-reported'; + const file = writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'uuid-a1-1', messageId: 'msg-a1-1', outputTokens: 10 }), + ]); + + await collectSubagentTranscriptEvents(sessionId, file); + + expect(await subagentNeedsBackstop(sessionId, file)).toBe(false); + }); + + it('is true again once more content is appended after the last reported offset', async () => { + const { subagentNeedsBackstop, collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-backstop-grown'; + const file = writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'uuid-a1-1', messageId: 'msg-a1-1', outputTokens: 10 }), + ]); + await collectSubagentTranscriptEvents(sessionId, file); + + writeFileSync( + file.filePath, + usageLine({ uuid: 'uuid-a1-2', messageId: 'msg-a1-2', outputTokens: 20 }) + '\n', + { flag: 'a' } + ); + + expect(await subagentNeedsBackstop(sessionId, file)).toBe(true); }); }); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts index 9d607ce2b..89d932c83 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts @@ -1,24 +1,30 @@ /** - * Tests for the `agent.subagent.usage` builder: `findSubagentFiles` and - * `buildSubagentUsageEvent`. + * Tests for the `agent.subagent.usage` builder: `findSubagentFiles`, `readSubagentMeta`, + * `buildUsageTotals`, and `buildSubagentUsageEvent`. * * Fixture layout, built fresh per test under a temp dir: * /.jsonl — trivial main-transcript placeholder * //subagents/agent-.jsonl — one subagent transcript per fixture - * //subagents/agent-.meta.json — sidecar (toolUseId/agentType/spawnDepth) + * //subagents/agent-.meta.json — sidecar (toolUseId/agentType/spawnDepth/description) * * Three subagents are used throughout: * - "a1": sidecar OMITS spawnDepth (top-level subagent — spawn_depth must default to 0), * two usage-bearing transcript lines (exercises summing across >1 request). * - "a2": sidecar INCLUDES spawnDepth: 2 (nested subagent — pass-through), one usage line. - * - "a3": sidecar includes toolUseId/agentType, one usage line. + * - "a3": sidecar includes toolUseId/agentType/description, one usage line. */ import { describe, it, expect, afterEach } from 'vitest'; import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'; import { tmpdir } from 'node:os'; import { join } from 'node:path'; -import { findSubagentFiles, buildSubagentUsageEvent, type SubagentFile } from '../subagent-usage.js'; +import { + findSubagentFiles, + readSubagentMeta, + buildUsageTotals, + buildSubagentUsageEvent, + type SubagentFile, +} from '../subagent-usage.js'; import { parseUsageLine, buildUsageRequestEvent } from '../usage-request.js'; import type { OpenUsageRequest } from '../parse-state.js'; @@ -69,6 +75,7 @@ interface FixtureMeta { toolUseId?: string; agentType?: string; spawnDepth?: number; + description?: string; } async function buildFixture( @@ -90,13 +97,28 @@ async function buildFixture( return mainTranscriptPath; } +function usageRequest(overrides: Partial = {}): OpenUsageRequest { + return { + requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + speed: 'standard', inferenceGeo: '', serviceTier: 'standard', + inputTokens: 0, cacheCreation5mTokens: 0, cacheCreation1hTokens: 0, + cacheReadTokens: 0, outputTokens: 0, webSearchRequests: 0, webFetchRequests: 0, + scopeKind: 'agent', scopeName: '', agentId: 'a1', + stopReason: 'end_turn', isApiError: false, gitBranch: 'main', + ...overrides, + }; +} + describe('findSubagentFiles', () => { it('discovers all subagent files with their sidecar fields, defaulting spawnDepth to undefined when the sidecar omits it', async () => { const sessionId = 'session-subagent-1'; const mainTranscriptPath = await buildFixture(sessionId, { a1: { lines: [usageLine('msg-a1-1', 100, 50)], meta: { toolUseId: 'tool-a1' } }, a2: { lines: [usageLine('msg-a2-1', 10, 5)], meta: { agentType: 'reviewer', spawnDepth: 2 } }, - a3: { lines: [usageLine('msg-a3-1', 7, 3)], meta: { toolUseId: 'tool-a3', agentType: 'coder' } }, + a3: { + lines: [usageLine('msg-a3-1', 7, 3)], + meta: { toolUseId: 'tool-a3', agentType: 'coder', description: 'Fix the bug' }, + }, }); tmpDir = join(mainTranscriptPath, '..'); @@ -114,6 +136,7 @@ describe('findSubagentFiles', () => { expect(a1.toolUseId).toBe('tool-a1'); expect(a1.agentType).toBeUndefined(); expect(a1.spawnDepth).toBeUndefined(); + expect(a1.description).toBeUndefined(); const a2 = byId('a2'); expect(a2.agentType).toBe('reviewer'); @@ -123,6 +146,7 @@ describe('findSubagentFiles', () => { const a3 = byId('a3'); expect(a3.toolUseId).toBe('tool-a3'); expect(a3.agentType).toBe('coder'); + expect(a3.description).toBe('Fix the bug'); expect(a3.spawnDepth).toBeUndefined(); }); @@ -142,33 +166,69 @@ describe('findSubagentFiles', () => { }); }); +describe('readSubagentMeta', () => { + it('reads toolUseId/agentType/spawnDepth/description from the sidecar next to a given .jsonl path', async () => { + tmpDir = await mkdtemp(join(tmpdir(), 'codemie-subagent-meta-')); + const jsonlPath = join(tmpDir, 'agent-x1.jsonl'); + await writeFile(jsonlPath, ''); + await writeFile( + join(tmpDir, 'agent-x1.meta.json'), + JSON.stringify({ toolUseId: 'tool-x1', agentType: 'explore', spawnDepth: 1, description: 'Investigate' }) + ); + + const meta = await readSubagentMeta(jsonlPath); + + expect(meta).toEqual({ toolUseId: 'tool-x1', agentType: 'explore', spawnDepth: 1, description: 'Investigate' }); + }); + + it('resolves to {} when the sidecar is missing, without throwing', async () => { + const meta = await readSubagentMeta('C:/nonexistent/agent-ghost.jsonl'); + expect(meta).toEqual({}); + }); +}); + +describe('buildUsageTotals', () => { + it('sums token/call fields per (model, speed, inference_geo, service_tier, scope_kind, scope_name) group', () => { + const reqs: OpenUsageRequest[] = [ + usageRequest({ model: 'claude-haiku-4-5', inputTokens: 900, outputTokens: 2100 }), + usageRequest({ model: 'claude-haiku-4-5', inputTokens: 100, outputTokens: 400 }), + usageRequest({ model: 'claude-opus-5-5', speed: 'fast', inputTokens: 50, outputTokens: 10 }), + ]; + + const totals = buildUsageTotals(reqs); + + expect(totals).toHaveLength(2); + const haiku = totals.find((t) => t.model === 'claude-haiku-4-5'); + expect(haiku).toMatchObject({ input_tokens: 1000, output_tokens: 2500, api_calls: 2 }); + const opus = totals.find((t) => t.model === 'claude-opus-5-5'); + expect(opus).toMatchObject({ speed: 'fast', input_tokens: 50, output_tokens: 10, api_calls: 1 }); + }); + + it('returns [] for no usage requests', () => { + expect(buildUsageTotals([])).toEqual([]); + }); +}); + describe('buildSubagentUsageEvent', () => { - it('sums token/cache fields across usageRequests, defaults spawn_depth to 0 when the file omits it, and passes caller-built maps through verbatim', () => { + it('groups usageRequests into usage[], sums tool maps into flat totals plus a tools breakdown, and passes skills through verbatim', () => { const file: SubagentFile = { agentId: 'a1', filePath: '/tmp/agent-a1.jsonl', toolUseId: 'tool-a1' }; const reqs: OpenUsageRequest[] = [ - { - requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', - speed: 'standard', inferenceGeo: '', serviceTier: 'standard', - inputTokens: 100, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, + usageRequest({ + requestId: 'r1', inputTokens: 100, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, cacheReadTokens: 3, outputTokens: 50, webSearchRequests: 1, webFetchRequests: 0, - scopeKind: 'agent', scopeName: '', agentId: 'a1', - stopReason: 'end_turn', isApiError: false, gitBranch: 'main', - }, - { - requestId: 'r2', model: 'm', modelRaw: 'm-raw', timestamp: 't2', - speed: 'standard', inferenceGeo: '', serviceTier: 'standard', - inputTokens: 10, cacheCreation5mTokens: 4, cacheCreation1hTokens: 0, + }), + usageRequest({ + requestId: 'r2', inputTokens: 10, cacheCreation5mTokens: 4, cacheCreation1hTokens: 0, cacheReadTokens: 1, outputTokens: 5, webSearchRequests: 0, webFetchRequests: 2, - scopeKind: 'agent', scopeName: '', agentId: 'a1', - stopReason: 'tool_use', isApiError: false, gitBranch: 'main', - }, + stopReason: 'tool_use', + }), ]; const toolCalls = { Read: 3, Edit: 1 }; const toolErrors = { Edit: 1 }; - const skillsInvoked = { brainstorming: 1 }; + const skills = { brainstorming: 1 }; const event = buildSubagentUsageEvent( - 'session-1', file, reqs, toolCalls, toolErrors, skillsInvoked, '2026-10-01T00:00:00.000Z', 1500 + 'session-1', file, reqs, toolCalls, toolErrors, 4, skills, '2026-10-01T00:00:00.000Z', '2026-10-01T00:05:00.000Z', 1500 ); expect(event).toEqual({ @@ -177,48 +237,71 @@ describe('buildSubagentUsageEvent', () => { agent_id: 'a1', tool_use_id: 'tool-a1', agent_type: '', - spawn_depth: 0, description: '', workflow_run: '', + spawn_depth: 0, worktree: '', started_at: '2026-10-01T00:00:00.000Z', + ended_at: '2026-10-01T00:05:00.000Z', duration_ms: 1500, - input_tokens: 110, - cache_creation_5m_tokens: 5, - cache_creation_1h_tokens: 2, - cache_read_tokens: 4, - output_tokens: 55, - web_search_requests: 1, - web_fetch_requests: 2, + model: 'm', api_calls: 2, - tool_calls: toolCalls, - tool_errors: toolErrors, - skills_invoked: skillsInvoked, + tool_calls: 4, + tool_results: 4, + tool_errors: 1, + tools: { Read: { calls: 3, errors: 0 }, Edit: { calls: 1, errors: 1 } }, + skills, + commands: [], + compactions: [], + usage: [ + { + model: 'm', model_raw: 'm-raw', speed: 'standard', inference_geo: '', service_tier: 'standard', + scope_kind: 'agent', scope_name: '', + input_tokens: 110, cache_creation_5m_tokens: 5, cache_creation_1h_tokens: 2, cache_read_tokens: 4, + output_tokens: 55, web_search_requests: 1, web_fetch_requests: 2, api_calls: 2, + }, + ], }); }); - it('passes a present spawn_depth through verbatim instead of defaulting to 0', () => { - const file: SubagentFile = { agentId: 'a2', filePath: '/tmp/agent-a2.jsonl', spawnDepth: 2 }; + it('passes a present spawn_depth through verbatim instead of defaulting to 0, and sources description from the sidecar-derived file field', () => { + const file: SubagentFile = { agentId: 'a2', filePath: '/tmp/agent-a2.jsonl', spawnDepth: 2, description: 'Review the diff' }; - const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, {}, '', 0); + const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, 0, {}, '', '', 0); expect(event.spawn_depth).toBe(2); + expect(event.description).toBe('Review the diff'); expect(event.api_calls).toBe(0); + expect(event.usage).toEqual([]); }); - it('never fabricates description/workflow_run/worktree — always empty string', () => { + it('never fabricates agent_type/description/workflow_run/worktree — always empty string when absent', () => { const file: SubagentFile = { agentId: 'a3', filePath: '/tmp/agent-a3.jsonl' }; - const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, {}, '', 0); + const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, 0, {}, '', '', 0); + expect(event.agent_type).toBe(''); expect(event.description).toBe(''); expect(event.workflow_run).toBe(''); expect(event.worktree).toBe(''); }); + + it('picks the usage[] row with the most api_calls as the top-level model', () => { + const file: SubagentFile = { agentId: 'a1', filePath: '/tmp/agent-a1.jsonl' }; + const reqs: OpenUsageRequest[] = [ + usageRequest({ model: 'claude-haiku-4-5', requestId: 'r1' }), + usageRequest({ model: 'claude-opus-5-5', requestId: 'r2' }), + usageRequest({ model: 'claude-opus-5-5', requestId: 'r3' }), + ]; + + const event = buildSubagentUsageEvent('session-1', file, reqs, {}, {}, 0, {}, '', '', 0); + + expect(event.model).toBe('claude-opus-5-5'); + }); }); -describe('cross-check: agent.subagent.usage summed tokens vs agent.usage.request (scope_kind: agent)', () => { - it('summing token fields across three agent.subagent.usage events equals summing the same fields across every agent.usage.request record derived from the same fixture', async () => { +describe('cross-check: agent.subagent.usage usage[] totals vs agent.usage.request (scope_kind: agent)', () => { + it('summing usage[] token fields across three agent.subagent.usage events equals summing the same fields across every agent.usage.request record derived from the same fixture', async () => { const sessionId = 'session-subagent-cross'; const mainTranscriptPath = await buildFixture(sessionId, { a1: { @@ -254,7 +337,7 @@ describe('cross-check: agent.subagent.usage summed tokens vs agent.usage.request reqs.forEach((r) => expect(r.scopeKind).toBe('agent')); subagentUsageEvents.push( - buildSubagentUsageEvent(sessionId, file, reqs, {}, {}, {}, '2026-10-01T00:00:00.000Z', 0) + buildSubagentUsageEvent(sessionId, file, reqs, {}, {}, 0, {}, '2026-10-01T00:00:00.000Z', '2026-10-01T00:00:00.000Z', 0) ); for (const req of reqs) { @@ -267,6 +350,12 @@ describe('cross-check: agent.subagent.usage summed tokens vs agent.usage.request expect(usageRequestEvents).toHaveLength(4); usageRequestEvents.forEach((e) => expect(e.scope_kind).toBe('agent')); + const sumUsageField = (records: Array>, field: string): number => + records.reduce((total, r) => { + const usage = r.usage as Array>; + return total + usage.reduce((rowTotal, row) => rowTotal + Number(row[field] ?? 0), 0); + }, 0); + const sumField = (records: Array>, field: string): number => records.reduce((total, r) => total + Number(r[field] ?? 0), 0); @@ -281,7 +370,7 @@ describe('cross-check: agent.subagent.usage summed tokens vs agent.usage.request ]; for (const [subagentField, requestField] of tokenFieldPairs) { - expect(sumField(subagentUsageEvents, subagentField)).toBe(sumField(usageRequestEvents, requestField)); + expect(sumUsageField(subagentUsageEvents, subagentField)).toBe(sumField(usageRequestEvents, requestField)); } // api_calls across the three subagent.usage events equals the total number of @@ -289,7 +378,7 @@ describe('cross-check: agent.subagent.usage summed tokens vs agent.usage.request expect(sumField(subagentUsageEvents, 'api_calls')).toBe(usageRequestEvents.length); // Known concrete totals: input 100+10+200+30=340, output 50+5+80+15=150. - expect(sumField(subagentUsageEvents, 'input_tokens')).toBe(340); - expect(sumField(subagentUsageEvents, 'output_tokens')).toBe(150); + expect(sumUsageField(subagentUsageEvents, 'input_tokens')).toBe(340); + expect(sumUsageField(subagentUsageEvents, 'output_tokens')).toBe(150); }); }); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 0bfcb78bb..d1a2aa623 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -17,7 +17,7 @@ * this). */ -import { readFile } from 'node:fs/promises'; +import { readFile, stat } from 'node:fs/promises'; import { loadParseState, saveParseState, withParseStateLock } from './parse-state.js'; import { readNewLines } from './transcript-reader.js'; import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from './usage-request.js'; @@ -311,14 +311,18 @@ export async function collectMainTranscriptEvents( interface SubagentScanResult { toolCalls: Record; toolErrors: Record; + toolResults: number; skillsInvoked: Record; startedAt: string; + endedAt: string; durationMs: number; + /** True when the file is missing, unreadable, or has no parseable lines — the phantom guard. */ + isEmpty: boolean; } /** - * Recompute one subagent transcript's tool-call/tool-error/skill-invocation aggregates and - * timing span from byte 0 of its own file (the subagent-transcript analogue of + * Recompute one subagent transcript's tool-call/tool-error/tool-result/skill-invocation + * aggregates and timing span from byte 0 of its own file (the subagent-transcript analogue of * {@link buildFullAccumulator}'s "recompute fresh each time" approach — `TranscriptParseState` * has no persisted field for any of these either). * @@ -326,23 +330,28 @@ interface SubagentScanResult { * `tool_result.is_error` correlation {@link buildFullAccumulator} uses, but tally into two * parallel `Record` maps (not the combined `{calls, errors}` shape * `SessionSummaryAccumulator` uses) to match {@link buildSubagentUsageEvent}'s own - * `tool_calls`/`tool_errors` parameter shapes. + * `tool_calls`/`tool_errors` parameter shapes; `toolResults` reuses {@link countToolResults}. * - `skillsInvoked` is `extractNamedInvocations(parsedLines).skillInvocations`, taken verbatim. - * - `startedAt` is the first parsed line's `timestamp`, or `''` when the file is empty/unreadable. - * - `durationMs` is `Date.parse(lastLine.timestamp) - Date.parse(firstLine.timestamp)`, guarded by - * `Number.isFinite` (covers a missing/unparseable timestamp on either end, and a single-line - * file) so it is never `NaN` — falls back to `0`. + * - `startedAt`/`endedAt` are the first/last *timestamped* line via {@link firstTimestamp} (real + * subagent transcripts interleave untimestamped lines, e.g. `attachment`, at either end). + * - `durationMs` is `Date.parse(endedAt) - Date.parse(startedAt)`, guarded by `Number.isFinite` + * (covers a missing/unparseable timestamp on either end, and a single-line file) so it is + * never `NaN` — falls back to `0`. * * Never throws: a missing/unreadable file or an empty file both resolve to the emptiest - * defensible result; a malformed individual line is skipped rather than aborting the whole scan. + * defensible result with `isEmpty: true`; a malformed individual line is skipped rather than + * aborting the whole scan. */ async function scanSubagentTranscript(filePath: string): Promise { const empty: SubagentScanResult = { toolCalls: {}, toolErrors: {}, + toolResults: 0, skillsInvoked: {}, startedAt: '', + endedAt: '', durationMs: 0, + isEmpty: true, }; let raw: string; @@ -382,13 +391,47 @@ async function scanSubagentTranscript(filePath: string): Promise { + try { + const state = await loadParseState(sessionId); + const offset = state.subagentOffsets[subagentFile.agentId]; + if (offset === undefined) { + return true; + } + const info = await stat(subagentFile.filePath); + return info.size !== offset; + } catch { + return true; + } } /** @@ -403,14 +446,16 @@ async function scanSubagentTranscript(filePath: string): Promise r.scopeKind === 'agent' && r.agentId === subagentFile.agentId ); - const { toolCalls, toolErrors, skillsInvoked, startedAt, durationMs } = - await scanSubagentTranscript(subagentFile.filePath); - - const subagentEvent = buildSubagentUsageEvent( - sessionId, - subagentFile, - usageRequestsForAgent, - toolCalls, - toolErrors, - skillsInvoked, - startedAt, - durationMs - ); - events.push(subagentEvent); + const scan = await scanSubagentTranscript(subagentFile.filePath); + if (!scan.isEmpty) { + events.push( + buildSubagentUsageEvent( + sessionId, + subagentFile, + usageRequestsForAgent, + scan.toolCalls, + scan.toolErrors, + scan.toolResults, + scan.skillsInvoked, + scan.startedAt, + scan.endedAt, + scan.durationMs + ) + ); + } await saveParseState(sessionId, state); return events; diff --git a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts index 1af1f1cac..837d6410e 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts @@ -4,14 +4,16 @@ * Discovery (`findSubagentFiles`) uses the same path convention as the private * `findSubagentFiles` in `src/agents/plugins/claude/claude.session.ts` * (`//subagents/agent-*.jsonl` + sibling `.meta.json`) but returns - * only the narrower {@link SubagentFile} shape this event needs. + * only the narrower {@link SubagentFile} shape this event needs. The sidecar reader is also + * exported ({@link readSubagentMeta}) so a `SubagentStop` hook — which only sees its own single + * transcript path, not the whole `subagents/` directory — can fill in the same fields. * - * The event builder (`buildSubagentUsageEvent`) sums an already-scoped `OpenUsageRequest[]` for - * token/cache fields and passes the caller-built `tool_calls`/`tool_errors`/`skills_invoked` maps - * through verbatim; it has no access to the subagent transcript itself. + * The event builder (`buildSubagentUsageEvent`) sums an already-scoped `OpenUsageRequest[]` into + * the contract's `usage[]` rows (one per distinct model/speed/inference_geo/service_tier/ + * scope_kind/scope_name) and passes the caller-built tool-call/tool-error/skill maps through as + * the contract's `tools`/`skills` objects; it has no access to the subagent transcript itself. * - * `description`/`workflow_run`/`worktree` have no known source and are always empty strings, - * never fabricated. + * `workflow_run`/`worktree` have no known source and are always empty strings, never fabricated. */ import { readdir, readFile } from 'node:fs/promises'; @@ -24,20 +26,44 @@ export interface SubagentFile { toolUseId?: string; agentType?: string; spawnDepth?: number; + description?: string; +} + +interface SubagentMeta { + toolUseId?: string; + agentType?: string; + spawnDepth?: number; + description?: string; +} + +/** + * Read one subagent's `.meta.json` sidecar (same path, `.jsonl` swapped for `.meta.json`). + * Never throws: a missing or malformed sidecar resolves to `{}`. + */ +export async function readSubagentMeta(jsonlFilePath: string): Promise { + const meta: SubagentMeta = {}; + try { + const metaPath = jsonlFilePath.replace(/\.jsonl$/, '.meta.json'); + const metaRaw = JSON.parse(await readFile(metaPath, 'utf-8')) as Record; + if (typeof metaRaw.toolUseId === 'string') meta.toolUseId = metaRaw.toolUseId; + if (typeof metaRaw.agentType === 'string') meta.agentType = metaRaw.agentType; + if (typeof metaRaw.spawnDepth === 'number') meta.spawnDepth = metaRaw.spawnDepth; + if (typeof metaRaw.description === 'string') meta.description = metaRaw.description; + } catch { + // sidecar absent or malformed — proceed without it. + } + return meta; } /** * Discover subagent transcript files for a main transcript at `mainTranscriptPath`. * * Looks under `//subagents/` (where `sessionId` is `mainTranscriptPath`'s - * own basename, minus `.jsonl`) for `agent-*.jsonl` files, reading each one's sibling - * `.meta.json` sidecar (when present and parseable) for `toolUseId`/`agentType`/ - * `spawnDepth`. Never reads `mainTranscriptPath`'s own content — only its path is used to derive - * the subagents directory. + * own basename, minus `.jsonl`) for `agent-*.jsonl` files, reading each one's sidecar via + * {@link readSubagentMeta}. * * Never throws: a missing subagents directory, an unreadable directory, or any other failure all - * resolve to `[]`. A missing or malformed per-agent `.meta.json` sidecar is likewise swallowed — - * that agent is still returned, just without the sidecar-derived fields. + * resolve to `[]`. */ export async function findSubagentFiles(mainTranscriptPath: string): Promise { try { @@ -53,23 +79,8 @@ export async function findSubagentFiles(mainTranscriptPath: string): Promise => { const agentId = f.replace(/^agent-/, '').replace(/\.jsonl$/, ''); const filePath = join(subagentsDir, f); - - let toolUseId: string | undefined; - let agentType: string | undefined; - let spawnDepth: number | undefined; - - try { - const metaRaw = JSON.parse( - await readFile(join(subagentsDir, f.replace(/\.jsonl$/, '.meta.json')), 'utf-8') - ) as Record; - if (typeof metaRaw.toolUseId === 'string') toolUseId = metaRaw.toolUseId; - if (typeof metaRaw.agentType === 'string') agentType = metaRaw.agentType; - if (typeof metaRaw.spawnDepth === 'number') spawnDepth = metaRaw.spawnDepth; - } catch { - // meta file absent or malformed — proceed without it. - } - - return { agentId, filePath, toolUseId, agentType, spawnDepth }; + const meta = await readSubagentMeta(filePath); + return { agentId, filePath, ...meta }; }) ); @@ -79,19 +90,118 @@ export async function findSubagentFiles(mainTranscriptPath: string): Promise(); + + for (const r of usageRequests) { + const key = [r.model, r.speed, r.inferenceGeo, r.serviceTier, r.scopeKind, r.scopeName].join('\u0000'); + const existing = groups.get(key); + if (existing) { + existing.input_tokens += r.inputTokens; + existing.cache_creation_5m_tokens += r.cacheCreation5mTokens; + existing.cache_creation_1h_tokens += r.cacheCreation1hTokens; + existing.cache_read_tokens += r.cacheReadTokens; + existing.output_tokens += r.outputTokens; + existing.web_search_requests += r.webSearchRequests; + existing.web_fetch_requests += r.webFetchRequests; + existing.api_calls += 1; + } else { + groups.set(key, { + model: r.model, + model_raw: r.modelRaw, + speed: r.speed, + inference_geo: r.inferenceGeo, + service_tier: r.serviceTier, + scope_kind: r.scopeKind, + scope_name: r.scopeName, + input_tokens: r.inputTokens, + cache_creation_5m_tokens: r.cacheCreation5mTokens, + cache_creation_1h_tokens: r.cacheCreation1hTokens, + cache_read_tokens: r.cacheReadTokens, + output_tokens: r.outputTokens, + web_search_requests: r.webSearchRequests, + web_fetch_requests: r.webFetchRequests, + api_calls: 1, + }); + } + } + + return [...groups.values()]; +} + +/** The `model` of the `usage[]` row with the most `api_calls` — the contract's top-level `model`. */ +function primaryModel(totals: SubagentUsageTotal[]): string { + let best: SubagentUsageTotal | undefined; + for (const total of totals) { + if (!best || total.api_calls > best.api_calls) { + best = total; + } + } + return best?.model ?? ''; +} + +/** Sum of a `{name: count}` map's values — the contract's flat `tool_calls`/`tool_errors` totals. */ +function sumCounts(counts: Record): number { + return Object.values(counts).reduce((total, n) => total + n, 0); +} + +/** Merge per-tool call/error counts into the contract's `tools: {tool: {calls, errors}}`. */ +function buildToolsBreakdown( + toolCalls: Record, + toolErrors: Record +): Record { + const tools: Record = {}; + for (const [name, calls] of Object.entries(toolCalls)) { + tools[name] = { calls, errors: toolErrors[name] ?? 0 }; + } + for (const [name, errors] of Object.entries(toolErrors)) { + if (!(name in tools)) { + tools[name] = { calls: 0, errors }; + } + } + return tools; +} + /** * Build the `agent.subagent.usage` event payload for one subagent file. * * `usageRequests` is an array of {@link OpenUsageRequest} the caller has already parsed and - * scoped to this one subagent (`scope_kind: 'agent'`) — this function only sums it, it does not - * filter or scope it itself. `toolCalls`/`toolErrors`/`skillsInvoked` are likewise caller-built - * `Record` maps (keyed by tool/skill name) and are passed through verbatim. + * scoped to this one subagent (`scope_kind: 'agent'`) — summed into `usage[]` by + * {@link buildUsageTotals}, never into flat token fields (the contract has none for this event). + * `toolCalls`/`toolErrors` are caller-built `Record` maps (keyed by tool name), + * folded into the flat `tool_calls`/`tool_errors` totals and the per-tool `tools` breakdown. + * `skills` is passed through verbatim — it is already the contract's `{name: count}` shape. * - * `started_at`/`duration_ms` are forwarded verbatim from the caller, which derives them from the - * subagent transcript's own first/last line timestamps — not this function's job. + * `startedAt`/`endedAt`/`durationMs`/`toolResults` are forwarded verbatim from the caller, which + * derives them from the subagent transcript's own lines — not this function's job. * * `spawn_depth` defaults to `0` when `file.spawnDepth` is absent (top-level subagents, whose - * sidecar omits the field — not treated as an error). + * sidecar omits the field — not treated as an error). `description` defaults to `''` when the + * sidecar has none. * * Carries its own explicit `type`, so `event_id`/`schema_version` are stamped later, by the adapter base class at hook time. */ @@ -101,12 +211,13 @@ export function buildSubagentUsageEvent( usageRequests: OpenUsageRequest[], toolCalls: Record, toolErrors: Record, - skillsInvoked: Record, + toolResults: number, + skills: Record, startedAt: string, + endedAt: string, durationMs: number ): Record { - const sum = (selector: (req: OpenUsageRequest) => number): number => - usageRequests.reduce((total, req) => total + selector(req), 0); + const usage = buildUsageTotals(usageRequests); return { type: 'agent.subagent.usage', @@ -114,22 +225,22 @@ export function buildSubagentUsageEvent( agent_id: file.agentId, tool_use_id: file.toolUseId ?? '', agent_type: file.agentType ?? '', - spawn_depth: file.spawnDepth ?? 0, - description: '', + description: file.description ?? '', workflow_run: '', + spawn_depth: file.spawnDepth ?? 0, worktree: '', started_at: startedAt, + ended_at: endedAt, duration_ms: durationMs, - input_tokens: sum((r) => r.inputTokens), - cache_creation_5m_tokens: sum((r) => r.cacheCreation5mTokens), - cache_creation_1h_tokens: sum((r) => r.cacheCreation1hTokens), - cache_read_tokens: sum((r) => r.cacheReadTokens), - output_tokens: sum((r) => r.outputTokens), - web_search_requests: sum((r) => r.webSearchRequests), - web_fetch_requests: sum((r) => r.webFetchRequests), + model: primaryModel(usage), api_calls: usageRequests.length, - tool_calls: toolCalls, - tool_errors: toolErrors, - skills_invoked: skillsInvoked, + tool_calls: sumCounts(toolCalls), + tool_results: toolResults, + tool_errors: sumCounts(toolErrors), + tools: buildToolsBreakdown(toolCalls, toolErrors), + skills, + commands: [], + compactions: [], + usage, }; } From bf115130c333f43d9544e34db58c012a2e49ef70 Mon Sep 17 00:00:00 2001 From: Dzmitry Halai Date: Fri, 9 Oct 2026 11:43:32 +0300 Subject: [PATCH 32/35] fix(analytics): make claude-code-otlp parse-state lock and save atomic --- .../transcript/__tests__/parse-state.test.ts | 83 ++++++++++++++++++- .../transcript/parse-state.ts | 79 +++++++++++------- 2 files changed, 131 insertions(+), 31 deletions(-) diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts index 6f5e9ac35..08f92f09d 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts @@ -8,9 +8,10 @@ */ import { describe, it, expect, beforeEach, afterEach } from 'vitest'; -import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs'; +import { mkdtempSync, mkdirSync, writeFileSync, readFileSync, existsSync, rmSync, utimesSync } from 'node:fs'; +import { spawn } from 'node:child_process'; import { tmpdir } from 'node:os'; -import { join } from 'node:path'; +import { dirname, join } from 'node:path'; import { createParseState, loadParseState, @@ -22,6 +23,14 @@ import { let codemieHome: string; +/** Spawn-and-wait a throwaway process so its pid is guaranteed dead, for lock-reclaim tests. */ +async function exitedPid(): Promise { + return new Promise((resolve) => { + const child = spawn(process.execPath, ['-e', 'process.exit(0)']); + child.on('exit', () => resolve(child.pid as number)); + }); +} + beforeEach(() => { codemieHome = mkdtempSync(join(tmpdir(), 'codemie-home-')); process.env.CODEMIE_HOME = codemieHome; @@ -104,6 +113,17 @@ describe('loadParseState', () => { }); }); +describe('saveParseState', () => { + it('leaves no temp file behind after a successful save', async () => { + const sessionId = 'session-atomic'; + await saveParseState(sessionId, createParseState()); + + const filePath = join(codemieHome, 'analytics', 'state', `${sessionId}.json`); + expect(existsSync(filePath)).toBe(true); + expect(existsSync(`${filePath}.tmp`)).toBe(false); + }); +}); + describe('withParseStateLock', () => { it('still runs fn when the lock cannot be released cleanly, and leaves no stale lock file behind', async () => { const sessionId = 'session-lock-cleanup'; @@ -114,4 +134,63 @@ describe('withParseStateLock', () => { const second = await withParseStateLock(sessionId, async () => 'done-again'); expect(second).toBe('done-again'); }); + + it('serializes concurrent acquirers so two never run inside the critical section at once', async () => { + const sessionId = 'session-mutex'; + const events: string[] = []; + const run = (label: string) => + withParseStateLock(sessionId, async () => { + events.push(`${label}:enter`); + await new Promise((resolve) => setTimeout(resolve, 20)); + events.push(`${label}:exit`); + }); + + await Promise.all([run('a'), run('b'), run('c')]); + + // Each label's enter/exit pair stays adjacent — no other label's enter lands between them. + expect(events).toHaveLength(6); + for (let i = 0; i < events.length; i += 2) { + const label = events[i].split(':')[0]; + expect(events[i]).toBe(`${label}:enter`); + expect(events[i + 1]).toBe(`${label}:exit`); + } + }); + + it('reclaims a lock left behind by a process that has since exited', async () => { + const sessionId = 'session-dead-holder'; + const lockPath = join(codemieHome, 'analytics', 'state', `${sessionId}.json.lock`); + mkdirSync(dirname(lockPath), { recursive: true }); + writeFileSync(lockPath, String(await exitedPid())); + + const result = await withParseStateLock(sessionId, async () => 'reclaimed'); + + expect(result).toBe('reclaimed'); + }); + + it('does not reclaim a lock whose holder process is still alive', async () => { + const sessionId = 'session-live-holder'; + const lockPath = join(codemieHome, 'analytics', 'state', `${sessionId}.json.lock`); + mkdirSync(dirname(lockPath), { recursive: true }); + writeFileSync(lockPath, String(process.pid)); + + const pending = withParseStateLock(sessionId, async () => 'acquired'); + await new Promise((resolve) => setTimeout(resolve, 150)); + + // Still exactly the lock we wrote — this test's own pid is alive, so it was never reclaimed. + expect(readFileSync(lockPath, 'utf-8')).toBe(String(process.pid)); + + rmSync(lockPath, { force: true }); // simulate that holder releasing it + await expect(pending).resolves.toBe('acquired'); + }); + + it('reclaims an empty lock only once it is old enough to be abandoned', async () => { + const sessionId = 'session-empty-lock'; + const lockPath = join(codemieHome, 'analytics', 'state', `${sessionId}.json.lock`); + mkdirSync(dirname(lockPath), { recursive: true }); + writeFileSync(lockPath, ''); + const old = new Date(Date.now() - 60_000); + utimesSync(lockPath, old, old); + + await expect(withParseStateLock(sessionId, async () => 'reclaimed')).resolves.toBe('reclaimed'); + }); }); diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts index 948f63260..beeabab9d 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -8,7 +8,7 @@ * state to disk between parse passes, keyed by session id. */ -import { mkdir, open, readFile, rm, stat, writeFile } from 'node:fs/promises'; +import { mkdir, readFile, rename, rm, stat, writeFile } from 'node:fs/promises'; import { dirname } from 'node:path'; import { getCodemiePath } from '@/utils/paths.js'; @@ -84,64 +84,88 @@ export async function loadParseState(sessionId: string): Promise { const filePath = getParseStatePath(sessionId); + const tmpPath = `${filePath}.tmp`; await mkdir(dirname(filePath), { recursive: true }); - await writeFile(filePath, JSON.stringify(state, null, 2), 'utf-8'); + await writeFile(tmpPath, JSON.stringify(state, null, 2), 'utf-8'); + await rename(tmpPath, filePath); } const LOCK_RETRY_MS = 25; -const LOCK_STALE_MS = 5_000; +const LOCK_WAIT_BUDGET_MS = 10_000; +const EMPTY_LOCK_STALE_MS = 5_000; +// Windows reports EPERM/EACCES/EBUSY, not EEXIST, when creating a lock whose deletion is pending. +const LOCK_CONTENTION_CODES = new Set(['EEXIST', 'EPERM', 'EACCES', 'EBUSY']); function getLockPath(sessionId: string): string { return `${getParseStatePath(sessionId)}.lock`; } -async function isLockStale(lockPath: string): Promise { +/** A pid we cannot signal for any reason other than ESRCH (e.g. EPERM) is treated as alive. */ +function isPidAlive(pid: number): boolean { try { - const info = await stat(lockPath); - return Date.now() - info.mtimeMs > LOCK_STALE_MS; - } catch { - return true; // disappeared between our EEXIST and this check — treat as gone + process.kill(pid, 0); + return true; + } catch (err) { + return (err as NodeJS.ErrnoException).code !== 'ESRCH'; + } +} + +/** Remove the lock only if it still holds `expected`, so a lock created meanwhile is never removed. */ +async function removeLockIfUnchanged(lockPath: string, expected: string): Promise { + if ((await readFile(lockPath, 'utf-8').catch(() => null)) === expected) { + await rm(lockPath, { force: true }).catch(() => {}); } } /** * Serialize one session's load-mutate-save parse-state cycle across concurrent hook processes - * (e.g. sibling `SubagentStop` fires for the same session) via an exclusive-create lock file. - * Each hook fire is a fresh CLI process, so this cannot use an in-memory mutex. + * (e.g. sibling `SubagentStop` fires for the same session) via an exclusive-create lock file + * holding the owner's pid. Each hook fire is a fresh CLI process, so an in-memory mutex won't do. * - * A lock older than {@link LOCK_STALE_MS} is treated as abandoned (its holder crashed before - * releasing it) and stolen rather than awaited forever. Likewise, if the lock cannot be acquired - * within a bounded wait, `fn` still runs unlocked rather than hanging the hook indefinitely — - * occasional lost contention here is strictly better than analytics never shipping at all. + * A lock is reclaimed only when its owner pid is dead, never by age, so a slow holder keeps it. + * If the lock is not acquired within {@link LOCK_WAIT_BUDGET_MS}, this throws instead of running + * `fn` unlocked; the caller skips the pass and the next hook resumes from the saved offsets. */ export async function withParseStateLock(sessionId: string, fn: () => Promise): Promise { const lockPath = getLockPath(sessionId); await mkdir(dirname(lockPath), { recursive: true }); - const deadline = Date.now() + LOCK_STALE_MS * 2; - let acquired = false; + const deadline = Date.now() + LOCK_WAIT_BUDGET_MS; for (;;) { try { - const handle = await open(lockPath, 'wx'); - await handle.close(); - acquired = true; + await writeFile(lockPath, String(process.pid), { flag: 'wx' }); break; } catch (err) { - if ((err as NodeJS.ErrnoException).code !== 'EEXIST') { - break; // can't lock (e.g. permissions) — proceed unlocked rather than block forever + if (!LOCK_CONTENTION_CODES.has((err as NodeJS.ErrnoException).code ?? '')) { + throw err; } - if (await isLockStale(lockPath)) { - await rm(lockPath, { force: true }).catch(() => {}); + if (Date.now() > deadline) { + throw new Error(`Timed out waiting for the parse-state lock (session ${sessionId})`); + } + + const holder = await readFile(lockPath, 'utf-8').catch(() => null); + const holderPid = Number.parseInt(holder ?? '', 10); + if (holder !== null && !Number.isNaN(holderPid) && !isPidAlive(holderPid)) { + await removeLockIfUnchanged(lockPath, holder); continue; } - if (Date.now() > deadline) { - break; // gave the lock a fair wait; proceed unlocked rather than hang the hook + if (holder === '') { + // Created but not yet written, or its creator died in between. + const info = await stat(lockPath).catch(() => null); + if (info && Date.now() - info.mtimeMs > EMPTY_LOCK_STALE_MS) { + await removeLockIfUnchanged(lockPath, holder); + continue; + } } await new Promise((resolve) => setTimeout(resolve, LOCK_RETRY_MS)); } @@ -150,9 +174,6 @@ export async function withParseStateLock(sessionId: string, fn: () => Promise try { return await fn(); } finally { - // Only release a lock we hold — when we proceeded unlocked, the file belongs to another process. - if (acquired) { - await rm(lockPath, { force: true }).catch(() => {}); - } + await rm(lockPath, { force: true }).catch(() => {}); } } From 472b7ef9186d647e6cfe53561f4736c82e87cf36 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Fri, 9 Oct 2026 13:21:05 +0300 Subject: [PATCH 33/35] fix(analytics): align agent.usage.request identity with the CLI analytics contract request_id is now the API request id (top-level requestId, empty when the transcript has none) and message_id carries message.id. Requests are grouped by requestId, else message.id, plus model, through a single usageRequestKey helper. This lets the backend pair direct Claude Code requests with their OTel api_request records instead of counting them twice. Generated with AI --- .../fixtures/transcript-usage-direct.jsonl | 3 + .../transcript/__tests__/orchestrator.test.ts | 120 +++++++++++++++++- .../transcript/__tests__/parse-state.test.ts | 1 + .../__tests__/subagent-usage.test.ts | 2 +- .../__tests__/usage-request.test.ts | 95 ++++++++++++-- .../transcript/orchestrator.ts | 14 +- .../transcript/parse-state.ts | 5 +- .../transcript/usage-request.ts | 32 ++++- 8 files changed, 244 insertions(+), 28 deletions(-) create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage-direct.jsonl diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage-direct.jsonl b/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage-direct.jsonl new file mode 100644 index 000000000..b3249e144 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/fixtures/transcript-usage-direct.jsonl @@ -0,0 +1,3 @@ +{"sessionId":"session-usage-direct","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-09T09:53:24.591Z","uuid":"3cbae8dc-b9ab-49c2-831b-dc9d54cdb562","requestId":"req_011CfrTbrJgSQRWgsdY5FZFc","message":{"id":"msg_011CfrTbrWLhWiotBBJC62pV","role":"assistant","model":"claude-sonnet-5","stop_reason":"tool_use","usage":{"input_tokens":2,"cache_creation_input_tokens":22509,"cache_read_input_tokens":37433,"output_tokens":87,"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"service_tier":"standard","cache_creation":{"ephemeral_1h_input_tokens":22509,"ephemeral_5m_input_tokens":0},"speed":"standard","inference_geo":"global"}}} +{"sessionId":"session-usage-direct","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-09T09:53:25.413Z","uuid":"a37e9e67-d5d7-4571-a069-0df29ab1bf43","requestId":"req_011CfrTbrJgSQRWgsdY5FZFc","message":{"id":"msg_011CfrTbrWLhWiotBBJC62pV","role":"assistant","model":"claude-sonnet-5","stop_reason":"tool_use","usage":{"input_tokens":2,"cache_creation_input_tokens":22509,"cache_read_input_tokens":37433,"output_tokens":87,"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"service_tier":"standard","cache_creation":{"ephemeral_1h_input_tokens":22509,"ephemeral_5m_input_tokens":0},"speed":"standard","inference_geo":"global"}}} +{"sessionId":"session-usage-direct","gitBranch":"main","cwd":"/repo","timestamp":"2026-10-09T09:53:52.777Z","uuid":"2438b44d-0f86-4b8c-aae7-82f6814c5889","requestId":"req_011CfrTdvHcHvpWuQRM5zwz8","message":{"id":"msg_011CfrTdvaxJgBE1TDcwFnce","role":"assistant","model":"claude-sonnet-5","stop_reason":"tool_use","usage":{"input_tokens":2,"cache_creation_input_tokens":251,"cache_read_input_tokens":59942,"output_tokens":155,"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"service_tier":"standard","cache_creation":{"ephemeral_1h_input_tokens":251,"ephemeral_5m_input_tokens":0},"speed":"standard","inference_geo":"global"}}} diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index 8c44911e7..ff2eaeb5a 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -20,6 +20,7 @@ let transcriptDir: string; function usageLine(opts: { uuid: string; messageId: string; + requestId?: string; outputTokens: number; gitBranch?: string; stopReason?: string; @@ -30,6 +31,7 @@ function usageLine(opts: { cwd: '/repo', timestamp: opts.timestamp ?? '2026-10-01T00:00:00.000Z', uuid: opts.uuid, + ...(opts.requestId ? { requestId: opts.requestId } : {}), message: { id: opts.messageId, role: 'assistant', @@ -105,7 +107,7 @@ function writeSubagentFixture( } describe('collectMainTranscriptEvents — idempotent reparse', () => { - it('returns agent.usage.request events with identical request_id/model pairs across a crash-before-save re-parse', async () => { + it('returns agent.usage.request events with identical request_id/message_id/model triples across a crash-before-save re-parse', async () => { const { collectMainTranscriptEvents } = await import('../orchestrator.js'); const { saveParseState, createParseState } = await import('../parse-state.js'); @@ -120,7 +122,7 @@ describe('collectMainTranscriptEvents — idempotent reparse', () => { const firstPairs = first .filter((e) => e.type === 'agent.usage.request') - .map((e) => `${e.request_id}::${e.model}`) + .map((e) => `${e.request_id}::${e.message_id}::${e.model}`) .sort(); expect(firstPairs).toHaveLength(2); @@ -134,7 +136,7 @@ describe('collectMainTranscriptEvents — idempotent reparse', () => { const secondPairs = second .filter((e) => e.type === 'agent.usage.request') - .map((e) => `${e.request_id}::${e.model}`) + .map((e) => `${e.request_id}::${e.message_id}::${e.model}`) .sort(); expect(secondPairs).toHaveLength(2); @@ -310,6 +312,118 @@ describe('collectMainTranscriptEvents — tool-call accumulation', () => { }); }); +describe('collectMainTranscriptEvents — request identity', () => { + const usageEventsOf = (events: ForwardedEvent[]): ForwardedEvent[] => + events.filter((e) => e.type === 'agent.usage.request'); + + it('direct shape: merges the rows of one response into one event carrying request_id and message_id', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const transcriptPath = writeTranscript('transcript-direct.jsonl', [ + usageLine({ uuid: 'u1', messageId: 'msg_a', requestId: 'req_a', outputTokens: 10 }), + usageLine({ uuid: 'u2', messageId: 'msg_a', requestId: 'req_a', outputTokens: 40 }), + usageLine({ uuid: 'u3', messageId: 'msg_b', requestId: 'req_b', outputTokens: 20 }), + ]); + + const usage = usageEventsOf(await collectMainTranscriptEvents('session-direct', transcriptPath, 'Stop')); + + expect(usage.map((e) => [e.request_id, e.message_id]).sort()).toEqual([ + ['req_a', 'msg_a'], + ['req_b', 'msg_b'], + ]); + expect(usage.find((e) => e.request_id === 'req_a')?.output_tokens).toBe(40); + }); + + it('proxy shape: distinct message ids stay distinct events with an empty request_id', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const transcriptPath = writeTranscript('transcript-proxy.jsonl', [ + usageLine({ uuid: 'u1', messageId: 'msg_bdrk_1', outputTokens: 10 }), + usageLine({ uuid: 'u2', messageId: 'msg_bdrk_2', outputTokens: 20 }), + ]); + + const usage = usageEventsOf(await collectMainTranscriptEvents('session-proxy', transcriptPath, 'Stop')); + + expect(usage).toHaveLength(2); + expect(usage.every((e) => e.request_id === '')).toBe(true); + expect(usage.map((e) => e.message_id).sort()).toEqual(['msg_bdrk_1', 'msg_bdrk_2']); + }); + + it('mixed transcript: a request with requestId and one without are not merged', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const transcriptPath = writeTranscript('transcript-mixed.jsonl', [ + usageLine({ uuid: 'u1', messageId: 'msg_a', requestId: 'req_a', outputTokens: 10 }), + usageLine({ uuid: 'u2', messageId: 'msg_b', outputTokens: 20 }), + ]); + + const usage = usageEventsOf(await collectMainTranscriptEvents('session-mixed', transcriptPath, 'Stop')); + + expect(usage.map((e) => [e.request_id, e.message_id]).sort()).toEqual([ + ['', 'msg_b'], + ['req_a', 'msg_a'], + ]); + }); + + it('direct shape: summary api_calls counts a multi-row response once', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const transcriptPath = writeTranscript('transcript-direct-summary.jsonl', [ + usageLine({ uuid: 'u1', messageId: 'msg_a', requestId: 'req_a', outputTokens: 10 }), + usageLine({ uuid: 'u2', messageId: 'msg_a', requestId: 'req_a', outputTokens: 40 }), + usageLine({ uuid: 'u3', messageId: 'msg_b', requestId: 'req_b', outputTokens: 20 }), + ]); + + const events = await collectMainTranscriptEvents('session-direct-summary', transcriptPath, 'Stop'); + const summary = events.find((e) => e.type === 'agent.session.summary'); + + expect(summary?.api_calls).toBe(2); + expect(summary?.models).toEqual(['claude-sonnet-4-5-20250929']); + }); + + it('direct shape: a reparse after a state reset yields the same request_id/message_id set', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + const { saveParseState, createParseState } = await import('../parse-state.js'); + + const sessionId = 'session-direct-reparse'; + const transcriptPath = writeTranscript('transcript-direct-reparse.jsonl', [ + usageLine({ uuid: 'u1', messageId: 'msg_a', requestId: 'req_a', outputTokens: 10 }), + usageLine({ uuid: 'u2', messageId: 'msg_a', requestId: 'req_a', outputTokens: 40 }), + usageLine({ uuid: 'u3', messageId: 'msg_b', requestId: 'req_b', outputTokens: 20 }), + ]); + const idsOf = (events: ForwardedEvent[]): string[] => + usageEventsOf(events).map((e) => `${e.request_id}::${e.message_id}`).sort(); + + const first = idsOf(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + await saveParseState(sessionId, createParseState()); + const second = idsOf(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + + expect(first).toEqual(['req_a::msg_a', 'req_b::msg_b']); + expect(second).toEqual(first); + }); +}); + +describe('collectSubagentTranscriptEvents — request identity', () => { + it('emits request_id and message_id on agent.usage.request events, merging rows of one response', async () => { + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-sub-identity'; + const file = writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'u1', messageId: 'msg_a', requestId: 'req_a', outputTokens: 10 }), + usageLine({ uuid: 'u2', messageId: 'msg_a', requestId: 'req_a', outputTokens: 40 }), + usageLine({ uuid: 'u3', messageId: 'msg_bdrk_b', outputTokens: 20 }), + ]); + + const events = await collectSubagentTranscriptEvents(sessionId, file); + const usage = events.filter((e) => e.type === 'agent.usage.request'); + + expect(usage.map((e) => [e.request_id, e.message_id]).sort()).toEqual([ + ['', 'msg_bdrk_b'], + ['req_a', 'msg_a'], + ]); + }); +}); + describe('collectSubagentTranscriptEvents — SessionEnd backstop (three subagents, one pre-advanced)', () => { it('returns exactly three agent.subagent.usage events — one per subagent, including the one whose own SubagentStop already advanced its offset — never a fourth', async () => { const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts index 08f92f09d..ad93955b9 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts @@ -74,6 +74,7 @@ describe('loadParseState', () => { it('round-trips openRequests and branchCounts exactly through saveParseState', async () => { const openRequest: OpenUsageRequest = { requestId: 'req1', + messageId: 'msg1', model: 'claude-3-5-sonnet', modelRaw: 'claude-3-5-sonnet-20241022', timestamp: '2026-10-01T00:00:00.000Z', diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts index 89d932c83..ca8bd3230 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts @@ -99,7 +99,7 @@ async function buildFixture( function usageRequest(overrides: Partial = {}): OpenUsageRequest { return { - requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + requestId: 'r1', messageId: 'm1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', speed: 'standard', inferenceGeo: '', serviceTier: 'standard', inputTokens: 0, cacheCreation5mTokens: 0, cacheCreation1hTokens: 0, cacheReadTokens: 0, outputTokens: 0, webSearchRequests: 0, webFetchRequests: 0, diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts index 210b2f55f..67f162875 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts @@ -2,7 +2,13 @@ * Tests for `agent.usage.request` extraction and merge: `parseUsageLine`, * `mergeUsageRequest`, and `buildUsageRequestEvent`. * - * Fixture: `fixtures/transcript-usage.jsonl` — one line with no `message.usage` + * Fixtures: `fixtures/transcript-usage.jsonl` is the proxy shape (no top-level `requestId`, + * so `requestId === ''` and `messageId` carries `message.id`); + * `fixtures/transcript-usage-direct.jsonl` is the direct Claude Code shape (top-level + * `requestId = req_...` plus `message.id = msg_...`, two rows of one response and one row of a + * second). + * + * `fixtures/transcript-usage.jsonl` — one line with no `message.usage` * (must parse to `null`), two lines sharing the same `message.id` where the * second has a higher `output_tokens` and a non-empty `stop_reason` the first * lacks (feeds `mergeUsageRequest`), and one fully-populated "normal" line @@ -12,14 +18,17 @@ import { describe, it, expect, beforeAll } from 'vitest'; import { readFile } from 'node:fs/promises'; import { join } from 'node:path'; -import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from '../usage-request.js'; +import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent, usageRequestKey } from '../usage-request.js'; import type { OpenUsageRequest } from '../parse-state.js'; let lines: string[]; +let directLines: string[]; beforeAll(async () => { const raw = await readFile(join(__dirname, 'fixtures', 'transcript-usage.jsonl'), 'utf-8'); lines = raw.split('\n').filter((l) => l.trim().length > 0); + const directRaw = await readFile(join(__dirname, 'fixtures', 'transcript-usage-direct.jsonl'), 'utf-8'); + directLines = directRaw.split('\n').filter((l) => l.trim().length > 0); }); describe('parseUsageLine', () => { @@ -34,7 +43,7 @@ describe('parseUsageLine', () => { expect(parseUsageLine('not valid json {{{', 'main', '', '')).toBeNull(); }); - it('returns null for a usage-bearing line with no message.id, instead of collapsing it onto a shared ::model key', () => { + it('returns null for a usage-bearing line with neither requestId nor message.id, instead of collapsing it onto a shared ::model key', () => { const line = JSON.stringify({ timestamp: '2026-10-01T00:00:04.000Z', message: { @@ -53,8 +62,9 @@ describe('parseUsageLine', () => { expect(req).not.toBeNull(); const r = req as OpenUsageRequest; - // request_id comes from message.id, not any top-level requestId. - expect(r.requestId).toBe('msg_normal_1'); + // No top-level requestId in the proxy shape; message.id lands in messageId. + expect(r.requestId).toBe(''); + expect(r.messageId).toBe('msg_normal_1'); // modelRaw is the transcript's own literal message.model (unresolved). expect(r.modelRaw).toBe('claude-sonnet-4-5-20250929'); // model is resolved via parseBackendModelName (x-litellm-model-name) first. @@ -82,6 +92,25 @@ describe('parseUsageLine', () => { expect(r.agentId).toBe(''); }); + it('reads the top-level requestId and message.id separately from a direct-session line', () => { + const r = parseUsageLine(directLines[0], 'main', '', '') as OpenUsageRequest; + + expect(r.requestId).toBe('req_011CfrTbrJgSQRWgsdY5FZFc'); + expect(r.messageId).toBe('msg_011CfrTbrWLhWiotBBJC62pV'); + }); + + it('keeps a line with a top-level requestId but no message.id', () => { + const line = JSON.stringify({ + requestId: 'req_only', + message: { role: 'assistant', model: 'm', usage: { input_tokens: 1, output_tokens: 1 } }, + }); + + const r = parseUsageLine(line, 'main', '', '') as OpenUsageRequest; + + expect(r.requestId).toBe('req_only'); + expect(r.messageId).toBe(''); + }); + it('passes scopeKind/scopeName/agentId through verbatim from its own parameters', () => { const req = parseUsageLine(lines[3], 'agent', 'reviewer', 'agent-42'); @@ -101,7 +130,7 @@ describe('mergeUsageRequest', () => { const a = first as OpenUsageRequest; const b = second as OpenUsageRequest; - expect(a.requestId).toBe(b.requestId); + expect(a.messageId).toBe(b.messageId); expect(a.outputTokens).toBe(50); expect(a.stopReason).toBe(''); expect(b.outputTokens).toBe(120); @@ -119,9 +148,28 @@ describe('mergeUsageRequest', () => { expect(b.outputTokens).toBe(120); }); + it('merges the two rows of one direct-session response into one record with both ids', () => { + const a = parseUsageLine(directLines[0], 'main', '', '') as OpenUsageRequest; + const b = parseUsageLine(directLines[1], 'main', '', '') as OpenUsageRequest; + + expect(usageRequestKey(a)).toBe(usageRequestKey(b)); + const merged = mergeUsageRequest(a, b); + + expect(merged.requestId).toBe('req_011CfrTbrJgSQRWgsdY5FZFc'); + expect(merged.messageId).toBe('msg_011CfrTbrWLhWiotBBJC62pV'); + expect(merged.outputTokens).toBe(Math.max(a.outputTokens, b.outputTokens)); + }); + + it('keeps messageId when only the earlier record has it', () => { + const a = parseUsageLine(lines[1], 'main', '', '') as OpenUsageRequest; + const merged = mergeUsageRequest(a, { ...a, messageId: '' }); + + expect(merged.messageId).toBe('msg_pair_1'); + }); + it('takes every numeric field as Math.max of the two inputs', () => { const a: OpenUsageRequest = { - requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + requestId: 'r1', messageId: 'm1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', speed: 'standard', inferenceGeo: '', serviceTier: 'standard', inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, @@ -150,7 +198,7 @@ describe('mergeUsageRequest', () => { it('does not mutate either input and returns a new object', () => { const a: OpenUsageRequest = { - requestId: 'r1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', + requestId: 'r1', messageId: 'm1', model: 'm', modelRaw: 'm-raw', timestamp: 't1', speed: '', inferenceGeo: '', serviceTier: '', inputTokens: 1, cacheCreation5mTokens: 0, cacheCreation1hTokens: 0, cacheReadTokens: 0, outputTokens: 1, webSearchRequests: 0, webFetchRequests: 0, @@ -175,7 +223,7 @@ describe('mergeUsageRequest', () => { describe('buildUsageRequestEvent', () => { it('maps every OpenUsageRequest field to its snake_case event field, with an explicit type', () => { const req: OpenUsageRequest = { - requestId: 'req-1', model: 'resolved-model', modelRaw: 'literal-model', timestamp: '2026-10-01T00:00:00.000Z', + requestId: 'req-1', messageId: 'msg-1', model: 'resolved-model', modelRaw: 'literal-model', timestamp: '2026-10-01T00:00:00.000Z', speed: 'fast', inferenceGeo: 'us', serviceTier: 'priority', inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, @@ -189,6 +237,7 @@ describe('buildUsageRequestEvent', () => { type: 'agent.usage.request', session_id: 'session-123', request_id: 'req-1', + message_id: 'msg-1', model_raw: 'literal-model', model: 'resolved-model', speed: 'fast', @@ -214,3 +263,31 @@ describe('buildUsageRequestEvent', () => { expect(event).not.toHaveProperty('schema_version'); }); }); + +describe('usageRequestKey', () => { + it('prefers requestId over messageId', () => { + expect(usageRequestKey({ requestId: 'req_1', messageId: 'msg_1', model: 'm' })).toBe('req_1::m'); + }); + + it('falls back to messageId when requestId is empty', () => { + expect(usageRequestKey({ requestId: '', messageId: 'msg_1', model: 'm' })).toBe('msg_1::m'); + }); + + it('gives different keys to the same messageId under different models', () => { + const a = usageRequestKey({ requestId: '', messageId: 'msg_1', model: 'm1' }); + const b = usageRequestKey({ requestId: '', messageId: 'msg_1', model: 'm2' }); + + expect(a).not.toBe(b); + }); +}); + +describe('buildUsageRequestEvent - direct session', () => { + it('emits the API request id as request_id and message.id as message_id', () => { + const req = parseUsageLine(directLines[2], 'main', '', '') as OpenUsageRequest; + + const event = buildUsageRequestEvent('session-direct', req); + + expect(event.request_id).toBe('req_011CfrTdvHcHvpWuQRM5zwz8'); + expect(event.message_id).toBe('msg_011CfrTdvaxJgBE1TDcwFnce'); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index d1a2aa623..860faa08c 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -20,7 +20,7 @@ import { readFile, stat } from 'node:fs/promises'; import { loadParseState, saveParseState, withParseStateLock } from './parse-state.js'; import { readNewLines } from './transcript-reader.js'; -import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent } from './usage-request.js'; +import { parseUsageLine, mergeUsageRequest, buildUsageRequestEvent, usageRequestKey } from './usage-request.js'; import { updateBranchCounts, buildSessionSummaryEvent, @@ -167,12 +167,12 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ // Pass 2: models, one count per distinct request — a request can span several // streaming/finalizing transcript lines, so lines are deduped by the same - // `${requestId}::${model}` key `state.openRequests` uses before counting. + // `usageRequestKey()` key `state.openRequests` uses before counting. const modelByRequestKey = new Map(); for (const line of rawLines) { const parsedUsage = parseUsageLine(line, 'main', '', ''); if (parsedUsage) { - modelByRequestKey.set(`${parsedUsage.requestId}::${parsedUsage.model}`, parsedUsage.model); + modelByRequestKey.set(usageRequestKey(parsedUsage), parsedUsage.model); } } for (const model of modelByRequestKey.values()) { @@ -216,7 +216,7 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ * * - Loads persisted state, reads only the lines appended since `state.mainOffset`. * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, - * keyed by `${requestId}::${model}` (matching `parse-state.ts`'s documented key shape), and + * keyed by `usageRequestKey()`, and * updates `state.branchCounts` from every new line's `gitBranch` (regardless of whether that * line carried usage). * - Returns one `agent.usage.request` JSON string per request key touched by this pass. @@ -262,7 +262,7 @@ export async function collectMainTranscriptEvents( const parsed = parseUsageLine(line, 'main', '', ''); if (parsed) { - const key = `${parsed.requestId}::${parsed.model}`; + const key = usageRequestKey(parsed); const existing = state.openRequests[key]; state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; touchedKeys.add(key); @@ -442,7 +442,7 @@ export async function subagentNeedsBackstop(sessionId: string, subagentFile: Sub * `state.subagentOffsets[subagentFile.agentId]` (defaulting to 0 for a never-before-seen * agent). * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, - * scoped `scopeKind: 'agent'`, keyed by `${requestId}::${model}` — same merge/key convention + * scoped `scopeKind: 'agent'`, keyed by `usageRequestKey()` — same merge/key convention * `collectMainTranscriptEvents` uses for the main transcript. * - Returns one `agent.usage.request` JSON string per request key touched by *this* pass (no new * lines means no new events — a no-op reparse returns nothing at this layer). @@ -481,7 +481,7 @@ export async function collectSubagentTranscriptEvents( for (const line of lines) { const parsed = parseUsageLine(line, 'agent', '', subagentFile.agentId); if (parsed) { - const key = `${parsed.requestId}::${parsed.model}`; + const key = usageRequestKey(parsed); const existing = state.openRequests[key]; state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; touchedKeys.add(key); diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts index beeabab9d..8679fe269 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -13,7 +13,10 @@ import { dirname } from 'node:path'; import { getCodemiePath } from '@/utils/paths.js'; export interface OpenUsageRequest { + /** API request id (`req_...`); `''` when the transcript has none (e.g. behind the proxy). */ requestId: string; + /** `message.id` (`msg_...`). */ + messageId: string; model: string; modelRaw: string; timestamp: string; @@ -38,7 +41,7 @@ export interface OpenUsageRequest { export interface TranscriptParseState { mainOffset: number; subagentOffsets: Record; - openRequests: Record; // key: `${requestId}::${model}` + openRequests: Record; // key: usageRequestKey() activeSkill: string; branchCounts: Record; compactionCount: number; diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index 8da8600b1..fd8b42e6b 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -9,9 +9,13 @@ * Claude Code can write more than one JSONL row for the same API response (progressive * streaming chunks, or a later row that fills in `stop_reason` once the turn finishes), so * callers parse every candidate line and merge same-identity records with - * {@link mergeUsageRequest} — this module trusts the caller to key records by - * `${requestId}::${model}` (see `parse-state.ts`'s `openRequests`) before merging; it does + * {@link mergeUsageRequest} — this module trusts the caller to key records with + * {@link usageRequestKey} (see `parse-state.ts`'s `openRequests`) before merging; it does * not itself check that two records it is asked to merge actually share that identity. + * + * Request identity follows the CLI analytics contract: `requestId` is the API request id + * (`req_...`, the transcript's top-level `requestId`; `''` when the transcript has none, as + * behind the proxy) and `messageId` is `message.id` (`msg_...`). */ import { type RoutingHeaderSource } from '../../../../utils/routing-headers.mjs'; @@ -27,6 +31,7 @@ interface TranscriptUsageLine { timestamp?: string; gitBranch?: string; isApiError?: boolean; + requestId?: string; message?: RoutingHeaderSource & { id?: string; model?: string; @@ -78,10 +83,11 @@ export function parseUsageLine( return null; } - // openRequests keys on `${requestId}::${model}` — an empty requestId would collide every such - // line in the session into one record instead of being skipped. - const requestId = parsed.message?.id ?? ''; - if (!requestId) { + // openRequests keys on `${requestId || messageId}::${model}` — with neither id every such + // line in the session would collide into one record, so it is skipped instead. + const requestId = parsed.requestId ?? ''; + const messageId = parsed.message?.id ?? ''; + if (!requestId && !messageId) { return null; } @@ -94,6 +100,7 @@ export function parseUsageLine( return { requestId, + messageId, model, modelRaw, timestamp: parsed.timestamp ?? '', @@ -117,9 +124,18 @@ export function parseUsageLine( }; } +/** + * Identity key for a logical request: the API request id when the transcript has one, else + * `message.id`, plus the resolved model. Single source of truth for every `openRequests` / + * dedupe key. + */ +export function usageRequestKey(r: { requestId: string; messageId: string; model: string }): string { + return `${r.requestId || r.messageId}::${r.model}`; +} + /** * Merge two {@link OpenUsageRequest} records the caller has already identified as the same - * logical request (same `requestId`+`model` — this function does not verify that itself). + * logical request (same {@link usageRequestKey} — this function does not verify that itself). * Every numeric field takes the max of the two (a later streaming/finalizing row only ever adds * usage, never subtracts it); every non-numeric field takes `b`'s value when non-empty, else * falls back to `a`'s — so a later row that fills in a previously-empty field (e.g. @@ -131,6 +147,7 @@ export function parseUsageLine( export function mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): OpenUsageRequest { return { requestId: b.requestId || a.requestId, + messageId: b.messageId || a.messageId, model: b.model || a.model, modelRaw: b.modelRaw || a.modelRaw, timestamp: b.timestamp || a.timestamp, @@ -163,6 +180,7 @@ export function buildUsageRequestEvent(sessionId: string, req: OpenUsageRequest) type: 'agent.usage.request', session_id: sessionId, request_id: req.requestId, + message_id: req.messageId, model_raw: req.modelRaw, model: req.model, speed: req.speed, From 93b7cf52337e876001976e9ffed51976341db153 Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Fri, 9 Oct 2026 13:55:58 +0300 Subject: [PATCH 34/35] fix(analytics): fill agent.session.summary gaps from transcript signals Derive the summary fields the backend reads only from the summary once one exists, instead of sending null, omitting them or sending the wrong shape: - lines_added/lines_removed from applied edit diffs (null until measured) - commands in call order with repeats; primary_command is the first - turns from real user prompts - compactions, compaction_count and compaction_pre_tokens from compact_boundary lines (replaces the PreCompact hook counter) - client_versions, git_branch and title (latest ai-title) - branch_counts count user lines only; branch_dominant ties go to the later branch Generated with AI --- .../transcript/__tests__/orchestrator.test.ts | 170 ++++++++++- .../transcript/__tests__/parse-state.test.ts | 2 - .../__tests__/session-signals.test.ts | 268 ++++++++++++++++++ .../__tests__/session-summary.test.ts | 186 +++++++++++- .../transcript/orchestrator.ts | 69 +++-- .../transcript/parse-state.ts | 2 - .../transcript/session-signals.ts | 186 ++++++++++++ .../transcript/session-summary.ts | 44 ++- .../claude-named-invocations.test.ts | 38 ++- .../session/claude-named-invocations.ts | 23 +- 10 files changed, 929 insertions(+), 59 deletions(-) create mode 100644 src/agents/plugins/claude-code-otlp/transcript/__tests__/session-signals.test.ts create mode 100644 src/agents/plugins/claude-code-otlp/transcript/session-signals.ts diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index ff2eaeb5a..7d1ea0061 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -186,12 +186,28 @@ describe('collectMainTranscriptEvents — PreCompact trigger', () => { }); }); -describe('collectMainTranscriptEvents — compaction_count', () => { - it('persists one increment per PreCompact trigger and surfaces the cumulative count on a later summary', async () => { +/** A `type: 'user'` transcript line on `gitBranch`, with optional extra fields. */ +function userLine(content: unknown, extra: Record = {}, gitBranch = 'feature'): string { + return JSON.stringify({ + type: 'user', + gitBranch, + cwd: '/repo', + timestamp: '2026-10-01T00:00:00.000Z', + message: { role: 'user', content }, + ...extra, + }); +} + +function commandText(name: string): string { + return `/${name}\n${name}\n`; +} + +describe('collectMainTranscriptEvents — compactions', () => { + it('does not count PreCompact hook fires; compaction_count comes from compact_boundary lines', async () => { const { collectMainTranscriptEvents } = await import('../orchestrator.js'); - const sessionId = 'session-compaction'; - const transcriptPath = writeTranscript('transcript-compaction.jsonl', [ + const sessionId = 'session-compaction-hook'; + const transcriptPath = writeTranscript('transcript-compaction-hook.jsonl', [ usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), ]); @@ -199,9 +215,149 @@ describe('collectMainTranscriptEvents — compaction_count', () => { await collectMainTranscriptEvents(sessionId, transcriptPath, 'PreCompact'); const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); - const summaryEvents = events.filter((e) => e.type === 'agent.session.summary'); - expect(summaryEvents).toHaveLength(1); - expect(summaryEvents[0].compaction_count).toBe(2); + const summary = events.filter((e) => e.type === 'agent.session.summary')[0]; + expect(summary.compaction_count).toBe(0); + expect(summary.compactions).toEqual([]); + }); + + it('reports a boundary line as a completed compaction on the next summary', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-compaction-boundary'; + const transcriptPath = writeTranscript('transcript-compaction-boundary.jsonl', [ + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + JSON.stringify({ + type: 'system', + subtype: 'compact_boundary', + timestamp: '2026-10-01T00:10:00.000Z', + compactMetadata: { trigger: 'auto', preTokens: 1000, postTokens: 300, durationMs: 60_000 }, + }), + ]); + + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + const summary = events.filter((e) => e.type === 'agent.session.summary')[0]; + + expect(summary.compaction_count).toBe(1); + expect(summary.compaction_pre_tokens).toBe(1000); + }); +}); + +describe('collectMainTranscriptEvents — summary transcript signals', () => { + async function runFinalSummary( + sessionId: string, + lines: string[] + ): Promise { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + const transcriptPath = writeTranscript(`${sessionId}.jsonl`, lines); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd')); + return events.filter((e) => e.type === 'agent.session.summary')[0]; + } + + const editResult = userLine( + [{ type: 'tool_result', tool_use_id: 't1', content: 'ok' }], + { toolUseResult: { structuredPatch: [{ lines: [' ctx', '-old', '+new1', '+new2'] }] } } + ); + + it('derives lines, turns, commands, compactions, versions, title and branch from the transcript', async () => { + const summary = await runFinalSummary('session-signals', [ + JSON.stringify({ type: 'ai-title', aiTitle: 'First title' }), + userLine('hello', { version: '2.1.295' }), + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + editResult, + userLine(commandText('commit')), + userLine(commandText('plan')), + userLine(commandText('plan')), + JSON.stringify({ + type: 'system', + subtype: 'compact_boundary', + timestamp: '2026-10-01T00:10:00.000Z', + compactMetadata: { trigger: 'auto', preTokens: 1000, postTokens: 300, durationMs: 60_000 }, + }), + JSON.stringify({ type: 'ai-title', aiTitle: 'Latest title' }), + ]); + + expect(summary.lines_added).toBe(2); + expect(summary.lines_removed).toBe(1); + // 'hello' + three command lines; the tool_result-only line is not a prompt. + expect(summary.turns).toBe(4); + expect(summary.commands).toEqual(['commit', 'plan', 'plan']); + expect(summary.primary_command).toBe('commit'); + expect(summary.compaction_count).toBe(1); + expect(summary.compactions).toEqual([ + { + start: '2026-10-01T00:09:00.000Z', + end: '2026-10-01T00:10:00.000Z', + duration_ms: 60_000, + trigger: 'auto', + pre_tokens: 1000, + post_tokens: 300, + dropped_tokens: 700, + }, + ]); + expect(summary.client_versions).toEqual(['2.1.295']); + expect(summary.title).toBe('Latest title'); + expect(summary.git_branch).toBe('feature'); + }); + + it('leaves lines_* null when the transcript has no applied edit result', async () => { + const summary = await runFinalSummary('session-no-edits', [ + userLine('hello'), + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50 }), + ]); + + expect(summary.lines_added).toBeNull(); + expect(summary.lines_removed).toBeNull(); + }); + + it('recomputes the signals from the whole file on a later pass (cumulative, not just new lines)', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-cumulative'; + const transcriptPath = writeTranscript('transcript-cumulative.jsonl', [userLine('first'), editResult]); + await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); + + writeFileSync(transcriptPath, [userLine('first'), editResult, userLine('second'), editResult].map((l) => l + '\n').join('')); + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'SessionEnd')); + const summary = events.filter((e) => e.type === 'agent.session.summary')[0]; + + expect(summary.turns).toBe(2); + expect(summary.lines_added).toBe(4); + expect(summary.lines_removed).toBe(2); + }); +}); + +describe('collectMainTranscriptEvents — branch_counts', () => { + it('counts user lines only, not assistant usage lines', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-branch-user-only'; + const transcriptPath = writeTranscript('transcript-branch-user-only.jsonl', [ + userLine('one', {}, 'feature'), + usageLine({ uuid: 'uuid-1', messageId: 'msg-1', outputTokens: 50, gitBranch: 'main' }), + usageLine({ uuid: 'uuid-2', messageId: 'msg-2', outputTokens: 50, gitBranch: 'main' }), + userLine('two', {}, 'feature'), + ]); + + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + const summary = events.filter((e) => e.type === 'agent.session.summary')[0]; + + expect(summary.branch_counts).toEqual({ feature: 2 }); + expect(summary.branch_dominant).toBe('feature'); + }); + + it('resolves a tied branch_dominant to the later branch', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-branch-tie'; + const transcriptPath = writeTranscript('transcript-branch-tie.jsonl', [ + userLine('one', {}, 'main'), + userLine('two', {}, 'feature'), + ]); + + const events = parseAll(await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop')); + const summary = events.filter((e) => e.type === 'agent.session.summary')[0]; + + expect(summary.branch_dominant).toBe('feature'); }); }); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts index ad93955b9..f8e626bc9 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts @@ -49,7 +49,6 @@ describe('createParseState', () => { openRequests: {}, activeSkill: '', branchCounts: {}, - compactionCount: 0, }); }); }); @@ -102,7 +101,6 @@ describe('loadParseState', () => { openRequests: { 'req1::claude-3-5-sonnet': openRequest }, activeSkill: 'brainstorming', branchCounts: { main: 3, feature: 1 }, - compactionCount: 2, }; await saveParseState('session-roundtrip', state); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-signals.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-signals.test.ts new file mode 100644 index 000000000..2c7132f63 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-signals.test.ts @@ -0,0 +1,268 @@ +/** + * Tests for the pure `agent.session.summary` transcript-signal extractors: edit-diff line + * counts, user turns, compactions, client versions and the session title. + */ + +import { describe, it, expect } from 'vitest'; +import { + collectClientVersions, + collectCompactions, + collectEditLineStats, + countTurns, + latestTitle, + type SignalLine, +} from '../session-signals.js'; + +function patchResult(...hunkLines: string[][]): SignalLine { + return { + type: 'user', + toolUseResult: { structuredPatch: hunkLines.map((lines) => ({ lines })) }, + }; +} + +function boundary(timestamp: string, compactMetadata: SignalLine['compactMetadata']): SignalLine { + return { type: 'system', subtype: 'compact_boundary', timestamp, compactMetadata }; +} + +describe('collectEditLineStats', () => { + it('returns null for both counts when no result reports a diff', () => { + expect(collectEditLineStats([])).toEqual({ linesAdded: null, linesRemoved: null }); + expect( + collectEditLineStats([ + { type: 'user', message: { content: 'hello' } }, + { type: 'user', toolUseResult: 'plain string result' }, + { type: 'user', toolUseResult: { stdout: 'x' } }, + ]) + ).toEqual({ linesAdded: null, linesRemoved: null }); + }); + + it('counts + and - lines of a structuredPatch and ignores context lines', () => { + expect(collectEditLineStats([patchResult([' ctx', '-old', '+new1', '+new2', ' ctx'])])).toEqual({ + linesAdded: 2, + linesRemoved: 1, + }); + }); + + it('sums every hunk of a patch and every result in the transcript', () => { + const stats = collectEditLineStats([ + patchResult(['-a', '+b'], ['+c']), + patchResult(['-d']), + ]); + expect(stats).toEqual({ linesAdded: 2, linesRemoved: 2 }); + }); + + it('reports 0 (not null) once a diff was measured that changed nothing on one side', () => { + expect(collectEditLineStats([patchResult(['+only added'])])).toEqual({ + linesAdded: 1, + linesRemoved: 0, + }); + }); + + it('counts the lines of a created file, without counting a trailing newline as a line', () => { + const create = (content: string): SignalLine => ({ + type: 'user', + toolUseResult: { type: 'create', content }, + }); + expect(collectEditLineStats([create('a\nb\nc')]).linesAdded).toBe(3); + expect(collectEditLineStats([create('a\nb\nc\n')]).linesAdded).toBe(3); + expect(collectEditLineStats([create('')]).linesAdded).toBe(0); + expect(collectEditLineStats([create('a\nb')]).linesRemoved).toBe(0); + }); + + it('skips a result of type create that has no string content', () => { + expect( + collectEditLineStats([{ type: 'user', toolUseResult: { type: 'create' } }]) + ).toEqual({ linesAdded: null, linesRemoved: null }); + }); + + it('tolerates malformed hunks', () => { + const line: SignalLine = { + type: 'user', + toolUseResult: { structuredPatch: [null, {}, { lines: 'nope' }, { lines: [1, '+ok'] }] }, + }; + expect(collectEditLineStats([line])).toEqual({ linesAdded: 1, linesRemoved: 0 }); + }); +}); + +describe('countTurns', () => { + it('counts string-content and text-block user prompts', () => { + expect( + countTurns([ + { type: 'user', message: { content: 'plain prompt' } }, + { type: 'user', message: { content: [{ type: 'text', text: 'block prompt' }] } }, + ]) + ).toBe(2); + }); + + it('does not count tool_result-only user lines', () => { + expect( + countTurns([{ type: 'user', message: { content: [{ type: 'tool_result', tool_use_id: 't1' }] } }]) + ).toBe(0); + }); + + it('does not count meta, sidechain or compaction-summary lines', () => { + const content = 'text'; + expect( + countTurns([ + { type: 'user', isMeta: true, message: { content } }, + { type: 'user', isSidechain: true, message: { content } }, + { type: 'user', isCompactSummary: true, message: { content } }, + ]) + ).toBe(0); + }); + + it('does not count assistant or system lines', () => { + expect( + countTurns([ + { type: 'assistant', message: { content: [{ type: 'text', text: 'reply' }] } }, + { type: 'system' }, + ]) + ).toBe(0); + }); + + it('returns 0 for no lines', () => { + expect(countTurns([])).toBe(0); + }); +}); + +describe('collectCompactions', () => { + it('maps a compact_boundary line to the contract element shape', () => { + const [compaction] = collectCompactions([ + boundary('2026-10-01T00:10:00.000Z', { + trigger: 'auto', + preTokens: 1000, + postTokens: 300, + durationMs: 60_000, + }), + ]); + + expect(compaction).toEqual({ + start: '2026-10-01T00:09:00.000Z', + end: '2026-10-01T00:10:00.000Z', + duration_ms: 60_000, + trigger: 'auto', + pre_tokens: 1000, + post_tokens: 300, + dropped_tokens: 700, + }); + }); + + it('returns one element per boundary, in transcript order', () => { + const compactions = collectCompactions([ + boundary('2026-10-01T00:10:00.000Z', { trigger: 'auto', preTokens: 1 }), + { type: 'user', message: { content: 'prompt' } }, + boundary('2026-10-01T00:20:00.000Z', { trigger: 'manual', preTokens: 2 }), + ]); + expect(compactions.map((c) => c.trigger)).toEqual(['auto', 'manual']); + }); + + it('ignores system lines of other subtypes and non-system lines', () => { + expect( + collectCompactions([ + { type: 'system', subtype: 'local_command' }, + { type: 'user', subtype: 'compact_boundary' }, + { type: 'user', isCompactSummary: true }, + ]) + ).toEqual([]); + }); + + it('uses null for unreported numbers and derives nothing from them', () => { + const [compaction] = collectCompactions([boundary('2026-10-01T00:10:00.000Z', { trigger: 'auto' })]); + expect(compaction).toMatchObject({ + start: null, + duration_ms: null, + pre_tokens: null, + post_tokens: null, + dropped_tokens: null, + }); + }); + + it('leaves dropped_tokens null when post exceeds pre', () => { + const [compaction] = collectCompactions([ + boundary('2026-10-01T00:10:00.000Z', { preTokens: 100, postTokens: 300, durationMs: 1000 }), + ]); + expect(compaction.dropped_tokens).toBeNull(); + expect(compaction.pre_tokens).toBe(100); + expect(compaction.post_tokens).toBe(300); + }); + + it('leaves start null for an unparseable timestamp and keeps end as given', () => { + const [compaction] = collectCompactions([boundary('not-a-date', { durationMs: 1000 })]); + expect(compaction.start).toBeNull(); + expect(compaction.end).toBe('not-a-date'); + }); + + it('uses an empty end and null start when the boundary line has no timestamp', () => { + const [compaction] = collectCompactions([ + { type: 'system', subtype: 'compact_boundary', compactMetadata: { durationMs: 1000 } }, + ]); + expect(compaction.end).toBe(''); + expect(compaction.start).toBeNull(); + }); + + it('rejects negative and non-numeric metadata values', () => { + const [compaction] = collectCompactions([ + boundary('2026-10-01T00:10:00.000Z', { + preTokens: -5, + postTokens: '10', + durationMs: Number.NaN, + trigger: 3, + }), + ]); + expect(compaction).toMatchObject({ + start: null, + duration_ms: null, + pre_tokens: null, + post_tokens: null, + dropped_tokens: null, + trigger: '', + }); + }); +}); + +describe('collectClientVersions', () => { + it('returns distinct versions in order of first appearance', () => { + expect( + collectClientVersions([ + { version: '2.1.283' }, + { version: '2.1.284' }, + { version: '2.1.283' }, + ]) + ).toEqual(['2.1.283', '2.1.284']); + }); + + it('skips lines without a usable version', () => { + expect(collectClientVersions([{}, { version: '' }, { version: 5 as unknown as string }])).toEqual([]); + }); +}); + +describe('latestTitle', () => { + it('returns the latest ai-title', () => { + expect( + latestTitle([ + { type: 'ai-title', aiTitle: 'First' }, + { type: 'user' }, + { type: 'ai-title', aiTitle: 'Second' }, + ]) + ).toBe('Second'); + }); + + it('skips blank titles and falls back to an earlier one', () => { + expect( + latestTitle([ + { type: 'ai-title', aiTitle: 'Earlier' }, + { type: 'ai-title', aiTitle: ' ' }, + ]) + ).toBe('Earlier'); + }); + + it('trims and cuts the title to 200 characters', () => { + const title = latestTitle([{ type: 'ai-title', aiTitle: ` ${'x'.repeat(250)} ` }]); + expect(title).toBe('x'.repeat(200)); + }); + + it('ignores aiTitle on lines of other types and returns "" when there is none', () => { + expect(latestTitle([{ type: 'user', aiTitle: 'not a title' }])).toBe(''); + expect(latestTitle([])).toBe(''); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts index 668821165..18f477684 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/session-summary.test.ts @@ -12,6 +12,7 @@ import { type SessionSummaryAccumulator, } from '../session-summary.js'; import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; +import type { Compaction } from '../session-signals.js'; function emptyAccumulator(): SessionSummaryAccumulator { return { @@ -20,7 +21,14 @@ function emptyAccumulator(): SessionSummaryAccumulator { toolResults: 0, filesEdited: new Set(), filesWritten: new Set(), - compactionCount: 0, + linesAdded: null, + linesRemoved: null, + turns: 0, + compactions: [], + clientVersions: [], + commandsInOrder: [], + title: '', + lastGitBranch: '', }; } @@ -36,6 +44,11 @@ describe('branchDominant', () => { it('returns "" for an empty map', () => { expect(branchDominant({})).toBe(''); }); + + it('breaks a tie in favour of the later branch', () => { + expect(branchDominant({ main: 2, feature: 2 })).toBe('feature'); + expect(branchDominant({ main: 2, feature: 2, hotfix: 1 })).toBe('feature'); + }); }); describe('primaryModel', () => { @@ -46,6 +59,10 @@ describe('primaryModel', () => { it('returns "" for an empty map', () => { expect(primaryModel({})).toBe(''); }); + + it('breaks a tie in favour of the first model', () => { + expect(primaryModel({ 'claude-sonnet-4-5': 3, 'claude-opus-4-1': 3 })).toBe('claude-sonnet-4-5'); + }); }); describe('updateBranchCounts', () => { @@ -177,7 +194,7 @@ describe('buildSessionSummaryEvent', () => { expect(event.tool_results).toBe(8); }); - it('derives skills, agents and primary_command from a constructed NamedInvocationCounts', () => { + it('derives skills and agents from a constructed NamedInvocationCounts', () => { const named: NamedInvocationCounts = { skillInvocations: { 'codemie:msgraph': 2, brainstorming: 1 }, agentInvocations: { Explore: 2 }, @@ -196,8 +213,40 @@ describe('buildSessionSummaryEvent', () => { expect(event.skills).toEqual({ 'codemie:msgraph': 2, brainstorming: 1 }); expect(event.agents).toEqual({ Explore: 2 }); - expect(event.primary_command).toBe('deploy'); - expect(event.commands).toEqual(Object.keys(named.commandInvocations)); + }); + + it('reports commands in call order with repeats, and the first one as primary_command', () => { + const acc = emptyAccumulator(); + acc.commandsInOrder = ['init', 'deploy', 'deploy', 'init', 'deploy']; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.commands).toEqual(['init', 'deploy', 'deploy', 'init', 'deploy']); + // 'deploy' is the most frequent, but the contract's primary_command is the first. + expect(event.primary_command).toBe('init'); + }); + + it('reports an empty commands array and an empty primary_command when no command was run', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.commands).toEqual([]); + expect(event.primary_command).toBe(''); }); it('reports models as an array with the primary model first', () => { @@ -218,11 +267,10 @@ describe('buildSessionSummaryEvent', () => { expect(event.primary_model).toBe('claude-opus-4-1'); }); - it('reports lines_* as null and files_changed/written/edited as counts', () => { + it('reports files_changed/written/edited as counts and the branch fields from branchCounts', () => { const acc = emptyAccumulator(); acc.filesEdited = new Set(['b.ts']); acc.filesWritten = new Set(['a.ts']); - acc.compactionCount = 2; const branchCounts = { main: 3, feature: 7 }; @@ -236,16 +284,118 @@ describe('buildSessionSummaryEvent', () => { '2026-10-01T01:00:00.000Z' ); - expect(event.lines_added).toBeNull(); - expect(event.lines_removed).toBeNull(); expect(event.files_changed).toBe(2); expect(event.files_written).toBe(1); expect(event.files_edited).toBe(1); - expect(event.compaction_count).toBe(2); expect(event.branch_counts).toBe(branchCounts); expect(event.branch_dominant).toBe('feature'); }); + it('reports lines_* as null while unmeasured (never 0), and as the accumulated counts otherwise', () => { + const unmeasured = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + expect(unmeasured.lines_added).toBeNull(); + expect(unmeasured.lines_removed).toBeNull(); + + const acc = emptyAccumulator(); + acc.linesAdded = 120; + acc.linesRemoved = 0; + const measured = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + expect(measured.lines_added).toBe(120); + expect(measured.lines_removed).toBe(0); + }); + + it('derives compaction_count, compaction_pre_tokens and compactions from acc.compactions', () => { + const compactions: Compaction[] = [ + { + start: '2026-10-01T00:09:00.000Z', + end: '2026-10-01T00:10:00.000Z', + duration_ms: 60_000, + trigger: 'auto', + pre_tokens: 1000, + post_tokens: 300, + dropped_tokens: 700, + }, + { + start: null, + end: '2026-10-01T00:20:00.000Z', + duration_ms: null, + trigger: 'manual', + pre_tokens: 500, + post_tokens: null, + dropped_tokens: null, + }, + ]; + const acc = emptyAccumulator(); + acc.compactions = compactions; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.compaction_count).toBe(2); + expect(event.compaction_pre_tokens).toBe(1500); + expect(event.compactions).toEqual(compactions); + }); + + it('reports no compactions as a zero count, a zero sum and an empty array', () => { + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + emptyAccumulator(), + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.compaction_count).toBe(0); + expect(event.compaction_pre_tokens).toBe(0); + expect(event.compactions).toEqual([]); + }); + + it('passes turns, client_versions and git_branch from the accumulator', () => { + const acc = emptyAccumulator(); + acc.turns = 7; + acc.clientVersions = ['2.1.283', '2.1.284']; + acc.lastGitBranch = 'feature/ABC-123'; + + const event = buildSessionSummaryEvent( + 'session-1', + 'final', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '2026-10-01T01:00:00.000Z' + ); + + expect(event.turns).toBe(7); + expect(event.client_versions).toEqual(['2.1.283', '2.1.284']); + expect(event.git_branch).toBe('feature/ABC-123'); + }); + it('counts a file touched by both Edit and Write once in files_changed', () => { const acc = emptyAccumulator(); acc.filesEdited = new Set(['a.ts']); @@ -266,8 +416,8 @@ describe('buildSessionSummaryEvent', () => { expect(event.files_edited).toBe(1); }); - it('emits title as a literal empty string (no identified source)', () => { - const event = buildSessionSummaryEvent( + it('emits title as an empty string when the accumulator has none, and the accumulator title otherwise', () => { + const none = buildSessionSummaryEvent( 'session-1', 'incremental', emptyAccumulator(), @@ -276,8 +426,20 @@ describe('buildSessionSummaryEvent', () => { '2026-10-01T00:00:00.000Z', '' ); + expect(none.title).toBe(''); - expect(event.title).toBe(''); + const acc = emptyAccumulator(); + acc.title = 'Review preflight plan'; + const titled = buildSessionSummaryEvent( + 'session-1', + 'incremental', + acc, + emptyNamed(), + {}, + '2026-10-01T00:00:00.000Z', + '' + ); + expect(titled.title).toBe('Review preflight plan'); }); it('does not include an apiCalls/api_calls field (left to the caller/orchestrator to merge)', () => { diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index 860faa08c..aa24bd3ab 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -27,7 +27,18 @@ import { type SessionSummaryAccumulator, type NamedInvocationCounts, } from './session-summary.js'; -import { extractNamedInvocations } from '@/agents/plugins/claude/session/claude-named-invocations.js'; +import { + extractNamedInvocations, + extractOrderedNamedInvocations, +} from '@/agents/plugins/claude/session/claude-named-invocations.js'; +import { + collectClientVersions, + collectCompactions, + collectEditLineStats, + countTurns, + latestTitle, + type SignalLine, +} from './session-signals.js'; import { type SubagentFile, buildSubagentUsageEvent } from './subagent-usage.js'; // Re-exported so callers (e.g. claude-code-otlp.plugin.ts) can import both `SubagentFile` and @@ -46,10 +57,8 @@ interface ContentBlock { input?: { file_path?: unknown; path?: unknown }; } -interface TranscriptLine { - timestamp?: string; +interface TranscriptLine extends SignalLine { gitBranch?: string; - message?: { content?: unknown }; } /** @@ -95,7 +104,14 @@ function emptyAccumulator(): SessionSummaryAccumulator { toolResults: 0, filesEdited: new Set(), filesWritten: new Set(), - compactionCount: 0, + linesAdded: null, + linesRemoved: null, + turns: 0, + compactions: [], + clientVersions: [], + commandsInOrder: [], + title: '', + lastGitBranch: '', }; } @@ -130,10 +146,11 @@ function countToolResults(parsedLines: TranscriptLine[]): number { * - `toolCalls[*].errors` is derived from a sibling `tool_result` block's `is_error`/`isError` * flag (the same pattern `claude.session.ts`/`claude.metrics-processor.ts` already use for * tool-use_id → error lookups) when one is found; otherwise a tool call's `.errors` stays 0. - * - Lines added/removed are not computed — an `Edit`/`Write` tool_use's `input` carries the - * *proposed* edit, not a diff stat — so the event builder sends them as `null`, never `0`. - * - `compactionCount` defaults to 0 — no verified in-transcript signal was found (`PreCompact` is - * a hook event, not a transcript line). + * - Lines added/removed come from the applied diff on the `tool_result` line + * (`toolUseResult.structuredPatch`, or a `create` result's `content`), not from the `tool_use` + * input, which only carries the *proposed* edit. They stay `null` until one such result is seen. + * - Compactions come from `compact_boundary` system lines, not from the `PreCompact` hook, which + * fires before the boundary line exists. */ async function buildFullAccumulator(transcriptPath: string): Promise<{ acc: SessionSummaryAccumulator; @@ -147,7 +164,7 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ try { raw = await readFile(transcriptPath, 'utf-8'); } catch { - return { acc, named: extractNamedInvocations([]), startedAt: '', endedAt: '' }; + return { acc, named: extractOrderedNamedInvocations([]), startedAt: '', endedAt: '' }; } const rawLines = raw.split('\n').filter((line) => line.trim().length > 0); @@ -201,7 +218,20 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ } } - const named = extractNamedInvocations(parsedLines); + const named = extractOrderedNamedInvocations(parsedLines); + acc.commandsInOrder = named.commandsInOrder; + + const { linesAdded, linesRemoved } = collectEditLineStats(parsedLines); + acc.linesAdded = linesAdded; + acc.linesRemoved = linesRemoved; + acc.turns = countTurns(parsedLines); + acc.compactions = collectCompactions(parsedLines); + acc.clientVersions = collectClientVersions(parsedLines); + acc.title = latestTitle(parsedLines); + acc.lastGitBranch = + [...parsedLines].reverse().find((line) => typeof line.gitBranch === 'string' && line.gitBranch) + ?.gitBranch ?? ''; + // Real transcripts interleave non-message lines (file-history-snapshot, cost-state, ...) // without a `timestamp`, including at index 0/length-1 — so the first/last *timestamped* // line is used, not literally the first/last line. @@ -217,8 +247,8 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ * - Loads persisted state, reads only the lines appended since `state.mainOffset`. * - Derives/merges `agent.usage.request` records for those new lines into `state.openRequests`, * keyed by `usageRequestKey()`, and - * updates `state.branchCounts` from every new line's `gitBranch` (regardless of whether that - * line carried usage). + * updates `state.branchCounts` from every new user line's `gitBranch` (regardless of whether + * that line carried usage). * - Returns one `agent.usage.request` JSON string per request key touched by this pass. * - On `Stop`/`SessionEnd` only, also returns exactly one `agent.session.summary` event * (`phase: 'incremental'` on `Stop`, `'final'` on `SessionEnd`) built from a fresh full-file @@ -250,13 +280,17 @@ export async function collectMainTranscriptEvents( const touchedKeys = new Set(); for (const line of lines) { let rawGitBranch = ''; + let isUserLine = false; try { - rawGitBranch = (JSON.parse(line) as { gitBranch?: string })?.gitBranch ?? ''; + const head = JSON.parse(line) as { gitBranch?: string; type?: string }; + rawGitBranch = head?.gitBranch ?? ''; + isUserLine = head?.type === 'user'; } catch { // Malformed line: still attempt usage parsing below (which has its own try/catch), but // there is no branch to record from it. } - if (rawGitBranch) { + // The contract's branch counts are user lines per branch. + if (rawGitBranch && isUserLine) { updateBranchCounts(state.branchCounts, rawGitBranch); } @@ -270,10 +304,6 @@ export async function collectMainTranscriptEvents( } state.mainOffset = nextOffset; - if (trigger === 'PreCompact') { - state.compactionCount += 1; - } - const events: Record[] = []; for (const key of touchedKeys) { events.push(buildUsageRequestEvent(sessionId, state.openRequests[key])); @@ -281,7 +311,6 @@ export async function collectMainTranscriptEvents( if (trigger === 'Stop' || trigger === 'SessionEnd') { const { acc, named, startedAt, endedAt } = await buildFullAccumulator(transcriptPath); - acc.compactionCount = state.compactionCount; const phase = trigger === 'SessionEnd' ? 'final' : 'incremental'; const summaryEvent = buildSessionSummaryEvent( sessionId, diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts index 8679fe269..ebf14fd81 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -44,7 +44,6 @@ export interface TranscriptParseState { openRequests: Record; // key: usageRequestKey() activeSkill: string; branchCounts: Record; - compactionCount: number; } /** @@ -57,7 +56,6 @@ export function createParseState(): TranscriptParseState { openRequests: {}, activeSkill: '', branchCounts: {}, - compactionCount: 0, }; } diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-signals.ts b/src/agents/plugins/claude-code-otlp/transcript/session-signals.ts new file mode 100644 index 000000000..b33be9ec7 --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/transcript/session-signals.ts @@ -0,0 +1,186 @@ +/** + * Transcript signals for `agent.session.summary`. + * + * Pure functions over already-parsed main-transcript lines: edit-diff line counts, user turns, + * compaction boundaries, client versions and the session title. No I/O — the orchestrator reads + * and parses the file once and hands the lines in. + */ + +/** The subset of a transcript line these extractors read; every field is optional. */ +export interface SignalLine { + type?: string; + subtype?: string; + timestamp?: string; + version?: string; + aiTitle?: string; + isMeta?: boolean; + isSidechain?: boolean; + isCompactSummary?: boolean; + message?: { content?: unknown }; + toolUseResult?: unknown; + compactMetadata?: { + trigger?: unknown; + preTokens?: unknown; + postTokens?: unknown; + durationMs?: unknown; + }; +} + +/** One compaction, in the contract's `compactions[]` element shape (`null` = not reported). */ +export interface Compaction { + start: string | null; + end: string; + duration_ms: number | null; + trigger: string; + pre_tokens: number | null; + post_tokens: number | null; + dropped_tokens: number | null; +} + +/** Lines added/removed by edits whose result carried a diff; `null` when no edit was measured. */ +export interface EditLineStats { + linesAdded: number | null; + linesRemoved: number | null; +} + +/** The contract's cap for `title`. */ +const TITLE_MAX_LENGTH = 200; + +function asNumber(value: unknown): number | null { + return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : null; +} + +/** `+`/`-` line counts of a `structuredPatch` (an array of `{ lines: string[] }` hunks). */ +function countPatchLines(patch: unknown[]): { added: number; removed: number } { + let added = 0; + let removed = 0; + for (const hunk of patch) { + const lines = (hunk as { lines?: unknown } | null)?.lines; + if (!Array.isArray(lines)) continue; + for (const line of lines) { + if (typeof line !== 'string') continue; + if (line.startsWith('+')) added += 1; + else if (line.startsWith('-')) removed += 1; + } + } + return { added, removed }; +} + +/** Lines in a file body; a trailing newline does not start another line. */ +function countContentLines(content: string): number { + if (content.length === 0) return 0; + const lines = content.split('\n').length; + return content.endsWith('\n') ? lines - 1 : lines; +} + +/** + * Sum the diff stats of every tool result that reports one. + * + * Claude Code attaches the applied change to the user line carrying the `tool_result`: + * `toolUseResult.structuredPatch` for edits, and `toolUseResult.type === 'create'` plus the file + * `content` for new files. A failed edit has neither, so only applied edits are counted. + */ +export function collectEditLineStats(lines: readonly SignalLine[]): EditLineStats { + let linesAdded: number | null = null; + let linesRemoved: number | null = null; + + for (const line of lines) { + const result = line.toolUseResult; + if (!result || typeof result !== 'object') continue; + const { structuredPatch, type, content } = result as { + structuredPatch?: unknown; + type?: unknown; + content?: unknown; + }; + + let added: number | null = null; + let removed = 0; + if (Array.isArray(structuredPatch)) { + const counts = countPatchLines(structuredPatch); + added = counts.added; + removed = counts.removed; + } else if (type === 'create' && typeof content === 'string') { + added = countContentLines(content); + } + if (added === null) continue; + + linesAdded = (linesAdded ?? 0) + added; + linesRemoved = (linesRemoved ?? 0) + removed; + } + + return { linesAdded, linesRemoved }; +} + +/** + * Count real user prompts: user lines with text (a plain string or a `text` block) that are not + * meta, sidechain or compaction-summary lines. Lines holding only `tool_result` blocks are not + * prompts. + */ +export function countTurns(lines: readonly SignalLine[]): number { + let turns = 0; + for (const line of lines) { + if (line.type !== 'user' || line.isMeta || line.isSidechain || line.isCompactSummary) continue; + const content = line.message?.content; + if (typeof content === 'string') { + turns += 1; + } else if ( + Array.isArray(content) && + content.some((block) => (block as { type?: unknown } | null)?.type === 'text') + ) { + turns += 1; + } + } + return turns; +} + +/** + * One {@link Compaction} per `compact_boundary` system line. `end` is the boundary line's + * timestamp and `start` is `end - durationMs`; `dropped_tokens` is `pre - post`. + */ +export function collectCompactions(lines: readonly SignalLine[]): Compaction[] { + const compactions: Compaction[] = []; + for (const line of lines) { + if (line.type !== 'system' || line.subtype !== 'compact_boundary') continue; + + const meta = line.compactMetadata ?? {}; + const end = typeof line.timestamp === 'string' ? line.timestamp : ''; + const durationMs = asNumber(meta.durationMs); + const endMs = Date.parse(end); + const pre = asNumber(meta.preTokens); + const post = asNumber(meta.postTokens); + + compactions.push({ + start: + durationMs !== null && Number.isFinite(endMs) + ? new Date(endMs - durationMs).toISOString() + : null, + end, + duration_ms: durationMs, + trigger: typeof meta.trigger === 'string' ? meta.trigger : '', + pre_tokens: pre, + post_tokens: post, + dropped_tokens: pre !== null && post !== null && pre >= post ? pre - post : null, + }); + } + return compactions; +} + +/** Distinct client versions, in order of first appearance. */ +export function collectClientVersions(lines: readonly SignalLine[]): string[] { + const versions = new Set(); + for (const line of lines) { + if (typeof line.version === 'string' && line.version) versions.add(line.version); + } + return [...versions]; +} + +/** The latest `ai-title`, cut to the contract's 200 characters; `''` when there is none. */ +export function latestTitle(lines: readonly SignalLine[]): string { + for (let i = lines.length - 1; i >= 0; i -= 1) { + const line = lines[i]; + if (line.type === 'ai-title' && typeof line.aiTitle === 'string' && line.aiTitle.trim()) { + return line.aiTitle.trim().slice(0, TITLE_MAX_LENGTH); + } + } + return ''; +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts index d1dfd8207..b691b8387 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/session-summary.ts @@ -13,10 +13,11 @@ * `api_calls` is omitted here: no input carries a request count. The orchestrator, which owns * the full set of `agent.usage.request` records, merges it in afterward. * - * `title` has no identified source and is always `''`, never fabricated. + * `title` is the transcript's latest `ai-title` (`''` when it has none), never fabricated. */ import type { NamedInvocationCounts } from '@/agents/plugins/claude/session/claude-named-invocations.js'; +import type { Compaction } from './session-signals.js'; export type { NamedInvocationCounts }; @@ -27,7 +28,17 @@ export interface SessionSummaryAccumulator { toolResults: number; filesEdited: Set; filesWritten: Set; - compactionCount: number; + /** `null` until an edit result reports a diff — unknown is never sent as `0`. */ + linesAdded: number | null; + linesRemoved: number | null; + turns: number; + compactions: Compaction[]; + clientVersions: string[]; + /** Slash commands in call order, repeats kept, no leading slash. */ + commandsInOrder: string[]; + title: string; + /** Branch of the last transcript line that carries one. */ + lastGitBranch: string; } /** @@ -46,14 +57,15 @@ export function updateBranchCounts(counts: Record, branch: strin /** * Return the key with the highest value in `counts`, or `''` when `counts` is empty. - * On a tie, the first-encountered key (in `Object.entries()` iteration order) wins. + * On a tie, the first-encountered key (in `Object.entries()` iteration order) wins, or the + * last-encountered one when `tieToLater` is set. */ -function maxKey(counts: Record): string { +function maxKey(counts: Record, tieToLater = false): string { let best = ''; let bestValue = -Infinity; for (const [key, value] of Object.entries(counts)) { - if (value > bestValue) { + if (tieToLater ? value >= bestValue : value > bestValue) { best = key; bestValue = value; } @@ -67,9 +79,9 @@ export function primaryModel(models: Record): string { return maxKey(models); } -/** The branch with the highest count in `counts`, or `''` when empty. */ +/** The branch with the highest count in `counts` (a tie goes to the later one), or `''` when empty. */ export function branchDominant(counts: Record): string { - return maxKey(counts); + return maxKey(counts, true); } /** Distinct normalised models, primary first, per the contract's `models` field. */ @@ -102,6 +114,7 @@ export function buildSessionSummaryEvent( (totals, t) => ({ calls: totals.calls + t.calls, errors: totals.errors + t.errors }), { calls: 0, errors: 0 } ); + const compactionPreTokens = acc.compactions.reduce((sum, c) => sum + (c.pre_tokens ?? 0), 0); return { type: 'agent.session.summary', @@ -122,16 +135,21 @@ export function buildSessionSummaryEvent( tools: acc.toolCalls, skills: named.skillInvocations, agents: named.agentInvocations, - commands: Object.keys(named.commandInvocations), - primary_command: maxKey(named.commandInvocations), - lines_added: null, - lines_removed: null, + commands: acc.commandsInOrder, + primary_command: acc.commandsInOrder[0] ?? '', + turns: acc.turns, + lines_added: acc.linesAdded, + lines_removed: acc.linesRemoved, files_changed: filesChanged.size, files_written: acc.filesWritten.size, files_edited: acc.filesEdited.size, - compaction_count: acc.compactionCount, + compaction_count: acc.compactions.length, + compaction_pre_tokens: compactionPreTokens, + compactions: acc.compactions, branch_counts: branchCounts, branch_dominant: branchDominant(branchCounts), - title: '', + git_branch: acc.lastGitBranch, + client_versions: acc.clientVersions, + title: acc.title, }; } diff --git a/src/agents/plugins/claude/session/__tests__/claude-named-invocations.test.ts b/src/agents/plugins/claude/session/__tests__/claude-named-invocations.test.ts index 692f9c14e..4bcebfaae 100644 --- a/src/agents/plugins/claude/session/__tests__/claude-named-invocations.test.ts +++ b/src/agents/plugins/claude/session/__tests__/claude-named-invocations.test.ts @@ -4,7 +4,7 @@ * session adapter's parse path (native/untracked sessions). */ import { describe, it, expect } from 'vitest'; -import { extractNamedInvocations } from '../claude-named-invocations.js'; +import { extractNamedInvocations, extractOrderedNamedInvocations } from '../claude-named-invocations.js'; function toolUse(name: string, input: Record) { return { message: { role: 'assistant', content: [{ type: 'tool_use', name, input }] } }; @@ -113,3 +113,39 @@ describe('extractNamedInvocations', () => { expect(out.commandInvocations).toEqual({}); }); }); + +describe('extractOrderedNamedInvocations', () => { + it('keeps slash commands in call order with repeats, and strips the slash', () => { + const out = extractOrderedNamedInvocations([ + userStringContent(commandWrapper('plan')), + userText('a prompt in between'), + userText(commandWrapper('commit')), + userStringContent(commandWrapper('plan')), + ]); + expect(out.commandsInOrder).toEqual(['plan', 'commit', 'plan']); + expect(out.commandInvocations).toEqual({ plan: 2, commit: 1 }); + }); + + it('returns an empty list when there are no genuine commands', () => { + const out = extractOrderedNamedInvocations([ + userText('docs mention /plan only'), + userText('just a plain prompt'), + ]); + expect(out.commandsInOrder).toEqual([]); + }); + + it('returns the same counts as extractNamedInvocations', () => { + const messages = [ + toolUse('Skill', { skill: 'codemie:msgraph' }), + toolUse('Agent', { subagent_type: 'Explore' }), + userStringContent(commandWrapper('init')), + ]; + const { commandsInOrder: _ignored, ...ordered } = extractOrderedNamedInvocations(messages); + expect(ordered).toEqual(extractNamedInvocations(messages)); + }); + + it('keeps extractNamedInvocations free of the ordered list', () => { + const out = extractNamedInvocations([userStringContent(commandWrapper('plan'))]); + expect(Object.keys(out).sort()).toEqual(['agentInvocations', 'commandInvocations', 'skillInvocations']); + }); +}); diff --git a/src/agents/plugins/claude/session/claude-named-invocations.ts b/src/agents/plugins/claude/session/claude-named-invocations.ts index f0832b820..279c13f7a 100644 --- a/src/agents/plugins/claude/session/claude-named-invocations.ts +++ b/src/agents/plugins/claude/session/claude-named-invocations.ts @@ -22,6 +22,11 @@ export interface NamedInvocationCounts { commandInvocations: Record; } +/** {@link NamedInvocationCounts} plus the slash commands in call order, repeats kept. */ +export interface OrderedNamedInvocations extends NamedInvocationCounts { + commandsInOrder: string[]; +} + interface RawBlock { type?: string; name?: string; @@ -44,9 +49,20 @@ function bump(map: Record, key: string): void { * Safe against missing or malformed fields — anything that isn't a recognized shape is skipped. */ export function extractNamedInvocations(messages: readonly unknown[]): NamedInvocationCounts { + const { skillInvocations, agentInvocations, commandInvocations } = + extractOrderedNamedInvocations(messages); + return { skillInvocations, agentInvocations, commandInvocations }; +} + +/** + * Same extraction as {@link extractNamedInvocations}, additionally keeping the slash commands in + * the order they were invoked (repeats kept, no leading slash). + */ +export function extractOrderedNamedInvocations(messages: readonly unknown[]): OrderedNamedInvocations { const skillInvocations: Record = {}; const agentInvocations: Record = {}; const commandInvocations: Record = {}; + const commandsInOrder: string[] = []; // Count slash commands from a text payload, but only when it carries the CLI's // `` sibling — that distinguishes a real invocation from prose that @@ -59,7 +75,10 @@ export function extractNamedInvocations(messages: readonly unknown[]): NamedInvo let match: RegExpExecArray | null; while ((match = COMMAND_TAG.exec(text)) !== null) { const cmd = match[1].replace(/^\//, '').trim(); - if (cmd) bump(commandInvocations, cmd); + if (cmd) { + bump(commandInvocations, cmd); + commandsInOrder.push(cmd); + } } }; @@ -95,5 +114,5 @@ export function extractNamedInvocations(messages: readonly unknown[]): NamedInvo } } - return { skillInvocations, agentInvocations, commandInvocations }; + return { skillInvocations, agentInvocations, commandInvocations, commandsInOrder }; } From 5ca7bfab1a86337df9c4abcf5f75977ba815408f Mon Sep 17 00:00:00 2001 From: Uladzislau Mamantau Date: Fri, 9 Oct 2026 17:22:26 +0300 Subject: [PATCH 35/35] fix(analytics): close the remaining PR#609 minor-field gaps in claude-code-otlp Fills agent.subagent.usage's worktree from the triggering hook's cwd, threads agent_type through agent.usage.request end to end, tracks a Skill tool_use as a scope_kind='skill' context (cleared on the next genuine user turn), counts MultiEdit/NotebookEdit toward files_edited/files_changed, and rejects an empty or path-traversing session_id before it ever reaches a filesystem path. Generated with AI Co-Authored-By: codemie-ai --- .../__tests__/claude-code-otlp.plugin.test.ts | 6 +- .../__tests__/claude-code-otlp.types.test.ts | 47 +++++ .../claude-code-otlp.plugin.ts | 6 +- .../claude-code-otlp.types.ts | 11 ++ .../transcript/__tests__/orchestrator.test.ts | 165 ++++++++++++++++++ .../transcript/__tests__/parse-state.test.ts | 1 + .../__tests__/subagent-usage.test.ts | 16 +- .../__tests__/usage-request.test.ts | 52 ++++-- .../transcript/orchestrator.ts | 77 ++++++-- .../transcript/parse-state.ts | 2 + .../transcript/subagent-usage.ts | 8 +- .../transcript/usage-request.ts | 12 +- 12 files changed, 362 insertions(+), 41 deletions(-) create mode 100644 src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.types.test.ts diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts index a298b6be6..4add83589 100644 --- a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.plugin.test.ts @@ -244,6 +244,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { filePath: '/tmp/agent-sub-1.jsonl', toolUseId: 'tu-1', agentType: 'explore', + cwd: '/repo', }); }); @@ -289,6 +290,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { agentType: 'explore', spawnDepth: 2, description: 'task', + cwd: '/repo', }); }); @@ -334,7 +336,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledTimes(2); expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( 'sid-1', - { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl' } + { agentId: 'sub-1', filePath: '/tmp/agent-sub-1.jsonl', cwd: '/repo' } ); }); @@ -355,7 +357,7 @@ describe('ClaudeCodeOtlpPlugin.processOtlpEvent dispatch', () => { expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledTimes(1); expect(collectSubagentTranscriptEventsMock).toHaveBeenCalledWith( 'sid-1', - { agentId: 'pending', filePath: '/tmp/agent-pending.jsonl' } + { agentId: 'pending', filePath: '/tmp/agent-pending.jsonl', cwd: '/repo' } ); }); diff --git a/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.types.test.ts b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.types.test.ts new file mode 100644 index 000000000..e1bdd0eff --- /dev/null +++ b/src/agents/plugins/claude-code-otlp/__tests__/claude-code-otlp.types.test.ts @@ -0,0 +1,47 @@ +import { describe, it, expect } from 'vitest'; +import { isClaudeCodeHookInput } from '../claude-code-otlp.types.js'; + +function hookInput(overrides: Record = {}): Record { + return { + session_id: 'sid-1', + cwd: '/repo', + hook_event_name: 'Stop', + ...overrides, + }; +} + +describe('isClaudeCodeHookInput', () => { + it('accepts a payload with the three required fields', () => { + expect(isClaudeCodeHookInput(hookInput())).toBe(true); + }); + + it('rejects non-object values', () => { + expect(isClaudeCodeHookInput(null)).toBe(false); + expect(isClaudeCodeHookInput(undefined)).toBe(false); + expect(isClaudeCodeHookInput('not an object')).toBe(false); + expect(isClaudeCodeHookInput(42)).toBe(false); + }); + + it('rejects a payload missing cwd or hook_event_name', () => { + expect(isClaudeCodeHookInput(hookInput({ cwd: undefined }))).toBe(false); + expect(isClaudeCodeHookInput(hookInput({ hook_event_name: undefined }))).toBe(false); + }); + + it('rejects an empty session_id — it is used verbatim as a filename, so an empty value would collide every session onto one shared state file', () => { + expect(isClaudeCodeHookInput(hookInput({ session_id: '' }))).toBe(false); + }); + + it('rejects a session_id carrying a path separator or a ".." segment (path traversal)', () => { + for (const unsafe of ['../../x', '..\\x', 'a/b', 'a\\b', '..', 'foo/../bar']) { + expect(isClaudeCodeHookInput(hookInput({ session_id: unsafe }))).toBe(false); + } + }); + + it('accepts a normal UUID-shaped session_id', () => { + expect(isClaudeCodeHookInput(hookInput({ session_id: 'dd7d411a-cb45-4cd4-a7df-8d0695546e55' }))).toBe(true); + }); + + it('rejects a non-string session_id', () => { + expect(isClaudeCodeHookInput(hookInput({ session_id: 123 }))).toBe(false); + }); +}); diff --git a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts index bf92b3033..65d71ce8a 100644 --- a/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts +++ b/src/agents/plugins/claude-code-otlp/claude-code-otlp.plugin.ts @@ -120,7 +120,10 @@ export class ClaudeCodeOtlpPlugin extends OtlpAgentAdapter; return ( typeof candidate.session_id === 'string' && + isSafeSessionId(candidate.session_id) && typeof candidate.cwd === 'string' && typeof candidate.hook_event_name === 'string' ); } + +/** + * `session_id` is used verbatim as a filename/path segment (parse-state, its lock, subagent + * discovery), so an empty value (would collide with every other empty-id session on one shared + * file) or one carrying a path separator or `..` segment (path traversal) is rejected here, + * before anything downstream ever reads it. + */ +function isSafeSessionId(sessionId: string): boolean { + return sessionId.length > 0 && !/[\\/]|\.\./.test(sessionId); +} diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts index 7d1ea0061..e51bb34c2 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/orchestrator.test.ts @@ -202,6 +202,43 @@ function commandText(name: string): string { return `/${name}\n${name}\n`; } +/** An assistant usage line, optionally invoking the `Skill` tool alongside its own usage. */ +function skillAwareLine(opts: { + uuid: string; + messageId: string; + outputTokens: number; + skill?: string; +}): string { + const content: unknown[] = [{ type: 'text', text: 'ok' }]; + if (opts.skill) { + content.push({ type: 'tool_use', id: `tool-${opts.uuid}`, name: 'Skill', input: { skill: opts.skill } }); + } + return JSON.stringify({ + gitBranch: 'main', + cwd: '/repo', + timestamp: '2026-10-01T00:00:00.000Z', + uuid: opts.uuid, + message: { + id: opts.messageId, + role: 'assistant', + model: 'claude-sonnet-4-5-20250929', + stop_reason: '', + content, + usage: { + input_tokens: 100, + output_tokens: opts.outputTokens, + cache_read_input_tokens: 5, + cache_creation_input_tokens: 0, + service_tier: 'standard', + speed: 'standard', + inference_geo: '', + cache_creation: { ephemeral_1h_input_tokens: 0, ephemeral_5m_input_tokens: 0 }, + server_tool_use: { web_search_requests: 0, web_fetch_requests: 0 }, + }, + }, + }); +} + describe('collectMainTranscriptEvents — compactions', () => { it('does not count PreCompact hook fires; compaction_count comes from compact_boundary lines', async () => { const { collectMainTranscriptEvents } = await import('../orchestrator.js'); @@ -381,6 +418,77 @@ describe('collectMainTranscriptEvents — api_calls', () => { }); }); +describe('collectMainTranscriptEvents — skill scoping', () => { + it('scopes the request that invokes a Skill tool as main, scopes subsequent requests as skill, and clears on the next genuine user turn', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-skill-scope'; + const transcriptPath = writeTranscript('transcript-skill-scope.jsonl', [ + // The invoking turn itself still belongs to whatever was active before it (main, here). + skillAwareLine({ uuid: 'u1', messageId: 'msg-1', outputTokens: 10, skill: 'brainstorming' }), + // Now inside the skill's context. + skillAwareLine({ uuid: 'u2', messageId: 'msg-2', outputTokens: 20 }), + skillAwareLine({ uuid: 'u3', messageId: 'msg-3', outputTokens: 30 }), + // A fresh user prompt leaves the skill context. + userLine('new unrelated prompt'), + skillAwareLine({ uuid: 'u4', messageId: 'msg-4', outputTokens: 40 }), + ]); + + const events = await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); + const usage = events.filter((e) => e.type === 'agent.usage.request'); + const byMessageId = (id: string): Record => { + const found = usage.find((e) => e.message_id === id); + if (!found) throw new Error(`missing usage event for ${id}`); + return found; + }; + + expect(byMessageId('msg-1')).toMatchObject({ scope_kind: 'main', scope_name: '' }); + expect(byMessageId('msg-2')).toMatchObject({ scope_kind: 'skill', scope_name: 'brainstorming' }); + expect(byMessageId('msg-3')).toMatchObject({ scope_kind: 'skill', scope_name: 'brainstorming' }); + expect(byMessageId('msg-4')).toMatchObject({ scope_kind: 'main', scope_name: '' }); + }); + + it('switches scope_name when a second, different Skill is invoked while already inside one', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-skill-switch'; + const transcriptPath = writeTranscript('transcript-skill-switch.jsonl', [ + skillAwareLine({ uuid: 'u1', messageId: 'msg-1', outputTokens: 10, skill: 'brainstorming' }), + skillAwareLine({ uuid: 'u2', messageId: 'msg-2', outputTokens: 20, skill: 'code-review' }), + skillAwareLine({ uuid: 'u3', messageId: 'msg-3', outputTokens: 30 }), + ]); + + const events = await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); + const usage = events.filter((e) => e.type === 'agent.usage.request'); + + // msg-1 invokes 'brainstorming' but is itself scoped by whatever was active before it (main). + expect(usage.find((e) => e.message_id === 'msg-1')).toMatchObject({ scope_kind: 'main', scope_name: '' }); + // msg-2 invokes 'code-review' but is itself still scoped under 'brainstorming' (active since msg-1). + expect(usage.find((e) => e.message_id === 'msg-2')).toMatchObject({ scope_kind: 'skill', scope_name: 'brainstorming' }); + expect(usage.find((e) => e.message_id === 'msg-3')).toMatchObject({ scope_kind: 'skill', scope_name: 'code-review' }); + }); + + it('persists activeSkill across hook invocations (fresh CLI process per hook)', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-skill-persisted'; + const transcriptPath = writeTranscript('transcript-skill-persisted.jsonl', [ + skillAwareLine({ uuid: 'u1', messageId: 'msg-1', outputTokens: 10, skill: 'brainstorming' }), + ]); + await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); + + writeFileSync( + transcriptPath, + skillAwareLine({ uuid: 'u2', messageId: 'msg-2', outputTokens: 20 }) + '\n', + { flag: 'a' } + ); + const events = await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); + const usage = events.filter((e) => e.type === 'agent.usage.request'); + + expect(usage.find((e) => e.message_id === 'msg-2')).toMatchObject({ scope_kind: 'skill', scope_name: 'brainstorming' }); + }); +}); + describe('collectMainTranscriptEvents — SessionEnd trigger', () => { it('returns a final-phase summary with an ended_at key present', async () => { const { collectMainTranscriptEvents } = await import('../orchestrator.js'); @@ -466,6 +574,34 @@ describe('collectMainTranscriptEvents — tool-call accumulation', () => { expect(summary?.tool_errors).toBe(1); expect(summary?.tool_results).toBe(2); }); + + it('counts MultiEdit (file_path) and NotebookEdit (notebook_path) toward files_edited/files_changed, not files_written', async () => { + const { collectMainTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-tools-multiedit-notebook'; + const toolLine = JSON.stringify({ + gitBranch: 'main', + timestamp: '2026-10-01T00:00:00.000Z', + message: { + role: 'assistant', + content: [ + { type: 'tool_use', id: 'tool-1', name: 'MultiEdit', input: { file_path: '/repo/a.ts', edits: [] } }, + { type: 'tool_use', id: 'tool-2', name: 'NotebookEdit', input: { notebook_path: '/repo/nb.ipynb' } }, + ], + }, + }); + const transcriptPath = writeTranscript('transcript-multiedit-notebook.jsonl', [toolLine]); + + const events = await collectMainTranscriptEvents(sessionId, transcriptPath, 'Stop'); + const summary = events.find((e) => e.type === 'agent.session.summary'); + + expect(summary?.files_written).toBe(0); + expect(summary?.files_edited).toBe(2); + expect(summary?.files_changed).toBe(2); + const tools = summary?.tools as Record; + expect(tools.MultiEdit).toEqual({ calls: 1, errors: 0 }); + expect(tools.NotebookEdit).toEqual({ calls: 1, errors: 0 }); + }); }); describe('collectMainTranscriptEvents — request identity', () => { @@ -578,6 +714,35 @@ describe('collectSubagentTranscriptEvents — request identity', () => { ['req_a', 'msg_a'], ]); }); + + it('carries the SubagentFile.agentType through onto every agent.usage.request event', async () => { + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-sub-agent-type'; + const file = writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'u1', messageId: 'msg_a', outputTokens: 10 }), + ]); + + const events = await collectSubagentTranscriptEvents(sessionId, { ...file, agentType: 'Explore' }); + const usage = events.filter((e) => e.type === 'agent.usage.request'); + + expect(usage).toHaveLength(1); + expect(usage[0].agent_type).toBe('Explore'); + }); + + it('defaults agent_type to empty when the SubagentFile has none', async () => { + const { collectSubagentTranscriptEvents } = await import('../orchestrator.js'); + + const sessionId = 'session-sub-no-agent-type'; + const file = writeSubagentFixture(sessionId, 'a1', [ + usageLine({ uuid: 'u1', messageId: 'msg_a', outputTokens: 10 }), + ]); + + const events = await collectSubagentTranscriptEvents(sessionId, file); + const usage = events.filter((e) => e.type === 'agent.usage.request'); + + expect(usage[0].agent_type).toBe(''); + }); }); describe('collectSubagentTranscriptEvents — SessionEnd backstop (three subagents, one pre-advanced)', () => { diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts index f8e626bc9..33b3e35e8 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/parse-state.test.ts @@ -90,6 +90,7 @@ describe('loadParseState', () => { scopeKind: 'main', scopeName: '', agentId: '', + agentType: '', stopReason: 'end_turn', isApiError: false, gitBranch: 'main', diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts index ca8bd3230..5d6139e09 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/subagent-usage.test.ts @@ -103,7 +103,7 @@ function usageRequest(overrides: Partial = {}): OpenUsageReque speed: 'standard', inferenceGeo: '', serviceTier: 'standard', inputTokens: 0, cacheCreation5mTokens: 0, cacheCreation1hTokens: 0, cacheReadTokens: 0, outputTokens: 0, webSearchRequests: 0, webFetchRequests: 0, - scopeKind: 'agent', scopeName: '', agentId: 'a1', + scopeKind: 'agent', scopeName: '', agentId: 'a1', agentType: '', stopReason: 'end_turn', isApiError: false, gitBranch: 'main', ...overrides, }; @@ -275,7 +275,7 @@ describe('buildSubagentUsageEvent', () => { expect(event.usage).toEqual([]); }); - it('never fabricates agent_type/description/workflow_run/worktree — always empty string when absent', () => { + it('never fabricates agent_type/description/workflow_run — always empty string when absent; worktree also defaults to empty when the caller has no cwd', () => { const file: SubagentFile = { agentId: 'a3', filePath: '/tmp/agent-a3.jsonl' }; const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, 0, {}, '', '', 0); @@ -286,6 +286,16 @@ describe('buildSubagentUsageEvent', () => { expect(event.worktree).toBe(''); }); + it('sources worktree from the caller-supplied cwd — the only field in agent.subagent.usage that is not sidecar-derived', () => { + const file: SubagentFile = { agentId: 'a1', filePath: '/tmp/agent-a1.jsonl', cwd: '/repo/worktrees/feature' }; + + const event = buildSubagentUsageEvent('session-1', file, [], {}, {}, 0, {}, '', '', 0); + + expect(event.worktree).toBe('/repo/worktrees/feature'); + // workflow_run has no known source regardless of cwd — still never fabricated. + expect(event.workflow_run).toBe(''); + }); + it('picks the usage[] row with the most api_calls as the top-level model', () => { const file: SubagentFile = { agentId: 'a1', filePath: '/tmp/agent-a1.jsonl' }; const reqs: OpenUsageRequest[] = [ @@ -330,7 +340,7 @@ describe('cross-check: agent.subagent.usage usage[] totals vs agent.usage.reques const lines = raw.split('\n').filter((l) => l.trim().length > 0); const reqs: OpenUsageRequest[] = lines - .map((line) => parseUsageLine(line, 'agent', '', file.agentId)) + .map((line) => parseUsageLine(line, 'agent', '', file.agentId, file.agentType ?? '')) .filter((r): r is OpenUsageRequest => r !== null); expect(reqs.length).toBeGreaterThan(0); diff --git a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts index 67f162875..cf181afe5 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/__tests__/usage-request.test.ts @@ -34,13 +34,13 @@ beforeAll(async () => { describe('parseUsageLine', () => { it('returns null for a line with no message.usage block', () => { expect(lines).toHaveLength(4); - const result = parseUsageLine(lines[0], 'main', '', ''); + const result = parseUsageLine(lines[0], 'main', '', '', ''); expect(result).toBeNull(); }); it('returns null for malformed JSON instead of throwing', () => { - expect(() => parseUsageLine('not valid json {{{', 'main', '', '')).not.toThrow(); - expect(parseUsageLine('not valid json {{{', 'main', '', '')).toBeNull(); + expect(() => parseUsageLine('not valid json {{{', 'main', '', '', '')).not.toThrow(); + expect(parseUsageLine('not valid json {{{', 'main', '', '', '')).toBeNull(); }); it('returns null for a usage-bearing line with neither requestId nor message.id, instead of collapsing it onto a shared ::model key', () => { @@ -53,11 +53,11 @@ describe('parseUsageLine', () => { }, }); - expect(parseUsageLine(line, 'main', '', '')).toBeNull(); + expect(parseUsageLine(line, 'main', '', '', '')).toBeNull(); }); it('extracts every field from a fully-populated line', () => { - const req = parseUsageLine(lines[3], 'main', '', ''); + const req = parseUsageLine(lines[3], 'main', '', '', ''); expect(req).not.toBeNull(); const r = req as OpenUsageRequest; @@ -93,7 +93,7 @@ describe('parseUsageLine', () => { }); it('reads the top-level requestId and message.id separately from a direct-session line', () => { - const r = parseUsageLine(directLines[0], 'main', '', '') as OpenUsageRequest; + const r = parseUsageLine(directLines[0], 'main', '', '', '') as OpenUsageRequest; expect(r.requestId).toBe('req_011CfrTbrJgSQRWgsdY5FZFc'); expect(r.messageId).toBe('msg_011CfrTbrWLhWiotBBJC62pV'); @@ -105,25 +105,32 @@ describe('parseUsageLine', () => { message: { role: 'assistant', model: 'm', usage: { input_tokens: 1, output_tokens: 1 } }, }); - const r = parseUsageLine(line, 'main', '', '') as OpenUsageRequest; + const r = parseUsageLine(line, 'main', '', '', '') as OpenUsageRequest; expect(r.requestId).toBe('req_only'); expect(r.messageId).toBe(''); }); - it('passes scopeKind/scopeName/agentId through verbatim from its own parameters', () => { - const req = parseUsageLine(lines[3], 'agent', 'reviewer', 'agent-42'); + it('passes scopeKind/scopeName/agentId/agentType through verbatim from its own parameters', () => { + const req = parseUsageLine(lines[3], 'agent', 'reviewer', 'agent-42', 'Explore'); expect(req?.scopeKind).toBe('agent'); expect(req?.scopeName).toBe('reviewer'); expect(req?.agentId).toBe('agent-42'); + expect(req?.agentType).toBe('Explore'); + }); + + it('defaults agentType to empty for a main-scoped line', () => { + const req = parseUsageLine(lines[3], 'main', '', '', ''); + + expect(req?.agentType).toBe(''); }); }); describe('mergeUsageRequest', () => { it('keeps the max output_tokens and the non-empty stop_reason across two records for the same message.id', () => { - const first = parseUsageLine(lines[1], 'main', '', ''); - const second = parseUsageLine(lines[2], 'main', '', ''); + const first = parseUsageLine(lines[1], 'main', '', '', ''); + const second = parseUsageLine(lines[2], 'main', '', '', ''); expect(first).not.toBeNull(); expect(second).not.toBeNull(); @@ -149,8 +156,8 @@ describe('mergeUsageRequest', () => { }); it('merges the two rows of one direct-session response into one record with both ids', () => { - const a = parseUsageLine(directLines[0], 'main', '', '') as OpenUsageRequest; - const b = parseUsageLine(directLines[1], 'main', '', '') as OpenUsageRequest; + const a = parseUsageLine(directLines[0], 'main', '', '', '') as OpenUsageRequest; + const b = parseUsageLine(directLines[1], 'main', '', '', '') as OpenUsageRequest; expect(usageRequestKey(a)).toBe(usageRequestKey(b)); const merged = mergeUsageRequest(a, b); @@ -161,7 +168,7 @@ describe('mergeUsageRequest', () => { }); it('keeps messageId when only the earlier record has it', () => { - const a = parseUsageLine(lines[1], 'main', '', '') as OpenUsageRequest; + const a = parseUsageLine(lines[1], 'main', '', '', '') as OpenUsageRequest; const merged = mergeUsageRequest(a, { ...a, messageId: '' }); expect(merged.messageId).toBe('msg_pair_1'); @@ -173,7 +180,7 @@ describe('mergeUsageRequest', () => { speed: 'standard', inferenceGeo: '', serviceTier: 'standard', inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, - scopeKind: 'main', scopeName: '', agentId: '', + scopeKind: 'main', scopeName: '', agentId: '', agentType: '', stopReason: '', isApiError: false, gitBranch: 'main', }; const b: OpenUsageRequest = { @@ -202,7 +209,7 @@ describe('mergeUsageRequest', () => { speed: '', inferenceGeo: '', serviceTier: '', inputTokens: 1, cacheCreation5mTokens: 0, cacheCreation1hTokens: 0, cacheReadTokens: 0, outputTokens: 1, webSearchRequests: 0, webFetchRequests: 0, - scopeKind: 'main', scopeName: '', agentId: '', + scopeKind: 'main', scopeName: '', agentId: '', agentType: '', stopReason: '', isApiError: false, gitBranch: '', }; const b: OpenUsageRequest = { ...a, outputTokens: 2, stopReason: 'end_turn', isApiError: true }; @@ -218,6 +225,14 @@ describe('mergeUsageRequest', () => { // isApiError: once true, stays true across merges. expect(merged.isApiError).toBe(true); }); + + it('merges agentType like every other non-numeric field: b wins when non-empty, else a', () => { + const first = parseUsageLine(lines[3], 'agent', '', 'agent-1', 'Explore') as OpenUsageRequest; + const laterWithoutAgentType = { ...first, agentType: '' }; + + expect(mergeUsageRequest(first, laterWithoutAgentType).agentType).toBe('Explore'); + expect(mergeUsageRequest(laterWithoutAgentType, first).agentType).toBe('Explore'); + }); }); describe('buildUsageRequestEvent', () => { @@ -227,7 +242,7 @@ describe('buildUsageRequestEvent', () => { speed: 'fast', inferenceGeo: 'us', serviceTier: 'priority', inputTokens: 10, cacheCreation5mTokens: 1, cacheCreation1hTokens: 2, cacheReadTokens: 3, outputTokens: 4, webSearchRequests: 5, webFetchRequests: 6, - scopeKind: 'skill', scopeName: 'brainstorming', agentId: 'agent-7', + scopeKind: 'skill', scopeName: 'brainstorming', agentId: 'agent-7', agentType: 'Explore', stopReason: 'end_turn', isApiError: false, gitBranch: 'main', }; @@ -253,6 +268,7 @@ describe('buildUsageRequestEvent', () => { scope_kind: 'skill', scope_name: 'brainstorming', agent_id: 'agent-7', + agent_type: 'Explore', stop_reason: 'end_turn', is_api_error: false, git_branch: 'main', @@ -283,7 +299,7 @@ describe('usageRequestKey', () => { describe('buildUsageRequestEvent - direct session', () => { it('emits the API request id as request_id and message.id as message_id', () => { - const req = parseUsageLine(directLines[2], 'main', '', '') as OpenUsageRequest; + const req = parseUsageLine(directLines[2], 'main', '', '', '') as OpenUsageRequest; const event = buildUsageRequestEvent('session-direct', req); diff --git a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts index aa24bd3ab..702b4b857 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/orchestrator.ts @@ -54,7 +54,7 @@ interface ContentBlock { tool_use_id?: string; is_error?: boolean; isError?: boolean; - input?: { file_path?: unknown; path?: unknown }; + input?: { file_path?: unknown; path?: unknown; notebook_path?: unknown; skill?: unknown }; } interface TranscriptLine extends SignalLine { @@ -128,6 +128,36 @@ function countToolResults(parsedLines: TranscriptLine[]): number { return count; } +/** + * The `Skill` tool's invoked name (`input.skill`) on `parsed`'s assistant content, or `''` when + * this line does not invoke one. Mirrors `claude-named-invocations.ts`'s own `Skill` detection, + * but only needs the single name on one line rather than a session-wide count. + */ +function findSkillInvocation(parsed: TranscriptLine): string { + const content = parsed.message?.content; + if (!Array.isArray(content)) return ''; + for (const item of content as ContentBlock[]) { + if (item?.type === 'tool_use' && item.name === 'Skill' && typeof item.input?.skill === 'string') { + const skill = item.input.skill.trim(); + if (skill) return skill; + } + } + return ''; +} + +/** + * Whether `parsed` is a genuine new user turn (a real prompt, not a line carrying only + * `tool_result` blocks) — the same criteria `session-signals.ts`'s `countTurns` uses. + */ +function isGenuineUserTurn(parsed: TranscriptLine): boolean { + if (parsed.type !== 'user' || parsed.isMeta || parsed.isSidechain || parsed.isCompactSummary) { + return false; + } + const content = parsed.message?.content; + if (typeof content === 'string') return true; + return Array.isArray(content) && content.some((block) => (block as { type?: unknown } | null)?.type === 'text'); +} + /** * Recompute the full-session summary accumulator, named-invocation counts, and session start * time from byte 0 of the main transcript. @@ -187,7 +217,9 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ // `usageRequestKey()` key `state.openRequests` uses before counting. const modelByRequestKey = new Map(); for (const line of rawLines) { - const parsedUsage = parseUsageLine(line, 'main', '', ''); + // scopeKind/scopeName/agentId/agentType don't affect this count (only `.model` is read), so + // 'main' is used uniformly rather than re-deriving collectMainTranscriptEvents's active-skill state. + const parsedUsage = parseUsageLine(line, 'main', '', '', ''); if (parsedUsage) { modelByRequestKey.set(usageRequestKey(parsedUsage), parsedUsage.model); } @@ -210,10 +242,12 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ } acc.toolCalls[item.name] = entry; - const filePath = item.input?.file_path ?? item.input?.path; + const filePath = item.input?.file_path ?? item.input?.path ?? item.input?.notebook_path; if (typeof filePath === 'string' && filePath) { if (item.name === 'Write') acc.filesWritten.add(filePath); - if (item.name === 'Edit') acc.filesEdited.add(filePath); + if (item.name === 'Edit' || item.name === 'MultiEdit' || item.name === 'NotebookEdit') { + acc.filesEdited.add(filePath); + } } } } @@ -259,9 +293,15 @@ async function buildFullAccumulator(transcriptPath: string): Promise<{ * Never forwards anything itself — the caller is responsible for sending the returned events to * the spool (exactly one place in the pipeline does that). * - * Scoping: no reliable transcript signal marks a *main*-transcript turn entering/exiting a - * "skill context", so every main-transcript usage record is scoped `scopeKind: 'main'`, - * `scopeName: ''`. `state.activeSkill` is deliberately neither read nor written. + * Scoping: `state.activeSkill` tracks which skill (if any) the main transcript is currently + * "inside" — set to a skill's name when a `Skill` tool_use invokes it ({@link findSkillInvocation}, + * applied *after* scoping that same line's own usage record, since the invoking turn itself still + * belongs to whatever was active before it), and cleared on the next genuine new user turn + * ({@link isGenuineUserTurn}) — a fresh prompt is treated as leaving any skill context the + * previous turn's work was under. Every usage record is scoped `scopeKind: 'skill'`, + * `scopeName: state.activeSkill` while a skill is active, else `scopeKind: 'main'`, `scopeName: ''`. + * This is a heuristic, not a tracked boundary Claude Code itself reports — a skill whose work + * spans multiple turns without an intervening user prompt is scoped as one continuous span. * * Swallows every error internally — never throws into `processOtlpEvent`. */ @@ -281,26 +321,41 @@ export async function collectMainTranscriptEvents( for (const line of lines) { let rawGitBranch = ''; let isUserLine = false; + let head: TranscriptLine | null = null; try { - const head = JSON.parse(line) as { gitBranch?: string; type?: string }; + head = JSON.parse(line) as TranscriptLine; rawGitBranch = head?.gitBranch ?? ''; isUserLine = head?.type === 'user'; } catch { // Malformed line: still attempt usage parsing below (which has its own try/catch), but - // there is no branch to record from it. + // there is no branch/skill/turn signal to read from it. } // The contract's branch counts are user lines per branch. if (rawGitBranch && isUserLine) { updateBranchCounts(state.branchCounts, rawGitBranch); } - const parsed = parseUsageLine(line, 'main', '', ''); + if (head && isGenuineUserTurn(head)) { + state.activeSkill = ''; + } + + const scopeKind = state.activeSkill ? 'skill' : 'main'; + const parsed = parseUsageLine(line, scopeKind, state.activeSkill, '', ''); if (parsed) { const key = usageRequestKey(parsed); const existing = state.openRequests[key]; state.openRequests[key] = existing ? mergeUsageRequest(existing, parsed) : parsed; touchedKeys.add(key); } + + // Applied after this line's own usage is scoped — the turn that invokes a skill still + // belongs to whatever context was active before it. + if (head) { + const skill = findSkillInvocation(head); + if (skill) { + state.activeSkill = skill; + } + } } state.mainOffset = nextOffset; @@ -508,7 +563,7 @@ export async function collectSubagentTranscriptEvents( const touchedKeys = new Set(); for (const line of lines) { - const parsed = parseUsageLine(line, 'agent', '', subagentFile.agentId); + const parsed = parseUsageLine(line, 'agent', '', subagentFile.agentId, subagentFile.agentType ?? ''); if (parsed) { const key = usageRequestKey(parsed); const existing = state.openRequests[key]; diff --git a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts index ebf14fd81..a3409c1df 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/parse-state.ts @@ -33,6 +33,8 @@ export interface OpenUsageRequest { scopeKind: 'main' | 'skill' | 'agent'; scopeName: string; agentId: string; + /** The subagent's type (e.g. `'Explore'`); `''` for `scopeKind: 'main' | 'skill'`. */ + agentType: string; stopReason: string; isApiError: boolean; gitBranch: string; diff --git a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts index 837d6410e..570086bab 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/subagent-usage.ts @@ -13,7 +13,9 @@ * scope_kind/scope_name) and passes the caller-built tool-call/tool-error/skill maps through as * the contract's `tools`/`skills` objects; it has no access to the subagent transcript itself. * - * `workflow_run`/`worktree` have no known source and are always empty strings, never fabricated. + * `workflow_run` has no known source and is always an empty string, never fabricated. `worktree` + * is the hook's own `cwd` (the contract's "its worktree path") — the caller fills it in from the + * triggering hook payload, since no sidecar or transcript signal carries it. */ import { readdir, readFile } from 'node:fs/promises'; @@ -27,6 +29,8 @@ export interface SubagentFile { agentType?: string; spawnDepth?: number; description?: string; + /** The triggering hook's `cwd` — the contract's `worktree` path. `''` when the caller has none. */ + cwd?: string; } interface SubagentMeta { @@ -228,7 +232,7 @@ export function buildSubagentUsageEvent( description: file.description ?? '', workflow_run: '', spawn_depth: file.spawnDepth ?? 0, - worktree: '', + worktree: file.cwd ?? '', started_at: startedAt, ended_at: endedAt, duration_ms: durationMs, diff --git a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts index fd8b42e6b..6ffb7fec8 100644 --- a/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts +++ b/src/agents/plugins/claude-code-otlp/transcript/usage-request.ts @@ -61,15 +61,16 @@ interface TranscriptUsageLine { * carries no `message.usage` block (not a billable API response — e.g. a plain user/system * message) or is not valid JSON. * - * `scopeKind`/`scopeName`/`agentId` are passed through verbatim from the caller, which already - * knows which transcript (main vs. a named skill context vs. a subagent transcript) `line` came - * from — this function has no way to derive that from the line itself. + * `scopeKind`/`scopeName`/`agentId`/`agentType` are passed through verbatim from the caller, which + * already knows which transcript (main vs. a named skill context vs. a subagent transcript) `line` + * came from — this function has no way to derive that from the line itself. */ export function parseUsageLine( line: string, scopeKind: 'main' | 'skill' | 'agent', scopeName: string, - agentId: string + agentId: string, + agentType: string ): OpenUsageRequest | null { let parsed: TranscriptUsageLine; try { @@ -117,6 +118,7 @@ export function parseUsageLine( scopeKind, scopeName, agentId, + agentType, // Sibling of usage on message, not nested inside it. stopReason: parsed.message?.stop_reason ?? '', isApiError: Boolean(parsed.isApiError), @@ -164,6 +166,7 @@ export function mergeUsageRequest(a: OpenUsageRequest, b: OpenUsageRequest): Ope scopeKind: b.scopeKind || a.scopeKind, scopeName: b.scopeName || a.scopeName, agentId: b.agentId || a.agentId, + agentType: b.agentType || a.agentType, stopReason: b.stopReason || a.stopReason, isApiError: b.isApiError || a.isApiError, gitBranch: b.gitBranch || a.gitBranch, @@ -196,6 +199,7 @@ export function buildUsageRequestEvent(sessionId: string, req: OpenUsageRequest) scope_kind: req.scopeKind, scope_name: req.scopeName, agent_id: req.agentId, + agent_type: req.agentType, stop_reason: req.stopReason, is_api_error: req.isApiError, git_branch: req.gitBranch,