From bd47ceb4078829a8c4807eba9d6f2b37708b91cc Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:21:20 +0200 Subject: [PATCH 01/36] Add privacy-safe Claude Code OTel event adapter --- shared/claude_otel_adapter.py | 386 ++++++++++++++++++++++++++++++++++ 1 file changed, 386 insertions(+) create mode 100644 shared/claude_otel_adapter.py diff --git a/shared/claude_otel_adapter.py b/shared/claude_otel_adapter.py new file mode 100644 index 00000000..742b9465 --- /dev/null +++ b/shared/claude_otel_adapter.py @@ -0,0 +1,386 @@ +from __future__ import annotations + +"""Translate Claude Code OTLP log events into privacy-safe OWG agent evidence. + +Claude Code can emit prompt/response text, tool parameters/content, raw API bodies, +filesystem paths and identity attributes. This adapter deliberately ignores all of +that. It reads only a small structural allowlist from documented Claude Code events. +""" + +from datetime import datetime, timezone +import hashlib +import re +from typing import Any, Iterator + + +_SUPPORTED_EVENTS = frozenset({ + "claude_code.user_prompt", + "claude_code.api_request", + "claude_code.api_error", + "claude_code.api_refusal", + "claude_code.tool_result", + "claude_code.tool_decision", + "claude_code.api_retries_exhausted", + "claude_code.subagent_completed", +}) +_HUMAN_DECISION_SOURCES = frozenset({ + "user_permanent", + "user_temporary", + "user_abort", + "user_reject", +}) +_SAFE_LABEL = re.compile(r"^[A-Za-z][A-Za-z0-9_.:/-]{0,199}$") + + +def _value(value: Any) -> Any: + if not isinstance(value, dict): + return value + for key in ("stringValue", "intValue", "doubleValue", "boolValue", "bytesValue"): + if key in value: + return value[key] + return None + + +def _attrs(raw: Any) -> dict[str, Any]: + if isinstance(raw, dict): + return dict(raw) + out: dict[str, Any] = {} + if isinstance(raw, list): + for item in raw: + if not isinstance(item, dict): + continue + key = str(item.get("key") or "").strip() + if key: + out[key] = _value(item.get("value")) + return out + + +def _text(value: Any, limit: int = 240) -> str: + return re.sub(r"\s+", " ", str(value or "")).strip()[:limit] + + +def _safe_label(value: Any, *, default: str = "", limit: int = 160) -> str: + raw = _text(value, min(limit, 200)) + if not raw or not _SAFE_LABEL.fullmatch(raw): + return default + return raw[:limit] + + +def _bool(value: Any) -> bool | None: + if isinstance(value, bool): + return value + low = _text(value, 20).lower() + if low in {"true", "1", "yes"}: + return True + if low in {"false", "0", "no"}: + return False + return None + + +def _int(value: Any) -> int | None: + try: + number = int(value) + except Exception: + return None + return number if 0 <= number <= 1_000_000_000 else None + + +def _duration_seconds(value: Any) -> float: + try: + millis = float(value or 0) + except Exception: + return 0.0 + return round(max(0.0, min(millis / 1000.0, 7 * 24 * 60 * 60)), 6) + + +def _iso_from_nanos(value: Any) -> str: + try: + nanos = int(value) + except Exception: + return "" + try: + return datetime.fromtimestamp(nanos / 1_000_000_000, timezone.utc).isoformat() + except Exception: + return "" + + +def _observed_at(attrs: dict[str, Any], native: dict[str, Any]) -> str: + timestamp = _text(attrs.get("event.timestamp"), 80) + if timestamp: + return timestamp + for key in ("timeUnixNano", "time_unix_nano", "observedTimeUnixNano", "observed_time_unix_nano"): + parsed = _iso_from_nanos(native.get(key)) + if parsed: + return parsed + return datetime.now(timezone.utc).isoformat() + + +def _hash(value: Any) -> str: + raw = _text(value, 500) + return hashlib.sha256(raw.encode("utf-8")).hexdigest()[:24] if raw else "" + + +def _body_text(body: Any) -> str: + value = _value(body) + return _text(value, 120) if isinstance(value, (str, int, float, bool)) else "" + + +def _event_name(record: dict[str, Any], attrs: dict[str, Any]) -> str: + raw = _text(attrs.get("event.name"), 120).lower() + if raw.startswith("claude_code."): + return raw + if raw: + candidate = f"claude_code.{raw}" + if candidate in _SUPPORTED_EVENTS: + return candidate + body = _body_text(record.get("body")).lower() + if body.startswith("claude_code."): + return body + return "" + + +def _iter_records(payload: dict[str, Any], max_records: int) -> Iterator[tuple[dict[str, Any], dict[str, Any], dict[str, Any]]]: + seen = 0 + for resource_log in payload.get("resourceLogs") or []: + if not isinstance(resource_log, dict): + continue + resource = resource_log.get("resource") if isinstance(resource_log.get("resource"), dict) else {} + resource_attrs = _attrs(resource.get("attributes")) + for scope_log in resource_log.get("scopeLogs") or []: + if not isinstance(scope_log, dict): + continue + for record in scope_log.get("logRecords") or []: + if not isinstance(record, dict): + continue + seen += 1 + if seen > max_records: + raise ValueError(f"Claude Code OTLP payload exceeds {max_records} records") + yield record, _attrs(record.get("attributes")), resource_attrs + + # Deterministic small form for adapter tests and custom relays. + for record in payload.get("records") or []: + if not isinstance(record, dict): + continue + seen += 1 + if seen > max_records: + raise ValueError(f"Claude Code OTLP payload exceeds {max_records} records") + yield record, _attrs(record.get("attributes")), _attrs(payload.get("resource")) + + +def _usage(attrs: dict[str, Any]) -> dict[str, int]: + out: dict[str, int] = {} + mapping = { + "input_tokens": "input_tokens", + "output_tokens": "output_tokens", + "cached_input_tokens": "cache_read_tokens", + } + for target, source in mapping.items(): + amount = _int(attrs.get(source)) + if amount is not None: + out[target] = amount + if "input_tokens" in out or "output_tokens" in out: + out["total_tokens"] = out.get("input_tokens", 0) + out.get("output_tokens", 0) + return out + + +def _tool_category(name: str, attrs: dict[str, Any]) -> str: + low = name.lower() + tool_source = _text(attrs.get("tool_source"), 80).lower() + if tool_source in {"mcp", "sdk_host_builtin_mcp"} or low.startswith("mcp__") or low == "mcp_tool": + return "mcp" + if low in {"bash", "shell", "terminal", "computer"} or any(token in low for token in ("exec", "command", "powershell")): + return "shell" + if any(token in low for token in ("read", "write", "edit", "file", "notebook")): + return "filesystem" + if any(token in low for token in ("search", "grep", "glob", "find", "lookup")): + return "search" + if any(token in low for token in ("webfetch", "browser", "playwright", "chrome")): + return "browser" + if any(token in low for token in ("github", "git", "code")): + return "code" + return "other" if name else "none" + + +def _event_id(run_id: str, event_name: str, attrs: dict[str, Any]) -> str: + if event_name in {"claude_code.tool_result", "claude_code.tool_decision"}: + discriminator = _text(attrs.get("tool_use_id"), 240) or _text(attrs.get("event.sequence"), 80) + elif event_name in {"claude_code.api_request", "claude_code.api_error", "claude_code.api_refusal"}: + discriminator = ( + _text(attrs.get("request_id"), 240) + or _text(attrs.get("client_request_id"), 240) + or _text(attrs.get("event.sequence"), 80) + ) + elif event_name == "claude_code.subagent_completed": + discriminator = "|".join([ + _text(attrs.get("event.sequence"), 80), + _safe_label(attrs.get("agent_type"), default="subagent", limit=80), + ]) + else: + discriminator = _text(attrs.get("event.sequence"), 80) or _text(attrs.get("event.timestamp"), 80) + material = f"{run_id}\x1f{event_name}\x1f{discriminator}".encode("utf-8") + return "claude-otel:" + hashlib.sha256(material).hexdigest()[:40] + + +def _base( + *, + attrs: dict[str, Any], + resource_attrs: dict[str, Any], + native: dict[str, Any], + event_name: str, + operation: str, + status: str, + defaults: dict[str, Any], + tool_name: str = "", + span_id: str = "", +) -> dict[str, Any] | None: + session_id = _text(attrs.get("session.id") or resource_attrs.get("session.id"), 128) + prompt_id = _text(attrs.get("prompt.id"), 128) + run_id = prompt_id or session_id + if not run_id: + return None + model = _safe_label(attrs.get("model"), limit=200) + return { + "event_id": _event_id(run_id, event_name, attrs), + "observed_at": _observed_at(attrs, native), + "organization_id": _text(defaults.get("organization_id"), 128), + "actor_id": _text(defaults.get("actor_id"), 128), + "device_id": _text(defaults.get("device_id"), 128) or "claude-local", + "sensor_id": "agent:claude-code-otel", + "session_id": session_id or run_id, + "agent_name": _safe_label(defaults.get("agent_name"), default="Claude-Code", limit=160), + "provider": "anthropic", + "framework": "claude-code", + "model": model, + "operation": operation, + "status": status, + "observation_level": "native_trace", + "run_id": run_id, + "trace_id": session_id or run_id, + "span_id": span_id, + "tool_name": tool_name, + "tool_category": _tool_category(tool_name, attrs), + "duration_seconds": _duration_seconds(attrs.get("duration_ms")), + "usage": _usage(attrs), + } + + +def claude_otel_to_agent_events( + payload: dict[str, Any], + *, + defaults: dict[str, Any] | None = None, + max_records: int = 1000, +) -> tuple[list[dict[str, Any]], dict[str, int]]: + """Project documented Claude Code log events without copying event content.""" + if not isinstance(payload, dict): + raise ValueError("Claude Code OpenTelemetry payload must be an object") + defaults = dict(defaults or {}) + events: list[dict[str, Any]] = [] + seen = 0 + ignored = 0 + + for native, attrs, resource_attrs in _iter_records(payload, max_records): + seen += 1 + event_name = _event_name(native, attrs) + if event_name not in _SUPPORTED_EVENTS: + ignored += 1 + continue + + projected: dict[str, Any] | None = None + if event_name == "claude_code.user_prompt": + projected = _base( + attrs=attrs, + resource_attrs=resource_attrs, + native=native, + event_name=event_name, + operation="run_started", + status="running", + defaults=defaults, + ) + + elif event_name in {"claude_code.api_request", "claude_code.api_error", "claude_code.api_refusal"}: + status = "success" if event_name == "claude_code.api_request" else "error" if event_name == "claude_code.api_error" else "denied" + request_id = _text(attrs.get("request_id") or attrs.get("client_request_id"), 128) + projected = _base( + attrs=attrs, + resource_attrs=resource_attrs, + native=native, + event_name=event_name, + operation="model_call", + status=status, + defaults=defaults, + span_id=("request:" + _hash(request_id)) if request_id else "", + ) + + elif event_name == "claude_code.tool_result": + tool_name = _safe_label(attrs.get("tool_name"), default="unknown-tool", limit=160) + success = _bool(attrs.get("success")) + tool_id = _text(attrs.get("tool_use_id"), 128) + projected = _base( + attrs=attrs, + resource_attrs=resource_attrs, + native=native, + event_name=event_name, + operation="tool_call", + status="success" if success is True else "error" if success is False else "unknown", + defaults=defaults, + tool_name=tool_name, + span_id=("tool:" + _hash(tool_id)) if tool_id else "", + ) + + elif event_name == "claude_code.tool_decision": + source = _text(attrs.get("source"), 80).lower() + if source not in _HUMAN_DECISION_SOURCES: + ignored += 1 + continue + decision = _text(attrs.get("decision"), 40).lower() + tool_name = _safe_label(attrs.get("tool_name"), default="unknown-tool", limit=160) + tool_id = _text(attrs.get("tool_use_id"), 128) + projected = _base( + attrs=attrs, + resource_attrs=resource_attrs, + native=native, + event_name=event_name, + operation="human_approval_received", + status="success" if decision == "accept" else "denied" if decision == "reject" else "unknown", + defaults=defaults, + tool_name=tool_name, + span_id=("tool:" + _hash(tool_id)) if tool_id else "", + ) + + elif event_name == "claude_code.subagent_completed": + agent_type = _safe_label(attrs.get("agent_type"), default="subagent", limit=80) + projected = _base( + attrs=attrs, + resource_attrs=resource_attrs, + native=native, + event_name=event_name, + operation="handoff", + status="success", + defaults=defaults, + tool_name=f"subagent:{agent_type}", + ) + + elif event_name == "claude_code.api_retries_exhausted": + projected = _base( + attrs=attrs, + resource_attrs=resource_attrs, + native=native, + event_name=event_name, + operation="error", + status="error", + defaults=defaults, + ) + + if projected is None: + ignored += 1 + continue + events.append(projected) + + unique: dict[str, dict[str, Any]] = {} + for event in events: + unique.setdefault(str(event["event_id"]), event) + return list(unique.values()), { + "records_seen": seen, + "records_ignored": ignored, + "agent_events": len(unique), + } From 791b017fb71b8ca960cd084ae809be0735b7663a Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:21:36 +0200 Subject: [PATCH 02/36] Ingest Claude Code structural OTel events --- server/agent_ingest.py | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/server/agent_ingest.py b/server/agent_ingest.py index 18a11dfb..c0229eff 100644 --- a/server/agent_ingest.py +++ b/server/agent_ingest.py @@ -8,6 +8,7 @@ from shared.agent_evidence import AgentEvidenceError, agent_event_to_evidence from shared.agent_ingress_validation import validate_agent_ingress_event from shared.capture_control import filter_recordable +from shared.claude_otel_adapter import claude_otel_to_agent_events from shared.codex_otel_adapter import codex_otel_to_agent_events from shared.otel_agent_adapter import otel_payload_to_agent_events from .db import insert_events @@ -16,6 +17,7 @@ MAX_AGENT_BATCH_BYTES = 2_000_000 MAX_OTEL_SPANS = 1000 MAX_CODEX_OTEL_RECORDS = 1000 +MAX_CLAUDE_OTEL_RECORDS = 1000 def _bounded_json_size(value: Any, *, maximum: int = MAX_AGENT_BATCH_BYTES) -> None: @@ -102,3 +104,27 @@ def ingest_codex_otel_payload( "projected": len(events), "inserted": inserted, } + + +def ingest_claude_otel_payload( + payload: dict[str, Any], + *, + defaults: dict[str, Any] | None = None, +) -> dict[str, int]: + """Ingest Claude Code OTLP logs through a strict structural allowlist.""" + _bounded_json_size({"payload": payload, "defaults": defaults or {}}) + projected, stats = claude_otel_to_agent_events( + payload, + defaults=defaults, + max_records=MAX_CLAUDE_OTEL_RECORDS, + ) + if len(projected) > MAX_AGENT_EVENTS * 2: + raise AgentEvidenceError("Claude Code OpenTelemetry projection produced too many agent events") + events = _validated_events(projected) + inserted = _insert_recordable(events) + return { + "records_seen": int(stats.get("records_seen") or 0), + "records_ignored": int(stats.get("records_ignored") or 0), + "projected": len(events), + "inserted": inserted, + } From 4080d86a5e5d8db56561a03805d2557319f0413e Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:21:59 +0200 Subject: [PATCH 03/36] Expose write-only Claude Code OTel ingest endpoint --- server/agent_routes.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/server/agent_routes.py b/server/agent_routes.py index 6eab1065..223960e0 100644 --- a/server/agent_routes.py +++ b/server/agent_routes.py @@ -11,6 +11,7 @@ from .agent_ingest import ( MAX_AGENT_BATCH_BYTES, ingest_agent_payloads, + ingest_claude_otel_payload, ingest_codex_otel_payload, ingest_otel_payload, ) @@ -34,6 +35,7 @@ AGENT_EVENT_PATH = "/agent-ingest/v1/events" AGENT_OTEL_PATH = "/agent-ingest/v1/otel" AGENT_CODEX_OTEL_PATH = "/agent-ingest/v1/codex-otel" +AGENT_CLAUDE_OTEL_PATH = "/agent-ingest/v1/claude-otel" class OTelDefaults(BaseModel): @@ -140,6 +142,22 @@ async def ingest_codex_otel(request: Request) -> Response: return Response(status_code=202) +@router.post(AGENT_CLAUDE_OTEL_PATH, status_code=202) +async def ingest_claude_otel(request: Request) -> Response: + """Accept Claude Code OTLP/HTTP JSON logs on a write-only local endpoint.""" + _require_agent_write_bearer(request) + payload = await _read_bounded_json(request) + if not isinstance(payload, dict): + raise HTTPException(status_code=422, detail="OpenTelemetry payload must be an object") + defaults_raw = payload.pop("openworkgraph", {}) + try: + defaults = OTelDefaults.model_validate(defaults_raw if isinstance(defaults_raw, dict) else {}).model_dump() + ingest_claude_otel_payload(payload, defaults=defaults) + except (AgentEvidenceError, ValueError) as exc: + raise HTTPException(status_code=422, detail=str(exc)) from exc + return Response(status_code=202) + + @router.get("/v1/agent-workflows") def get_agent_workflows( request: Request, From 524245b01669f3a1f7a7b1f3eb01f206fabdaded Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:22:48 +0200 Subject: [PATCH 04/36] Correlate Claude hooks by prompt and observe subagent handoffs --- shared/claude_code_adapter.py | 117 ++++++++++++++++++++++++++-------- 1 file changed, 89 insertions(+), 28 deletions(-) diff --git a/shared/claude_code_adapter.py b/shared/claude_code_adapter.py index 2480e6bf..6fc49cf2 100644 --- a/shared/claude_code_adapter.py +++ b/shared/claude_code_adapter.py @@ -59,7 +59,7 @@ def _duration_seconds(payload: dict[str, Any]) -> float: def _tool_category(name: str) -> str: low = name.lower() - if low.startswith("mcp__"): + if low.startswith("mcp__") or low == "mcp_tool": return "mcp" if low in {"bash", "shell", "terminal", "computer"} or any( token in low for token in ("exec", "command", "powershell") @@ -76,6 +76,21 @@ def _tool_category(name: str) -> str: return "other" if name else "none" +def _prompt_run_id(payload: dict[str, Any], session_id: str) -> str: + # Claude Code v2.1.196+ gives every within-turn hook the same prompt_id used + # by its OTel prompt.id. Prefer it so hook evidence and OTel evidence land in + # one execution. Older versions safely fall back to the session boundary. + return _text(payload.get("prompt_id"), 128) or session_id + + +def _hook_agent_name(payload: dict[str, Any]) -> str: + agent_id = _text(payload.get("agent_id"), 128) + if not agent_id: + return "Claude Code" + agent_type = _safe_label(payload.get("agent_type"), default="subagent", limit=80) + return f"Claude Code/{agent_type}" + + def _base_event( payload: dict[str, Any], *, @@ -87,20 +102,26 @@ def _base_event( event_key: str, agent_name: str = "Claude Code", span_id: str = "", + parent_span_id: str = "", tool_name: str = "", + model: str = "", ) -> dict[str, Any]: return { - "event_id": _event_id(trace_id, event_key, span_id, run_id), + "event_id": _event_id(trace_id, event_key, span_id, run_id, agent_name), "observed_at": observed_at, + "sensor_id": "agent:claude-code-hook", + "session_id": trace_id, "agent_name": agent_name, "provider": "anthropic", "framework": "claude-code", + "model": model, "operation": operation, "status": status, "observation_level": "native_trace", "run_id": run_id, "trace_id": trace_id, "span_id": span_id, + "parent_span_id": parent_span_id, "tool_name": tool_name, "tool_category": _tool_category(tool_name), "duration_seconds": _duration_seconds(payload), @@ -112,11 +133,11 @@ def claude_hook_to_agent_events( *, observed_at: str | None = None, ) -> list[dict[str, Any]]: - """Project one Claude Code hook invocation into zero or one safe events. + """Project one Claude Code hook invocation into privacy-safe structural events. - Hooks that expose content but do not add reliable structural workflow signal - (for example UserPromptSubmit, PreToolUse, MessageDisplay, and Stop) are - intentionally ignored. Tool calls are recorded only after success/failure. + Content-bearing hooks remain ignored. Within-turn events prefer prompt_id, so + they correlate with Claude Code OTel without exposing prompt content. A + SubagentStart produces both the parent handoff and a child-run boundary. """ if not isinstance(payload, dict): return [] @@ -124,11 +145,13 @@ def claude_hook_to_agent_events( if hook not in _SUPPORTED_EVENTS: return [] - session_id = _text(payload.get("session_id"), 240) + session_id = _text(payload.get("session_id"), 128) if not session_id: return [] timestamp = _text(observed_at, 80) or _now_iso() trace_id = session_id + prompt_run_id = _prompt_run_id(payload, session_id) + agent_name = _hook_agent_name(payload) if hook == "SessionStart": return [_base_event( @@ -139,6 +162,7 @@ def claude_hook_to_agent_events( run_id=session_id, trace_id=trace_id, event_key=hook, + model=_safe_label(payload.get("model"), limit=160), )] if hook == "SessionEnd": @@ -153,81 +177,118 @@ def claude_hook_to_agent_events( )] if hook in {"PostToolUse", "PostToolUseFailure"}: - tool_use_id = _text(payload.get("tool_use_id"), 240) + tool_use_id = _text(payload.get("tool_use_id"), 128) tool_name = _safe_label(payload.get("tool_name"), default="unknown-tool", limit=160) return [_base_event( payload, operation="tool_call", status="success" if hook == "PostToolUse" else "error", observed_at=timestamp, - run_id=session_id, + run_id=prompt_run_id, trace_id=trace_id, span_id=tool_use_id, event_key=hook, tool_name=tool_name, + agent_name=agent_name, )] if hook == "PermissionRequest": - tool_use_id = _text(payload.get("tool_use_id"), 240) + tool_use_id = _text(payload.get("tool_use_id"), 128) tool_name = _safe_label(payload.get("tool_name"), default="unknown-tool", limit=160) return [_base_event( payload, operation="human_approval_requested", status="running", observed_at=timestamp, - run_id=session_id, + run_id=prompt_run_id, trace_id=trace_id, span_id=tool_use_id, event_key=hook, tool_name=tool_name, + agent_name=agent_name, )] if hook == "PermissionDenied": - # Claude Code emits PermissionDenied in auto mode. Do not mislabel an - # automatic policy denial as a human approval/denial decision. - tool_use_id = _text(payload.get("tool_use_id"), 240) + # PermissionDenied can be an automatic policy denial. Do not turn it into + # a human decision; Claude's tool_decision OTel event identifies the + # actual decision source when richer telemetry is connected. + tool_use_id = _text(payload.get("tool_use_id"), 128) tool_name = _safe_label(payload.get("tool_name"), default="unknown-tool", limit=160) return [_base_event( payload, operation="error", status="denied", observed_at=timestamp, - run_id=session_id, + run_id=prompt_run_id, trace_id=trace_id, span_id=tool_use_id, event_key=hook, tool_name=tool_name, + agent_name=agent_name, )] - if hook in {"SubagentStart", "SubagentStop"}: - agent_id = _text(payload.get("agent_id"), 240) - if not agent_id: + if hook == "SubagentStart": + child_id = _text(payload.get("agent_id"), 128) + if not child_id: + return [] + child_type = _safe_label(payload.get("agent_type"), default="subagent", limit=80) + child_name = f"Claude Code/{child_type}" + # One event belongs to the parent execution and records the delegation; + # the other opens the child execution under the same prompt correlation. + return [ + _base_event( + payload, + operation="handoff", + status="running", + observed_at=timestamp, + run_id=prompt_run_id, + trace_id=trace_id, + span_id=child_id, + event_key="SubagentHandoff", + tool_name=f"subagent:{child_type}", + agent_name="Claude Code", + ), + _base_event( + payload, + operation="run_started", + status="running", + observed_at=timestamp, + run_id=prompt_run_id, + trace_id=trace_id, + span_id=child_id, + event_key=hook, + agent_name=child_name, + ), + ] + + if hook == "SubagentStop": + child_id = _text(payload.get("agent_id"), 128) + if not child_id: return [] - agent_type = _safe_label(payload.get("agent_type"), default="subagent", limit=80) - sub_run_id = f"{session_id}:{agent_id}" + child_type = _safe_label(payload.get("agent_type"), default="subagent", limit=80) return [_base_event( payload, - operation="run_started" if hook == "SubagentStart" else "run_finished", - status="running" if hook == "SubagentStart" else "unknown", + operation="run_finished", + status="unknown", observed_at=timestamp, - run_id=sub_run_id, + run_id=prompt_run_id, trace_id=trace_id, - span_id=agent_id, + span_id=child_id, event_key=hook, - agent_name=f"Claude Code/{agent_type}", + agent_name=f"Claude Code/{child_type}", )] if hook == "StopFailure": - prompt_id = _text(payload.get("prompt_id"), 240) return [_base_event( payload, operation="error", status="error", observed_at=timestamp, - run_id=session_id, + run_id=prompt_run_id, trace_id=trace_id, - span_id=prompt_id, + span_id=_text(payload.get("prompt_id"), 128), event_key=hook, + agent_name=agent_name, )] return [] From 36f5f218f178f0e6e0b29ec8fdce1087789dc10a Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:24:12 +0200 Subject: [PATCH 05/36] Add capability-aware agent observability summaries --- server/agent_observability.py | 188 ++++++++++++++++++++++++++++++++++ 1 file changed, 188 insertions(+) create mode 100644 server/agent_observability.py diff --git a/server/agent_observability.py b/server/agent_observability.py new file mode 100644 index 00000000..74de6491 --- /dev/null +++ b/server/agent_observability.py @@ -0,0 +1,188 @@ +from __future__ import annotations + +"""Capability-aware summaries over canonical agent evidence. + +Counts answer "what did OWG observe?" while capabilities answer "could the active +adapter observe this signal at all?" Keeping those separate prevents unsupported +signals from being rendered as misleading zeroes. +""" + +from collections import Counter +from typing import Any + +from .context_execution_linkage import _agent_groups, _meta, _one_execution + + +_SIGNAL_KEYS = ( + "model_call", + "tool_call", + "handoff", + "human_approval_requested", + "human_approval_received", + "token_usage", + "model_identity", + "duration", +) + + +def _sensor_ids(events: list[dict[str, Any]]) -> set[str]: + return {str(event.get("sensor_id") or "").strip() for event in events if event.get("sensor_id")} + + +def _framework(events: list[dict[str, Any]]) -> str: + for event in events: + meta, _trace = _meta(event) + agent = meta.get("agent") if isinstance(meta.get("agent"), dict) else {} + value = str(agent.get("framework") or "").strip().lower() + if value: + return value + return "" + + +def _capability(status: str, basis: str) -> dict[str, str]: + return {"status": status, "basis": basis} + + +def _capabilities(events: list[dict[str, Any]]) -> dict[str, dict[str, str]]: + sensors = _sensor_ids(events) + framework = _framework(events) + caps = {key: _capability("unknown", "adapter_capability_not_declared") for key in _SIGNAL_KEYS} + + if framework == "claude-code": + has_otel = "agent:claude-code-otel" in sensors + has_hooks = "agent:claude-code-hook" in sensors + if has_otel: + for key in ("model_call", "tool_call", "handoff", "human_approval_received", "token_usage", "model_identity", "duration"): + caps[key] = _capability("observable", "claude_code_otel_logs") + if has_hooks: + for key in ("tool_call", "handoff", "human_approval_requested", "duration"): + caps[key] = _capability("observable", "claude_code_hooks") + if not has_otel: + for key in ("model_call", "human_approval_received", "token_usage"): + caps[key] = _capability("not_observable", "claude_code_hooks_do_not_emit_signal") + if not has_hooks: + caps["human_approval_requested"] = _capability("not_observable", "claude_code_otel_reports_decision_not_prompt_open") + return caps + + if framework == "codex": + for key in ("model_call", "tool_call", "handoff", "human_approval_received", "model_identity", "duration"): + caps[key] = _capability("observable", "codex_otel") + caps["human_approval_requested"] = _capability("not_observable", "codex_otel_decision_surface_has_no_request_event") + caps["token_usage"] = _capability("partial", "codex_usage_is_span_or_turn_dependent") + return caps + + if framework.startswith("openai-agents"): + for key in ("model_call", "tool_call", "handoff", "token_usage", "model_identity", "duration"): + caps[key] = _capability("observable", "openai_agents_tracing_processor") + for key in ("human_approval_requested", "human_approval_received"): + caps[key] = _capability("not_observable", "openai_agents_processor_has_no_approval_lifecycle_signal") + return caps + + if any(sensor.startswith("otel:") or sensor == "agent:otel" for sensor in sensors): + for key in ("model_call", "tool_call", "token_usage", "model_identity", "duration"): + caps[key] = _capability("observable", "generic_genai_otel_semantic_conventions") + for key in ("handoff", "human_approval_requested", "human_approval_received"): + caps[key] = _capability("not_observable", "generic_genai_otel_contract_has_no_portable_signal") + return caps + + +def _effective_events(events: list[dict[str, Any]]) -> list[dict[str, Any]]: + """Prefer richer Claude OTel evidence where hooks report the same tool call. + + OWG intentionally keeps both canonical rows. This read-only projection avoids + double-counting after a user upgrades an existing hook installation to OTel. + Handoff initiation remains hook-preferred when both feeds are present, because + the OTel subagent event is emitted on completion rather than delegation start. + """ + framework = _framework(events) + sensors = _sensor_ids(events) + if framework != "claude-code" or not {"agent:claude-code-hook", "agent:claude-code-otel"}.issubset(sensors): + return events + + hook_handoff_observed = any( + str(event.get("sensor_id") or "") == "agent:claude-code-hook" + and str(_meta(event)[0].get("operation") or "") == "handoff" + for event in events + ) + output: list[dict[str, Any]] = [] + for event in events: + meta, _trace = _meta(event) + operation = str(meta.get("operation") or "") + sensor = str(event.get("sensor_id") or "") + if operation == "tool_call" and sensor == "agent:claude-code-hook": + continue + if operation == "handoff" and hook_handoff_observed and sensor == "agent:claude-code-otel": + continue + output.append(event) + return output + + +def _usage_totals(events: list[dict[str, Any]]) -> dict[str, int]: + totals: Counter[str] = Counter() + for event in events: + meta, _trace = _meta(event) + usage = meta.get("usage") if isinstance(meta.get("usage"), dict) else {} + for key in ("input_tokens", "output_tokens", "cached_input_tokens", "total_tokens"): + try: + value = int(usage.get(key)) + except Exception: + continue + if value >= 0: + totals[key] += value + return dict(totals) + + +def _models(events: list[dict[str, Any]]) -> list[str]: + values: set[str] = set() + for event in events: + meta, _trace = _meta(event) + agent = meta.get("agent") if isinstance(meta.get("agent"), dict) else {} + model = str(agent.get("model") or "").strip() + if model: + values.add(model[:200]) + return sorted(values) + + +def enrich_agent_execution_payload(payload: dict[str, Any], raw_events: list[dict[str, Any]]) -> dict[str, Any]: + groups: dict[str, list[dict[str, Any]]] = {} + for events in _agent_groups(raw_events): + if not events: + continue + execution = _one_execution(events) + groups[str(execution.get("execution_id") or "")] = events + + for run in payload.get("executions") or []: + if not isinstance(run, dict): + continue + events = groups.get(str(run.get("execution_id") or ""), []) + if not events: + continue + effective = _effective_events(events) + counts = Counter(str(_meta(event)[0].get("operation") or "unknown") for event in effective) + caps = _capabilities(events) + sensors = sorted(_sensor_ids(events)) + run["observed_operation_counts"] = dict(sorted(counts.items())) + run["signal_capabilities"] = caps + run["telemetry_sources"] = sensors + run["usage_totals"] = _usage_totals(effective) + run["models_observed"] = _models(effective) + run["telemetry_depth"] = ( + "rich_native_events_plus_hooks" + if {"agent:claude-code-hook", "agent:claude-code-otel"}.issubset(set(sensors)) + else "rich_native_events" + if "agent:claude-code-otel" in sensors or _framework(events) in {"codex", "openai-agents-python"} + else "hooks_only" + if "agent:claude-code-hook" in sensors + else "provider_neutral_structural" + ) + + payload["capability_semantics"] = { + "observable": "the active adapter is designed to emit this structural signal; zero means none was observed in this evidence window", + "partial": "the active adapter can expose the signal only on some runtime paths or span shapes", + "not_observable": "the active adapter does not expose this signal; do not interpret absence as zero runtime activity", + "unknown": "the integration did not declare whether this signal is observable", + } + return payload + + +__all__ = ["enrich_agent_execution_payload"] From f201a26fc99ff41e9f2f0547fe231c5a02d8c9e8 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:24:20 +0200 Subject: [PATCH 06/36] Expose signal capability and rich run summaries --- server/agent_execution_trace_routes.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/server/agent_execution_trace_routes.py b/server/agent_execution_trace_routes.py index f2dbdcb7..c11ce89c 100644 --- a/server/agent_execution_trace_routes.py +++ b/server/agent_execution_trace_routes.py @@ -5,6 +5,7 @@ from fastapi import APIRouter, HTTPException, Request from .agent_execution_traces import agent_execution_traces +from .agent_observability import enrich_agent_execution_payload from .agent_read_auth import agent_read_authorized from .procedural_memory import load_recent_evidence @@ -41,6 +42,7 @@ def get_agent_execution_traces( limit=limit, max_events_per_execution=max_events_per_execution, ) + payload = enrich_agent_execution_payload(payload, raw) except (TypeError, ValueError) as exc: raise HTTPException(status_code=422, detail="invalid agent-execution-trace query") from exc return { From 9a04c1424864387361a7ddae8f076ecf27187fd0 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:27:40 +0200 Subject: [PATCH 07/36] Enrich Codex turn, usage and multi-agent telemetry --- shared/codex_otel_adapter.py | 148 +++++++++++++++++++++++++---------- 1 file changed, 106 insertions(+), 42 deletions(-) diff --git a/shared/codex_otel_adapter.py b/shared/codex_otel_adapter.py index f276e6cf..cd5fc849 100644 --- a/shared/codex_otel_adapter.py +++ b/shared/codex_otel_adapter.py @@ -2,10 +2,10 @@ """Translate Codex OTLP JSON into structural OpenWorkGraph agent evidence. -Codex can export diagnostic log records containing prompts, account identifiers, -tool arguments, tool output, and error strings. This module intentionally reads -only a small structural allowlist. It never copies OTLP record bodies or arbitrary -attributes into canonical evidence. +Codex telemetry can contain prompts, account identifiers, tool arguments/results, +inter-agent message content and error strings. This adapter only projects a small +structural allowlist and never copies record bodies or arbitrary attributes into +canonical evidence. """ from datetime import datetime, timezone @@ -19,6 +19,7 @@ "codex.tool_result", "codex.tool_decision", "codex.api_request", + "codex.agent_communication", }) _SAFE_LABEL = re.compile(r"^[A-Za-z][A-Za-z0-9_.:/-]{0,199}$") @@ -63,9 +64,10 @@ def _bool(value: Any) -> bool | None: def _int(value: Any) -> int | None: try: - return int(value) + number = int(value) except Exception: return None + return number if 0 <= number <= 1_000_000_000 else None def _duration_seconds(value: Any) -> float: @@ -112,8 +114,30 @@ def _hash_part(value: Any) -> str: return hashlib.sha256(raw.encode("utf-8")).hexdigest()[:24] +def _event_name(attrs: dict[str, Any]) -> str: + raw = _text(attrs.get("event.name"), 160) + if raw in _SUPPORTED_EVENTS: + return raw + + # Some Codex builds have exported a tracing call-site in event.name instead + # of the semantic name. Infer only from low-cardinality structural fields; + # never inspect record bodies, arguments, output or inter-agent content. + if attrs.get("communication_id") and attrs.get("kind") and attrs.get("state"): + return "codex.agent_communication" + if attrs.get("call_id") and attrs.get("decision") is not None and attrs.get("source") is not None: + return "codex.tool_decision" + if attrs.get("call_id") and attrs.get("tool_name") and attrs.get("success") is not None: + return "codex.tool_result" + if attrs.get("attempt") is not None and ( + "http.response.status_code" in attrs or "error.message" in attrs or "endpoint" in attrs + ): + return "codex.api_request" + if attrs.get("provider_name") and "approval_policy" in attrs and "sandbox_policy" in attrs: + return "codex.conversation_starts" + return "" + + def _event_id(conversation_id: str, event_name: str, attrs: dict[str, Any]) -> str: - """Stable across Codex log + trace copies without exposing native IDs.""" if event_name == "codex.conversation_starts": discriminator = "conversation-start" elif event_name == "codex.tool_result": @@ -134,6 +158,12 @@ def _event_id(conversation_id: str, event_name: str, attrs: dict[str, Any]) -> s _text(attrs.get("attempt"), 40), "" if request_hash else _text(attrs.get("event.timestamp"), 80), ]) + elif event_name == "codex.agent_communication": + discriminator = "agent-communication|" + "|".join([ + _hash_part(attrs.get("communication_id")), + _text(attrs.get("kind"), 40), + _text(attrs.get("state"), 40), + ]) else: discriminator = _text(attrs.get("event.timestamp"), 80) material = f"{conversation_id}\x1f{event_name}\x1f{discriminator}".encode("utf-8") @@ -157,6 +187,24 @@ def _tool_category(name: str, namespace: str = "") -> str: return "other" if name else "none" +def _merge_span_structure(event_attrs: dict[str, Any], span_attrs: dict[str, Any]) -> dict[str, Any]: + out = dict(event_attrs) + for key in ( + "conversation.id", + "thread.id", + "turn.id", + "model", + "gen_ai.request.model", + "gen_ai.usage.input_tokens", + "gen_ai.usage.output_tokens", + "gen_ai.usage.cache_read.input_tokens", + "codex.usage.total_tokens", + ): + if key not in out and key in span_attrs: + out[key] = span_attrs[key] + return out + + def _iter_records(payload: dict[str, Any], max_records: int) -> Iterator[tuple[dict[str, Any], dict[str, Any]]]: seen = 0 @@ -174,8 +222,6 @@ def _iter_records(payload: dict[str, Any], max_records: int) -> Iterator[tuple[d raise ValueError(f"Codex OTLP payload exceeds {max_records} records") yield record, _attrs(record.get("attributes")) - # Codex also emits trace-safe telemetry as span events. Read event attributes, - # never arbitrary span attributes, span names, or native event bodies. for resource_span in payload.get("resourceSpans") or []: if not isinstance(resource_span, dict): continue @@ -185,6 +231,7 @@ def _iter_records(payload: dict[str, Any], max_records: int) -> Iterator[tuple[d for span in scope_span.get("spans") or []: if not isinstance(span, dict): continue + span_attrs = _attrs(span.get("attributes")) for event in span.get("events") or []: if not isinstance(event, dict): continue @@ -194,9 +241,8 @@ def _iter_records(payload: dict[str, Any], max_records: int) -> Iterator[tuple[d synthetic = dict(event) if "timeUnixNano" not in synthetic and span.get("endTimeUnixNano"): synthetic["timeUnixNano"] = span.get("endTimeUnixNano") - yield synthetic, _attrs(event.get("attributes")) + yield synthetic, _merge_span_structure(_attrs(event.get("attributes")), span_attrs) - # Small simplified form for deterministic adapter tests/integrators. for record in payload.get("records") or []: if not isinstance(record, dict): continue @@ -206,6 +252,23 @@ def _iter_records(payload: dict[str, Any], max_records: int) -> Iterator[tuple[d yield record, _attrs(record.get("attributes")) +def _usage(attrs: dict[str, Any]) -> dict[str, int]: + mapping = { + "input_tokens": "gen_ai.usage.input_tokens", + "output_tokens": "gen_ai.usage.output_tokens", + "cached_input_tokens": "gen_ai.usage.cache_read.input_tokens", + "total_tokens": "codex.usage.total_tokens", + } + out: dict[str, int] = {} + for target, source in mapping.items(): + amount = _int(attrs.get(source)) + if amount is not None: + out[target] = amount + if "total_tokens" not in out and ("input_tokens" in out or "output_tokens" in out): + out["total_tokens"] = out.get("input_tokens", 0) + out.get("output_tokens", 0) + return out + + def _base( attrs: dict[str, Any], native: dict[str, Any], @@ -218,16 +281,18 @@ def _base( tool_category: str = "none", span_id: str = "", ) -> dict[str, Any] | None: - conversation_id = _text(attrs.get("conversation.id"), 240) + conversation_id = _text(attrs.get("conversation.id") or attrs.get("thread.id") or defaults.get("run_id"), 128) if not conversation_id: return None - model = _safe_label(attrs.get("model"), limit=200) + turn_id = _text(attrs.get("turn.id"), 128) + run_id = turn_id or conversation_id + model = _safe_label(attrs.get("model") or attrs.get("gen_ai.request.model"), limit=200) return { "event_id": _event_id(conversation_id, event_name, attrs), "observed_at": _observed_at(attrs, native), - "organization_id": _text(defaults.get("organization_id"), 240), - "actor_id": _text(defaults.get("actor_id"), 240), - "device_id": _text(defaults.get("device_id"), 240) or "codex-local", + "organization_id": _text(defaults.get("organization_id"), 128), + "actor_id": _text(defaults.get("actor_id"), 128), + "device_id": _text(defaults.get("device_id"), 128) or "codex-local", "sensor_id": "agent:codex-otel", "agent_name": _safe_label(defaults.get("agent_name"), default="Codex", limit=160), "provider": "openai", @@ -236,13 +301,14 @@ def _base( "operation": operation, "status": status, "observation_level": "native_trace", - "run_id": conversation_id, + "run_id": run_id, "trace_id": conversation_id, "span_id": span_id, - "workflow_id": _text(defaults.get("workflow_id"), 240), + "workflow_id": _text(defaults.get("workflow_id"), 128), "tool_name": tool_name, "tool_category": tool_category, "duration_seconds": _duration_seconds(attrs.get("duration_ms")), + "usage": _usage(attrs), } @@ -261,7 +327,7 @@ def codex_otel_to_agent_events( for native, attrs in _iter_records(payload, max_records): seen += 1 - event_name = _text(attrs.get("event.name"), 100) + event_name = _event_name(attrs) if event_name not in _SUPPORTED_EVENTS: ignored += 1 continue @@ -269,12 +335,8 @@ def codex_otel_to_agent_events( projected: dict[str, Any] | None = None if event_name == "codex.conversation_starts": projected = _base( - attrs, - native, - defaults=defaults, - event_name=event_name, - operation="run_started", - status="running", + attrs, native, defaults=defaults, event_name=event_name, + operation="run_started", status="running", ) elif event_name == "codex.tool_result": @@ -282,20 +344,15 @@ def codex_otel_to_agent_events( namespace = _safe_label(attrs.get("tool_namespace"), limit=160) success = _bool(attrs.get("success")) projected = _base( - attrs, - native, - defaults=defaults, - event_name=event_name, + attrs, native, defaults=defaults, event_name=event_name, operation="tool_call", status="success" if success is True else "error" if success is False else "unknown", tool_name=tool_name, tool_category=_tool_category(tool_name, namespace), - span_id=_text(attrs.get("call_id"), 240), + span_id=("call:" + _hash_part(attrs.get("call_id"))) if attrs.get("call_id") else "", ) elif event_name == "codex.tool_decision": - # Codex can resolve approvals through a user OR an automated reviewer. - # Only explicit user-sourced decisions are human approval evidence. source = _text(attrs.get("source"), 80).lower() if source != "user": ignored += 1 @@ -303,34 +360,41 @@ def codex_otel_to_agent_events( tool_name = _safe_label(attrs.get("tool_name"), default="unknown-tool", limit=160) namespace = _safe_label(attrs.get("tool_namespace"), limit=160) decision = _text(attrs.get("decision"), 80).lower() - denied = any(token in decision for token in ("deny", "denied", "reject", "cancel")) + denied = any(token in decision for token in ("deny", "denied", "reject", "cancel", "abort")) approved = any(token in decision for token in ("approve", "approved", "allow")) projected = _base( - attrs, - native, - defaults=defaults, - event_name=event_name, + attrs, native, defaults=defaults, event_name=event_name, operation="human_approval_received", status="denied" if denied else "success" if approved else "unknown", tool_name=tool_name, tool_category=_tool_category(tool_name, namespace), - span_id=_text(attrs.get("call_id"), 240), + span_id=("call:" + _hash_part(attrs.get("call_id"))) if attrs.get("call_id") else "", ) elif event_name == "codex.api_request": status_code = _int(attrs.get("http.response.status_code")) - # We only inspect whether an error exists; its text is never copied. has_error = bool(_text(attrs.get("error.message"), 1)) success = status_code is not None and 200 <= status_code <= 299 and not has_error projected = _base( - attrs, - native, - defaults=defaults, - event_name=event_name, + attrs, native, defaults=defaults, event_name=event_name, operation="model_call", status="success" if success else "error" if has_error or (status_code or 0) >= 400 else "unknown", ) + elif event_name == "codex.agent_communication": + # Inter-agent messages/results are communication, not necessarily a + # delegation. Count only a new spawn as a handoff. + if _text(attrs.get("state"), 40).lower() != "send" or _text(attrs.get("kind"), 40).lower() != "spawn": + ignored += 1 + continue + communication_id = _text(attrs.get("communication_id"), 128) + projected = _base( + attrs, native, defaults=defaults, event_name=event_name, + operation="handoff", status="success", + tool_name="agent:spawn", tool_category="other", + span_id=("communication:" + _hash_part(communication_id)) if communication_id else "", + ) + if projected is None: ignored += 1 continue From 1f42c11e30fc7926ffd686cf06f0e6c7ee4a7003 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:31:26 +0200 Subject: [PATCH 08/36] Add privacy-safe Claude Code OTel configuration helper --- adapters/claude_code_otel.py | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) create mode 100644 adapters/claude_code_otel.py diff --git a/adapters/claude_code_otel.py b/adapters/claude_code_otel.py new file mode 100644 index 00000000..0ffbba4b --- /dev/null +++ b/adapters/claude_code_otel.py @@ -0,0 +1,29 @@ +from __future__ import annotations + +"""Build the Claude Code settings env needed for structural OWG telemetry only.""" + + +def env_settings(*, token: str, base_url: str) -> dict[str, str]: + base = str(base_url or "").rstrip("/") + return { + "CLAUDE_CODE_ENABLE_TELEMETRY": "1", + "OTEL_LOGS_EXPORTER": "otlp", + "OTEL_EXPORTER_OTLP_LOGS_PROTOCOL": "http/json", + "OTEL_EXPORTER_OTLP_LOGS_ENDPOINT": f"{base}/agent-ingest/v1/claude-otel", + "OTEL_EXPORTER_OTLP_LOGS_HEADERS": f"Authorization=Bearer {token}", + # Claude leaves these content surfaces off by default. OWG writes the + # explicit zeroes as defense in depth; the ingest adapter independently + # allowlists structural attributes and would discard content anyway. + "OTEL_LOG_USER_PROMPTS": "0", + "OTEL_LOG_ASSISTANT_RESPONSES": "0", + "OTEL_LOG_TOOL_DETAILS": "0", + "OTEL_LOG_TOOL_CONTENT": "0", + "OTEL_LOG_RAW_API_BODIES": "0", + } + + +def settings_fragment(*, token: str, base_url: str) -> dict[str, dict[str, str]]: + return {"env": env_settings(token=token, base_url=base_url)} + + +__all__ = ["env_settings", "settings_fragment"] From aa364477aecf7158b10b957c4e72718f91d32c5f Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:31:59 +0200 Subject: [PATCH 09/36] Make Claude one-click setup manage rich telemetry safely --- server/agent_config_writer.py | 117 +++++++++++++++++++++++++++++----- 1 file changed, 102 insertions(+), 15 deletions(-) diff --git a/server/agent_config_writer.py b/server/agent_config_writer.py index 2f21f8e2..faed9749 100644 --- a/server/agent_config_writer.py +++ b/server/agent_config_writer.py @@ -70,7 +70,7 @@ def _atomic_write(path: Path, text: str) -> None: raise -# --- Claude Code: hooks in ~/.claude/settings.json --------------------------- +# --- Claude Code: hooks + logs-only OTel in ~/.claude/settings.json ---------- def _load_claude(path: Path) -> dict[str, Any]: if not path.exists(): @@ -87,6 +87,9 @@ def _load_claude(path: Path) -> dict[str, Any]: hooks = data.get("hooks") if hooks is not None and not isinstance(hooks, dict): raise ConfigConflict(f"'hooks' in {path} has an unexpected shape; use manual setup") + env = data.get("env") + if env is not None and not isinstance(env, dict): + raise ConfigConflict(f"'env' in {path} has an unexpected shape; use manual setup") return data @@ -125,24 +128,92 @@ def _strip_owg_hooks(data: dict[str, Any]) -> int: return removed -def claude_status() -> dict[str, Any]: - path = claude_settings_path() - try: - data = _load_claude(path) - except ConfigConflict as exc: - return {"configured": False, "path": str(path), "error": str(exc)} - configured = any( +def _claude_hooks_configured(data: dict[str, Any]) -> bool: + return any( _is_owg_handler(h) for groups in (data.get("hooks") or {}).values() if isinstance(groups, list) for group in groups if isinstance(group, dict) for h in (group.get("hooks") or []) if isinstance(group.get("hooks"), list) ) - return {"configured": configured, "path": str(path)} -def claude_connect(fragment_factory: Callable[[], dict]) -> dict[str, Any]: +def _claude_env_matches(data: dict[str, Any], managed_env: dict[str, str] | None) -> bool: + if not managed_env: + return True + existing = data.get("env") if isinstance(data.get("env"), dict) else {} + return all(str(existing.get(key) or "") == str(value) for key, value in managed_env.items()) + + +def _validate_claude_env_conflicts(data: dict[str, Any], managed_env: dict[str, str]) -> None: + existing = data.get("env") if isinstance(data.get("env"), dict) else {} + conflicts = [ + key for key, desired in managed_env.items() + if key in existing and str(existing.get(key)) != str(desired) + ] + if conflicts: + joined = ", ".join(sorted(conflicts)) + raise ConfigConflict( + f"Claude Code already has different telemetry settings for {joined}. " + "OpenWorkGraph will not overwrite them; use manual setup or an OTLP collector/tee." + ) + + +def _merge_claude_env(data: dict[str, Any], managed_env: dict[str, str]) -> None: + if not managed_env: + return + env = data.setdefault("env", {}) + if not isinstance(env, dict): + raise ConfigConflict("Claude Code 'env' has an unexpected shape; use manual setup") + for key, value in managed_env.items(): + env[key] = str(value) + + +def _strip_matching_claude_env(data: dict[str, Any], managed_env: dict[str, str]) -> int: + """Remove only OWG values that still exactly match what OWG would write. + + If the user changed a value after connecting, leave it untouched. This makes + Disconnect reversible without claiming ownership of unrelated telemetry keys. + """ + env = data.get("env") + if not isinstance(env, dict): + return 0 + removed = 0 + for key, expected in managed_env.items(): + if key in env and str(env.get(key)) == str(expected): + del env[key] + removed += 1 + if not env: + data.pop("env", None) + return removed + + +def claude_status(managed_env: dict[str, str] | None = None) -> dict[str, Any]: + path = claude_settings_path() + try: + data = _load_claude(path) + except ConfigConflict as exc: + return {"configured": False, "path": str(path), "error": str(exc)} + hooks_configured = _claude_hooks_configured(data) + telemetry_configured = _claude_env_matches(data, managed_env) + return { + "configured": hooks_configured and telemetry_configured, + "hooks_configured": hooks_configured, + "telemetry_configured": telemetry_configured, + "path": str(path), + } + + +def claude_connect( + fragment_factory: Callable[[], dict], + managed_env: dict[str, str] | None = None, +) -> dict[str, Any]: path = claude_settings_path() data = _load_claude(path) + desired_env = dict(managed_env or {}) + # Fail before changing hooks if a user already owns a conflicting telemetry + # destination or privacy flag. + _validate_claude_env_conflicts(data, desired_env) + # Replace, never duplicate: older OpenWorkGraph handlers (e.g. the pre-0.89 # command/args form) are removed before the current ones are added. _strip_owg_hooks(data) @@ -152,22 +223,38 @@ def claude_connect(fragment_factory: Callable[[], dict]) -> dict[str, Any]: if not isinstance(existing, list): raise ConfigConflict(f"hooks.{event} in {path} has an unexpected shape; use manual setup") existing.extend(groups) + _merge_claude_env(data, desired_env) + backup = _backup(path) _atomic_write(path, json.dumps(data, indent=2, ensure_ascii=False) + "\n") - return {"configured": True, "path": str(path), "backup": backup, - "note": "Takes effect in new Claude Code sessions."} + return { + "configured": True, + "hooks_configured": True, + "telemetry_configured": bool(desired_env), + "path": str(path), + "backup": backup, + "note": "Takes effect in new Claude Code sessions.", + } -def claude_disconnect() -> dict[str, Any]: +def claude_disconnect(managed_env: dict[str, str] | None = None) -> dict[str, Any]: path = claude_settings_path() if not path.exists(): return {"configured": False, "path": str(path), "backup": None} data = _load_claude(path) - if not _strip_owg_hooks(data): + removed_hooks = _strip_owg_hooks(data) + removed_env = _strip_matching_claude_env(data, dict(managed_env or {})) + if not (removed_hooks or removed_env): return {"configured": False, "path": str(path), "backup": None} backup = _backup(path) _atomic_write(path, json.dumps(data, indent=2, ensure_ascii=False) + "\n") - return {"configured": False, "path": str(path), "backup": backup} + return { + "configured": False, + "hooks_configured": False, + "telemetry_configured": False, + "path": str(path), + "backup": backup, + } # --- Codex: [otel] in ~/.codex/config.toml ------------------------------------ From 8208aa8d12db899e32262917fa6a837183bc1887 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:32:40 +0200 Subject: [PATCH 10/36] Wire rich Claude telemetry into one-click agent setup --- server/agent_dashboard_control_plane.py | 67 +++++++++++++++++++------ 1 file changed, 51 insertions(+), 16 deletions(-) diff --git a/server/agent_dashboard_control_plane.py b/server/agent_dashboard_control_plane.py index 6b74d1cb..793be78f 100644 --- a/server/agent_dashboard_control_plane.py +++ b/server/agent_dashboard_control_plane.py @@ -10,16 +10,15 @@ treats an integration as active only when structural telemetry is observed. """ -import json import shlex import sys -from pathlib import Path from typing import Any from fastapi import Request from fastapi.responses import HTMLResponse, JSONResponse, Response -from adapters.claude_code_hook import settings_fragment +from adapters.claude_code_hook import settings_fragment as claude_hook_settings +from adapters.claude_code_otel import env_settings as claude_otel_env from adapters.codex_config import config_snippet from server import agent_config_writer as writer from server.agent_auth import ensure_agent_ingest_token @@ -28,13 +27,28 @@ DASHBOARD_SCRIPT = ROOT / "dashboard" / "agent_control_plane.js" +OBSERVABILITY_SCRIPT = ROOT / "dashboard" / "agent_observability_v090.js" _SCRIPT_MARKER = '' +_OBSERVABILITY_MARKER = '' def _base_url(request: Request) -> str: return str(request.base_url).rstrip("/") +def _claude_env(request: Request) -> dict[str, str]: + return claude_otel_env( + token=ensure_agent_ingest_token(), + base_url=_base_url(request), + ) + + +def _claude_settings(request: Request) -> dict[str, Any]: + settings = dict(claude_hook_settings()) + settings["env"] = _claude_env(request) + return settings + + def agent_setup_payload(request: Request) -> dict[str, Any]: """Return reviewable setup material for the local dashboard owner. @@ -78,10 +92,12 @@ def agent_setup_payload(request: Request) -> dict[str, Any]: "integrations": { "claude_code": { "label": "Claude Code", - "method": "native_lifecycle_hooks", + "method": "native_hooks_plus_otel_logs", "command": f"cd {shlex.quote(str(ROOT))} && {shlex.quote(sys.executable)} -m adapters.claude_code_hook --print-settings", "one_click": True, - "settings": settings_fragment(), + "settings": _claude_settings(request), + "otel_endpoint": f"{base_url}/agent-ingest/v1/claude-otel", + "telemetry_depth": "rich_structural", "events": [ "SessionStart", "SessionEnd", @@ -92,9 +108,18 @@ def agent_setup_payload(request: Request) -> dict[str, Any]: "SubagentStart", "SubagentStop", "StopFailure", + "claude_code.user_prompt", + "claude_code.api_request", + "claude_code.api_error", + "claude_code.api_refusal", + "claude_code.tool_result", + "claude_code.tool_decision", + "claude_code.api_retries_exhausted", + "claude_code.subagent_completed", ], - "instructions": "Click Connect to add these hooks to ~/.claude/settings.json (other settings are preserved and a backup is written), or merge the hooks object manually into ~/.claude/settings.json or an intentional project-scoped .claude/settings.json.", + "instructions": "Click Connect to add OpenWorkGraph's hooks and logs-only OpenTelemetry settings to ~/.claude/settings.json. Other settings are preserved, conflicting existing telemetry is never overwritten, and a backup is written. Manual setup can merge the shown hooks and env objects instead.", "failure_mode": "fail_open_async", + "content_logging_enabled": False, }, "codex": { "label": "Codex", @@ -105,13 +130,15 @@ def agent_setup_payload(request: Request) -> dict[str, Any]: "endpoint": f"{base_url}/agent-ingest/v1/codex-otel", "instructions": "Click Connect to add a managed [otel] block to ~/.codex/config.toml (refused if you already have your own [otel] settings), or merge these keys manually into the existing [otel] section. Do not create a second [otel] table.", "logs_enabled_by_owg": False, + "telemetry_depth": "native_trace", }, "openai_agents": { "label": "OpenAI Agents SDK", "method": "additional_tracing_processor", "python": "from adapters.openai_agents import install_openai_agents_processor\n\ninstall_openai_agents_processor()", - "instructions": "Register OpenWorkGraph as an additional tracing processor in the agent application. Existing SDK tracing remains in place.", + "instructions": "Register OpenWorkGraph as an additional tracing processor in the agent application. Existing SDK tracing remains in place. OWG projects model spans, tools, handoffs, hierarchy, usage when the SDK exposes it, timings and errors without serializing span content.", "one_click": False, + "telemetry_depth": "native_trace", }, "otel": { "label": "Generic OpenTelemetry", @@ -119,7 +146,8 @@ def agent_setup_payload(request: Request) -> dict[str, Any]: "endpoint": otel_endpoint, "posix": posix_otel, "powershell": powershell_otel, - "instructions": "Use the signal-specific traces endpoint exactly as shown. OpenWorkGraph currently accepts OTLP/HTTP JSON here, not protobuf or gRPC.", + "instructions": "Use the signal-specific traces endpoint exactly as shown. OpenWorkGraph currently accepts OTLP/HTTP JSON here, not protobuf or gRPC. Portable GenAI model/tool spans are supported; provider-specific handoff or approval signals require a native adapter.", + "telemetry_depth": "portable_genai", }, "custom": { "label": "Custom structural agent", @@ -140,22 +168,22 @@ def get_agent_setup(request: Request) -> JSONResponse: _ONE_CLICK = { "claude_code": { - "status": writer.claude_status, - "connect": lambda request: writer.claude_connect(settings_fragment), - "disconnect": writer.claude_disconnect, + "status": lambda request: writer.claude_status(_claude_env(request)), + "connect": lambda request: writer.claude_connect(claude_hook_settings, _claude_env(request)), + "disconnect": lambda request: writer.claude_disconnect(_claude_env(request)), }, "codex": { - "status": writer.codex_status, + "status": lambda request: writer.codex_status(), "connect": lambda request: writer.codex_connect( config_snippet(token=ensure_agent_ingest_token(), base_url=_base_url(request)) ), - "disconnect": writer.codex_disconnect, + "disconnect": lambda request: writer.codex_disconnect(), }, } -def get_agent_config_status() -> JSONResponse: - status = {kind: ops["status"]() for kind, ops in _ONE_CLICK.items()} +def get_agent_config_status(request: Request) -> JSONResponse: + status = {kind: ops["status"](request) for kind, ops in _ONE_CLICK.items()} return JSONResponse({"integrations": status}, headers={"Cache-Control": "no-store"}) @@ -170,7 +198,7 @@ async def change_agent_config(request: Request) -> JSONResponse: if ops is None or action not in {"connect", "disconnect"}: return JSONResponse({"detail": "unsupported agent or action"}, status_code=400) try: - result = ops["connect"](request) if action == "connect" else ops["disconnect"]() + result = ops[action](request) except writer.ConfigConflict as exc: return JSONResponse({"detail": str(exc), "manual_setup_required": True}, status_code=409) except OSError: @@ -185,6 +213,10 @@ def agent_control_plane_script() -> Response: return Response(DASHBOARD_SCRIPT.read_text(encoding="utf-8"), media_type="application/javascript") +def agent_observability_script() -> Response: + return Response(OBSERVABILITY_SCRIPT.read_text(encoding="utf-8"), media_type="application/javascript") + + async def _inject_agent_control_plane(request: Request, call_next): response = await call_next(request) if request.method.upper() != "GET" or request.url.path != "/" or response.status_code != 200: @@ -202,6 +234,8 @@ async def _inject_agent_control_plane(request: Request, call_next): return response if _SCRIPT_MARKER not in text: text = text.replace("", _SCRIPT_MARKER + "\n") + if _OBSERVABILITY_MARKER not in text: + text = text.replace("", _OBSERVABILITY_MARKER + "\n") headers = dict(response.headers) headers.pop("content-length", None) return HTMLResponse(text, status_code=response.status_code, headers=headers) @@ -212,6 +246,7 @@ def _install() -> None: app.add_api_route("/v1/agent-config", get_agent_config_status, methods=["GET"]) app.add_api_route("/v1/agent-config", change_agent_config, methods=["POST"]) app.add_api_route("/agent-control-plane.js", agent_control_plane_script, methods=["GET"]) + app.add_api_route("/agent-observability-v090.js", agent_observability_script, methods=["GET"]) app.middleware("http")(_inject_agent_control_plane) From 4cca0dbe6c11dca613432bbc6d08a525c9346187 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:33:18 +0200 Subject: [PATCH 11/36] Make agent dashboard capability-aware instead of showing fake zeroes --- dashboard/agent_observability_v090.js | 180 ++++++++++++++++++++++++++ 1 file changed, 180 insertions(+) create mode 100644 dashboard/agent_observability_v090.js diff --git a/dashboard/agent_observability_v090.js b/dashboard/agent_observability_v090.js new file mode 100644 index 00000000..81a0c7b9 --- /dev/null +++ b/dashboard/agent_observability_v090.js @@ -0,0 +1,180 @@ +(() => { + let latest=null; + let loading=false; + let patchScheduled=false; + + const esc=value=>{ + if(typeof window.esc==='function')return window.esc(String(value??'')); + const node=document.createElement('div');node.textContent=String(value??'');return node.innerHTML; + }; + const fmt=value=>typeof window.fmtTime==='function'?window.fmtTime(value):`${Math.round(Number(value)||0)}s`; + + function count(run,name){ + const source=run?.observed_operation_counts||run?.operation_counts||{}; + return Number(source[name]||0); + } + + function capability(run,name){ + const value=run?.signal_capabilities?.[name]; + return value&&typeof value==='object'?value:{status:'unknown',basis:'adapter_capability_not_declared'}; + } + + function supported(run,name){ + return ['observable','partial'].includes(capability(run,name).status); + } + + function signalText(run,name,label){ + const cap=capability(run,name); + if(!['observable','partial'].includes(cap.status)){ + return `— ${esc(label)}`; + } + const qualifier=cap.status==='partial'?'≈ ':''; + return `${qualifier}${count(run,name)} ${esc(label)}`; + } + + function tokenText(run){ + const cap=capability(run,'token_usage'); + if(!['observable','partial'].includes(cap.status))return ''; + const usage=run?.usage_totals||{}; + const total=Number(usage.total_tokens||0); + if(!total&&!Object.keys(usage).length)return cap.status==='partial'?'tokens not reported on this run':'0 tokens observed'; + const input=Number(usage.input_tokens||0),output=Number(usage.output_tokens||0),cached=Number(usage.cached_input_tokens||0); + return `${total.toLocaleString()} tokens${input||output?` · ${input.toLocaleString()} in · ${output.toLocaleString()} out`:''}${cached?` · ${cached.toLocaleString()} cached`:''}`; + } + + function modelText(run){ + const models=Array.isArray(run?.models_observed)?run.models_observed.filter(Boolean):[]; + if(models.length)return `${models.map(esc).join(' · ')}`; + const cap=capability(run,'model_identity'); + return ['not_observable','unknown'].includes(cap.status)?'model identity not observable':''; + } + + function depthLabel(run){ + const value=String(run?.telemetry_depth||''); + const labels={ + rich_native_events_plus_hooks:'rich native + hooks', + rich_native_events:'rich native', + hooks_only:'hooks only', + provider_neutral_structural:'portable structural', + }; + return labels[value]||value.replaceAll('_',' ')||'structural telemetry'; + } + + function aggregate(runs,name){ + const observable=runs.filter(run=>supported(run,name)); + if(!observable.length)return {text:'—',title:'This signal is not observable with the integrations in this view.'}; + const total=observable.reduce((sum,run)=>sum+count(run,name),0); + const mixed=observable.length!==runs.length; + return {text:`${total}${mixed?'*':''}`,title:mixed?'Some runs use integrations that cannot observe this signal.':'Observed structural events.'}; + } + + function patchMetrics(runs){ + const mappings=[ + ['agentToolCount','tool_call'], + ['agentApprovalCount','human_approval_requested'], + ['agentHandoffCount','handoff'], + ]; + for(const [id,key] of mappings){ + const el=document.querySelector(`#${id}`);if(!el)continue; + const value=aggregate(runs,key);el.textContent=value.text;el.title=value.title; + } + } + + function groupRuns(runs){ + const grouped=new Map(); + for(const run of runs){ + const agent=run?.agent||{}; + const name=String(agent.name||'Agent'),provider=String(agent.provider||''),framework=String(agent.framework||''); + const key=`${name}\u0000${provider}\u0000${framework}`; + const row=grouped.get(key)||{name,provider,framework,runs:[],success:0,failures:0,duration:0,durationCount:0}; + row.runs.push(run); + if(run.outcome_status==='success')row.success+=1; + if(['error','denied','cancelled'].includes(String(run.outcome_status||'')))row.failures+=1; + const start=Date.parse(String(run.started_at||'')),end=Date.parse(String(run.ended_at||'')); + if(Number.isFinite(start)&&Number.isFinite(end)&&end>=start){row.duration+=(end-start)/1000;row.durationCount+=1;} + grouped.set(key,row); + } + return [...grouped.values()].sort((a,b)=>b.runs.length-a.runs.length); + } + + function groupedSignal(row,name){ + const value=aggregate(row.runs,name); + return `${esc(value.text)}`; + } + + function groupedTokens(row){ + const supportedRuns=row.runs.filter(run=>supported(run,'token_usage')); + if(!supportedRuns.length)return '—'; + const total=supportedRuns.reduce((sum,run)=>sum+Number(run?.usage_totals?.total_tokens||0),0); + return total?total.toLocaleString():'0'; + } + + function patchReports(runs){ + const host=document.querySelector('#agentReportTable'); + if(!host||host.querySelector('.owg-rich-agent-report'))return; + const rows=groupRuns(runs); + if(!rows.length)return; + host.innerHTML=`${rows.map(row=>``).join('')}
AgentRunsSuccessfulFailuresModel callsTool callsHandoffsApproval requestsTokensAvg duration
${esc(row.name)}
${esc([row.provider,row.framework].filter(Boolean).join(' · ')||'Provider/framework not reported')}
${row.runs.length}${row.success}${row.failures}${groupedSignal(row,'model_call')}${groupedSignal(row,'tool_call')}${groupedSignal(row,'handoff')}${groupedSignal(row,'human_approval_requested')}${groupedTokens(row)}${row.durationCount?fmt(row.duration/row.durationCount):'—'}
`; + } + + function patchRunRows(runs){ + const byId=new Map(runs.map(run=>[String(run.execution_id||''),run])); + const table=document.querySelector('#agentRunsTable table'); + if(!table||table.classList.contains('owg-capability-aware'))return; + for(const button of table.querySelectorAll('[data-agent-execution]')){ + const run=byId.get(String(button.dataset.agentExecution||'')); + const row=button.closest('tr');if(!run||!row)continue; + const cells=row.querySelectorAll('td');if(cells.length<6)continue; + cells[4].innerHTML=`
${signalText(run,'model_call','model')} · ${signalText(run,'tool_call','tools')} · ${signalText(run,'handoff','handoffs')}
${signalText(run,'human_approval_requested','approval requests')} · ${count(run,'error')} explicit errors
${modelText(run)?`
${modelText(run)}
`:''}${tokenText(run)?`
${tokenText(run)}
`:''}`; + const existing=cells[5].innerHTML; + cells[5].innerHTML=`${existing}
${esc(depthLabel(run))}
`; + } + table.classList.add('owg-capability-aware'); + } + + function patchNote(){ + const note=document.querySelector('#agentCoverageNote'); + if(!note)return; + const base=String(note.textContent||'').replace(/\s*0 = observed zero\..*$/,'').trim(); + note.textContent=`${base}${base?' ':''}0 = observed zero. — = this integration cannot observe that signal; it does not mean zero runtime activity.`; + } + + function patch(){ + patchScheduled=false; + if(!latest)return; + const runs=Array.isArray(latest.executions)?latest.executions:[]; + patchMetrics(runs);patchReports(runs);patchRunRows(runs);patchNote(); + } + + function schedulePatch(){ + if(patchScheduled)return;patchScheduled=true;requestAnimationFrame(patch); + } + + async function refresh(force=false){ + if(loading)return; + const active=document.querySelector('[role="tab"][aria-selected="true"]')?.dataset.tab; + if(!force&&active!=='agents')return; + loading=true; + try{ + await window.__owgAuthReady; + const response=await fetch('/v1/agent-execution-traces?limit=50&evidence_limit=25000&max_events_per_execution=100',{cache:'no-store'}); + if(response.ok){latest=await response.json();schedulePatch();} + }finally{loading=false;} + } + + function install(){ + document.querySelector('#tab-agents')?.addEventListener('click',()=>setTimeout(()=>refresh(true),0)); + document.querySelector('#agentRefreshButton')?.addEventListener('click',()=>setTimeout(()=>refresh(true),30)); + const observer=new MutationObserver(()=>{ + if(!latest)return; + const report=document.querySelector('#agentReportTable'); + const table=document.querySelector('#agentRunsTable table'); + if((report&&!report.querySelector('.owg-rich-agent-report'))||(table&&!table.classList.contains('owg-capability-aware')))schedulePatch(); + }); + observer.observe(document.body,{childList:true,subtree:true}); + refresh(false); + setInterval(()=>{if(!document.hidden)refresh(false);},5000); + } + + if(document.readyState==='loading')document.addEventListener('DOMContentLoaded',install);else install(); +})(); From 1bce0053ca2f779a159a787d4a15ee3867e55eb7 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:35:54 +0200 Subject: [PATCH 12/36] Enrich OpenAI Agents response usage and telemetry identity --- adapters/openai_agents.py | 42 ++++++++++++++++++++++++++++++++------- 1 file changed, 35 insertions(+), 7 deletions(-) diff --git a/adapters/openai_agents.py b/adapters/openai_agents.py index a8dbb496..fa59a88a 100644 --- a/adapters/openai_agents.py +++ b/adapters/openai_agents.py @@ -28,6 +28,7 @@ class _TracingProcessor: # type: ignore[no-redef] _SAFE_LABEL = re.compile(r"^[A-Za-z][A-Za-z0-9_.:/-]{0,199}$") _USAGE_KEYS = frozenset({"input_tokens", "output_tokens", "cached_input_tokens", "total_tokens"}) +_SENSOR_ID = "agent:openai-agents" def _now_iso() -> str: @@ -86,21 +87,34 @@ def _duration_seconds(span: Any) -> float: def _usage(value: Any) -> dict[str, int]: - if not isinstance(value, dict): + """Read only numeric token counters from dict- or object-shaped SDK usage.""" + if value is None: return {} out: dict[str, int] = {} for key in _USAGE_KEYS: - if key not in value: + if isinstance(value, dict): + raw = value.get(key) + else: + raw = getattr(value, key, None) + if raw is None: continue try: - amount = int(value[key]) + amount = int(raw) except Exception: continue if 0 <= amount <= 1_000_000_000: out[key] = amount + if "total_tokens" not in out and ("input_tokens" in out or "output_tokens" in out): + out["total_tokens"] = out.get("input_tokens", 0) + out.get("output_tokens", 0) return out +def _response_model(data: Any) -> str: + """Read only the response model identifier, never response content.""" + response = getattr(data, "response", None) + return _safe_label(getattr(response, "model", ""), limit=200) if response is not None else "" + + def _tool_category(name: str, *, is_mcp: bool = False) -> str: if is_mcp: return "mcp" @@ -212,6 +226,7 @@ def on_trace_start(self, trace: Any) -> None: self._emit({ "event_id": _event_id("trace-start", raw_trace), "observed_at": _now_iso(), + "sensor_id": _SENSOR_ID, "agent_name": "OpenAI-Agents-SDK", "provider": "openai", "framework": "openai-agents-python", @@ -232,6 +247,7 @@ def on_trace_end(self, trace: Any) -> None: self._emit({ "event_id": _event_id("trace-end", raw_trace), "observed_at": _now_iso(), + "sensor_id": _SENSOR_ID, "agent_name": "OpenAI-Agents-SDK", "provider": "openai", "framework": "openai-agents-python", @@ -271,6 +287,7 @@ def on_span_end(self, span: Any) -> None: base: dict[str, Any] = { "event_id": _event_id(span_type, raw_trace, raw_span), "observed_at": observed_at, + "sensor_id": _SENSOR_ID, "agent_name": agent_name, "provider": "openai", "framework": "openai-agents-python", @@ -285,10 +302,21 @@ def on_span_end(self, span: Any) -> None: if span_type in {"generation", "response", "transcription", "speech"}: base["operation"] = "model_call" - if span_type != "response": - base["model"] = _safe_label(getattr(data, "model", ""), limit=200) - if span_type == "generation": - base["usage"] = _usage(getattr(data, "usage", None)) + if span_type == "response": + model = _response_model(data) + if model: + base["model"] = model + usage = _usage(getattr(data, "usage", None)) + if usage: + base["usage"] = usage + else: + model = _safe_label(getattr(data, "model", ""), limit=200) + if model: + base["model"] = model + if span_type == "generation": + usage = _usage(getattr(data, "usage", None)) + if usage: + base["usage"] = usage self._emit(base) return From 36f9e1d39c8661d4d017250d300a731baf90e7e4 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:36:50 +0200 Subject: [PATCH 13/36] Update Claude hook tests for turn correlation and handoff edges --- tests/test_claude_code_adapter_v060.py | 42 ++++++++++++++++++++++---- 1 file changed, 36 insertions(+), 6 deletions(-) diff --git a/tests/test_claude_code_adapter_v060.py b/tests/test_claude_code_adapter_v060.py index e03c2866..852c9464 100644 --- a/tests/test_claude_code_adapter_v060.py +++ b/tests/test_claude_code_adapter_v060.py @@ -59,8 +59,9 @@ def test_session_subagent_permission_and_failure_mappings_are_structural(): ({"session_id": "s", "hook_event_name": "SessionStart"}, "run_started", "running"), ({"session_id": "s", "hook_event_name": "SessionEnd"}, "run_finished", "unknown"), ({"session_id": "s", "hook_event_name": "PermissionRequest", "tool_use_id": "t", "tool_name": "Edit"}, "human_approval_requested", "running"), - # PermissionDenied is an auto-mode policy denial in current Claude Code, - # so it must not be mislabeled as a human decision. + # PermissionDenied can be an automatic policy denial, so it must not be + # mislabeled as a human decision. The richer OTel decision event carries + # the decision source when connected. ({"session_id": "s", "hook_event_name": "PermissionDenied", "tool_use_id": "t", "tool_name": "Edit"}, "error", "denied"), ({"session_id": "s", "hook_event_name": "StopFailure", "prompt_id": "p", "error": "private failure details"}, "error", "error"), ] @@ -70,13 +71,27 @@ def test_session_subagent_permission_and_failure_mappings_are_structural(): assert event["status"] == status assert "private failure details" not in json.dumps(event) - [started] = claude_hook_to_agent_events( - {"session_id": "s", "hook_event_name": "SubagentStart", "agent_id": "child", "agent_type": "Explore"}, + started_events = claude_hook_to_agent_events( + { + "session_id": "s", + "prompt_id": "prompt-turn-1", + "hook_event_name": "SubagentStart", + "agent_id": "child", + "agent_type": "Explore", + }, observed_at="2026-09-25T00:00:00Z", ) + assert [event["operation"] for event in started_events] == ["handoff", "run_started"] + handoff, started = started_events + assert handoff["run_id"] == started["run_id"] == "prompt-turn-1" + assert handoff["agent_name"] == "Claude Code" + assert handoff["tool_name"] == "subagent:Explore" + assert started["agent_name"] == "Claude Code/Explore" + [stopped] = claude_hook_to_agent_events( { "session_id": "s", + "prompt_id": "prompt-turn-1", "hook_event_name": "SubagentStop", "agent_id": "child", "agent_type": "Explore", @@ -85,13 +100,28 @@ def test_session_subagent_permission_and_failure_mappings_are_structural(): }, observed_at="2026-09-25T00:00:05Z", ) - assert started["run_id"] == stopped["run_id"] == "s:child" - assert started["operation"] == "run_started" + assert stopped["run_id"] == "prompt-turn-1" assert stopped["operation"] == "run_finished" + assert stopped["agent_name"] == "Claude Code/Explore" assert "do not store this" not in json.dumps(stopped) assert "/private/child.jsonl" not in json.dumps(stopped) +def test_prompt_id_correlates_tool_and_permission_hooks_to_one_turn(): + base = {"session_id": "session-1", "prompt_id": "prompt-1"} + [tool] = claude_hook_to_agent_events( + {**base, "hook_event_name": "PostToolUse", "tool_use_id": "t1", "tool_name": "Read"}, + observed_at="2026-09-25T00:00:00Z", + ) + [approval] = claude_hook_to_agent_events( + {**base, "hook_event_name": "PermissionRequest", "tool_use_id": "t2", "tool_name": "Edit"}, + observed_at="2026-09-25T00:00:01Z", + ) + assert tool["run_id"] == approval["run_id"] == "prompt-1" + assert tool["trace_id"] == approval["trace_id"] == "session-1" + assert tool["sensor_id"] == approval["sensor_id"] == "agent:claude-code-hook" + + def test_retry_of_same_hook_event_is_idempotent(): payload = { "session_id": "s", From bd26f0c3f1178a28c0fad0c5458982769191a8b2 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:37:43 +0200 Subject: [PATCH 14/36] Test rich Claude, Codex, OpenAI and capability-aware observability --- tests/test_rich_agent_observability_v090.py | 297 ++++++++++++++++++++ 1 file changed, 297 insertions(+) create mode 100644 tests/test_rich_agent_observability_v090.py diff --git a/tests/test_rich_agent_observability_v090.py b/tests/test_rich_agent_observability_v090.py new file mode 100644 index 00000000..dcf45e38 --- /dev/null +++ b/tests/test_rich_agent_observability_v090.py @@ -0,0 +1,297 @@ +from __future__ import annotations + +import json + +from adapters.openai_agents import OpenWorkGraphTracingProcessor +from server.agent_execution_traces import agent_execution_traces +from server.agent_observability import enrich_agent_execution_payload +from shared.agent_evidence import agent_event_to_evidence +from shared.claude_code_adapter import claude_hook_to_agent_events +from shared.claude_otel_adapter import claude_otel_to_agent_events +from shared.codex_otel_adapter import codex_otel_to_agent_events + + +def _claude_record(name: str, **attrs): + return {"attributes": {"event.name": name, "session.id": "session-1", "prompt.id": "prompt-1", **attrs}} + + +def test_claude_otel_projects_rich_structure_and_never_content(): + payload = { + "records": [ + _claude_record("claude_code.user_prompt", **{"prompt_length": 900, "prompt": "SUPERSECRET"}), + _claude_record( + "claude_code.api_request", + model="claude-sonnet-5", + input_tokens=120, + output_tokens=30, + cache_read_tokens=50, + duration_ms=1250, + request_id="private-request-id", + response="SUPERSECRET response", + ), + _claude_record( + "claude_code.tool_result", + tool_name="Bash", + tool_use_id="private-tool-id", + success=True, + duration_ms=400, + tool_parameters="SUPERSECRET args", + tool_result="patient@example.com", + ), + _claude_record( + "claude_code.tool_decision", + tool_name="Edit", + tool_use_id="private-tool-approval", + source="user_temporary", + decision="accept", + ), + _claude_record( + "claude_code.tool_decision", + tool_name="Read", + tool_use_id="automatic", + source="config", + decision="accept", + ), + _claude_record("claude_code.subagent_completed", agent_type="Explore", duration_ms=2000), + _claude_record("claude_code.api_retries_exhausted", **{"error.message": "SUPERSECRET retry body"}), + ] + } + events, stats = claude_otel_to_agent_events(payload) + assert stats["records_seen"] == 7 + assert [event["operation"] for event in events] == [ + "run_started", + "model_call", + "tool_call", + "human_approval_received", + "handoff", + "error", + ] + model = next(event for event in events if event["operation"] == "model_call") + assert model["model"] == "claude-sonnet-5" + assert model["usage"] == { + "input_tokens": 120, + "output_tokens": 30, + "cached_input_tokens": 50, + "total_tokens": 150, + } + tool = next(event for event in events if event["operation"] == "tool_call") + assert tool["tool_name"] == "Bash" + assert tool["tool_category"] == "shell" + assert all(event["run_id"] == "prompt-1" for event in events) + assert all(event["sensor_id"] == "agent:claude-code-otel" for event in events) + + canonical = [agent_event_to_evidence(event) for event in events] + serialized = json.dumps(canonical) + for forbidden in ( + "SUPERSECRET", + "patient@example.com", + "private-request-id", + "private-tool-id", + "private-tool-approval", + "tool_parameters", + "tool_result", + ): + assert forbidden not in serialized + + +def test_claude_hooks_only_mark_model_and_tokens_not_observable(): + [tool] = claude_hook_to_agent_events( + { + "session_id": "session-1", + "prompt_id": "prompt-1", + "hook_event_name": "PostToolUse", + "tool_use_id": "tool-1", + "tool_name": "Read", + }, + observed_at="2026-09-27T10:00:00Z", + ) + raw = [agent_event_to_evidence(tool)] + payload = enrich_agent_execution_payload(agent_execution_traces(raw), raw) + [run] = payload["executions"] + assert run["observed_operation_counts"]["tool_call"] == 1 + assert run["signal_capabilities"]["tool_call"]["status"] == "observable" + assert run["signal_capabilities"]["model_call"]["status"] == "not_observable" + assert run["signal_capabilities"]["token_usage"]["status"] == "not_observable" + assert run["telemetry_depth"] == "hooks_only" + + +def test_claude_otel_plus_hooks_deduplicates_tool_count_and_exposes_rich_signals(): + [hook_tool] = claude_hook_to_agent_events( + { + "session_id": "session-1", + "prompt_id": "prompt-1", + "hook_event_name": "PostToolUse", + "tool_use_id": "tool-1", + "tool_name": "Read", + }, + observed_at="2026-09-27T10:00:01Z", + ) + otel, _ = claude_otel_to_agent_events({ + "records": [ + _claude_record("claude_code.user_prompt"), + _claude_record( + "claude_code.api_request", + model="claude-sonnet-5", + input_tokens=10, + output_tokens=5, + request_id="request-1", + ), + _claude_record( + "claude_code.tool_result", + tool_name="Read", + tool_use_id="tool-1", + success=True, + ), + ] + }) + raw = [agent_event_to_evidence(event) for event in [*otel, hook_tool]] + payload = enrich_agent_execution_payload(agent_execution_traces(raw), raw) + [run] = payload["executions"] + assert run["observed_operation_counts"]["tool_call"] == 1 + assert run["observed_operation_counts"]["model_call"] == 1 + assert run["usage_totals"]["total_tokens"] == 15 + assert run["models_observed"] == ["claude-sonnet-5"] + assert run["signal_capabilities"]["model_call"]["status"] == "observable" + assert run["signal_capabilities"]["human_approval_requested"]["status"] == "observable" + assert run["telemetry_depth"] == "rich_native_events_plus_hooks" + + +def test_codex_turn_usage_and_spawn_handoff_are_structural(): + payload = { + "records": [ + { + "attributes": { + "event.name": "codex.api_request", + "conversation.id": "conversation-private", + "turn.id": "turn-7", + "model": "gpt-5.6-codex", + "attempt": 1, + "http.response.status_code": 200, + "gen_ai.usage.input_tokens": 80, + "gen_ai.usage.output_tokens": 20, + "codex.usage.total_tokens": 100, + "event.timestamp": "2026-09-27T10:00:00Z", + "prompt": "SUPERSECRET", + } + }, + { + "attributes": { + "event.name": "codex.agent_communication", + "conversation.id": "conversation-private", + "turn.id": "turn-7", + "communication_id": "private-communication-id", + "kind": "spawn", + "state": "send", + "agents": "SUPERSECRET names", + "event.timestamp": "2026-09-27T10:00:01Z", + } + }, + { + "attributes": { + "event.name": "codex.agent_communication", + "conversation.id": "conversation-private", + "turn.id": "turn-7", + "communication_id": "private-message-id", + "kind": "message", + "state": "send", + "message": "patient@example.com SUPERSECRET", + } + }, + ] + } + events, stats = codex_otel_to_agent_events(payload) + assert stats["records_seen"] == 3 + assert [event["operation"] for event in events] == ["model_call", "handoff"] + model = events[0] + assert model["run_id"] == "turn-7" + assert model["model"] == "gpt-5.6-codex" + assert model["usage"]["total_tokens"] == 100 + assert events[1]["tool_name"] == "agent:spawn" + assert all(event["sensor_id"] == "agent:codex-otel" for event in events) + serialized = json.dumps([agent_event_to_evidence(event) for event in events]) + assert "SUPERSECRET" not in serialized + assert "patient@example.com" not in serialized + assert "private-communication-id" not in serialized + + +def test_codex_can_infer_semantic_event_from_structural_fields_when_event_name_is_callsite(): + events, _ = codex_otel_to_agent_events({ + "records": [{ + "attributes": { + "event.name": "codex_otel::event_manager::log_event", + "conversation.id": "conversation-1", + "call_id": "private-call", + "tool_name": "Bash", + "success": True, + } + }] + }) + assert len(events) == 1 + assert events[0]["operation"] == "tool_call" + assert events[0]["tool_name"] == "Bash" + + +class _CaptureSink: + def __init__(self): + self.events: list[dict] = [] + + def emit(self, event: dict) -> bool: + self.events.append(dict(event)) + return True + + def force_flush(self, *, timeout: float = 2.0) -> bool: + return True + + def shutdown(self, *, timeout: float = 2.0) -> None: + return None + + +class _UsageObject: + input_tokens = 31 + output_tokens = 9 + cached_input_tokens = 4 + total_tokens = 40 + secret_breakdown = "SUPERSECRET" + + +class _ResponseObject: + model = "gpt-5.6" + output = "SUPERSECRET response" + + +class _Data: + type = "response" + usage = _UsageObject() + response = _ResponseObject() + input = "patient@example.com SUPERSECRET" + + +class _Span: + span_id = "span-private" + trace_id = "trace-private" + parent_id = None + span_data = _Data() + error = None + started_at = "2026-09-27T10:00:00Z" + ended_at = "2026-09-27T10:00:02Z" + + +def test_openai_agents_response_span_emits_model_usage_without_response_content(): + sink = _CaptureSink() + processor = OpenWorkGraphTracingProcessor(sink=sink) + span = _Span() + processor.on_span_start(span) + processor.on_span_end(span) + [event] = sink.events + assert event["operation"] == "model_call" + assert event["sensor_id"] == "agent:openai-agents" + assert event["model"] == "gpt-5.6" + assert event["usage"] == { + "input_tokens": 31, + "output_tokens": 9, + "cached_input_tokens": 4, + "total_tokens": 40, + } + serialized = json.dumps(event) + assert "SUPERSECRET" not in serialized + assert "patient@example.com" not in serialized From b9134081a6641bef7ce0a745b7a9d6727644805f Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:38:10 +0200 Subject: [PATCH 15/36] Test safe rich Claude one-click telemetry configuration --- tests/test_agent_config_rich_v090.py | 94 ++++++++++++++++++++++++++++ 1 file changed, 94 insertions(+) create mode 100644 tests/test_agent_config_rich_v090.py diff --git a/tests/test_agent_config_rich_v090.py b/tests/test_agent_config_rich_v090.py new file mode 100644 index 00000000..c7b746d3 --- /dev/null +++ b/tests/test_agent_config_rich_v090.py @@ -0,0 +1,94 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from adapters.claude_code_hook import settings_fragment +from adapters.claude_code_otel import env_settings +from server import agent_config_writer as writer + + +@pytest.fixture +def claude_file(tmp_path, monkeypatch): + path = tmp_path / ".claude" / "settings.json" + monkeypatch.setenv("OWG_CLAUDE_SETTINGS_PATH", str(path)) + return path + + +def _managed_env() -> dict[str, str]: + return env_settings(token="write-only-token", base_url="http://127.0.0.1:8787") + + +def test_rich_claude_connect_adds_hooks_and_logs_only_otel_preserving_unrelated_env(claude_file): + claude_file.parent.mkdir(parents=True) + claude_file.write_text(json.dumps({ + "model": "opus", + "env": {"MY_COMPANY_SETTING": "keep-me"}, + })) + managed = _managed_env() + result = writer.claude_connect(settings_fragment, managed) + data = json.loads(claude_file.read_text()) + + assert result["configured"] is True + assert result["hooks_configured"] is True + assert result["telemetry_configured"] is True + assert data["env"]["MY_COMPANY_SETTING"] == "keep-me" + for key, value in managed.items(): + assert data["env"][key] == value + assert data["env"]["OTEL_LOGS_EXPORTER"] == "otlp" + assert data["env"]["OTEL_EXPORTER_OTLP_LOGS_PROTOCOL"] == "http/json" + assert data["env"]["OTEL_EXPORTER_OTLP_LOGS_ENDPOINT"].endswith("/agent-ingest/v1/claude-otel") + assert data["env"]["OTEL_LOG_USER_PROMPTS"] == "0" + assert data["env"]["OTEL_LOG_ASSISTANT_RESPONSES"] == "0" + assert data["env"]["OTEL_LOG_TOOL_DETAILS"] == "0" + assert data["env"]["OTEL_LOG_TOOL_CONTENT"] == "0" + assert data["env"]["OTEL_LOG_RAW_API_BODIES"] == "0" + status = writer.claude_status(managed) + assert status["configured"] is True + assert status["hooks_configured"] is True + assert status["telemetry_configured"] is True + + +def test_rich_claude_connect_refuses_to_overwrite_foreign_telemetry(claude_file): + original = { + "env": { + "OTEL_EXPORTER_OTLP_LOGS_ENDPOINT": "https://company-collector.example/v1/logs", + "COMPANY": "keep", + } + } + claude_file.parent.mkdir(parents=True) + claude_file.write_text(json.dumps(original)) + + with pytest.raises(writer.ConfigConflict): + writer.claude_connect(settings_fragment, _managed_env()) + + assert json.loads(claude_file.read_text()) == original + assert not list(claude_file.parent.glob("settings.json.owg-backup-*")) + + +def test_rich_claude_disconnect_removes_only_values_still_owned_by_owg(claude_file): + managed = _managed_env() + claude_file.parent.mkdir(parents=True) + claude_file.write_text(json.dumps({"env": {"COMPANY": "keep"}})) + writer.claude_connect(settings_fragment, managed) + + # The user changes one previously OWG-managed setting after setup. Disconnect + # must not delete or rewrite that value because OWG no longer owns it. + data = json.loads(claude_file.read_text()) + data["env"]["OTEL_LOGS_EXPORTER"] = "console" + claude_file.write_text(json.dumps(data)) + + result = writer.claude_disconnect(managed) + after = json.loads(claude_file.read_text()) + assert result["configured"] is False + assert after["env"] == {"COMPANY": "keep", "OTEL_LOGS_EXPORTER": "console"} + assert "hooks" not in after + + +def test_claude_manual_fragment_contains_no_content_collection_opt_in(): + managed = _managed_env() + forbidden_truthy = [key for key in managed if key.startswith("OTEL_LOG_") and managed[key] not in {"0", "false", "False"}] + assert forbidden_truthy == [] + assert "OTEL_TRACES_EXPORTER" not in managed # beta detailed traces are not required for v0.90 From 8463348cc734930d5290768f07d7e6c68608e932 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:38:17 +0200 Subject: [PATCH 16/36] Test capability-aware agent dashboard semantics --- tests/js/agent_observability_v090.test.mjs | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 tests/js/agent_observability_v090.test.mjs diff --git a/tests/js/agent_observability_v090.test.mjs b/tests/js/agent_observability_v090.test.mjs new file mode 100644 index 00000000..222e31b4 --- /dev/null +++ b/tests/js/agent_observability_v090.test.mjs @@ -0,0 +1,19 @@ +import fs from 'node:fs'; +import vm from 'node:vm'; +import assert from 'node:assert/strict'; + +const source = fs.readFileSync(new URL('../../dashboard/agent_observability_v090.js', import.meta.url), 'utf8'); + +assert.match(source, /signal_capabilities/); +assert.match(source, /observed_operation_counts/); +assert.match(source, /usage_totals/); +assert.match(source, /models_observed/); +assert.match(source, /0 = observed zero/); +assert.match(source, /— = this integration cannot observe that signal/); +assert.match(source, /model_call/); +assert.match(source, /human_approval_requested/); + +// Syntax-only compile. The script is an IIFE and expects a browser DOM, so do +// not execute it in Node; compilation still catches malformed template strings +// and accidental syntax regressions. +new vm.Script(source, {filename: 'agent_observability_v090.js'}); From 179d1c6e194d5cec3345645ced1996ee9b91922c Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:38:42 +0200 Subject: [PATCH 17/36] Bump rich agent observability release to v0.90.0 --- VERSION | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/VERSION b/VERSION index 5aee1345..ae02209b 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -0.89.0 +0.90.0 From fe295df3b5cc71c9ade1411e8b53375621a41327 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:38:50 +0200 Subject: [PATCH 18/36] Bump package version to v0.90.0 --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 61cf8337..b69b5346 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "workflow-observer" -version = "0.89.0" +version = "0.90.0" description = "Local-first work evidence, self-hosted organizational context gateway, REST API, and MCP access." requires-python = ">=3.11" license = {file = "LICENSE"} From 3dabe59120ce96a1c5f74b521daf7ef442b4f913 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:39:02 +0200 Subject: [PATCH 19/36] Bump MCPB version to v0.90.0 --- mcpb/manifest.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mcpb/manifest.json b/mcpb/manifest.json index f763aaad..96affe8a 100644 --- a/mcpb/manifest.json +++ b/mcpb/manifest.json @@ -2,7 +2,7 @@ "manifest_version": "0.3", "name": "openworkgraph-local", "display_name": "OpenWorkGraph", - "version": "0.89.0", + "version": "0.90.0", "description": "Connect Claude Desktop to the compact local OpenWorkGraph context surface.", "long_description": "Uses the OpenWorkGraph installation already running on this computer. New connections expose a compact read-oriented tool surface while the legacy 24-tool stdio entrypoint remains available for existing configurations. Workflow evidence remains in the local OpenWorkGraph store until Claude requests it through MCP. AI access can be turned off instantly from the OpenWorkGraph dashboard.", "author": { From abee53643ed9138e62dbb21bc53c6249204be6d0 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:39:40 +0200 Subject: [PATCH 20/36] Document rich capability-aware native agent observation --- docs/NATIVE_AGENT_ADAPTERS.md | 207 ++++++++++++++++------------------ 1 file changed, 95 insertions(+), 112 deletions(-) diff --git a/docs/NATIVE_AGENT_ADAPTERS.md b/docs/NATIVE_AGENT_ADAPTERS.md index 796343a7..2c212883 100644 --- a/docs/NATIVE_AGENT_ADAPTERS.md +++ b/docs/NATIVE_AGENT_ADAPTERS.md @@ -1,42 +1,40 @@ # Native agent adapters -OpenWorkGraph can observe supported agent runtimes through their native lifecycle/telemetry surfaces and project only structural workflow evidence into the same canonical event store used by desktop and browser capture. +OpenWorkGraph can observe supported agent runtimes through their native lifecycle and telemetry surfaces, then project only structural execution evidence into the canonical event store used by desktop and browser capture. -The native adapters are **optional**. Existing OpenWorkGraph behavior is unchanged until an agent runtime is explicitly configured to send events. +Native adapters are **optional**. Existing OpenWorkGraph behavior is unchanged until an agent runtime is explicitly configured to send events. ## Privacy model -Native agent payloads are treated as untrusted, potentially content-bearing input. +Native agent payloads are treated as untrusted, potentially content-bearing input. OpenWorkGraph stores only structural facts such as runtime identity, run/turn linkage, tool name/category, model identifier, token counters, status, duration, approval provenance when available, and parent/child execution relationships. -OpenWorkGraph stores structural facts such as: +The adapters deliberately do **not** persist prompts, assistant messages, reasoning/chain-of-thought, tool arguments, tool results/output, transcript paths, working-directory paths, account identifiers, arbitrary OpenTelemetry attributes, arbitrary span names, or raw log bodies. -- agent/runtime identity; -- run/conversation ID; -- tool name and coarse tool category; -- success/failure status; -- duration; -- model identifier when safely available; -- approval outcome only when native evidence establishes its provenance; -- trace/run relationships. +Native integrations reuse the dedicated write-only `.agent_ingest_token`. That token can submit evidence but cannot read work history, exports, summaries, agent reports, or MCP context. -The adapters deliberately do **not** persist: +## Capability semantics -- prompts or assistant messages; -- chain-of-thought/reasoning content; -- tool arguments; -- tool results/output; -- transcript paths; -- working-directory paths; -- account email/account IDs; -- arbitrary OpenTelemetry attributes or log bodies. +Agent reports separate **observed count** from **adapter capability**: -Native integrations reuse the dedicated write-only `.agent_ingest_token`. That token can submit agent evidence but cannot read `/v1/events`, exports, summaries, `/v1/agent-workflows`, or agent execution reports. +- `observable`: the active adapter is designed to expose the signal; `0` means zero matching events were observed in this evidence window. +- `partial`: the signal is exposed only on some runtime/span paths. +- `not_observable`: the integration does not expose that signal; the dashboard renders `—`, never a misleading zero. +- `unknown`: the integration has not declared a portable capability for that signal. + +Missing evidence is never treated as proof that an underlying agent action did not occur, and hidden model reasoning is never claimed as observable. ## Claude Code -Claude Code exposes lifecycle hooks such as `SessionStart`, `PostToolUse`, `PermissionRequest`, `SubagentStart`, and `SessionEnd`. Command hooks receive their native JSON payload on stdin. +### Rich observation: hooks + stable OpenTelemetry logs + +OpenWorkGraph combines two Claude Code surfaces: + +1. **Lifecycle hooks** for session boundaries, completed tools, permission requests, failures, and subagent start/stop. +2. **Claude Code OpenTelemetry log events** for per-prompt model calls, token usage, model identity, completed tools, permission decisions, retries and subagent completion. -OpenWorkGraph registers only lifecycle points that provide useful structural evidence: +The hook surface remains useful as a fail-open lifecycle/fallback channel. The OTel log surface supplies the signals hooks do not expose, especially model calls and token usage. + +OpenWorkGraph registers only content-safe hook points: ```text SessionStart @@ -50,117 +48,122 @@ SubagentStop StopFailure ``` -Content-heavy events such as `UserPromptSubmit`, `PreToolUse`, `MessageDisplay`, and `Stop` are intentionally not registered. - -`PermissionRequest` is represented as `human_approval_requested` because the hook fires as Claude Code is about to ask the user for permission and the OpenWorkGraph hook itself never supplies a decision. `PermissionDenied` is different: current Claude Code emits it for an automatic permission denial in auto mode, so OpenWorkGraph records it as a denied execution/error rather than falsely claiming that a human denied the action. +Content-heavy hooks such as `UserPromptSubmit`, `PreToolUse`, `MessageDisplay`, and `Stop` are intentionally not registered. -### One-click setup +The Claude OTel adapter recognizes these stable structural events: -On the dashboard's **Connect** tab, click **Connect** on the Claude Code card. OpenWorkGraph adds its hooks to `~/.claude/settings.json`: it writes a timestamped `settings.json.owg-backup-*` copy first, keeps every other setting and hook, replaces (never duplicates) earlier OpenWorkGraph hooks, and refuses to touch a file that is not valid JSON. **Disconnect** removes only OpenWorkGraph's handlers. Changes apply to new Claude Code sessions. The card still turns green only when telemetry actually arrives. +```text +claude_code.user_prompt +claude_code.api_request +claude_code.api_error +claude_code.api_refusal +claude_code.tool_result +claude_code.tool_decision +claude_code.api_retries_exhausted +claude_code.subagent_completed +``` -### Generate the settings fragment manually +`prompt.id` from OTel and `prompt_id` from current Claude Code hooks are used as a structural turn correlation key. This lets a single user request become one observed execution containing its model/tool/handoff sequence instead of collapsing an entire terminal session into one run. Older hook payloads without `prompt_id` safely fall back to the Claude session boundary. -From the same OpenWorkGraph installation/interpreter that you normally use: +`SubagentStart` produces a parent `handoff` event plus a child run boundary. `subagent_completed` enriches the parent execution from native telemetry. When hooks and OTel report the same completed tool, the read-side report prefers the richer OTel event so the tool is not double-counted. -```bash -python -m adapters.claude_code_hook --print-settings -``` +`PermissionRequest` is `human_approval_requested`. A Claude OTel `tool_decision` becomes `human_approval_received` only when its decision source is explicitly human (`user_temporary`, `user_permanent`, `user_reject`, or `user_abort`). Automated config/hook decisions are not attributed to a person. -Merge the printed `hooks` object into either: +### One-click setup -```text -~/.claude/settings.json -``` +On **Connect → Claude Code**, click **Connect**. OpenWorkGraph adds: -for your local user, or the relevant project `.claude/settings.json` if you intentionally want project-scoped configuration. +- its safe asynchronous hooks; and +- logs-only OTel settings pointing at the local write-only endpoint. -OpenWorkGraph edits Claude Code settings only when you click **Connect** or **Disconnect** on the dashboard; it never does so in the background. +It writes a timestamped backup first, preserves every unrelated setting/hook, replaces rather than duplicates older OWG hooks, and refuses to overwrite an existing conflicting telemetry destination. Disconnect removes only OWG hooks and telemetry values that still exactly match what OWG wrote; if the user changed a value afterwards, OWG leaves it alone. -The generated hook is a single shell command that first `cd`s into the OpenWorkGraph installation directory (the `adapters` package is not installed into the venv, and Claude Code runs hooks from the session's project directory) and then runs the interpreter with `-m adapters.claude_code_hook`. Both paths are shell-quoted, so directories containing spaces such as `Application Support` work. It also sets `"async": true`. OpenWorkGraph is observational and never returns a Claude Code control decision, so the bridge runs in the background rather than adding local HTTP latency to the triggering tool or lifecycle event. +The managed Claude settings explicitly keep content logging disabled: -### Failure behavior +```text +OTEL_LOG_USER_PROMPTS=0 +OTEL_LOG_ASSISTANT_RESPONSES=0 +OTEL_LOG_TOOL_DETAILS=0 +OTEL_LOG_TOOL_CONTENT=0 +OTEL_LOG_RAW_API_BODIES=0 +``` -The hook bridge is fail-open by design. Invalid JSON, an unavailable OpenWorkGraph server, authentication failure, or an unexpected adapter exception cannot block or approve Claude Code tool execution. Because the generated command hook is asynchronous, Claude Code proceeds without waiting for the OpenWorkGraph bridge to finish. +This is defense in depth. The server-side Claude adapter independently uses a strict allowlist and never copies content-bearing attributes even if a user changes upstream logging settings later. -Set: +The dedicated local endpoint is: ```text -OWG_AGENT_ADAPTER_DEBUG=1 +POST /agent-ingest/v1/claude-otel ``` -only when debugging. Even in debug mode the hook prints a fixed notice rather than exception details or native payload content. +The integration uses Claude's logs exporter only. OWG v0.90 does **not** require Claude's beta detailed-trace exporter. + +### Hook failure behavior + +Hooks remain asynchronous and fail-open. Invalid JSON, an unavailable OpenWorkGraph server, authentication failure, or an adapter exception cannot block or approve Claude Code execution. `OWG_AGENT_ADAPTER_DEBUG=1` prints only a fixed diagnostic notice, never native exception/payload content. ## Codex -Current Codex builds support OpenTelemetry log, trace, and metrics exporters. OpenWorkGraph's recommended integration uses the **trace exporter only** because Codex's diagnostic log stream may contain richer content such as tool arguments/results, whereas its trace-safe events expose the structural information OpenWorkGraph needs. +OpenWorkGraph's default Codex integration uses the native OTLP trace exporter and does not enable content-rich diagnostic log export. -OpenWorkGraph currently maps these Codex structural events: +The adapter maps structural evidence for: ```text -codex.conversation_starts -> run_started -codex.tool_result -> tool_call -codex.api_request -> model_call +conversation start -> run_started +API request -> model_call +completed tool -> tool_call +user-sourced decision -> human_approval_received +multi-agent spawn/send -> handoff ``` -The parser also understands `codex.tool_decision` if a compatible relay explicitly sends such a record, but the default configuration below does not enable Codex log export merely to obtain approval events. A tool decision is represented as `human_approval_received` only when Codex explicitly marks its decision source as `user`; decisions resolved by an automated reviewer, or records with no decision provenance, are ignored rather than attributed to a person. - -### One-click setup - -On the dashboard's **Connect** tab, click **Connect** on the Codex card. OpenWorkGraph appends a clearly marked, managed `[otel]` block to `~/.codex/config.toml` (or `$CODEX_HOME/config.toml`) after writing a timestamped backup, validates the result as TOML, and restarts nothing: Codex picks it up on its next start. If the file already has its own `[otel]` settings, or is not valid TOML, OpenWorkGraph changes nothing and asks you to merge manually. **Disconnect** removes only the managed block. +Where available, Codex turn IDs become run boundaries and parent span attributes contribute model identity and token counters. Multi-agent communication is conservative: only a structural `spawn` sent to another agent is treated as a handoff; ordinary inter-agent message/result content is ignored. -### Generate a safe trace-exporter snippet manually +Some Codex builds have emitted tracing call-site names in `event.name` rather than the semantic event name. The adapter can infer the supported event class from a small set of low-cardinality structural fields, never from record bodies, arguments, output, message content, or arbitrary span names. -Preview a snippet with a placeholder credential: +### One-click setup -```bash -python -m adapters.codex_config -``` +**Connect → Codex** appends a clearly marked managed `[otel]` block to `~/.codex/config.toml` (or `$CODEX_HOME/config.toml`) after writing a backup. If the file already has its own `[otel]` settings or invalid TOML, OWG changes nothing and requests manual merge. Disconnect removes only the managed block. -Or explicitly print a complete snippet containing the local **write-only** agent token: +The generated configuration disables prompt/agent-response/guardian logging and points only the trace exporter at: -```bash -python -m adapters.codex_config --with-token +```text +POST /agent-ingest/v1/codex-otel ``` -Merge those keys into `~/.codex/config.toml`. +## OpenAI Agents SDK -If your Codex config already has an `[otel]` section, merge the generated keys into that existing section rather than adding a second `[otel]` table. +For Python applications using the OpenAI Agents SDK, OWG registers an **additional tracing processor** rather than replacing the application's existing tracing. -The generated configuration is equivalent to: +The processor observes structural trace/run boundaries plus: -```toml -[otel] -log_user_prompt = false -log_agent_responses = false -log_guardian_assessments = false -trace_exporter = { otlp-http = { endpoint = "http://127.0.0.1:8787/agent-ingest/v1/codex-otel", headers = { Authorization = "Bearer WRITE_ONLY_TOKEN" }, protocol = "json" } } -``` +- generation/response/transcription/speech spans as `model_call`; +- function/MCP spans as `tool_call`; +- native handoff spans as `handoff`; +- parent-child span linkage; +- duration and error state; +- model identity and token usage where the SDK span exposes them. -No log exporter or metrics exporter is added by OpenWorkGraph. +Response span usage/model fields are read directly without serializing response content. Approval request/decision events are currently marked `not_observable` for this adapter rather than shown as zero. -### Codex ingestion behavior +Install in the agent application: -The dedicated endpoint is: +```python +from adapters.openai_agents import install_openai_agents_processor -```text -POST /agent-ingest/v1/codex-otel +install_openai_agents_processor() ``` -It accepts OTLP/HTTP JSON under the same 2 MB request bound as the generic agent-ingest path and processes at most 1,000 log/span-event records per request. - -The adapter uses an allowlist. OTLP `body`, prompt content, tool arguments/output, account identifiers, arbitrary span attributes, and arbitrary span names are ignored. Stable native request/call identifiers may be hashed solely to make repeated log/trace copies idempotent; the original identifiers are not added as content fields. +## Generic OpenTelemetry -## Generic OpenTelemetry trace export - -For runtimes that expose ordinary OTLP traces rather than a dedicated OpenWorkGraph adapter, the generic structural endpoint is: +For runtimes that expose ordinary GenAI OTLP traces rather than a dedicated OWG adapter: ```text POST /agent-ingest/v1/otel ``` -It currently accepts **OTLP/HTTP JSON only**. It does not accept protobuf bodies and OpenWorkGraph does not currently expose the conventional `/v1/traces` alias. +OWG accepts OTLP/HTTP JSON and projects a strict portable subset of GenAI semantic conventions. Portable model calls, tool calls, usage, model identity and timings can be observed when emitted. Provider-specific handoffs and human approvals are **not** guessed; those capabilities remain unavailable unless a native adapter provides them. -For an OpenTelemetry SDK/exporter that supports `http/json`, use the signal-specific standard variables so the endpoint path is used exactly as written: +Example: ```bash export OTEL_EXPORTER_OTLP_TRACES_ENDPOINT="http://127.0.0.1:8787/agent-ingest/v1/otel" @@ -168,36 +171,16 @@ export OTEL_EXPORTER_OTLP_TRACES_PROTOCOL="http/json" export OTEL_EXPORTER_OTLP_TRACES_HEADERS="Authorization=Bearer WRITE_ONLY_AGENT_TOKEN" ``` -Use the actual local `.agent_ingest_token` value in place of `WRITE_ONLY_AGENT_TOKEN`. - -Do **not** set only `OTEL_EXPORTER_OTLP_ENDPOINT=http://127.0.0.1:8787`: standard OTLP/HTTP exporters construct a signal path such as `/v1/traces` from the generic base endpoint, and that route is not an OpenWorkGraph ingest route. The trace-specific endpoint above is used as-is by compliant exporters. - -`http/json` support is optional across OpenTelemetry SDKs. If a particular runtime supports only `http/protobuf` or gRPC, do not send that binary payload to this JSON endpoint; use a JSON-capable exporter/relay or a native adapter instead. +Use the signal-specific endpoint exactly as shown. The generic endpoint currently accepts JSON, not OTLP protobuf or gRPC. ## Local-first boundary -The Claude bridge defaults to: - -```text -http://127.0.0.1:8787 -``` - -and refuses a non-loopback OpenWorkGraph API URL unless `OWG_AGENT_ALLOW_REMOTE=1` is explicitly set. It can only call `/agent-ingest/*` write routes. - -The normal OpenWorkGraph launcher binds the local API to `127.0.0.1`, so the native integrations remain local by default. +The normal OWG launcher binds the observer API to `127.0.0.1`. Native integrations receive only the write-only agent-ingest credential and remain local by default. When organization Gateway synchronization is enabled, agent evidence remains local unless the endpoint owner separately opts in with `gateway.local_policy.allow_agent_events: true`; see `docs/AGENT_GATEWAY_SHARING.md`. -## What this does not do - -These adapters do not: +## What these adapters do not do -- scrape terminal output or transcript files; -- alter agent prompts or instructions; -- inject context into an agent; -- proxy tool calls; -- change Claude Code/Codex approval decisions; -- make OpenWorkGraph a dependency for agent execution; -- claim to observe internal model reasoning. +They do not scrape terminal output or transcripts, alter agent prompts/instructions, inject context into an agent, proxy tool calls, change approval decisions, make OWG a dependency for agent execution, or claim access to internal model reasoning. -They provide structural execution evidence. Retrieval of learned organizational workflow context back into agents remains a separate layer built on top of the shared OpenWorkGraph evidence store. +They provide provider-neutral **structural execution evidence**. Retrieval of learned work context back into agents is a separate layer over the same canonical OpenWorkGraph store. From ee4c4d768ffee870db429ddf304fab771c6d4797 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:40:16 +0200 Subject: [PATCH 21/36] Test Claude OTel route auth and privacy --- tests/test_claude_otel_route_v090.py | 75 ++++++++++++++++++++++++++++ 1 file changed, 75 insertions(+) create mode 100644 tests/test_claude_otel_route_v090.py diff --git a/tests/test_claude_otel_route_v090.py b/tests/test_claude_otel_route_v090.py new file mode 100644 index 00000000..bf6d74ff --- /dev/null +++ b/tests/test_claude_otel_route_v090.py @@ -0,0 +1,75 @@ +from __future__ import annotations + +import json + +from fastapi import FastAPI +from fastapi.testclient import TestClient + +from server import db as server_db +from server.agent_auth import ensure_agent_ingest_token +from server.agent_routes import AGENT_CLAUDE_OTEL_PATH, router as agent_router +from server.local_auth import ensure_api_token + + +def _attr(key: str, value): + if isinstance(value, bool): + wrapped = {"boolValue": value} + elif isinstance(value, int): + wrapped = {"intValue": str(value)} + else: + wrapped = {"stringValue": str(value)} + return {"key": key, "value": wrapped} + + +def _payload() -> dict: + attrs = { + "event.name": "claude_code.api_request", + "session.id": "claude-route-session", + "prompt.id": "claude-route-prompt", + "model": "claude-sonnet-5", + "input_tokens": 25, + "output_tokens": 5, + "duration_ms": 500, + "request_id": "native-private-request-id", + "prompt": "SUPERSECRET prompt", + "response": "patient@example.com SUPERSECRET response", + } + return { + "resourceLogs": [{ + "scopeLogs": [{"logRecords": [{ + "body": {"stringValue": "SUPERSECRET log body"}, + "attributes": [_attr(k, v) for k, v in attrs.items()], + }]}], + }], + } + + +def _app() -> FastAPI: + server_db.init_db() + app = FastAPI() + app.include_router(agent_router) + return app + + +def test_claude_otel_route_requires_write_only_token_and_persists_only_structure(): + agent_headers = {"Authorization": f"Bearer {ensure_agent_ingest_token()}"} + api_headers = {"Authorization": f"Bearer {ensure_api_token()}"} + + with TestClient(_app()) as client: + assert client.post(AGENT_CLAUDE_OTEL_PATH, json=_payload()).status_code == 401 + assert client.post(AGENT_CLAUDE_OTEL_PATH, json=_payload(), headers=api_headers).status_code == 401 + response = client.post(AGENT_CLAUDE_OTEL_PATH, json=_payload(), headers=agent_headers) + assert response.status_code == 202, response.text + assert response.content == b"" + + rows = server_db.rows("SELECT * FROM events WHERE session_id = ?", ("claude-route-session",)) + assert len(rows) == 1 + row = rows[0] + assert row["event_type"] == "agent_model_call" + assert row["sensor_id"] == "agent:claude-code-otel" + assert row["metadata"]["agent"]["framework"] == "claude-code" + assert row["metadata"]["agent"]["model"] == "claude-sonnet-5" + assert row["metadata"]["usage"]["total_tokens"] == 30 + serialized = json.dumps(row, ensure_ascii=False) + for forbidden in ("SUPERSECRET", "patient@example.com", "native-private-request-id"): + assert forbidden not in serialized From 47e21b7804cd21e72c18fc8f807aa73cc4bb59d4 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:41:57 +0200 Subject: [PATCH 22/36] Expose rich capability-aware agent summaries through MCP --- mcp_server/agent_tools.py | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/mcp_server/agent_tools.py b/mcp_server/agent_tools.py index 3e437899..492ba78f 100644 --- a/mcp_server/agent_tools.py +++ b/mcp_server/agent_tools.py @@ -32,7 +32,11 @@ def _run_summary(execution: dict[str, Any]) -> dict[str, Any]: "run_finish_observed", "complete_boundary_observed", "event_count_total", + # operation_counts remains for backward compatibility. New clients + # should prefer observed_operation_counts because it applies native + # adapter de-duplication (for example Claude hooks + OTel). "operation_counts", + "observed_operation_counts", "tool_category_counts", "structural_steps", "structural_steps_truncated", @@ -40,6 +44,11 @@ def _run_summary(execution: dict[str, Any]) -> dict[str, Any]: "approval_received_count", "task_context_linkage_status", "observed_coverage", + "signal_capabilities", + "telemetry_sources", + "telemetry_depth", + "models_observed", + "usage_totals", "derived", "authoritative", ) @@ -65,9 +74,10 @@ def get_agent_runs( ) -> dict[str, Any]: """Return compact privacy-safe summaries of observed agent executions. - Results report only structural signals actually present in canonical - evidence. Missing coverage means not observed, not proof that the runtime - did not perform the underlying action. Native run/trace/span identifiers, + Results report observed structural counts separately from the active + adapter's capability to observe each signal. A numeric zero is meaningful + only when that signal is observable; not_observable/unknown must not be + interpreted as zero underlying activity. Native run/trace/span IDs, prompts, model-response content, tool arguments/results and hidden reasoning are not exposed. """ @@ -86,6 +96,7 @@ def get_agent_runs( **{key: value for key, value in result.items() if key != "executions"}, "executions": [_run_summary(item) for item in list(result.get("executions") or [])], "events_omitted_from_list_view": True, + "count_semantics": "prefer observed_operation_counts; interpret counts together with signal_capabilities", "detail_tool": "get_agent_execution_trace", } return core._finish(name, compact) From 6bf0dad2ebc4858690dbd814c241a8e7c8034e90 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:43:36 +0200 Subject: [PATCH 23/36] Fix optional Claude SessionStart model label --- shared/claude_code_adapter.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/shared/claude_code_adapter.py b/shared/claude_code_adapter.py index 6fc49cf2..c882dccd 100644 --- a/shared/claude_code_adapter.py +++ b/shared/claude_code_adapter.py @@ -162,7 +162,7 @@ def claude_hook_to_agent_events( run_id=session_id, trace_id=trace_id, event_key=hook, - model=_safe_label(payload.get("model"), limit=160), + model=_safe_label(payload.get("model"), default="", limit=160), )] if hook == "SessionEnd": From 863955efe4c8d592e6f3381566e13cd81351cedb Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:43:50 +0200 Subject: [PATCH 24/36] Update Claude label privacy test for explicit handoff plus child run --- tests/test_native_agent_label_privacy_v060.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/tests/test_native_agent_label_privacy_v060.py b/tests/test_native_agent_label_privacy_v060.py index 6e9b2bcb..181ca356 100644 --- a/tests/test_native_agent_label_privacy_v060.py +++ b/tests/test_native_agent_label_privacy_v060.py @@ -23,7 +23,7 @@ def test_claude_rejects_content_bearing_tool_and_agent_labels(): assert tool_event["tool_name"] == "unknown-tool" assert "patient@example.com" not in json.dumps(tool_event) - [agent_event] = claude_hook_to_agent_events( + subagent_events = claude_hook_to_agent_events( { "session_id": "s1", "hook_event_name": "SubagentStart", @@ -32,8 +32,13 @@ def test_claude_rejects_content_bearing_tool_and_agent_labels(): }, observed_at="2026-09-25T00:00:00Z", ) + assert len(subagent_events) == 2 + handoff, agent_event = subagent_events + assert handoff["operation"] == "handoff" + assert handoff["tool_name"] == "subagent:subagent" + assert agent_event["operation"] == "run_started" assert agent_event["agent_name"] == "Claude Code/subagent" - assert "patient@example.com" not in json.dumps(agent_event) + assert "patient@example.com" not in json.dumps(subagent_events) def test_codex_rejects_content_bearing_tool_and_model_labels(): From 96332925bff79a0078788c586b61ed0089cc07bc Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:44:02 +0200 Subject: [PATCH 25/36] Advance release-version test to v0.90.0 --- tests/test_release_version_v087.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_release_version_v087.py b/tests/test_release_version_v087.py index 710c7943..12994281 100644 --- a/tests/test_release_version_v087.py +++ b/tests/test_release_version_v087.py @@ -6,7 +6,7 @@ ROOT = Path(__file__).resolve().parents[1] -EXPECTED_VERSION = "0.89.0" +EXPECTED_VERSION = "0.90.0" def test_release_version_sources_are_aligned(): From 61f404ef93fa3f785623f26539511e8f3d7210b6 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:44:36 +0200 Subject: [PATCH 26/36] Preserve exact generic OTel transport guidance --- docs/NATIVE_AGENT_ADAPTERS.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/NATIVE_AGENT_ADAPTERS.md b/docs/NATIVE_AGENT_ADAPTERS.md index 2c212883..b87c098b 100644 --- a/docs/NATIVE_AGENT_ADAPTERS.md +++ b/docs/NATIVE_AGENT_ADAPTERS.md @@ -171,7 +171,7 @@ export OTEL_EXPORTER_OTLP_TRACES_PROTOCOL="http/json" export OTEL_EXPORTER_OTLP_TRACES_HEADERS="Authorization=Bearer WRITE_ONLY_AGENT_TOKEN" ``` -Use the signal-specific endpoint exactly as shown. The generic endpoint currently accepts JSON, not OTLP protobuf or gRPC. +Use the signal-specific endpoint exactly as shown. The generic endpoint accepts OTLP/HTTP JSON and **does not accept protobuf bodies**. It also **does not currently expose the conventional `/v1/traces` alias**; use `/agent-ingest/v1/otel` exactly. ## Local-first boundary From 232d44f0861fb40a31233d23f7dee2fbab89cbc8 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:46:17 +0200 Subject: [PATCH 27/36] Fix privacy assertion to test values rather than canonical field names --- tests/test_rich_agent_observability_v090.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/test_rich_agent_observability_v090.py b/tests/test_rich_agent_observability_v090.py index dcf45e38..9674ce2d 100644 --- a/tests/test_rich_agent_observability_v090.py +++ b/tests/test_rich_agent_observability_v090.py @@ -89,7 +89,6 @@ def test_claude_otel_projects_rich_structure_and_never_content(): "private-tool-id", "private-tool-approval", "tool_parameters", - "tool_result", ): assert forbidden not in serialized From b341ce198e201f87407446009b952ca543a9d590 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:48:26 +0200 Subject: [PATCH 28/36] Export Codex structural logs as well as traces --- adapters/codex_config.py | 31 ++++++++++++++++++++----------- 1 file changed, 20 insertions(+), 11 deletions(-) diff --git a/adapters/codex_config.py b/adapters/codex_config.py index c0e24ad7..3d56e6fc 100644 --- a/adapters/codex_config.py +++ b/adapters/codex_config.py @@ -1,10 +1,11 @@ from __future__ import annotations -"""Print a Codex OTLP trace-exporter snippet for OpenWorkGraph. +"""Print a privacy-safe Codex OTLP logs + trace exporter snippet for OWG. -This helper never edits ~/.codex/config.toml. With --with-token it intentionally -prints the least-privilege write token so an administrator can paste a complete -local-only configuration. +Codex's business events (API requests, completed tools, approval decisions and +multi-agent communication) are emitted on its OTLP log surface, while native +span hierarchy is emitted on the trace surface. OWG points both at the same +write-only structural endpoint and keeps all content-bearing opt-ins disabled. """ import argparse @@ -20,21 +21,29 @@ def _toml_string(value: str) -> str: return json.dumps(value) +def _exporter(endpoint: str, authorization: str) -> str: + return ( + "{ otlp-http = { endpoint = " + + _toml_string(endpoint) + + ", headers = { Authorization = " + + _toml_string(authorization) + + " }, protocol = \"json\" } }" + ) + + def config_snippet(*, token: str, base_url: str | None = None) -> str: endpoint = (base_url or _base_url()).rstrip("/") + "/agent-ingest/v1/codex-otel" authorization = f"Bearer {token}" + exporter = _exporter(endpoint, authorization) return "\n".join([ "[otel]", + # Keep every Codex content-bearing opt-in disabled. Structural business + # events still export and are then strict-allowlisted again server-side. "log_user_prompt = false", "log_agent_responses = false", "log_guardian_assessments = false", - ( - "trace_exporter = { otlp-http = { endpoint = " - + _toml_string(endpoint) - + ", headers = { Authorization = " - + _toml_string(authorization) - + " }, protocol = \"json\" } }" - ), + f"exporter = {exporter}", + f"trace_exporter = {exporter}", ]) From caf8371033e1b1ac2b5ce09fcbe0803ca41a5830 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:50:24 +0200 Subject: [PATCH 29/36] Describe rich Claude and Codex telemetry accurately in setup UI --- dashboard/agent_control_plane.js | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/dashboard/agent_control_plane.js b/dashboard/agent_control_plane.js index b661ffab..3789e02d 100644 --- a/dashboard/agent_control_plane.js +++ b/dashboard/agent_control_plane.js @@ -43,13 +43,13 @@ const section=document.createElement('div'); section.id='agent-observation-setup';section.className='card connection-section'; section.innerHTML=` -

Observe an agent

Instrument an agent's native lifecycle or trace surface so OpenWorkGraph can measure structural execution: runs, tools, handoffs, approvals, failures, timings and coverage. Prompts, responses, reasoning, tool arguments and tool results are not collected.
+

Observe an agent

Instrument an agent's native lifecycle or telemetry surface so OpenWorkGraph can measure structural execution: runs, models, tools, handoffs, approvals, failures, timings and coverage. Prompts, responses, reasoning, tool arguments and tool results are not collected.
Status means telemetry observed. The green badge appears only when evidence actually arrives. Connect adds OpenWorkGraph's entries to that agent's own settings file (a backup is written first and your other settings are kept); Disconnect removes only what OpenWorkGraph added.
- ${setupCard('claude_code','Native hooks','Claude Code','Observe lifecycle, tool use, failures, approval requests and subagent handoffs without capturing prompts or tool contents.')} - ${setupCard('codex','OTel trace','Codex','Use Codex trace export only. OpenWorkGraph does not enable the richer diagnostic log stream.')} - ${setupCard('openai_agents','Tracing processor','OpenAI Agents SDK','Register OpenWorkGraph as an additional tracing processor; existing SDK tracing remains active.')} - ${setupCard('otel','Provider-neutral','OpenTelemetry / custom','Send OTLP/HTTP JSON traces or canonical structural events from another agent runtime.')} + ${setupCard('claude_code','Hooks + OTel logs','Claude Code','Observe per-request model calls and tokens, tool use, approval requests/decisions, failures and subagent handoffs without capturing prompt or tool content.')} + ${setupCard('codex','OTel logs + trace','Codex','Observe structural API, tool, approval and multi-agent events plus trace hierarchy. Content-bearing log options stay disabled.')} + ${setupCard('openai_agents','Tracing processor','OpenAI Agents SDK','Register OpenWorkGraph as an additional tracing processor for model, tool, handoff, hierarchy, usage and timing signals.')} + ${setupCard('otel','Provider-neutral','OpenTelemetry / custom','Send portable GenAI OTLP/HTTP JSON traces or canonical structural events from another agent runtime.')}
`; const anchor=grid||panel.lastElementChild; if(anchor&&anchor.parentNode===panel)anchor.insertAdjacentElement('afterend',section);else panel.appendChild(section); @@ -114,7 +114,7 @@ renderConfigState(); const where=result.path?`${h(result.path)}`:'its settings file'; const backup=result.backup?`
Backup of the previous file: ${h(result.backup)}
`:''; - if(action==='connect')window.openModal?.(`${label} connected`,'Agent observation',`

OpenWorkGraph's observation hooks were added to ${where}. ${h(result.note||'')}

${backup}
The status turns green once the first telemetry arrives. Disconnect removes only OpenWorkGraph's entries.
`); + if(action==='connect')window.openModal?.(`${label} connected`,'Agent observation',`

OpenWorkGraph's observation settings were added to ${where}. ${h(result.note||'')}

${backup}
The status turns green once the first telemetry arrives. Disconnect removes only OpenWorkGraph's entries.
`); else window.openModal?.(`${label} disconnected`,'Agent observation',`

OpenWorkGraph's entries were removed from ${where}. Your other settings were left as they were.

${backup}`); }catch(_){ renderConfigState(); @@ -156,10 +156,10 @@ let title='Observe an agent',body=''; if(kind==='claude_code'){ const x=integrations.claude_code||{};title='Observe Claude Code'; - body=`

Prefer the Connect button, which does this for you. To do it by hand, merge the hooks object below into your Claude Code settings.

${privacyHtml(payload)}

Dashboard-generated settings

${codeBox(JSON.stringify(x.settings||{},null,2),'agentSetupCode')}

Equivalent command

${codeBox(x.command||'python -m adapters.claude_code_hook --print-settings','agentSetupCommand')}
Hooks are asynchronous and fail-open. If OpenWorkGraph is unavailable, Claude Code continues normally.
`; + body=`

Prefer the Connect button, which does this for you. To do it by hand, merge the hooks and env objects below into your Claude Code settings.

${privacyHtml(payload)}

Dashboard-generated settings

${codeBox(JSON.stringify(x.settings||{},null,2),'agentSetupCode')}

Equivalent hook command

${codeBox(x.command||'python -m adapters.claude_code_hook --print-settings','agentSetupCommand')}
Hooks are asynchronous and fail-open. OTel content logging is explicitly disabled; the OWG ingest path independently strict-allowlists structural fields.
`; }else if(kind==='codex'){ const x=integrations.codex||{};title='Observe Codex'; - body=`

Prefer the Connect button, which adds this for you unless you already have your own [otel] settings. To do it by hand, merge these keys into your existing [otel] section. The dashboard includes only the dedicated local write-only telemetry credential; it does not grant access to your OWG history.

${privacyHtml(payload)}${codeBox(x.config||'', 'agentSetupCode')}
OpenWorkGraph enables trace export only. User prompts, agent responses and guardian assessments remain disabled.
`; + body=`

Prefer the Connect button, which adds this for you unless you already have your own [otel] settings. To do it by hand, merge these keys into your existing [otel] section. The dashboard includes only the dedicated local write-only telemetry credential; it does not grant access to your OWG history.

${privacyHtml(payload)}${codeBox(x.config||'', 'agentSetupCode')}
OpenWorkGraph enables structural Codex log events and trace export. User prompts, agent responses and guardian assessments remain disabled, and content-bearing fields are discarded server-side.
`; }else if(kind==='openai_agents'){ const x=integrations.openai_agents||{};title='Observe OpenAI Agents SDK'; body=`

Add OpenWorkGraph as an additional tracing processor in the agent application's Python environment.

${privacyHtml(payload)}${codeBox(x.python||'', 'agentSetupCode')}
This does not replace existing SDK tracing and does not make OpenWorkGraph a dependency for the agent's control flow.
`; @@ -228,4 +228,4 @@ } if(document.readyState==='loading')document.addEventListener('DOMContentLoaded',install);else install(); -})(); +})(); \ No newline at end of file From 49bbcbc6294493257ba6ff00fe2bf055cdd8332f Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:50:56 +0200 Subject: [PATCH 30/36] Advertise Codex structural logs plus traces in agent setup --- server/agent_dashboard_control_plane.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/server/agent_dashboard_control_plane.py b/server/agent_dashboard_control_plane.py index 793be78f..5a3fbd72 100644 --- a/server/agent_dashboard_control_plane.py +++ b/server/agent_dashboard_control_plane.py @@ -123,14 +123,16 @@ def agent_setup_payload(request: Request) -> dict[str, Any]: }, "codex": { "label": "Codex", - "method": "otel_http_json_trace_export", + "method": "otel_http_json_logs_and_trace_export", "command": f"cd {shlex.quote(str(ROOT))} && {shlex.quote(sys.executable)} -m adapters.codex_config --with-token", "one_click": True, "config": config_snippet(token=token, base_url=base_url), "endpoint": f"{base_url}/agent-ingest/v1/codex-otel", - "instructions": "Click Connect to add a managed [otel] block to ~/.codex/config.toml (refused if you already have your own [otel] settings), or merge these keys manually into the existing [otel] section. Do not create a second [otel] table.", - "logs_enabled_by_owg": False, - "telemetry_depth": "native_trace", + "instructions": "Click Connect to add a managed [otel] block that sends structural Codex log events and traces to the local write-only endpoint (refused if you already have your own [otel] settings). User-prompt, agent-response and guardian-rationale content logging remains disabled.", + "logs_enabled_by_owg": True, + "structural_logs_enabled_by_owg": True, + "content_logging_enabled": False, + "telemetry_depth": "rich_native_events_plus_trace", }, "openai_agents": { "label": "OpenAI Agents SDK", From e398280055648aee928caa1720d2993dff1eeed6 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:51:25 +0200 Subject: [PATCH 31/36] Document Codex structural logs and trace export --- docs/NATIVE_AGENT_ADAPTERS.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/NATIVE_AGENT_ADAPTERS.md b/docs/NATIVE_AGENT_ADAPTERS.md index b87c098b..27614243 100644 --- a/docs/NATIVE_AGENT_ADAPTERS.md +++ b/docs/NATIVE_AGENT_ADAPTERS.md @@ -104,7 +104,7 @@ Hooks remain asynchronous and fail-open. Invalid JSON, an unavailable OpenWorkGr ## Codex -OpenWorkGraph's default Codex integration uses the native OTLP trace exporter and does not enable content-rich diagnostic log export. +OpenWorkGraph's Codex integration uses **both Codex's structural OTLP log exporter and trace exporter**. The log layer is required for Codex business events such as API requests, completed tools, approval decisions and multi-agent communication; the trace layer supplies native span hierarchy. Content-bearing opt-ins stay disabled, and the server independently strict-allowlists every accepted field. The adapter maps structural evidence for: @@ -124,7 +124,7 @@ Some Codex builds have emitted tracing call-site names in `event.name` rather th **Connect → Codex** appends a clearly marked managed `[otel]` block to `~/.codex/config.toml` (or `$CODEX_HOME/config.toml`) after writing a backup. If the file already has its own `[otel]` settings or invalid TOML, OWG changes nothing and requests manual merge. Disconnect removes only the managed block. -The generated configuration disables prompt/agent-response/guardian logging and points only the trace exporter at: +The generated configuration explicitly keeps prompt, agent-response and guardian-rationale logging disabled while pointing both the structural log exporter and trace exporter at the same local write-only endpoint: ```text POST /agent-ingest/v1/codex-otel From fcd15456d3f65c3a52a01751a17f6e2c8b36f100 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:52:00 +0200 Subject: [PATCH 32/36] Verify Codex one-click exports structural logs and traces --- tests/test_agent_config_rich_v090.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/tests/test_agent_config_rich_v090.py b/tests/test_agent_config_rich_v090.py index c7b746d3..01faa1ff 100644 --- a/tests/test_agent_config_rich_v090.py +++ b/tests/test_agent_config_rich_v090.py @@ -1,12 +1,14 @@ from __future__ import annotations import json +import tomllib from pathlib import Path import pytest from adapters.claude_code_hook import settings_fragment from adapters.claude_code_otel import env_settings +from adapters.codex_config import config_snippet from server import agent_config_writer as writer @@ -92,3 +94,15 @@ def test_claude_manual_fragment_contains_no_content_collection_opt_in(): forbidden_truthy = [key for key in managed if key.startswith("OTEL_LOG_") and managed[key] not in {"0", "false", "False"}] assert forbidden_truthy == [] assert "OTEL_TRACES_EXPORTER" not in managed # beta detailed traces are not required for v0.90 + + +def test_codex_fragment_exports_structural_logs_and_traces_but_no_content_opt_ins(): + snippet = config_snippet(token="write-only-token", base_url="http://127.0.0.1:8787") + parsed = tomllib.loads(snippet)["otel"] + assert "otlp-http" in parsed["exporter"] + assert "otlp-http" in parsed["trace_exporter"] + assert parsed["exporter"]["otlp-http"]["endpoint"].endswith("/agent-ingest/v1/codex-otel") + assert parsed["trace_exporter"]["otlp-http"]["endpoint"].endswith("/agent-ingest/v1/codex-otel") + assert parsed["log_user_prompt"] is False + assert parsed["log_agent_responses"] is False + assert parsed["log_guardian_assessments"] is False From e36cc14bd2d1c02a60c908be47b4760c14fa7980 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:52:58 +0200 Subject: [PATCH 33/36] Detect and upgrade older trace-only Codex OWG setup --- server/agent_config_writer.py | 35 ++++++++++++++++++++++++++++++----- 1 file changed, 30 insertions(+), 5 deletions(-) diff --git a/server/agent_config_writer.py b/server/agent_config_writer.py index faed9749..8c8751e0 100644 --- a/server/agent_config_writer.py +++ b/server/agent_config_writer.py @@ -285,7 +285,26 @@ def _parse_toml(text: str, path: Path) -> dict[str, Any]: def codex_status() -> dict[str, Any]: path = codex_config_path() text = path.read_text(encoding="utf-8") if path.exists() else "" - return {"configured": CODEX_BLOCK_START in text, "path": str(path)} + managed = CODEX_BLOCK_START in text + if not managed: + return {"configured": False, "partial_configured": False, "path": str(path)} + try: + otel = _parse_toml(text, path).get("otel") or {} + except ConfigConflict as exc: + return {"configured": False, "partial_configured": True, "path": str(path), "error": str(exc)} + rich = ( + isinstance(otel, dict) + and "exporter" in otel + and "trace_exporter" in otel + and otel.get("log_user_prompt") is False + and otel.get("log_agent_responses") is False + and otel.get("log_guardian_assessments") is False + ) + return { + "configured": bool(rich), + "partial_configured": not bool(rich), + "path": str(path), + } def codex_connect(snippet: str) -> dict[str, Any]: @@ -301,12 +320,18 @@ def codex_connect(snippet: str) -> dict[str, Any]: body = remainder.rstrip("\n") updated = (body + "\n\n" if body else "") + block parsed = _parse_toml(updated, path) - if "trace_exporter" not in (parsed.get("otel") or {}): + otel = parsed.get("otel") or {} + if not isinstance(otel, dict) or "exporter" not in otel or "trace_exporter" not in otel: raise ConfigConflict("Generated Codex configuration did not validate; use manual setup") backup = _backup(path) _atomic_write(path, updated) - return {"configured": True, "path": str(path), "backup": backup, - "note": "Takes effect the next time Codex starts."} + return { + "configured": True, + "partial_configured": False, + "path": str(path), + "backup": backup, + "note": "Takes effect the next time Codex starts.", + } def codex_disconnect() -> dict[str, Any]: @@ -320,4 +345,4 @@ def codex_disconnect() -> dict[str, Any]: _parse_toml(remainder, path) backup = _backup(path) _atomic_write(path, remainder) - return {"configured": False, "path": str(path), "backup": backup} + return {"configured": False, "partial_configured": False, "path": str(path), "backup": backup} From 3c8e77c778c96cc0bd76b7fefe1952d464fba802 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 00:53:28 +0200 Subject: [PATCH 34/36] Test upgrade from older trace-only Codex managed setup --- tests/test_agent_config_rich_v090.py | 34 ++++++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/tests/test_agent_config_rich_v090.py b/tests/test_agent_config_rich_v090.py index 01faa1ff..8f2ae885 100644 --- a/tests/test_agent_config_rich_v090.py +++ b/tests/test_agent_config_rich_v090.py @@ -19,6 +19,13 @@ def claude_file(tmp_path, monkeypatch): return path +@pytest.fixture +def codex_file(tmp_path, monkeypatch): + path = tmp_path / ".codex" / "config.toml" + monkeypatch.setenv("OWG_CODEX_CONFIG_PATH", str(path)) + return path + + def _managed_env() -> dict[str, str]: return env_settings(token="write-only-token", base_url="http://127.0.0.1:8787") @@ -106,3 +113,30 @@ def test_codex_fragment_exports_structural_logs_and_traces_but_no_content_opt_in assert parsed["log_user_prompt"] is False assert parsed["log_agent_responses"] is False assert parsed["log_guardian_assessments"] is False + + +def test_codex_trace_only_owg_block_is_detected_as_partial_and_upgraded(codex_file): + codex_file.parent.mkdir(parents=True) + endpoint = "http://127.0.0.1:8787/agent-ingest/v1/codex-otel" + old_block = "\n".join([ + writer.CODEX_BLOCK_START, + "[otel]", + "log_user_prompt = false", + "log_agent_responses = false", + "log_guardian_assessments = false", + f'trace_exporter = {{ otlp-http = {{ endpoint = "{endpoint}", headers = {{ Authorization = "Bearer tok" }}, protocol = "json" }} }}', + writer.CODEX_BLOCK_END, + "", + ]) + codex_file.write_text(old_block) + + before = writer.codex_status() + assert before["configured"] is False + assert before["partial_configured"] is True + + writer.codex_connect(config_snippet(token="tok", base_url="http://127.0.0.1:8787")) + after = writer.codex_status() + assert after["configured"] is True + assert after["partial_configured"] is False + parsed = tomllib.loads(codex_file.read_text())["otel"] + assert "exporter" in parsed and "trace_exporter" in parsed From e061d5ff53a7316a8f93684a859b6ed0604fd48e Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 10:18:05 +0200 Subject: [PATCH 35/36] Update Codex config test for structural logs and traces --- tests/test_codex_config_v060.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/test_codex_config_v060.py b/tests/test_codex_config_v060.py index 4039c52f..6ee50080 100644 --- a/tests/test_codex_config_v060.py +++ b/tests/test_codex_config_v060.py @@ -3,14 +3,14 @@ from adapters.codex_config import config_snippet -def test_codex_config_uses_trace_only_and_disables_content_logging(): +def test_codex_config_exports_structural_logs_and_traces_without_content_logging(): text = config_snippet(token="write-token", base_url="http://127.0.0.1:8787") assert "[otel]" in text assert "log_user_prompt = false" in text assert "log_agent_responses = false" in text assert "log_guardian_assessments = false" in text + assert "exporter =" in text assert "trace_exporter" in text - assert "agent-ingest/v1/codex-otel" in text - assert "Bearer write-token" in text - assert "exporter =" not in text.replace("trace_exporter =", "") + assert text.count("agent-ingest/v1/codex-otel") == 2 + assert text.count("Bearer write-token") == 2 assert "metrics_exporter" not in text From 4e93b2e4863b1a14af709a0ad5a7e6d877534bf3 Mon Sep 17 00:00:00 2001 From: Kinvectum <134240819+KAVentures@users.noreply.github.com> Date: Sun, 27 Sep 2026 10:35:50 +0200 Subject: [PATCH 36/36] Fix cross-platform file mode assertion on Windows --- tests/test_agent_one_click_v090.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/test_agent_one_click_v090.py b/tests/test_agent_one_click_v090.py index d5b99099..d5796729 100644 --- a/tests/test_agent_one_click_v090.py +++ b/tests/test_agent_one_click_v090.py @@ -87,7 +87,10 @@ def test_claude_disconnect_removes_only_openworkgraph(claude_file): def test_claude_connect_creates_missing_file_and_refuses_invalid_json(claude_file): writer.claude_connect(settings_fragment) assert writer.claude_status()["configured"] is True - assert oct(claude_file.stat().st_mode & 0o777) == "0o600" + # POSIX permission bits are meaningful on Unix-like systems. Windows reports + # synthesized mode bits (commonly 0666), so asserting 0600 there is invalid. + if os.name != "nt": + assert oct(claude_file.stat().st_mode & 0o777) == "0o600" claude_file.write_text("{not json") with pytest.raises(writer.ConfigConflict):