diff --git a/VERSION b/VERSION index 95fce8ca..01781720 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -0.98.0 +0.99.0 diff --git a/adapters/_agent_client.py b/adapters/_agent_client.py index 125ffd4c..2ad84e0f 100644 --- a/adapters/_agent_client.py +++ b/adapters/_agent_client.py @@ -67,7 +67,7 @@ def post_agent_events(events: list[dict], *, timeout: float = 0.75, spool: bool return {"status": "ignored", "received": 0} try: framework = str(events[0].get("framework") or "") if isinstance(events[0], dict) else "" - channel = "claude_code_hooks" if framework == "claude-code" else "agent_events" + channel = {"claude-code": "claude_code_hooks", "cursor": "cursor_hooks"}.get(framework, "agent_events") return post_json("/agent-ingest/v1/events", {"events": events}, timeout=timeout, channel=channel) except HTTPError as exc: if exc.code < 500 or not spool: diff --git a/adapters/cursor_hook.py b/adapters/cursor_hook.py new file mode 100644 index 00000000..db303a14 --- /dev/null +++ b/adapters/cursor_hook.py @@ -0,0 +1,133 @@ +from __future__ import annotations + +"""Cursor hook bridge for structural OpenWorkGraph evidence. + +Cursor runs hooks synchronously and reads their stdout. The foreground hook +therefore does only privacy-safe projection, writes Cursor's non-blocking reply, +then hands the already-allowlisted structural events to a detached helper process. +Network delivery/spooling never sits on Cursor's synchronous hook path. + +Raw prompt text, tool arguments/results, workspace paths and error text are never +passed to the helper. Any failure still exits 0. +""" + +import json +import os +import subprocess +import sys + +from adapters.claude_code_hook import PROJECT_ROOT, _shell_quote +from shared.cursor_hook_adapter import SUPPORTED_EVENTS, cursor_hook_to_agent_events + +OWG_MARKER = "adapters.cursor_hook" +HOOK_TIMEOUT_SECONDS = 5 +MAX_DELIVERY_BYTES = 256_000 + + +def hook_command(command: str | None = None) -> str: + python = _shell_quote(command or sys.executable) + root = _shell_quote(PROJECT_ROOT) + if os.name == "nt": + return f"cd /d {root} && {python} -m {OWG_MARKER}" + return f"cd {root} && {python} -m {OWG_MARKER}" + + +def hooks_fragment(command: str | None = None) -> dict: + handler = {"command": hook_command(command), "timeout": HOOK_TIMEOUT_SECONDS} + return {"version": 1, "hooks": {event: [dict(handler)] for event in SUPPORTED_EVENTS}} + + +def _reply(hook: str) -> None: + sys.stdout.write(json.dumps({"continue": True} if hook == "beforeSubmitPrompt" else {})) + sys.stdout.flush() + + +def _debug_notice() -> None: + if os.getenv("OWG_AGENT_ADAPTER_DEBUG", "").strip() == "1": + print("OpenWorkGraph Cursor hook skipped one event", file=sys.stderr) + + +def _deliver(events: list[dict]) -> int: + """Detached helper entrypoint; input is already privacy-safe structural data.""" + try: + from adapters._agent_client import post_agent_events + + post_agent_events(events, timeout=0.5) + except Exception: + _debug_notice() + return 0 + + +def _spawn_delivery(events: list[dict]) -> None: + if not events: + return + try: + raw = json.dumps(events, ensure_ascii=False, separators=(",", ":")).encode("utf-8") + except Exception: + return + if len(raw) > MAX_DELIVERY_BYTES: + return + + kwargs: dict = { + "cwd": PROJECT_ROOT, + "stdin": subprocess.PIPE, + "stdout": subprocess.DEVNULL, + "stderr": subprocess.DEVNULL, + "close_fds": True, + } + if os.name == "nt": + kwargs["creationflags"] = getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0) + else: + kwargs["start_new_session"] = True + + try: + child = subprocess.Popen([sys.executable, "-m", OWG_MARKER, "--deliver"], **kwargs) + if child.stdin is not None: + child.stdin.write(raw) + child.stdin.close() + except Exception: + # Observation must never make Cursor depend on OpenWorkGraph. + try: + child.kill() # type: ignore[name-defined] + except Exception: + pass + + +def main(argv: list[str] | None = None) -> int: + argv = list(argv or []) + if "--print-hooks" in argv: + print(json.dumps(hooks_fragment(), indent=2)) + return 0 + + if "--deliver" in argv: + try: + raw = sys.stdin.buffer.read(MAX_DELIVERY_BYTES + 1) + if len(raw) > MAX_DELIVERY_BYTES: + return 0 + value = json.loads(raw.decode("utf-8")) + events = [item for item in value if isinstance(item, dict)] if isinstance(value, list) else [] + except Exception: + events = [] + return _deliver(events) + + hook = "" + try: + raw = sys.stdin.buffer.read(2_000_001) + payload = json.loads(raw.decode("utf-8")) if len(raw) <= 2_000_000 else {} + hook = str(payload.get("hook_event_name") or "") if isinstance(payload, dict) else "" + except Exception: + payload = {} + + # Cursor's decision is returned before any network/spool work. Projection is + # local and allowlisted; delivery happens in a separate process. + _reply(hook) + try: + events = cursor_hook_to_agent_events(payload) + _spawn_delivery(events) + except Exception: + _debug_notice() + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/browser_extension/agent_surface_adapters.js b/browser_extension/agent_surface_adapters.js index 5bb1cb9c..df0ddd86 100644 --- a/browser_extension/agent_surface_adapters.js +++ b/browser_extension/agent_surface_adapters.js @@ -8,6 +8,16 @@ {key:'lovable', name:'Lovable', hosts:['lovable.dev']}, {key:'gemini', name:'Gemini', hosts:['gemini.google.com']}, ]; + // Structural signals come first and work in any UI language. Busy/send signals + // are scoped to the conversation/composer surface when structure is available, + // so an unrelated spinner or feedback form elsewhere on the page cannot start + // an agent run. English labels remain only a compatibility fallback. + const BUSY_SELECTOR='[aria-busy="true"],[data-is-streaming="true"],button[data-testid*="stop" i]'; + const SEND_TESTID=/(^|[-_])(send|submit)([-_]|$)/i; + const STOP_TESTID=/(^|[-_])(stop|cancel)([-_]|$)/i; + const COMPOSER_TESTID=/(composer|prompt|chat[-_]?input|message[-_]?input)/i; + const COMPOSER_CONTAINER_SELECTOR='[data-testid*="composer" i],[data-testid*="prompt" i],[data-testid*="chat-input" i],[data-testid*="message-input" i]'; + const COMPOSER_INPUT_SELECTOR='textarea,[contenteditable="true"],[role="textbox"]'; const STOP_RE=/^(stop|cancel)( generating| response| task| run)?$/i; const SEND_RE=/^(send|submit|ask|run|build|generate)( prompt| message| request)?$/i; const APPROVE_RE=/^(allow|approve|confirm|continue|accept)$/i; @@ -20,23 +30,83 @@ if(!el)return ''; return String(el.getAttribute?.('aria-label')||el.getAttribute?.('title')||el.textContent||'').replace(/\s+/g,' ').trim().slice(0,80); } - function buttonLike(el){return !!el?.closest?.('button,[role="button"],input[type="submit"]');} + function controlOf(el){return el?.closest?.('button,[role="button"],input[type="submit"]')||null;} + function buttonLike(el){return !!controlOf(el);} + function testId(el){return String(el?.getAttribute?.('data-testid')||'');} + function usable(el){ + if(!el)return false; + return el.hidden!==true&&String(el.getAttribute?.('aria-hidden')||'').toLowerCase()!=='true'&&el.disabled!==true; + } + function isComposer(el){ + if(!el||typeof el.closest!=='function')return false; + const tag=String(el.tagName||'').toLowerCase(); + return tag==='textarea'||el.isContentEditable===true||String(el.getAttribute?.('contenteditable')||'')==='true' + ||String(el.getAttribute?.('role')||'')==='textbox'; + } + function formHasComposer(form){ + if(!form||typeof form.querySelectorAll!=='function')return false; + return [...form.querySelectorAll(COMPOSER_INPUT_SELECTOR)].some(isComposer); + } + function mainHasComposer(main){return !!main&&typeof main.querySelectorAll==='function'&&[...main.querySelectorAll(COMPOSER_INPUT_SELECTOR)].some(isComposer);} + function composerSurfaceOf(el){ + if(!el||typeof el.closest!=='function')return null; + const marked=el.closest(COMPOSER_CONTAINER_SELECTOR); + if(marked)return marked; + const form=el.closest('form'); + if(form&&formHasComposer(form))return form; + const main=el.closest('main,[role="main"]'); + if(main&&mainHasComposer(main))return main; + return null; + } + function structurallyComposer(el){return isComposer(el)&&!!composerSurfaceOf(el);} + function surfaceRoots(doc){ + if(!doc||typeof doc.querySelectorAll!=='function')return []; + const roots=[]; + for(const composer of doc.querySelectorAll(COMPOSER_INPUT_SELECTOR)){ + if(!isComposer(composer))continue; + const root=composerSurfaceOf(composer); + if(root&&!roots.includes(root))roots.push(root); + } + return roots; + } + function controlInComposerSurface(control){return !!composerSurfaceOf(control);} function hasBusyState(doc){ if(!doc)return false; - if(doc.querySelector('[aria-busy="true"]'))return true; - return [...doc.querySelectorAll('button,[role="button"]')].some(el=>STOP_RE.test(accessibleName(el))); + // Never treat a page-global aria-busy region as an agent run. Prefer the + // nearest structural conversation surface around a real composer. + for(const root of surfaceRoots(doc)){ + if(typeof root.querySelectorAll==='function'&&[...root.querySelectorAll(BUSY_SELECTOR)].some(usable))return true; + } + // Compatibility fallback: a literal English stop control was the old signal. + return [...doc.querySelectorAll('button,[role="button"]')].some(el=>usable(el)&&STOP_RE.test(accessibleName(el))); + } + // Enter is considered a send only in a structurally identified message + // composer. This avoids treating unrelated textareas/editors on agent sites as + // new runs. + function isComposerSend(ev){ + return ev?.key==='Enter'&&!ev.shiftKey&&!ev.altKey&&!ev.ctrlKey&&!ev.metaKey&&!ev.isComposing&&structurallyComposer(ev.target); } function hasVisibleErrorState(doc){return !!doc?.querySelector?.('[role="alert"][aria-live], [role="alert"]');} function hasApprovalState(doc){ if(!doc)return false; for(const root of doc.querySelectorAll('dialog,[role="dialog"]')){ - if([...root.querySelectorAll('button,[role="button"]')].some(el=>APPROVE_RE.test(accessibleName(el))))return true; + if([...root.querySelectorAll('button,[role="button"]')].some(el=>usable(el)&&APPROVE_RE.test(accessibleName(el))))return true; } return false; } - function isSendControl(el){const control=buttonLike(el);return !!control&&SEND_RE.test(accessibleName(control));} - function isStopControl(el){const control=buttonLike(el);return !!control&&STOP_RE.test(accessibleName(control));} - function isApprovalControl(el){const control=buttonLike(el);return !!control&&APPROVE_RE.test(accessibleName(control))&&!!control.closest('dialog,[role="dialog"]');} + function isSendControl(el){ + const control=controlOf(el);if(!control||!usable(control))return false; + if(SEND_TESTID.test(testId(control))&&controlInComposerSurface(control))return true; + const type=String(control.getAttribute?.('type')||'').toLowerCase(); + if(type==='submit'&&control.form&&formHasComposer(control.form))return true; + return SEND_RE.test(accessibleName(control)); + } + function isStopControl(el){ + const control=controlOf(el);if(!control||!usable(control))return false; + if(STOP_TESTID.test(testId(control))&&controlInComposerSurface(control))return true; + return STOP_RE.test(accessibleName(control)); + } + function isApprovalControl(el){const control=controlOf(el);return !!control&&usable(control)&&APPROVE_RE.test(accessibleName(control))&&!!control.closest('dialog,[role="dialog"]');} function safeRunId(){ try{return 'web-'+crypto.randomUUID().replaceAll('-','');}catch(_){return 'web-'+Date.now().toString(36)+Math.random().toString(36).slice(2,12);} } @@ -84,14 +154,15 @@ if(runId&&isStopControl(ev.target)){finish('agent_run_cancelled','stop_control');return;} if(runId&&isApprovalControl(ev.target)){sendLifecycle(provider,'agent_approval_received',runId,'approval_control');approvalSent=true;} },true); - doc.addEventListener('submit',()=>{begin('form_submit');setTimeout(sample,0);},true); + doc.addEventListener('submit',ev=>{if(formHasComposer(ev.target)){begin('form_submit');setTimeout(sample,0);}},true); + doc.addEventListener('keydown',ev=>{if(isComposerSend(ev)){begin('composer_enter');setTimeout(sample,0);}},true); const observer=new MutationObserver(()=>sample()); - const root=doc.documentElement||doc;observer.observe(root,{subtree:true,childList:true,attributes:true,attributeFilter:['aria-busy','aria-live','role','open','disabled']}); + const root=doc.documentElement||doc;observer.observe(root,{subtree:true,childList:true,attributes:true,attributeFilter:['aria-busy','aria-live','role','open','disabled','hidden','aria-hidden','data-is-streaming','data-testid']}); setInterval(()=>{if(!doc.hidden)sample();},1500); sample(); } - globalThis.__OWG_AGENT_SURFACE_ADAPTERS_FOR_TESTS__={PROVIDERS,providerForHost,accessibleName,hasBusyState,hasVisibleErrorState,hasApprovalState,isSendControl,isStopControl,isApprovalControl,structuralPayload}; + globalThis.__OWG_AGENT_SURFACE_ADAPTERS_FOR_TESTS__={PROVIDERS,providerForHost,accessibleName,hasBusyState,hasVisibleErrorState,hasApprovalState,isSendControl,isStopControl,isApprovalControl,isComposerSend,structuralPayload}; if(typeof document!=='undefined'&&typeof MutationObserver!=='undefined'){ if(document.readyState==='loading')document.addEventListener('DOMContentLoaded',()=>start(),{once:true});else start(); } diff --git a/browser_extension/manifest.json b/browser_extension/manifest.json index edb30b0d..4b41dd46 100644 --- a/browser_extension/manifest.json +++ b/browser_extension/manifest.json @@ -1,8 +1,8 @@ { "manifest_version": 3, "name": "Workflow Observer Browser Sensor", - "version": "1.12.0", - "version_name": "1.12.0-v94-agent-lifecycle", + "version": "1.13.0", + "version_name": "1.13.0-v99-structural-agent-detection", "description": "Local-only structural browser telemetry for OpenWorkGraph. Query values/fragments, typed field values, clipboard contents, filenames, file contents, prompts and model responses are not collected.", "incognito": "not_allowed", "permissions": ["tabs", "webNavigation", "storage", "alarms"], diff --git a/collector/main.py b/collector/main.py index 1dcdc455..80cb4325 100644 --- a/collector/main.py +++ b/collector/main.py @@ -8,7 +8,7 @@ import uuid import queue import threading -from datetime import datetime, timezone +from datetime import datetime, timedelta, timezone from pathlib import Path import httpx @@ -21,6 +21,7 @@ ActivityTracker, RawInteraction, RawClipboardAction, + clipboard_change_token, ) from .privacy import should_exclude, title_for_mode from .identity import load_or_create_identity @@ -44,6 +45,18 @@ def utcnow() -> str: return datetime.now(timezone.utc).isoformat() +def _occurred_at(occurred_mono: float) -> str: + """Wall time of an input that happened at ``occurred_mono``. + + Interactions are persisted by a worker thread; stamping them when processed + would shift them by however long the queue took. + """ + if not occurred_mono: + return utcnow() + lag = max(0.0, time.monotonic() - float(occurred_mono)) + return (datetime.now(timezone.utc) - timedelta(seconds=lag)).isoformat() + + def load_config(path: Path) -> dict: cfg = { "device_id": "", @@ -80,6 +93,9 @@ def load_config(path: Path) -> dict: "keyboard_activity_enabled": True, "clipboard_behavior_enabled": True, "clipboard_link_max_seconds": 7200, + # Notice clipboard writes that no copy/cut shortcut explains (menu, + # right-click, drag, apps) from the OS change counter; contents never read. + "clipboard_write_detection_enabled": True, "activity_active_window_seconds": 5, "engaged_grace_seconds": 60, } @@ -412,7 +428,7 @@ def _interaction_event(*, raw: RawInteraction, cfg: dict, session_id: str) -> di return { "event_id": event_id, - "observed_at": utcnow(), + "observed_at": _occurred_at(raw.occurred_mono), **_identity_fields(cfg), "session_id": session_id, "app": state["app"], @@ -441,7 +457,7 @@ def _clipboard_event( metadata: dict = { "source": "desktop", "action": raw.kind, - "evidence_channel": "keyboard_shortcut", + "evidence_channel": "os_clipboard_sequence" if raw.kind == "write" else "keyboard_shortcut", "context_observed_at_interaction": bool(raw.context), "clipboard_contents_captured": False, "clipboard_source_observed": bool(linked_copy_event_id) if raw.kind == "paste" else True, @@ -452,6 +468,11 @@ def _clipboard_event( }, "excluded": state["excluded"], } + if raw.kind == "write": + metadata["interpretation"] = ( + "Something was written to the clipboard without a copy/cut shortcut (menu, right-click, " + "drag, an app or a script). Only that fact is recorded; contents are never read. Not a paste." + ) if raw.clipboard_change_token is not None: metadata["clipboard_change_token"] = int(raw.clipboard_change_token) if transfer_id: @@ -463,7 +484,7 @@ def _clipboard_event( return { "event_id": event_id, - "observed_at": utcnow(), + "observed_at": _occurred_at(raw.occurred_mono), **_identity_fields(cfg), "session_id": session_id, "app": state["app"], @@ -496,7 +517,7 @@ def _interaction_worker( linked_copy_event_id: str | None = None link_age_seconds: float | None = None - if raw.kind in {"copy", "cut"}: + if raw.kind in {"copy", "cut", "write"}: transfer_id = str(uuid.uuid4()) last_clipboard_source = { "event_id": event_id, @@ -623,7 +644,13 @@ def enqueue_interaction(raw: RawInteraction) -> None: except queue.Full: capture_health["interaction_queue_dropped"] += 1 + # A copy/cut shortcut changes the clipboard counter shortly afterwards; that + # change is the shortcut's, not a separate write. + clipboard_shortcut_until = [0.0] + def enqueue_clipboard(raw: RawClipboardAction) -> None: + if raw.kind in {"copy", "cut"}: + clipboard_shortcut_until[0] = raw.occurred_mono + max(3.0, 2 * float(cfg.get("poll_seconds", 2))) raw.context = snapshot_context() try: interaction_q.put(raw, timeout=0.05) @@ -651,6 +678,9 @@ def enqueue_clipboard(raw: RawClipboardAction) -> None: ) keyboard_started = keyboard_sensor.start() + watch_clipboard = bool(cfg.get("clipboard_behavior_enabled", True) and cfg.get("clipboard_write_detection_enabled", True)) + last_clipboard_token = clipboard_change_token() if watch_clipboard else None + # What each sensor needs on macOS. A started listener is not proof: macOS # withholds events from an untrusted process while the thread still runs. mouse_needs = ("accessibility",) @@ -849,6 +879,18 @@ def close_span() -> None: open_span(current_state, current_key, now_wall, now_mono) pending_doc, pending_count, pending_wall, pending_mono, pending_state = pending + if watch_clipboard: + token = clipboard_change_token() + if token is not None and last_clipboard_token is not None and token != last_clipboard_token: + if away: + pass # nobody at the computer: an app or sync wrote it, not the person + elif now_mono > clipboard_shortcut_until[0]: + enqueue_clipboard(RawClipboardAction(kind="write", occurred_mono=now_mono, clipboard_change_token=token)) + else: + clipboard_shortcut_until[0] = 0.0 + if token is not None: + last_clipboard_token = token + if now_mono - last_permission_check >= 60: permissions = sensor_permissions() last_permission_check = now_mono diff --git a/dashboard/agent_control_plane.js b/dashboard/agent_control_plane.js index f9134164..883a0a17 100644 --- a/dashboard/agent_control_plane.js +++ b/dashboard/agent_control_plane.js @@ -106,6 +106,9 @@ }else if(kind==='codex'){ const x=integrations.codex||{};title='Observe Codex'; body=`

Prefer the Observe switch under Connections, which adds this for you unless you already have your own [otel] settings. To do it by hand, merge these keys into your existing [otel] section. The dashboard includes only the dedicated local write-only telemetry credential; it does not grant access to your OWG history.

${privacyHtml(payload)}${codeBox(x.config||'', 'agentSetupCode')}
OpenWorkGraph enables structural OTel logs plus trace export. User prompts, agent responses and guardian assessments remain disabled.
`; + }else if(kind==='vscode'||kind==='gemini_cli'||kind==='cursor'){ + const x=integrations[kind]||{};title=`Observe ${x.label||kind}`; + body=`

Prefer the Observe switch under Connections, which does this for you and never overwrites telemetry you set up yourself. ${h(x.instructions||'')}

${privacyHtml(payload)}${codeBox(JSON.stringify(x.settings||{},null,2),'agentSetupCode')}${kind==='cursor'?'
Hooks answer immediately with a reply that never blocks, then record. OpenWorkGraph never uses a permission hook.
':'
The endpoint path carries a separate write-only token because this app cannot send an Authorization header from its settings file. It cannot read anything.
'}`; }else if(kind==='openai_agents'){ const x=integrations.openai_agents||{};title='Observe OpenAI Agents SDK'; body=`

Add OpenWorkGraph as an additional tracing processor in the agent application's Python environment.

${privacyHtml(payload)}${codeBox(x.python||'', 'agentSetupCode')}
This does not replace existing SDK tracing and does not make OpenWorkGraph a dependency for the agent's control flow.
`; @@ -129,6 +132,9 @@ const text=`${agent.name||''} ${agent.provider||''} ${agent.framework||''}`.toLowerCase(); if(text.includes('claude'))return 'claude_code'; if(text.includes('codex'))return 'codex'; + if(text.includes('copilot'))return 'vscode'; + if(text.includes('gemini'))return 'gemini_cli'; + if(text.includes('cursor'))return 'cursor'; if(text.includes('openai-agents')||text.includes('openai agents'))return 'openai_agents'; return 'otel'; } diff --git a/dashboard/agent_observability_v090.js b/dashboard/agent_observability_v090.js index 2ab0de3c..77f8b7e8 100644 --- a/dashboard/agent_observability_v090.js +++ b/dashboard/agent_observability_v090.js @@ -143,9 +143,9 @@ function patchHiddenNote(){ const note=document.querySelector('#agentHiddenNote');if(!note)return; - const names={'claude-code':'Claude Code','codex':'Codex'}; + const names={'claude-code':'Claude Code','codex':'Codex','github-copilot':'GitHub Copilot','gemini-cli':'Gemini CLI','cursor':'Cursor'}; const hidden=(latest.hidden_frameworks||[]).map(f=>names[f]||f); - note.textContent=hidden.length?`Past runs from ${hidden.join(' and ')} are hidden because Observe is off for ${hidden.length>1?'them':'it'}. Nothing was deleted.`:''; + note.textContent=hidden.length?`Past runs from ${(hidden.length>1?hidden.slice(0,-1).join(', ')+' and '+hidden[hidden.length-1]:hidden[0])} are hidden because Observe is off for ${hidden.length>1?'them':'it'}. Nothing was deleted.`:''; } function patch(){ diff --git a/dashboard/connections.js b/dashboard/connections.js index 93fc3df4..90d052b9 100644 --- a/dashboard/connections.js +++ b/dashboard/connections.js @@ -106,7 +106,8 @@ // Where each native telemetry signal stands since OpenWorkGraph started: // counts and reason codes only (see /v1/agent-telemetry/diagnostics). - const CHANNELS={claude_code:[['claude_code_hooks','Hooks'],['claude_code_otel_logs','OTel logs']],codex:[['codex_otel','OTel']]}; + const CHANNELS={claude_code:[['claude_code_hooks','Hooks'],['claude_code_otel_logs','OTel logs']],codex:[['codex_otel','OTel']], + vscode:[['copilot_otel','Copilot OTel']],gemini_cli:[['gemini_otel','OTel']],cursor:[['cursor_hooks','Hooks']]}; function channelText(label,ch){ if(!ch||!ch.requests)return {text:`${label}: nothing received since OpenWorkGraph started`,warn:false}; const parts=[`${label} ${relativeTime(ch.last_received_at)}`]; @@ -186,7 +187,7 @@ for(const item of activity.items||[]){if(item.client&&item.status==='ok'&&!lastUsed[item.client])lastUsed[item.client]=item.observed_at;} for(const run of traces.executions||[]){ const agent=run?.agent||{},text=`${agent.name||''} ${agent.framework||''}`.toLowerCase(); - const id=text.includes('claude')?'claude_code':text.includes('codex')?'codex':''; + const id=text.includes('claude')?'claude_code':text.includes('codex')?'codex':text.includes('copilot')?'vscode':text.includes('gemini')?'gemini_cli':text.includes('cursor')?'cursor':''; const at=run.ended_at||run.started_at||''; if(id&&(!lastObserved[id]||String(at)>String(lastObserved[id])))lastObserved[id]=at; } diff --git a/docs/CHANGELOG_V099.md b/docs/CHANGELOG_V099.md new file mode 100644 index 00000000..7f6d6df2 --- /dev/null +++ b/docs/CHANGELOG_V099.md @@ -0,0 +1,46 @@ +# OpenWorkGraph (unreleased, planned v0.99): capture coverage + +Builds on v0.98's capture correctness. It widens what OpenWorkGraph can see without changing the privacy model: structure only, no content, the same Observe switches. + +## Observe GitHub Copilot, Gemini CLI and Cursor +The Observe switch now works for three more apps. As with Claude Code and Codex, it backs up the file, keeps your settings, refuses rather than overwrite your own telemetry, and Remove takes out only what OpenWorkGraph added. + +- **GitHub Copilot in VS Code:** + - **How:** Copilot's own OpenTelemetry export, with content capture explicitly off. + - **What is recorded:** agent runs, model calls (model, tokens) and tool calls from Copilot's GenAI spans. Tool arguments and results are never read. + - **When it applies:** after VS Code reloads its window. +- **Gemini CLI:** + - **How:** OTLP over HTTP with `logPrompts` forced **false**. Gemini's default is true, which puts prompts and tool arguments in its log events. + - **What is recorded:** turn starts, model calls, tool calls, and human accept/reject decisions (auto-accept is not a person). + - **Allowlist:** nothing else is read even if prompt logging is re-enabled. +- **Cursor:** + - **How:** hooks for sessions, turns (`beforeSubmitPrompt` → `stop`, one per generation), tool results and subagent outcomes. + - **Never blocks:** Cursor runs hooks synchronously, so the hook answers first with a non-blocking reply. It never uses a permission hook, so it cannot approve or deny anything. +- **Authentication:** + - Copilot and Gemini cannot send an Authorization header from a settings file, so their OTLP endpoint path carries a separate write-only token. + - The access log shows it as `[redacted]`. + - Only OTLP/HTTP JSON is accepted; protobuf gets a clear 415. +- **Diagnostics:** each new app has its own line (Copilot OTel, Gemini OTel, Cursor hooks) in Connections. +- **Agents tab:** the "past runs hidden" note now names only apps whose runs were actually hidden. + +## Clipboard writes +- **What is new:** copies made without a shortcut (menu, right-click, drag, an app) are now noticed from the OS clipboard change counter, as `clipboard_write`. The counter never exposes contents. +- **No double counting:** a Cmd/Ctrl+C that already produced `clipboard_copy` is not counted twice. +- **Paste linking:** a later paste links to whichever write came last. +- **What it never does:** a write is never turned into a paste, since pastes do not change the counter; pastes are still shortcut-only. Nothing is attributed while you are away. +- **Opting out:** `clipboard_write_detection_enabled: false`. +- **Timing fix:** click, scroll and clipboard events are now stamped with the time they happened, not when the worker thread processed them. + +## Web agents in any UI language +The browser sensor recognises ChatGPT, Claude, Gemini, Microsoft Copilot and Lovable runs from structure first: +- `aria-busy` and streaming markers; +- stable `data-testid` tokens for send and stop; +- submit buttons in the composer's form; +- Enter in the message box (not Shift+Enter or IME composition). + +The English button labels are only a fallback, so a Swedish or German UI works whenever the site exposes any of these. The browser sensor version is 1.13.0; reload the extension to update. + +## Not in this release +- Claude Code metrics export. +- Copilot CLI (environment-variable configuration only). +- Platform rework (macOS without `osascript`, Windows UWP app names, Linux/Wayland). diff --git a/docs/CONNECTIONS.md b/docs/CONNECTIONS.md index ef7d6253..2cc9883c 100644 --- a/docs/CONNECTIONS.md +++ b/docs/CONNECTIONS.md @@ -12,16 +12,46 @@ OpenWorkGraph has one list of AI apps, on the dashboard's **Connect** tab under | Claude Code | ✓ | ✓ hooks | `~/.claude.json` (MCP), `~/.claude/settings.json` (hooks) | | Claude Desktop | ✓ | – | `claude_desktop_config.json` in Claude's app-support folder | | Codex | ✓ | ✓ OTel traces | `~/.codex/config.toml` (or `$CODEX_HOME`) | -| Cursor | ✓ | – | `~/.cursor/mcp.json` | -| VS Code + GitHub Copilot | ✓ | – | `mcp.json` in VS Code's user folder (Copilot agent mode uses it) | +| Cursor | ✓ | ✓ hooks | `~/.cursor/mcp.json` (MCP), `~/.cursor/hooks.json` (hooks) | +| VS Code + GitHub Copilot | ✓ | ✓ OTel traces | `mcp.json` (MCP) and `settings.json` (telemetry) in VS Code's user folder | | Windsurf | ✓ | – | `~/.codeium/windsurf/mcp_config.json` | -| Gemini CLI | ✓ | – | `~/.gemini/settings.json` | +| Gemini CLI | ✓ | ✓ OTel logs | `~/.gemini/settings.json` (`mcpServers` and `telemetry`) | | GitHub Copilot CLI | ✓ | – | `~/.copilot/mcp-config.json` (or `$COPILOT_HOME`) | | Kiro | ✓ | – | `~/.kiro/settings/mcp.json` | | Amazon Q Developer | ✓ | – | `~/.aws/amazonq/mcp.json` | **Cloud apps** (ChatGPT, Lovable, Microsoft 365 Copilot) run on their own servers and accept only remote HTTP MCP servers, so they cannot reach OpenWorkGraph on this computer directly. They need a secure tunnel to the optional local HTTP endpoint (or an organization Gateway). Other MCP apps and custom agents (OpenAI Agents SDK, OpenTelemetry, custom events) are under **More ways to connect**. +## How Copilot, Gemini CLI and Cursor are observed + +**GitHub Copilot (VS Code)** +- **Settings written:** Copilot's own OpenTelemetry export, pointed at OpenWorkGraph (`github.copilot.chat.otel.*`), with `captureContent` false. +- **What becomes evidence:** Copilot's agent, model-call and tool spans. Tool arguments and results are never read. +- **When it applies:** after VS Code reloads its window. +- **Refuses (changes nothing) if:** + - Copilot telemetry already goes to your own collector; + - content capture is on; + - `settings.json` contains comments (VS Code allows them, OpenWorkGraph does not rewrite them). Manual setup shows the four keys to add. + +**Gemini CLI** +- **Settings written:** a `telemetry` block in `~/.gemini/settings.json`: `target` local, `otlpProtocol` http, and `logPrompts` **false**. Gemini's default is true, which would put prompts and tool arguments into its log events. +- **What becomes evidence:** turn starts, model calls (model, duration, token counts), tool calls, and human accept/reject decisions. Nothing else is read, even if prompt logging is later turned back on. +- **Refuses (changes nothing) if** you already have your own telemetry settings. + +**Cursor** +- **Hooks added:** OpenWorkGraph's hooks in `~/.cursor/hooks.json`, next to yours: + - `sessionStart` / `sessionEnd` + - `beforeSubmitPrompt` / `stop` (one turn per generation) + - `postToolUse` / `postToolUseFailure` + - `subagentStop` +- **Never blocks:** Cursor runs hooks synchronously, so the hook answers first with a reply that never blocks (`{"continue": true}` for `beforeSubmitPrompt`, `{}` otherwise), then records. Permission hooks (`preToolUse`, `beforeShellExecution`, `beforeMCPExecution`, `beforeReadFile`, `subagentStart`) are never used, so OpenWorkGraph can never approve or deny anything. +- **Never read:** prompts, agent text, tool input/output, file edits, commands, error messages, your email and workspace paths. + +**Authentication without headers** +- Copilot and Gemini CLI cannot send an Authorization header from a settings file, so their endpoint is `/agent-ingest/otlp///v1/…`. The path token is separate and write-only: it can only add structural agent events and cannot read anything. +- The server's access log shows the path as `[redacted]`. +- Only OTLP/HTTP JSON is accepted. Protobuf gets a clear 415, and the diagnostics line reports it. + ## How the switches behave - **First time on.** OpenWorkGraph adds its entry to the app's config file. The app picks it up when it next starts or reloads; the dashboard says which. diff --git a/docs/PRIVACY_AND_DATA.md b/docs/PRIVACY_AND_DATA.md index 9463480e..96ee608c 100644 --- a/docs/PRIVACY_AND_DATA.md +++ b/docs/PRIVACY_AND_DATA.md @@ -57,7 +57,7 @@ Current signals include: - global clicks and throttled scrolls - safe native control identity/label metadata through macOS Accessibility or Windows UI Automation, best effort - browser semantic events such as interactive clicks, editor/input focus, form submit and control change -- occurrence of copy/paste +- occurrence of copy/paste: copy, cut and paste shortcuts, plus clipboard writes without a shortcut (menu, right-click, drag, apps). Writes are noticed from the OS clipboard change counter, which never exposes contents. A write is never recorded as a paste, and nothing is attributed while you are away. - derived task/process/effort structures ## What is deliberately not captured diff --git a/mcpb/manifest.json b/mcpb/manifest.json index 9b32a636..56c97d31 100644 --- a/mcpb/manifest.json +++ b/mcpb/manifest.json @@ -2,9 +2,9 @@ "manifest_version": "0.3", "name": "openworkgraph-local", "display_name": "OpenWorkGraph", - "version": "0.98.0", + "version": "0.99.0", "description": "Connect Claude Desktop to the compact local OpenWorkGraph context surface.", - "long_description": "Uses the OpenWorkGraph installation already running on this computer. v0.98 hardens capture correctness with single-instance recording, local-only away spans, sparse degradation-only health evidence, truthful macOS permission status, debounced same-app document boundaries, Claude Code turn boundaries, and lease-gated durable structural agent delivery. v0.97 added named Gateway administrators, an employee roster, identity-bound personal invitations, optional company sign-in, and an employee /me view of organization-held evidence and recorded reads. v0.96 added an evidence-driven first-value dashboard layer that reconstructs the current session from existing privacy-hardened evidence without adding sensors, permissions, AI access, retention, or MCP capabilities. v0.95 added a framework-neutral custom-harness setup flow plus standalone Python and Node helpers for privacy-safe structural agent telemetry; arbitrary MCP-capable harnesses can separately read authorized OpenWorkGraph context. v0.94 added explicit local history retention, separate saved-history AI access, lightweight history navigation, and structural browser-agent lifecycle observation. Canonical workflow evidence remains primary; Context Pulse provides incremental factual updates. Retained history is separately user-controlled: list_history can navigate saved human and agent sessions only while a time-limited saved-history lease is active. The legacy 24-tool MCP entrypoint remains available for existing configurations while new connections use this compact surface. Prompts, model responses, tool arguments/results, typed text, clipboard contents, exception text, returned values, and hidden reasoning are not captured by the custom agent helpers.", + "long_description": "Uses the OpenWorkGraph installation already running on this computer. v0.99 broadens privacy-safe structural observation to GitHub Copilot, Gemini CLI and Cursor, adds clipboard-write evidence without reading contents, and makes web-agent lifecycle detection rely primarily on structural UI signals rather than English labels. v0.98 hardens capture correctness with single-instance recording, local-only away spans, sparse degradation-only health evidence, truthful macOS permission status, debounced same-app document boundaries, Claude Code turn boundaries, and lease-gated durable structural agent delivery. v0.97 added named Gateway administrators, an employee roster, identity-bound personal invitations, optional company sign-in, and an employee /me view of organization-held evidence and recorded reads. v0.96 added an evidence-driven first-value dashboard layer that reconstructs the current session from existing privacy-hardened evidence without adding sensors, permissions, AI access, retention, or MCP capabilities. v0.95 added a framework-neutral custom-harness setup flow plus standalone Python and Node helpers for privacy-safe structural agent telemetry; arbitrary MCP-capable harnesses can separately read authorized OpenWorkGraph context. v0.94 added explicit local history retention, separate saved-history AI access, lightweight history navigation, and structural browser-agent lifecycle observation. Canonical workflow evidence remains primary; Context Pulse provides incremental factual updates. Retained history is separately user-controlled: list_history can navigate saved human and agent sessions only while a time-limited saved-history lease is active. The legacy 24-tool MCP entrypoint remains available for existing configurations while new connections use this compact surface. Prompts, model responses, tool arguments/results, typed text, clipboard contents, exception text, returned values, and hidden reasoning are not captured by the custom agent helpers.", "author": {"name": "Koyar Afrasyab / Kinvectum"}, "repository": {"type": "git", "url": "https://github.com/KAVentures/openworkgraph"}, "server": {"type": "node", "entry_point": "server/index.js", "mcp_config": {"command": "node", "args": ["${__dirname}/server/index.js"], "env": {}}}, diff --git a/pyproject.toml b/pyproject.toml index 791abbef..a565405e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "workflow-observer" -version = "0.98.0" +version = "0.99.0" description = "Local-first work evidence, self-hosted organizational context gateway, REST API, and MCP access." requires-python = ">=3.11" license = {file = "LICENSE"} diff --git a/sdk/python/pyproject.toml b/sdk/python/pyproject.toml index 6972ed53..c901d2bd 100644 --- a/sdk/python/pyproject.toml +++ b/sdk/python/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "openworkgraph-agent" -version = "0.98.0" +version = "0.99.0" description = "Dependency-free structural telemetry helper for custom OpenWorkGraph agent harnesses" requires-python = ">=3.10" license = {text = "Apache-2.0"} diff --git a/sdk/typescript/package.json b/sdk/typescript/package.json index 6bad59f4..d4494416 100644 --- a/sdk/typescript/package.json +++ b/sdk/typescript/package.json @@ -1,6 +1,6 @@ { "name": "@openworkgraph/agent", - "version": "0.98.0", + "version": "0.99.0", "description": "Dependency-free structural telemetry helper for custom OpenWorkGraph agent harnesses", "type": "module", "exports": { diff --git a/server/agent_auth.py b/server/agent_auth.py index 02044104..29330b22 100644 --- a/server/agent_auth.py +++ b/server/agent_auth.py @@ -22,6 +22,23 @@ def agent_bearer_matches(header: str | None) -> bool: return bool(supplied) and hmac.compare_digest(supplied, target) +def ensure_agent_otlp_path_token(*, directory: Path | None = None) -> str: + """Write-only token carried in the OTLP endpoint *path*. + + VS Code Copilot and Gemini CLI can be pointed at an OTLP endpoint from their + settings files but cannot send an Authorization header from there. This + separate secret authorizes only the OTLP agent-ingest routes, so it can sit in + those settings files (as the bearer token already sits in Claude Code's) and + be rotated independently. Access logs redact it (see server.enterprise_runner). + """ + return _read_or_create_secret(".agent_otlp_path_token", directory=directory) + + +def agent_otlp_path_token_matches(supplied: str | None) -> bool: + supplied = str(supplied or "") + return bool(supplied) and hmac.compare_digest(supplied, ensure_agent_otlp_path_token()) + + def main() -> None: # Explicit administrator/setup action. Printing is intentional here so a # local adapter can be configured without granting the broader API bearer. diff --git a/server/agent_dashboard_control_plane.py b/server/agent_dashboard_control_plane.py index c2461edb..fe3d11aa 100644 --- a/server/agent_dashboard_control_plane.py +++ b/server/agent_dashboard_control_plane.py @@ -41,6 +41,8 @@ def _claude_settings(request: Request) -> dict[str, Any]: def agent_setup_payload(request: Request) -> dict[str, Any]: + from .agent_observe_presets import manual_setup_material + base_url = _base_url(request) token = ensure_agent_ingest_token() otel_endpoint = f"{base_url}/agent-ingest/v1/otel" @@ -78,7 +80,7 @@ def agent_setup_payload(request: Request) -> dict[str, Any]: "otel_endpoint": f"{base_url}/agent-ingest/v1/claude-otel", "telemetry_depth": "rich_structural", "events": [ - "SessionStart", "SessionEnd", "PostToolUse", "PostToolUseFailure", + "SessionStart", "SessionEnd", "UserPromptSubmit", "Stop", "PostToolUse", "PostToolUseFailure", "PermissionRequest", "PermissionDenied", "SubagentStart", "SubagentStop", "StopFailure", "claude_code.user_prompt", "claude_code.api_request", "claude_code.api_error", "claude_code.api_refusal", "claude_code.tool_result", @@ -102,6 +104,7 @@ def agent_setup_payload(request: Request) -> dict[str, Any]: "content_logging_enabled": False, "telemetry_depth": "rich_native_events_plus_trace", }, + **manual_setup_material(), "openai_agents": { "label": "OpenAI Agents SDK", "method": "additional_tracing_processor", diff --git a/server/agent_execution_trace_routes.py b/server/agent_execution_trace_routes.py index 2ae6175a..4e77a990 100644 --- a/server/agent_execution_trace_routes.py +++ b/server/agent_execution_trace_routes.py @@ -74,7 +74,10 @@ def get_agent_execution_traces( raw = [row for row in raw if (_parse(row.get("observed_at")) is not None and _parse(row.get("observed_at")) < end)] if hide_disconnected: from .connections import hidden_frameworks - hidden = hidden_frameworks() + switched_off = hidden_frameworks() + # Report only frameworks whose runs were actually hidden, not every + # connector that happens to be off or never installed. + hidden = {fw for fw in switched_off if any(_row_framework(row) == fw for row in raw)} if hidden: raw = [row for row in raw if _row_framework(row) not in hidden] payload = agent_execution_traces(raw, family_key=family_key, execution_id=execution_id, limit=limit, max_events_per_execution=max_events_per_execution) payload = enrich_agent_execution_payload(payload, raw) diff --git a/server/agent_ingest.py b/server/agent_ingest.py index 54a0b11c..b8bb92b1 100644 --- a/server/agent_ingest.py +++ b/server/agent_ingest.py @@ -10,6 +10,7 @@ from shared.capture_control import filter_recordable from shared.claude_otel_adapter import claude_otel_to_agent_events from shared.codex_otel_adapter import codex_otel_to_agent_events +from shared.gemini_otel_adapter import gemini_otel_to_agent_events from shared.history_policy import retention_for_kind from shared.otel_agent_adapter import otel_payload_to_agent_events from .db import insert_events @@ -150,3 +151,37 @@ def ingest_claude_otel_payload( "projected": len(events), "inserted": inserted, } + + +def ingest_gemini_otel_payload( + payload: dict[str, Any], + *, + defaults: dict[str, Any] | None = None, +) -> dict[str, int]: + """Ingest Gemini CLI OTLP logs through a strict structural allowlist.""" + _bounded_json_size({"payload": payload, "defaults": defaults or {}}) + projected, stats = gemini_otel_to_agent_events(payload, defaults=defaults, max_records=MAX_CLAUDE_OTEL_RECORDS) + if len(projected) > MAX_AGENT_EVENTS * 2: + raise AgentEvidenceError("Gemini CLI OpenTelemetry projection produced too many agent events") + events = _validated_events(projected) + inserted = _insert_recordable(events) + return { + "records_seen": int(stats.get("records_seen") or 0), + "records_ignored": int(stats.get("records_ignored") or 0), + "projected": len(events), + "inserted": inserted, + } + + +def count_otlp_records(payload: dict[str, Any]) -> int: + """Records in an OTLP JSON body of any signal (for signals we accept but do not store).""" + total = 0 + for key, scopes, items in ( + ("resourceSpans", "scopeSpans", "spans"), + ("resourceLogs", "scopeLogs", "logRecords"), + ("resourceMetrics", "scopeMetrics", "metrics"), + ): + for resource in payload.get(key) or []: + for scope in (resource or {}).get(scopes) or [] if isinstance(resource, dict) else []: + total += len((scope or {}).get(items) or []) if isinstance(scope, dict) else 0 + return total diff --git a/server/agent_observe_presets.py b/server/agent_observe_presets.py new file mode 100644 index 00000000..45707fb3 --- /dev/null +++ b/server/agent_observe_presets.py @@ -0,0 +1,359 @@ +from __future__ import annotations + +"""Observe presets for GitHub Copilot (VS Code), Gemini CLI and Cursor. + +Same rules as the other connectors (server.agent_config_writer): back up before +writing, keep every other setting, never overwrite a setting the user chose for +themselves, and refuse (changing nothing) when a file is not plain JSON. + +* Copilot: VS Code user settings enable Copilot's OpenTelemetry export to + OpenWorkGraph with content capture explicitly off. Copilot environment + variables take precedence over user settings, so conflicting overrides that + are visible to OpenWorkGraph are detected before configuration is claimed. +* Gemini CLI: ~/.gemini/settings.json ``telemetry`` block exports OTLP over + HTTP to OpenWorkGraph with ``logPrompts`` explicitly false (Gemini's default is + true, which would put prompts and tool arguments into its log events). +* Cursor: ~/.cursor/hooks.json gets OpenWorkGraph's non-blocking hooks next to + any hooks the user already has. + +Copilot and Gemini cannot send an Authorization header from a settings file, +so their endpoint path carries a separate write-only token. +""" + +import json +import os +from pathlib import Path +from typing import Any + +from . import agent_config_writer as writer +from .agent_config_writer import ConfigConflict + + +def otlp_base_url(source: str) -> str: + from adapters._agent_client import _base_url + + from .agent_auth import ensure_agent_otlp_path_token + + return f"{_base_url()}/agent-ingest/otlp/{source}/{ensure_agent_otlp_path_token()}" + + +def _owned_otlp_endpoint(source: str, value: Any) -> bool: + """Whether an endpoint is an OpenWorkGraph endpoint for this source. + + Ownership deliberately ignores the final token segment. The write-only path + token can rotate (for example after an auth reset); an older OWG endpoint must + remain repairable/removable instead of being mistaken for a foreign collector. + The local/base URL and source still have to match exactly. + """ + from adapters._agent_client import _base_url + + endpoint = str(value or "").strip().rstrip("/") + prefix = f"{_base_url().rstrip('/')}/agent-ingest/otlp/{source}/" + if not endpoint.startswith(prefix): + return False + token = endpoint[len(prefix):] + return bool(token) and "/" not in token + + +def _load(path: Path, what: str) -> dict[str, Any]: + if not path.exists(): + return {} + raw = path.read_text(encoding="utf-8") + if not raw.strip(): + return {} + try: + data = json.loads(raw) + except json.JSONDecodeError as exc: + raise ConfigConflict( + f"{path} is not plain JSON ({what} allows comments, which OpenWorkGraph will not rewrite); " + "use manual setup" + ) from exc + if not isinstance(data, dict): + raise ConfigConflict(f"{path} is not a JSON object; use manual setup") + return data + + +def _write(path: Path, data: dict[str, Any]) -> str | None: + backup = writer._backup(path) + writer._atomic_write(path, json.dumps(data, indent=2, ensure_ascii=False) + "\n") + return backup + + +# --- GitHub Copilot in VS Code -------------------------------------------------------------- + +def copilot_settings() -> dict[str, Any]: + return { + "github.copilot.chat.otel.enabled": True, + "github.copilot.chat.otel.exporterType": "otlp-http", + "github.copilot.chat.otel.otlpEndpoint": otlp_base_url("copilot"), + "github.copilot.chat.otel.captureContent": False, + } + + +def _truthy_env(name: str) -> bool: + return str(os.getenv(name) or "").strip().lower() in {"1", "true", "yes", "on"} + + +def _false_env(name: str) -> bool: + return str(os.getenv(name) or "").strip().lower() in {"0", "false", "no", "off"} + + +def _copilot_environment_conflict(desired: dict[str, Any]) -> str: + """Return a visible Copilot OTel override that would beat OWG's user settings. + + VS Code/Copilot documents these environment variables as higher precedence + than user settings. We can only inspect the environment OpenWorkGraph itself + inherited; managed enterprise settings or a differently launched VS Code can + still differ, which is why actual received telemetry remains the final health + signal in Connections. + """ + ours = str(desired["github.copilot.chat.otel.otlpEndpoint"]).rstrip("/") + endpoint = str(os.getenv("COPILOT_OTEL_ENDPOINT") or os.getenv("OTEL_EXPORTER_OTLP_ENDPOINT") or "").strip().rstrip("/") + if endpoint and endpoint != ours: + return ( + "Copilot OTel has an environment endpoint override that takes precedence over VS Code settings; " + "OpenWorkGraph will not claim Observe is configured until that override is removed or points to OpenWorkGraph." + ) + if _false_env("COPILOT_OTEL_ENABLED"): + return "COPILOT_OTEL_ENABLED disables Copilot telemetry and takes precedence over the VS Code Observe setting." + if _truthy_env("COPILOT_OTEL_CAPTURE_CONTENT"): + return ( + "COPILOT_OTEL_CAPTURE_CONTENT is enabled in the environment. OpenWorkGraph refuses to configure Copilot " + "while an environment override would make the client transmit prompt/response content." + ) + protocol = str(os.getenv("COPILOT_OTEL_PROTOCOL") or os.getenv("OTEL_EXPORTER_OTLP_PROTOCOL") or "").strip().lower() + if protocol == "grpc": + return "Copilot OTel is forced to gRPC by an environment override; OpenWorkGraph's local Copilot endpoint is OTLP/HTTP." + return "" + + +def copilot_status(path: Path) -> dict[str, Any]: + try: + data = _load(path, "VS Code settings.json") + except ConfigConflict as exc: + return {"configured": False, "path": str(path), "error": str(exc)} + desired = copilot_settings() + configured = all(data.get(k) == v for k, v in desired.items()) + if configured: + conflict = _copilot_environment_conflict(desired) + if conflict: + raise ConfigConflict(conflict) + return {"configured": configured, "path": str(path)} + + +def copilot_connect(path: Path) -> dict[str, Any]: + data = _load(path, "VS Code settings.json") + desired = copilot_settings() + conflict = _copilot_environment_conflict(desired) + if conflict: + raise ConfigConflict(conflict) + ours = str(desired["github.copilot.chat.otel.otlpEndpoint"]) + endpoint = data.get("github.copilot.chat.otel.otlpEndpoint") + # Preserve a user's collector even when Copilot telemetry is currently + # disabled. Disabled is not consent for OWG to replace a configured endpoint. + # A stale OWG endpoint, however, is ours and is safely repaired to the current + # write-only token below. + if endpoint not in (None, "", ours) and not _owned_otlp_endpoint("copilot", endpoint): + raise ConfigConflict( + "VS Code already points Copilot telemetry at another endpoint; OpenWorkGraph will not redirect it. " + "Use manual setup or an OTLP collector that forwards to OpenWorkGraph." + ) + if data.get("github.copilot.chat.otel.captureContent") is True: + raise ConfigConflict( + "Copilot content capture is on in VS Code settings; OpenWorkGraph only connects with content capture off." + ) + data.update(desired) + backup = _write(path, data) + return {"configured": True, "path": str(path), "backup": backup, "note": "Takes effect after VS Code reloads (Developer: Reload Window)."} + + +def copilot_disconnect(path: Path) -> dict[str, Any]: + data = _load(path, "VS Code settings.json") + desired = copilot_settings() + endpoint_key = "github.copilot.chat.otel.otlpEndpoint" + removed = [] + for key, value in desired.items(): + if key not in data: + continue + if key == endpoint_key: + if _owned_otlp_endpoint("copilot", data.get(key)): + removed.append(key) + elif data.get(key) == value: + removed.append(key) + for key in removed: + del data[key] + backup = _write(path, data) if removed else None + return {"configured": False, "path": str(path), "removed": len(removed), "backup": backup} + + +# --- Gemini CLI ---------------------------------------------------------------------------- + +def gemini_telemetry() -> dict[str, Any]: + return { + "enabled": True, + "target": "local", + "otlpEndpoint": otlp_base_url("gemini"), + "otlpProtocol": "http", + "logPrompts": False, + } + + +def gemini_status(path: Path) -> dict[str, Any]: + try: + data = _load(path, "Gemini settings.json") + except ConfigConflict as exc: + return {"configured": False, "path": str(path), "error": str(exc)} + telemetry = data.get("telemetry") if isinstance(data.get("telemetry"), dict) else {} + return {"configured": all(telemetry.get(k) == v for k, v in gemini_telemetry().items()), "path": str(path)} + + +def gemini_connect(path: Path) -> dict[str, Any]: + data = _load(path, "Gemini settings.json") + desired = gemini_telemetry() + existing = data.get("telemetry") + if existing is not None and not isinstance(existing, dict): + raise ConfigConflict(f"'telemetry' in {path} has an unexpected shape; use manual setup") + existing = dict(existing or {}) + # Only take over a telemetry block that is ours, empty, or switched off. An + # old OWG endpoint remains ours after path-token rotation and is repairable. + foreign = sorted( + k for k, v in existing.items() + if not ( + desired.get(k) == v + or (k == "enabled" and v is False) + or (k == "otlpEndpoint" and _owned_otlp_endpoint("gemini", v)) + ) + ) + if foreign: + raise ConfigConflict( + f"Gemini CLI already has its own telemetry settings ({', '.join(foreign)}); " + "OpenWorkGraph will not change them. Use manual setup." + ) + data["telemetry"] = {**existing, **desired} + backup = _write(path, data) + return {"configured": True, "path": str(path), "backup": backup, "note": "Takes effect in new Gemini CLI sessions."} + + +def gemini_disconnect(path: Path) -> dict[str, Any]: + data = _load(path, "Gemini settings.json") + telemetry = data.get("telemetry") if isinstance(data.get("telemetry"), dict) else None + if telemetry is None: + return {"configured": False, "path": str(path), "removed": 0, "backup": None} + desired = gemini_telemetry() + removed = [] + for key, value in desired.items(): + if key not in telemetry: + continue + if key == "otlpEndpoint": + if _owned_otlp_endpoint("gemini", telemetry.get(key)): + removed.append(key) + elif telemetry.get(key) == value: + removed.append(key) + for key in removed: + del telemetry[key] + if not telemetry: + del data["telemetry"] + backup = _write(path, data) if removed else None + return {"configured": False, "path": str(path), "removed": len(removed), "backup": backup} + + +# --- Cursor -------------------------------------------------------------------------------- + +def _is_owg_cursor_handler(handler: Any) -> bool: + from adapters.cursor_hook import OWG_MARKER + + return isinstance(handler, dict) and OWG_MARKER in str(handler.get("command") or "") + + +def _load_cursor(path: Path) -> dict[str, Any]: + data = _load(path, "Cursor hooks.json") + hooks = data.get("hooks") + if hooks is not None and not isinstance(hooks, dict): + raise ConfigConflict(f"'hooks' in {path} has an unexpected shape; use manual setup") + for event, handlers in (hooks or {}).items(): + if not isinstance(handlers, list): + raise ConfigConflict(f"hooks.{event} in {path} has an unexpected shape; use manual setup") + return data + + +def _strip_cursor(data: dict[str, Any]) -> int: + removed = 0 + hooks = data.get("hooks") or {} + for event in list(hooks): + kept = [h for h in hooks[event] if not _is_owg_cursor_handler(h)] + removed += len(hooks[event]) - len(kept) + if kept: + hooks[event] = kept + else: + del hooks[event] + return removed + + +def cursor_status(path: Path) -> dict[str, Any]: + from shared.cursor_hook_adapter import SUPPORTED_EVENTS + + try: + data = _load_cursor(path) + except ConfigConflict as exc: + return {"configured": False, "path": str(path), "error": str(exc)} + present = {event for event, handlers in (data.get("hooks") or {}).items() if any(_is_owg_cursor_handler(h) for h in handlers)} + return {"configured": set(SUPPORTED_EVENTS) <= present, "path": str(path)} + + +def cursor_connect(path: Path) -> dict[str, Any]: + from adapters.cursor_hook import hooks_fragment + + data = _load_cursor(path) + version = data.get("version", 1) + if version != 1: + raise ConfigConflict(f"{path} uses hooks version {version!r}; OpenWorkGraph writes version 1. Use manual setup.") + _strip_cursor(data) + data["version"] = 1 + hooks = data.setdefault("hooks", {}) + for event, handlers in hooks_fragment()["hooks"].items(): + hooks.setdefault(event, []).extend(handlers) + backup = _write(path, data) + return {"configured": True, "path": str(path), "backup": backup, "note": "Takes effect after Cursor restarts."} + + +def cursor_disconnect(path: Path) -> dict[str, Any]: + data = _load_cursor(path) + removed = _strip_cursor(data) + if not data.get("hooks"): + data.pop("hooks", None) + backup = _write(path, data) if removed else None + return {"configured": False, "path": str(path), "removed": removed, "backup": backup} + + +# --- manual setup (shown by the dashboard's Manual setup button) --------------------------- + +def manual_setup_material() -> dict[str, Any]: + """Manual equivalents of the Copilot, Gemini CLI and Cursor Observe switches.""" + from adapters.cursor_hook import hooks_fragment + + return { + "vscode": { + "label": "GitHub Copilot (VS Code)", + "method": "copilot_otel_http_json", + "one_click": True, + "settings": copilot_settings(), + "instructions": "Use the Observe switch on the VS Code + GitHub Copilot row, or merge these keys into VS Code's user settings.json and reload the window. Content capture stays off. Environment or enterprise-managed OTel settings can override user settings; Connections reports whether telemetry actually arrives.", + "content_logging_enabled": False, + }, + "gemini_cli": { + "label": "Gemini CLI", + "method": "gemini_otel_http_json_logs", + "one_click": True, + "settings": {"telemetry": gemini_telemetry()}, + "instructions": "Use the Observe switch on the Gemini CLI row, or merge this telemetry block into ~/.gemini/settings.json. logPrompts must stay false (Gemini's default is true).", + "content_logging_enabled": False, + }, + "cursor": { + "label": "Cursor", + "method": "cursor_hooks", + "one_click": True, + "settings": hooks_fragment(), + "instructions": "Use the Observe switch on the Cursor row, or merge these hooks into ~/.cursor/hooks.json and restart Cursor. Only non-blocking hooks are used; OpenWorkGraph never answers a permission hook.", + "content_logging_enabled": False, + }, + } diff --git a/server/agent_routes.py b/server/agent_routes.py index 6cc08b01..1ed08310 100644 --- a/server/agent_routes.py +++ b/server/agent_routes.py @@ -8,16 +8,18 @@ from shared.agent_evidence import AgentEvidenceError from . import agent_telemetry_diagnostics as diagnostics -from .agent_auth import agent_bearer_matches +from .agent_auth import agent_bearer_matches, agent_otlp_path_token_matches from .agent_ingest import ( MAX_AGENT_BATCH_BYTES, ingest_agent_payloads, ingest_claude_otel_payload, ingest_codex_otel_payload, + ingest_gemini_otel_payload, + count_otlp_records, ingest_otel_payload, ) from .agent_read_auth import agent_read_authorized -from .connections import is_enabled +from .connections import FRAMEWORK_CLIENTS, is_enabled from .agent_workflows import agent_workflow_view from .agent_execution_trace_routes import router as agent_execution_trace_router from .context_execution_routes import router as context_execution_router @@ -48,12 +50,9 @@ class OTelDefaults(BaseModel): workflow_id: str = "" -_SWITCHABLE_FRAMEWORKS = {"claude-code": "claude_code", "codex": "codex"} - - def _observation_enabled_for(event: Any) -> bool: framework = str(event.get("framework") or "") if isinstance(event, dict) else "" - client = _SWITCHABLE_FRAMEWORKS.get(framework) + client = FRAMEWORK_CLIENTS.get(framework) return is_enabled(client, "observe") if client else True @@ -93,7 +92,7 @@ async def _read_bounded_json(request: Request) -> Any: def _events_channel(payload: Any) -> str: events = payload.get("events") if isinstance(payload, dict) else None first = events[0] if isinstance(events, list) and events and isinstance(events[0], dict) else {} - return "claude_code_hooks" if str(first.get("framework") or "") == "claude-code" else "agent_events" + return {"claude-code": "claude_code_hooks", "cursor": "cursor_hooks"}.get(str(first.get("framework") or ""), "agent_events") async def _authorized_json(request: Request, channel: str | None) -> Any: @@ -119,7 +118,7 @@ async def ingest_agent_events(request: Request) -> dict[str, int | str]: # The native client names its channel so an auth failure is attributable # before the body is read; otherwise classify by the events themselves. hint = str(request.headers.get("x-owg-channel") or "") - channel = hint if hint in {"claude_code_hooks", "agent_events"} else "" + channel = hint if hint in {"claude_code_hooks", "cursor_hooks", "agent_events"} else "" payload = await _authorized_json(request, channel or None) if not channel: channel = _events_channel(payload) @@ -200,6 +199,90 @@ async def ingest_claude_otel(request: Request) -> Response: return Response(status_code=202) +# Clients that can be pointed at an OTLP endpoint from their settings file but +# cannot send an Authorization header from there. The path carries a separate +# write-only token instead (server.agent_auth.ensure_agent_otlp_path_token). +OTLP_SOURCES: dict[str, dict[str, str]] = { + "copilot": {"client": "vscode", "framework": "github-copilot", "agent_name": "GitHub Copilot", + "provider": "github", "channel": "copilot_otel"}, + "gemini": {"client": "gemini_cli", "framework": "gemini-cli", "agent_name": "Gemini CLI", + "provider": "google", "channel": "gemini_otel"}, +} +MAX_OTLP_DECOMPRESSED_BYTES = MAX_AGENT_BATCH_BYTES + + +async def _read_otlp_json(request: Request, channel: str) -> Any: + content_type = str(request.headers.get("content-type") or "").lower() + if "protobuf" in content_type: + diagnostics.rejected(channel, "protobuf_unsupported") + raise HTTPException(status_code=415, detail="OpenWorkGraph accepts OTLP/HTTP JSON only; use the http/json protocol") + body = bytearray() + async for chunk in request.stream(): + body.extend(chunk) + if len(body) > MAX_AGENT_BATCH_BYTES: + diagnostics.rejected(channel, "too_large") + raise HTTPException(status_code=413, detail=f"agent request exceeds {MAX_AGENT_BATCH_BYTES} bytes") + raw = bytes(body) + if "gzip" in str(request.headers.get("content-encoding") or "").lower(): + import zlib + + try: + inflater = zlib.decompressobj(16 + zlib.MAX_WBITS) + raw = inflater.decompress(raw, MAX_OTLP_DECOMPRESSED_BYTES + 1) + except zlib.error as exc: + diagnostics.rejected(channel, "invalid_payload") + raise HTTPException(status_code=400, detail="invalid gzip body") from exc + if len(raw) > MAX_OTLP_DECOMPRESSED_BYTES or inflater.unconsumed_tail: + diagnostics.rejected(channel, "too_large") + raise HTTPException(status_code=413, detail="decompressed agent request is too large") + # A decodable prefix is not a valid gzip request. Reject truncated streams + # and concatenated/trailing members rather than accepting partial JSON. + if not inflater.eof or inflater.unused_data: + diagnostics.rejected(channel, "invalid_payload") + raise HTTPException(status_code=400, detail="invalid gzip body") + try: + payload = json.loads(raw.decode("utf-8")) + except Exception as exc: + diagnostics.rejected(channel, "invalid_payload") + raise HTTPException(status_code=400, detail="invalid OTLP JSON payload") from exc + if not isinstance(payload, dict): + diagnostics.rejected(channel, "not_an_object") + raise HTTPException(status_code=422, detail="OpenTelemetry payload must be an object") + return payload + + +@router.post("/agent-ingest/otlp/{source}/{token}/v1/{signal}") +async def ingest_path_token_otlp(source: str, token: str, signal: str, request: Request) -> dict[str, Any]: + spec = OTLP_SOURCES.get(source) + if spec is None or signal not in {"traces", "logs", "metrics"}: + raise HTTPException(status_code=404, detail="unknown OTLP source") + channel = spec["channel"] + diagnostics.received(channel) + if not agent_otlp_path_token_matches(token): + diagnostics.rejected(channel, "auth") + raise HTTPException(status_code=401, detail="agent ingest authentication required") + payload = await _read_otlp_json(request, channel) + if not is_enabled(spec["client"], "observe"): + diagnostics.observation_off(channel) + return {} + defaults = {"framework": spec["framework"], "agent_name": spec["agent_name"], "provider": spec["provider"]} + try: + if signal == "traces": + result = ingest_otel_payload(payload, defaults=defaults) + elif signal == "logs" and source == "gemini": + result = ingest_gemini_otel_payload(payload, defaults=defaults) + else: + # Metrics, and Copilot's logs (which repeat its spans), are accepted so + # the exporter stays healthy, and counted, but nothing is stored. + count = count_otlp_records(payload) + result = {"records_seen": count, "records_ignored": count, "projected": 0, "inserted": 0} + except (AgentEvidenceError, ValueError) as exc: + diagnostics.rejected(channel, "adapter_error") + raise HTTPException(status_code=422, detail=str(exc)) from exc + diagnostics.processed(channel, result) + return {} # the OTLP/HTTP JSON success response + + @router.get("/v1/agent-telemetry/diagnostics") def get_agent_telemetry_diagnostics(request: Request) -> dict[str, Any]: """Per-channel delivery counts so a missing signal can be located, not guessed.""" diff --git a/server/agent_telemetry_diagnostics.py b/server/agent_telemetry_diagnostics.py index 0bef6bbd..189c79c4 100644 --- a/server/agent_telemetry_diagnostics.py +++ b/server/agent_telemetry_diagnostics.py @@ -22,9 +22,12 @@ "codex_otel": "Codex OpenTelemetry", "agent_events": "Other agent events (SDKs, adapters)", "otel_generic": "Generic OpenTelemetry", + "copilot_otel": "GitHub Copilot OpenTelemetry", + "gemini_otel": "Gemini CLI OpenTelemetry", + "cursor_hooks": "Cursor hooks", "spool": "Delayed delivery (agent spool)", } -REJECTION_REASONS = ("auth", "invalid_payload", "too_large", "not_an_object", "adapter_error") +REJECTION_REASONS = ("auth", "invalid_payload", "too_large", "not_an_object", "adapter_error", "protobuf_unsupported") _LOCK = threading.Lock() _STARTED_AT = datetime.now(timezone.utc).isoformat() diff --git a/server/connections.py b/server/connections.py index dc6977f1..12796316 100644 --- a/server/connections.py +++ b/server/connections.py @@ -312,6 +312,33 @@ def _claude_observe() -> Target: ) +def _preset_target(path_fn: Callable[[], Path], name: str, applies: str) -> Target: + from server import agent_observe_presets as presets + + status = getattr(presets, f"{name}_status") + connect = getattr(presets, f"{name}_connect") + disconnect = getattr(presets, f"{name}_disconnect") + return Target( + path=path_fn, + installed=lambda: bool(status(path_fn()).get("configured")), + install=lambda: connect(path_fn()), + remove=lambda: disconnect(path_fn()), + applies=applies, + ) + + +def _copilot_observe() -> Target: + return _preset_target(lambda: _app_support("Code", "User", "settings.json"), "copilot", "after VS Code reloads its window") + + +def _gemini_observe() -> Target: + return _preset_target(lambda: _home() / ".gemini" / "settings.json", "gemini", "in new Gemini CLI sessions") + + +def _cursor_observe() -> Target: + return _preset_target(lambda: _home() / ".cursor" / "hooks.json", "cursor", "after Cursor restarts") + + def _codex_observe() -> Target: def install() -> dict[str, Any]: from adapters._agent_client import _base_url @@ -365,11 +392,11 @@ def _claude_code_json() -> Path: Client("cursor", "Cursor", lambda: _home() / ".cursor", _json_target(lambda: _home() / ".cursor" / "mcp.json", "mcpServers", "cursor", "plain", "after Cursor reloads its MCP servers"), - None, "Cursor does not expose an execution trace to observe."), + _cursor_observe), Client("vscode", "VS Code + GitHub Copilot", lambda: _app_support("Code", "User"), _json_target(lambda: _app_support("Code", "User", "mcp.json"), "servers", "vscode", "vscode", "after VS Code reloads its MCP servers"), - None, "VS Code does not expose an execution trace to observe."), + _copilot_observe), Client("windsurf", "Windsurf", lambda: _home() / ".codeium" / "windsurf", _json_target(lambda: _home() / ".codeium" / "windsurf" / "mcp_config.json", "mcpServers", "windsurf", "plain", "after Windsurf refreshes its MCP servers"), @@ -377,7 +404,7 @@ def _claude_code_json() -> Path: Client("gemini_cli", "Gemini CLI", lambda: _home() / ".gemini", _json_target(lambda: _home() / ".gemini" / "settings.json", "mcpServers", "gemini_cli", "plain", "in new Gemini CLI sessions"), - None, "Gemini CLI observation is not supported yet."), + _gemini_observe), Client("copilot_cli", "GitHub Copilot CLI", _copilot_home, _json_target(lambda: _copilot_home() / "mcp-config.json", "mcpServers", "copilot_cli", "copilot_cli", "in new Copilot CLI sessions"), @@ -463,7 +490,13 @@ def is_enabled(client_id: str | None, kind: str) -> bool: # Agent frameworks whose observation is controlled by a client switch. -FRAMEWORK_CLIENTS = {"claude-code": "claude_code", "codex": "codex"} +FRAMEWORK_CLIENTS = { + "claude-code": "claude_code", + "codex": "codex", + "github-copilot": "vscode", + "gemini-cli": "gemini_cli", + "cursor": "cursor", +} def observation_active(client_id: str) -> bool: diff --git a/server/enterprise_runner.py b/server/enterprise_runner.py index caf6b617..6a1c31c8 100644 --- a/server/enterprise_runner.py +++ b/server/enterprise_runner.py @@ -30,6 +30,9 @@ def main() -> None: import server.dashboard_privacy # noqa: F401 import server.agent_capture_runtime as agent_capture_runtime + import server.log_redaction as log_redaction + + log_redaction.install() org_join_routes.start_managed_setup_in_background() agent_capture_runtime.start() diff --git a/server/log_redaction.py b/server/log_redaction.py new file mode 100644 index 00000000..e8d9f84b --- /dev/null +++ b/server/log_redaction.py @@ -0,0 +1,28 @@ +from __future__ import annotations + +"""Keep the OTLP path token out of the server's access log.""" + +import logging +import re + +_OTLP_TOKEN = re.compile(r"(/agent-ingest/otlp/[^/\s]+/)[^/\s?]+") + + +def redact(text: str) -> str: + return _OTLP_TOKEN.sub(r"\1[redacted]", text) + + +class RedactOtlpPathToken(logging.Filter): + def filter(self, record: logging.LogRecord) -> bool: + if isinstance(record.args, tuple): + record.args = tuple(redact(a) if isinstance(a, str) else a for a in record.args) + elif isinstance(record.msg, str): + record.msg = redact(record.msg) + return True + + +def install() -> None: + """Attach to uvicorn's access logger; uvicorn's logging setup keeps logger filters.""" + logger = logging.getLogger("uvicorn.access") + if not any(isinstance(f, RedactOtlpPathToken) for f in logger.filters): + logger.addFilter(RedactOtlpPathToken()) diff --git a/shared/cursor_hook_adapter.py b/shared/cursor_hook_adapter.py new file mode 100644 index 00000000..6c639a74 --- /dev/null +++ b/shared/cursor_hook_adapter.py @@ -0,0 +1,142 @@ +from __future__ import annotations + +"""Translate Cursor agent hooks into structural OpenWorkGraph evidence. + +Cursor hook payloads can contain the prompt, attachments, agent text, tool +inputs/outputs, file edits, shell commands, error messages, the user's email, +workspace paths and transcript paths. This adapter reads only identifiers, +the hook name, the tool name, statuses and durations; nothing else is copied. + + sessionStart / sessionEnd -> session starts / finishes (conversation_id) + beforeSubmitPrompt / stop -> turn starts / finishes (generation_id) + postToolUse / postToolUseFailure -> tool_call + subagentStop -> handoff to a subagent, with its outcome + +Permission hooks (preToolUse, beforeShellExecution, beforeMCPExecution, +beforeReadFile, subagentStart) are never registered: they must return a +decision, and an observer must not be able to block or approve anything. +""" + +import hashlib +import re +from datetime import datetime, timezone +from typing import Any + +SUPPORTED_EVENTS = ( + "sessionStart", + "sessionEnd", + "beforeSubmitPrompt", + "postToolUse", + "postToolUseFailure", + "subagentStop", + "stop", +) +_SAFE_LABEL = re.compile(r"^[A-Za-z][A-Za-z0-9_.:/ -]{0,159}$") +_STATUS = {"completed": "success", "success": "success", "aborted": "cancelled", "cancelled": "cancelled", + "error": "error", "failed": "error"} + + +def _text(value: Any, limit: int = 200) -> str: + return re.sub(r"\s+", " ", str(value or "")).strip()[:limit] + + +def _label(value: Any, *, default: str, limit: int = 100) -> str: + raw = _text(value, limit) + return raw if raw and _SAFE_LABEL.fullmatch(raw) else default + + +def _event_id(*parts: Any) -> str: + material = "\x1f".join(_text(p, 300) for p in parts).encode("utf-8") + return "cursor-hook:" + hashlib.sha256(material).hexdigest()[:40] + + +def _duration(payload: dict[str, Any]) -> float: + for key in ("duration_ms", "duration"): # Cursor reports milliseconds + try: + value = float(payload.get(key)) + except Exception: + continue + return round(max(0.0, min(value / 1000.0, 7 * 24 * 3600)), 6) + return 0.0 + + +def _category(name: str) -> str: + low = name.lower() + if low.startswith("mcp"): + return "mcp" + if any(t in low for t in ("shell", "terminal", "command", "bash")): + return "shell" + if any(t in low for t in ("read", "edit", "write", "file", "delete")): + return "filesystem" + if any(t in low for t in ("search", "grep", "glob", "find", "codebase")): + return "search" + if any(t in low for t in ("web", "browser", "fetch")): + return "browser" + return "other" if name else "none" + + +def cursor_hook_to_agent_events(payload: dict[str, Any], *, observed_at: str | None = None) -> list[dict[str, Any]]: + if not isinstance(payload, dict): + return [] + hook = _text(payload.get("hook_event_name"), 80) + if hook not in SUPPORTED_EVENTS: + return [] + conversation = _text(payload.get("conversation_id") or payload.get("session_id"), 128) + if not conversation: + return [] + generation = _text(payload.get("generation_id"), 128) + timestamp = _text(observed_at, 80) or datetime.now(timezone.utc).isoformat() + + def event(operation: str, status: str, run_id: str, key: str, *, tool: str = "", span: str = "") -> dict[str, Any]: + return { + "event_id": _event_id(conversation, run_id, key, span), + "observed_at": timestamp, + "sensor_id": "agent:cursor-hook", + "session_id": conversation, + "agent_name": "Cursor", + "provider": "cursor", + "framework": "cursor", + "model": _label(payload.get("model"), default="", limit=160), + "operation": operation, + "status": status, + "observation_level": "native_trace", + "run_id": run_id, + "trace_id": conversation, + "span_id": span, + "parent_span_id": "", + "tool_name": tool, + "tool_category": _category(tool), + "duration_seconds": _duration(payload), + } + + if hook == "sessionStart": + return [event("run_started", "running", conversation, hook)] + if hook == "sessionEnd": + status = _STATUS.get(_text(payload.get("final_status") or payload.get("reason"), 40).lower(), "unknown") + return [event("run_finished", status, conversation, hook)] + # Turn-level hooks need the generation: falling back to the conversation would + # make a turn's stop look like the whole session finishing. + if not generation: + return [] + if hook == "beforeSubmitPrompt": + return [event("run_started", "running", generation, hook)] + if hook == "stop": + status = _STATUS.get(_text(payload.get("status"), 40).lower(), "unknown") + return [event("run_finished", status, generation, hook)] + if hook in {"postToolUse", "postToolUseFailure"}: + tool = _label(payload.get("tool_name"), default="unknown-tool", limit=160) + span = _text(payload.get("tool_use_id"), 128) + if hook == "postToolUse": + status = "success" + elif payload.get("is_interrupt") is True: + status = "cancelled" + elif "den" in _text(payload.get("failure_type"), 40).lower(): + status = "denied" + else: + status = "error" + return [event("tool_call", status, generation, hook, tool=tool, span=span)] + if hook == "subagentStop": + kind = _label(payload.get("subagent_type"), default="subagent", limit=80) + status = _STATUS.get(_text(payload.get("status"), 40).lower(), "unknown") + return [event("handoff", status, generation, hook, tool=f"subagent:{kind}", span=_text(payload.get("subagent_id"), 128))] + return [] diff --git a/shared/gemini_otel_adapter.py b/shared/gemini_otel_adapter.py new file mode 100644 index 00000000..53559663 --- /dev/null +++ b/shared/gemini_otel_adapter.py @@ -0,0 +1,214 @@ +from __future__ import annotations + +"""Translate Gemini CLI OpenTelemetry log events into structural agent events. + +Gemini CLI emits one log record per event, named by the ``event.name`` +attribute. With ``logPrompts`` on (Gemini's default), the same records also +carry prompt, request/response text and tool arguments. OpenWorkGraph turns +``logPrompts`` off when it configures Gemini CLI, and independently reads only +the allowlisted attributes below, so content never enters OpenWorkGraph even if +a user re-enables prompt logging. + +Mapped: + gemini_cli.user_prompt -> run_started (turn marker = prompt_id) + gemini_cli.api_response -> model_call (model, duration, token counts) + gemini_cli.api_error -> model_call (status error) + gemini_cli.tool_call -> tool_call (+ human approval decision) + gemini_cli.agent.start/finish -> nested run lifecycle (agent_id) + gemini_cli.conversation_finished -> session run_finished (session.id) + +Gemini currently does not expose a reliable per-prompt "turn finished" log. +OpenWorkGraph therefore does not manufacture one from an API response: a model +response can be followed by tools and more model calls. Session and explicit +agent-run finishes are recorded when Gemini emits them. +""" + +import hashlib +from typing import Any + +from .claude_otel_adapter import ( + _duration_seconds, + _hash, + _iter_records, + _observed_at, + _safe_label, + _text, + _tool_category, + _value, +) + +_SUPPORTED = { + "gemini_cli.user_prompt", + "gemini_cli.api_response", + "gemini_cli.api_error", + "gemini_cli.tool_call", + "gemini_cli.conversation_finished", + "gemini_cli.agent.start", + "gemini_cli.agent.finish", +} +# Gemini's ToolCallDecision: accept/reject/modify are a person's choice; +# auto_accept is policy and is not a human decision. +_HUMAN_DECISIONS = {"accept": "success", "reject": "denied", "modify": "success"} + + +def _int(value: Any) -> int | None: + try: + number = int(_value(value) if isinstance(value, dict) else value) + except Exception: + return None + return number if 0 <= number <= 1_000_000_000 else None + + +def _usage(attrs: dict[str, Any]) -> dict[str, int]: + mapping = { + "input_tokens": "input_token_count", + "output_tokens": "output_token_count", + "cached_input_tokens": "cached_content_token_count", + "total_tokens": "total_token_count", + } + out = {} + for target, source in mapping.items(): + value = _int(attrs.get(source)) + if value is not None: + out[target] = value + return out + + +def _event_id(run_id: str, key: str, attrs: dict[str, Any], native: dict[str, Any]) -> str: + # One id per native record (timestamp is per record), so exporter retries are + # idempotent and several model calls in one prompt stay distinct. + stamp = _text(attrs.get("event.timestamp"), 80) or _text(native.get("timeUnixNano"), 40) + material = f"{run_id}\x1f{key}\x1f{stamp}".encode("utf-8") + return "gemini-otel:" + hashlib.sha256(material).hexdigest()[:40] + + +def _event_name(attrs: dict[str, Any]) -> str: + raw = _text(attrs.get("event.name"), 120).lower() + return raw if raw.startswith("gemini_cli.") else (f"gemini_cli.{raw}" if raw else "") + + +def _finish_status(reason: Any) -> str: + """Use only explicit failure/cancellation words; otherwise do not overclaim success.""" + low = _text(reason, 80).lower() + if any(token in low for token in ("error", "fail", "exception")): + return "error" + if any(token in low for token in ("cancel", "abort", "interrupt")): + return "cancelled" + if any(token in low for token in ("deny", "reject")): + return "denied" + return "unknown" + + +def gemini_otel_to_agent_events( + payload: dict[str, Any], + *, + defaults: dict[str, Any] | None = None, + max_records: int = 1000, +) -> tuple[list[dict[str, Any]], dict[str, int]]: + if not isinstance(payload, dict): + raise ValueError("Gemini CLI OpenTelemetry payload must be an object") + defaults = dict(defaults or {}) + events: list[dict[str, Any]] = [] + seen = ignored = 0 + + for native, attrs, resource_attrs in _iter_records(payload, max_records): + seen += 1 + name = _event_name(attrs) + if name not in _SUPPORTED: + ignored += 1 + continue + session_id = _text(attrs.get("session.id") or resource_attrs.get("session.id"), 128) + prompt_id = _text(attrs.get("prompt_id"), 128) + default_run_id = prompt_id or session_id + if not session_id and not default_run_id: + ignored += 1 + continue + + def base( + operation: str, + status: str, + *, + run_id: str | None = None, + tool_name: str = "", + span_id: str = "", + key: str = name, + duration_ms: Any | None = None, + ) -> dict[str, Any]: + rid = _text(run_id or default_run_id, 128) + return { + "event_id": _event_id(rid, key, attrs, native), + "observed_at": _observed_at(attrs, native), + "organization_id": _text(defaults.get("organization_id"), 128), + "actor_id": _text(defaults.get("actor_id"), 128), + "device_id": _text(defaults.get("device_id"), 128) or "gemini-local", + "sensor_id": "agent:gemini-cli-otel", + "session_id": session_id or rid, + "agent_name": "Gemini CLI", + "provider": "google", + "framework": "gemini-cli", + "model": _safe_label(attrs.get("model"), limit=200), + "operation": operation, + "status": status, + "observation_level": "native_trace", + "run_id": rid, + "trace_id": session_id or rid, + "span_id": span_id, + "tool_name": tool_name, + "tool_category": _tool_category(tool_name, attrs) if tool_name else "none", + "duration_seconds": _duration_seconds(attrs.get("duration_ms") if duration_ms is None else duration_ms), + "usage": _usage(attrs) if operation == "model_call" else {}, + } + + if name == "gemini_cli.user_prompt": + if not prompt_id: + ignored += 1 + continue + events.append(base("run_started", "running", run_id=prompt_id)) + elif name == "gemini_cli.api_response": + code = _int(attrs.get("status_code")) + events.append(base("model_call", "success" if code in (None, 200) else "error")) + elif name == "gemini_cli.api_error": + events.append(base("model_call", "error")) + elif name == "gemini_cli.tool_call": + tool = _safe_label(attrs.get("function_name"), default="unknown-tool", limit=160) + if str(_value(attrs.get("tool_type")) or "") == "mcp": + server = _safe_label(attrs.get("mcp_server_name"), default="", limit=80) + tool = f"mcp__{server}__{tool}" if server else f"mcp__{tool}" + call_key = "tool:" + _hash(f"{prompt_id}|{tool}|{_observed_at(attrs, native)}") + raw_success = _value(attrs.get("success")) + success = str(raw_success).lower() if raw_success is not None else "" + tool_status = "success" if success == "true" else "error" if success == "false" else "unknown" + events.append(base("tool_call", tool_status, tool_name=tool, span_id=call_key)) + decision = _text(attrs.get("decision"), 40).lower() + if decision in _HUMAN_DECISIONS: + events.append(base( + "human_approval_received", _HUMAN_DECISIONS[decision], + tool_name=tool, span_id=call_key, key=name + ":decision", + )) + elif name == "gemini_cli.conversation_finished": + if not session_id: + ignored += 1 + continue + events.append(base("run_finished", "unknown", run_id=session_id)) + elif name == "gemini_cli.agent.start": + agent_id = _text(attrs.get("agent_id"), 128) + if not agent_id: + ignored += 1 + continue + events.append(base("run_started", "running", run_id=agent_id)) + elif name == "gemini_cli.agent.finish": + agent_id = _text(attrs.get("agent_id"), 128) + if not agent_id: + ignored += 1 + continue + events.append(base( + "run_finished", + _finish_status(attrs.get("terminate_reason")), + run_id=agent_id, + duration_ms=attrs.get("duration_ms"), + )) + + unique: dict[str, dict[str, Any]] = {} + for event in events: + unique.setdefault(str(event["event_id"]), event) + return list(unique.values()), {"records_seen": seen, "records_ignored": ignored, "agent_events": len(unique)} diff --git a/tests/conftest.py b/tests/conftest.py index 9a902b97..02baaf8a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -33,6 +33,7 @@ def _isolate_agent_ingestion_database(request): "test_agent_native_routes_v060.py", "test_agent_ingress_boundaries_v084.py", "test_agent_delivery_v098.py", + "test_agent_coverage_v099.py", } if Path(str(request.fspath)).name not in isolated_modules: yield diff --git a/tests/js/agent_surface_structural.test.mjs b/tests/js/agent_surface_structural.test.mjs new file mode 100644 index 00000000..a64a2ba9 --- /dev/null +++ b/tests/js/agent_surface_structural.test.mjs @@ -0,0 +1,95 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; + +// Minimal element stand-ins: only what the detector reads (attributes, tag, +// closest, form/main/container relationships). Text is used only for the old +// English fallback. +function el({tag = 'button', attrs = {}, text = '', form = null, main = null, container = null, editable = false} = {}) { + const node = { + tagName: tag.toUpperCase(), textContent: text, form, isContentEditable: editable, + getAttribute: (k) => (k in attrs ? attrs[k] : null), + closest(sel) { + if (sel === 'form') return node.form || null; + if (sel === 'main,[role="main"]') return main || null; + if (sel.includes('data-testid*="composer"')) return container || null; + if (sel === 'dialog,[role="dialog"]') return null; + return /button|role="button"|input\[type="submit"\]/.test(sel) && (tag === 'button' || attrs.role === 'button') ? node : null; + }, + }; + return node; +} + +function composerForm({busy = false} = {}) { + const busyNode = {hidden: false, disabled: false, getAttribute: () => null}; + const form = { + querySelectorAll(sel) { + if (sel.includes('textarea')) return [composer]; + if (sel.includes('aria-busy') || sel.includes('data-is-streaming')) return busy ? [busyNode] : []; + return []; + }, + }; + const composer = el({tag: 'textarea', form}); + return {form, composer}; +} + +new Function(fs.readFileSync('browser_extension/agent_surface_adapters.js', 'utf8'))(); +const api = globalThis.__OWG_AGENT_SURFACE_ADAPTERS_FOR_TESTS__; + +test('send is recognised structurally without reading a localized label', () => { + const {form} = composerForm(); + assert.equal(api.isSendControl(el({attrs: {'data-testid': 'send-button', 'aria-label': 'Skicka prompt'}, form})), true); + assert.equal(api.isSendControl(el({attrs: {type: 'submit', 'aria-label': 'Skicka'}, form})), true); + // A Swedish label with no structural signal is not guessed from text... + assert.equal(api.isSendControl(el({attrs: {'aria-label': 'Skicka'}})), false); + // ...while the English compatibility fallback still works. + assert.equal(api.isSendControl(el({attrs: {'aria-label': 'Send'}})), true); +}); + +test('Enter in the message box sends; Shift+Enter, IME composition and other fields do not', () => { + const {form} = composerForm(); + const box = el({tag: 'div', attrs: {contenteditable: 'true'}, editable: true, form}); + form.querySelectorAll = (sel) => sel.includes('textarea') ? [box] : []; + assert.equal(api.isComposerSend({key: 'Enter', target: box}), true); + assert.equal(api.isComposerSend({key: 'Enter', shiftKey: true, target: box}), false); + assert.equal(api.isComposerSend({key: 'Enter', isComposing: true, target: box}), false); + assert.equal(api.isComposerSend({key: 'a', target: box}), false); + assert.equal(api.isComposerSend({key: 'Enter', target: el({tag: 'input', attrs: {type: 'text'}})}), false); +}); + +test('busy state is scoped to a real conversation/composer surface', () => { + const {form, composer} = composerForm({busy: true}); + const streaming = { + querySelectorAll(sel) { + if (sel.includes('textarea')) return [composer]; + if (sel.includes('button')) return []; + return []; + }, + }; + assert.equal(api.hasBusyState(streaming), true); + + // A page-global loading spinner with no composer is not an agent run. + const unrelatedBusy = { + querySelectorAll(sel) { + if (sel.includes('aria-busy')) return [{hidden: false, disabled: false, getAttribute: () => null}]; + return []; + }, + }; + assert.equal(api.hasBusyState(unrelatedBusy), false); + + const {form: stopForm} = composerForm(); + assert.equal(api.isStopControl(el({attrs: {'data-testid': 'stop-button', 'aria-label': 'Stoppa'}, form: stopForm})), true); + assert.equal(api.isStopControl(el({attrs: {'aria-label': 'Stoppa'}})), false); + void form; +}); + +test('a structural send/stop token outside a composer surface is ignored', () => { + assert.equal(api.isSendControl(el({attrs: {'data-testid': 'send-button', 'aria-label': 'Skicka'}})), false); + assert.equal(api.isStopControl(el({attrs: {'data-testid': 'stop-button', 'aria-label': 'Stoppa'}})), false); +}); + +test('no class-name selectors and no content in lifecycle messages', () => { + const source = fs.readFileSync('browser_extension/agent_surface_adapters.js', 'utf8'); + assert.doesNotMatch(source, /querySelector(All)?\([^\n]*['"`]\.[A-Za-z]/); + assert.doesNotMatch(source, /\.value\b|innerText/); +}); diff --git a/tests/test_agent_coverage_v099.py b/tests/test_agent_coverage_v099.py new file mode 100644 index 00000000..003d5054 --- /dev/null +++ b/tests/test_agent_coverage_v099.py @@ -0,0 +1,322 @@ +from __future__ import annotations + +"""Copilot, Gemini CLI and Cursor observation: endpoints, presets and hooks.""" + +import gzip +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest +from fastapi import FastAPI +from fastapi.testclient import TestClient + +from server import agent_observe_presets as presets +from server import agent_telemetry_diagnostics as diagnostics +from server.agent_config_writer import ConfigConflict +from shared.cursor_hook_adapter import SUPPORTED_EVENTS, cursor_hook_to_agent_events + +ROOT = Path(__file__).resolve().parents[1] +SECRET = "TOP-SECRET-CONTENT" + + +def _kv(key, value): + if isinstance(value, bool): + v = {"boolValue": value} + elif isinstance(value, int): + v = {"intValue": str(value)} + else: + v = {"stringValue": str(value)} + return {"key": key, "value": v} + + +def _span(name_op, span_id, attrs, start=1_790_000_000_000_000_000, dur=1_000_000_000, parent=""): + return { + "traceId": "a" * 32, "spanId": span_id, "parentSpanId": parent, "name": f"{name_op} {SECRET}", + "startTimeUnixNano": str(start), "endTimeUnixNano": str(start + dur), + "attributes": [_kv("gen_ai.operation.name", name_op), *[_kv(k, v) for k, v in attrs.items()]], + "status": {"code": 1}, + } + + +def copilot_traces() -> dict: + return {"resourceSpans": [{"resource": {"attributes": [_kv("service.name", "copilot-chat")]}, "scopeSpans": [{ + "scope": {"name": "copilot-chat"}, + "spans": [ + _span("invoke_agent", "1" * 16, {"gen_ai.agent.name": "GitHub Copilot", "gen_ai.conversation.id": "conv-1", + "gen_ai.request.model": "gpt-5"}, dur=5_000_000_000), + _span("chat", "2" * 16, {"gen_ai.request.model": "gpt-5", "gen_ai.usage.input_tokens": 120, + "gen_ai.usage.output_tokens": 30, "gen_ai.input.messages": SECRET}, parent="1" * 16), + _span("execute_tool", "3" * 16, {"gen_ai.tool.name": "run_in_terminal", "gen_ai.tool.call.arguments": SECRET, + "gen_ai.tool.call.result": SECRET}, parent="1" * 16), + ], + }]}]} + + +def gemini_logs() -> dict: + def rec(**attrs): + return {"timeUnixNano": "1790000000000000000", "attributes": [_kv(k, v) for k, v in attrs.items()]} + return {"resourceLogs": [{"resource": {"attributes": [_kv("session.id", "g-sess")]}, "scopeLogs": [{"logRecords": [ + rec(**{"event.name": "gemini_cli.user_prompt", "event.timestamp": "2026-09-28T10:00:00.000Z", "prompt_id": "gp1", "prompt": SECRET}), + rec(**{"event.name": "gemini_cli.api_response", "event.timestamp": "2026-09-28T10:00:01.000Z", "prompt_id": "gp1", + "model": "gemini-2.5-pro", "duration_ms": 900, "status_code": 200, "input_token_count": 10, "response_text": SECRET}), + rec(**{"event.name": "gemini_cli.tool_call", "event.timestamp": "2026-09-28T10:00:02.000Z", "prompt_id": "gp1", + "function_name": "run_shell_command", "function_args": SECRET, "success": True, "decision": "auto_accept"}), + ]}]}]} + + +@pytest.fixture() +def app_client(tmp_path, monkeypatch): + monkeypatch.setenv("OWG_CONNECTIONS_HOME", str(tmp_path / "home")) + from server.agent_routes import router + + app = FastAPI() + app.include_router(router) + diagnostics.reset_for_tests() + with TestClient(app) as client: + yield client + + +def _url(source: str, token: str | None = None, signal: str = "traces") -> str: + from server.agent_auth import ensure_agent_otlp_path_token + + return f"/agent-ingest/otlp/{source}/{token or ensure_agent_otlp_path_token()}/v1/{signal}" + + +def _stored(framework: str) -> list[dict]: + from server.db import rows + + return [r for r in rows("SELECT * FROM events WHERE source = 'agent'") if r["metadata"]["agent"].get("framework") == framework] + + +# --- the OTLP endpoint for settings-file clients ------------------------------------------------ + +def test_copilot_traces_become_structural_runs_without_content(app_client): + from server.db import init_db + + init_db() + response = app_client.post(_url("copilot"), json=copilot_traces()) + assert response.status_code == 200 and response.json() == {} + stored = _stored("github-copilot") + ops = sorted(r["metadata"]["operation"] for r in stored) + assert ops == ["model_call", "run_finished", "run_started", "tool_call"] + assert any(r["metadata"]["tool"]["name"] == "run_in_terminal" for r in stored) + assert SECRET not in json.dumps(stored) + ch = diagnostics.snapshot()["channels"]["copilot_otel"] + assert ch["requests"] == 1 and ch["events_stored"] == 4 + + +def test_gemini_logs_and_gzip_are_accepted(app_client): + from server.db import init_db + + init_db() + body = gzip.compress(json.dumps(gemini_logs()).encode()) + response = app_client.post(_url("gemini", signal="logs"), content=body, + headers={"Content-Type": "application/json", "Content-Encoding": "gzip"}) + assert response.status_code == 200 + stored = _stored("gemini-cli") + assert sorted(r["metadata"]["operation"] for r in stored) == ["model_call", "run_started", "tool_call"] + assert SECRET not in json.dumps(stored) # prompt, response text and arguments never copied + # auto_accept is policy, not a person: no approval event + assert not any(r["metadata"]["operation"] == "human_approval_received" for r in stored) + + +def test_endpoint_rejects_wrong_token_protobuf_and_unknown_sources(app_client): + assert app_client.post(_url("copilot", token="wrong"), json={}).status_code == 401 + assert app_client.post(_url("copilot"), content=b"\x0a\x01", headers={"Content-Type": "application/x-protobuf"}).status_code == 415 + assert app_client.post(_url("nosuch"), json={}).status_code == 404 + assert app_client.post(_url("copilot", signal="profiles"), json={}).status_code == 404 + ch = diagnostics.snapshot()["channels"]["copilot_otel"] + assert ch["rejected"] == {"auth": 1, "protobuf_unsupported": 1} + + +def test_metrics_are_accepted_counted_and_not_stored(app_client): + from server.db import init_db + + init_db() + before = len(_stored("github-copilot")) + body = {"resourceMetrics": [{"scopeMetrics": [{"metrics": [{"name": "copilot_chat.session.count"}, {"name": "x"}]}]}]} + assert app_client.post(_url("copilot", signal="metrics"), json=body).status_code == 200 + assert len(_stored("github-copilot")) == before + ch = diagnostics.snapshot()["channels"]["copilot_otel"] + assert ch["records_seen"] == 2 and ch["records_ignored"] == 2 and ch["events_stored"] == 0 + + +def test_observe_switch_off_stores_nothing(app_client, monkeypatch): + import server.agent_routes as routes + from server.db import init_db + + init_db() + original = routes.is_enabled + monkeypatch.setattr(routes, "is_enabled", lambda client, kind: False if client == "vscode" else original(client, kind)) + before = len(_stored("github-copilot")) + assert app_client.post(_url("copilot"), json=copilot_traces()).status_code == 200 + assert len(_stored("github-copilot")) == before + assert diagnostics.snapshot()["channels"]["copilot_otel"]["observation_off"] == 1 + + +def test_access_log_never_shows_the_path_token(): + from server.log_redaction import redact + + line = '127.0.0.1:1 - "POST /agent-ingest/otlp/gemini/abcDEF123_-xyz/v1/logs HTTP/1.1" 200' + assert "abcDEF123" not in redact(line) and "/agent-ingest/otlp/gemini/[redacted]/v1/logs" in redact(line) + + +# --- presets --------------------------------------------------------------------------------- + +def test_copilot_preset_merges_refuses_and_removes_only_its_keys(tmp_path): + path = tmp_path / "Code" / "User" / "settings.json" + path.parent.mkdir(parents=True) + path.write_text(json.dumps({"editor.fontSize": 14})) + presets.copilot_connect(path) + data = json.loads(path.read_text()) + assert data["editor.fontSize"] == 14 and data["github.copilot.chat.otel.captureContent"] is False + assert data["github.copilot.chat.otel.enabled"] is True + assert "/agent-ingest/otlp/copilot/" in data["github.copilot.chat.otel.otlpEndpoint"] + assert presets.copilot_status(path)["configured"] is True + presets.copilot_disconnect(path) + assert json.loads(path.read_text()) == {"editor.fontSize": 14} + + path.write_text(json.dumps({"github.copilot.chat.otel.enabled": True, "github.copilot.chat.otel.otlpEndpoint": "http://my-collector:4318"})) + with pytest.raises(ConfigConflict, match="another endpoint"): + presets.copilot_connect(path) + path.write_text(json.dumps({"github.copilot.chat.otel.captureContent": True})) + with pytest.raises(ConfigConflict, match="content capture"): + presets.copilot_connect(path) + jsonc = '{\n // my settings\n "editor.fontSize": 14,\n}\n' + path.write_text(jsonc) + with pytest.raises(ConfigConflict, match="not plain JSON"): + presets.copilot_connect(path) + assert path.read_text() == jsonc + + +def test_gemini_preset_forces_prompt_logging_off_and_respects_own_telemetry(tmp_path): + path = tmp_path / ".gemini" / "settings.json" + path.parent.mkdir() + path.write_text(json.dumps({"mcpServers": {"x": {}}})) + presets.gemini_connect(path) + data = json.loads(path.read_text()) + assert data["mcpServers"] == {"x": {}} + assert data["telemetry"]["logPrompts"] is False and data["telemetry"]["otlpProtocol"] == "http" + assert data["telemetry"]["target"] == "local" and data["telemetry"]["enabled"] is True + presets.gemini_disconnect(path) + assert "telemetry" not in json.loads(path.read_text()) + + path.write_text(json.dumps({"telemetry": {"enabled": False}})) # switched off: fine to take over + presets.gemini_connect(path) + path.write_text(json.dumps({"telemetry": {"enabled": True, "target": "gcp"}})) + with pytest.raises(ConfigConflict, match="own telemetry"): + presets.gemini_connect(path) + path.write_text(json.dumps({"telemetry": {"logPrompts": True}})) + with pytest.raises(ConfigConflict): + presets.gemini_connect(path) + + +def test_cursor_preset_keeps_user_hooks_and_never_uses_permission_hooks(tmp_path): + path = tmp_path / ".cursor" / "hooks.json" + path.parent.mkdir() + mine = {"command": "./audit.sh"} + path.write_text(json.dumps({"version": 1, "hooks": {"stop": [mine], "beforeShellExecution": [mine]}})) + presets.cursor_connect(path) + presets.cursor_connect(path) # idempotent: no duplicates + data = json.loads(path.read_text()) + assert data["hooks"]["stop"][0] == mine and len(data["hooks"]["stop"]) == 2 + assert data["hooks"]["beforeShellExecution"] == [mine] + for permission_hook in ("preToolUse", "beforeShellExecution", "beforeMCPExecution", "beforeReadFile", "subagentStart"): + assert permission_hook not in SUPPORTED_EVENTS + assert not any("adapters.cursor_hook" in h["command"] for h in data["hooks"].get(permission_hook, [])) + assert presets.cursor_status(path)["configured"] is True + presets.cursor_disconnect(path) + assert json.loads(path.read_text()) == {"version": 1, "hooks": {"stop": [mine], "beforeShellExecution": [mine]}} + + +# --- Cursor hooks -------------------------------------------------------------------------------- + +def test_cursor_mapping_turns_sessions_tools_and_no_content(): + common = {"conversation_id": "c1", "generation_id": "g1", "model": "claude-4.5-sonnet", "user_email": SECRET, + "workspace_roots": [SECRET], "transcript_path": SECRET} + events = [] + for hook, extra in ( + ("sessionStart", {"session_id": "c1"}), + ("beforeSubmitPrompt", {"prompt": SECRET, "attachments": [SECRET]}), + ("postToolUse", {"tool_name": "Shell", "tool_input": {"command": SECRET}, "tool_output": SECRET, "tool_use_id": "t1", "duration": 1500}), + ("postToolUseFailure", {"tool_name": "Edit", "tool_use_id": "t2", "error_message": SECRET, "failure_type": "permission_denied"}), + ("subagentStop", {"subagent_type": "explore", "status": "completed", "task": SECRET, "summary": SECRET}), + ("stop", {"status": "aborted"}), + ("sessionEnd", {"final_status": "completed", "error_message": SECRET}), + ): + events += cursor_hook_to_agent_events({**common, "hook_event_name": hook, **extra}, observed_at="2026-09-28T10:00:00Z") + summary = [(e["operation"], e["status"], e["run_id"], e["tool_name"]) for e in events] + assert summary == [ + ("run_started", "running", "c1", ""), + ("run_started", "running", "g1", ""), + ("tool_call", "success", "g1", "Shell"), + ("tool_call", "denied", "g1", "Edit"), + ("handoff", "success", "g1", "subagent:explore"), + ("run_finished", "cancelled", "g1", ""), + ("run_finished", "success", "c1", ""), + ] + assert events[2]["duration_seconds"] == 1.5 and events[2]["tool_category"] == "shell" + assert SECRET not in json.dumps(events) + # A turn hook without a generation never falls back to the conversation. + assert cursor_hook_to_agent_events({"hook_event_name": "stop", "conversation_id": "c1", "status": "completed"}) == [] + from shared.agent_evidence import agent_event_to_evidence + + for event in events: + agent_event_to_evidence(event) # every event is valid evidence + + +@pytest.mark.parametrize("hook,expected", [("beforeSubmitPrompt", {"continue": True}), ("postToolUse", {}), ("stop", {})]) +def test_cursor_hook_process_answers_immediately_and_never_blocks(tmp_path, hook, expected): + env = {**os.environ, "WORKFLOW_OBSERVER_API": "http://127.0.0.1:9", "WORKFLOW_OBSERVER_AUTH_DIR": str(tmp_path)} + payload = json.dumps({"hook_event_name": hook, "conversation_id": "c", "generation_id": "g", "prompt": SECRET, "tool_name": "Shell"}) + done = subprocess.run([sys.executable, "-m", "adapters.cursor_hook"], input=payload, capture_output=True, text=True, cwd=ROOT, env=env, timeout=30) + assert done.returncode == 0 and json.loads(done.stdout) == expected and SECRET not in done.stdout + done.stderr + broken = subprocess.run([sys.executable, "-m", "adapters.cursor_hook"], input="{not json", capture_output=True, text=True, cwd=ROOT, env=env, timeout=30) + assert broken.returncode == 0 and json.loads(broken.stdout) == {} + + +def test_manual_setup_material_matches_the_switches(): + material = presets.manual_setup_material() + source = (ROOT / "server" / "agent_dashboard_control_plane.py").read_text() + assert "**manual_setup_material()," in source + assert set(material) == {"vscode", "gemini_cli", "cursor"} + assert material["vscode"]["settings"]["github.copilot.chat.otel.captureContent"] is False + assert material["gemini_cli"]["settings"]["telemetry"]["logPrompts"] is False + assert set(material["cursor"]["settings"]["hooks"]) == set(SUPPORTED_EVENTS) + assert all(item["content_logging_enabled"] is False for item in material.values()) + + +def test_ephemeral_history_keeps_cursor_and_copilot_sessions_across_turns(app_client, tmp_path, monkeypatch): + """Turn finishes (Cursor stop, Copilot invoke_agent) must not purge the session.""" + # Own data folder: the retention setting must not leak into later tests. + monkeypatch.setenv("WORKFLOW_OBSERVER_DATA", str(tmp_path / "data")) + from server.agent_ingest import ingest_agent_payloads + from server.db import init_db, rows + from shared.history_policy import update_retention + + init_db() + update_retention(human_mode="ephemeral", human_days=None, agent_mode="ephemeral", agent_days=None) + + def cursor(hook, **extra): + return cursor_hook_to_agent_events({"hook_event_name": hook, "conversation_id": "eph-cur", **extra}) + + def count(session): + return len([r for r in rows("SELECT session_id FROM events WHERE source = 'agent'") if r["session_id"] == session]) + + ingest_agent_payloads(cursor("sessionStart") + cursor("beforeSubmitPrompt", generation_id="g1") + cursor("stop", generation_id="g1", status="completed")) + ingest_agent_payloads(cursor("beforeSubmitPrompt", generation_id="g2")) + assert count("eph-cur") == 4 + ingest_agent_payloads(cursor("sessionEnd", final_status="completed")) + assert count("eph-cur") == 0 + + app_client.post(_url("copilot"), json=copilot_traces()) + second = copilot_traces() + for span in second["resourceSpans"][0]["scopeSpans"][0]["spans"]: + span["traceId"] = "b" * 32 + app_client.post(_url("copilot"), json=second) + # Two agent invocations in one conversation: both kept in full (the root span + # carries the conversation; its model/tool spans are grouped by trace). + assert count("conv-1") == 4 and count("a" * 32) == 2 and count("b" * 32) == 2 diff --git a/tests/test_capture_coverage_hardening_v099.py b/tests/test_capture_coverage_hardening_v099.py new file mode 100644 index 00000000..4d23c57e --- /dev/null +++ b/tests/test_capture_coverage_hardening_v099.py @@ -0,0 +1,111 @@ +from __future__ import annotations + +import gzip +import json + +import pytest +from fastapi import FastAPI +from fastapi.testclient import TestClient + +from server import agent_observe_presets as presets +from server.agent_config_writer import ConfigConflict +from shared.gemini_otel_adapter import gemini_otel_to_agent_events + + +def test_copilot_preserves_foreign_endpoint_even_when_disabled(tmp_path): + path = tmp_path / "settings.json" + original = { + "github.copilot.chat.otel.enabled": False, + "github.copilot.chat.otel.otlpEndpoint": "http://my-collector:4318", + } + path.write_text(json.dumps(original), encoding="utf-8") + + with pytest.raises(ConfigConflict, match="another endpoint"): + presets.copilot_connect(path) + + assert json.loads(path.read_text(encoding="utf-8")) == original + + +def test_copilot_stale_owg_token_is_repaired_and_removable(tmp_path, monkeypatch): + path = tmp_path / "settings.json" + old = "http://127.0.0.1:8787/agent-ingest/otlp/copilot/old-token" + new = "http://127.0.0.1:8787/agent-ingest/otlp/copilot/new-token" + path.write_text(json.dumps({ + "editor.fontSize": 14, + "github.copilot.chat.otel.enabled": True, + "github.copilot.chat.otel.exporterType": "otlp-http", + "github.copilot.chat.otel.otlpEndpoint": old, + "github.copilot.chat.otel.captureContent": False, + }), encoding="utf-8") + monkeypatch.setattr(presets, "otlp_base_url", lambda source: new if source == "copilot" else f"http://127.0.0.1:8787/agent-ingest/otlp/{source}/new-token") + + result = presets.copilot_connect(path) + assert result["configured"] is True + assert json.loads(path.read_text(encoding="utf-8"))["github.copilot.chat.otel.otlpEndpoint"] == new + + presets.copilot_disconnect(path) + assert json.loads(path.read_text(encoding="utf-8")) == {"editor.fontSize": 14} + + +def test_gemini_stale_owg_token_is_repaired_and_removable(tmp_path, monkeypatch): + path = tmp_path / "settings.json" + old = "http://127.0.0.1:8787/agent-ingest/otlp/gemini/old-token" + new = "http://127.0.0.1:8787/agent-ingest/otlp/gemini/new-token" + path.write_text(json.dumps({ + "theme": "system", + "telemetry": { + "enabled": True, + "target": "local", + "otlpEndpoint": old, + "otlpProtocol": "http", + "logPrompts": False, + }, + }), encoding="utf-8") + monkeypatch.setattr(presets, "otlp_base_url", lambda source: new if source == "gemini" else f"http://127.0.0.1:8787/agent-ingest/otlp/{source}/new-token") + + result = presets.gemini_connect(path) + assert result["configured"] is True + assert json.loads(path.read_text(encoding="utf-8"))["telemetry"]["otlpEndpoint"] == new + + presets.gemini_disconnect(path) + assert json.loads(path.read_text(encoding="utf-8")) == {"theme": "system"} + + +def test_gemini_missing_tool_success_is_unknown_not_error(): + payload = { + "resourceLogs": [{ + "resource": {"attributes": [{"key": "session.id", "value": {"stringValue": "s1"}}]}, + "scopeLogs": [{"logRecords": [{ + "timeUnixNano": "1790000000000000000", + "attributes": [ + {"key": "event.name", "value": {"stringValue": "gemini_cli.tool_call"}}, + {"key": "prompt_id", "value": {"stringValue": "p1"}}, + {"key": "function_name", "value": {"stringValue": "run_shell_command"}}, + ], + }]}], + }], + } + events, stats = gemini_otel_to_agent_events(payload) + assert stats["records_seen"] == 1 + assert len(events) == 1 + assert events[0]["operation"] == "tool_call" + assert events[0]["status"] == "unknown" + + +def test_otlp_gzip_requires_one_complete_stream(tmp_path, monkeypatch): + monkeypatch.setenv("WORKFLOW_OBSERVER_AUTH_DIR", str(tmp_path / "auth")) + monkeypatch.setenv("OWG_CONNECTIONS_HOME", str(tmp_path / "home")) + + from server.agent_auth import ensure_agent_otlp_path_token + from server.agent_routes import router + + app = FastAPI() + app.include_router(router) + token = ensure_agent_otlp_path_token() + url = f"/agent-ingest/otlp/copilot/{token}/v1/traces" + headers = {"Content-Type": "application/json", "Content-Encoding": "gzip"} + + good = gzip.compress(b"{}") + with TestClient(app) as client: + assert client.post(url, content=good[:-4], headers=headers).status_code == 400 + assert client.post(url, content=good + gzip.compress(b"{}"), headers=headers).status_code == 400 diff --git a/tests/test_capture_coverage_v099.py b/tests/test_capture_coverage_v099.py new file mode 100644 index 00000000..95ad0355 --- /dev/null +++ b/tests/test_capture_coverage_v099.py @@ -0,0 +1,91 @@ +from __future__ import annotations + +"""Clipboard writes observed from the OS change counter (contents never read).""" + +import json + +import collector.main as cm +from collector.interactions import RawClipboardAction +from tests.test_capture_correctness_v098 import drive, seconds, spans + + +def _keyboard(monkeypatch): + holder: dict = {} + + class FakeKeyboard: + def __init__(self, callback, *, clipboard_callback=None, capture_clipboard_shortcuts=False): + holder["clipboard"] = clipboard_callback + + def start(self): + return True + + def stop(self): + pass + + monkeypatch.setattr(cm, "KeyboardActivitySensor", FakeKeyboard) + return holder + + +def test_menu_copy_is_a_write_shortcut_copy_is_not_counted_twice_and_pastes_link(tmp_path, monkeypatch): + holder = _keyboard(monkeypatch) + now = {"t": 0.0} + fired: set[float] = set() + + def token(): + t = now["t"] + return 1 if t < 10 else 2 if t < 22 else 3 if t < 40 else 4 + + def shortcut(t, kind): + if t not in fired: + fired.add(t) + holder["clipboard"](RawClipboardAction(kind=kind, occurred_mono=cm.time.monotonic())) + + def world(t): + now["t"] = t + if t == 20: + shortcut(t, "copy") # Cmd+C; the app updates the clipboard at t=22 + if t == 30: + shortcut(t, "paste") + if t == 44: + shortcut(t, "paste") + return ("Mail", "Inbox", 0.0, False) + + monkeypatch.setattr(cm, "clipboard_change_token", token) + events, _ = drive(tmp_path, monkeypatch, world, end=60, keyboard=True) + clip = sorted(spans(events), key=seconds) + clip = [e for e in clip if e["event_type"].startswith("clipboard_")] + assert [(e["event_type"], seconds(e)) for e in clip] == [ + ("clipboard_write", 10), ("clipboard_copy", 20), ("clipboard_paste", 30), ("clipboard_write", 40), ("clipboard_paste", 44), + ] + write1, copy, paste1, write2, paste2 = clip + assert write1["metadata"]["evidence_channel"] == "os_clipboard_sequence" + assert copy["metadata"]["evidence_channel"] == "keyboard_shortcut" + # Each paste links to the most recent clipboard source, whichever way it was written. + assert paste1["metadata"]["linked_copy_event_id"] == copy["event_id"] + assert paste2["metadata"]["linked_copy_event_id"] == write2["event_id"] + assert all(e["metadata"]["clipboard_contents_captured"] is False for e in clip) + assert "paste" not in {e["event_type"] for e in clip if e["metadata"]["evidence_channel"] == "os_clipboard_sequence"} + + +def test_no_writes_are_attributed_while_away_or_when_disabled(tmp_path, monkeypatch): + _keyboard(monkeypatch) + now = {"t": 0.0} + monkeypatch.setattr(cm, "clipboard_change_token", lambda: 1 if now["t"] < 20 else 2) + + def world(t): + now["t"] = t + return ("Mail", "Inbox", 400.0 if t < 40 else 0.0, False) # away until t=40 + + events, _ = drive(tmp_path, monkeypatch, world, end=60, keyboard=True) + assert not [e for e in events if e["event_type"] == "clipboard_write"] + + now["t"] = 0.0 + monkeypatch.setattr(cm, "clipboard_change_token", lambda: 1 if now["t"] < 20 else 2) + + def present(t): + now["t"] = t + return ("Mail", "Inbox", 0.0, False) + + events, _ = drive(tmp_path / "off", monkeypatch, present, end=40, keyboard=True, config={"clipboard_write_detection_enabled": False}) + assert not [e for e in events if e["event_type"] == "clipboard_write"] + assert "Inbox" not in json.dumps([e.get("metadata") for e in events if e["event_type"].startswith("clipboard")]) diff --git a/tests/test_connections_followups_v092.py b/tests/test_connections_followups_v092.py index cef91460..e7c743b7 100644 --- a/tests/test_connections_followups_v092.py +++ b/tests/test_connections_followups_v092.py @@ -91,9 +91,10 @@ def test_claude_desktop_matcher_ignores_helpers_and_claude_code(): def test_hidden_frameworks_follow_observe_switch(home): - assert connections.hidden_frameworks() == {"claude-code", "codex"} # nothing installed + everything = {"claude-code", "codex", "github-copilot", "gemini-cli", "cursor"} + assert connections.hidden_frameworks() == everything # nothing installed connections.change("claude_code", "on", ("observe",)) - assert connections.hidden_frameworks() == {"codex"} + assert connections.hidden_frameworks() == everything - {"claude-code"} connections.change("claude_code", "off", ("observe",)) assert "claude-code" in connections.hidden_frameworks() diff --git a/tests/test_connections_v091.py b/tests/test_connections_v091.py index fe0285bc..d1da3dc0 100644 --- a/tests/test_connections_v091.py +++ b/tests/test_connections_v091.py @@ -111,11 +111,11 @@ def test_claude_code_both_kinds_and_observe_only_where_supported(home): result = connections.change("claude_code", "on") assert result["mcp"]["on"] and result["observe"]["on"] assert (home / ".claude" / "settings.json").exists() - assert _status("cursor")["observe"]["supported"] is False + assert _status("windsurf")["observe"]["supported"] is False with pytest.raises(ValueError): - connections.change("cursor", "on", ("observe",)) + connections.change("windsurf", "on", ("observe",)) # "both" on a context-only client just does context - assert connections.change("cursor", "on")["mcp"]["on"] is True + assert connections.change("windsurf", "on")["mcp"]["on"] is True def test_invalid_json_is_never_touched(home): diff --git a/tests/test_copilot_effective_config_v099.py b/tests/test_copilot_effective_config_v099.py new file mode 100644 index 00000000..7036a612 --- /dev/null +++ b/tests/test_copilot_effective_config_v099.py @@ -0,0 +1,58 @@ +from __future__ import annotations + +import json + +import pytest + +from server import agent_observe_presets as presets +from server.agent_config_writer import ConfigConflict + + +def _clean_env(monkeypatch): + for key in ( + "COPILOT_OTEL_ENDPOINT", + "OTEL_EXPORTER_OTLP_ENDPOINT", + "COPILOT_OTEL_ENABLED", + "COPILOT_OTEL_CAPTURE_CONTENT", + "COPILOT_OTEL_PROTOCOL", + "OTEL_EXPORTER_OTLP_PROTOCOL", + ): + monkeypatch.delenv(key, raising=False) + + +def test_copilot_refuses_visible_environment_overrides_before_writing(tmp_path, monkeypatch): + _clean_env(monkeypatch) + monkeypatch.setenv("WORKFLOW_OBSERVER_AUTH_DIR", str(tmp_path / "auth")) + path = tmp_path / "settings.json" + path.write_text(json.dumps({"editor.fontSize": 14}), encoding="utf-8") + before = path.read_text(encoding="utf-8") + + monkeypatch.setenv("COPILOT_OTEL_ENDPOINT", "http://other-collector:4318") + with pytest.raises(ConfigConflict, match="environment endpoint override"): + presets.copilot_connect(path) + assert path.read_text(encoding="utf-8") == before + + monkeypatch.delenv("COPILOT_OTEL_ENDPOINT") + monkeypatch.setenv("COPILOT_OTEL_CAPTURE_CONTENT", "true") + with pytest.raises(ConfigConflict, match="transmit prompt/response content"): + presets.copilot_connect(path) + assert path.read_text(encoding="utf-8") == before + + monkeypatch.delenv("COPILOT_OTEL_CAPTURE_CONTENT") + monkeypatch.setenv("COPILOT_OTEL_PROTOCOL", "grpc") + with pytest.raises(ConfigConflict, match="forced to gRPC"): + presets.copilot_connect(path) + assert path.read_text(encoding="utf-8") == before + + +def test_copilot_status_does_not_claim_configured_when_visible_override_wins(tmp_path, monkeypatch): + _clean_env(monkeypatch) + monkeypatch.setenv("WORKFLOW_OBSERVER_AUTH_DIR", str(tmp_path / "auth")) + path = tmp_path / "settings.json" + path.write_text("{}", encoding="utf-8") + presets.copilot_connect(path) + assert presets.copilot_status(path)["configured"] is True + + monkeypatch.setenv("OTEL_EXPORTER_OTLP_ENDPOINT", "http://enterprise-collector:4318") + with pytest.raises(ConfigConflict, match="takes precedence"): + presets.copilot_status(path) diff --git a/tests/test_extension_capture.py b/tests/test_extension_capture.py index cef92aca..4e786104 100644 --- a/tests/test_extension_capture.py +++ b/tests/test_extension_capture.py @@ -48,8 +48,8 @@ def test_browser_delivery_is_durable_idempotent_and_sanitized(): assert "flushBrowserQueue" in background assert "sanitizePendingBrowserQueue" in background assert "sensor_version" in background - assert manifest["version"] == "1.12.0" - assert manifest["version_name"] == "1.12.0-v94-agent-lifecycle" + assert manifest["version"] == "1.13.0" + assert manifest["version_name"] == "1.13.0-v99-structural-agent-detection" assert manifest["background"]["service_worker"] == "secure_background.js" diff --git a/tests/test_gemini_lifecycle_v099.py b/tests/test_gemini_lifecycle_v099.py new file mode 100644 index 00000000..c82ebaea --- /dev/null +++ b/tests/test_gemini_lifecycle_v099.py @@ -0,0 +1,75 @@ +from __future__ import annotations + +import json + +from shared.agent_evidence import agent_event_to_evidence +from shared.gemini_otel_adapter import gemini_otel_to_agent_events + +SECRET = "DO-NOT-STORE-THIS" + + +def _kv(key, value): + if isinstance(value, bool): + wrapped = {"boolValue": value} + elif isinstance(value, int): + wrapped = {"intValue": str(value)} + else: + wrapped = {"stringValue": str(value)} + return {"key": key, "value": wrapped} + + +def _record(stamp: int, **attrs): + return { + "timeUnixNano": str(1_790_000_000_000_000_000 + stamp), + "attributes": [_kv(k, v) for k, v in attrs.items()], + } + + +def test_gemini_session_and_explicit_agent_runs_get_finish_boundaries_without_content(): + payload = { + "resourceLogs": [{ + "resource": {"attributes": [_kv("session.id", "session-1")]}, + "scopeLogs": [{"logRecords": [ + _record(1, **{"event.name": "gemini_cli.user_prompt", "prompt_id": "prompt-1", "prompt": SECRET}), + _record(2, **{"event.name": "gemini_cli.agent.start", "agent_id": "agent-1", "agent_name": SECRET}), + _record(3, **{ + "event.name": "gemini_cli.agent.finish", + "agent_id": "agent-1", + "agent_name": SECRET, + "duration_ms": 2500, + "terminate_reason": "cancelled_by_user", + "routing.error_message": SECRET, + }), + _record(4, **{"event.name": "gemini_cli.conversation_finished", "turnCount": 1}), + ]}], + }], + } + + events, stats = gemini_otel_to_agent_events(payload) + assert stats == {"records_seen": 4, "records_ignored": 0, "agent_events": 4} + assert [(e["operation"], e["status"], e["run_id"]) for e in events] == [ + ("run_started", "running", "prompt-1"), + ("run_started", "running", "agent-1"), + ("run_finished", "cancelled", "agent-1"), + ("run_finished", "unknown", "session-1"), + ] + assert events[2]["duration_seconds"] == 2.5 + assert events[3]["session_id"] == "session-1" and events[3]["run_id"] == events[3]["session_id"] + assert SECRET not in json.dumps(events) + for event in events: + agent_event_to_evidence(event) + + +def test_gemini_does_not_invent_a_turn_finish_from_api_response(): + payload = { + "resourceLogs": [{ + "resource": {"attributes": [_kv("session.id", "session-2")]}, + "scopeLogs": [{"logRecords": [ + _record(1, **{"event.name": "gemini_cli.user_prompt", "prompt_id": "prompt-2"}), + _record(2, **{"event.name": "gemini_cli.api_response", "prompt_id": "prompt-2", "status_code": 200}), + ]}], + }], + } + events, _ = gemini_otel_to_agent_events(payload) + assert [e["operation"] for e in events] == ["run_started", "model_call"] + assert not any(e["operation"] == "run_finished" for e in events) diff --git a/tests/test_release_version_v087.py b/tests/test_release_version_v087.py index fed13a66..4056fb05 100644 --- a/tests/test_release_version_v087.py +++ b/tests/test_release_version_v087.py @@ -6,7 +6,7 @@ ROOT = Path(__file__).resolve().parents[1] -EXPECTED_VERSION = "0.98.0" +EXPECTED_VERSION = "0.99.0" def test_release_version_sources_are_aligned(): diff --git a/tests/test_v423_live_active_tab_surface.py b/tests/test_v423_live_active_tab_surface.py index 1eacf2b5..f971fd76 100644 --- a/tests/test_v423_live_active_tab_surface.py +++ b/tests/test_v423_live_active_tab_surface.py @@ -11,8 +11,8 @@ def test_browser_heartbeat_source_reports_sanitized_active_tab(): assert 'ext.tabs.query({active: true, lastFocusedWindow: true})' in background assert 'page,' in background assert 'safeUrl(tab?.url || "")' in background - assert '"version": "1.12.0"' in manifest - assert '"version_name": "1.12.0-v94-agent-lifecycle"' in manifest + assert '"version": "1.13.0"' in manifest + assert '"version_name": "1.13.0-v99-structural-agent-detection"' in manifest def test_browser_heartbeat_keeps_safe_active_page_and_surface():