diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index c7966f494..a913a0532 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -236,6 +236,14 @@ starts the Runtime and its daemon authenticates and initiates the Core connectio Core verifies principal ownership and the exact Environment binding. These are management responsibilities, not separate execution architectures. +Runtime telemetry uses a separate read-only service boundary documented in +[`contracts/agents-api/runtime-observability.md`](contracts/agents-api/runtime-observability.md). +Resolve durable Session, Environment and Runtime-instance identity before selecting +a provider source. Observation never extends a lease or changes compute lifecycle. +Keep observed zero, unavailable data and unsupported Runtime modes distinct. Metrics +may inform operators, but automatic suspension requires durable Core-owned activity +state and must not use a monitoring backend as lifecycle authority. + In V1, our daemon fills the user-side executor role. Users deploy daemon, the selected harness, local tools and workspace together. Do not require Codex `exec-server`, a service-side harness, registry/Noise transport or remote tool diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index d179f0e0f..11cbd4591 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -2329,10 +2329,11 @@ test("presents Dashboard page-chain results and System boundaries without extra await expect.poll(async () => { const entries = await fixtureRequests(request); return [count(entries, "/v1/agents"), count(entries, "/v1/agents/sessions")]; - }).toEqual([count(before, "/v1/agents") + 1, count(before, "/v1/agents/sessions") + 1]); + }).toEqual([count(before, "/v1/agents") + 1, count(before, "/v1/agents/sessions") + 2]); const after = await fixtureRequests(request); expect(count(after, "/v1/agents")).toBe(count(before, "/v1/agents") + 1); - expect(count(after, "/v1/agents/sessions")).toBe(count(before, "/v1/agents/sessions") + 1); + expect(count(after, "/v1/agents/sessions")).toBe(count(before, "/v1/agents/sessions") + 2); + expect(count(after, "/v1/agents/runtime-observations")).toBe(count(before, "/v1/agents/runtime-observations") + 1); for (const path of detailPaths) expect(count(after, path)).toBe(count(before, path)); await attachScreenshot(page, testInfo, "desktop-dashboard-loaded-snapshot"); @@ -2387,11 +2388,12 @@ test("presents Dashboard page-chain results and System boundaries without extra return [count(entries, "/v1/agents"), count(entries, "/v1/agents/sessions")]; }).toEqual([ count(beforeSystemRefresh, "/v1/agents") + 1, - count(beforeSystemRefresh, "/v1/agents/sessions") + 1, + count(beforeSystemRefresh, "/v1/agents/sessions") + 2, ]); const afterSystemRefresh = await fixtureRequests(request); expect(count(afterSystemRefresh, "/v1/agents")).toBe(count(beforeSystemRefresh, "/v1/agents") + 1); - expect(count(afterSystemRefresh, "/v1/agents/sessions")).toBe(count(beforeSystemRefresh, "/v1/agents/sessions") + 1); + expect(count(afterSystemRefresh, "/v1/agents/sessions")).toBe(count(beforeSystemRefresh, "/v1/agents/sessions") + 2); + expect(count(afterSystemRefresh, "/v1/agents/runtime-observations")).toBe(count(beforeSystemRefresh, "/v1/agents/runtime-observations") + 1); for (const path of detailPaths) expect(count(afterSystemRefresh, path)).toBe(count(beforeSystemRefresh, path)); await attachScreenshot(page, testInfo, "desktop-system-contract-boundary"); @@ -2480,7 +2482,9 @@ test("publishes Dashboard counts only after every top-level Agent and Session pa await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Agents" })).toContainText("3"); await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Sessions" })).toContainText("2"); expect(agentAfters).toEqual([null, "agent_b"]); - expect(sessionAfters).toEqual([null, "session_snapshot"]); + expect(sessionAfters).toHaveLength(4); + expect(sessionAfters.filter((after) => after === null)).toHaveLength(2); + expect(sessionAfters.filter((after) => after === "session_snapshot")).toHaveLength(2); }); test("keeps the previous Dashboard result when pagination exceeds the safety limit", async ({ page, request }) => { @@ -2523,8 +2527,9 @@ test("keeps the previous Dashboard result when pagination exceeds the safety lim await refresh.click(); await expect.poll(() => reads).toBe(100); await expect(refresh).toBeEnabled(); - await expect(dashboard).toContainText("Using the last successful snapshot"); + await expect(dashboard).toContainText("Snapshot incomplete"); await expect(dashboard).toContainText("collection pagination exceeded the Web safety limit"); + await expect(dashboard).toContainText("Sessions changed while Runtime observations were loading"); await expect(loadedAgents).toContainText("3"); await dashboard.getByRole("button", { name: "Connection settings" }).click(); const connectionDialog = page.getByRole("dialog", { name: "Connect an Agent Core" }); diff --git a/apps/web/e2e/fixture-core.mjs b/apps/web/e2e/fixture-core.mjs index b950ef6b6..55f4a4154 100644 --- a/apps/web/e2e/fixture-core.mjs +++ b/apps/web/e2e/fixture-core.mjs @@ -996,6 +996,15 @@ const server = http.createServer(async (request, response) => { return sendJson(response, created, 201); } + // The legacy Web fixture uses human-readable Session IDs for interaction + // assertions. Runtime observation resources require canonical UUIDs, so this + // fixture advertises an empty, valid collection instead of inventing a false + // identity join. Positive Runtime rendering is covered by the typed component + // and coordinator tests with canonical identities. + if (request.method === "GET" && url.pathname === "/v1/agents/runtime-observations") { + return sendJson(response, page([])); + } + if (request.method === "GET" && url.pathname === "/v1/agents/sessions") { trackAbort(response, "sessionListReads"); const control = consumeControl("sessionList"); diff --git a/apps/web/package.json b/apps/web/package.json index 98bf865e2..2a7f1da60 100644 --- a/apps/web/package.json +++ b/apps/web/package.json @@ -11,6 +11,7 @@ }, "dependencies": { "@agents-core-web/agents-client": "workspace:*", + "@tanstack/react-table": "^8.21.3", "lucide-react": "^1.22.0", "react": "^19.2.7", "react-dom": "^19.2.7", diff --git a/apps/web/src/App.tsx b/apps/web/src/App.tsx index e4a7d6f85..c0925e986 100644 --- a/apps/web/src/App.tsx +++ b/apps/web/src/App.tsx @@ -33,6 +33,12 @@ import { requestAgentUpdate, } from "./features/agents/agent-actions"; import { DashboardView } from "./features/dashboard/DashboardView"; +import { + loadRuntimeDashboardSnapshot, + RUNTIME_SNAPSHOT_REFRESH_MS, + RUNTIME_SNAPSHOT_TIMEOUT_MS, + type RuntimeDashboardSnapshot, +} from "./features/dashboard/runtime-snapshot"; import { SessionsView, type SessionDetailState, @@ -270,6 +276,10 @@ export function App() { const [sessionCollectionState, setSessionCollectionState] = useState("connecting"); const [sessionCollectionError, setSessionCollectionError] = useState(null); const [sessionCollectionHasSnapshot, setSessionCollectionHasSnapshot] = useState(false); + const [runtimeSnapshot, setRuntimeSnapshot] = useState(null); + const [runtimeCollectionState, setRuntimeCollectionState] = useState("connecting"); + const [runtimeCollectionError, setRuntimeCollectionError] = useState(null); + const [runtimeCollectionHasSnapshot, setRuntimeCollectionHasSnapshot] = useState(false); const [sessionAgentFilter, setSessionAgentFilter] = useState(null); const [filteredSessions, setFilteredSessions] = useState([]); const [filteredSessionCollectionState, setFilteredSessionCollectionState] = useState("connecting"); @@ -305,6 +315,8 @@ export function App() { const sessionCollectionRequestRef = useRef(0); const agentCollectionAbortRef = useRef(null); const sessionCollectionAbortRef = useRef(null); + const runtimeCollectionAbortRef = useRef(null); + const runtimeCollectionRequestRef = useRef(0); const filteredSessionCollectionAbortRef = useRef(null); const filteredSessionCollectionRequestRef = useRef(0); const sessionAgentFilterRef = useRef(sessionAgentFilter); @@ -516,6 +528,54 @@ export function App() { } }, [core, coreGeneration, notify]); + const refreshRuntimeSnapshot = useCallback(async () => { + if (coreGeneration !== connectionGenerationRef.current) return false; + runtimeCollectionAbortRef.current?.abort(); + const controller = new AbortController(); + runtimeCollectionAbortRef.current = controller; + const request = runtimeCollectionRequestRef.current + 1; + runtimeCollectionRequestRef.current = request; + let timedOut = false; + const timeout = window.setTimeout(() => { + timedOut = true; + controller.abort(); + }, RUNTIME_SNAPSHOT_TIMEOUT_MS); + setRuntimeCollectionState("connecting"); + setRuntimeCollectionError(null); + try { + const result = await settleCollection(() => loadRuntimeDashboardSnapshot( + core, + () => sessionCollectionRevisionRef.current, + controller.signal, + )); + if ( + coreGeneration !== connectionGenerationRef.current || + request !== runtimeCollectionRequestRef.current + ) return false; + if (result.status === "rejected") { + if (isAbort(result.reason) && !timedOut) return false; + const message = timedOut + ? "Runtime snapshot exceeded the 15 second Web refresh budget." + : errorMessage(result.reason); + setRuntimeCollectionState("failed"); + setRuntimeCollectionError(message); + return false; + } + if (result.value === null) { + setRuntimeCollectionState("failed"); + setRuntimeCollectionError("Session data changed while Runtime observations were loading. The previous complete snapshot was retained."); + return false; + } + setRuntimeSnapshot(result.value); + setRuntimeCollectionHasSnapshot(true); + setRuntimeCollectionState("ready"); + return true; + } finally { + window.clearTimeout(timeout); + if (runtimeCollectionAbortRef.current === controller) runtimeCollectionAbortRef.current = null; + } + }, [core, coreGeneration]); + const refreshFilteredSessions = useCallback(async (agentId: string) => { if ( coreGeneration !== connectionGenerationRef.current || @@ -855,9 +915,10 @@ export function App() { void refreshSessions(); void refreshVaults(); void refreshEnvironmentTemplates(); + void refreshRuntimeSnapshot(); const filter = sessionAgentFilterRef.current; if (filter) void refreshFilteredSessions(filter); - }, [refreshAgents, refreshEnvironmentTemplates, refreshFilteredSessions, refreshSessions, refreshVaults]); + }, [refreshAgents, refreshEnvironmentTemplates, refreshFilteredSessions, refreshRuntimeSnapshot, refreshSessions, refreshVaults]); const changeSessionAgentFilter = useCallback((agentId: string | null) => { if (sessionAgentFilterRef.current === agentId) return; @@ -894,6 +955,10 @@ export function App() { setSessions([]); setAgentCollectionHasSnapshot(false); setSessionCollectionHasSnapshot(false); + setRuntimeSnapshot(null); + setRuntimeCollectionState("connecting"); + setRuntimeCollectionError(null); + setRuntimeCollectionHasSnapshot(false); setItems([]); setTurns([]); setEnvironmentObservations(new Map()); @@ -908,7 +973,29 @@ export function App() { void refreshSessions(); void refreshVaults(); void refreshEnvironmentTemplates(); - }, [refreshAgents, refreshEnvironmentTemplates, refreshSessions, refreshVaults]); + void refreshRuntimeSnapshot(); + }, [refreshAgents, refreshEnvironmentTemplates, refreshRuntimeSnapshot, refreshSessions, refreshVaults]); + + useEffect(() => { + if (view !== "dashboard") return; + let timer: number | null = null; + const schedule = () => { + const jitter = Math.floor(Math.random() * 5_000); + timer = window.setTimeout(() => { + if (!document.hidden) void refreshRuntimeSnapshot(); + schedule(); + }, RUNTIME_SNAPSHOT_REFRESH_MS + jitter); + }; + const onVisibilityChange = () => { + if (!document.hidden) void refreshRuntimeSnapshot(); + }; + document.addEventListener("visibilitychange", onVisibilityChange); + schedule(); + return () => { + if (timer !== null) window.clearTimeout(timer); + document.removeEventListener("visibilitychange", onVisibilityChange); + }; + }, [refreshRuntimeSnapshot, view]); useEffect(() => { filteredSessionCollectionAbortRef.current?.abort(); @@ -1943,6 +2030,7 @@ export function App() { connectionGenerationRef.current += 1; agentCollectionRequestRef.current += 1; sessionCollectionRequestRef.current += 1; + runtimeCollectionRequestRef.current += 1; filteredSessionCollectionRequestRef.current += 1; vaultCollectionRequestRef.current += 1; agentCollectionRevisionRef.current = 0; @@ -1966,6 +2054,8 @@ export function App() { agentCollectionAbortRef.current = null; sessionCollectionAbortRef.current?.abort(); sessionCollectionAbortRef.current = null; + runtimeCollectionAbortRef.current?.abort(); + runtimeCollectionAbortRef.current = null; filteredSessionCollectionAbortRef.current?.abort(); filteredSessionCollectionAbortRef.current = null; vaultCollectionAbortRef.current?.abort(); @@ -1981,6 +2071,10 @@ export function App() { setSessionCollectionState("connecting"); setSessionCollectionError(null); setSessionCollectionHasSnapshot(false); + setRuntimeSnapshot(null); + setRuntimeCollectionState("connecting"); + setRuntimeCollectionError(null); + setRuntimeCollectionHasSnapshot(false); sessionAgentFilterRef.current = null; filteredSessionsRef.current = []; setSessionAgentFilter(null); @@ -2108,6 +2202,10 @@ export function App() { sessionCollectionState={sessionCollectionState} sessionCollectionError={sessionCollectionError} sessionCollectionHasSnapshot={sessionCollectionHasSnapshot} + runtimeSnapshot={runtimeSnapshot} + runtimeCollectionState={runtimeCollectionState} + runtimeCollectionError={runtimeCollectionError} + runtimeCollectionHasSnapshot={runtimeCollectionHasSnapshot} onRefresh={refreshDashboard} onCreateAgent={openAgentSetup} onStartSession={() => openSessionSetup()} @@ -2208,6 +2306,7 @@ export function App() { refreshing={ agentCollectionState === "connecting" || sessionCollectionState === "connecting" || + runtimeCollectionState === "connecting" || vaultCollectionState === "connecting" } onRefresh={refreshDashboard} diff --git a/apps/web/src/features/dashboard/DashboardView.css b/apps/web/src/features/dashboard/DashboardView.css index 999874c09..13778f0fb 100644 --- a/apps/web/src/features/dashboard/DashboardView.css +++ b/apps/web/src/features/dashboard/DashboardView.css @@ -613,10 +613,567 @@ border-radius: 50%; } +.dashboard-runtime-panel { + margin-top: 16px; + overflow: hidden; +} + +.dashboard-runtime-freshness { + color: var(--fg-muted); + font-family: var(--font-mono); + font-size: 10px; + font-variant-numeric: tabular-nums; + line-height: 15px; + text-align: right; +} + +.dashboard-runtime-summary { + display: grid; + grid-template-columns: repeat(4, minmax(0, 1fr)); + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-metric { + display: grid; + min-width: 0; + min-height: 106px; + padding: 16px; + grid-template-columns: 32px minmax(0, 1fr); + align-items: flex-start; + gap: 10px; +} + +.dashboard-runtime-metric + .dashboard-runtime-metric { + border-left: 1px solid var(--line); +} + +.dashboard-runtime-metric-icon { + display: inline-flex; + width: 32px; + height: 32px; + align-items: center; + justify-content: center; + color: var(--accent); + background: color-mix(in srgb, var(--accent) 8%, var(--surface)); + border: 1px solid color-mix(in srgb, var(--accent) 18%, var(--line)); + border-radius: 8px; +} + +.dashboard-runtime-metric > span:last-child { + display: grid; + min-width: 0; + gap: 2px; +} + +.dashboard-runtime-metric small, +.dashboard-runtime-metric > span:last-child > span { + color: var(--fg-muted); + font-size: 10px; + line-height: 14px; +} + +.dashboard-runtime-metric strong { + overflow: hidden; + font-family: var(--font-mono); + font-size: 17px; + font-variant-numeric: tabular-nums slashed-zero; + font-weight: 550; + line-height: 24px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-insights { + display: grid; + grid-template-columns: minmax(0, 1.15fr) minmax(0, .85fr); + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-insight { + min-width: 0; + padding: 14px; +} + +.dashboard-runtime-insight + .dashboard-runtime-insight { + border-left: 1px solid var(--line); +} + +.dashboard-runtime-insight > header, +.dashboard-runtime-targets > header { + display: flex; + align-items: flex-start; + justify-content: space-between; + gap: 12px; +} + +.dashboard-runtime-insight h3, +.dashboard-runtime-targets h3, +.dashboard-runtime-insight p, +.dashboard-runtime-targets p { + margin: 0; +} + +.dashboard-runtime-insight h3, +.dashboard-runtime-targets h3 { + font-size: 12px; + font-weight: 600; + line-height: 17px; +} + +.dashboard-runtime-insight header p, +.dashboard-runtime-targets header p, +.dashboard-runtime-insight header > span, +.dashboard-runtime-targets header > span { + color: var(--fg-muted); + font-size: 9px; + line-height: 13px; +} + +.dashboard-runtime-insight header > span, +.dashboard-runtime-targets header > span { + flex: 0 0 auto; + font-family: var(--font-mono); + font-variant-numeric: tabular-nums; +} + +.dashboard-runtime-health-grid { + display: grid; + margin-top: 12px; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 7px; +} + +.dashboard-runtime-health { + display: grid; + min-width: 0; + padding: 8px 9px 8px 19px; + position: relative; + gap: 1px; + color: var(--fg); + text-align: left; + background: var(--surface-subtle); + border: 1px solid var(--line); + border-radius: 6px; +} + +.dashboard-runtime-health::before { + width: 6px; + height: 6px; + position: absolute; + top: 12px; + left: 8px; + content: ""; + background: var(--fg-muted); + border-radius: 50%; +} + +.dashboard-runtime-health-observed::before { + background: var(--success); +} + +.dashboard-runtime-health-unavailable::before { + background: var(--warning); +} + +.dashboard-runtime-health:hover, +.dashboard-runtime-health:focus-visible { + background: var(--hover); + border-color: color-mix(in srgb, var(--accent) 35%, var(--line)); +} + +.dashboard-runtime-health strong, +.dashboard-runtime-health span { + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-health strong { + font-size: 10px; + font-weight: 550; + line-height: 14px; +} + +.dashboard-runtime-health span, +.dashboard-runtime-health-more { + color: var(--fg-muted); + font-size: 9px; + line-height: 13px; +} + +.dashboard-runtime-health-more { + align-self: center; + padding-left: 4px; +} + +.dashboard-runtime-coverage-grid { + display: grid; + margin-top: 12px; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 7px; +} + +.dashboard-runtime-coverage-grid > div { + display: grid; + min-width: 0; + padding: 9px 10px; + gap: 1px; + background: var(--surface-subtle); + border: 1px solid var(--line); + border-radius: 6px; +} + +.dashboard-runtime-coverage-grid strong { + font-family: var(--font-mono); + font-size: 10px; + font-variant-numeric: tabular-nums; + font-weight: 600; + line-height: 14px; +} + +.dashboard-runtime-coverage-grid span { + color: var(--fg-muted); + font-size: 9px; + line-height: 13px; +} + +.dashboard-runtime-coverage-warning strong { + color: var(--warning); +} + +.dashboard-runtime-coverage-unsupported strong { + color: var(--fg-muted); +} + +.dashboard-runtime-targets { + min-width: 0; +} + +.dashboard-runtime-targets > header { + min-height: 58px; + align-items: center; + padding: 11px 14px; + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-toolbar { + display: flex; + min-height: 48px; + align-items: center; + padding: 8px 14px; + gap: 8px; + background: var(--surface-subtle); + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-toolbar label { + display: flex; + align-items: center; + gap: 6px; + color: var(--fg-muted); + font-size: 9px; +} + +.dashboard-runtime-toolbar input, +.dashboard-runtime-toolbar select { + height: 30px; + color: var(--fg); + font: inherit; + background: var(--surface); + border: 1px solid var(--line); + border-radius: 5px; +} + +.dashboard-runtime-search { + width: min(360px, 100%); + height: 30px; + margin-right: auto; + padding: 0 8px; + background: var(--surface); + border: 1px solid var(--line); + border-radius: 5px; +} + +.dashboard-runtime-search input { + width: 100%; + min-width: 0; + height: auto; + padding: 0; + background: transparent; + border: 0; + outline: 0; +} + +.dashboard-runtime-toolbar select { + min-width: 112px; + padding: 0 24px 0 8px; +} + +.dashboard-runtime-table-scroll { + overflow-x: auto; +} + +.dashboard-runtime-table { + width: 100%; + min-width: 1080px; + border-collapse: collapse; + font-size: 10px; + line-height: 14px; +} + +.dashboard-runtime-table th { + height: 34px; + padding: 0 10px; + color: var(--fg-muted); + font-size: 9px; + font-weight: 600; + text-align: left; + background: var(--surface-subtle); + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-table th button { + display: inline-flex; + align-items: center; + padding: 0; + gap: 4px; + color: inherit; + font: inherit; + background: transparent; + border: 0; +} + +.dashboard-runtime-table td { + height: 68px; + padding: 9px 10px; + vertical-align: middle; + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-table tbody tr:hover { + background: var(--hover); +} + +.dashboard-runtime-table th:first-child, +.dashboard-runtime-table td:first-child { + width: 220px; + padding-left: 14px; +} + +.dashboard-runtime-target-identity, +.dashboard-runtime-table-value { + display: grid; + min-width: 0; + gap: 2px; +} + +.dashboard-runtime-target-identity { + position: relative; +} + +.dashboard-runtime-target-identity > button { + width: fit-content; + max-width: 100%; + padding: 0; + overflow: hidden; + color: var(--fg); + font: inherit; + font-weight: 600; + text-align: left; + text-overflow: ellipsis; + white-space: nowrap; + background: transparent; + border: 0; +} + +.dashboard-runtime-target-identity > button:hover, +.dashboard-runtime-target-identity > button:focus-visible { + color: var(--accent); +} + +.dashboard-runtime-target-identity > small, +.dashboard-runtime-table-value small { + overflow: hidden; + color: var(--fg-muted); + font-size: 9px; + line-height: 13px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-target-identity details { + color: var(--fg-muted); + font-size: 9px; +} + +.dashboard-runtime-target-identity summary { + width: fit-content; + cursor: pointer; +} + +.dashboard-runtime-target-identity dl { + display: grid; + width: 320px; + margin: 6px 0 0; + padding: 8px; + position: absolute; + z-index: 2; + gap: 4px; + background: var(--surface); + border: 1px solid var(--line); + border-radius: 6px; + box-shadow: var(--shadow-control); +} + +.dashboard-runtime-target-identity dl > div { + display: grid; + grid-template-columns: 72px minmax(0, 1fr); +} + +.dashboard-runtime-target-identity dt, +.dashboard-runtime-target-identity dd { + margin: 0; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-target-identity dd { + color: var(--fg); + font-family: var(--font-mono); +} + +.dashboard-runtime-status { + display: inline-flex; + width: fit-content; + align-items: center; + gap: 6px; + font-size: 10px; + font-weight: 550; + line-height: 15px; +} + +.dashboard-runtime-status > span { + width: 7px; + height: 7px; + background: var(--fg-muted); + border-radius: 50%; +} + +.dashboard-runtime-status-observed > span { + background: var(--success); + box-shadow: 0 0 0 3px color-mix(in srgb, var(--success) 11%, transparent); +} + +.dashboard-runtime-status-unavailable { + color: var(--warning); +} + +.dashboard-runtime-status-unavailable > span { + background: var(--warning); +} + +.dashboard-runtime-status-unsupported { + color: var(--fg-muted); +} + +.dashboard-runtime-table-value strong { + overflow: hidden; + font-family: var(--font-mono); + font-size: 12px; + font-variant-numeric: tabular-nums slashed-zero; + font-weight: 550; + line-height: 17px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-bar { + display: block; + width: 100%; + max-width: 110px; + height: 3px; + margin-top: 2px; + overflow: hidden; + background: var(--surface-subtle); + border-radius: 999px; +} + +.dashboard-runtime-bar i { + display: block; + height: 100%; + background: var(--accent); + border-radius: inherit; +} + +.dashboard-runtime-no-results { + margin: 0; + padding: 20px 14px; + color: var(--fg-muted); + font-size: 10px; + text-align: center; + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-pagination { + display: flex; + min-height: 46px; + align-items: center; + justify-content: space-between; + padding: 8px 14px; + color: var(--fg-muted); + font-family: var(--font-mono); + font-size: 9px; +} + +.dashboard-runtime-pagination > div { + display: flex; + gap: 6px; +} + +.dashboard-runtime-pagination button { + display: inline-flex; + min-height: 28px; + align-items: center; + padding: 4px 8px; + gap: 4px; + color: var(--fg); + font: inherit; + background: var(--surface-subtle); + border: 1px solid var(--line); + border-radius: 5px; +} + +.dashboard-runtime-pagination button:disabled { + cursor: not-allowed; + opacity: .45; +} + @media (max-width: 980px) { .dashboard-primary-grid { grid-template-columns: 1fr; } + + .dashboard-runtime-summary { + grid-template-columns: repeat(2, minmax(0, 1fr)); + } + + .dashboard-runtime-metric:nth-child(3) { + border-left: 0; + } + + .dashboard-runtime-metric:nth-child(n + 3) { + border-top: 1px solid var(--line); + } + + .dashboard-runtime-insights { + grid-template-columns: 1fr; + } + + .dashboard-runtime-insight + .dashboard-runtime-insight { + border-top: 1px solid var(--line); + border-left: 0; + } } @media (max-width: 760px) { @@ -670,6 +1227,26 @@ .dashboard-metric:nth-child(n + 3) { border-top: 1px solid var(--line); } + + .dashboard-runtime-toolbar { + align-items: stretch; + flex-wrap: wrap; + } + + .dashboard-runtime-search { + width: 100%; + flex-basis: 100%; + margin-right: 0; + } + + .dashboard-runtime-toolbar label:not(.dashboard-runtime-search) { + flex: 1; + } + + .dashboard-runtime-toolbar select { + min-width: 0; + flex: 1; + } } @media (max-width: 520px) { @@ -701,6 +1278,27 @@ align-items: flex-start; } + .dashboard-runtime-summary { + grid-template-columns: 1fr; + } + + .dashboard-runtime-metric + .dashboard-runtime-metric, + .dashboard-runtime-metric:nth-child(3) { + border-top: 1px solid var(--line); + border-left: 0; + } + + .dashboard-runtime-health-grid, + .dashboard-runtime-coverage-grid { + grid-template-columns: 1fr; + } + + .dashboard-runtime-targets > header { + align-items: flex-start; + flex-direction: column; + gap: 4px; + } + .dashboard-panel header p { white-space: normal; } diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index f7c617ba4..c0887eba7 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -1,7 +1,7 @@ import { renderToStaticMarkup } from "react-dom/server"; import { describe, expect, it } from "vitest"; -import type { AgentSession, SavedAgent } from "@agents-core-web/agents-client"; +import type { AgentSession, RuntimeObservation, SavedAgent } from "@agents-core-web/agents-client"; import { DashboardView, type DashboardViewProps } from "./DashboardView"; @@ -74,6 +74,10 @@ function render(overrides: Partial = {}): string { sessionCollectionState="ready" sessionCollectionError={null} sessionCollectionHasSnapshot + runtimeSnapshot={{ sessions: [], observations: [], loadedAt: 1_700_000_000_000 }} + runtimeCollectionState="ready" + runtimeCollectionError={null} + runtimeCollectionHasSnapshot {...callbacks} {...overrides} />, @@ -121,6 +125,21 @@ describe("Dashboard loaded-result presentation", () => { expect(html).not.toContain("Sessions: Agent core request failed (502)"); }); + it("keeps a Runtime-only 503 scoped to the optional observation feature", () => { + const html = render({ + runtimeSnapshot: null, + runtimeCollectionState: "failed", + runtimeCollectionError: "Agent core request failed (503).", + runtimeCollectionHasSnapshot: false, + }); + + expect(html).toContain("Runtime: Agent core request failed (503)."); + expect(html).toContain("Runtime observations unavailable"); + expect(html).not.toContain("Agent Core backend is not ready"); + expect(html).not.toContain("Core backend is offline"); + expect(html).not.toContain("Open startup guide"); + }); + it("renders a compact actionable overview while preserving Environment qualifications", () => { const selfHosted: AgentSession["environment"] = { type: "self_hosted", @@ -174,6 +193,80 @@ describe("Dashboard loaded-result presentation", () => { expect(html).not.toContain("Execution ready"); }); + it("renders current Docker resources without inventing a CPU percentage or history", () => { + const hosted = session("11111111-1111-4111-8111-111111111111", { + metadata: { title: "Managed research" }, + environment: { + type: "openai_hosted", + id: "22222222-2222-4222-8222-222222222222", + capability_directories: [], + network: { access: "enabled", allowed_domains: [] }, + packages: { npm: [], python: [], system: [] }, + files: [], + plugins: [], + skills: [], + }, + usage: { + input_tokens: 30, + output_tokens: 12, + total_tokens: 42, + input_tokens_details: { cached_tokens: 7 }, + output_tokens_details: { reasoning_tokens: 3 }, + }, + }); + const observation: RuntimeObservation = { + id: hosted.id, + object: "agent.runtime_observation", + session_id: hosted.id, + environment_id: "22222222-2222-4222-8222-222222222222", + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: "33333333-3333-4333-8333-333333333333", + device_id: null, + connection_generation: null, + }, + status: "observed", + reason: null, + allocation_created_at: 1_700_000_000, + resolved_at: 1_700_000_100, + observed_at: 1_700_000_090, + started_at: 1_700_000_010, + cpu: { usage_seconds_total: 73.5, capacity_cores: 2, usage_cores: null, utilization_ratio: null }, + memory: { usage_bytes: 536_870_912, limit_bytes: 2_147_483_648 }, + }; + const html = render({ + runtimeSnapshot: { sessions: [hosted], observations: [observation], loadedAt: 1_700_000_100_000 }, + }); + + expect(html).toContain("Runtime monitoring"); + expect(html).toContain("1/1 managed observed"); + expect(html).toContain("CPU time / capacity"); + expect(html).toContain("1m 13s / 2 cores"); + expect(html).toContain("512 MiB / 2.00 GiB"); + expect(html).toContain("Runtime health"); + expect(html).toContain("Observed · 2023-11-14 22:14 UTC"); + expect(html).not.toContain("Observed · 10s"); + expect(html).toContain("Usage coverage"); + expect(html).toContain("CPU 1/1"); + expect(html).toContain("Memory 1/1"); + expect(html).toContain("Tokens 1/1"); + expect(html).toContain("Unsupported 0"); + expect(html).toContain("Runtime targets"); + expect(html).toContain("Search Runtime targets"); + expect(html).toContain("All statuses"); + expect(html).toContain("All modes"); + expect(html).toContain(''); + expect(html).toContain("CPU time"); + expect(html).toContain("Managed research"); + expect(html).toContain("Identity"); + expect(html).toContain("Unknown remains unknown, never zero"); + expect(html).not.toContain("CPU %"); + expect(html).not.toContain("CPU now"); + expect(html).not.toContain("historical chart"); + }); + it("keeps partial Usage out of the primary overview", () => { const html = render({ sessions: [session("usage-unknown")] }); diff --git a/apps/web/src/features/dashboard/DashboardView.tsx b/apps/web/src/features/dashboard/DashboardView.tsx index c2cd9786d..4fd6cc68c 100644 --- a/apps/web/src/features/dashboard/DashboardView.tsx +++ b/apps/web/src/features/dashboard/DashboardView.tsx @@ -14,12 +14,15 @@ import { StatusIcon, type StatusKind } from "../../components/StatusIcon"; import { backendFailureStatus } from "../../lib/core-readiness"; import { buildDashboardSnapshot, + buildRuntimeDashboardModel, dashboardEnvironmentLabel, dashboardStatusLabel, formatDashboardTimestamp, type DashboardCollectionState, type DashboardSessionRow, } from "./dashboard-model"; +import { RuntimeObservabilityContent } from "./RuntimeObservabilityContent"; +import type { RuntimeDashboardSnapshot } from "./runtime-snapshot"; import "./DashboardView.css"; export interface DashboardViewProps { @@ -31,6 +34,10 @@ export interface DashboardViewProps { sessionCollectionState: DashboardCollectionState; sessionCollectionError: string | null; sessionCollectionHasSnapshot: boolean; + runtimeSnapshot: RuntimeDashboardSnapshot | null; + runtimeCollectionState: DashboardCollectionState; + runtimeCollectionError: string | null; + runtimeCollectionHasSnapshot: boolean; onRefresh: () => void; onCreateAgent: () => void; onStartSession: () => void; @@ -232,6 +239,10 @@ export function DashboardView({ sessionCollectionState, sessionCollectionError, sessionCollectionHasSnapshot, + runtimeSnapshot, + runtimeCollectionState, + runtimeCollectionError, + runtimeCollectionHasSnapshot, onRefresh, onCreateAgent, onStartSession, @@ -241,21 +252,28 @@ export function DashboardView({ onOpenSession, }: DashboardViewProps) { const snapshot = useMemo(() => buildDashboardSnapshot(agents, sessions, 6, 5), [agents, sessions]); + const runtimeModel = useMemo(() => runtimeSnapshot + ? buildRuntimeDashboardModel(runtimeSnapshot.sessions, runtimeSnapshot.observations) + : null, [runtimeSnapshot]); const agentsAvailable = collectionHasSnapshot(agentCollectionState, agentCollectionHasSnapshot); const sessionsAvailable = collectionHasSnapshot(sessionCollectionState, sessionCollectionHasSnapshot); - const refreshing = agentCollectionState === "connecting" || sessionCollectionState === "connecting"; + const runtimeAvailable = runtimeCollectionState === "ready" || runtimeCollectionHasSnapshot; + const refreshing = agentCollectionState === "connecting" || sessionCollectionState === "connecting" || runtimeCollectionState === "connecting"; const hasStaleSnapshot = ( (agentCollectionState === "failed" && agentCollectionHasSnapshot) || - (sessionCollectionState === "failed" && sessionCollectionHasSnapshot) + (sessionCollectionState === "failed" && sessionCollectionHasSnapshot) || + (runtimeCollectionState === "failed" && runtimeCollectionHasSnapshot) ); - const hasUnavailableSource = !agentsAvailable || !sessionsAvailable; + const hasUnavailableSource = !agentsAvailable || !sessionsAvailable || !runtimeAvailable; const attentionCount = snapshot.statusCounts.requires_action + snapshot.statusCounts.failed; - const sourceErrors: Array = [ + const sourceErrors: Array = [ agentCollectionState === "failed" ? ["Agents", agentCollectionError] as const : null, sessionCollectionState === "failed" ? ["Sessions", sessionCollectionError] as const : null, - ].filter((entry): entry is readonly ["Agents" | "Sessions", string | null] => entry !== null); - const backendFailureStatuses = sourceErrors.map(([, error]) => backendFailureStatus(error)); - const backendUnavailable = sourceErrors.length > 0 && backendFailureStatuses.every(Boolean); + runtimeCollectionState === "failed" ? ["Runtime", runtimeCollectionError] as const : null, + ].filter((entry): entry is readonly ["Agents" | "Sessions" | "Runtime", string | null] => entry !== null); + const coreSourceErrors = sourceErrors.filter(([label]) => label !== "Runtime"); + const backendFailureStatuses = coreSourceErrors.map(([, error]) => backendFailureStatus(error)); + const backendUnavailable = coreSourceErrors.length > 0 && backendFailureStatuses.every(Boolean); const backendFailureDetail = Array.from(new Set(backendFailureStatuses.filter(Boolean))).map((status) => ( status === "network" ? "network failure" : `HTTP ${status}` )).join(" / "); @@ -304,6 +322,7 @@ export function DashboardView({
+
@@ -359,6 +378,45 @@ export function DashboardView({ +
+
+
+

Runtime monitoring

+

Current provider samples joined to an exact complete Session snapshot · no historical series

+
+ {runtimeModel ? ( + + {runtimeModel.summary.observedRuntimeCount}/{runtimeModel.summary.managedRuntimeCount} managed observed + {runtimeModel.summary.newestResolvedAt === null ? "" : ` · ${formatDashboardTimestamp(runtimeModel.summary.newestResolvedAt)}`} + + ) : null} +
+ {!runtimeAvailable || !runtimeSnapshot || !runtimeModel ? ( +

+

+ ) : ( + <> + {runtimeModel.rows.length ? ( + + ) : ( +

No Session-owned Runtime contexts in this snapshot.

+ )} + + )} +
+
diff --git a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx new file mode 100644 index 000000000..c580856cc --- /dev/null +++ b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx @@ -0,0 +1,349 @@ +import { + ChevronDown, + ChevronLeft, + ChevronRight, + ChevronsUpDown, + ChevronUp, + Cpu, + Gauge, + MemoryStick, + Search, + Server, +} from "lucide-react"; +import { useMemo, useState, type ReactNode } from "react"; +import { + flexRender, + getCoreRowModel, + getFilteredRowModel, + getPaginationRowModel, + getSortedRowModel, + useReactTable, + type ColumnDef, + type FilterFn, + type SortingState, +} from "@tanstack/react-table"; + +import { + buildRuntimeDashboardModel, + dashboardEnvironmentLabel, + dashboardStatusLabel, + formatDashboardBytes, + formatDashboardDuration, + formatDashboardTimestamp, + formatDashboardTokens, + runtimeObservationStatusLabel, + type RuntimeDashboardRow, +} from "./dashboard-model"; +import type { RuntimeDashboardSnapshot } from "./runtime-snapshot"; + +const PAGE_SIZE = 10; + +function RuntimeMetric({ + icon, + label, + value, + detail, +}: { + icon: ReactNode; + label: string; + value: string; + detail: string; +}) { + return ( +
+ + + {label} + {value} + {detail} + +
+ ); +} + +function percent(usage: number | null | undefined, limit: number | null | undefined): number | null { + if (typeof usage !== "number" || typeof limit !== "number" || limit <= 0) return null; + return Math.min(100, Math.max(0, usage / limit * 100)); +} + +function coverage(known: number, total: number): string { + if (total === 0) return "No observed Runtimes"; + return `${Math.round(known / total * 100)}% coverage`; +} + +function runtimeModeLabel(row: RuntimeDashboardRow): string { + if (row.observation.mode === "openai_hosted") { + const provider = row.observation.provider_type; + return provider ? `Managed ${provider === "docker" ? "Docker" : provider}` : "Managed"; + } + return dashboardEnvironmentLabel(row.session.environmentProfile); +} + +function observationTimestamp(row: RuntimeDashboardRow, loadedAt: number): number | null { + const timestamp = row.observation.status === "observed" + ? row.observation.observed_at + : row.observation.resolved_at; + if (!Number.isSafeInteger(timestamp) || timestamp < 0 || timestamp > Math.floor(loadedAt / 1_000)) return null; + return timestamp; +} + +function healthDetail(row: RuntimeDashboardRow, loadedAt: number): string { + const status = runtimeObservationStatusLabel(row.observation); + const timestamp = observationTimestamp(row, loadedAt); + return timestamp === null ? status : `${status} · ${formatDashboardTimestamp(timestamp)}`; +} + +function SortHeader({ + label, + sorted, + onClick, +}: { + label: string; + sorted: false | "asc" | "desc"; + onClick: (event: unknown) => void; +}) { + const Icon = sorted === "asc" ? ChevronUp : sorted === "desc" ? ChevronDown : ChevronsUpDown; + return ( + + ); +} + +const runtimeGlobalFilter: FilterFn = (row, _columnId, value) => { + const query = String(value).trim().toLocaleLowerCase(); + if (!query) return true; + const item = row.original; + return [ + item.session.title, + item.session.agentLabel, + item.observation.session_id, + item.observation.environment_id, + item.observation.instance.allocation_id, + item.observation.provider_type, + item.observation.status, + item.observation.reason, + runtimeModeLabel(item), + ].some((candidate) => typeof candidate === "string" && candidate.toLocaleLowerCase().includes(query)); +}; + +function RuntimeTargets({ + rows, + onOpenSession, +}: { + rows: RuntimeDashboardRow[]; + onOpenSession: (sessionId: string) => void; +}) { + const [sorting, setSorting] = useState([]); + const [globalFilter, setGlobalFilter] = useState(""); + const [statusFilter, setStatusFilter] = useState("all"); + const [modeFilter, setModeFilter] = useState("all"); + const filteredRows = useMemo(() => rows.filter((row) => ( + (statusFilter === "all" || row.observation.status === statusFilter) && + (modeFilter === "all" || row.observation.mode === modeFilter) + )), [modeFilter, rows, statusFilter]); + const columns = useMemo[]>(() => [{ + id: "session", + accessorFn: (row) => row.session.title, + header: ({ column }) => undefined)} />, + cell: ({ row }) => { + const item = row.original; + return ( +
+ + {item.session.agentLabel} +
+ Identity +
+
Session
{item.observation.session_id}
+
Environment
{item.observation.environment_id ?? "Not applicable"}
+
Allocation
{item.observation.instance.allocation_id ?? "Not available"}
+
Resolved
{formatDashboardTimestamp(item.observation.resolved_at)}
+
+
+
+ ); + }, + }, { + id: "mode", + accessorFn: runtimeModeLabel, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {runtimeModeLabel(row.original)}, + }, { + id: "status", + accessorFn: (row) => row.observation.status, + header: ({ column }) => undefined)} />, + cell: ({ row }) => ( + + + ), + }, { + id: "cpu", + accessorFn: (row) => row.observation.cpu?.usage_seconds_total ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => { + const cpu = row.original.observation.status === "observed" ? row.original.observation.cpu : null; + return {formatDashboardDuration(cpu?.usage_seconds_total ?? null)}{typeof cpu?.capacity_cores === "number" ? `${cpu.capacity_cores.toLocaleString("en-US")} cores` : "Capacity unknown"}; + }, + }, { + id: "memory", + accessorFn: (row) => row.observation.memory?.usage_bytes ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => { + const memory = row.original.observation.status === "observed" ? row.original.observation.memory : null; + const memoryPercent = percent(memory?.usage_bytes, memory?.limit_bytes); + return ( + + {formatDashboardBytes(memory?.usage_bytes ?? null)} + {memory?.limit_bytes == null ? "Limit unknown" : `of ${formatDashboardBytes(memory.limit_bytes)}`} + {memoryPercent !== null ? : null} + + ); + }, + }, { + id: "uptime", + accessorFn: (row) => row.computeUptimeSeconds ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {formatDashboardDuration(row.original.computeUptimeSeconds)}{row.original.allocationAgeSeconds === null ? "Allocation age unknown" : `${formatDashboardDuration(row.original.allocationAgeSeconds)} allocated`}, + }, { + id: "sessionStatus", + accessorFn: (row) => row.session.status, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {dashboardStatusLabel(row.original.session.status)}, + }, { + id: "tokens", + accessorFn: (row) => row.session.totalTokens ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {formatDashboardTokens(row.original.session.totalTokens)}{row.original.session.totalTokens === null ? "Not reported" : "Session reported"}, + }], [onOpenSession]); + const table = useReactTable({ + data: filteredRows, + columns, + state: { sorting, globalFilter }, + onSortingChange: setSorting, + onGlobalFilterChange: setGlobalFilter, + globalFilterFn: runtimeGlobalFilter, + getCoreRowModel: getCoreRowModel(), + getFilteredRowModel: getFilteredRowModel(), + getSortedRowModel: getSortedRowModel(), + getPaginationRowModel: getPaginationRowModel(), + initialState: { pagination: { pageIndex: 0, pageSize: PAGE_SIZE } }, + }); + const visibleRows = table.getFilteredRowModel().rows.length; + + return ( +
+
+
+

Runtime targets

+

Read-only Session navigation · missing measurements remain unknown

+
+ {visibleRows.toLocaleString("en-US")} visible +
+
+ + + +
+
+
+ + {table.getHeaderGroups().map((headerGroup) => ( + + {headerGroup.headers.map((header) => )} + + ))} + + + {table.getRowModel().rows.map((row) => ( + + {row.getVisibleCells().map((cell) => )} + + ))} + +
{flexRender(header.column.columnDef.header, header.getContext())}
{flexRender(cell.column.columnDef.cell, cell.getContext())}
+ {visibleRows === 0 ?

No Runtime targets match these filters.

: null} + + {table.getPageCount() > 1 ? ( +
+ Page {table.getState().pagination.pageIndex + 1} of {table.getPageCount()} +
+ + +
+
+ ) : null} + + ); +} + +export function RuntimeObservabilityContent({ + snapshot, + stale, + onOpenSession, +}: { + snapshot: RuntimeDashboardSnapshot; + stale: boolean; + onOpenSession: (sessionId: string) => void; +}) { + const model = useMemo(() => buildRuntimeDashboardModel(snapshot.sessions, snapshot.observations), [snapshot]); + const summary = model.summary; + + return ( + <> +
+ } label="Active Runtimes" value={summary.observedRuntimeCount.toLocaleString("en-US")} detail={`${summary.managedRuntimeCount} managed · ${summary.unavailableRuntimeCount} unavailable`} /> + } label="CPU time / capacity" value={summary.cpuUsageSecondsTotal === null && summary.cpuCapacityCores === null ? "No current sample" : `${formatDashboardDuration(summary.cpuUsageSecondsTotal)} / ${summary.cpuCapacityCores?.toLocaleString("en-US") ?? "—"} cores`} detail={`${summary.cpuCoverageCount}/${summary.observedRuntimeCount} observed Runtimes report CPU`} /> + } label="Memory" value={summary.memoryUsageBytes === null && summary.memoryLimitBytes === null ? "No current sample" : `${formatDashboardBytes(summary.memoryUsageBytes)} / ${formatDashboardBytes(summary.memoryLimitBytes)}`} detail={`${summary.memoryCoverageCount}/${summary.observedRuntimeCount} observed Runtimes report memory`} /> + } label="Reported tokens" value={formatDashboardTokens(summary.totalTokens)} detail={`${summary.tokenCoverageCount}/${summary.sessionCount} Sessions report usage`} /> +
+ +
+
+

Runtime health

Status and observation freshness by Session

{summary.sessionCount} contexts
+
+ {model.rows.slice(0, 8).map((row) => ( + + ))} + {model.rows.length > 8 ? +{model.rows.length - 8} more : null} +
+
+
+

Usage coverage

Unknown remains unknown, never zero

{stale ? "retained snapshot" : "current snapshot"}
+
+
CPU {summary.cpuCoverageCount}/{summary.observedRuntimeCount}{coverage(summary.cpuCoverageCount, summary.observedRuntimeCount)}
+
Memory {summary.memoryCoverageCount}/{summary.observedRuntimeCount}{coverage(summary.memoryCoverageCount, summary.observedRuntimeCount)}
+
Tokens {summary.tokenCoverageCount}/{summary.sessionCount}{coverage(summary.tokenCoverageCount, summary.sessionCount)}
+
Unavailable {summary.unavailableRuntimeCount}current observations
+
Unsupported {summary.unsupportedRuntimeCount}self-hosted or none
+
+
+
+ + + + ); +} diff --git a/apps/web/src/features/dashboard/dashboard-model.test.ts b/apps/web/src/features/dashboard/dashboard-model.test.ts index e638c883f..03a167088 100644 --- a/apps/web/src/features/dashboard/dashboard-model.test.ts +++ b/apps/web/src/features/dashboard/dashboard-model.test.ts @@ -1,13 +1,17 @@ import { describe, expect, it } from "vitest"; -import type { AgentSession, SavedAgent, TokenUsage } from "@agents-core-web/agents-client"; +import type { AgentSession, RuntimeObservation, SavedAgent, TokenUsage } from "@agents-core-web/agents-client"; import { buildDashboardSnapshot, + buildRuntimeDashboardModel, dashboardEnvironmentLabel, dashboardEnvironmentProfile, dashboardStatusLabel, formatDashboardTimestamp, + formatDashboardBytes, + formatDashboardDuration, + runtimeObservationStatusLabel, } from "./dashboard-model"; function agent(id: string, overrides: Partial = {}): SavedAgent { @@ -228,4 +232,140 @@ describe("Dashboard loaded-snapshot model", () => { model: "fixture/fallback", }); }); + + it("aggregates only present Runtime measurements and preserves coverage", () => { + const managed = session("11111111-1111-4111-8111-111111111111", { usage: usage(21) }); + const unsupported = session("22222222-2222-4222-8222-222222222222"); + const observations: RuntimeObservation[] = [{ + id: managed.id, + object: "agent.runtime_observation", + session_id: managed.id, + environment_id: "33333333-3333-4333-8333-333333333333", + mode: "openai_hosted", + provider_type: "docker", + instance: { kind: "managed_allocation", allocation_id: "44444444-4444-4444-8444-444444444444", device_id: null, connection_generation: null }, + status: "observed", + reason: null, + allocation_created_at: 100, + resolved_at: 220, + observed_at: 210, + started_at: 150, + cpu: { usage_seconds_total: 3.5, capacity_cores: 2, usage_cores: null, utilization_ratio: null }, + memory: { usage_bytes: 512, limit_bytes: 2048 }, + }, { + id: unsupported.id, + object: "agent.runtime_observation", + session_id: unsupported.id, + environment_id: null, + mode: "none", + provider_type: null, + instance: { kind: "none", allocation_id: null, device_id: null, connection_generation: null }, + status: "unsupported", + reason: "runtime_mode_not_observable", + allocation_created_at: null, + resolved_at: 225, + observed_at: null, + started_at: null, + cpu: null, + memory: null, + }]; + const model = buildRuntimeDashboardModel([managed, unsupported], observations); + + expect(model.summary).toMatchObject({ + sessionCount: 2, + managedRuntimeCount: 1, + observedRuntimeCount: 1, + unavailableRuntimeCount: 0, + unsupportedRuntimeCount: 1, + cpuUsageSecondsTotal: 3.5, + cpuCapacityCores: 2, + cpuCoverageCount: 1, + memoryUsageBytes: 512, + memoryLimitBytes: 2048, + memoryCoverageCount: 1, + totalTokens: 21, + tokenCoverageCount: 1, + oldestResolvedAt: 220, + newestResolvedAt: 225, + }); + expect(model.rows[0]?.computeUptimeSeconds).toBe(60); + expect(model.rows[0]?.allocationAgeSeconds).toBe(120); + expect(runtimeObservationStatusLabel(observations[0]!)).toBe("Observed"); + expect(formatDashboardBytes(2048)).toBe("2.00 KiB"); + expect(formatDashboardDuration(90)).toBe("1m 30s"); + }); + + it("does not infer a released allocation lifetime from the current resolution time", () => { + const stopped = session("11111111-1111-4111-8111-111111111111"); + const observation: RuntimeObservation = { + id: stopped.id, + object: "agent.runtime_observation", + session_id: stopped.id, + environment_id: "33333333-3333-4333-8333-333333333333", + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: "44444444-4444-4444-8444-444444444444", + device_id: null, + connection_generation: null, + }, + status: "unavailable", + reason: "runtime_not_running", + allocation_created_at: 100, + resolved_at: 10_000, + observed_at: null, + started_at: null, + cpu: null, + memory: null, + }; + + expect(buildRuntimeDashboardModel([stopped], [observation]).rows[0]?.allocationAgeSeconds).toBeNull(); + }); + + it("does not count capacity-only or limit-only samples as usage coverage", () => { + const managed = session("11111111-1111-4111-8111-111111111111", { + environment: { + type: "openai_hosted", + id: "33333333-3333-4333-8333-333333333333", + capability_directories: [], + network: { access: "disabled", allowed_domains: [] }, + packages: { npm: [], python: [], system: [] }, + files: [], + plugins: [], + skills: [], + }, + }); + const observation: RuntimeObservation = { + id: managed.id, + object: "agent.runtime_observation", + session_id: managed.id, + environment_id: "33333333-3333-4333-8333-333333333333", + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: "44444444-4444-4444-8444-444444444444", + device_id: null, + connection_generation: null, + }, + status: "observed", + reason: null, + allocation_created_at: null, + resolved_at: 220, + observed_at: 210, + started_at: null, + cpu: { usage_seconds_total: null, capacity_cores: 2, usage_cores: null, utilization_ratio: null }, + memory: { usage_bytes: null, limit_bytes: 2048 }, + }; + + expect(buildRuntimeDashboardModel([managed], [observation]).summary).toMatchObject({ + cpuUsageSecondsTotal: null, + cpuCapacityCores: 2, + cpuCoverageCount: 0, + memoryUsageBytes: null, + memoryLimitBytes: 2048, + memoryCoverageCount: 0, + }); + }); }); diff --git a/apps/web/src/features/dashboard/dashboard-model.ts b/apps/web/src/features/dashboard/dashboard-model.ts index 2239bce16..5c31b55f1 100644 --- a/apps/web/src/features/dashboard/dashboard-model.ts +++ b/apps/web/src/features/dashboard/dashboard-model.ts @@ -1,5 +1,6 @@ import type { AgentSession, + RuntimeObservation, SavedAgent, SessionStatus, TokenUsage, @@ -45,6 +46,36 @@ export interface DashboardSnapshot { recentSessions: DashboardSessionRow[]; } +export interface RuntimeDashboardRow { + session: DashboardSessionRow; + observation: RuntimeObservation; + computeUptimeSeconds: number | null; + allocationAgeSeconds: number | null; +} + +export interface RuntimeDashboardSummary { + sessionCount: number; + managedRuntimeCount: number; + observedRuntimeCount: number; + unavailableRuntimeCount: number; + unsupportedRuntimeCount: number; + cpuUsageSecondsTotal: number | null; + cpuCapacityCores: number | null; + cpuCoverageCount: number; + memoryUsageBytes: number | null; + memoryLimitBytes: number | null; + memoryCoverageCount: number; + totalTokens: number | null; + tokenCoverageCount: number; + oldestResolvedAt: number | null; + newestResolvedAt: number | null; +} + +export interface RuntimeDashboardModel { + summary: RuntimeDashboardSummary; + rows: RuntimeDashboardRow[]; +} + const sessionStatuses = new Set([ "idle", "in_progress", @@ -229,3 +260,189 @@ export function formatDashboardTimestamp(value: number | null): string { if (seconds === null) return "Unknown"; return `${new Date(seconds * 1_000).toISOString().slice(0, 16).replace("T", " ")} UTC`; } + +function safeFiniteNonNegative(value: unknown): number | null { + return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null; +} + +function safeAdd(left: number, right: number): number | null { + const value = left + right; + return Number.isSafeInteger(left) && Number.isSafeInteger(right) && Number.isSafeInteger(value) ? value : null; +} + +function elapsedSeconds(start: number | null, end: number | null): number | null { + if (start === null || end === null || end < start) return null; + return end - start; +} + +export function buildRuntimeDashboardModel( + sessions: readonly AgentSession[], + observations: readonly RuntimeObservation[], +): RuntimeDashboardModel { + const sessionsById = new Map(sessions.map((session) => [session.id, session])); + const rows: RuntimeDashboardRow[] = []; + let managedRuntimeCount = 0; + let observedRuntimeCount = 0; + let unavailableRuntimeCount = 0; + let unsupportedRuntimeCount = 0; + let cpuUsageSecondsTotal = 0; + let cpuUsageKnown = false; + let cpuUsageSafe = true; + let cpuCapacityCores = 0; + let cpuCapacityKnown = false; + let cpuCapacitySafe = true; + let cpuCoverageCount = 0; + let memoryUsageBytes = 0; + let memoryUsageKnown = false; + let memoryUsageSafe = true; + let memoryLimitBytes = 0; + let memoryLimitKnown = false; + let memoryLimitSafe = true; + let memoryCoverageCount = 0; + let totalTokens = 0; + let tokensKnown = false; + let tokensSafe = true; + let tokenCoverageCount = 0; + let oldestResolvedAt: number | null = null; + let newestResolvedAt: number | null = null; + + for (const observation of observations) { + const session = sessionsById.get(observation.session_id); + if (!session) continue; + const sessionRow = toSessionRow(session); + const resolvedAt = canonicalTimestamp(observation.resolved_at); + if (resolvedAt !== null) { + oldestResolvedAt = oldestResolvedAt === null ? resolvedAt : Math.min(oldestResolvedAt, resolvedAt); + newestResolvedAt = newestResolvedAt === null ? resolvedAt : Math.max(newestResolvedAt, resolvedAt); + } + if (observation.mode === "openai_hosted") managedRuntimeCount += 1; + if (observation.status === "observed") { + observedRuntimeCount += 1; + const cpuUsage = safeFiniteNonNegative(observation.cpu?.usage_seconds_total); + const cpuCapacity = safeFiniteNonNegative(observation.cpu?.capacity_cores); + if (cpuUsage !== null) { + const next = cpuUsageSecondsTotal + cpuUsage; + if (Number.isFinite(next)) { + cpuUsageSecondsTotal = next; + cpuUsageKnown = true; + } else cpuUsageSafe = false; + } + if (cpuCapacity !== null) { + const next = cpuCapacityCores + cpuCapacity; + if (Number.isFinite(next)) { + cpuCapacityCores = next; + cpuCapacityKnown = true; + } else cpuCapacitySafe = false; + } + if (cpuUsage !== null) cpuCoverageCount += 1; + + const memoryUsage = safeNonNegativeInteger(observation.memory?.usage_bytes); + const memoryLimit = safeNonNegativeInteger(observation.memory?.limit_bytes); + if (memoryUsage !== null) { + const next = safeAdd(memoryUsageBytes, memoryUsage); + if (next !== null) { + memoryUsageBytes = next; + memoryUsageKnown = true; + } else memoryUsageSafe = false; + } + if (memoryLimit !== null) { + const next = safeAdd(memoryLimitBytes, memoryLimit); + if (next !== null) { + memoryLimitBytes = next; + memoryLimitKnown = true; + } else memoryLimitSafe = false; + } + if (memoryUsage !== null) memoryCoverageCount += 1; + } else if (observation.status === "unavailable") { + unavailableRuntimeCount += 1; + } else { + unsupportedRuntimeCount += 1; + } + + if (sessionRow.totalTokens !== null) { + const next = safeAdd(totalTokens, sessionRow.totalTokens); + if (next !== null) { + totalTokens = next; + tokensKnown = true; + } else tokensSafe = false; + tokenCoverageCount += 1; + } + rows.push({ + session: sessionRow, + observation, + computeUptimeSeconds: observation.status === "observed" + ? elapsedSeconds(canonicalTimestamp(observation.started_at), canonicalTimestamp(observation.observed_at)) + : null, + allocationAgeSeconds: observation.mode === "openai_hosted" && observation.reason !== "runtime_not_running" + ? elapsedSeconds(canonicalTimestamp(observation.allocation_created_at), resolvedAt) + : null, + }); + } + + const statusOrder = { observed: 0, unavailable: 1, unsupported: 2 } as const; + rows.sort((left, right) => ( + statusOrder[left.observation.status] - statusOrder[right.observation.status] + || compareSessionRows(left.session, right.session) + )); + return { + summary: { + sessionCount: rows.length, + managedRuntimeCount, + observedRuntimeCount, + unavailableRuntimeCount, + unsupportedRuntimeCount, + cpuUsageSecondsTotal: cpuUsageKnown && cpuUsageSafe ? cpuUsageSecondsTotal : null, + cpuCapacityCores: cpuCapacityKnown && cpuCapacitySafe ? cpuCapacityCores : null, + cpuCoverageCount, + memoryUsageBytes: memoryUsageKnown && memoryUsageSafe ? memoryUsageBytes : null, + memoryLimitBytes: memoryLimitKnown && memoryLimitSafe ? memoryLimitBytes : null, + memoryCoverageCount, + totalTokens: tokensKnown && tokensSafe ? totalTokens : null, + tokenCoverageCount, + oldestResolvedAt, + newestResolvedAt, + }, + rows, + }; +} + +export function formatDashboardBytes(value: number | null): string { + if (value === null) return "Unavailable"; + const units = ["B", "KiB", "MiB", "GiB", "TiB"]; + let amount = value; + let index = 0; + while (amount >= 1024 && index < units.length - 1) { + amount /= 1024; + index += 1; + } + const digits = amount >= 100 || index === 0 ? 0 : amount >= 10 ? 1 : 2; + return `${amount.toFixed(digits)} ${units[index]}`; +} + +export function formatDashboardDuration(value: number | null): string { + if (value === null || !Number.isFinite(value) || value < 0) return "Unavailable"; + const seconds = Math.floor(value); + if (seconds < 60) return `${seconds}s`; + const minutes = Math.floor(seconds / 60); + if (minutes < 60) return `${minutes}m ${seconds % 60}s`; + const hours = Math.floor(minutes / 60); + if (hours < 24) return `${hours}h ${minutes % 60}m`; + return `${Math.floor(hours / 24)}d ${hours % 24}h`; +} + +export function formatDashboardTokens(value: number | null): string { + if (value === null) return "Unavailable"; + return value.toLocaleString("en-US"); +} + +export function runtimeObservationStatusLabel(observation: RuntimeObservation): string { + if (observation.status === "observed") return "Observed"; + if (observation.status === "unsupported") return "Unsupported"; + switch (observation.reason) { + case "allocation_pending": return "Allocation pending"; + case "runtime_not_running": return "Not running"; + case "source_not_configured": return "Source unavailable"; + case "sample_timeout": return "Sample timeout"; + case "sample_unavailable": return "Sample unavailable"; + } +} diff --git a/apps/web/src/features/dashboard/runtime-snapshot.test.ts b/apps/web/src/features/dashboard/runtime-snapshot.test.ts new file mode 100644 index 000000000..4e852e59a --- /dev/null +++ b/apps/web/src/features/dashboard/runtime-snapshot.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from "vitest"; + +import type { + AgentCore, + AgentSession, + RuntimeObservation, +} from "@agents-core-web/agents-client"; + +import { + loadRuntimeDashboardSnapshot, + RuntimeSnapshotIncompleteError, +} from "./runtime-snapshot"; + +const session = { id: "11111111-1111-4111-8111-111111111111" } as AgentSession; +const observation = { + id: session.id, + session_id: session.id, +} as RuntimeObservation; + +function core( + sessions: AgentSession[], + observations: RuntimeObservation[], +): Pick { + return { + listSessions: async () => ({ object: "list", data: sessions, has_more: false, first_id: session.id, last_id: session.id }), + listRuntimeObservations: async () => ({ object: "list", data: observations, has_more: false, first_id: session.id, last_id: session.id }), + }; +} + +describe("Runtime Dashboard snapshot coordination", () => { + it("publishes only exact Session and observation identity sets", async () => { + const value = await loadRuntimeDashboardSnapshot(core([session], [observation]), () => 4); + expect(value?.sessions).toEqual([session]); + expect(value?.observations).toEqual([observation]); + + await expect(loadRuntimeDashboardSnapshot(core([session], []), () => 4)).rejects.toBeInstanceOf( + RuntimeSnapshotIncompleteError, + ); + }); + + it("discards a candidate when local Session state changes during collection", async () => { + let revision = 1; + const changing = core([session], [observation]); + changing.listRuntimeObservations = async () => { + revision += 1; + return { object: "list", data: [observation], has_more: false, first_id: session.id, last_id: session.id }; + }; + + await expect(loadRuntimeDashboardSnapshot(changing, () => revision)).resolves.toBeNull(); + }); + + it("fails closed when the target budget is exceeded", async () => { + await expect(loadRuntimeDashboardSnapshot(core([session], [observation]), () => 1, undefined, 0)).rejects.toThrow( + "target budget", + ); + }); +}); diff --git a/apps/web/src/features/dashboard/runtime-snapshot.ts b/apps/web/src/features/dashboard/runtime-snapshot.ts new file mode 100644 index 000000000..79d794406 --- /dev/null +++ b/apps/web/src/features/dashboard/runtime-snapshot.ts @@ -0,0 +1,59 @@ +import type { + AgentCore, + AgentSession, + RuntimeObservation, +} from "@agents-core-web/agents-client"; + +import { listAllCollectionPages } from "../../lib/collection-pagination"; + +export const RUNTIME_SNAPSHOT_TARGET_LIMIT = 10_000; +export const RUNTIME_SNAPSHOT_TIMEOUT_MS = 15_000; +export const RUNTIME_SNAPSHOT_REFRESH_MS = 30_000; + +export interface RuntimeDashboardSnapshot { + sessions: AgentSession[]; + observations: RuntimeObservation[]; + loadedAt: number; +} + +export class RuntimeSnapshotIncompleteError extends Error { + constructor(message: string) { + super(message); + this.name = "RuntimeSnapshotIncompleteError"; + } +} + +function identitySet(values: readonly { id: string }[]): Set { + return new Set(values.map((value) => value.id)); +} + +function setsEqual(left: ReadonlySet, right: ReadonlySet): boolean { + if (left.size !== right.size) return false; + for (const value of left) if (!right.has(value)) return false; + return true; +} + +export async function loadRuntimeDashboardSnapshot( + core: Pick, + readSessionRevision: () => number, + signal?: AbortSignal, + targetLimit = RUNTIME_SNAPSHOT_TARGET_LIMIT, +): Promise { + const revision = readSessionRevision(); + const [sessions, observations] = await Promise.all([ + listAllCollectionPages((options) => core.listSessions(options), signal), + listAllCollectionPages((options) => core.listRuntimeObservations(options), signal), + ]); + signal?.throwIfAborted(); + + if (revision !== readSessionRevision()) return null; + if (sessions.length > targetLimit || observations.length > targetLimit) { + throw new RuntimeSnapshotIncompleteError("Runtime snapshot exceeded the Web target budget."); + } + if (!setsEqual(identitySet(sessions), identitySet(observations))) { + throw new RuntimeSnapshotIncompleteError( + "Sessions changed while Runtime observations were loading. The previous complete snapshot was retained.", + ); + } + return { sessions, observations, loadedAt: Date.now() }; +} diff --git a/contracts/agents-api/README.md b/contracts/agents-api/README.md index 9cfad97fe..49250c862 100644 --- a/contracts/agents-api/README.md +++ b/contracts/agents-api/README.md @@ -102,6 +102,16 @@ paths start at `/vaults`, not `/agents/vaults`. | vaults | create, retrieve, list, delete | Create/retrieve/list/delete with independent tenant persistence, stored status filtering, atomic Credential cascade and frozen Session attachments; archive semantics and full hosted lifecycle parity remain missing | | vaults.credentials | create, retrieve, update, list, delete | Static-bearer and OAuth create/retrieve/list/replacement/deletion with scoped encrypted storage and dispatch-time refresh; Session attachment and exact-URL HTTPS MCP binding; archive semantics and full hosted lifecycle parity remain missing | +## Core extension inventory + +The operations below are implemented public Core extensions. They are excluded +from the 42-operation upstream inventory and must not be counted as OpenAI Agents +compatibility. + +| Extension | Operations | Current coverage | +| --- | --- | --- | +| Runtime observations | `GET /v1/agents/runtime-observations`; `GET /v1/agents/sessions/{session_id}/runtime-observation` | Current, read-only, tenant-scoped Session contexts with stable Session-keyset pagination, bounded concurrent sampling, Docker metrics, explicit unsupported/unavailable states, strict `packages/agents-client` projection, and no lifecycle mutation. Kubernetes, E2B, self-hosted telemetry, history, CPU-rate derivation, and automatic idle policy remain unimplemented. See [Runtime observation API](runtime-observability-api.md). | + For each resource, verify the referenced request/response unions and observable behavior, not just the route. Non-text initial input, configuration options, text/image content, function results, environment variants, full Item/SSE diff --git a/contracts/agents-api/openapi.yaml b/contracts/agents-api/openapi.yaml index c3a8fcfef..d81a55fa5 100644 --- a/contracts/agents-api/openapi.yaml +++ b/contracts/agents-api/openapi.yaml @@ -1027,6 +1027,180 @@ definitions: required: - type type: object + v1.RuntimeCPUObservation: + properties: + capacity_cores: + minimum: 5e-324 + type: number + x-nullable: true + usage_cores: + minimum: 0 + type: number + x-nullable: true + usage_seconds_total: + minimum: 0 + type: number + x-nullable: true + utilization_ratio: + minimum: 0 + type: number + x-nullable: true + required: + - capacity_cores + - usage_cores + - usage_seconds_total + - utilization_ratio + type: object + v1.RuntimeInstance: + properties: + allocation_id: + format: uuid + type: string + x-nullable: true + connection_generation: + format: uuid + type: string + x-nullable: true + device_id: + format: uuid + type: string + x-nullable: true + kind: + enum: + - managed_allocation + - self_hosted_connection + - none + type: string + required: + - allocation_id + - connection_generation + - device_id + - kind + type: object + v1.RuntimeMemoryObservation: + properties: + limit_bytes: + minimum: 1 + type: integer + x-nullable: true + usage_bytes: + minimum: 0 + type: integer + x-nullable: true + required: + - limit_bytes + - usage_bytes + type: object + v1.RuntimeObservation: + properties: + allocation_created_at: + minimum: 0 + type: integer + x-nullable: true + cpu: + allOf: + - $ref: '#/definitions/v1.RuntimeCPUObservation' + x-nullable: true + environment_id: + format: uuid + type: string + x-nullable: true + id: + format: uuid + type: string + instance: + $ref: '#/definitions/v1.RuntimeInstance' + memory: + allOf: + - $ref: '#/definitions/v1.RuntimeMemoryObservation' + x-nullable: true + mode: + enum: + - none + - self_hosted + - openai_hosted + type: string + object: + enum: + - agent.runtime_observation + type: string + observed_at: + minimum: 0 + type: integer + x-nullable: true + provider_type: + type: string + x-nullable: true + reason: + enum: + - runtime_mode_not_observable + - allocation_pending + - runtime_not_running + - source_not_configured + - sample_timeout + - sample_unavailable + type: string + x-nullable: true + resolved_at: + minimum: 0 + type: integer + session_id: + format: uuid + type: string + started_at: + minimum: 0 + type: integer + x-nullable: true + status: + enum: + - observed + - unsupported + - unavailable + type: string + required: + - allocation_created_at + - cpu + - environment_id + - id + - instance + - memory + - mode + - object + - observed_at + - provider_type + - reason + - resolved_at + - session_id + - started_at + - status + type: object + v1.RuntimeObservationList: + properties: + data: + items: + $ref: '#/definitions/v1.RuntimeObservation' + type: array + first_id: + format: uuid + type: string + x-nullable: true + has_more: + type: boolean + last_id: + format: uuid + type: string + x-nullable: true + object: + enum: + - list + type: string + required: + - data + - first_id + - has_more + - last_id + - object + type: object v1.SavedAgent: properties: created_at: @@ -2703,6 +2877,68 @@ paths: summary: Update an Environment Template tags: - Environment Templates + /agents/runtime-observations: + get: + description: Core extension listing one current Runtime context per tenant-owned + Session in Session creation order. Each row has an independent resolved_at + and optional provider observed_at; the page is not an atomic telemetry snapshot. + parameters: + - description: agents=v1 + in: header + name: OpenAI-Beta + required: true + type: string + - description: Last observation ID from the previous page + in: query + name: after + type: string + - default: 20 + description: Page size + in: query + maximum: 100 + minimum: 1 + name: limit + type: integer + - default: desc + description: Session creation order + enum: + - asc + - desc + in: query + name: order + type: string + produces: + - application/json + responses: + "200": + description: OK + schema: + $ref: '#/definitions/v1.RuntimeObservationList' + "400": + description: Bad Request + schema: + $ref: '#/definitions/v1.ErrorResponse' + "401": + description: Unauthorized + schema: + $ref: '#/definitions/v1.ErrorResponse' + "404": + description: Not Found + schema: + $ref: '#/definitions/v1.ErrorResponse' + "500": + description: Internal Server Error + schema: + $ref: '#/definitions/v1.ErrorResponse' + "503": + description: Service Unavailable + schema: + $ref: '#/definitions/v1.ErrorResponse' + security: + - BearerAuth: [] + summary: List current Runtime observations + tags: + - Runtime observations /agents/sessions: get: description: Cursor and results are scoped to the authenticated execution tenant. @@ -3489,6 +3725,53 @@ paths: summary: List persisted execution Items tags: - Items + /agents/sessions/{session_id}/runtime-observation: + get: + description: Core extension returning one tenant-scoped, read-only current Runtime + observation. It never provisions, renews, restarts, pauses or stops compute. + parameters: + - description: agents=v1 + in: header + name: OpenAI-Beta + required: true + type: string + - description: Session ID + in: path + name: session_id + required: true + type: string + produces: + - application/json + responses: + "200": + description: OK + schema: + $ref: '#/definitions/v1.RuntimeObservation' + "400": + description: Bad Request + schema: + $ref: '#/definitions/v1.ErrorResponse' + "401": + description: Unauthorized + schema: + $ref: '#/definitions/v1.ErrorResponse' + "404": + description: Not Found + schema: + $ref: '#/definitions/v1.ErrorResponse' + "500": + description: Internal Server Error + schema: + $ref: '#/definitions/v1.ErrorResponse' + "503": + description: Service Unavailable + schema: + $ref: '#/definitions/v1.ErrorResponse' + security: + - BearerAuth: [] + summary: Retrieve a Session Runtime observation + tags: + - Runtime observations /agents/sessions/{session_id}/subagents: get: description: Includes nested and closed Subagents. Cursors belong to the same diff --git a/contracts/agents-api/runtime-observability-api.md b/contracts/agents-api/runtime-observability-api.md new file mode 100644 index 000000000..4d6345a51 --- /dev/null +++ b/contracts/agents-api/runtime-observability-api.md @@ -0,0 +1,271 @@ +# Runtime observation API + +Status: Phase 2 and initial Core Web consumption implemented. The current-snapshot routes, strict +`packages/agents-client` projection, and generated `openapi.yaml` contract are +implemented and consumed by the Dashboard through complete Session/observation +identity joins. Historical queries and lifecycle controls remain outside this phase. + +This is an Agents Core extension, not an upstream OpenAI Agents resource. The +implementation must record that status in the coverage ledger and generated +OpenAPI contract. + +## Routes + +### List current Runtime observations + +```http +GET /v1/agents/runtime-observations?after={target_id}&limit=20&order=desc +OpenAI-Beta: agents=v1 +Authorization: Bearer ... +``` + +| Field | Rules | +| --- | --- | +| `after` | Observation ID from the previous page. Optional, supplied once. | +| `limit` | Integer 1–100, default 20. | +| `order` | `asc` or `desc`, default `desc`. | + +The list contains one current Runtime context for every Session visible to the +authenticated tenant, including explicit `none`, unsupported `self_hosted`, and +released managed contexts. Ordering uses the same Session creation-time and ID +keyset as the Session list. An observation ID is the Session UUID, so pagination +does not change when the underlying Runtime incarnation changes. Pages are not an +atomic telemetry snapshot; every row has its own `resolved_at`, and a successful +provider sample has its own `observed_at`. A client completes the entire page chain +before publishing a new Dashboard snapshot. + +```json +{ + "object": "list", + "data": [ + { + "id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", + "object": "agent.runtime_observation", + "session_id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", + "environment_id": "6c02fb71-5fa8-4298-93e8-57c6625a3fc2", + "mode": "openai_hosted", + "provider_type": "docker", + "instance": { + "kind": "managed_allocation", + "allocation_id": "d23ab94e-e40b-45bd-93a2-444f1f74642b", + "device_id": "2e434f4f-76aa-4e54-a707-4757036d90ef", + "connection_generation": null + }, + "status": "observed", + "reason": null, + "allocation_created_at": 1789951200, + "resolved_at": 1789953021, + "observed_at": 1789953020, + "started_at": 1789951220, + "cpu": { + "usage_seconds_total": 482.75, + "capacity_cores": 2.0, + "usage_cores": null, + "utilization_ratio": null + }, + "memory": { + "usage_bytes": 805306368, + "limit_bytes": 2147483648 + } + } + ], + "has_more": false, + "first_id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", + "last_id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3" +} +``` + +### Retrieve one Session's current Runtime observation + +```http +GET /v1/agents/sessions/{session_id}/runtime-observation +OpenAI-Beta: agents=v1 +Authorization: Bearer ... +``` + +This returns the same object shape as a list item. It never starts a Turn, creates +an Environment, provisions compute, renews a lease, or changes lifecycle state. + +A valid `environment:none` Session returns `200` with status `unsupported`; the +Session exists but has no attributable Runtime instance. A missing or foreign +Session returns the existing indistinguishable not-found error. + +## Resource schema + +### `RuntimeObservation` + +| Field | Type | Required | Semantics | +| --- | --- | --- | --- | +| `id` | string | yes | Session UUID; stable identity of this current-observation resource and its list cursor. | +| `object` | literal | yes | `agent.runtime_observation`. | +| `session_id` | string | yes | Authorized Core Session. | +| `environment_id` | string or null | yes | Null only for mode `none`. | +| `mode` | enum | yes | `none`, `self_hosted`, `openai_hosted`. | +| `provider_type` | string or null | yes | Forward-compatible safe source kind such as `docker`; null when no provider applies. Clients must not treat an unknown nonempty value as an error. | +| `instance` | object | yes | Provider-neutral current incarnation identity; explicit `kind=none` when no compute applies. | +| `status` | enum | yes | `observed`, `unsupported`, `unavailable`. | +| `reason` | enum or null | yes | Safe reason when status is not `observed`. | +| `allocation_created_at` | integer or null | yes | Unix seconds for managed allocation age. | +| `resolved_at` | integer | yes | Unix seconds when Core resolved identity and status for this row. | +| `observed_at` | integer or null | yes | Provider sample time; null without a sample. | +| `started_at` | integer or null | yes | Current compute incarnation start time. | +| `cpu` | object or null | yes | Null when no CPU fields were observed. | +| `memory` | object or null | yes | Null when no memory fields were observed. | + +### `RuntimeInstance` + +```json +{ + "kind": "managed_allocation", + "allocation_id": "alloc_...", + "device_id": "device_...", + "connection_generation": null +} +``` + +`kind` is `managed_allocation`, `self_hosted_connection`, or `none`. For a managed +context, `allocation_id` is the incarnation key; for self-hosted, the current +`connection_generation` is the incarnation key. Fields that do not apply are +explicit nulls. Provider-native container IDs, pod names, host paths, credentials, +and raw labels are not public fields. + +### `RuntimeCPUObservation` + +```json +{ + "usage_seconds_total": 482.75, + "capacity_cores": 2.0, + "usage_cores": 1.42, + "utilization_ratio": 0.71 +} +``` + +All fields are `number | null`. Values are finite and nonnegative; +`capacity_cores`, when present, is greater than zero. Numeric zero is observed +zero. Null is unavailable. `usage_cores` is the cumulative CPU delta divided by +the observation-time delta for two ordered samples of the same incarnation. +`utilization_ratio` is `usage_cores / capacity_cores`. It is not clamped: a value +above 1 is retained as provider/accounting evidence and is not interpreted as a +lifecycle signal. Both derived fields are null after a cache restart or whenever +either source sample is absent or invalid. The API never derives CPU rate from a +single sample. + +### `RuntimeMemoryObservation` + +```json +{ + "usage_bytes": 805306368, + "limit_bytes": 2147483648 +} +``` + +Both fields are `integer | null`. Values are nonnegative and safe JSON integers. +Zero usage is observed zero. A missing or unlimited provider limit is null. + +## Status and reason matrix + +| Status | Allowed reason | +| --- | --- | +| `observed` | null | +| `unsupported` | `runtime_mode_not_observable` | +| `unavailable` | `allocation_pending`, `runtime_not_running`, `source_not_configured`, `sample_timeout`, `sample_unavailable` | + +Ownership mismatch, malformed durable identity, corrupt provider evidence, and +authorization failure are not downgraded to unavailable rows. + +## Error responses + +Use the existing Agents API error envelope. + +| HTTP | Code | When | +| --- | --- | --- | +| 400 | `unsupported_parameter` | Unknown or duplicate query fields. | +| 400 | `invalid_request` | Empty or invalid limits, order, or malformed cursor. | +| 401 | `authentication_error` | Missing or invalid API authentication. | +| 404 | `not_found` | Missing or foreign Session/cursor, indistinguishably. | +| 500 | `internal_error` | Integrity, ownership, or invalid provider evidence. | +| 503 | `execution_unavailable` | Required Runtime observation service is not configured. | + +Errors never include provider raw responses or credentials. + +## Freshness and caching + +- Return `Cache-Control: no-store`. +- The Phase 2 implementation performs bounded direct reads and has no observation + cache. A later internal cache may coalesce reads for at most five seconds. +- `observed_at` is authoritative for freshness; HTTP response time is not. +- Clients mark samples stale according to their own explicit threshold. +- `ETag` is not proposed because observations change independently. + +## Client contract + +`packages/agents-client` exposes: + +```ts +type RuntimeObservationStatus = "observed" | "unsupported" | "unavailable"; +type RuntimeObservationReason = + | "runtime_mode_not_observable" + | "allocation_pending" + | "runtime_not_running" + | "source_not_configured" + | "sample_timeout" + | "sample_unavailable"; + +type RuntimeObservation = + | RuntimeObservedObservation + | RuntimeUnavailableObservation + | RuntimeNoneObservation + | RuntimeSelfHostedObservation; + +interface RuntimeObservationList { + object: "list"; + data: RuntimeObservation[]; + has_more: boolean; + first_id: string | null; + last_id: string | null; +} + +interface AgentCore { + listRuntimeObservations(options?: { + after?: string; + limit?: number; + order?: "asc" | "desc"; + }): Promise; + + retrieveRuntimeObservation(sessionId: string): Promise; +} +``` + +These exported variants discriminate on `status` and `mode`; their instance, +reason, timestamps, CPU, and memory fields narrow accordingly. The exact variant +definitions live in `packages/agents-client/src/types.ts` and mirror the status +and reason matrix above. + +The client validates every required field, enum, nullability rule, timestamp, and +finite number. The current pinned contract rejects unknown additive fields so an +unreviewed server expansion cannot silently cross the browser boundary. Malformed +data rejects the whole page; Web does not publish a partial snapshot. + +The generated OpenAPI 2 schema records field-level required/nullability rules, +UUID formats, reason enums, and numeric minima. OpenAPI 2 +cannot encode the complete cross-field discriminated union. The matrix above is +normative for wire consumers; the server projection and strict TypeScript +projector enforce it, and the exported TypeScript type prevents invalid +status/mode combinations in typed consumers. + +Web also applies a configured whole-refresh budget. If `has_more` remains true +when that budget is exhausted, it retains the prior complete snapshot and marks +the refresh incomplete; it does not publish partial values as global totals. +After both Runtime-observation and Session traversals complete, Web also requires +their Session ID sets to be identical. A mismatch caused by concurrent creation or +deletion makes the candidate incomplete and prevents publication. + +## Deliberately excluded + +- Token usage: use existing Session/Turn Usage. +- Billing and cost: product/backend concern. +- Historical series: optional later capability with a separate contract. +- Container logs and command output. +- Provider credentials or native configuration. +- Start, stop, pause, resume, restart, renew, or delete operations. +- Idle classification and automatic shutdown. diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md new file mode 100644 index 000000000..f3522662d --- /dev/null +++ b/contracts/agents-api/runtime-observability-design.md @@ -0,0 +1,401 @@ +# Runtime observability and Dashboard design + +Status: Phase 1 provider abstraction/Docker sampling, Phase 2 current-snapshot +API/client contract, and the initial Core Web current-snapshot Dashboard are +implemented. The history backend, additional providers, and lifecycle automation +described below are not implemented. + +## 1. Problem statement + +Operators need one Dashboard that answers four separate questions without +confusing their sources of truth: + +1. Which Runtime instances currently belong to which tenant, Session, and + Environment? +2. What compute is allocated and what is it consuming now? +3. How long has allocation, compute, and model work been active? +4. How many model tokens have been reported for the corresponding Sessions? + +The design must work across managed Docker now and later managed Kubernetes, +E2B, and authenticated self-hosted Runtime deployments. Metrics are operational +evidence. They must not become execution or lifecycle authority. + +## 2. Goals + +- Resolve every sample through durable Core identity before provider access. +- Keep provider-specific collection behind one source interface. +- Preserve observed zero, unavailable measurements, and unsupported modes as + different states. +- Provide a bounded read-only API suitable for Core Web and other operators. +- Let Web combine Runtime observations with existing Session and Turn usage + without copying execution truth into the browser. +- Keep current snapshots independent from an optional history backend. +- Define a safe path to future idle shutdown without implementing it implicitly. + +## 3. Non-goals + +- Redefining the pinned OpenAI Agents resources. +- Adding product users, organizations, billing, or authorization tables to Core. +- Treating a Session, daemon socket, container, pod, native harness Session, or + Turn as the same identity. +- Estimating missing CPU, memory, token, or duration values. +- Using telemetry, heartbeat age, low CPU, or Prometheus state to stop compute. +- Adding lifecycle actions to the first Dashboard release. +- Storing time-series samples in PostgreSQL. + +## 4. Source-of-truth model + +| Concern | Authority | Notes | +| --- | --- | --- | +| Tenant and Session ownership | Core database | Every public read is tenant-scoped. | +| Environment placement | Session configuration and Environment row | `none`, `self_hosted`, or `openai_hosted`. | +| Observation resource identity | Session ID | One current observation resource exists per tenant-owned Session. | +| Managed Runtime identity | `runtime_allocations` | Allocation and provider key identify the compute incarnation. | +| Self-hosted Runtime identity | Environment connection generation | Future telemetry must be generation-fenced. | +| Container/pod resource values | Selected provider source | Read-only, point-in-time evidence. | +| Turn state and busy duration | Core Turns | Never inferred from CPU. | +| Token usage | Existing Session/Turn usage | Missing native usage remains unknown. | +| Historical resource series | Optional telemetry backend | Not execution or lifecycle authority. | +| Idle shutdown decision | Future durable Core control state | Separate design and migration. | + +## 5. Identity chain + +```text +managed +tenant_id -> session_id -> environment_id -> runtime_allocation_id + -> provider_key -> provider-owned container/pod/instance + +self-hosted (future) +tenant_id -> session_id -> environment_id + -> device_id + connection_generation -> authenticated Runtime report + +none +tenant_id -> session_id + -> no Session-owned Runtime instance +``` + +Provider-native identifiers are never accepted from browser input. The resolver +starts from the authorized tenant and Session, loads the committed Environment and +allocation, and only then selects the configured source by persisted provider key. +The provider independently verifies its labels or equivalent ownership metadata. + +## 6. Component architecture + +```mermaid +flowchart LR + Web[Core Web Dashboard] --> Client[packages/agents-client] + Client --> API[Agents API read handlers] + API --> Service[runtimeobs.Service] + Service --> Resolver[durable identity resolver] + Resolver --> DB[(Core PostgreSQL)] + Service --> Registry[provider source registry] + Registry --> Docker[Docker Inspect and one-shot Stats] + Registry -. future .-> K8s[Kubernetes Metrics API or cAdvisor] + Registry -. future .-> E2B[E2B metrics adapter] + Registry -. future .-> Self[authenticated daemon telemetry] + Service -. optional export .-> Telemetry[OTLP or Prometheus pipeline] + Telemetry -. future history reads .-> History[operator history adapter] +``` + +### 6.1 `runtimeobs` + +Owns provider-neutral identity, mode resolution, source selection, sample +validation, and observation status. It must not import provider SDKs or mutate +Runtime lifecycle. + +### 6.2 Provider sources + +Each source receives a fully resolved target and returns one normalized sample. +A source must verify target ownership, make only bounded read calls, preserve +missing fields, return cumulative CPU seconds, and never create, renew, restart, +pause, or stop compute. + +Docker uses Inspect followed by non-streaming one-shot Stats. Kubernetes should +retain pod UID, container identity, and restart boundaries. E2B must use an +API-supported instance identity rather than display names. Self-hosted metrics +require authenticated daemon messages fenced by the current connection generation. + +### 6.3 API composition + +The API resolves durable rows first, samples sources with bounded concurrency, +maps only ordinary absence/timeouts to safe unavailable reasons, and fails closed +on ownership or integrity errors. It returns current observations only. + +### 6.4 Web composition + +Web loads the complete paginated Runtime observation collection before publishing +a new Dashboard snapshot. The collection follows the same Session creation-time +and ID keyset as the Session list, so a Runtime incarnation change cannot invalidate +pagination. It separately uses existing Session/Turn reads for status and tokens +and joins only by exact Session identity. Before publication, the set of Session +IDs from both complete traversals must be identical. Concurrent Session creation +or deletion can make the sets differ because the APIs have no shared snapshot +token; Web then discards the candidate, marks the refresh incomplete, and keeps +the previous successful snapshot visibly stale. + +The browser enforces a configured refresh budget for total pages, targets, and +elapsed time. Exhausting that budget is an incomplete refresh: Web retains the +previous complete snapshot and does not relabel partial aggregates as tenant-wide. + +## 7. Normalized sample + +```go +type Sample struct { + ObservedAt time.Time + StartedAt *time.Time + + CPUUsageSecondsTotal *float64 + CPUCapacityCores *float64 + MemoryUsageBytes *uint64 + MemoryLimitBytes *uint64 +} +``` + +Pointer presence is semantic. `0` means observed zero; `nil` means unavailable. +CPU percentage is derived from the delta between two cumulative samples and their +observation times. A single sample cannot truthfully supply CPU percentage. + +The API projection may additionally expose `usage_cores` and `utilization_ratio` +only when the service has two ordered samples for the same Runtime incarnation. +A future process-local observation cache may keep the previous cumulative value +for this calculation. Phase 2 intentionally leaves both derived fields null +because it has only one provider sample per request. Cache loss must make the +derived fields temporarily null; it must never change the cumulative source +measurement or lifecycle state. + +## 8. Duration semantics + +| UI label | Calculation | Meaning | +| --- | --- | --- | +| Allocation age | allocation `created_at` to `released_at` or now | Age of Core's allocation record. | +| Compute uptime | provider `started_at` to sample `observed_at` | Age of the current compute incarnation. | +| Busy duration | Turn `started_at` to `completed_at` or now | Time model work has been active. | +| Idle duration | future durable `idle_since` | Not available in the current design. | + +Container restart resets compute uptime but not allocation age. Dashboard labels +must not collapse these values into one generic Runtime duration. + +## 9. Collection behavior + +### 9.1 Current snapshot path + +- List one current target context for each tenant-owned Session in the same stable + Session creation-time and ID order used by the Session list. A released managed + allocation remains attributable but reports `runtime_not_running`; `none` and + unsupported `self_hosted` remain explicit rows rather than disappearing. +- Default page size 20, maximum 100. +- Sample at most eight providers concurrently. +- Default per-source budget two seconds and whole-request budget ten seconds. +- Do not retry a source call inside the HTTP request. +- An optional process-local singleflight/cache may coalesce identical reads for up + to five seconds and retain the previous cumulative sample for CPU-rate + calculation. It is an optimization only and may be lost on restart. +- Do not write samples to the Core database. + +### 9.2 Error classification + +| Condition | API result | +| --- | --- | +| Mode `none` or unsupported `self_hosted` | Row status `unsupported`. | +| Managed allocation not created yet | `unavailable`, reason `allocation_pending`. | +| Owned Runtime absent or stopped | `unavailable`, reason `runtime_not_running`. | +| Source not configured | `unavailable`, reason `source_not_configured`. | +| Source deadline | `unavailable`, reason `sample_timeout`. | +| Ownership mismatch or invalid durable identity | Fail the request and log a sanitized integrity error. | +| Database/authentication failure | Existing safe API error mapping. | + +Raw Docker, Kubernetes, E2B, daemon, host, credential, or network diagnostics are +never returned to the browser. + +## 10. Historical metrics + +Current API reads and history are separate capabilities. The initial API does not +provide charts over time. A later operator-configured adapter may query an +OTLP/Prometheus-compatible backend. Core must not make that backend mandatory for +Session execution or current snapshot reads. + +Recommended instruments are: + +- `agents.runtime.cpu.usage` cumulative seconds; +- `agents.runtime.cpu.capacity` cores; +- `agents.runtime.memory.usage` bytes; +- `agents.runtime.memory.limit` bytes; +- `agents.runtime.sample` success/unavailable count; and +- `agents.runtime.sample.duration` seconds. + +Provider type, Runtime mode, and coarse status are safe low-cardinality labels. +High-cardinality identities require tenant-scoped access and retention policies; +they are not global Prometheus labels by default. + +## 11. Dashboard information architecture + +### 11.1 Overview + +- Active managed Runtime count. +- Observed CPU usage and known configured capacity. +- Observed memory usage and known limits. +- Reported Session tokens, together with the reporting Session count. +- Data freshness and source coverage. + +Aggregates include only present measurements. Each total states its denominator, +for example, `6.4 / 12 cores across 6 of 8 active Runtimes`. Unknown is never added +as zero. + +### 11.2 Runtime table + +Each row shows Session, Agent/harness when already available from the Session +snapshot, mode, observation status, cumulative CPU time, memory, compute uptime, +Session state, and reported tokens. The semantic table supports local search, +status and mode filters, sortable columns, and bounded pagination over the last +complete snapshot. Rows navigate to the existing Session view. No stop, restart, +pause, or delete actions appear in the first release. + +### 11.3 Detail view + +The detail surface shows exact Session/Environment/allocation identity, provider +type, observation timestamps, allocation age, compute uptime, and safe unavailable +reason. It displays only the Core-owned identifiers explicitly present in the +public contract. Provider-native container IDs, pod names, instance names, host +paths, and raw labels are never displayed. + +### 11.4 States + +- **Loading:** no previous complete Runtime snapshot. +- **Fresh:** every page loaded and each row carries its own resolution time; a + provider sample also carries its independent observation time. +- **Stale:** refresh failed; previous complete snapshot retained. +- **Unavailable row:** identity is valid, measurement is temporarily absent. +- **Unsupported row:** mode is recognized but has no qualified source. +- **Integrity failure:** do not publish a partial replacement snapshot. + +### 11.5 Web implementation shape + +The Web implementation belongs in Core Web, not the Core service layer. It uses +`packages/agents-client` as the only Runtime-observation transport and keeps four +seams separate: + +1. A client/parser module validates one page and exposes list and Session-scoped + retrieval methods. +2. A refresh coordinator loads all observation pages plus the canonical Session + collection, applies page/target/time budgets, requires exact equality of their + Session ID sets, and atomically swaps only a complete joined snapshot. A set + mismatch is an incomplete refresh, not a partial success. +3. A feature-local state model retains `last_complete`, current refresh status, + local filters, and the selected time range. It aborts an overlapping refresh + and marks old data stale after a failed or incomplete refresh. +4. Presentational components render summary metrics, a compact health matrix, + explicit usage coverage, the Runtime table, and an identity detail surface. + Trend components and chart dependencies are absent unless a later history + capability and contract are configured. + +The initial implementation uses a 30-second Web cadence plus up to five seconds +of jitter and a 15-second whole-refresh budget. These are Web configuration, not +API guarantees. Web pauses periodic reads when hidden, refreshes when visibility +returns, and adds jitter so multiple browsers do not synchronize. Filtering is +local to the last complete snapshot and never changes tenant authorization or +provider selection. + +## 12. Token usage boundary + +Runtime observations do not duplicate token usage. Web uses the existing canonical +Session Usage snapshot and joins it to Runtime rows by `session_id`. The Dashboard +shows both total and coverage, such as `1.84M reported by 7/8 Sessions`. Missing or +incomplete native usage remains unknown. + +Cost and billing stay outside this Core API. A product may join billing in its own +authorized backend, never by exposing product credentials to Core Web. + +## 13. Security and tenancy + +- Authenticate with the existing Agents API mechanism. +- Scope resolution to the authenticated tenant before provider access. +- Do not accept provider key, allocation ID, container ID, pod UID, or device ID + as an authority-bearing query parameter. +- Bound per-request list size, concurrency, response bytes, and source deadlines; + Web separately bounds a complete multi-page refresh. +- Sanitize logs through `internal/obs/log`. +- Never return credentials, environment variables, Docker raw JSON, daemon status + payloads, host paths, image registry credentials, or backend credentials. +- Rate-limit collection separately from ordinary Session reads. + +## 14. Data model impact + +Current snapshot and Dashboard work require no migration. Existing +`runtime_allocations`, `environments`, Sessions, Turns, and Usage are sufficient. +No time-series table is proposed. + +Automatic idle shutdown is a separate feature. It requires durable fields such as +`activity_revision`, `idle_since`, and `shutdown_requested_at` with fenced state +transitions. That migration cannot read a monitoring backend as authority. + +## 15. Delivery plan + +### Phase 1: provider-neutral foundation + +Implemented in the Docker observability foundation. + +- `runtimeobs` identity, resolver, source, sample, and service. +- Managed Docker Inspect/Stats source. +- CPU, memory, and current compute start time. +- Explicit unsupported and unavailable states. + +### Phase 2: current snapshot API + +Implemented by the Runtime Observation extension routes and +`packages/agents-client`. The generated OpenAPI contract records the extension; +this does not add an upstream OpenAI operation. + +- Add extension types under `contracts/agents-api/v1`. +- Add collection and Session-scoped handlers. +- Add `packages/agents-client` methods and raw HTTP/client coverage. +- Add bounded concurrency, timeout, authorization, and error tests. +- Regenerate the public OpenAPI contract. + +### Phase 3: Web Dashboard + +- Add Runtime observations as a third independent Dashboard collection. +- Publish only complete traversals and retain the previous snapshot on failure. +- Join existing Session Usage and Turn status by exact Session ID. +- Add responsive, keyboard-accessible current-resource views. + +### Phase 4: optional history + +- Add telemetry exporter and qualified operator backend. +- Define a separate history query adapter and retention/security policy. +- Add trend charts only when this capability is advertised. + +### Phase 5: additional sources + +- Kubernetes, E2B, and generation-fenced self-hosted telemetry. +- Each source requires independent mechanism and deployment acceptance. + +### Phase 6: idle policy + +- Separate durable activity and shutdown state machine. +- No automatic action until race, fencing, recovery, and operator-control + acceptance is complete. + +## 16. Acceptance criteria + +- Every observation proves tenant, Session, Environment, and Runtime-instance + association. +- Managed Docker emits correct present/absent semantics and never mutates compute. +- Unsupported modes never look like zero usage. +- Collection calls are bounded and one ordinary unavailable source invents no data. +- Ownership/integrity mismatch fails closed. +- Web never publishes a partial page traversal as a current snapshot. +- Web rejects cross-collection Session membership skew, including concurrent + Session create/delete cases, before publishing tenant-wide aggregates. +- Token totals report coverage and do not estimate missing usage. +- No lifecycle action is reachable from the first Dashboard. +- No new database table is required for current snapshots or history export. + +## 17. Recorded design decisions + +1. The API is a documented Core extension rather than an upstream OpenAI resource. +2. Current snapshots have no atomic cross-row time semantics; every row + exposes its own `observed_at`. +3. History is optional and external, not a PostgreSQL sample table. +4. The first Web release has no lifecycle controls. +5. `self_hosted` remains visibly unsupported until authenticated, + generation-fenced telemetry is qualified. diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md new file mode 100644 index 000000000..658a136ed --- /dev/null +++ b/contracts/agents-api/runtime-observability.md @@ -0,0 +1,77 @@ +# Runtime observability contract + +This document defines the internal Runtime observation boundary. It does not add +an Agents API resource or change the pinned public protocol. + +## Ownership and identity + +Runtime telemetry is attributed to durable Core identity before it is sampled: + +```text +managed: tenant_id -> session_id -> environment_id -> runtime_allocation_id +self-hosted: tenant_id -> session_id -> environment_id -> device_id + connection_generation +none: tenant_id -> session_id (no Session-owned Runtime instance) +``` + +The managed allocation's persisted `provider_key` selects exactly one configured +observation source. A provider must independently verify the allocation labels or +equivalent ownership data. A Session, daemon connection, process, container, and +native harness Session are different identities and must not be substituted for +one another. + +The first implementation supports managed Docker allocations. `self_hosted` and +`none` are recognized but explicitly unsupported. A future self-hosted source must +use authenticated daemon telemetry fenced by the current connection generation. +Core must not attribute shared host statistics to an `environment:none` Session. + +## Sample semantics + +One sample contains: + +- `observed_at`, the provider observation time; +- `started_at`, the current compute incarnation start time; +- cumulative CPU usage in seconds; +- configured CPU capacity in cores, when known; +- current memory usage in bytes; and +- configured memory limit in bytes, when known. + +Measurements are optional. A present pointer with value zero means the provider +observed zero. An absent measurement means it was unavailable and must never be +rendered or aggregated as zero. A whole observation has one of three states: +`observed`, `unsupported`, or `unavailable`. Provider and permission failures are +errors, not ordinary unavailability. + +Docker reports cumulative cgroup CPU time and current cgroup memory usage. CPU and +memory capacity come from the inspected container configuration. Inspect and Stats +are read-only; observation must not renew, restart, create, or stop the container. +The Docker `StartedAt` value defines current compute uptime and resets after a +container restart. + +## Duration boundaries + +These durations answer different questions and must remain separate: + +- allocation age: `runtime_allocations.created_at` through `released_at` or now; +- compute uptime: provider `started_at` through `observed_at`; and +- busy Turn duration: `turns.started_at` through `completed_at` or now. + +This phase supplies compute uptime evidence and retains the existing durable +allocation and Turn timestamps. It does not infer idle time. CPU quietness, +heartbeat age, connection status, and `kept_at` are not authoritative idle state. + +Future automatic suspension requires a separate durable control model, including +an activity revision and timestamps such as `idle_since` and +`shutdown_requested_at`. Metrics, an in-memory cache, or a monitoring backend must +not become the lifecycle authority. + +## First-phase boundary + +The first phase adds no migration, public endpoint, Web view, metrics backend, +token aggregation, Kubernetes/E2B source, or automatic lifecycle action. The +internal source interface is intended to admit those providers without changing +Session attribution or the existing sandbox lifecycle interface. + +The review proposal for later API and Web phases is split into the +[full design](runtime-observability-design.md) and the +[proposed public extension](runtime-observability-api.md). Neither document marks +those later phases as implemented. diff --git a/contracts/agents-api/v1/runtime_observations.go b/contracts/agents-api/v1/runtime_observations.go new file mode 100644 index 000000000..d82d8d514 --- /dev/null +++ b/contracts/agents-api/v1/runtime_observations.go @@ -0,0 +1,46 @@ +package v1 + +type RuntimeObservation struct { + ID string `json:"id" binding:"required" format:"uuid"` + Object string `json:"object" enums:"agent.runtime_observation" binding:"required"` + SessionID string `json:"session_id" binding:"required" format:"uuid"` + EnvironmentID *string `json:"environment_id" extensions:"x-nullable" binding:"required" format:"uuid"` + Mode string `json:"mode" enums:"none,self_hosted,openai_hosted" binding:"required"` + ProviderType *string `json:"provider_type" extensions:"x-nullable" binding:"required" pattern:"^[a-z][a-z0-9_]{0,31}$"` + Instance RuntimeInstance `json:"instance" binding:"required"` + Status string `json:"status" enums:"observed,unsupported,unavailable" binding:"required"` + Reason *string `json:"reason" extensions:"x-nullable" binding:"required" enums:"runtime_mode_not_observable,allocation_pending,runtime_not_running,source_not_configured,sample_timeout,sample_unavailable"` + AllocationCreatedAt *int64 `json:"allocation_created_at" extensions:"x-nullable" binding:"required" minimum:"0"` + ResolvedAt int64 `json:"resolved_at" binding:"required" minimum:"0"` + ObservedAt *int64 `json:"observed_at" extensions:"x-nullable" binding:"required" minimum:"0"` + StartedAt *int64 `json:"started_at" extensions:"x-nullable" binding:"required" minimum:"0"` + CPU *RuntimeCPUObservation `json:"cpu" extensions:"x-nullable" binding:"required"` + Memory *RuntimeMemoryObservation `json:"memory" extensions:"x-nullable" binding:"required"` +} + +type RuntimeInstance struct { + Kind string `json:"kind" enums:"managed_allocation,self_hosted_connection,none" binding:"required"` + AllocationID *string `json:"allocation_id" extensions:"x-nullable" binding:"required" format:"uuid"` + DeviceID *string `json:"device_id" extensions:"x-nullable" binding:"required" format:"uuid"` + ConnectionGeneration *string `json:"connection_generation" extensions:"x-nullable" binding:"required" format:"uuid"` +} + +type RuntimeCPUObservation struct { + UsageSecondsTotal *float64 `json:"usage_seconds_total" extensions:"x-nullable" binding:"required" minimum:"0"` + CapacityCores *float64 `json:"capacity_cores" extensions:"x-nullable" binding:"required" minimum:"5e-324"` + UsageCores *float64 `json:"usage_cores" extensions:"x-nullable" binding:"required" minimum:"0"` + UtilizationRatio *float64 `json:"utilization_ratio" extensions:"x-nullable" binding:"required" minimum:"0"` +} + +type RuntimeMemoryObservation struct { + UsageBytes *uint64 `json:"usage_bytes" extensions:"x-nullable" binding:"required" minimum:"0"` + LimitBytes *uint64 `json:"limit_bytes" extensions:"x-nullable" binding:"required" minimum:"1"` +} + +type RuntimeObservationList struct { + Object string `json:"object" enums:"list" binding:"required"` + Data []RuntimeObservation `json:"data" binding:"required"` + HasMore bool `json:"has_more" binding:"required"` + FirstID *string `json:"first_id" extensions:"x-nullable" binding:"required" format:"uuid"` + LastID *string `json:"last_id" extensions:"x-nullable" binding:"required" format:"uuid"` +} diff --git a/docs/web/README.md b/docs/web/README.md index 3703cddeb..398ccacc1 100644 --- a/docs/web/README.md +++ b/docs/web/README.md @@ -11,7 +11,8 @@ credentials or execution into the browser. ## What you can do -- **Operate from one Dashboard** — see loaded Agents, active Sessions, work that needs +- **Operate from one Dashboard** — see loaded Agents, active Sessions, current Runtime + CPU/memory evidence, compute uptime, reported token coverage, work that needs attention, recent activity, and the two common create flows. - **Build reusable Agents** — start from a blank Agent or a practical template, then configure its model, instructions, text behavior, Functions, and HTTP MCP servers. @@ -29,8 +30,12 @@ credentials or execution into the browser. ### Dashboard Dashboard is the starting point. It summarizes the current Agent and Session results, -highlights Sessions that need attention, and links directly to Agent creation or a new -Session. +loads a complete tenant-scoped Runtime observation snapshot, shows current Docker +resource evidence and coverage without inventing missing values, highlights Sessions +that need attention, and links directly to Agent creation or a new Session. Runtime +health and coverage cards sit above a searchable, filterable, sortable, paginated +semantic table. Historical charts remain absent until an operator configures a +separate history capability. ### Agents @@ -61,7 +66,7 @@ service is not mistaken for a ready model execution path. | Area | User experience | | --- | --- | -| Dashboard | Agent and Session overview, attention queue, recent activity, quick actions | +| Dashboard | Agent and Session overview, current Runtime CPU/memory/uptime and token coverage, attention queue, recent activity, quick actions | | Agents | Create, search, inspect, edit, delete, use templates, and start Sessions | | Sessions | Durable conversation history, Agent filtering, live events, cancellation, retry and continuation | | Trace | Turn history, usage when reported by Core, command output, Function and patch activity | diff --git a/docs/web/README.zh-CN.md b/docs/web/README.zh-CN.md index f444beddb..77b804710 100644 --- a/docs/web/README.zh-CN.md +++ b/docs/web/README.zh-CN.md @@ -10,8 +10,8 @@ Core 部署提供完整的产品界面,同时让凭据和执行能力始终留 ## 可以做什么 -- **通过 Dashboard 统一管理**:查看 Agent、活跃 Session、需要关注的工作、最近活动 - 和常用创建入口。 +- **通过 Dashboard 统一管理**:查看 Agent、活跃 Session、Runtime 当前 CPU/内存 + 证据、计算运行时长、Token 覆盖率、需要关注的工作、最近活动和常用创建入口。 - **创建可复用 Agent**:从空白配置或实用模板开始,设置模型、指令、文本行为、 Function 和 HTTP MCP 服务。 - **运行持久化对话**:创建 Session、发送消息、查看实时事件、重新打开历史工作、 @@ -27,8 +27,11 @@ Core 部署提供完整的产品界面,同时让凭据和执行能力始终留 ### Dashboard -Dashboard 是默认首页,集中展示当前 Agent 和 Session 结果、需要关注的 Session, -并可直接进入创建 Agent 或启动 Session 的流程。 +Dashboard 是默认首页,集中展示当前 Agent 和 Session 结果,加载完整的租户级 Runtime +观测快照,并在不把缺失值伪装成 0 的前提下展示当前 Docker 资源和数据覆盖率。页面也 +展示 Runtime health 与 coverage 卡片,以及支持搜索、状态/模式筛选、排序、分页的语义 +表格;同时展示需要关注的 Session,并可直接进入创建 Agent 或启动 Session 的流程。 +在运维方配置独立历史能力之前,页面不会伪造历史趋势图,也不会引入趋势图组件。 ### Agents @@ -56,7 +59,7 @@ System 展示当前 Core 对 Web 暴露的能力,并区分 API 访问、Vault | 区域 | 用户可以完成的工作 | | --- | --- | -| Dashboard | 查看 Agent/Session 概览、关注队列、最近活动和快捷入口 | +| Dashboard | 查看 Agent/Session 概览、Runtime 当前 CPU/内存/运行时长与 Token 覆盖率、关注队列、最近活动和快捷入口 | | Agents | 创建、搜索、查看、编辑、删除、使用模板并启动 Session | | Sessions | 持久化对话、按 Agent 筛选、实时事件、取消、重试和继续执行 | | Trace | 查看 Turn 历史、Core 报告的 Usage、命令输出、Function 和 Patch 活动 | diff --git a/packages/agents-client/src/client.test.ts b/packages/agents-client/src/client.test.ts index 13c0a237d..facd5e497 100644 --- a/packages/agents-client/src/client.test.ts +++ b/packages/agents-client/src/client.test.ts @@ -103,6 +103,42 @@ function messageItem(overrides: Record = {}): Record = {}): Record { + return { + id: runtimeSessionId, + object: "agent.runtime_observation", + session_id: runtimeSessionId, + environment_id: runtimeEnvironmentId, + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: runtimeAllocationId, + device_id: runtimeDeviceId, + connection_generation: null, + }, + status: "observed", + reason: null, + allocation_created_at: 10, + resolved_at: 30, + observed_at: 20, + started_at: 10, + cpu: { + usage_seconds_total: 0, + capacity_cores: 2, + usage_cores: null, + utilization_ratio: null, + }, + memory: { usage_bytes: 0, limit_bytes: 1024 }, + ...overrides, + }; +} + describe("OpenAIAgentsClient", () => { afterEach(() => vi.unstubAllGlobals()); @@ -2385,4 +2421,121 @@ describe("OpenAIAgentsClient", () => { ).rejects.toThrow("createSession only supports the JSON response"); expect(calls).toHaveLength(0); }); + + it("retrieves a Runtime observation, preserves observed zeroes, and encodes the Session ID", async () => { + const calls: FetchCall[] = []; + const client = new OpenAIAgentsClient({ + baseUrl: "https://core.example/v1", + fetch: recordingFetch(jsonResponse(runtimeObservation()), calls), + }); + + await expect(client.retrieveRuntimeObservation(runtimeSessionId)).resolves.toMatchObject({ + id: runtimeSessionId, + cpu: { usage_seconds_total: 0 }, + memory: { usage_bytes: 0 }, + }); + expect(String(calls[0]?.input)).toBe( + `https://core.example/v1/agents/sessions/${runtimeSessionId}/runtime-observation`, + ); + }); + + it("lists Runtime observations with stable pagination metadata and query serialization", async () => { + const calls: FetchCall[] = []; + const body = { + object: "list", + data: [runtimeObservation()], + has_more: true, + first_id: runtimeSessionId, + last_id: runtimeSessionId, + }; + const client = new OpenAIAgentsClient({ + baseUrl: "https://core.example/v1/", + fetch: recordingFetch(jsonResponse(body), calls), + }); + + await expect(client.listRuntimeObservations({ + after: runtimeSessionId, limit: 1, order: "asc", + })).resolves.toMatchObject(body); + expect(String(calls[0]?.input)).toBe( + `https://core.example/v1/agents/runtime-observations?after=${runtimeSessionId}&limit=1&order=asc`, + ); + }); + + it("accepts an unsupported none-mode Runtime observation with explicit nulls", async () => { + const value = runtimeObservation({ + environment_id: null, + mode: "none", + provider_type: null, + instance: { kind: "none", allocation_id: null, device_id: null, connection_generation: null }, + status: "unsupported", + reason: "runtime_mode_not_observable", + allocation_created_at: null, + observed_at: null, + started_at: null, + cpu: null, + memory: null, + }); + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse(value), []) }); + await expect(client.retrieveRuntimeObservation(runtimeSessionId)).resolves.toMatchObject(value); + }); + + it.each([ + ["unknown field", () => ({ ...runtimeObservation(), provider_native_id: "hidden" })], + ["foreign Session", () => ({ ...runtimeObservation(), session_id: "55555555-5555-4555-8555-555555555555" })], + ["invalid status/reason", () => ({ ...runtimeObservation(), status: "observed", reason: "sample_timeout" })], + ["invalid mode/instance", () => ({ ...runtimeObservation(), mode: "none" })], + ["negative CPU", () => ({ ...runtimeObservation(), cpu: { + usage_seconds_total: -1, capacity_cores: 2, usage_cores: null, utilization_ratio: null, + } })], + ["non-numeric CPU", () => ({ ...runtimeObservation(), cpu: { + usage_seconds_total: "NaN", capacity_cores: 2, usage_cores: null, utilization_ratio: null, + } })], + ["zero CPU capacity", () => ({ ...runtimeObservation(), cpu: { + usage_seconds_total: 1, capacity_cores: 0, usage_cores: null, utilization_ratio: null, + } })], + ["unsafe memory", () => ({ ...runtimeObservation(), memory: { + usage_bytes: Number.MAX_SAFE_INTEGER + 1, limit_bytes: 1024, + } })], + ["zero memory limit", () => ({ ...runtimeObservation(), memory: { + usage_bytes: 1, limit_bytes: 0, + } })], + ])("rejects a Runtime observation with %s", async (_label, build) => { + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse(build()), []) }); + await expect(client.retrieveRuntimeObservation(runtimeSessionId)).rejects.toMatchObject({ + status: 502, + code: "invalid_runtime_observation", + }); + }); + + it.each([ + ["mismatched first_id", { + object: "list", data: [runtimeObservation()], has_more: false, + first_id: runtimeEnvironmentId, last_id: runtimeSessionId, + }], + ["duplicate IDs", { + object: "list", data: [runtimeObservation(), runtimeObservation()], has_more: false, + first_id: runtimeSessionId, last_id: runtimeSessionId, + }], + ["empty continuation", { + object: "list", data: [], has_more: true, first_id: null, last_id: null, + }], + ])("rejects a Runtime observation list with %s", async (_label, body) => { + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse(body), []) }); + await expect(client.listRuntimeObservations()).rejects.toMatchObject({ + status: 502, + code: "invalid_runtime_observation", + }); + }); + + it.each([ + { after: "not-a-uuid" }, + { limit: 0 }, + { limit: 101 }, + { order: "sideways" }, + ])("rejects invalid Runtime observation pagination before fetch", async (options) => { + const calls: FetchCall[] = []; + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse({}), calls) }); + await expect(client.listRuntimeObservations(options as never)).rejects.toThrow(TypeError); + expect(calls).toHaveLength(0); + }); }); diff --git a/packages/agents-client/src/client.ts b/packages/agents-client/src/client.ts index c491cf554..24cc80cac 100644 --- a/packages/agents-client/src/client.ts +++ b/packages/agents-client/src/client.ts @@ -46,6 +46,8 @@ import type { StreamError, UpdateAgentInput, ReplaceVaultCredentialTokenInput, + RuntimeObservation, + RuntimeObservationList, Vault, VaultCredential, VaultCredentialDeleted, @@ -214,6 +216,18 @@ const errorEventFields = new Set(["type", "event_id", "session_id", "error"]); const unsafeUnknownEventFields = new Set([ "session", "turn", "turn_id", "item", "item_id", "output_index", "content_index", "part", "delta", "text", "error", "environment", ]); +const runtimeObservationFields = new Set([ + "id", "object", "session_id", "environment_id", "mode", "provider_type", "instance", "status", "reason", + "allocation_created_at", "resolved_at", "observed_at", "started_at", "cpu", "memory", +]); +const runtimeInstanceFields = new Set(["kind", "allocation_id", "device_id", "connection_generation"]); +const runtimeCPUFields = new Set(["usage_seconds_total", "capacity_cores", "usage_cores", "utilization_ratio"]); +const runtimeMemoryFields = new Set(["usage_bytes", "limit_bytes"]); +const runtimeObservationReasons = new Set([ + "runtime_mode_not_observable", "allocation_pending", "runtime_not_running", + "source_not_configured", "sample_timeout", "sample_unavailable", +]); +const runtimeProviderTypePattern = /^[a-z][a-z0-9_]{0,31}$/; function utf8Length(value: string): number { return new TextEncoder().encode(value).length; } @@ -905,6 +919,164 @@ function projectAgentSession( return session; } +function invalidRuntimeObservation(message = "Agent Core returned an invalid Runtime observation."): never { + throw new AgentCoreError(message, 502, "invalid_runtime_observation"); +} + +function nullableRuntimeNumber(value: unknown): number | null { + if (value === null) return null; + if (typeof value !== "number" || !Number.isFinite(value) || value < 0) { + return invalidRuntimeObservation(); + } + return value; +} + +function nullableRuntimeInteger(value: unknown): number | null { + const projected = nullableRuntimeNumber(value); + if (projected !== null && !Number.isSafeInteger(projected)) return invalidRuntimeObservation(); + return projected; +} + +function projectRuntimeObservation(value: unknown, expectedSessionId?: string): RuntimeObservation { + if (!isRecord(value) || !exactFields(value, runtimeObservationFields)) { + return invalidRuntimeObservation(); + } + const id = canonicalUuid(value.id); + const sessionId = canonicalUuid(value.session_id); + const environmentId = value.environment_id === null ? null : canonicalUuid(value.environment_id); + if ( + id === null || sessionId === null || id !== sessionId || + (expectedSessionId !== undefined && !sameUuid(sessionId, expectedSessionId)) || + value.object !== "agent.runtime_observation" || + (value.mode !== "none" && value.mode !== "self_hosted" && value.mode !== "openai_hosted") || + !(value.provider_type === null || ( + typeof value.provider_type === "string" && runtimeProviderTypePattern.test(value.provider_type) + )) || + !isRecord(value.instance) || !exactFields(value.instance, runtimeInstanceFields) || + (value.status !== "observed" && value.status !== "unsupported" && value.status !== "unavailable") || + !(value.reason === null || ( + typeof value.reason === "string" && runtimeObservationReasons.has(value.reason) + )) || + !isNonnegativeInteger(value.resolved_at) + ) return invalidRuntimeObservation(); + + const allocationId = value.instance.allocation_id === null ? null : canonicalUuid(value.instance.allocation_id); + const deviceId = value.instance.device_id === null ? null : canonicalUuid(value.instance.device_id); + const connectionGeneration = value.instance.connection_generation === null + ? null + : canonicalUuid(value.instance.connection_generation); + if ( + (value.instance.allocation_id !== null && allocationId === null) || + (value.instance.device_id !== null && deviceId === null) || + (value.instance.connection_generation !== null && connectionGeneration === null) + ) return invalidRuntimeObservation(); + + const allocationCreatedAt = nullableRuntimeInteger(value.allocation_created_at); + const observedAt = nullableRuntimeInteger(value.observed_at); + const startedAt = nullableRuntimeInteger(value.started_at); + const isNone = value.mode === "none"; + const isSelfHosted = value.mode === "self_hosted"; + const isManaged = value.mode === "openai_hosted"; + if ( + (isNone && ( + value.instance.kind !== "none" || environmentId !== null || value.provider_type !== null || + allocationId !== null || deviceId !== null || connectionGeneration !== null || allocationCreatedAt !== null + )) || + (isSelfHosted && ( + value.instance.kind !== "self_hosted_connection" || environmentId === null || + allocationId !== null || allocationCreatedAt !== null + )) || + (isManaged && ( + value.instance.kind !== "managed_allocation" || environmentId === null || connectionGeneration !== null || + (allocationId === null && (deviceId !== null || allocationCreatedAt !== null)) + )) + ) return invalidRuntimeObservation(); + + const observed = value.status === "observed"; + if ( + (observed && ( + !isManaged || allocationId === null || value.reason !== null || observedAt === null || + observedAt > value.resolved_at + )) || + (!observed && ( + observedAt !== null || startedAt !== null || value.cpu !== null || value.memory !== null + )) || + (value.status === "unsupported" && ( + (!isNone && !isSelfHosted) || value.reason !== "runtime_mode_not_observable" + )) || + (value.status === "unavailable" && ( + !isManaged || value.reason === null || value.reason === "runtime_mode_not_observable" + )) || + (startedAt !== null && observedAt !== null && startedAt > observedAt) || + (allocationCreatedAt !== null && allocationCreatedAt > value.resolved_at) + ) return invalidRuntimeObservation(); + + let cpu: RuntimeObservation["cpu"] = null; + if (value.cpu !== null) { + if (!observed || !isRecord(value.cpu) || !exactFields(value.cpu, runtimeCPUFields)) { + return invalidRuntimeObservation(); + } + cpu = { + usage_seconds_total: nullableRuntimeNumber(value.cpu.usage_seconds_total), + capacity_cores: nullableRuntimeNumber(value.cpu.capacity_cores), + usage_cores: nullableRuntimeNumber(value.cpu.usage_cores), + utilization_ratio: nullableRuntimeNumber(value.cpu.utilization_ratio), + }; + if ( + Object.values(cpu).every((entry) => entry === null) || + (cpu.capacity_cores !== null && cpu.capacity_cores === 0) + ) return invalidRuntimeObservation(); + } + + let memory: RuntimeObservation["memory"] = null; + if (value.memory !== null) { + if (!observed || !isRecord(value.memory) || !exactFields(value.memory, runtimeMemoryFields)) { + return invalidRuntimeObservation(); + } + memory = { + usage_bytes: nullableRuntimeInteger(value.memory.usage_bytes), + limit_bytes: nullableRuntimeInteger(value.memory.limit_bytes), + }; + if ( + (memory.usage_bytes === null && memory.limit_bytes === null) || + memory.limit_bytes === 0 + ) return invalidRuntimeObservation(); + } + + return { + id, object: "agent.runtime_observation", session_id: sessionId, environment_id: environmentId, + mode: value.mode, provider_type: value.provider_type, instance: { + kind: value.instance.kind as RuntimeObservation["instance"]["kind"], + allocation_id: allocationId, device_id: deviceId, connection_generation: connectionGeneration, + }, + status: value.status, reason: value.reason as RuntimeObservation["reason"], + allocation_created_at: allocationCreatedAt, resolved_at: value.resolved_at, + observed_at: observedAt, started_at: startedAt, cpu, memory, + } as RuntimeObservation; +} + +function projectRuntimeObservationList(value: unknown, options?: PageOptions): RuntimeObservationList { + if ( + !isRecord(value) || !exactFields(value, vaultListFields) || value.object !== "list" || + !Array.isArray(value.data) || typeof value.has_more !== "boolean" + ) return invalidRuntimeObservation("Agent Core returned an invalid Runtime observation list."); + const limit = options?.limit ?? 20; + if ( + !Number.isSafeInteger(limit) || limit < 1 || limit > 100 || + (options?.order !== undefined && options.order !== "asc" && options.order !== "desc") || + value.data.length > limit + ) return invalidRuntimeObservation("Agent Core returned an invalid Runtime observation list."); + const data = value.data.map((entry) => projectRuntimeObservation(entry)); + const firstId = data[0]?.id ?? null; + const lastId = data[data.length - 1]?.id ?? null; + if ( + new Set(data.map((entry) => entry.id)).size !== data.length || + value.first_id !== firstId || value.last_id !== lastId || + (value.has_more && data.length === 0) + ) return invalidRuntimeObservation("Agent Core returned an invalid Runtime observation list."); + return { object: "list", data, has_more: value.has_more, first_id: firstId, last_id: lastId }; +} + function projectStreamError(value: unknown): StreamError { if ( !isRecord(value) || !exactFields(value, streamErrorFields) || @@ -1778,6 +1950,31 @@ export class OpenAIAgentsClient implements AgentCore { return { ...page, data: page.data.map((session) => projectAgentSession(session)) }; } + async listRuntimeObservations(options?: PageOptions): Promise { + if ( + (options?.after !== undefined && canonicalUuid(options.after) === null) || + (options?.limit !== undefined && ( + !Number.isSafeInteger(options.limit) || options.limit < 1 || options.limit > 100 + )) || + (options?.order !== undefined && options.order !== "asc" && options.order !== "desc") + ) throw new TypeError("Runtime observation pagination options are invalid."); + const params = new URLSearchParams(); + addPageOptions(params, options); + const value = await this.request( + withQuery("/agents/runtime-observations", params), + { signal: options?.signal }, + ); + return projectRuntimeObservationList(value, options); + } + + async retrieveRuntimeObservation(sessionId: string, options?: ReadOptions): Promise { + const value = await this.request( + `/agents/sessions/${encodeURIComponent(sessionId)}/runtime-observation`, + { signal: options?.signal }, + ); + return projectRuntimeObservation(value, sessionId); + } + async createSession(input: CreateSessionInput, idempotencyKey = createIdempotencyKey()): Promise { if ((input as { stream?: boolean }).stream === true) { throw new TypeError("createSession only supports the JSON response; connect streamEvents after creation."); diff --git a/packages/agents-client/src/protocol-types.test.ts b/packages/agents-client/src/protocol-types.test.ts index 3c1384129..ddf80f36e 100644 --- a/packages/agents-client/src/protocol-types.test.ts +++ b/packages/agents-client/src/protocol-types.test.ts @@ -29,6 +29,8 @@ import type { OpenAIHostedAgentEnvironmentInput, OpenAIHostedAgentEnvironmentResource, RequiredAction, + RuntimeObservation, + RuntimeUnavailableReason, SelfHostedAgentEnvironment, SavedAgentToolInput, SourceFile, @@ -48,6 +50,25 @@ import type { VaultCredential, } from "./types"; +describe("Runtime Observation discriminated contract", () => { + it("narrows status, reason, mode, instance, and sample presence together", () => { + type Observed = Extract; + type Unavailable = Extract; + type NoneMode = Extract; + type SelfHosted = Extract; + + expectTypeOf().toEqualTypeOf<"openai_hosted">(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf<"none">(); + expectTypeOf().toEqualTypeOf<"self_hosted_connection">(); + }); +}); + describe("Parsar dadf64a7 basic managed Environment profile", () => { it("pins omitted/default, explicit-enabled, and explicit-disabled network input", () => { const inputs = hostedDadf64.inputs as Record; diff --git a/packages/agents-client/src/types.ts b/packages/agents-client/src/types.ts index ca2ab73f7..8d38d2a57 100644 --- a/packages/agents-client/src/types.ts +++ b/packages/agents-client/src/types.ts @@ -715,6 +715,119 @@ export interface CreateSessionStreamOptions extends StreamOptions { onSession: (session: AgentSession) => void; } +export type RuntimeObservationStatus = "observed" | "unsupported" | "unavailable"; +export type RuntimeObservationReason = + | "runtime_mode_not_observable" + | "allocation_pending" + | "runtime_not_running" + | "source_not_configured" + | "sample_timeout" + | "sample_unavailable"; + +export type RuntimeUnavailableReason = Exclude; + +export interface RuntimeCPUObservation { + usage_seconds_total: number | null; + capacity_cores: number | null; + usage_cores: number | null; + utilization_ratio: number | null; +} + +export interface RuntimeMemoryObservation { + usage_bytes: number | null; + limit_bytes: number | null; +} + +interface RuntimeObservationBase { + id: string; + object: "agent.runtime_observation"; + session_id: string; + resolved_at: number; +} + +export interface RuntimeObservedObservation extends RuntimeObservationBase { + environment_id: string; + mode: "openai_hosted"; + provider_type: string | null; + instance: { + kind: "managed_allocation"; + allocation_id: string; + device_id: string | null; + connection_generation: null; + }; + status: "observed"; + reason: null; + allocation_created_at: number | null; + observed_at: number; + started_at: number | null; + cpu: RuntimeCPUObservation | null; + memory: RuntimeMemoryObservation | null; +} + +export interface RuntimeUnavailableObservation extends RuntimeObservationBase { + environment_id: string; + mode: "openai_hosted"; + provider_type: string | null; + instance: { + kind: "managed_allocation"; + allocation_id: string | null; + device_id: string | null; + connection_generation: null; + }; + status: "unavailable"; + reason: RuntimeUnavailableReason; + allocation_created_at: number | null; + observed_at: null; + started_at: null; + cpu: null; + memory: null; +} + +export interface RuntimeNoneObservation extends RuntimeObservationBase { + environment_id: null; + mode: "none"; + provider_type: null; + instance: { kind: "none"; allocation_id: null; device_id: null; connection_generation: null }; + status: "unsupported"; + reason: "runtime_mode_not_observable"; + allocation_created_at: null; + observed_at: null; + started_at: null; + cpu: null; + memory: null; +} + +export interface RuntimeSelfHostedObservation extends RuntimeObservationBase { + environment_id: string; + mode: "self_hosted"; + provider_type: string | null; + instance: { + kind: "self_hosted_connection"; + allocation_id: null; + device_id: string | null; + connection_generation: string | null; + }; + status: "unsupported"; + reason: "runtime_mode_not_observable"; + allocation_created_at: null; + observed_at: null; + started_at: null; + cpu: null; + memory: null; +} + +export type RuntimeObservation = + | RuntimeObservedObservation + | RuntimeUnavailableObservation + | RuntimeNoneObservation + | RuntimeSelfHostedObservation; + +export interface RuntimeObservationList extends ListPage { + object: "list"; + first_id: string | null; + last_id: string | null; +} + export interface AgentCore { listAgents(options?: PageOptions): Promise>; createAgent(input: CreateAgentInput): Promise; @@ -731,6 +844,8 @@ export interface AgentCore { replaceVaultCredentialToken(vaultId: string, credentialId: string, input: ReplaceVaultCredentialTokenInput): Promise; deleteVaultCredential(vaultId: string, credentialId: string): Promise; listSessions(options?: PageOptions & { agentId?: string }): Promise>; + listRuntimeObservations(options?: PageOptions): Promise; + retrieveRuntimeObservation(sessionId: string, options?: ReadOptions): Promise; createSession(input: CreateSessionInput, idempotencyKey?: string): Promise; createSessionStream( input: Omit, diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index dd71da556..d46c38103 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -23,6 +23,9 @@ importers: '@agents-core-web/agents-client': specifier: workspace:* version: link:../../packages/agents-client + '@tanstack/react-table': + specifier: ^8.21.3 + version: 8.21.3(react-dom@19.3.0(react@19.3.0))(react@19.3.0) lucide-react: specifier: ^1.22.0 version: 1.47.0(react@19.3.0) @@ -493,6 +496,17 @@ packages: '@standard-schema/spec@1.1.0': resolution: {integrity: sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==} + '@tanstack/react-table@8.21.3': + resolution: {integrity: sha512-5nNMTSETP4ykGegmVkhjcS8tTLW6Vl4axfEGQN3v0zdHYbK4UfoqfPChclTrJ4EoK9QynqAu9oUf8VEmrpZ5Ww==} + engines: {node: '>=12'} + peerDependencies: + react: '>=16.8' + react-dom: '>=16.8' + + '@tanstack/table-core@8.21.3': + resolution: {integrity: sha512-ldZXEhOBb8Is7xLs01fR3YEc3DERiz5silj8tnGkFZytt1abEvl/GhUmCE0PMLaMPTa3Jk4HbKmRlHmu+gCftg==} + engines: {node: '>=12'} + '@types/chai@5.2.3': resolution: {integrity: sha512-Mw558oeA9fFbv65/y4mHtXDs9bPnFMZAL/jxdPFUpOHHIXX91mcgEHbS5Lahr+pwZFR8A7GQleRWeI6cGFC2UA==} @@ -1763,6 +1777,14 @@ snapshots: '@standard-schema/spec@1.1.0': {} + '@tanstack/react-table@8.21.3(react-dom@19.3.0(react@19.3.0))(react@19.3.0)': + dependencies: + '@tanstack/table-core': 8.21.3 + react: 19.3.0 + react-dom: 19.3.0(react@19.3.0) + + '@tanstack/table-core@8.21.3': {} + '@types/chai@5.2.3': dependencies: '@types/deep-eql': 4.0.2 diff --git a/services/agents-api/cmd/server/main.go b/services/agents-api/cmd/server/main.go index e1fc46947..46603fadc 100644 --- a/services/agents-api/cmd/server/main.go +++ b/services/agents-api/cmd/server/main.go @@ -28,6 +28,7 @@ import ( "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtime" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeenrollment" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" "github.com/jackc/pgx/v5/pgxpool" ) @@ -93,9 +94,24 @@ func run() error { if err := executionStore.EnsureProjectScopes(ready, auth.ProjectScopes()); err != nil { return err } + observationSources := map[string]runtimeobs.Source{} + if managed != nil { + source, ok := managed.Provider.(runtimeobs.Source) + if ok { + observationSources[managed.InstallationID] = source + } + } + resolver, err := runtimeobs.NewResolver(executionStore) + if err != nil { + return err + } + observationService, err := runtimeobs.NewService(resolver, observationSources) + if err != nil { + return err + } var workerDone chan error var worker *execution.Worker - options := []api.Option{api.WithSubagents(executionStore), api.WithSkills(executionStore), api.WithSourceFiles(executionStore), api.WithSessionArtifacts(executionStore)} + options := []api.Option{api.WithSubagents(executionStore), api.WithSkills(executionStore), api.WithSourceFiles(executionStore), api.WithSessionArtifacts(executionStore), api.WithRuntimeObservations(observationService)} var daemonHandler http.Handler var registry *gateway.Registry if wsURL := os.Getenv("AGENTS_API_DAEMON_WS_URL"); wsURL != "" { diff --git a/services/agents-api/internal/api/handler.go b/services/agents-api/internal/api/handler.go index bcbb51fc9..3d923eb6a 100644 --- a/services/agents-api/internal/api/handler.go +++ b/services/agents-api/internal/api/handler.go @@ -35,20 +35,21 @@ type ResourceStore interface { } type Handler struct { - policy execution.Policy - store ResourceStore - auth *Authenticator - harnesses map[string]bool - engine string - inputs InputSubmitter - executorURL string - hostedEnvironments bool - directoryReader EnvironmentDirectoryReader - fileWriter EnvironmentFileWriter - skills SkillStore - sourceFiles SourceFileStore - artifacts SessionArtifactStore - subagents SubagentStore + policy execution.Policy + store ResourceStore + auth *Authenticator + harnesses map[string]bool + engine string + inputs InputSubmitter + executorURL string + hostedEnvironments bool + directoryReader EnvironmentDirectoryReader + fileWriter EnvironmentFileWriter + skills SkillStore + sourceFiles SourceFileStore + artifacts SessionArtifactStore + subagents SubagentStore + runtimeObservations RuntimeObservationService } func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ...Option) (http.Handler, error) { @@ -100,6 +101,8 @@ func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ... r.Post("/agents/sessions", h.createSession) r.Get("/agents/sessions", h.listSessions) r.Get("/agents/sessions/{session_id}", h.getSession) + r.Get("/agents/sessions/{session_id}/runtime-observation", h.getRuntimeObservation) + r.Get("/agents/runtime-observations", h.listRuntimeObservations) r.Post("/agents/sessions/{session_id}", h.updateSession) r.Delete("/agents/sessions/{session_id}", h.deleteSession) r.Post("/agents/sessions/{session_id}/events", h.createEvents) diff --git a/services/agents-api/internal/api/handler_test.go b/services/agents-api/internal/api/handler_test.go index dbf08bcf7..b1c3befe2 100644 --- a/services/agents-api/internal/api/handler_test.go +++ b/services/agents-api/internal/api/handler_test.go @@ -19,8 +19,19 @@ import ( type recordingStore struct { ResourceStore - tenant string - input store.CreateSessionInput + tenant string + input store.CreateSessionInput + sessions []store.Session + nextSessionCursor string + listTenant string + listAfter string + listLimit int + listAscending bool +} + +func (s *recordingStore) ListSessions(_ context.Context, tenant, after string, limit int, ascending bool, _ *string) (store.SessionPage, error) { + s.listTenant, s.listAfter, s.listLimit, s.listAscending = tenant, after, limit, ascending + return store.SessionPage{Sessions: append([]store.Session(nil), s.sessions...), NextCursor: s.nextSessionCursor}, nil } func (s *recordingStore) GetSession(ctx context.Context, tenant, id string) (store.Session, error) { diff --git a/services/agents-api/internal/api/runtime_observations.go b/services/agents-api/internal/api/runtime_observations.go new file mode 100644 index 000000000..54d061a2d --- /dev/null +++ b/services/agents-api/internal/api/runtime_observations.go @@ -0,0 +1,221 @@ +package api + +import ( + "context" + "errors" + "net/http" + "regexp" + "sync" + "time" + + v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/go-chi/chi/v5" +) + +const ( + runtimeObservationConcurrency = 8 + runtimeObservationSourceBudget = 2 * time.Second + runtimeObservationRequestBudget = 10 * time.Second +) + +var runtimeProviderTypePattern = regexp.MustCompile(`^[a-z][a-z0-9_]{0,31}$`) + +type RuntimeObservationService interface { + ObserveSession(context.Context, string, string) (runtimeobs.Observation, error) +} + +func WithRuntimeObservations(service RuntimeObservationService) Option { + return func(h *Handler) { h.runtimeObservations = service } +} + +// @Summary Retrieve a Session Runtime observation +// @Description Core extension returning one tenant-scoped, read-only current Runtime observation. It never provisions, renews, restarts, pauses or stops compute. +// @Tags Runtime observations +// @Produce json +// @Security BearerAuth +// @Param OpenAI-Beta header string true "agents=v1" +// @Param session_id path string true "Session ID" +// @Success 200 {object} v1.RuntimeObservation +// @Failure 400,401,404,500,503 {object} v1.ErrorResponse +// @Router /agents/sessions/{session_id}/runtime-observation [get] +func (h *Handler) getRuntimeObservation(w http.ResponseWriter, r *http.Request) { + if len(r.URL.Query()) != 0 { + writeError(w, http.StatusBadRequest, "unsupported_parameter", "Runtime observation retrieval does not accept query parameters.") + return + } + if h.runtimeObservations == nil { + writeError(w, http.StatusServiceUnavailable, "execution_unavailable", "Runtime observation is not configured on this service.") + return + } + ctx, cancel := context.WithTimeout(r.Context(), runtimeObservationSourceBudget) + defer cancel() + observation, err := h.runtimeObservations.ObserveSession(ctx, tenantID(r), chi.URLParam(r, "session_id")) + if err != nil { + writeStoreError(w, r, err) + return + } + response, err := runtimeObservationResponse(observation) + if err != nil { + writeStoreError(w, r, err) + return + } + writeJSON(w, http.StatusOK, response) +} + +// @Summary List current Runtime observations +// @Description Core extension listing one current Runtime context per tenant-owned Session in Session creation order. Each row has an independent resolved_at and optional provider observed_at; the page is not an atomic telemetry snapshot. +// @Tags Runtime observations +// @Produce json +// @Security BearerAuth +// @Param OpenAI-Beta header string true "agents=v1" +// @Param after query string false "Last observation ID from the previous page" +// @Param limit query int false "Page size" minimum(1) maximum(100) default(20) +// @Param order query string false "Session creation order" Enums(asc,desc) default(desc) +// @Success 200 {object} v1.RuntimeObservationList +// @Failure 400,401,404,500,503 {object} v1.ErrorResponse +// @Router /agents/runtime-observations [get] +func (h *Handler) listRuntimeObservations(w http.ResponseWriter, r *http.Request) { + if h.runtimeObservations == nil { + writeError(w, http.StatusServiceUnavailable, "execution_unavailable", "Runtime observation is not configured on this service.") + return + } + options, ok := readPage(w, r) + if !ok { + return + } + ctx, cancel := context.WithTimeout(r.Context(), runtimeObservationRequestBudget) + defer cancel() + page, err := h.store.ListSessions(ctx, tenantID(r), options.after, options.limit, options.ascending, nil) + if err != nil { + writeStoreError(w, r, err) + return + } + observations := make([]runtimeobs.Observation, len(page.Sessions)) + semaphore := make(chan struct{}, runtimeObservationConcurrency) + work, stop := context.WithCancel(ctx) + defer stop() + var wait sync.WaitGroup + var once sync.Once + var firstErr error + for index, session := range page.Sessions { + wait.Add(1) + go func(index int, sessionID string) { + defer wait.Done() + select { + case semaphore <- struct{}{}: + defer func() { <-semaphore }() + case <-work.Done(): + return + } + sampleCtx, sampleCancel := context.WithTimeout(work, runtimeObservationSourceBudget) + defer sampleCancel() + value, err := h.runtimeObservations.ObserveSession(sampleCtx, tenantID(r), sessionID) + if err != nil { + once.Do(func() { firstErr = err; stop() }) + return + } + observations[index] = value + }(index, session.ID) + } + wait.Wait() + if firstErr != nil { + writeStoreError(w, r, firstErr) + return + } + if err := ctx.Err(); err != nil { + writeError(w, http.StatusServiceUnavailable, "execution_unavailable", "Runtime observation collection exceeded its request budget.") + return + } + response := v1.RuntimeObservationList{Object: "list", Data: make([]v1.RuntimeObservation, 0, len(observations)), HasMore: page.NextCursor != ""} + for _, observation := range observations { + item, err := runtimeObservationResponse(observation) + if err != nil { + writeStoreError(w, r, err) + return + } + response.Data = append(response.Data, item) + } + if len(response.Data) > 0 { + response.FirstID = &response.Data[0].ID + response.LastID = &response.Data[len(response.Data)-1].ID + } + writeJSON(w, http.StatusOK, response) +} + +func runtimeObservationResponse(observation runtimeobs.Observation) (v1.RuntimeObservation, error) { + if observation.Target.SessionID == "" || observation.ResolvedAt.IsZero() || observation.ResolvedAt.Unix() < 0 { + return v1.RuntimeObservation{}, errors.New("invalid Runtime observation identity") + } + if !observation.Target.Instance.AllocationCreatedAt.IsZero() && + (observation.Target.Instance.AllocationCreatedAt.Unix() < 0 || observation.Target.Instance.AllocationCreatedAt.After(observation.ResolvedAt)) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime allocation creation time") + } + if observation.Sample != nil { + if observation.Sample.ObservedAt.IsZero() || observation.Sample.ObservedAt.Unix() < 0 || observation.Sample.ObservedAt.After(observation.ResolvedAt) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime sample time") + } + if observation.Sample.StartedAt != nil && + (observation.Sample.StartedAt.IsZero() || observation.Sample.StartedAt.Unix() < 0 || observation.Sample.StartedAt.After(observation.Sample.ObservedAt)) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime start time") + } + } + result := v1.RuntimeObservation{ + ID: observation.Target.SessionID, Object: "agent.runtime_observation", SessionID: observation.Target.SessionID, + Mode: string(observation.Target.Mode), Status: string(observation.Status), ResolvedAt: observation.ResolvedAt.Unix(), + } + if observation.Target.EnvironmentID != "" { + result.EnvironmentID = &observation.Target.EnvironmentID + } + if observation.ProviderType != "" { + if !runtimeProviderTypePattern.MatchString(observation.ProviderType) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime observation provider type") + } + result.ProviderType = &observation.ProviderType + } + if observation.Reason != "" { + result.Reason = &observation.Reason + } + switch observation.Target.Mode { + case runtimeobs.ModeManaged: + result.Instance.Kind = "managed_allocation" + if observation.Target.Instance.AllocationID != "" { + result.Instance.AllocationID = &observation.Target.Instance.AllocationID + } + if observation.Target.Instance.DeviceID != "" { + result.Instance.DeviceID = &observation.Target.Instance.DeviceID + } + if !observation.Target.Instance.AllocationCreatedAt.IsZero() { + created := observation.Target.Instance.AllocationCreatedAt.Unix() + result.AllocationCreatedAt = &created + } + case runtimeobs.ModeSelfHosted: + result.Instance.Kind = "self_hosted_connection" + if observation.Target.Instance.DeviceID != "" { + result.Instance.DeviceID = &observation.Target.Instance.DeviceID + } + if observation.Target.Instance.ConnectionGeneration != "" { + result.Instance.ConnectionGeneration = &observation.Target.Instance.ConnectionGeneration + } + case runtimeobs.ModeNone: + result.Instance.Kind = "none" + default: + return v1.RuntimeObservation{}, errors.New("invalid Runtime observation mode") + } + if observation.Sample == nil { + return result, nil + } + observedAt := observation.Sample.ObservedAt.Unix() + result.ObservedAt = &observedAt + if observation.Sample.StartedAt != nil { + startedAt := observation.Sample.StartedAt.Unix() + result.StartedAt = &startedAt + } + if observation.Sample.CPUUsageSecondsTotal != nil || observation.Sample.CPUCapacityCores != nil { + result.CPU = &v1.RuntimeCPUObservation{UsageSecondsTotal: observation.Sample.CPUUsageSecondsTotal, CapacityCores: observation.Sample.CPUCapacityCores} + } + if observation.Sample.MemoryUsageBytes != nil || observation.Sample.MemoryLimitBytes != nil { + result.Memory = &v1.RuntimeMemoryObservation{UsageBytes: observation.Sample.MemoryUsageBytes, LimitBytes: observation.Sample.MemoryLimitBytes} + } + return result, nil +} diff --git a/services/agents-api/internal/api/runtime_observations_test.go b/services/agents-api/internal/api/runtime_observations_test.go new file mode 100644 index 000000000..29768be94 --- /dev/null +++ b/services/agents-api/internal/api/runtime_observations_test.go @@ -0,0 +1,224 @@ +package api + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + "time" + + v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/google/uuid" +) + +type runtimeObservationFixture struct { + values map[string]runtimeobs.Observation +} + +func (f runtimeObservationFixture) ObserveSession(_ context.Context, tenant, session string) (runtimeobs.Observation, error) { + value := f.values[session] + value.Target.TenantID = tenant + return value, nil +} + +type runtimeObservationServiceFunc func(context.Context, string, string) (runtimeobs.Observation, error) + +func (f runtimeObservationServiceFunc) ObserveSession(ctx context.Context, tenant, session string) (runtimeobs.Observation, error) { + return f(ctx, tenant, session) +} + +func runtimeObservationRequest(handler http.Handler, path string) *httptest.ResponseRecorder { + request := httptest.NewRequest(http.MethodGet, path, nil) + request.Header.Set("Authorization", "Bearer test-api-key") + request.Header.Set("OpenAI-Beta", "agents=v1") + response := httptest.NewRecorder() + handler.ServeHTTP(response, request) + return response +} + +func TestRuntimeObservationRoutesUseSessionIdentityAndExactNullability(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + sessionID := uuid.NewString() + service := runtimeObservationFixture{values: map[string]runtimeobs.Observation{ + sessionID: {Target: runtimeobs.Target{SessionID: sessionID, Mode: runtimeobs.ModeNone}, Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now}, + }} + handler, saved, _ := testHandler(t, WithRuntimeObservations(service)) + saved.sessions = []store.Session{{ID: sessionID, CreatedAt: now}} + + for _, path := range []string{"/v1/agents/sessions/" + sessionID + "/runtime-observation", "/v1/agents/runtime-observations?limit=1"} { + response := runtimeObservationRequest(handler, path) + if response.Code != http.StatusOK { + t.Fatalf("%s returned %d: %s", path, response.Code, response.Body) + } + if path[len(path)-7:] == "limit=1" { + var page v1.RuntimeObservationList + if json.Unmarshal(response.Body.Bytes(), &page) != nil || len(page.Data) != 1 || page.FirstID == nil || *page.FirstID != sessionID || page.LastID == nil || *page.LastID != sessionID { + t.Fatalf("invalid observation page: %s", response.Body) + } + continue + } + var value v1.RuntimeObservation + if json.Unmarshal(response.Body.Bytes(), &value) != nil || value.ID != sessionID || value.SessionID != sessionID || value.Instance.Kind != "none" || value.EnvironmentID != nil || value.ProviderType != nil || value.ObservedAt != nil || value.CPU != nil || value.Memory != nil || value.Reason == nil || *value.Reason != "runtime_mode_not_observable" { + t.Fatalf("invalid unsupported observation: %s", response.Body) + } + } +} + +func TestRuntimeObservationRoutesRequireConfiguredServiceAndRejectQueries(t *testing.T) { + handler, _, _ := testHandler(t) + missing := runtimeObservationRequest(handler, "/v1/agents/runtime-observations") + if missing.Code != http.StatusServiceUnavailable { + t.Fatalf("unconfigured service returned %d: %s", missing.Code, missing.Body) + } + + sessionID := uuid.NewString() + now := time.Now().UTC() + service := runtimeObservationFixture{values: map[string]runtimeobs.Observation{ + sessionID: {Target: runtimeobs.Target{SessionID: sessionID, Mode: runtimeobs.ModeNone}, Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now}, + }} + handler, _, _ = testHandler(t, WithRuntimeObservations(service)) + invalid := runtimeObservationRequest(handler, "/v1/agents/sessions/"+sessionID+"/runtime-observation?provider=docker") + if invalid.Code != http.StatusBadRequest { + t.Fatalf("unsupported query returned %d: %s", invalid.Code, invalid.Body) + } +} + +func TestRuntimeObservationResponsePreservesObservedZero(t *testing.T) { + zeroCPU := float64(0) + zeroMemory := uint64(0) + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + sessionID, environmentID := uuid.NewString(), uuid.NewString() + value, err := runtimeObservationResponse(runtimeobs.Observation{ + Target: runtimeobs.Target{SessionID: sessionID, EnvironmentID: environmentID, Mode: runtimeobs.ModeManaged, Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), DeviceID: uuid.NewString(), AllocationCreatedAt: now.Add(-time.Hour)}}, + Status: runtimeobs.StatusObserved, ProviderType: "docker", ResolvedAt: now, + Sample: &runtimeobs.Sample{ObservedAt: now, CPUUsageSecondsTotal: &zeroCPU, MemoryUsageBytes: &zeroMemory}, + }) + if err != nil || value.CPU == nil || value.CPU.UsageSecondsTotal == nil || *value.CPU.UsageSecondsTotal != 0 || value.Memory == nil || value.Memory.UsageBytes == nil || *value.Memory.UsageBytes != 0 { + t.Fatalf("observed zero was lost: %+v %v", value, err) + } +} + +func TestRuntimeObservationResponseRejectsTimesOutsidePublicContract(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + preEpoch := time.Unix(-1, 0).UTC() + base := runtimeobs.Observation{ + Target: runtimeobs.Target{ + SessionID: uuid.NewString(), EnvironmentID: uuid.NewString(), Mode: runtimeobs.ModeManaged, + Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), DeviceID: uuid.NewString(), AllocationCreatedAt: now.Add(-time.Hour)}, + }, + Status: runtimeobs.StatusObserved, ResolvedAt: now, + Sample: &runtimeobs.Sample{ObservedAt: now, StartedAt: timePointer(now.Add(-time.Minute))}, + } + for _, mutate := range []func(*runtimeobs.Observation){ + func(value *runtimeobs.Observation) { value.Target.Instance.AllocationCreatedAt = now.Add(time.Second) }, + func(value *runtimeobs.Observation) { value.Target.Instance.AllocationCreatedAt = preEpoch }, + func(value *runtimeobs.Observation) { value.Sample.ObservedAt = preEpoch }, + func(value *runtimeobs.Observation) { value.Sample.StartedAt = &preEpoch }, + } { + observation := base + sample := *base.Sample + observation.Sample = &sample + mutate(&observation) + if _, err := runtimeObservationResponse(observation); err == nil { + t.Fatalf("invalid Runtime time accepted: %+v", observation) + } + } +} + +func timePointer(value time.Time) *time.Time { return &value } + +func TestRuntimeObservationListPreservesStoreOrderAndTenantPagination(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + sessionIDs := []string{uuid.NewString(), uuid.NewString(), uuid.NewString()} + var expectedTenant string + service := runtimeObservationServiceFunc(func(_ context.Context, tenant, session string) (runtimeobs.Observation, error) { + if tenant != expectedTenant { + return runtimeobs.Observation{}, errors.New("unexpected tenant") + } + if session == sessionIDs[0] { + time.Sleep(20 * time.Millisecond) + } + return runtimeobs.Observation{ + Target: runtimeobs.Target{SessionID: session, Mode: runtimeobs.ModeNone}, + Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now, + }, nil + }) + handler, saved, tenant := testHandler(t, WithRuntimeObservations(service)) + expectedTenant = tenant + for _, id := range sessionIDs { + saved.sessions = append(saved.sessions, store.Session{ID: id, CreatedAt: now}) + } + saved.nextSessionCursor = "next" + + response := runtimeObservationRequest(handler, "/v1/agents/runtime-observations?after=cursor&limit=3&order=asc") + if response.Code != http.StatusOK { + t.Fatalf("list returned %d: %s", response.Code, response.Body) + } + var page v1.RuntimeObservationList + if err := json.Unmarshal(response.Body.Bytes(), &page); err != nil { + t.Fatal(err) + } + if len(page.Data) != len(sessionIDs) || !page.HasMore || saved.listTenant != tenant || saved.listAfter != "cursor" || saved.listLimit != 3 || !saved.listAscending { + t.Fatalf("pagination binding was not preserved: page=%+v store=%+v", page, saved) + } + for index, item := range page.Data { + if item.ID != sessionIDs[index] { + t.Fatalf("concurrent collection reordered page: %+v", page.Data) + } + } +} + +func TestRuntimeObservationListBoundsCollectionConcurrency(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + var active, maximum atomic.Int32 + service := runtimeObservationServiceFunc(func(_ context.Context, _, session string) (runtimeobs.Observation, error) { + current := active.Add(1) + defer active.Add(-1) + for current > maximum.Load() && !maximum.CompareAndSwap(maximum.Load(), current) { + } + time.Sleep(15 * time.Millisecond) + return runtimeobs.Observation{ + Target: runtimeobs.Target{SessionID: session, Mode: runtimeobs.ModeNone}, + Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now, + }, nil + }) + handler, saved, _ := testHandler(t, WithRuntimeObservations(service)) + for range 20 { + saved.sessions = append(saved.sessions, store.Session{ID: uuid.NewString(), CreatedAt: now}) + } + response := runtimeObservationRequest(handler, "/v1/agents/runtime-observations?limit=20") + if response.Code != http.StatusOK { + t.Fatalf("list returned %d: %s", response.Code, response.Body) + } + if got := maximum.Load(); got == 0 || got > runtimeObservationConcurrency { + t.Fatalf("collection concurrency = %d, want 1..%d", got, runtimeObservationConcurrency) + } +} + +func TestRuntimeObservationListRejectsWholePageOnIntegrityFailure(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + validID, invalidID := uuid.NewString(), uuid.NewString() + service := runtimeObservationFixture{values: map[string]runtimeobs.Observation{ + validID: { + Target: runtimeobs.Target{SessionID: validID, Mode: runtimeobs.ModeNone}, + Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now, + }, + invalidID: {Target: runtimeobs.Target{Mode: runtimeobs.ModeNone}, Status: runtimeobs.StatusUnsupported, ResolvedAt: now}, + }} + handler, saved, _ := testHandler(t, WithRuntimeObservations(service)) + saved.sessions = []store.Session{{ID: validID, CreatedAt: now}, {ID: invalidID, CreatedAt: now}} + + response := runtimeObservationRequest(handler, "/v1/agents/runtime-observations?limit=2") + if response.Code != http.StatusInternalServerError { + t.Fatalf("integrity failure returned %d: %s", response.Code, response.Body) + } + var envelope v1.ErrorResponse + if err := json.Unmarshal(response.Body.Bytes(), &envelope); err != nil || envelope.Error.Code == "" { + t.Fatalf("integrity failure leaked a partial page: %s", response.Body) + } +} diff --git a/services/agents-api/internal/runtimeobs/identity.go b/services/agents-api/internal/runtimeobs/identity.go new file mode 100644 index 000000000..17e38f7d4 --- /dev/null +++ b/services/agents-api/internal/runtimeobs/identity.go @@ -0,0 +1,23 @@ +package runtimeobs + +import "time" + +// Instance is one provider-owned Runtime incarnation. AllocationID is present +// for managed compute. DeviceID and ConnectionGeneration are reserved for a +// future authenticated self-hosted telemetry source. +type Instance struct { + AllocationID string + ProviderKey string + DeviceID string + ConnectionGeneration string + AllocationState string + AllocationCreatedAt time.Time +} + +// Target binds telemetry to durable Core identity. A Session is not itself a +// process or sandbox, so callers must retain the complete binding. +type Target struct { + TenantID, SessionID, EnvironmentID string + Mode Mode + Instance Instance +} diff --git a/services/agents-api/internal/runtimeobs/resolver.go b/services/agents-api/internal/runtimeobs/resolver.go new file mode 100644 index 000000000..58a5f28df --- /dev/null +++ b/services/agents-api/internal/runtimeobs/resolver.go @@ -0,0 +1,86 @@ +package runtimeobs + +import ( + "context" + "encoding/json" + "errors" + "fmt" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +var ( + ErrUnavailable = errors.New("Runtime observation unavailable") + ErrNotRunning = errors.New("Runtime is not running") +) + +type sessionStore interface { + GetSession(context.Context, string, string) (store.Session, error) + GetRuntimeAllocation(context.Context, string, string) (store.RuntimeAllocation, error) +} + +type Resolver struct{ store sessionStore } + +func NewResolver(s sessionStore) (*Resolver, error) { + if s == nil { + return nil, errors.New("Runtime observation store is required") + } + return &Resolver{store: s}, nil +} + +func (r *Resolver) Resolve(ctx context.Context, tenantID, sessionID string) (Target, error) { + session, err := r.store.GetSession(ctx, tenantID, sessionID) + if err != nil { + return Target{}, fmt.Errorf("resolve Runtime Session: %w", err) + } + var configuration struct { + Environment *struct { + Type string `json:"type"` + } `json:"environment"` + } + if err := json.Unmarshal(session.Configuration, &configuration); err != nil || configuration.Environment == nil { + return Target{}, errors.New("invalid stored Runtime environment configuration") + } + target := Target{TenantID: session.TenantID, SessionID: session.ID, Mode: Mode(configuration.Environment.Type)} + switch target.Mode { + case ModeNone: + if session.Environment != nil { + return Target{}, errors.New("environment:none unexpectedly has a durable Environment") + } + return target, nil + case ModeSelfHosted: + if session.Environment == nil { + return Target{}, errors.New("self-hosted Session is missing its Environment") + } + if session.Environment.TenantID != session.TenantID || session.Environment.SessionID != session.ID { + return Target{}, errors.New("self-hosted Environment does not match resolved ownership") + } + target.EnvironmentID = session.Environment.ID + return target, nil + case ModeManaged: + if session.Environment == nil { + return Target{}, errors.New("managed Session is missing its Environment") + } + if session.Environment.TenantID != session.TenantID || session.Environment.SessionID != session.ID { + return Target{}, errors.New("managed Environment does not match resolved ownership") + } + target.EnvironmentID = session.Environment.ID + allocation, err := r.store.GetRuntimeAllocation(ctx, tenantID, target.EnvironmentID) + if errors.Is(err, store.ErrNotFound) { + return target, ErrUnavailable + } + if err != nil { + return Target{}, fmt.Errorf("resolve Runtime allocation: %w", err) + } + if allocation.TenantID != tenantID || allocation.SessionID != session.ID || allocation.EnvironmentID != target.EnvironmentID { + return Target{}, errors.New("Runtime allocation does not match resolved ownership") + } + target.Instance = Instance{ + AllocationID: allocation.ID, ProviderKey: allocation.ProviderKey, DeviceID: allocation.DeviceID, + AllocationState: allocation.State, AllocationCreatedAt: allocation.CreatedAt, + } + return target, nil + default: + return Target{}, errors.New("invalid stored Runtime environment type") + } +} diff --git a/services/agents-api/internal/runtimeobs/resolver_test.go b/services/agents-api/internal/runtimeobs/resolver_test.go new file mode 100644 index 000000000..7edb4ba48 --- /dev/null +++ b/services/agents-api/internal/runtimeobs/resolver_test.go @@ -0,0 +1,124 @@ +package runtimeobs + +import ( + "context" + "errors" + "testing" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +type resolverStore struct { + session store.Session + allocation store.RuntimeAllocation + allocationErr error +} + +func (s resolverStore) GetSession(context.Context, string, string) (store.Session, error) { + return s.session, nil +} + +func (s resolverStore) GetRuntimeAllocation(context.Context, string, string) (store.RuntimeAllocation, error) { + return s.allocation, s.allocationErr +} + +func TestResolverBindsManagedSessionEnvironmentAndAllocation(t *testing.T) { + r, err := NewResolver(resolverStore{ + session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}}, + allocation: store.RuntimeAllocation{ + ID: "allocation", TenantID: "tenant", SessionID: "session", EnvironmentID: "environment", + ProviderKey: "provider", DeviceID: "device", + }, + }) + if err != nil { + t.Fatal(err) + } + target, err := r.Resolve(t.Context(), "tenant", "session") + if err != nil { + t.Fatal(err) + } + if target.TenantID != "tenant" || target.SessionID != "session" || target.EnvironmentID != "environment" || target.Mode != ModeManaged || target.Instance.AllocationID != "allocation" || target.Instance.ProviderKey != "provider" || target.Instance.DeviceID != "device" { + t.Fatalf("incorrect managed identity binding: %+v", target) + } +} + +func TestResolverKeepsUnsupportedModesDistinct(t *testing.T) { + for _, tc := range []struct { + mode string + environment *store.Environment + }{ + {mode: "none"}, + {mode: "self_hosted", environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}}, + } { + r, err := NewResolver(resolverStore{session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"` + tc.mode + `"}}`), Environment: tc.environment}}) + if err != nil { + t.Fatal(err) + } + target, err := r.Resolve(t.Context(), "tenant", "session") + if err != nil || string(target.Mode) != tc.mode { + t.Fatalf("mode %s was not resolved accurately: %+v %v", tc.mode, target, err) + } + } +} + +func TestResolverReportsManagedAllocationAsUnavailable(t *testing.T) { + r, err := NewResolver(resolverStore{ + session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}}, + allocationErr: store.ErrNotFound, + }) + if err != nil { + t.Fatal(err) + } + target, err := r.Resolve(t.Context(), "tenant", "session") + if !errors.Is(err, ErrUnavailable) || target.EnvironmentID != "environment" || target.Mode != ModeManaged { + t.Fatalf("allocation absence was not preserved: %+v %v", target, err) + } +} + +func TestResolverRejectsMismatchedEnvironmentOwnership(t *testing.T) { + for _, mode := range []string{"self_hosted", "openai_hosted"} { + for _, environment := range []store.Environment{ + {ID: "environment", TenantID: "other", SessionID: "session"}, + {ID: "environment", TenantID: "tenant", SessionID: "other"}, + } { + resolver, err := NewResolver(resolverStore{session: store.Session{ + ID: "session", TenantID: "tenant", + Configuration: []byte(`{"environment":{"type":"` + mode + `"}}`), Environment: &environment, + }}) + if err != nil { + t.Fatal(err) + } + if _, err := resolver.Resolve(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("mismatched %s Environment accepted: %+v", mode, environment) + } + } + } +} + +func TestResolverRejectsMismatchedAllocationOwnership(t *testing.T) { + base := store.RuntimeAllocation{ + ID: "allocation", TenantID: "tenant", SessionID: "session", EnvironmentID: "environment", + ProviderKey: "provider", DeviceID: "device", + } + for _, mutate := range []func(*store.RuntimeAllocation){ + func(value *store.RuntimeAllocation) { value.TenantID = "other" }, + func(value *store.RuntimeAllocation) { value.SessionID = "other" }, + func(value *store.RuntimeAllocation) { value.EnvironmentID = "other" }, + } { + allocation := base + mutate(&allocation) + resolver, err := NewResolver(resolverStore{ + session: store.Session{ + ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), + Environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}, + }, + allocation: allocation, + }) + if err != nil { + t.Fatal(err) + } + if _, err := resolver.Resolve(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("mismatched allocation accepted: %+v", allocation) + } + } +} diff --git a/services/agents-api/internal/runtimeobs/sample.go b/services/agents-api/internal/runtimeobs/sample.go new file mode 100644 index 000000000..5b36e81f1 --- /dev/null +++ b/services/agents-api/internal/runtimeobs/sample.go @@ -0,0 +1,60 @@ +// Package runtimeobs resolves durable Session identity to provider-owned Runtime +// observations. It is telemetry only and never owns Runtime lifecycle decisions. +package runtimeobs + +import ( + "errors" + "math" + "time" +) + +type Mode string + +const ( + ModeNone Mode = "none" + ModeSelfHosted Mode = "self_hosted" + ModeManaged Mode = "openai_hosted" +) + +type Status string + +const ( + StatusObserved Status = "observed" + StatusUnsupported Status = "unsupported" + StatusUnavailable Status = "unavailable" +) + +// Sample contains provider-neutral cumulative counters and current gauges. +// Pointer fields distinguish an observed zero from an unavailable measurement. +type Sample struct { + ObservedAt time.Time + StartedAt *time.Time + + CPUUsageSecondsTotal *float64 + CPUCapacityCores *float64 + MemoryUsageBytes *uint64 + MemoryLimitBytes *uint64 +} + +func (s Sample) validate(now time.Time) error { + if s.ObservedAt.IsZero() || s.ObservedAt.Unix() < 0 || s.ObservedAt.After(now) { + return errors.New("invalid Runtime observation time") + } + if s.StartedAt != nil && (s.StartedAt.IsZero() || s.StartedAt.Unix() < 0 || s.StartedAt.After(s.ObservedAt)) { + return errors.New("invalid Runtime start time") + } + if s.CPUUsageSecondsTotal != nil && (*s.CPUUsageSecondsTotal < 0 || math.IsNaN(*s.CPUUsageSecondsTotal) || math.IsInf(*s.CPUUsageSecondsTotal, 0)) { + return errors.New("invalid Runtime CPU usage") + } + if s.CPUCapacityCores != nil && (*s.CPUCapacityCores <= 0 || math.IsNaN(*s.CPUCapacityCores) || math.IsInf(*s.CPUCapacityCores, 0)) { + return errors.New("invalid Runtime CPU capacity") + } + const maxSafeJSONInteger = uint64(1<<53 - 1) + if s.MemoryUsageBytes != nil && *s.MemoryUsageBytes > maxSafeJSONInteger { + return errors.New("Runtime memory usage exceeds the public JSON integer range") + } + if s.MemoryLimitBytes != nil && (*s.MemoryLimitBytes == 0 || *s.MemoryLimitBytes > maxSafeJSONInteger) { + return errors.New("invalid Runtime memory limit") + } + return nil +} diff --git a/services/agents-api/internal/runtimeobs/service.go b/services/agents-api/internal/runtimeobs/service.go new file mode 100644 index 000000000..e56d17348 --- /dev/null +++ b/services/agents-api/internal/runtimeobs/service.go @@ -0,0 +1,102 @@ +package runtimeobs + +import ( + "context" + "errors" + "fmt" + "time" +) + +type Observation struct { + Target Target + Status Status + Sample *Sample + Reason string + ProviderType string + ResolvedAt time.Time +} + +type Service struct { + resolver TargetResolver + sources map[string]Source + now func() time.Time +} + +func NewService(resolver TargetResolver, sources map[string]Source) (*Service, error) { + if resolver == nil { + return nil, errors.New("Runtime observation resolver is required") + } + copySources := make(map[string]Source, len(sources)) + for key, source := range sources { + if key == "" || source == nil { + return nil, errors.New("invalid Runtime observation source") + } + copySources[key] = source + } + return &Service{resolver: resolver, sources: copySources, now: time.Now}, nil +} + +func (s *Service) ObserveSession(ctx context.Context, tenantID, sessionID string) (Observation, error) { + target, err := s.resolver.Resolve(ctx, tenantID, sessionID) + resolvedAt := s.now() + if errors.Is(err, ErrUnavailable) { + if target.TenantID != tenantID || target.SessionID != sessionID || target.Mode != ModeManaged || target.EnvironmentID == "" { + return Observation{}, errors.New("Runtime observation resolver returned invalid pending allocation identity") + } + return Observation{Target: target, Status: StatusUnavailable, Reason: "allocation_pending", ResolvedAt: resolvedAt}, nil + } + if err != nil { + return Observation{}, err + } + if target.TenantID != tenantID || target.SessionID != sessionID { + return Observation{}, errors.New("Runtime observation resolver returned mismatched ownership") + } + if (target.Mode == ModeNone && target.EnvironmentID != "") || + ((target.Mode == ModeSelfHosted || target.Mode == ModeManaged) && target.EnvironmentID == "") { + return Observation{}, errors.New("Runtime observation resolver returned mismatched Environment identity") + } + if target.Mode == ModeNone || target.Mode == ModeSelfHosted { + return Observation{Target: target, Status: StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: resolvedAt}, nil + } + if target.Mode != ModeManaged || target.Instance.AllocationID == "" || target.Instance.ProviderKey == "" { + return Observation{}, errors.New("invalid managed Runtime observation target") + } + if !target.Instance.AllocationCreatedAt.IsZero() && + (target.Instance.AllocationCreatedAt.Unix() < 0 || target.Instance.AllocationCreatedAt.After(resolvedAt)) { + return Observation{}, errors.New("invalid managed Runtime allocation creation time") + } + switch target.Instance.AllocationState { + case "creating": + return Observation{Target: target, Status: StatusUnavailable, Reason: "allocation_pending", ResolvedAt: resolvedAt}, nil + case "cleanup_pending", "released": + return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_not_running", ResolvedAt: resolvedAt}, nil + case "running": + default: + return Observation{}, errors.New("invalid managed Runtime allocation state") + } + source, ok := s.sources[target.Instance.ProviderKey] + if !ok { + return Observation{Target: target, Status: StatusUnavailable, Reason: "source_not_configured", ResolvedAt: resolvedAt}, nil + } + providerType := "" + if typed, ok := source.(interface{ ObservationProviderType() string }); ok { + providerType = typed.ObservationProviderType() + } + sample, err := source.Observe(ctx, target) + if errors.Is(err, context.DeadlineExceeded) { + return Observation{Target: target, Status: StatusUnavailable, Reason: "sample_timeout", ProviderType: providerType, ResolvedAt: s.now()}, nil + } + if errors.Is(err, ErrNotRunning) { + return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_not_running", ProviderType: providerType, ResolvedAt: s.now()}, nil + } + if errors.Is(err, ErrUnavailable) { + return Observation{Target: target, Status: StatusUnavailable, Reason: "sample_unavailable", ProviderType: providerType, ResolvedAt: s.now()}, nil + } + if err != nil { + return Observation{}, fmt.Errorf("observe Runtime: %w", err) + } + if err := sample.validate(s.now()); err != nil { + return Observation{}, err + } + return Observation{Target: target, Status: StatusObserved, Sample: &sample, ProviderType: providerType, ResolvedAt: s.now()}, nil +} diff --git a/services/agents-api/internal/runtimeobs/service_test.go b/services/agents-api/internal/runtimeobs/service_test.go new file mode 100644 index 000000000..6cb76682f --- /dev/null +++ b/services/agents-api/internal/runtimeobs/service_test.go @@ -0,0 +1,232 @@ +package runtimeobs + +import ( + "context" + "errors" + "math" + "testing" + "time" +) + +type fixedResolver struct { + target Target + err error +} + +func (r fixedResolver) Resolve(_ context.Context, tenant, session string) (Target, error) { + target := r.target + if target.TenantID == "" { + target.TenantID = tenant + } + if target.SessionID == "" { + target.SessionID = session + } + return target, r.err +} + +type fixedSource struct { + sample Sample + err error + calls int +} + +func (s *fixedSource) Observe(context.Context, Target) (Sample, error) { + s.calls++ + return s.sample, s.err +} + +type typedSource struct { + *fixedSource + providerType string +} + +func (s typedSource) ObservationProviderType() string { return s.providerType } + +type blockingSource struct{} + +func (blockingSource) Observe(ctx context.Context, _ Target) (Sample, error) { + <-ctx.Done() + return Sample{}, ctx.Err() +} + +func (blockingSource) ObservationProviderType() string { return "docker" } + +func TestServiceDoesNotCallSourcesForUnsupportedModes(t *testing.T) { + for _, mode := range []Mode{ModeNone, ModeSelfHosted} { + source := &fixedSource{} + target := Target{Mode: mode} + if mode == ModeSelfHosted { + target.EnvironmentID = "environment" + } + service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnsupported || observation.Reason != "runtime_mode_not_observable" || source.calls != 0 { + t.Fatalf("unsupported mode touched a source: %+v %v calls=%d", observation, err, source.calls) + } + } +} + +func TestServicePreservesUnavailableAndObservedZero(t *testing.T) { + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}} + service, err := NewService(fixedResolver{target: target}, nil) + if err != nil { + t.Fatal(err) + } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "source_not_configured" || observation.Sample != nil { + t.Fatalf("missing source was not unavailable: %+v %v", observation, err) + } + + zeroCPU := float64(0) + zeroMemory := uint64(0) + now := time.Date(2026, 9, 22, 1, 0, 0, 0, time.UTC) + source := &fixedSource{sample: Sample{ObservedAt: now, CPUUsageSecondsTotal: &zeroCPU, MemoryUsageBytes: &zeroMemory}} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + service.now = func() time.Time { return now } + observation, err = service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusObserved || observation.Sample == nil || observation.Sample.CPUUsageSecondsTotal == nil || observation.Sample.MemoryUsageBytes == nil { + t.Fatalf("observed zero was lost: %+v %v", observation, err) + } +} + +func TestServiceMapsOnlyDeclaredUnavailability(t *testing.T) { + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}} + for _, tc := range []struct { + err error + wantReason string + wantError bool + }{ + {err: ErrUnavailable, wantReason: "sample_unavailable"}, + {err: ErrNotRunning, wantReason: "runtime_not_running"}, + {err: context.DeadlineExceeded, wantReason: "sample_timeout"}, + {err: errors.New("Docker permission denied"), wantError: true}, + } { + service, err := NewService(fixedResolver{target: target}, map[string]Source{ + "provider": typedSource{fixedSource: &fixedSource{err: tc.err}, providerType: "docker"}, + }) + if err != nil { + t.Fatal(err) + } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if (err != nil) != tc.wantError { + t.Fatalf("wrong error classification: %+v %v", observation, err) + } + if !tc.wantError && (observation.Status != StatusUnavailable || observation.Reason != tc.wantReason || observation.ProviderType != "docker") { + t.Fatalf("declared unavailability was not mapped: %+v", observation) + } + } +} + +func TestServiceMapsAnActualSourceDeadlineWithoutLeakingIt(t *testing.T) { + target := Target{ + EnvironmentID: "environment", Mode: ModeManaged, + Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}, + } + service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": blockingSource{}}) + if err != nil { + t.Fatal(err) + } + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Millisecond) + defer cancel() + observation, err := service.ObserveSession(ctx, "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "sample_timeout" || observation.ProviderType != "docker" { + t.Fatalf("source deadline was not safely classified: %+v %v", observation, err) + } +} + +func TestServiceClassifiesResolverAndTerminalAllocationUnavailability(t *testing.T) { + now := time.Date(2026, 9, 22, 1, 0, 0, 0, time.UTC) + service, err := NewService(fixedResolver{target: Target{SessionID: "session", EnvironmentID: "environment", Mode: ModeManaged}, err: ErrUnavailable}, nil) + if err != nil { + t.Fatal(err) + } + service.now = func() time.Time { return now } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "allocation_pending" || !observation.ResolvedAt.Equal(now) { + t.Fatalf("pending allocation was not classified: %+v %v", observation, err) + } + + source := &fixedSource{} + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "creating"}} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + observation, err = service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "allocation_pending" || source.calls != 0 { + t.Fatalf("creating allocation reached its provider: %+v %v calls=%d", observation, err, source.calls) + } + + target = Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "released"}} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": &fixedSource{}}) + if err != nil { + t.Fatal(err) + } + observation, err = service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "runtime_not_running" { + t.Fatalf("released allocation was not classified: %+v %v", observation, err) + } + + source = &fixedSource{} + target = Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{ + AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running", AllocationCreatedAt: time.Now().Add(time.Hour), + }} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + if _, err := service.ObserveSession(t.Context(), "tenant", "session"); err == nil || source.calls != 0 { + t.Fatalf("future allocation creation reached its provider: %v calls=%d", err, source.calls) + } +} + +func TestServiceRejectsUnsafeProviderSamples(t *testing.T) { + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}} + now := time.Date(2026, 9, 22, 1, 0, 0, 0, time.UTC) + preEpoch := time.Unix(-1, 0).UTC() + tooLarge := uint64(1 << 53) + for _, sample := range []Sample{ + {ObservedAt: preEpoch}, + {ObservedAt: now, StartedAt: &preEpoch}, + {ObservedAt: now, CPUUsageSecondsTotal: float64Pointer(-1)}, + {ObservedAt: now, CPUUsageSecondsTotal: float64Pointer(math.NaN())}, + {ObservedAt: now, CPUUsageSecondsTotal: float64Pointer(math.Inf(1))}, + {ObservedAt: now, CPUCapacityCores: float64Pointer(0)}, + {ObservedAt: now, MemoryUsageBytes: &tooLarge}, + {ObservedAt: now, MemoryLimitBytes: &tooLarge}, + } { + service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": &fixedSource{sample: sample}}) + if err != nil { + t.Fatal(err) + } + service.now = func() time.Time { return now } + if _, err := service.ObserveSession(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("unsafe sample accepted: %+v", sample) + } + } +} + +func TestServiceRejectsMismatchedResolvedOwnership(t *testing.T) { + for _, target := range []Target{ + {TenantID: "other", SessionID: "session", Mode: ModeNone}, + {TenantID: "tenant", SessionID: "other", Mode: ModeNone}, + {TenantID: "tenant", SessionID: "session", EnvironmentID: "unexpected", Mode: ModeNone}, + {TenantID: "tenant", SessionID: "session", Mode: ModeSelfHosted}, + } { + service, err := NewService(fixedResolver{target: target}, nil) + if err != nil { + t.Fatal(err) + } + if _, err := service.ObserveSession(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("mismatched ownership accepted: %+v", target) + } + } +} + +func float64Pointer(value float64) *float64 { return &value } diff --git a/services/agents-api/internal/runtimeobs/source.go b/services/agents-api/internal/runtimeobs/source.go new file mode 100644 index 000000000..4044456ab --- /dev/null +++ b/services/agents-api/internal/runtimeobs/source.go @@ -0,0 +1,13 @@ +package runtimeobs + +import "context" + +// Source reads one provider-owned Runtime instance. Implementations must verify +// ownership before returning data and must not renew, restart, or stop compute. +type Source interface { + Observe(context.Context, Target) (Sample, error) +} + +type TargetResolver interface { + Resolve(context.Context, string, string) (Target, error) +} diff --git a/services/agents-api/internal/sandbox/docker/provider_test.go b/services/agents-api/internal/sandbox/docker/provider_test.go index c99ef6387..7216208f9 100644 --- a/services/agents-api/internal/sandbox/docker/provider_test.go +++ b/services/agents-api/internal/sandbox/docker/provider_test.go @@ -12,6 +12,7 @@ import ( "testing" "time" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/sandbox" "github.com/containerd/errdefs" "github.com/google/uuid" @@ -66,7 +67,8 @@ func TestDockerProviderLifecycle(t *testing.T) { t.Fatal(e) } defer c.Close() - p, e := New(c, Config{InstallationID: uuid.NewString(), Image: image, Network: "bridge", Seccomp: string(seccomp)}) + installationID := uuid.NewString() + p, e := New(c, Config{InstallationID: installationID, Image: image, Network: "bridge", Seccomp: string(seccomp)}) if e != nil { t.Fatal(e) } @@ -111,6 +113,13 @@ func TestDockerProviderLifecycle(t *testing.T) { if info.State != "running" || info.ProviderID == "" { t.Fatalf("bad compute observation: %+v", info) } + resources, e := p.Observe(ctx, runtimeobs.Target{ + TenantID: b.TenantID, SessionID: b.SessionID, EnvironmentID: b.EnvironmentID, Mode: runtimeobs.ModeManaged, + Instance: runtimeobs.Instance{AllocationID: b.AllocationID, ProviderKey: installationID, DeviceID: b.DeviceID}, + }) + if e != nil || resources.StartedAt == nil || resources.CPUUsageSecondsTotal == nil || resources.MemoryUsageBytes == nil || resources.CPUCapacityCores == nil || resources.MemoryLimitBytes == nil { + t.Fatalf("bad resource observation: %+v %v", resources, e) + } inspected, e := p.inspect(ctx, b.Reference) if e != nil { t.Fatal(e) diff --git a/services/agents-api/internal/sandbox/docker/resources.go b/services/agents-api/internal/sandbox/docker/resources.go new file mode 100644 index 000000000..f0f057de0 --- /dev/null +++ b/services/agents-api/internal/sandbox/docker/resources.go @@ -0,0 +1,93 @@ +package docker + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/sandbox" + "github.com/moby/moby/api/types/container" + "github.com/moby/moby/client" +) + +var _ runtimeobs.Source = (*Provider)(nil) + +func (*Provider) ObservationProviderType() string { return "docker" } + +// Observe is read-only. Inspect verifies allocation ownership before Docker +// statistics are requested; it never renews or changes the container. +func (p *Provider) Observe(ctx context.Context, target runtimeobs.Target) (runtimeobs.Sample, error) { + if target.Mode != runtimeobs.ModeManaged || target.Instance.AllocationID == "" { + return runtimeobs.Sample{}, sandbox.ErrInvalid + } + if target.Instance.ProviderKey != p.config.InstallationID { + return runtimeobs.Sample{}, sandbox.ErrOwnership + } + reference := sandbox.Reference{TenantID: target.TenantID, EnvironmentID: target.EnvironmentID, AllocationID: target.Instance.AllocationID} + inspected, err := p.inspect(ctx, reference) + if errors.Is(err, sandbox.ErrNotFound) { + return runtimeobs.Sample{}, runtimeobs.ErrNotRunning + } + if err != nil { + return runtimeobs.Sample{}, err + } + if inspected.Container.State == nil || !inspected.Container.State.Running { + return runtimeobs.Sample{}, runtimeobs.ErrNotRunning + } + result, err := p.client.ContainerStats(ctx, inspected.Container.ID, client.ContainerStatsOptions{Stream: false, IncludePreviousSample: false}) + if err != nil { + return runtimeobs.Sample{}, fmt.Errorf("read Docker Runtime statistics: %w", err) + } + defer result.Body.Close() + var stats dockerStatsResponse + if err := json.NewDecoder(result.Body).Decode(&stats); err != nil { + return runtimeobs.Sample{}, fmt.Errorf("decode Docker Runtime statistics: %w", err) + } + return sampleFromDocker(inspected.Container, stats) +} + +type dockerStatsResponse struct { + Read time.Time `json:"read"` + CPUStats *struct { + CPUUsage *struct { + TotalUsage *uint64 `json:"total_usage"` + } `json:"cpu_usage"` + } `json:"cpu_stats"` + MemoryStats *struct { + Usage *uint64 `json:"usage"` + } `json:"memory_stats"` +} + +func sampleFromDocker(inspected container.InspectResponse, stats dockerStatsResponse) (runtimeobs.Sample, error) { + if inspected.State == nil || inspected.HostConfig == nil || !inspected.State.Running || stats.Read.IsZero() { + return runtimeobs.Sample{}, errors.New("incomplete Docker Runtime observation") + } + startedAt, err := time.Parse(time.RFC3339Nano, inspected.State.StartedAt) + if err != nil || startedAt.IsZero() || startedAt.After(stats.Read) { + return runtimeobs.Sample{}, errors.New("invalid Docker Runtime start time") + } + sample := runtimeobs.Sample{ + ObservedAt: stats.Read, + StartedAt: &startedAt, + } + if stats.CPUStats != nil && stats.CPUStats.CPUUsage != nil && stats.CPUStats.CPUUsage.TotalUsage != nil { + cpuUsage := float64(*stats.CPUStats.CPUUsage.TotalUsage) / float64(time.Second) + sample.CPUUsageSecondsTotal = &cpuUsage + } + if stats.MemoryStats != nil && stats.MemoryStats.Usage != nil { + memoryUsage := *stats.MemoryStats.Usage + sample.MemoryUsageBytes = &memoryUsage + } + if inspected.HostConfig.NanoCPUs > 0 { + capacity := float64(inspected.HostConfig.NanoCPUs) / 1_000_000_000 + sample.CPUCapacityCores = &capacity + } + if inspected.HostConfig.Memory > 0 { + limit := uint64(inspected.HostConfig.Memory) + sample.MemoryLimitBytes = &limit + } + return sample, nil +} diff --git a/services/agents-api/internal/sandbox/docker/resources_test.go b/services/agents-api/internal/sandbox/docker/resources_test.go new file mode 100644 index 000000000..8a21dbff6 --- /dev/null +++ b/services/agents-api/internal/sandbox/docker/resources_test.go @@ -0,0 +1,166 @@ +package docker + +import ( + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/google/uuid" + "github.com/moby/moby/api/types/container" + "github.com/moby/moby/client" +) + +func TestObserveVerifiesOwnershipThenReadsOneShotStats(t *testing.T) { + installationID := uuid.NewString() + target := runtimeobs.Target{ + TenantID: uuid.NewString(), SessionID: uuid.NewString(), EnvironmentID: uuid.NewString(), Mode: runtimeobs.ModeManaged, + Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), ProviderKey: installationID, DeviceID: uuid.NewString()}, + } + observed := time.Now().UTC().Truncate(time.Microsecond) + started := observed.Add(-time.Minute) + statsRead := false + omitMeasurements := false + running := true + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch { + case r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/json"): + _ = json.NewEncoder(w).Encode(map[string]any{ + "Id": "container-id", + "State": map[string]any{"Status": "running", "Running": running, "StartedAt": started.Format(time.RFC3339Nano)}, + "HostConfig": map[string]any{"NanoCpus": 2_000_000_000, "Memory": 2048}, + "Config": map[string]any{"Labels": map[string]string{ + labelPrefix + "installation": installationID, + labelPrefix + "tenant": target.TenantID, + labelPrefix + "environment": target.EnvironmentID, + labelPrefix + "allocation": target.Instance.AllocationID, + }}, + }) + case r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/containers/container-id/stats"): + if r.URL.Query().Get("stream") != "false" || r.URL.Query().Get("one-shot") != "true" { + t.Errorf("stats request was not one-shot: %s", r.URL.RawQuery) + } + statsRead = true + if omitMeasurements { + _ = json.NewEncoder(w).Encode(map[string]any{"read": observed}) + return + } + _ = json.NewEncoder(w).Encode(container.StatsResponse{ + Read: observed, + CPUStats: container.CPUStats{CPUUsage: container.CPUUsage{TotalUsage: 1_500_000_000}}, + MemoryStats: container.MemoryStats{Usage: 1024}, + }) + default: + http.NotFound(w, r) + } + })) + defer server.Close() + c, err := client.New(client.WithHost(server.URL)) + if err != nil { + t.Fatal(err) + } + defer c.Close() + p, err := New(c, Config{InstallationID: installationID, Image: "fixture@sha256:" + strings.Repeat("a", 64), Network: "bridge", Seccomp: `{}`}) + if err != nil { + t.Fatal(err) + } + sample, err := p.Observe(t.Context(), target) + if err != nil || !statsRead || sample.CPUUsageSecondsTotal == nil || *sample.CPUUsageSecondsTotal != 1.5 || sample.MemoryUsageBytes == nil || *sample.MemoryUsageBytes != 1024 { + t.Fatalf("bad one-shot observation: %+v %v stats=%v", sample, err, statsRead) + } + omitMeasurements = true + missing, err := p.Observe(t.Context(), target) + if err != nil || missing.CPUUsageSecondsTotal != nil || missing.MemoryUsageBytes != nil || missing.CPUCapacityCores == nil || missing.MemoryLimitBytes == nil { + t.Fatalf("missing Docker measurements became zero: %+v %v", missing, err) + } + running = false + statsRead = false + if _, err := p.Observe(t.Context(), target); !errors.Is(err, runtimeobs.ErrNotRunning) || statsRead { + t.Fatalf("stopped Runtime was not classified before stats: %v stats=%v", err, statsRead) + } + running = true + foreign := target + foreign.Instance.ProviderKey = uuid.NewString() + statsRead = false + if _, err := p.Observe(t.Context(), foreign); err == nil || statsRead { + t.Fatal("foreign provider identity reached Docker stats") + } +} + +func TestSampleFromDockerPreservesObservedZeroAndConfiguredCapacity(t *testing.T) { + observed := time.Date(2026, 9, 22, 1, 2, 3, 0, time.UTC) + started := observed.Add(-5 * time.Minute) + zero := uint64(0) + sample, err := sampleFromDocker(container.InspectResponse{ + State: &container.State{Running: true, StartedAt: started.Format(time.RFC3339Nano)}, + HostConfig: &container.HostConfig{Resources: container.Resources{NanoCPUs: 2_000_000_000, Memory: 2 * 1024 * 1024 * 1024}}, + }, testDockerStats(observed, &zero, &zero)) + if err != nil { + t.Fatal(err) + } + if sample.ObservedAt != observed || sample.StartedAt == nil || !sample.StartedAt.Equal(started) { + t.Fatalf("lost Docker observation time: %+v", sample) + } + if sample.CPUUsageSecondsTotal == nil || *sample.CPUUsageSecondsTotal != 0 || sample.MemoryUsageBytes == nil || *sample.MemoryUsageBytes != 0 { + t.Fatalf("observed zero became unavailable: %+v", sample) + } + if sample.CPUCapacityCores == nil || *sample.CPUCapacityCores != 2 || sample.MemoryLimitBytes == nil || *sample.MemoryLimitBytes != 2*1024*1024*1024 { + t.Fatalf("lost configured capacity: %+v", sample) + } +} + +func TestSampleFromDockerNormalizesCumulativeCPU(t *testing.T) { + observed := time.Now().UTC() + started := observed.Add(-time.Hour) + cpu, memory := uint64(2_500_000_000), uint64(4096) + sample, err := sampleFromDocker(container.InspectResponse{ + State: &container.State{Running: true, StartedAt: started.Format(time.RFC3339Nano)}, + HostConfig: &container.HostConfig{}, + }, testDockerStats(observed, &cpu, &memory)) + if err != nil { + t.Fatal(err) + } + if sample.CPUUsageSecondsTotal == nil || *sample.CPUUsageSecondsTotal != 2.5 || sample.MemoryUsageBytes == nil || *sample.MemoryUsageBytes != 4096 { + t.Fatalf("bad Docker normalization: %+v", sample) + } + if sample.CPUCapacityCores != nil || sample.MemoryLimitBytes != nil { + t.Fatalf("invented unconfigured capacity: %+v", sample) + } +} + +func TestSampleFromDockerRejectsIncompleteState(t *testing.T) { + observed := time.Now().UTC() + for _, inspected := range []container.InspectResponse{ + {}, + {State: &container.State{Running: false}, HostConfig: &container.HostConfig{}}, + {State: &container.State{Running: true, StartedAt: "invalid"}, HostConfig: &container.HostConfig{}}, + } { + if _, err := sampleFromDocker(inspected, dockerStatsResponse{Read: observed}); err == nil { + t.Fatal("accepted incomplete Docker state") + } + } +} + +func testDockerStats(observed time.Time, cpu, memory *uint64) dockerStatsResponse { + stats := dockerStatsResponse{Read: observed} + if cpu != nil { + stats.CPUStats = &struct { + CPUUsage *struct { + TotalUsage *uint64 `json:"total_usage"` + } `json:"cpu_usage"` + }{CPUUsage: &struct { + TotalUsage *uint64 `json:"total_usage"` + }{TotalUsage: cpu}} + } + if memory != nil { + stats.MemoryStats = &struct { + Usage *uint64 `json:"usage"` + }{Usage: memory} + } + return stats +}