From b0085d6562d01c2ce615e9728200793d67bb09e4 Mon Sep 17 00:00:00 2001 From: sam Date: Tue, 22 Sep 2026 14:47:30 +0800 Subject: [PATCH 1/6] Add Docker runtime observability foundation --- CONTRIBUTING.md | 8 + contracts/agents-api/runtime-observability.md | 73 ++++++++ .../internal/runtimeobs/identity.go | 19 +++ .../internal/runtimeobs/resolver.go | 71 ++++++++ .../internal/runtimeobs/resolver_test.go | 73 ++++++++ .../agents-api/internal/runtimeobs/sample.go | 55 ++++++ .../agents-api/internal/runtimeobs/service.go | 66 ++++++++ .../internal/runtimeobs/service_test.go | 91 ++++++++++ .../agents-api/internal/runtimeobs/source.go | 13 ++ .../internal/sandbox/docker/provider_test.go | 11 +- .../internal/sandbox/docker/resources.go | 91 ++++++++++ .../internal/sandbox/docker/resources_test.go | 158 ++++++++++++++++++ 12 files changed, 728 insertions(+), 1 deletion(-) create mode 100644 contracts/agents-api/runtime-observability.md create mode 100644 services/agents-api/internal/runtimeobs/identity.go create mode 100644 services/agents-api/internal/runtimeobs/resolver.go create mode 100644 services/agents-api/internal/runtimeobs/resolver_test.go create mode 100644 services/agents-api/internal/runtimeobs/sample.go create mode 100644 services/agents-api/internal/runtimeobs/service.go create mode 100644 services/agents-api/internal/runtimeobs/service_test.go create mode 100644 services/agents-api/internal/runtimeobs/source.go create mode 100644 services/agents-api/internal/sandbox/docker/resources.go create mode 100644 services/agents-api/internal/sandbox/docker/resources_test.go diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 58b311ac6..c979f5e62 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -197,6 +197,14 @@ starts the Runtime and its daemon authenticates and initiates the Core connectio Core verifies principal ownership and the exact Environment binding. These are management responsibilities, not separate execution architectures. +Runtime telemetry uses a separate read-only service boundary documented in +[`contracts/agents-api/runtime-observability.md`](contracts/agents-api/runtime-observability.md). +Resolve durable Session, Environment and Runtime-instance identity before selecting +a provider source. Observation never extends a lease or changes compute lifecycle. +Keep observed zero, unavailable data and unsupported Runtime modes distinct. Metrics +may inform operators, but automatic suspension requires durable Core-owned activity +state and must not use a monitoring backend as lifecycle authority. + In V1, our daemon fills the user-side executor role. Users deploy daemon, the selected harness, local tools and workspace together. Do not require Codex `exec-server`, a service-side harness, registry/Noise transport or remote tool diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md new file mode 100644 index 000000000..61d64ccc5 --- /dev/null +++ b/contracts/agents-api/runtime-observability.md @@ -0,0 +1,73 @@ +# Runtime observability contract + +This document defines the internal Runtime observation boundary. It does not add +an Agents API resource or change the pinned public protocol. + +## Ownership and identity + +Runtime telemetry is attributed to durable Core identity before it is sampled: + +```text +managed: tenant_id -> session_id -> environment_id -> runtime_allocation_id +self-hosted: tenant_id -> session_id -> environment_id -> device_id + connection_generation +none: tenant_id -> session_id (no Session-owned Runtime instance) +``` + +The managed allocation's persisted `provider_key` selects exactly one configured +observation source. A provider must independently verify the allocation labels or +equivalent ownership data. A Session, daemon connection, process, container, and +native harness Session are different identities and must not be substituted for +one another. + +The first implementation supports managed Docker allocations. `self_hosted` and +`none` are recognized but explicitly unsupported. A future self-hosted source must +use authenticated daemon telemetry fenced by the current connection generation. +Core must not attribute shared host statistics to an `environment:none` Session. + +## Sample semantics + +One sample contains: + +- `observed_at`, the provider observation time; +- `started_at`, the current compute incarnation start time; +- cumulative CPU usage in seconds; +- configured CPU capacity in cores, when known; +- current memory usage in bytes; and +- configured memory limit in bytes, when known. + +Measurements are optional. A present pointer with value zero means the provider +observed zero. An absent measurement means it was unavailable and must never be +rendered or aggregated as zero. A whole observation has one of three states: +`observed`, `unsupported`, or `unavailable`. Provider and permission failures are +errors, not ordinary unavailability. + +Docker reports cumulative cgroup CPU time and current cgroup memory usage. CPU and +memory capacity come from the inspected container configuration. Inspect and Stats +are read-only; observation must not renew, restart, create, or stop the container. +The Docker `StartedAt` value defines current compute uptime and resets after a +container restart. + +## Duration boundaries + +These durations answer different questions and must remain separate: + +- allocation age: `runtime_allocations.created_at` through `released_at` or now; +- compute uptime: provider `started_at` through `observed_at`; and +- busy Turn duration: `turns.started_at` through `completed_at` or now. + +This phase supplies compute uptime evidence and retains the existing durable +allocation and Turn timestamps. It does not infer idle time. CPU quietness, +heartbeat age, connection status, and `kept_at` are not authoritative idle state. + +Future automatic suspension requires a separate durable control model, including +an activity revision and timestamps such as `idle_since` and +`shutdown_requested_at`. Metrics, an in-memory cache, or a monitoring backend must +not become the lifecycle authority. + +## First-phase boundary + +The first phase adds no migration, public endpoint, Web view, metrics backend, +token aggregation, Kubernetes/E2B source, or automatic lifecycle action. The +internal source interface is intended to admit those providers without changing +Session attribution or the existing sandbox lifecycle interface. + diff --git a/services/agents-api/internal/runtimeobs/identity.go b/services/agents-api/internal/runtimeobs/identity.go new file mode 100644 index 000000000..7e33f4bb9 --- /dev/null +++ b/services/agents-api/internal/runtimeobs/identity.go @@ -0,0 +1,19 @@ +package runtimeobs + +// Instance is one provider-owned Runtime incarnation. AllocationID is present +// for managed compute. DeviceID and ConnectionGeneration are reserved for a +// future authenticated self-hosted telemetry source. +type Instance struct { + AllocationID string + ProviderKey string + DeviceID string + ConnectionGeneration string +} + +// Target binds telemetry to durable Core identity. A Session is not itself a +// process or sandbox, so callers must retain the complete binding. +type Target struct { + TenantID, SessionID, EnvironmentID string + Mode Mode + Instance Instance +} diff --git a/services/agents-api/internal/runtimeobs/resolver.go b/services/agents-api/internal/runtimeobs/resolver.go new file mode 100644 index 000000000..7b1047c0c --- /dev/null +++ b/services/agents-api/internal/runtimeobs/resolver.go @@ -0,0 +1,71 @@ +package runtimeobs + +import ( + "context" + "encoding/json" + "errors" + "fmt" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +var ErrUnavailable = errors.New("Runtime observation unavailable") + +type sessionStore interface { + GetSession(context.Context, string, string) (store.Session, error) + GetRuntimeAllocation(context.Context, string, string) (store.RuntimeAllocation, error) +} + +type Resolver struct{ store sessionStore } + +func NewResolver(s sessionStore) (*Resolver, error) { + if s == nil { + return nil, errors.New("Runtime observation store is required") + } + return &Resolver{store: s}, nil +} + +func (r *Resolver) Resolve(ctx context.Context, tenantID, sessionID string) (Target, error) { + session, err := r.store.GetSession(ctx, tenantID, sessionID) + if err != nil { + return Target{}, fmt.Errorf("resolve Runtime Session: %w", err) + } + var configuration struct { + Environment *struct { + Type string `json:"type"` + } `json:"environment"` + } + if err := json.Unmarshal(session.Configuration, &configuration); err != nil || configuration.Environment == nil { + return Target{}, errors.New("invalid stored Runtime environment configuration") + } + target := Target{TenantID: session.TenantID, SessionID: session.ID, Mode: Mode(configuration.Environment.Type)} + switch target.Mode { + case ModeNone: + if session.Environment != nil { + return Target{}, errors.New("environment:none unexpectedly has a durable Environment") + } + return target, nil + case ModeSelfHosted: + if session.Environment == nil { + return Target{}, errors.New("self-hosted Session is missing its Environment") + } + target.EnvironmentID = session.Environment.ID + return target, nil + case ModeManaged: + if session.Environment == nil { + return Target{}, errors.New("managed Session is missing its Environment") + } + target.EnvironmentID = session.Environment.ID + allocation, err := r.store.GetRuntimeAllocation(ctx, tenantID, target.EnvironmentID) + if errors.Is(err, store.ErrNotFound) { + return target, ErrUnavailable + } + if err != nil { + return Target{}, fmt.Errorf("resolve Runtime allocation: %w", err) + } + target.Instance = Instance{AllocationID: allocation.ID, ProviderKey: allocation.ProviderKey, DeviceID: allocation.DeviceID} + return target, nil + default: + return Target{}, errors.New("invalid stored Runtime environment type") + } +} diff --git a/services/agents-api/internal/runtimeobs/resolver_test.go b/services/agents-api/internal/runtimeobs/resolver_test.go new file mode 100644 index 000000000..738c79f2c --- /dev/null +++ b/services/agents-api/internal/runtimeobs/resolver_test.go @@ -0,0 +1,73 @@ +package runtimeobs + +import ( + "context" + "errors" + "testing" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +type resolverStore struct { + session store.Session + allocation store.RuntimeAllocation + allocationErr error +} + +func (s resolverStore) GetSession(context.Context, string, string) (store.Session, error) { + return s.session, nil +} + +func (s resolverStore) GetRuntimeAllocation(context.Context, string, string) (store.RuntimeAllocation, error) { + return s.allocation, s.allocationErr +} + +func TestResolverBindsManagedSessionEnvironmentAndAllocation(t *testing.T) { + r, err := NewResolver(resolverStore{ + session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment"}}, + allocation: store.RuntimeAllocation{ID: "allocation", ProviderKey: "provider", DeviceID: "device"}, + }) + if err != nil { + t.Fatal(err) + } + target, err := r.Resolve(t.Context(), "tenant", "session") + if err != nil { + t.Fatal(err) + } + if target.TenantID != "tenant" || target.SessionID != "session" || target.EnvironmentID != "environment" || target.Mode != ModeManaged || target.Instance.AllocationID != "allocation" || target.Instance.ProviderKey != "provider" || target.Instance.DeviceID != "device" { + t.Fatalf("incorrect managed identity binding: %+v", target) + } +} + +func TestResolverKeepsUnsupportedModesDistinct(t *testing.T) { + for _, tc := range []struct { + mode string + environment *store.Environment + }{ + {mode: "none"}, + {mode: "self_hosted", environment: &store.Environment{ID: "environment"}}, + } { + r, err := NewResolver(resolverStore{session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"` + tc.mode + `"}}`), Environment: tc.environment}}) + if err != nil { + t.Fatal(err) + } + target, err := r.Resolve(t.Context(), "tenant", "session") + if err != nil || string(target.Mode) != tc.mode { + t.Fatalf("mode %s was not resolved accurately: %+v %v", tc.mode, target, err) + } + } +} + +func TestResolverReportsManagedAllocationAsUnavailable(t *testing.T) { + r, err := NewResolver(resolverStore{ + session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment"}}, + allocationErr: store.ErrNotFound, + }) + if err != nil { + t.Fatal(err) + } + target, err := r.Resolve(t.Context(), "tenant", "session") + if !errors.Is(err, ErrUnavailable) || target.EnvironmentID != "environment" || target.Mode != ModeManaged { + t.Fatalf("allocation absence was not preserved: %+v %v", target, err) + } +} diff --git a/services/agents-api/internal/runtimeobs/sample.go b/services/agents-api/internal/runtimeobs/sample.go new file mode 100644 index 000000000..bf653c174 --- /dev/null +++ b/services/agents-api/internal/runtimeobs/sample.go @@ -0,0 +1,55 @@ +// Package runtimeobs resolves durable Session identity to provider-owned Runtime +// observations. It is telemetry only and never owns Runtime lifecycle decisions. +package runtimeobs + +import ( + "errors" + "time" +) + +type Mode string + +const ( + ModeNone Mode = "none" + ModeSelfHosted Mode = "self_hosted" + ModeManaged Mode = "openai_hosted" +) + +type Status string + +const ( + StatusObserved Status = "observed" + StatusUnsupported Status = "unsupported" + StatusUnavailable Status = "unavailable" +) + +// Sample contains provider-neutral cumulative counters and current gauges. +// Pointer fields distinguish an observed zero from an unavailable measurement. +type Sample struct { + ObservedAt time.Time + StartedAt *time.Time + + CPUUsageSecondsTotal *float64 + CPUCapacityCores *float64 + MemoryUsageBytes *uint64 + MemoryLimitBytes *uint64 +} + +func (s Sample) validate(now time.Time) error { + if s.ObservedAt.IsZero() || s.ObservedAt.After(now) { + return errors.New("invalid Runtime observation time") + } + if s.StartedAt != nil && (s.StartedAt.IsZero() || s.StartedAt.After(s.ObservedAt)) { + return errors.New("invalid Runtime start time") + } + if s.CPUUsageSecondsTotal != nil && *s.CPUUsageSecondsTotal < 0 { + return errors.New("invalid Runtime CPU usage") + } + if s.CPUCapacityCores != nil && *s.CPUCapacityCores <= 0 { + return errors.New("invalid Runtime CPU capacity") + } + if s.MemoryLimitBytes != nil && *s.MemoryLimitBytes == 0 { + return errors.New("invalid Runtime memory limit") + } + return nil +} diff --git a/services/agents-api/internal/runtimeobs/service.go b/services/agents-api/internal/runtimeobs/service.go new file mode 100644 index 000000000..e05f7c678 --- /dev/null +++ b/services/agents-api/internal/runtimeobs/service.go @@ -0,0 +1,66 @@ +package runtimeobs + +import ( + "context" + "errors" + "fmt" + "time" +) + +type Observation struct { + Target Target + Status Status + Sample *Sample + Reason string +} + +type Service struct { + resolver TargetResolver + sources map[string]Source + now func() time.Time +} + +func NewService(resolver TargetResolver, sources map[string]Source) (*Service, error) { + if resolver == nil { + return nil, errors.New("Runtime observation resolver is required") + } + copySources := make(map[string]Source, len(sources)) + for key, source := range sources { + if key == "" || source == nil { + return nil, errors.New("invalid Runtime observation source") + } + copySources[key] = source + } + return &Service{resolver: resolver, sources: copySources, now: time.Now}, nil +} + +func (s *Service) ObserveSession(ctx context.Context, tenantID, sessionID string) (Observation, error) { + target, err := s.resolver.Resolve(ctx, tenantID, sessionID) + if errors.Is(err, ErrUnavailable) { + return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_allocation_unavailable"}, nil + } + if err != nil { + return Observation{}, err + } + if target.Mode == ModeNone || target.Mode == ModeSelfHosted { + return Observation{Target: target, Status: StatusUnsupported, Reason: "runtime_mode_not_observable"}, nil + } + if target.Mode != ModeManaged || target.Instance.AllocationID == "" || target.Instance.ProviderKey == "" { + return Observation{}, errors.New("invalid managed Runtime observation target") + } + source, ok := s.sources[target.Instance.ProviderKey] + if !ok { + return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_source_unavailable"}, nil + } + sample, err := source.Observe(ctx, target) + if errors.Is(err, ErrUnavailable) { + return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_sample_unavailable"}, nil + } + if err != nil { + return Observation{}, fmt.Errorf("observe Runtime: %w", err) + } + if err := sample.validate(s.now()); err != nil { + return Observation{}, err + } + return Observation{Target: target, Status: StatusObserved, Sample: &sample}, nil +} diff --git a/services/agents-api/internal/runtimeobs/service_test.go b/services/agents-api/internal/runtimeobs/service_test.go new file mode 100644 index 000000000..bff6bf33d --- /dev/null +++ b/services/agents-api/internal/runtimeobs/service_test.go @@ -0,0 +1,91 @@ +package runtimeobs + +import ( + "context" + "errors" + "testing" + "time" +) + +type fixedResolver struct { + target Target + err error +} + +func (r fixedResolver) Resolve(context.Context, string, string) (Target, error) { + return r.target, r.err +} + +type fixedSource struct { + sample Sample + err error + calls int +} + +func (s *fixedSource) Observe(context.Context, Target) (Sample, error) { + s.calls++ + return s.sample, s.err +} + +func TestServiceDoesNotCallSourcesForUnsupportedModes(t *testing.T) { + for _, mode := range []Mode{ModeNone, ModeSelfHosted} { + source := &fixedSource{} + service, err := NewService(fixedResolver{target: Target{Mode: mode}}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnsupported || source.calls != 0 { + t.Fatalf("unsupported mode touched a source: %+v %v calls=%d", observation, err, source.calls) + } + } +} + +func TestServicePreservesUnavailableAndObservedZero(t *testing.T) { + target := Target{Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider"}} + service, err := NewService(fixedResolver{target: target}, nil) + if err != nil { + t.Fatal(err) + } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Sample != nil { + t.Fatalf("missing source was not unavailable: %+v %v", observation, err) + } + + zeroCPU := float64(0) + zeroMemory := uint64(0) + now := time.Date(2026, 9, 22, 1, 0, 0, 0, time.UTC) + source := &fixedSource{sample: Sample{ObservedAt: now, CPUUsageSecondsTotal: &zeroCPU, MemoryUsageBytes: &zeroMemory}} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + service.now = func() time.Time { return now } + observation, err = service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusObserved || observation.Sample == nil || observation.Sample.CPUUsageSecondsTotal == nil || observation.Sample.MemoryUsageBytes == nil { + t.Fatalf("observed zero was lost: %+v %v", observation, err) + } +} + +func TestServiceMapsOnlyDeclaredUnavailability(t *testing.T) { + target := Target{Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider"}} + for _, tc := range []struct { + err error + wantError bool + }{ + {err: ErrUnavailable}, + {err: errors.New("Docker permission denied"), wantError: true}, + } { + service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": &fixedSource{err: tc.err}}) + if err != nil { + t.Fatal(err) + } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if (err != nil) != tc.wantError { + t.Fatalf("wrong error classification: %+v %v", observation, err) + } + if !tc.wantError && observation.Status != StatusUnavailable { + t.Fatalf("declared unavailability was not mapped: %+v", observation) + } + } +} diff --git a/services/agents-api/internal/runtimeobs/source.go b/services/agents-api/internal/runtimeobs/source.go new file mode 100644 index 000000000..4044456ab --- /dev/null +++ b/services/agents-api/internal/runtimeobs/source.go @@ -0,0 +1,13 @@ +package runtimeobs + +import "context" + +// Source reads one provider-owned Runtime instance. Implementations must verify +// ownership before returning data and must not renew, restart, or stop compute. +type Source interface { + Observe(context.Context, Target) (Sample, error) +} + +type TargetResolver interface { + Resolve(context.Context, string, string) (Target, error) +} diff --git a/services/agents-api/internal/sandbox/docker/provider_test.go b/services/agents-api/internal/sandbox/docker/provider_test.go index c99ef6387..7216208f9 100644 --- a/services/agents-api/internal/sandbox/docker/provider_test.go +++ b/services/agents-api/internal/sandbox/docker/provider_test.go @@ -12,6 +12,7 @@ import ( "testing" "time" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/sandbox" "github.com/containerd/errdefs" "github.com/google/uuid" @@ -66,7 +67,8 @@ func TestDockerProviderLifecycle(t *testing.T) { t.Fatal(e) } defer c.Close() - p, e := New(c, Config{InstallationID: uuid.NewString(), Image: image, Network: "bridge", Seccomp: string(seccomp)}) + installationID := uuid.NewString() + p, e := New(c, Config{InstallationID: installationID, Image: image, Network: "bridge", Seccomp: string(seccomp)}) if e != nil { t.Fatal(e) } @@ -111,6 +113,13 @@ func TestDockerProviderLifecycle(t *testing.T) { if info.State != "running" || info.ProviderID == "" { t.Fatalf("bad compute observation: %+v", info) } + resources, e := p.Observe(ctx, runtimeobs.Target{ + TenantID: b.TenantID, SessionID: b.SessionID, EnvironmentID: b.EnvironmentID, Mode: runtimeobs.ModeManaged, + Instance: runtimeobs.Instance{AllocationID: b.AllocationID, ProviderKey: installationID, DeviceID: b.DeviceID}, + }) + if e != nil || resources.StartedAt == nil || resources.CPUUsageSecondsTotal == nil || resources.MemoryUsageBytes == nil || resources.CPUCapacityCores == nil || resources.MemoryLimitBytes == nil { + t.Fatalf("bad resource observation: %+v %v", resources, e) + } inspected, e := p.inspect(ctx, b.Reference) if e != nil { t.Fatal(e) diff --git a/services/agents-api/internal/sandbox/docker/resources.go b/services/agents-api/internal/sandbox/docker/resources.go new file mode 100644 index 000000000..2f444c961 --- /dev/null +++ b/services/agents-api/internal/sandbox/docker/resources.go @@ -0,0 +1,91 @@ +package docker + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/sandbox" + "github.com/moby/moby/api/types/container" + "github.com/moby/moby/client" +) + +var _ runtimeobs.Source = (*Provider)(nil) + +// Observe is read-only. Inspect verifies allocation ownership before Docker +// statistics are requested; it never renews or changes the container. +func (p *Provider) Observe(ctx context.Context, target runtimeobs.Target) (runtimeobs.Sample, error) { + if target.Mode != runtimeobs.ModeManaged || target.Instance.AllocationID == "" { + return runtimeobs.Sample{}, sandbox.ErrInvalid + } + if target.Instance.ProviderKey != p.config.InstallationID { + return runtimeobs.Sample{}, sandbox.ErrOwnership + } + reference := sandbox.Reference{TenantID: target.TenantID, EnvironmentID: target.EnvironmentID, AllocationID: target.Instance.AllocationID} + inspected, err := p.inspect(ctx, reference) + if errors.Is(err, sandbox.ErrNotFound) { + return runtimeobs.Sample{}, runtimeobs.ErrUnavailable + } + if err != nil { + return runtimeobs.Sample{}, err + } + if inspected.Container.State == nil || !inspected.Container.State.Running { + return runtimeobs.Sample{}, runtimeobs.ErrUnavailable + } + result, err := p.client.ContainerStats(ctx, inspected.Container.ID, client.ContainerStatsOptions{Stream: false, IncludePreviousSample: false}) + if err != nil { + return runtimeobs.Sample{}, fmt.Errorf("read Docker Runtime statistics: %w", err) + } + defer result.Body.Close() + var stats dockerStatsResponse + if err := json.NewDecoder(result.Body).Decode(&stats); err != nil { + return runtimeobs.Sample{}, fmt.Errorf("decode Docker Runtime statistics: %w", err) + } + return sampleFromDocker(inspected.Container, stats) +} + +type dockerStatsResponse struct { + Read time.Time `json:"read"` + CPUStats *struct { + CPUUsage *struct { + TotalUsage *uint64 `json:"total_usage"` + } `json:"cpu_usage"` + } `json:"cpu_stats"` + MemoryStats *struct { + Usage *uint64 `json:"usage"` + } `json:"memory_stats"` +} + +func sampleFromDocker(inspected container.InspectResponse, stats dockerStatsResponse) (runtimeobs.Sample, error) { + if inspected.State == nil || inspected.HostConfig == nil || !inspected.State.Running || stats.Read.IsZero() { + return runtimeobs.Sample{}, errors.New("incomplete Docker Runtime observation") + } + startedAt, err := time.Parse(time.RFC3339Nano, inspected.State.StartedAt) + if err != nil || startedAt.IsZero() || startedAt.After(stats.Read) { + return runtimeobs.Sample{}, errors.New("invalid Docker Runtime start time") + } + sample := runtimeobs.Sample{ + ObservedAt: stats.Read, + StartedAt: &startedAt, + } + if stats.CPUStats != nil && stats.CPUStats.CPUUsage != nil && stats.CPUStats.CPUUsage.TotalUsage != nil { + cpuUsage := float64(*stats.CPUStats.CPUUsage.TotalUsage) / float64(time.Second) + sample.CPUUsageSecondsTotal = &cpuUsage + } + if stats.MemoryStats != nil && stats.MemoryStats.Usage != nil { + memoryUsage := *stats.MemoryStats.Usage + sample.MemoryUsageBytes = &memoryUsage + } + if inspected.HostConfig.NanoCPUs > 0 { + capacity := float64(inspected.HostConfig.NanoCPUs) / 1_000_000_000 + sample.CPUCapacityCores = &capacity + } + if inspected.HostConfig.Memory > 0 { + limit := uint64(inspected.HostConfig.Memory) + sample.MemoryLimitBytes = &limit + } + return sample, nil +} diff --git a/services/agents-api/internal/sandbox/docker/resources_test.go b/services/agents-api/internal/sandbox/docker/resources_test.go new file mode 100644 index 000000000..ea4db1096 --- /dev/null +++ b/services/agents-api/internal/sandbox/docker/resources_test.go @@ -0,0 +1,158 @@ +package docker + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/google/uuid" + "github.com/moby/moby/api/types/container" + "github.com/moby/moby/client" +) + +func TestObserveVerifiesOwnershipThenReadsOneShotStats(t *testing.T) { + installationID := uuid.NewString() + target := runtimeobs.Target{ + TenantID: uuid.NewString(), SessionID: uuid.NewString(), EnvironmentID: uuid.NewString(), Mode: runtimeobs.ModeManaged, + Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), ProviderKey: installationID, DeviceID: uuid.NewString()}, + } + observed := time.Now().UTC().Truncate(time.Microsecond) + started := observed.Add(-time.Minute) + statsRead := false + omitMeasurements := false + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch { + case r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/json"): + _ = json.NewEncoder(w).Encode(map[string]any{ + "Id": "container-id", + "State": map[string]any{"Status": "running", "Running": true, "StartedAt": started.Format(time.RFC3339Nano)}, + "HostConfig": map[string]any{"NanoCpus": 2_000_000_000, "Memory": 2048}, + "Config": map[string]any{"Labels": map[string]string{ + labelPrefix + "installation": installationID, + labelPrefix + "tenant": target.TenantID, + labelPrefix + "environment": target.EnvironmentID, + labelPrefix + "allocation": target.Instance.AllocationID, + }}, + }) + case r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/containers/container-id/stats"): + if r.URL.Query().Get("stream") != "false" || r.URL.Query().Get("one-shot") != "true" { + t.Errorf("stats request was not one-shot: %s", r.URL.RawQuery) + } + statsRead = true + if omitMeasurements { + _ = json.NewEncoder(w).Encode(map[string]any{"read": observed}) + return + } + _ = json.NewEncoder(w).Encode(container.StatsResponse{ + Read: observed, + CPUStats: container.CPUStats{CPUUsage: container.CPUUsage{TotalUsage: 1_500_000_000}}, + MemoryStats: container.MemoryStats{Usage: 1024}, + }) + default: + http.NotFound(w, r) + } + })) + defer server.Close() + c, err := client.New(client.WithHost(server.URL)) + if err != nil { + t.Fatal(err) + } + defer c.Close() + p, err := New(c, Config{InstallationID: installationID, Image: "fixture@sha256:" + strings.Repeat("a", 64), Network: "bridge", Seccomp: `{}`}) + if err != nil { + t.Fatal(err) + } + sample, err := p.Observe(t.Context(), target) + if err != nil || !statsRead || sample.CPUUsageSecondsTotal == nil || *sample.CPUUsageSecondsTotal != 1.5 || sample.MemoryUsageBytes == nil || *sample.MemoryUsageBytes != 1024 { + t.Fatalf("bad one-shot observation: %+v %v stats=%v", sample, err, statsRead) + } + omitMeasurements = true + missing, err := p.Observe(t.Context(), target) + if err != nil || missing.CPUUsageSecondsTotal != nil || missing.MemoryUsageBytes != nil || missing.CPUCapacityCores == nil || missing.MemoryLimitBytes == nil { + t.Fatalf("missing Docker measurements became zero: %+v %v", missing, err) + } + foreign := target + foreign.Instance.ProviderKey = uuid.NewString() + statsRead = false + if _, err := p.Observe(t.Context(), foreign); err == nil || statsRead { + t.Fatal("foreign provider identity reached Docker stats") + } +} + +func TestSampleFromDockerPreservesObservedZeroAndConfiguredCapacity(t *testing.T) { + observed := time.Date(2026, 9, 22, 1, 2, 3, 0, time.UTC) + started := observed.Add(-5 * time.Minute) + zero := uint64(0) + sample, err := sampleFromDocker(container.InspectResponse{ + State: &container.State{Running: true, StartedAt: started.Format(time.RFC3339Nano)}, + HostConfig: &container.HostConfig{Resources: container.Resources{NanoCPUs: 2_000_000_000, Memory: 2 * 1024 * 1024 * 1024}}, + }, testDockerStats(observed, &zero, &zero)) + if err != nil { + t.Fatal(err) + } + if sample.ObservedAt != observed || sample.StartedAt == nil || !sample.StartedAt.Equal(started) { + t.Fatalf("lost Docker observation time: %+v", sample) + } + if sample.CPUUsageSecondsTotal == nil || *sample.CPUUsageSecondsTotal != 0 || sample.MemoryUsageBytes == nil || *sample.MemoryUsageBytes != 0 { + t.Fatalf("observed zero became unavailable: %+v", sample) + } + if sample.CPUCapacityCores == nil || *sample.CPUCapacityCores != 2 || sample.MemoryLimitBytes == nil || *sample.MemoryLimitBytes != 2*1024*1024*1024 { + t.Fatalf("lost configured capacity: %+v", sample) + } +} + +func TestSampleFromDockerNormalizesCumulativeCPU(t *testing.T) { + observed := time.Now().UTC() + started := observed.Add(-time.Hour) + cpu, memory := uint64(2_500_000_000), uint64(4096) + sample, err := sampleFromDocker(container.InspectResponse{ + State: &container.State{Running: true, StartedAt: started.Format(time.RFC3339Nano)}, + HostConfig: &container.HostConfig{}, + }, testDockerStats(observed, &cpu, &memory)) + if err != nil { + t.Fatal(err) + } + if sample.CPUUsageSecondsTotal == nil || *sample.CPUUsageSecondsTotal != 2.5 || sample.MemoryUsageBytes == nil || *sample.MemoryUsageBytes != 4096 { + t.Fatalf("bad Docker normalization: %+v", sample) + } + if sample.CPUCapacityCores != nil || sample.MemoryLimitBytes != nil { + t.Fatalf("invented unconfigured capacity: %+v", sample) + } +} + +func TestSampleFromDockerRejectsIncompleteState(t *testing.T) { + observed := time.Now().UTC() + for _, inspected := range []container.InspectResponse{ + {}, + {State: &container.State{Running: false}, HostConfig: &container.HostConfig{}}, + {State: &container.State{Running: true, StartedAt: "invalid"}, HostConfig: &container.HostConfig{}}, + } { + if _, err := sampleFromDocker(inspected, dockerStatsResponse{Read: observed}); err == nil { + t.Fatal("accepted incomplete Docker state") + } + } +} + +func testDockerStats(observed time.Time, cpu, memory *uint64) dockerStatsResponse { + stats := dockerStatsResponse{Read: observed} + if cpu != nil { + stats.CPUStats = &struct { + CPUUsage *struct { + TotalUsage *uint64 `json:"total_usage"` + } `json:"cpu_usage"` + }{CPUUsage: &struct { + TotalUsage *uint64 `json:"total_usage"` + }{TotalUsage: cpu}} + } + if memory != nil { + stats.MemoryStats = &struct { + Usage *uint64 `json:"usage"` + }{Usage: memory} + } + return stats +} From 2a0715f0ea876e85d7d54127cbae097254ce23d0 Mon Sep 17 00:00:00 2001 From: sam Date: Tue, 22 Sep 2026 15:39:51 +0800 Subject: [PATCH 2/6] Document runtime observability API and dashboard design --- .../agents-api/runtime-observability-api.md | 281 +++++++++++++ .../runtime-observability-design.md | 388 ++++++++++++++++++ contracts/agents-api/runtime-observability.md | 4 + 3 files changed, 673 insertions(+) create mode 100644 contracts/agents-api/runtime-observability-api.md create mode 100644 contracts/agents-api/runtime-observability-design.md diff --git a/contracts/agents-api/runtime-observability-api.md b/contracts/agents-api/runtime-observability-api.md new file mode 100644 index 000000000..4e607a9d7 --- /dev/null +++ b/contracts/agents-api/runtime-observability-api.md @@ -0,0 +1,281 @@ +# Runtime observation API proposal + +Status: review proposal. These routes are not implemented and are not yet present +in `openapi.yaml`. + +This is an Agents Core extension, not an upstream OpenAI Agents resource. The +implementation must record that status in the coverage ledger and generated +OpenAPI contract. + +## Routes + +### List current Runtime observations + +```http +GET /v1/agents/runtime-observations?after={target_id}&limit=20&order=desc +OpenAI-Beta: agents=v1 +Authorization: Bearer ... +``` + +| Field | Rules | +| --- | --- | +| `after` | Observation ID from the previous page. Optional, supplied once. | +| `limit` | Integer 1–100, default 20. | +| `order` | `asc` or `desc`, default `desc`. | + +The list contains one current Runtime context for every Session visible to the +authenticated tenant, including explicit `none`, unsupported `self_hosted`, and +released managed contexts. Ordering uses the same Session creation-time and ID +keyset as the Session list. An observation ID is the Session UUID, so pagination +does not change when the underlying Runtime incarnation changes. Pages are not an +atomic telemetry snapshot; every row has its own `resolved_at`, and a successful +provider sample has its own `observed_at`. A client completes the entire page chain +before publishing a new Dashboard snapshot. + +```json +{ + "object": "list", + "data": [ + { + "id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", + "object": "agent.runtime_observation", + "session_id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", + "environment_id": "env_...", + "mode": "openai_hosted", + "provider_type": "docker", + "instance": { + "kind": "managed_allocation", + "allocation_id": "alloc_...", + "device_id": "device_...", + "connection_generation": null + }, + "status": "observed", + "reason": null, + "allocation_created_at": 1789951200, + "resolved_at": 1789953021, + "observed_at": 1789953020, + "started_at": 1789951220, + "cpu": { + "usage_seconds_total": 482.75, + "capacity_cores": 2.0, + "usage_cores": 1.42, + "utilization_ratio": 0.71 + }, + "memory": { + "usage_bytes": 805306368, + "limit_bytes": 2147483648 + } + } + ], + "has_more": false, + "first_id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", + "last_id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3" +} +``` + +### Retrieve one Session's current Runtime observation + +```http +GET /v1/agents/sessions/{session_id}/runtime-observation +OpenAI-Beta: agents=v1 +Authorization: Bearer ... +``` + +This returns the same object shape as a list item. It never starts a Turn, creates +an Environment, provisions compute, renews a lease, or changes lifecycle state. + +A valid `environment:none` Session returns `200` with status `unsupported`; the +Session exists but has no attributable Runtime instance. A missing or foreign +Session returns the existing indistinguishable not-found error. + +## Resource schema + +### `RuntimeObservation` + +| Field | Type | Required | Semantics | +| --- | --- | --- | --- | +| `id` | string | yes | Session UUID; stable identity of this current-observation resource and its list cursor. | +| `object` | literal | yes | `agent.runtime_observation`. | +| `session_id` | string | yes | Authorized Core Session. | +| `environment_id` | string or null | yes | Null only for mode `none`. | +| `mode` | enum | yes | `none`, `self_hosted`, `openai_hosted`. | +| `provider_type` | string or null | yes | Forward-compatible safe source kind such as `docker`; null when no provider applies. Clients must not treat an unknown nonempty value as an error. | +| `instance` | object | yes | Provider-neutral current incarnation identity; explicit `kind=none` when no compute applies. | +| `status` | enum | yes | `observed`, `unsupported`, `unavailable`. | +| `reason` | enum or null | yes | Safe reason when status is not `observed`. | +| `allocation_created_at` | integer or null | yes | Unix seconds for managed allocation age. | +| `resolved_at` | integer | yes | Unix seconds when Core resolved identity and status for this row. | +| `observed_at` | integer or null | yes | Provider sample time; null without a sample. | +| `started_at` | integer or null | yes | Current compute incarnation start time. | +| `cpu` | object or null | yes | Null when no CPU fields were observed. | +| `memory` | object or null | yes | Null when no memory fields were observed. | + +### `RuntimeInstance` + +```json +{ + "kind": "managed_allocation", + "allocation_id": "alloc_...", + "device_id": "device_...", + "connection_generation": null +} +``` + +`kind` is `managed_allocation`, `self_hosted_connection`, or `none`. For a managed +context, `allocation_id` is the incarnation key; for self-hosted, the current +`connection_generation` is the incarnation key. Fields that do not apply are +explicit nulls. Provider-native container IDs, pod names, host paths, credentials, +and raw labels are not public fields. + +### `RuntimeCPUObservation` + +```json +{ + "usage_seconds_total": 482.75, + "capacity_cores": 2.0, + "usage_cores": 1.42, + "utilization_ratio": 0.71 +} +``` + +All fields are `number | null`. Values are finite and nonnegative; +`capacity_cores`, when present, is greater than zero. Numeric zero is observed +zero. Null is unavailable. `usage_cores` is the cumulative CPU delta divided by +the observation-time delta for two ordered samples of the same incarnation. +`utilization_ratio` is `usage_cores / capacity_cores`. It is not clamped: a value +above 1 is retained as provider/accounting evidence and is not interpreted as a +lifecycle signal. Both derived fields are null after a cache restart or whenever +either source sample is absent or invalid. The API never derives CPU rate from a +single sample. + +### `RuntimeMemoryObservation` + +```json +{ + "usage_bytes": 805306368, + "limit_bytes": 2147483648 +} +``` + +Both fields are `integer | null`. Values are nonnegative and safe JSON integers. +Zero usage is observed zero. A missing or unlimited provider limit is null. + +## Status and reason matrix + +| Status | Allowed reason | +| --- | --- | +| `observed` | null | +| `unsupported` | `runtime_mode_not_observable` | +| `unavailable` | `allocation_pending`, `runtime_not_running`, `source_not_configured`, `sample_timeout`, `sample_unavailable` | + +Ownership mismatch, malformed durable identity, corrupt provider evidence, and +authorization failure are not downgraded to unavailable rows. + +## Error responses + +Use the existing Agents API error envelope. + +| HTTP | Code | When | +| --- | --- | --- | +| 400 | `unsupported_parameter` | Unknown or duplicate query fields. | +| 400 | `invalid_request` | Empty or invalid limits, order, or malformed cursor. | +| 401 | `authentication_error` | Missing or invalid API authentication. | +| 404 | `not_found` | Missing or foreign Session/cursor, indistinguishably. | +| 429 | `rate_limit_exceeded` | Runtime sampling read budget exceeded. | +| 500 | `internal_error` | Integrity, ownership, or invalid provider evidence. | +| 503 | `execution_unavailable` | Required Runtime observation service is not configured. | + +Errors never include provider raw responses or credentials. + +## Freshness and caching + +- Return `Cache-Control: no-store`. +- An internal cache may coalesce reads for at most five seconds. +- `observed_at` is authoritative for freshness; HTTP response time is not. +- Clients mark samples stale according to their own explicit threshold. +- `ETag` is not proposed because observations change independently. + +## Client contract + +`packages/agents-client` should expose: + +```ts +type RuntimeObservationStatus = "observed" | "unsupported" | "unavailable"; +type RuntimeObservationReason = + | "runtime_mode_not_observable" + | "allocation_pending" + | "runtime_not_running" + | "source_not_configured" + | "sample_timeout" + | "sample_unavailable"; + +interface RuntimeObservation { + id: string; + object: "agent.runtime_observation"; + session_id: string; + environment_id: string | null; + mode: "none" | "self_hosted" | "openai_hosted"; + provider_type: string | null; + instance: { + kind: "managed_allocation" | "self_hosted_connection" | "none"; + allocation_id: string | null; + device_id: string | null; + connection_generation: string | null; + }; + status: RuntimeObservationStatus; + reason: RuntimeObservationReason | null; + allocation_created_at: number | null; + resolved_at: number; + observed_at: number | null; + started_at: number | null; + cpu: { + usage_seconds_total: number | null; + capacity_cores: number | null; + usage_cores: number | null; + utilization_ratio: number | null; + } | null; + memory: { + usage_bytes: number | null; + limit_bytes: number | null; + } | null; +} + +interface RuntimeObservationPage { + object: "list"; + data: RuntimeObservation[]; + has_more: boolean; + first_id: string | null; + last_id: string | null; +} + +interface RuntimeObservationClient { + list(options?: { + after?: string; + limit?: number; + order?: "asc" | "desc"; + }): Promise; + + retrieveForSession(sessionId: string): Promise; +} +``` + +The client validates every required field, enum, nullability rule, timestamp, and +finite number. Unknown additive fields are ignored. Malformed data rejects the +whole page; Web does not publish a partial snapshot. + +Web also applies a configured whole-refresh budget. If `has_more` remains true +when that budget is exhausted, it retains the prior complete snapshot and marks +the refresh incomplete; it does not publish partial values as global totals. +After both Runtime-observation and Session traversals complete, Web also requires +their Session ID sets to be identical. A mismatch caused by concurrent creation or +deletion makes the candidate incomplete and prevents publication. + +## Deliberately excluded + +- Token usage: use existing Session/Turn Usage. +- Billing and cost: product/backend concern. +- Historical series: optional later capability with a separate contract. +- Container logs and command output. +- Provider credentials or native configuration. +- Start, stop, pause, resume, restart, renew, or delete operations. +- Idle classification and automatic shutdown. diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md new file mode 100644 index 000000000..70ca2fe63 --- /dev/null +++ b/contracts/agents-api/runtime-observability-design.md @@ -0,0 +1,388 @@ +# Runtime observability and Dashboard design + +Status: review proposal. Phase 1 provider abstraction and Docker sampling are +implemented; the public API, Web integration, history backend, additional +providers, and lifecycle automation described below are not implemented. + +## 1. Problem statement + +Operators need one Dashboard that answers four separate questions without +confusing their sources of truth: + +1. Which Runtime instances currently belong to which tenant, Session, and + Environment? +2. What compute is allocated and what is it consuming now? +3. How long has allocation, compute, and model work been active? +4. How many model tokens have been reported for the corresponding Sessions? + +The design must work across managed Docker now and later managed Kubernetes, +E2B, and authenticated self-hosted Runtime deployments. Metrics are operational +evidence. They must not become execution or lifecycle authority. + +## 2. Goals + +- Resolve every sample through durable Core identity before provider access. +- Keep provider-specific collection behind one source interface. +- Preserve observed zero, unavailable measurements, and unsupported modes as + different states. +- Provide a bounded read-only API suitable for Core Web and other operators. +- Let Web combine Runtime observations with existing Session and Turn usage + without copying execution truth into the browser. +- Keep current snapshots independent from an optional history backend. +- Define a safe path to future idle shutdown without implementing it implicitly. + +## 3. Non-goals + +- Redefining the pinned OpenAI Agents resources. +- Adding product users, organizations, billing, or authorization tables to Core. +- Treating a Session, daemon socket, container, pod, native harness Session, or + Turn as the same identity. +- Estimating missing CPU, memory, token, or duration values. +- Using telemetry, heartbeat age, low CPU, or Prometheus state to stop compute. +- Adding lifecycle actions to the first Dashboard release. +- Storing time-series samples in PostgreSQL. + +## 4. Source-of-truth model + +| Concern | Authority | Notes | +| --- | --- | --- | +| Tenant and Session ownership | Core database | Every public read is tenant-scoped. | +| Environment placement | Session configuration and Environment row | `none`, `self_hosted`, or `openai_hosted`. | +| Observation resource identity | Session ID | One current observation resource exists per tenant-owned Session. | +| Managed Runtime identity | `runtime_allocations` | Allocation and provider key identify the compute incarnation. | +| Self-hosted Runtime identity | Environment connection generation | Future telemetry must be generation-fenced. | +| Container/pod resource values | Selected provider source | Read-only, point-in-time evidence. | +| Turn state and busy duration | Core Turns | Never inferred from CPU. | +| Token usage | Existing Session/Turn usage | Missing native usage remains unknown. | +| Historical resource series | Optional telemetry backend | Not execution or lifecycle authority. | +| Idle shutdown decision | Future durable Core control state | Separate design and migration. | + +## 5. Identity chain + +```text +managed +tenant_id -> session_id -> environment_id -> runtime_allocation_id + -> provider_key -> provider-owned container/pod/instance + +self-hosted (future) +tenant_id -> session_id -> environment_id + -> device_id + connection_generation -> authenticated Runtime report + +none +tenant_id -> session_id + -> no Session-owned Runtime instance +``` + +Provider-native identifiers are never accepted from browser input. The resolver +starts from the authorized tenant and Session, loads the committed Environment and +allocation, and only then selects the configured source by persisted provider key. +The provider independently verifies its labels or equivalent ownership metadata. + +## 6. Component architecture + +```mermaid +flowchart LR + Web[Core Web Dashboard] --> Client[packages/agents-client] + Client --> API[Agents API read handlers] + API --> Service[runtimeobs.Service] + Service --> Resolver[durable identity resolver] + Resolver --> DB[(Core PostgreSQL)] + Service --> Registry[provider source registry] + Registry --> Docker[Docker Inspect and one-shot Stats] + Registry -. future .-> K8s[Kubernetes Metrics API or cAdvisor] + Registry -. future .-> E2B[E2B metrics adapter] + Registry -. future .-> Self[authenticated daemon telemetry] + Service -. optional export .-> Telemetry[OTLP or Prometheus pipeline] + Telemetry -. future history reads .-> History[operator history adapter] +``` + +### 6.1 `runtimeobs` + +Owns provider-neutral identity, mode resolution, source selection, sample +validation, and observation status. It must not import provider SDKs or mutate +Runtime lifecycle. + +### 6.2 Provider sources + +Each source receives a fully resolved target and returns one normalized sample. +A source must verify target ownership, make only bounded read calls, preserve +missing fields, return cumulative CPU seconds, and never create, renew, restart, +pause, or stop compute. + +Docker uses Inspect followed by non-streaming one-shot Stats. Kubernetes should +retain pod UID, container identity, and restart boundaries. E2B must use an +API-supported instance identity rather than display names. Self-hosted metrics +require authenticated daemon messages fenced by the current connection generation. + +### 6.3 API composition + +The API resolves durable rows first, samples sources with bounded concurrency, +maps only ordinary absence/timeouts to safe unavailable reasons, and fails closed +on ownership or integrity errors. It returns current observations only. + +### 6.4 Web composition + +Web loads the complete paginated Runtime observation collection before publishing +a new Dashboard snapshot. The collection follows the same Session creation-time +and ID keyset as the Session list, so a Runtime incarnation change cannot invalidate +pagination. It separately uses existing Session/Turn reads for status and tokens +and joins only by exact Session identity. Before publication, the set of Session +IDs from both complete traversals must be identical. Concurrent Session creation +or deletion can make the sets differ because the APIs have no shared snapshot +token; Web then discards the candidate, marks the refresh incomplete, and keeps +the previous successful snapshot visibly stale. + +The browser enforces a configured refresh budget for total pages, targets, and +elapsed time. Exhausting that budget is an incomplete refresh: Web retains the +previous complete snapshot and does not relabel partial aggregates as tenant-wide. + +## 7. Normalized sample + +```go +type Sample struct { + ObservedAt time.Time + StartedAt *time.Time + + CPUUsageSecondsTotal *float64 + CPUCapacityCores *float64 + MemoryUsageBytes *uint64 + MemoryLimitBytes *uint64 +} +``` + +Pointer presence is semantic. `0` means observed zero; `nil` means unavailable. +CPU percentage is derived from the delta between two cumulative samples and their +observation times. A single sample cannot truthfully supply CPU percentage. + +The API projection may additionally expose `usage_cores` and `utilization_ratio` +only when the service has two ordered samples for the same Runtime incarnation. +The process-local observation cache keeps the previous cumulative value for this +calculation. Its loss makes the derived fields temporarily null; it never changes +the cumulative source measurement or lifecycle state. + +## 8. Duration semantics + +| UI label | Calculation | Meaning | +| --- | --- | --- | +| Allocation age | allocation `created_at` to `released_at` or now | Age of Core's allocation record. | +| Compute uptime | provider `started_at` to sample `observed_at` | Age of the current compute incarnation. | +| Busy duration | Turn `started_at` to `completed_at` or now | Time model work has been active. | +| Idle duration | future durable `idle_since` | Not available in the current design. | + +Container restart resets compute uptime but not allocation age. Dashboard labels +must not collapse these values into one generic Runtime duration. + +## 9. Collection behavior + +### 9.1 Current snapshot path + +- List one current target context for each tenant-owned Session in the same stable + Session creation-time and ID order used by the Session list. A released managed + allocation remains attributable but reports `runtime_not_running`; `none` and + unsupported `self_hosted` remain explicit rows rather than disappearing. +- Default page size 20, maximum 100. +- Sample at most eight providers concurrently. +- Default per-source budget two seconds and whole-request budget ten seconds. +- Do not retry a source call inside the HTTP request. +- An optional process-local singleflight/cache may coalesce identical reads for up + to five seconds and retain the previous cumulative sample for CPU-rate + calculation. It is an optimization only and may be lost on restart. +- Do not write samples to the Core database. + +### 9.2 Error classification + +| Condition | API result | +| --- | --- | +| Mode `none` or unsupported `self_hosted` | Row status `unsupported`. | +| Managed allocation not created yet | `unavailable`, reason `allocation_pending`. | +| Owned Runtime absent or stopped | `unavailable`, reason `runtime_not_running`. | +| Source not configured | `unavailable`, reason `source_not_configured`. | +| Source deadline | `unavailable`, reason `sample_timeout`. | +| Ownership mismatch or invalid durable identity | Fail the request and log a sanitized integrity error. | +| Database/authentication failure | Existing safe API error mapping. | + +Raw Docker, Kubernetes, E2B, daemon, host, credential, or network diagnostics are +never returned to the browser. + +## 10. Historical metrics + +Current API reads and history are separate capabilities. The initial API does not +provide charts over time. A later operator-configured adapter may query an +OTLP/Prometheus-compatible backend. Core must not make that backend mandatory for +Session execution or current snapshot reads. + +Recommended instruments are: + +- `agents.runtime.cpu.usage` cumulative seconds; +- `agents.runtime.cpu.capacity` cores; +- `agents.runtime.memory.usage` bytes; +- `agents.runtime.memory.limit` bytes; +- `agents.runtime.sample` success/unavailable count; and +- `agents.runtime.sample.duration` seconds. + +Provider type, Runtime mode, and coarse status are safe low-cardinality labels. +High-cardinality identities require tenant-scoped access and retention policies; +they are not global Prometheus labels by default. + +## 11. Dashboard information architecture + +### 11.1 Overview + +- Active managed Runtime count. +- Observed CPU usage and known configured capacity. +- Observed memory usage and known limits. +- Reported Session tokens, together with the reporting Session count. +- Data freshness and source coverage. + +Aggregates include only present measurements. Each total states its denominator, +for example, `6.4 / 12 cores across 6 of 8 active Runtimes`. Unknown is never added +as zero. + +### 11.2 Runtime table + +Each row shows Session, Agent/harness when already available from the Session +snapshot, mode, observation status, CPU, memory, compute uptime, Turn state, and +reported tokens. Rows navigate to the existing Session view. No stop, restart, +pause, or delete actions appear in the first release. + +### 11.3 Detail view + +The detail surface shows exact Session/Environment/allocation identity, provider +type, observation timestamps, allocation age, compute uptime, and safe unavailable +reason. It displays only the Core-owned identifiers explicitly present in the +public contract. Provider-native container IDs, pod names, instance names, host +paths, and raw labels are never displayed. + +### 11.4 States + +- **Loading:** no previous complete Runtime snapshot. +- **Fresh:** every page loaded and each row carries its own resolution time; a + provider sample also carries its independent observation time. +- **Stale:** refresh failed; previous complete snapshot retained. +- **Unavailable row:** identity is valid, measurement is temporarily absent. +- **Unsupported row:** mode is recognized but has no qualified source. +- **Integrity failure:** do not publish a partial replacement snapshot. + +### 11.5 Web implementation shape + +The Web change belongs in Core Web, not the Core service repository. It uses +`packages/agents-client` as the only Runtime-observation transport and keeps four +seams separate: + +1. A client/parser module validates one page and exposes list and Session-scoped + retrieval methods. +2. A refresh coordinator loads all observation pages plus the canonical Session + collection, applies page/target/time budgets, requires exact equality of their + Session ID sets, and atomically swaps only a complete joined snapshot. A set + mismatch is an incomplete refresh, not a partial success. +3. A feature-local state model retains `last_complete`, current refresh status, + local filters, and the selected time range. It aborts an overlapping refresh + and marks old data stale after a failed or incomplete refresh. +4. Presentational components render summary coverage, the Runtime table, and a + Session detail surface. Trend components are absent unless a later history + capability and contract are configured. + +The initial refresh cadence is an operator-configured value, not an API guarantee. +Web pauses periodic reads when hidden, refreshes when visibility returns, and adds +jitter so multiple browsers do not synchronize. Filtering is local to the last +complete snapshot and never changes tenant authorization or provider selection. + +## 12. Token usage boundary + +Runtime observations do not duplicate token usage. Web uses the existing canonical +Session Usage snapshot and joins it to Runtime rows by `session_id`. The Dashboard +shows both total and coverage, such as `1.84M reported by 7/8 Sessions`. Missing or +incomplete native usage remains unknown. + +Cost and billing stay outside this Core API. A product may join billing in its own +authorized backend, never by exposing product credentials to Core Web. + +## 13. Security and tenancy + +- Authenticate with the existing Agents API mechanism. +- Scope resolution to the authenticated tenant before provider access. +- Do not accept provider key, allocation ID, container ID, pod UID, or device ID + as an authority-bearing query parameter. +- Bound per-request list size, concurrency, response bytes, and source deadlines; + Web separately bounds a complete multi-page refresh. +- Sanitize logs through `internal/obs/log`. +- Never return credentials, environment variables, Docker raw JSON, daemon status + payloads, host paths, image registry credentials, or backend credentials. +- Rate-limit collection separately from ordinary Session reads. + +## 14. Data model impact + +Current snapshot and Dashboard work require no migration. Existing +`runtime_allocations`, `environments`, Sessions, Turns, and Usage are sufficient. +No time-series table is proposed. + +Automatic idle shutdown is a separate feature. It requires durable fields such as +`activity_revision`, `idle_since`, and `shutdown_requested_at` with fenced state +transitions. That migration cannot read a monitoring backend as authority. + +## 15. Delivery plan + +### Phase 1: provider-neutral foundation + +- `runtimeobs` identity, resolver, source, sample, and service. +- Managed Docker Inspect/Stats source. +- CPU, memory, and current compute start time. +- Explicit unsupported and unavailable states. + +### Phase 2: current snapshot API + +- Add extension types under `contracts/agents-api/v1`. +- Add collection and Session-scoped handlers. +- Add `packages/agents-client` methods and raw HTTP/client coverage. +- Add bounded concurrency, timeout, authorization, and error tests. +- Regenerate the public OpenAPI contract. + +### Phase 3: Web Dashboard + +- Add Runtime observations as a third independent Dashboard collection. +- Publish only complete traversals and retain the previous snapshot on failure. +- Join existing Session Usage and Turn status by exact Session ID. +- Add responsive, keyboard-accessible current-resource views. + +### Phase 4: optional history + +- Add telemetry exporter and qualified operator backend. +- Define a separate history query adapter and retention/security policy. +- Add trend charts only when this capability is advertised. + +### Phase 5: additional sources + +- Kubernetes, E2B, and generation-fenced self-hosted telemetry. +- Each source requires independent mechanism and deployment acceptance. + +### Phase 6: idle policy + +- Separate durable activity and shutdown state machine. +- No automatic action until race, fencing, recovery, and operator-control + acceptance is complete. + +## 16. Acceptance criteria + +- Every observation proves tenant, Session, Environment, and Runtime-instance + association. +- Managed Docker emits correct present/absent semantics and never mutates compute. +- Unsupported modes never look like zero usage. +- Collection calls are bounded and one ordinary unavailable source invents no data. +- Ownership/integrity mismatch fails closed. +- Web never publishes a partial page traversal as a current snapshot. +- Web rejects cross-collection Session membership skew, including concurrent + Session create/delete cases, before publishing tenant-wide aggregates. +- Token totals report coverage and do not estimate missing usage. +- No lifecycle action is reachable from the first Dashboard. +- No new database table is required for current snapshots or history export. + +## 17. Review decisions required + +1. Accept the proposed API as a documented Core extension rather than an upstream + OpenAI resource. +2. Accept current snapshots without atomic cross-row time semantics; every row + exposes its own `observed_at`. +3. Confirm that history is optional and external, not a PostgreSQL sample table. +4. Confirm that the first Web release has no lifecycle controls. +5. Choose whether `self_hosted` remains visibly unsupported until authenticated + generation-fenced telemetry is qualified. diff --git a/contracts/agents-api/runtime-observability.md b/contracts/agents-api/runtime-observability.md index 61d64ccc5..658a136ed 100644 --- a/contracts/agents-api/runtime-observability.md +++ b/contracts/agents-api/runtime-observability.md @@ -71,3 +71,7 @@ token aggregation, Kubernetes/E2B source, or automatic lifecycle action. The internal source interface is intended to admit those providers without changing Session attribution or the existing sandbox lifecycle interface. +The review proposal for later API and Web phases is split into the +[full design](runtime-observability-design.md) and the +[proposed public extension](runtime-observability-api.md). Neither document marks +those later phases as implemented. From 45ff7f7caa4453ee9cae0650f133a558bb6cf014 Mon Sep 17 00:00:00 2001 From: sam Date: Tue, 22 Sep 2026 18:07:27 +0800 Subject: [PATCH 3/6] feat(agents-api): expose runtime observations --- contracts/agents-api/README.md | 10 + contracts/agents-api/openapi.yaml | 283 ++++++++++++++++++ .../agents-api/runtime-observability-api.md | 84 +++--- .../runtime-observability-design.md | 33 +- .../agents-api/v1/runtime_observations.go | 46 +++ packages/agents-client/src/client.test.ts | 153 ++++++++++ packages/agents-client/src/client.ts | 197 ++++++++++++ .../agents-client/src/protocol-types.test.ts | 21 ++ packages/agents-client/src/types.ts | 115 +++++++ services/agents-api/cmd/server/main.go | 21 +- services/agents-api/internal/api/handler.go | 31 +- .../agents-api/internal/api/handler_test.go | 15 +- .../internal/api/runtime_observations.go | 221 ++++++++++++++ .../internal/api/runtime_observations_test.go | 224 ++++++++++++++ .../internal/runtimeobs/identity.go | 4 + .../internal/runtimeobs/resolver.go | 19 +- .../internal/runtimeobs/resolver_test.go | 59 +++- .../agents-api/internal/runtimeobs/sample.go | 15 +- .../agents-api/internal/runtimeobs/service.go | 54 +++- .../internal/runtimeobs/service_test.go | 165 +++++++++- .../internal/sandbox/docker/resources.go | 6 +- .../internal/sandbox/docker/resources_test.go | 10 +- 22 files changed, 1674 insertions(+), 112 deletions(-) create mode 100644 contracts/agents-api/v1/runtime_observations.go create mode 100644 services/agents-api/internal/api/runtime_observations.go create mode 100644 services/agents-api/internal/api/runtime_observations_test.go diff --git a/contracts/agents-api/README.md b/contracts/agents-api/README.md index c331ddb01..5049296c6 100644 --- a/contracts/agents-api/README.md +++ b/contracts/agents-api/README.md @@ -98,6 +98,16 @@ paths start at `/vaults`, not `/agents/vaults`. | vaults | create, retrieve, list, delete | Create/retrieve/list/delete with independent tenant persistence, stored status filtering, atomic Credential cascade and frozen Session attachments; archive semantics and full hosted lifecycle parity remain missing | | vaults.credentials | create, retrieve, update, list, delete | Static-bearer create/retrieve/list/token replacement/deletion with scoped encrypted storage; Session attachment and exact-URL HTTPS MCP binding; OAuth, archive semantics and full hosted lifecycle parity remain missing | +## Core extension inventory + +The operations below are implemented public Core extensions. They are excluded +from the 42-operation upstream inventory and must not be counted as OpenAI Agents +compatibility. + +| Extension | Operations | Current coverage | +| --- | --- | --- | +| Runtime observations | `GET /v1/agents/runtime-observations`; `GET /v1/agents/sessions/{session_id}/runtime-observation` | Current, read-only, tenant-scoped Session contexts with stable Session-keyset pagination, bounded concurrent sampling, Docker metrics, explicit unsupported/unavailable states, strict `packages/agents-client` projection, and no lifecycle mutation. Kubernetes, E2B, self-hosted telemetry, history, CPU-rate derivation, and automatic idle policy remain unimplemented. See [Runtime observation API](runtime-observability-api.md). | + For each resource, verify the referenced request/response unions and observable behavior, not just the route. Non-text initial input, configuration options, text/image content, function results, environment variants, full Item/SSE diff --git a/contracts/agents-api/openapi.yaml b/contracts/agents-api/openapi.yaml index 3578bf4ff..90a206045 100644 --- a/contracts/agents-api/openapi.yaml +++ b/contracts/agents-api/openapi.yaml @@ -868,6 +868,180 @@ definitions: required: - type type: object + v1.RuntimeCPUObservation: + properties: + capacity_cores: + minimum: 5e-324 + type: number + x-nullable: true + usage_cores: + minimum: 0 + type: number + x-nullable: true + usage_seconds_total: + minimum: 0 + type: number + x-nullable: true + utilization_ratio: + minimum: 0 + type: number + x-nullable: true + required: + - capacity_cores + - usage_cores + - usage_seconds_total + - utilization_ratio + type: object + v1.RuntimeInstance: + properties: + allocation_id: + format: uuid + type: string + x-nullable: true + connection_generation: + format: uuid + type: string + x-nullable: true + device_id: + format: uuid + type: string + x-nullable: true + kind: + enum: + - managed_allocation + - self_hosted_connection + - none + type: string + required: + - allocation_id + - connection_generation + - device_id + - kind + type: object + v1.RuntimeMemoryObservation: + properties: + limit_bytes: + minimum: 1 + type: integer + x-nullable: true + usage_bytes: + minimum: 0 + type: integer + x-nullable: true + required: + - limit_bytes + - usage_bytes + type: object + v1.RuntimeObservation: + properties: + allocation_created_at: + minimum: 0 + type: integer + x-nullable: true + cpu: + allOf: + - $ref: '#/definitions/v1.RuntimeCPUObservation' + x-nullable: true + environment_id: + format: uuid + type: string + x-nullable: true + id: + format: uuid + type: string + instance: + $ref: '#/definitions/v1.RuntimeInstance' + memory: + allOf: + - $ref: '#/definitions/v1.RuntimeMemoryObservation' + x-nullable: true + mode: + enum: + - none + - self_hosted + - openai_hosted + type: string + object: + enum: + - agent.runtime_observation + type: string + observed_at: + minimum: 0 + type: integer + x-nullable: true + provider_type: + type: string + x-nullable: true + reason: + enum: + - runtime_mode_not_observable + - allocation_pending + - runtime_not_running + - source_not_configured + - sample_timeout + - sample_unavailable + type: string + x-nullable: true + resolved_at: + minimum: 0 + type: integer + session_id: + format: uuid + type: string + started_at: + minimum: 0 + type: integer + x-nullable: true + status: + enum: + - observed + - unsupported + - unavailable + type: string + required: + - allocation_created_at + - cpu + - environment_id + - id + - instance + - memory + - mode + - object + - observed_at + - provider_type + - reason + - resolved_at + - session_id + - started_at + - status + type: object + v1.RuntimeObservationList: + properties: + data: + items: + $ref: '#/definitions/v1.RuntimeObservation' + type: array + first_id: + format: uuid + type: string + x-nullable: true + has_more: + type: boolean + last_id: + format: uuid + type: string + x-nullable: true + object: + enum: + - list + type: string + required: + - data + - first_id + - has_more + - last_id + - object + type: object v1.SavedAgent: properties: created_at: @@ -2581,6 +2755,68 @@ paths: summary: Update an Environment Template tags: - Environment Templates + /agents/runtime-observations: + get: + description: Core extension listing one current Runtime context per tenant-owned + Session in Session creation order. Each row has an independent resolved_at + and optional provider observed_at; the page is not an atomic telemetry snapshot. + parameters: + - description: agents=v1 + in: header + name: OpenAI-Beta + required: true + type: string + - description: Last observation ID from the previous page + in: query + name: after + type: string + - default: 20 + description: Page size + in: query + maximum: 100 + minimum: 1 + name: limit + type: integer + - default: desc + description: Session creation order + enum: + - asc + - desc + in: query + name: order + type: string + produces: + - application/json + responses: + "200": + description: OK + schema: + $ref: '#/definitions/v1.RuntimeObservationList' + "400": + description: Bad Request + schema: + $ref: '#/definitions/v1.ErrorResponse' + "401": + description: Unauthorized + schema: + $ref: '#/definitions/v1.ErrorResponse' + "404": + description: Not Found + schema: + $ref: '#/definitions/v1.ErrorResponse' + "500": + description: Internal Server Error + schema: + $ref: '#/definitions/v1.ErrorResponse' + "503": + description: Service Unavailable + schema: + $ref: '#/definitions/v1.ErrorResponse' + security: + - BearerAuth: [] + summary: List current Runtime observations + tags: + - Runtime observations /agents/sessions: get: description: Cursor and results are scoped to the authenticated execution tenant. @@ -3348,6 +3584,53 @@ paths: summary: List persisted execution Items tags: - Items + /agents/sessions/{session_id}/runtime-observation: + get: + description: Core extension returning one tenant-scoped, read-only current Runtime + observation. It never provisions, renews, restarts, pauses or stops compute. + parameters: + - description: agents=v1 + in: header + name: OpenAI-Beta + required: true + type: string + - description: Session ID + in: path + name: session_id + required: true + type: string + produces: + - application/json + responses: + "200": + description: OK + schema: + $ref: '#/definitions/v1.RuntimeObservation' + "400": + description: Bad Request + schema: + $ref: '#/definitions/v1.ErrorResponse' + "401": + description: Unauthorized + schema: + $ref: '#/definitions/v1.ErrorResponse' + "404": + description: Not Found + schema: + $ref: '#/definitions/v1.ErrorResponse' + "500": + description: Internal Server Error + schema: + $ref: '#/definitions/v1.ErrorResponse' + "503": + description: Service Unavailable + schema: + $ref: '#/definitions/v1.ErrorResponse' + security: + - BearerAuth: [] + summary: Retrieve a Session Runtime observation + tags: + - Runtime observations /agents/sessions/{session_id}/subagents: get: description: Includes nested and closed Subagents. Cursors belong to the same diff --git a/contracts/agents-api/runtime-observability-api.md b/contracts/agents-api/runtime-observability-api.md index 4e607a9d7..34394ffd3 100644 --- a/contracts/agents-api/runtime-observability-api.md +++ b/contracts/agents-api/runtime-observability-api.md @@ -1,7 +1,9 @@ # Runtime observation API proposal -Status: review proposal. These routes are not implemented and are not yet present -in `openapi.yaml`. +Status: Phase 2 implemented. The current-snapshot routes, strict +`packages/agents-client` projection, and generated `openapi.yaml` contract are +implemented. Core Web integration, historical queries, and lifecycle controls +remain outside this phase. This is an Agents Core extension, not an upstream OpenAI Agents resource. The implementation must record that status in the coverage ledger and generated @@ -40,13 +42,13 @@ before publishing a new Dashboard snapshot. "id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", "object": "agent.runtime_observation", "session_id": "6c77d3a2-71d6-4ed5-884f-687aecda02a3", - "environment_id": "env_...", + "environment_id": "6c02fb71-5fa8-4298-93e8-57c6625a3fc2", "mode": "openai_hosted", "provider_type": "docker", "instance": { "kind": "managed_allocation", - "allocation_id": "alloc_...", - "device_id": "device_...", + "allocation_id": "d23ab94e-e40b-45bd-93a2-444f1f74642b", + "device_id": "2e434f4f-76aa-4e54-a707-4757036d90ef", "connection_generation": null }, "status": "observed", @@ -58,8 +60,8 @@ before publishing a new Dashboard snapshot. "cpu": { "usage_seconds_total": 482.75, "capacity_cores": 2.0, - "usage_cores": 1.42, - "utilization_ratio": 0.71 + "usage_cores": null, + "utilization_ratio": null }, "memory": { "usage_bytes": 805306368, @@ -181,7 +183,6 @@ Use the existing Agents API error envelope. | 400 | `invalid_request` | Empty or invalid limits, order, or malformed cursor. | | 401 | `authentication_error` | Missing or invalid API authentication. | | 404 | `not_found` | Missing or foreign Session/cursor, indistinguishably. | -| 429 | `rate_limit_exceeded` | Runtime sampling read budget exceeded. | | 500 | `internal_error` | Integrity, ownership, or invalid provider evidence. | | 503 | `execution_unavailable` | Required Runtime observation service is not configured. | @@ -190,14 +191,15 @@ Errors never include provider raw responses or credentials. ## Freshness and caching - Return `Cache-Control: no-store`. -- An internal cache may coalesce reads for at most five seconds. +- The Phase 2 implementation performs bounded direct reads and has no observation + cache. A later internal cache may coalesce reads for at most five seconds. - `observed_at` is authoritative for freshness; HTTP response time is not. - Clients mark samples stale according to their own explicit threshold. - `ETag` is not proposed because observations change independently. ## Client contract -`packages/agents-client` should expose: +`packages/agents-client` exposes: ```ts type RuntimeObservationStatus = "observed" | "unsupported" | "unavailable"; @@ -209,38 +211,13 @@ type RuntimeObservationReason = | "sample_timeout" | "sample_unavailable"; -interface RuntimeObservation { - id: string; - object: "agent.runtime_observation"; - session_id: string; - environment_id: string | null; - mode: "none" | "self_hosted" | "openai_hosted"; - provider_type: string | null; - instance: { - kind: "managed_allocation" | "self_hosted_connection" | "none"; - allocation_id: string | null; - device_id: string | null; - connection_generation: string | null; - }; - status: RuntimeObservationStatus; - reason: RuntimeObservationReason | null; - allocation_created_at: number | null; - resolved_at: number; - observed_at: number | null; - started_at: number | null; - cpu: { - usage_seconds_total: number | null; - capacity_cores: number | null; - usage_cores: number | null; - utilization_ratio: number | null; - } | null; - memory: { - usage_bytes: number | null; - limit_bytes: number | null; - } | null; -} +type RuntimeObservation = + | RuntimeObservedObservation + | RuntimeUnavailableObservation + | RuntimeNoneObservation + | RuntimeSelfHostedObservation; -interface RuntimeObservationPage { +interface RuntimeObservationList { object: "list"; data: RuntimeObservation[]; has_more: boolean; @@ -248,20 +225,33 @@ interface RuntimeObservationPage { last_id: string | null; } -interface RuntimeObservationClient { - list(options?: { +interface AgentCore { + listRuntimeObservations(options?: { after?: string; limit?: number; order?: "asc" | "desc"; - }): Promise; + }): Promise; - retrieveForSession(sessionId: string): Promise; + retrieveRuntimeObservation(sessionId: string): Promise; } ``` +These exported variants discriminate on `status` and `mode`; their instance, +reason, timestamps, CPU, and memory fields narrow accordingly. The exact variant +definitions live in `packages/agents-client/src/types.ts` and mirror the status +and reason matrix above. + The client validates every required field, enum, nullability rule, timestamp, and -finite number. Unknown additive fields are ignored. Malformed data rejects the -whole page; Web does not publish a partial snapshot. +finite number. The current pinned contract rejects unknown additive fields so an +unreviewed server expansion cannot silently cross the browser boundary. Malformed +data rejects the whole page; Web does not publish a partial snapshot. + +The generated OpenAPI 2 schema records field-level required/nullability rules, +UUID formats, reason enums, and numeric minima. OpenAPI 2 +cannot encode the complete cross-field discriminated union. The matrix above is +normative for wire consumers; the server projection and strict TypeScript +projector enforce it, and the exported TypeScript type prevents invalid +status/mode combinations in typed consumers. Web also applies a configured whole-refresh budget. If `has_more` remains true when that budget is exhausted, it retains the prior complete snapshot and marks diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md index 70ca2fe63..dd9dfa16d 100644 --- a/contracts/agents-api/runtime-observability-design.md +++ b/contracts/agents-api/runtime-observability-design.md @@ -1,8 +1,8 @@ # Runtime observability and Dashboard design -Status: review proposal. Phase 1 provider abstraction and Docker sampling are -implemented; the public API, Web integration, history backend, additional -providers, and lifecycle automation described below are not implemented. +Status: Phase 1 provider abstraction/Docker sampling and Phase 2 current-snapshot +API/client contract are implemented. Core Web integration, history backend, +additional providers, and lifecycle automation described below are not implemented. ## 1. Problem statement @@ -156,9 +156,11 @@ observation times. A single sample cannot truthfully supply CPU percentage. The API projection may additionally expose `usage_cores` and `utilization_ratio` only when the service has two ordered samples for the same Runtime incarnation. -The process-local observation cache keeps the previous cumulative value for this -calculation. Its loss makes the derived fields temporarily null; it never changes -the cumulative source measurement or lifecycle state. +A future process-local observation cache may keep the previous cumulative value +for this calculation. Phase 2 intentionally leaves both derived fields null +because it has only one provider sample per request. Cache loss must make the +derived fields temporarily null; it must never change the cumulative source +measurement or lifecycle state. ## 8. Duration semantics @@ -324,6 +326,8 @@ transitions. That migration cannot read a monitoring backend as authority. ### Phase 1: provider-neutral foundation +Implemented in the Docker observability foundation. + - `runtimeobs` identity, resolver, source, sample, and service. - Managed Docker Inspect/Stats source. - CPU, memory, and current compute start time. @@ -331,6 +335,10 @@ transitions. That migration cannot read a monitoring backend as authority. ### Phase 2: current snapshot API +Implemented by the Runtime Observation extension routes and +`packages/agents-client`. The generated OpenAPI contract records the extension; +this does not add an upstream OpenAI operation. + - Add extension types under `contracts/agents-api/v1`. - Add collection and Session-scoped handlers. - Add `packages/agents-client` methods and raw HTTP/client coverage. @@ -376,13 +384,12 @@ transitions. That migration cannot read a monitoring backend as authority. - No lifecycle action is reachable from the first Dashboard. - No new database table is required for current snapshots or history export. -## 17. Review decisions required +## 17. Recorded design decisions -1. Accept the proposed API as a documented Core extension rather than an upstream - OpenAI resource. -2. Accept current snapshots without atomic cross-row time semantics; every row +1. The API is a documented Core extension rather than an upstream OpenAI resource. +2. Current snapshots have no atomic cross-row time semantics; every row exposes its own `observed_at`. -3. Confirm that history is optional and external, not a PostgreSQL sample table. -4. Confirm that the first Web release has no lifecycle controls. -5. Choose whether `self_hosted` remains visibly unsupported until authenticated +3. History is optional and external, not a PostgreSQL sample table. +4. The first Web release has no lifecycle controls. +5. `self_hosted` remains visibly unsupported until authenticated, generation-fenced telemetry is qualified. diff --git a/contracts/agents-api/v1/runtime_observations.go b/contracts/agents-api/v1/runtime_observations.go new file mode 100644 index 000000000..d82d8d514 --- /dev/null +++ b/contracts/agents-api/v1/runtime_observations.go @@ -0,0 +1,46 @@ +package v1 + +type RuntimeObservation struct { + ID string `json:"id" binding:"required" format:"uuid"` + Object string `json:"object" enums:"agent.runtime_observation" binding:"required"` + SessionID string `json:"session_id" binding:"required" format:"uuid"` + EnvironmentID *string `json:"environment_id" extensions:"x-nullable" binding:"required" format:"uuid"` + Mode string `json:"mode" enums:"none,self_hosted,openai_hosted" binding:"required"` + ProviderType *string `json:"provider_type" extensions:"x-nullable" binding:"required" pattern:"^[a-z][a-z0-9_]{0,31}$"` + Instance RuntimeInstance `json:"instance" binding:"required"` + Status string `json:"status" enums:"observed,unsupported,unavailable" binding:"required"` + Reason *string `json:"reason" extensions:"x-nullable" binding:"required" enums:"runtime_mode_not_observable,allocation_pending,runtime_not_running,source_not_configured,sample_timeout,sample_unavailable"` + AllocationCreatedAt *int64 `json:"allocation_created_at" extensions:"x-nullable" binding:"required" minimum:"0"` + ResolvedAt int64 `json:"resolved_at" binding:"required" minimum:"0"` + ObservedAt *int64 `json:"observed_at" extensions:"x-nullable" binding:"required" minimum:"0"` + StartedAt *int64 `json:"started_at" extensions:"x-nullable" binding:"required" minimum:"0"` + CPU *RuntimeCPUObservation `json:"cpu" extensions:"x-nullable" binding:"required"` + Memory *RuntimeMemoryObservation `json:"memory" extensions:"x-nullable" binding:"required"` +} + +type RuntimeInstance struct { + Kind string `json:"kind" enums:"managed_allocation,self_hosted_connection,none" binding:"required"` + AllocationID *string `json:"allocation_id" extensions:"x-nullable" binding:"required" format:"uuid"` + DeviceID *string `json:"device_id" extensions:"x-nullable" binding:"required" format:"uuid"` + ConnectionGeneration *string `json:"connection_generation" extensions:"x-nullable" binding:"required" format:"uuid"` +} + +type RuntimeCPUObservation struct { + UsageSecondsTotal *float64 `json:"usage_seconds_total" extensions:"x-nullable" binding:"required" minimum:"0"` + CapacityCores *float64 `json:"capacity_cores" extensions:"x-nullable" binding:"required" minimum:"5e-324"` + UsageCores *float64 `json:"usage_cores" extensions:"x-nullable" binding:"required" minimum:"0"` + UtilizationRatio *float64 `json:"utilization_ratio" extensions:"x-nullable" binding:"required" minimum:"0"` +} + +type RuntimeMemoryObservation struct { + UsageBytes *uint64 `json:"usage_bytes" extensions:"x-nullable" binding:"required" minimum:"0"` + LimitBytes *uint64 `json:"limit_bytes" extensions:"x-nullable" binding:"required" minimum:"1"` +} + +type RuntimeObservationList struct { + Object string `json:"object" enums:"list" binding:"required"` + Data []RuntimeObservation `json:"data" binding:"required"` + HasMore bool `json:"has_more" binding:"required"` + FirstID *string `json:"first_id" extensions:"x-nullable" binding:"required" format:"uuid"` + LastID *string `json:"last_id" extensions:"x-nullable" binding:"required" format:"uuid"` +} diff --git a/packages/agents-client/src/client.test.ts b/packages/agents-client/src/client.test.ts index 13c0a237d..facd5e497 100644 --- a/packages/agents-client/src/client.test.ts +++ b/packages/agents-client/src/client.test.ts @@ -103,6 +103,42 @@ function messageItem(overrides: Record = {}): Record = {}): Record { + return { + id: runtimeSessionId, + object: "agent.runtime_observation", + session_id: runtimeSessionId, + environment_id: runtimeEnvironmentId, + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: runtimeAllocationId, + device_id: runtimeDeviceId, + connection_generation: null, + }, + status: "observed", + reason: null, + allocation_created_at: 10, + resolved_at: 30, + observed_at: 20, + started_at: 10, + cpu: { + usage_seconds_total: 0, + capacity_cores: 2, + usage_cores: null, + utilization_ratio: null, + }, + memory: { usage_bytes: 0, limit_bytes: 1024 }, + ...overrides, + }; +} + describe("OpenAIAgentsClient", () => { afterEach(() => vi.unstubAllGlobals()); @@ -2385,4 +2421,121 @@ describe("OpenAIAgentsClient", () => { ).rejects.toThrow("createSession only supports the JSON response"); expect(calls).toHaveLength(0); }); + + it("retrieves a Runtime observation, preserves observed zeroes, and encodes the Session ID", async () => { + const calls: FetchCall[] = []; + const client = new OpenAIAgentsClient({ + baseUrl: "https://core.example/v1", + fetch: recordingFetch(jsonResponse(runtimeObservation()), calls), + }); + + await expect(client.retrieveRuntimeObservation(runtimeSessionId)).resolves.toMatchObject({ + id: runtimeSessionId, + cpu: { usage_seconds_total: 0 }, + memory: { usage_bytes: 0 }, + }); + expect(String(calls[0]?.input)).toBe( + `https://core.example/v1/agents/sessions/${runtimeSessionId}/runtime-observation`, + ); + }); + + it("lists Runtime observations with stable pagination metadata and query serialization", async () => { + const calls: FetchCall[] = []; + const body = { + object: "list", + data: [runtimeObservation()], + has_more: true, + first_id: runtimeSessionId, + last_id: runtimeSessionId, + }; + const client = new OpenAIAgentsClient({ + baseUrl: "https://core.example/v1/", + fetch: recordingFetch(jsonResponse(body), calls), + }); + + await expect(client.listRuntimeObservations({ + after: runtimeSessionId, limit: 1, order: "asc", + })).resolves.toMatchObject(body); + expect(String(calls[0]?.input)).toBe( + `https://core.example/v1/agents/runtime-observations?after=${runtimeSessionId}&limit=1&order=asc`, + ); + }); + + it("accepts an unsupported none-mode Runtime observation with explicit nulls", async () => { + const value = runtimeObservation({ + environment_id: null, + mode: "none", + provider_type: null, + instance: { kind: "none", allocation_id: null, device_id: null, connection_generation: null }, + status: "unsupported", + reason: "runtime_mode_not_observable", + allocation_created_at: null, + observed_at: null, + started_at: null, + cpu: null, + memory: null, + }); + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse(value), []) }); + await expect(client.retrieveRuntimeObservation(runtimeSessionId)).resolves.toMatchObject(value); + }); + + it.each([ + ["unknown field", () => ({ ...runtimeObservation(), provider_native_id: "hidden" })], + ["foreign Session", () => ({ ...runtimeObservation(), session_id: "55555555-5555-4555-8555-555555555555" })], + ["invalid status/reason", () => ({ ...runtimeObservation(), status: "observed", reason: "sample_timeout" })], + ["invalid mode/instance", () => ({ ...runtimeObservation(), mode: "none" })], + ["negative CPU", () => ({ ...runtimeObservation(), cpu: { + usage_seconds_total: -1, capacity_cores: 2, usage_cores: null, utilization_ratio: null, + } })], + ["non-numeric CPU", () => ({ ...runtimeObservation(), cpu: { + usage_seconds_total: "NaN", capacity_cores: 2, usage_cores: null, utilization_ratio: null, + } })], + ["zero CPU capacity", () => ({ ...runtimeObservation(), cpu: { + usage_seconds_total: 1, capacity_cores: 0, usage_cores: null, utilization_ratio: null, + } })], + ["unsafe memory", () => ({ ...runtimeObservation(), memory: { + usage_bytes: Number.MAX_SAFE_INTEGER + 1, limit_bytes: 1024, + } })], + ["zero memory limit", () => ({ ...runtimeObservation(), memory: { + usage_bytes: 1, limit_bytes: 0, + } })], + ])("rejects a Runtime observation with %s", async (_label, build) => { + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse(build()), []) }); + await expect(client.retrieveRuntimeObservation(runtimeSessionId)).rejects.toMatchObject({ + status: 502, + code: "invalid_runtime_observation", + }); + }); + + it.each([ + ["mismatched first_id", { + object: "list", data: [runtimeObservation()], has_more: false, + first_id: runtimeEnvironmentId, last_id: runtimeSessionId, + }], + ["duplicate IDs", { + object: "list", data: [runtimeObservation(), runtimeObservation()], has_more: false, + first_id: runtimeSessionId, last_id: runtimeSessionId, + }], + ["empty continuation", { + object: "list", data: [], has_more: true, first_id: null, last_id: null, + }], + ])("rejects a Runtime observation list with %s", async (_label, body) => { + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse(body), []) }); + await expect(client.listRuntimeObservations()).rejects.toMatchObject({ + status: 502, + code: "invalid_runtime_observation", + }); + }); + + it.each([ + { after: "not-a-uuid" }, + { limit: 0 }, + { limit: 101 }, + { order: "sideways" }, + ])("rejects invalid Runtime observation pagination before fetch", async (options) => { + const calls: FetchCall[] = []; + const client = new OpenAIAgentsClient({ fetch: recordingFetch(jsonResponse({}), calls) }); + await expect(client.listRuntimeObservations(options as never)).rejects.toThrow(TypeError); + expect(calls).toHaveLength(0); + }); }); diff --git a/packages/agents-client/src/client.ts b/packages/agents-client/src/client.ts index c2f32630b..dc4393f57 100644 --- a/packages/agents-client/src/client.ts +++ b/packages/agents-client/src/client.ts @@ -43,6 +43,8 @@ import type { StreamError, UpdateAgentInput, ReplaceVaultCredentialTokenInput, + RuntimeObservation, + RuntimeObservationList, Vault, VaultCredential, VaultCredentialDeleted, @@ -234,6 +236,18 @@ const knownItemTypes = new Set([ ]); const itemStatuses = new Set(["in_progress", "completed", "failed", "incomplete"]); const turnStatuses = new Set(["queued", "in_progress", "waiting", "completed", "failed", "cancelled"]); +const runtimeObservationFields = new Set([ + "id", "object", "session_id", "environment_id", "mode", "provider_type", "instance", "status", "reason", + "allocation_created_at", "resolved_at", "observed_at", "started_at", "cpu", "memory", +]); +const runtimeInstanceFields = new Set(["kind", "allocation_id", "device_id", "connection_generation"]); +const runtimeCPUFields = new Set(["usage_seconds_total", "capacity_cores", "usage_cores", "utilization_ratio"]); +const runtimeMemoryFields = new Set(["usage_bytes", "limit_bytes"]); +const runtimeObservationReasons = new Set([ + "runtime_mode_not_observable", "allocation_pending", "runtime_not_running", + "source_not_configured", "sample_timeout", "sample_unavailable", +]); +const runtimeProviderTypePattern = /^[a-z][a-z0-9_]{0,31}$/; function exactFields(value: Record, fields: Set): boolean { const keys = Object.keys(value); @@ -995,6 +1009,164 @@ function projectAgentSession( return session; } +function invalidRuntimeObservation(message = "Agent Core returned an invalid Runtime observation."): never { + throw new AgentCoreError(message, 502, "invalid_runtime_observation"); +} + +function nullableRuntimeNumber(value: unknown): number | null { + if (value === null) return null; + if (typeof value !== "number" || !Number.isFinite(value) || value < 0) { + return invalidRuntimeObservation(); + } + return value; +} + +function nullableRuntimeInteger(value: unknown): number | null { + const projected = nullableRuntimeNumber(value); + if (projected !== null && !Number.isSafeInteger(projected)) return invalidRuntimeObservation(); + return projected; +} + +function projectRuntimeObservation(value: unknown, expectedSessionId?: string): RuntimeObservation { + if (!isRecord(value) || !exactFields(value, runtimeObservationFields)) { + return invalidRuntimeObservation(); + } + const id = canonicalUuid(value.id); + const sessionId = canonicalUuid(value.session_id); + const environmentId = value.environment_id === null ? null : canonicalUuid(value.environment_id); + if ( + id === null || sessionId === null || id !== sessionId || + (expectedSessionId !== undefined && !sameUuid(sessionId, expectedSessionId)) || + value.object !== "agent.runtime_observation" || + (value.mode !== "none" && value.mode !== "self_hosted" && value.mode !== "openai_hosted") || + !(value.provider_type === null || ( + typeof value.provider_type === "string" && runtimeProviderTypePattern.test(value.provider_type) + )) || + !isRecord(value.instance) || !exactFields(value.instance, runtimeInstanceFields) || + (value.status !== "observed" && value.status !== "unsupported" && value.status !== "unavailable") || + !(value.reason === null || ( + typeof value.reason === "string" && runtimeObservationReasons.has(value.reason) + )) || + !isNonnegativeInteger(value.resolved_at) + ) return invalidRuntimeObservation(); + + const allocationId = value.instance.allocation_id === null ? null : canonicalUuid(value.instance.allocation_id); + const deviceId = value.instance.device_id === null ? null : canonicalUuid(value.instance.device_id); + const connectionGeneration = value.instance.connection_generation === null + ? null + : canonicalUuid(value.instance.connection_generation); + if ( + (value.instance.allocation_id !== null && allocationId === null) || + (value.instance.device_id !== null && deviceId === null) || + (value.instance.connection_generation !== null && connectionGeneration === null) + ) return invalidRuntimeObservation(); + + const allocationCreatedAt = nullableRuntimeInteger(value.allocation_created_at); + const observedAt = nullableRuntimeInteger(value.observed_at); + const startedAt = nullableRuntimeInteger(value.started_at); + const isNone = value.mode === "none"; + const isSelfHosted = value.mode === "self_hosted"; + const isManaged = value.mode === "openai_hosted"; + if ( + (isNone && ( + value.instance.kind !== "none" || environmentId !== null || value.provider_type !== null || + allocationId !== null || deviceId !== null || connectionGeneration !== null || allocationCreatedAt !== null + )) || + (isSelfHosted && ( + value.instance.kind !== "self_hosted_connection" || environmentId === null || + allocationId !== null || allocationCreatedAt !== null + )) || + (isManaged && ( + value.instance.kind !== "managed_allocation" || environmentId === null || connectionGeneration !== null || + (allocationId === null && (deviceId !== null || allocationCreatedAt !== null)) + )) + ) return invalidRuntimeObservation(); + + const observed = value.status === "observed"; + if ( + (observed && ( + !isManaged || allocationId === null || value.reason !== null || observedAt === null || + observedAt > value.resolved_at + )) || + (!observed && ( + observedAt !== null || startedAt !== null || value.cpu !== null || value.memory !== null + )) || + (value.status === "unsupported" && ( + (!isNone && !isSelfHosted) || value.reason !== "runtime_mode_not_observable" + )) || + (value.status === "unavailable" && ( + !isManaged || value.reason === null || value.reason === "runtime_mode_not_observable" + )) || + (startedAt !== null && observedAt !== null && startedAt > observedAt) || + (allocationCreatedAt !== null && allocationCreatedAt > value.resolved_at) + ) return invalidRuntimeObservation(); + + let cpu: RuntimeObservation["cpu"] = null; + if (value.cpu !== null) { + if (!observed || !isRecord(value.cpu) || !exactFields(value.cpu, runtimeCPUFields)) { + return invalidRuntimeObservation(); + } + cpu = { + usage_seconds_total: nullableRuntimeNumber(value.cpu.usage_seconds_total), + capacity_cores: nullableRuntimeNumber(value.cpu.capacity_cores), + usage_cores: nullableRuntimeNumber(value.cpu.usage_cores), + utilization_ratio: nullableRuntimeNumber(value.cpu.utilization_ratio), + }; + if ( + Object.values(cpu).every((entry) => entry === null) || + (cpu.capacity_cores !== null && cpu.capacity_cores === 0) + ) return invalidRuntimeObservation(); + } + + let memory: RuntimeObservation["memory"] = null; + if (value.memory !== null) { + if (!observed || !isRecord(value.memory) || !exactFields(value.memory, runtimeMemoryFields)) { + return invalidRuntimeObservation(); + } + memory = { + usage_bytes: nullableRuntimeInteger(value.memory.usage_bytes), + limit_bytes: nullableRuntimeInteger(value.memory.limit_bytes), + }; + if ( + (memory.usage_bytes === null && memory.limit_bytes === null) || + memory.limit_bytes === 0 + ) return invalidRuntimeObservation(); + } + + return { + id, object: "agent.runtime_observation", session_id: sessionId, environment_id: environmentId, + mode: value.mode, provider_type: value.provider_type, instance: { + kind: value.instance.kind as RuntimeObservation["instance"]["kind"], + allocation_id: allocationId, device_id: deviceId, connection_generation: connectionGeneration, + }, + status: value.status, reason: value.reason as RuntimeObservation["reason"], + allocation_created_at: allocationCreatedAt, resolved_at: value.resolved_at, + observed_at: observedAt, started_at: startedAt, cpu, memory, + } as RuntimeObservation; +} + +function projectRuntimeObservationList(value: unknown, options?: PageOptions): RuntimeObservationList { + if ( + !isRecord(value) || !exactFields(value, vaultListFields) || value.object !== "list" || + !Array.isArray(value.data) || typeof value.has_more !== "boolean" + ) return invalidRuntimeObservation("Agent Core returned an invalid Runtime observation list."); + const limit = options?.limit ?? 20; + if ( + !Number.isSafeInteger(limit) || limit < 1 || limit > 100 || + (options?.order !== undefined && options.order !== "asc" && options.order !== "desc") || + value.data.length > limit + ) return invalidRuntimeObservation("Agent Core returned an invalid Runtime observation list."); + const data = value.data.map((entry) => projectRuntimeObservation(entry)); + const firstId = data[0]?.id ?? null; + const lastId = data[data.length - 1]?.id ?? null; + if ( + new Set(data.map((entry) => entry.id)).size !== data.length || + value.first_id !== firstId || value.last_id !== lastId || + (value.has_more && data.length === 0) + ) return invalidRuntimeObservation("Agent Core returned an invalid Runtime observation list."); + return { object: "list", data, has_more: value.has_more, first_id: firstId, last_id: lastId }; +} + function projectStreamError(value: unknown): StreamError { if ( !isRecord(value) || !exactFields(value, streamErrorFields) || @@ -2016,6 +2188,31 @@ export class OpenAIAgentsClient implements AgentCore { return { ...page, data: page.data.map((session) => projectAgentSession(session)) }; } + async listRuntimeObservations(options?: PageOptions): Promise { + if ( + (options?.after !== undefined && canonicalUuid(options.after) === null) || + (options?.limit !== undefined && ( + !Number.isSafeInteger(options.limit) || options.limit < 1 || options.limit > 100 + )) || + (options?.order !== undefined && options.order !== "asc" && options.order !== "desc") + ) throw new TypeError("Runtime observation pagination options are invalid."); + const params = new URLSearchParams(); + addPageOptions(params, options); + const value = await this.request( + withQuery("/agents/runtime-observations", params), + { signal: options?.signal }, + ); + return projectRuntimeObservationList(value, options); + } + + async retrieveRuntimeObservation(sessionId: string, options?: ReadOptions): Promise { + const value = await this.request( + `/agents/sessions/${encodeURIComponent(sessionId)}/runtime-observation`, + { signal: options?.signal }, + ); + return projectRuntimeObservation(value, sessionId); + } + async createSession(input: CreateSessionInput, idempotencyKey = createIdempotencyKey()): Promise { if ((input as { stream?: boolean }).stream === true) { throw new TypeError("createSession only supports the JSON response; connect streamEvents after creation."); diff --git a/packages/agents-client/src/protocol-types.test.ts b/packages/agents-client/src/protocol-types.test.ts index 3c1384129..ddf80f36e 100644 --- a/packages/agents-client/src/protocol-types.test.ts +++ b/packages/agents-client/src/protocol-types.test.ts @@ -29,6 +29,8 @@ import type { OpenAIHostedAgentEnvironmentInput, OpenAIHostedAgentEnvironmentResource, RequiredAction, + RuntimeObservation, + RuntimeUnavailableReason, SelfHostedAgentEnvironment, SavedAgentToolInput, SourceFile, @@ -48,6 +50,25 @@ import type { VaultCredential, } from "./types"; +describe("Runtime Observation discriminated contract", () => { + it("narrows status, reason, mode, instance, and sample presence together", () => { + type Observed = Extract; + type Unavailable = Extract; + type NoneMode = Extract; + type SelfHosted = Extract; + + expectTypeOf().toEqualTypeOf<"openai_hosted">(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf(); + expectTypeOf().toEqualTypeOf<"none">(); + expectTypeOf().toEqualTypeOf<"self_hosted_connection">(); + }); +}); + describe("Parsar dadf64a7 basic managed Environment profile", () => { it("pins omitted/default, explicit-enabled, and explicit-disabled network input", () => { const inputs = hostedDadf64.inputs as Record; diff --git a/packages/agents-client/src/types.ts b/packages/agents-client/src/types.ts index 74a156353..8bc3b0f6b 100644 --- a/packages/agents-client/src/types.ts +++ b/packages/agents-client/src/types.ts @@ -682,6 +682,119 @@ export interface CreateSessionStreamOptions extends StreamOptions { onSession: (session: AgentSession) => void; } +export type RuntimeObservationStatus = "observed" | "unsupported" | "unavailable"; +export type RuntimeObservationReason = + | "runtime_mode_not_observable" + | "allocation_pending" + | "runtime_not_running" + | "source_not_configured" + | "sample_timeout" + | "sample_unavailable"; + +export type RuntimeUnavailableReason = Exclude; + +export interface RuntimeCPUObservation { + usage_seconds_total: number | null; + capacity_cores: number | null; + usage_cores: number | null; + utilization_ratio: number | null; +} + +export interface RuntimeMemoryObservation { + usage_bytes: number | null; + limit_bytes: number | null; +} + +interface RuntimeObservationBase { + id: string; + object: "agent.runtime_observation"; + session_id: string; + resolved_at: number; +} + +export interface RuntimeObservedObservation extends RuntimeObservationBase { + environment_id: string; + mode: "openai_hosted"; + provider_type: string | null; + instance: { + kind: "managed_allocation"; + allocation_id: string; + device_id: string | null; + connection_generation: null; + }; + status: "observed"; + reason: null; + allocation_created_at: number | null; + observed_at: number; + started_at: number | null; + cpu: RuntimeCPUObservation | null; + memory: RuntimeMemoryObservation | null; +} + +export interface RuntimeUnavailableObservation extends RuntimeObservationBase { + environment_id: string; + mode: "openai_hosted"; + provider_type: string | null; + instance: { + kind: "managed_allocation"; + allocation_id: string | null; + device_id: string | null; + connection_generation: null; + }; + status: "unavailable"; + reason: RuntimeUnavailableReason; + allocation_created_at: number | null; + observed_at: null; + started_at: null; + cpu: null; + memory: null; +} + +export interface RuntimeNoneObservation extends RuntimeObservationBase { + environment_id: null; + mode: "none"; + provider_type: null; + instance: { kind: "none"; allocation_id: null; device_id: null; connection_generation: null }; + status: "unsupported"; + reason: "runtime_mode_not_observable"; + allocation_created_at: null; + observed_at: null; + started_at: null; + cpu: null; + memory: null; +} + +export interface RuntimeSelfHostedObservation extends RuntimeObservationBase { + environment_id: string; + mode: "self_hosted"; + provider_type: string | null; + instance: { + kind: "self_hosted_connection"; + allocation_id: null; + device_id: string | null; + connection_generation: string | null; + }; + status: "unsupported"; + reason: "runtime_mode_not_observable"; + allocation_created_at: null; + observed_at: null; + started_at: null; + cpu: null; + memory: null; +} + +export type RuntimeObservation = + | RuntimeObservedObservation + | RuntimeUnavailableObservation + | RuntimeNoneObservation + | RuntimeSelfHostedObservation; + +export interface RuntimeObservationList extends ListPage { + object: "list"; + first_id: string | null; + last_id: string | null; +} + export interface AgentCore { listAgents(options?: PageOptions): Promise>; createAgent(input: CreateAgentInput): Promise; @@ -698,6 +811,8 @@ export interface AgentCore { replaceVaultCredentialToken(vaultId: string, credentialId: string, input: ReplaceVaultCredentialTokenInput): Promise; deleteVaultCredential(vaultId: string, credentialId: string): Promise; listSessions(options?: PageOptions & { agentId?: string }): Promise>; + listRuntimeObservations(options?: PageOptions): Promise; + retrieveRuntimeObservation(sessionId: string, options?: ReadOptions): Promise; createSession(input: CreateSessionInput, idempotencyKey?: string): Promise; createSessionStream( input: Omit, diff --git a/services/agents-api/cmd/server/main.go b/services/agents-api/cmd/server/main.go index 757c2be80..d7c2daf88 100644 --- a/services/agents-api/cmd/server/main.go +++ b/services/agents-api/cmd/server/main.go @@ -28,6 +28,7 @@ import ( "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtime" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeenrollment" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" "github.com/jackc/pgx/v5/pgxpool" ) @@ -89,9 +90,27 @@ func run() error { if err := executionStore.EnsureProjectScopes(ready, auth.ProjectScopes()); err != nil { return err } + observationSources := map[string]runtimeobs.Source{} + if managed != nil { + for key, provider := range managed.Providers { + source, ok := provider.(runtimeobs.Source) + if !ok { + continue + } + observationSources[key] = source + } + } + resolver, err := runtimeobs.NewResolver(executionStore) + if err != nil { + return err + } + observationService, err := runtimeobs.NewService(resolver, observationSources) + if err != nil { + return err + } var workerDone chan error var worker *execution.Worker - options := []api.Option{api.WithSubagents(executionStore), api.WithSkills(executionStore), api.WithSourceFiles(executionStore), api.WithSessionArtifacts(executionStore)} + options := []api.Option{api.WithSubagents(executionStore), api.WithSkills(executionStore), api.WithSourceFiles(executionStore), api.WithSessionArtifacts(executionStore), api.WithRuntimeObservations(observationService)} var daemonHandler http.Handler var registry *gateway.Registry if wsURL := os.Getenv("AGENTS_API_DAEMON_WS_URL"); wsURL != "" { diff --git a/services/agents-api/internal/api/handler.go b/services/agents-api/internal/api/handler.go index 4fa78bdbb..a154843ef 100644 --- a/services/agents-api/internal/api/handler.go +++ b/services/agents-api/internal/api/handler.go @@ -35,20 +35,21 @@ type ResourceStore interface { } type Handler struct { - policy execution.Policy - store ResourceStore - auth *Authenticator - harnesses map[string]bool - engine string - inputs InputSubmitter - executorURL string - hostedEnvironments bool - directoryReader EnvironmentDirectoryReader - fileWriter EnvironmentFileWriter - skills SkillStore - sourceFiles SourceFileStore - artifacts SessionArtifactStore - subagents SubagentStore + policy execution.Policy + store ResourceStore + auth *Authenticator + harnesses map[string]bool + engine string + inputs InputSubmitter + executorURL string + hostedEnvironments bool + directoryReader EnvironmentDirectoryReader + fileWriter EnvironmentFileWriter + skills SkillStore + sourceFiles SourceFileStore + artifacts SessionArtifactStore + subagents SubagentStore + runtimeObservations RuntimeObservationService } func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ...Option) (http.Handler, error) { @@ -100,6 +101,8 @@ func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ... r.Post("/agents/sessions", h.createSession) r.Get("/agents/sessions", h.listSessions) r.Get("/agents/sessions/{session_id}", h.getSession) + r.Get("/agents/sessions/{session_id}/runtime-observation", h.getRuntimeObservation) + r.Get("/agents/runtime-observations", h.listRuntimeObservations) r.Post("/agents/sessions/{session_id}", h.updateSession) r.Delete("/agents/sessions/{session_id}", h.deleteSession) r.Post("/agents/sessions/{session_id}/events", h.createEvents) diff --git a/services/agents-api/internal/api/handler_test.go b/services/agents-api/internal/api/handler_test.go index dbf08bcf7..b1c3befe2 100644 --- a/services/agents-api/internal/api/handler_test.go +++ b/services/agents-api/internal/api/handler_test.go @@ -19,8 +19,19 @@ import ( type recordingStore struct { ResourceStore - tenant string - input store.CreateSessionInput + tenant string + input store.CreateSessionInput + sessions []store.Session + nextSessionCursor string + listTenant string + listAfter string + listLimit int + listAscending bool +} + +func (s *recordingStore) ListSessions(_ context.Context, tenant, after string, limit int, ascending bool, _ *string) (store.SessionPage, error) { + s.listTenant, s.listAfter, s.listLimit, s.listAscending = tenant, after, limit, ascending + return store.SessionPage{Sessions: append([]store.Session(nil), s.sessions...), NextCursor: s.nextSessionCursor}, nil } func (s *recordingStore) GetSession(ctx context.Context, tenant, id string) (store.Session, error) { diff --git a/services/agents-api/internal/api/runtime_observations.go b/services/agents-api/internal/api/runtime_observations.go new file mode 100644 index 000000000..54d061a2d --- /dev/null +++ b/services/agents-api/internal/api/runtime_observations.go @@ -0,0 +1,221 @@ +package api + +import ( + "context" + "errors" + "net/http" + "regexp" + "sync" + "time" + + v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/go-chi/chi/v5" +) + +const ( + runtimeObservationConcurrency = 8 + runtimeObservationSourceBudget = 2 * time.Second + runtimeObservationRequestBudget = 10 * time.Second +) + +var runtimeProviderTypePattern = regexp.MustCompile(`^[a-z][a-z0-9_]{0,31}$`) + +type RuntimeObservationService interface { + ObserveSession(context.Context, string, string) (runtimeobs.Observation, error) +} + +func WithRuntimeObservations(service RuntimeObservationService) Option { + return func(h *Handler) { h.runtimeObservations = service } +} + +// @Summary Retrieve a Session Runtime observation +// @Description Core extension returning one tenant-scoped, read-only current Runtime observation. It never provisions, renews, restarts, pauses or stops compute. +// @Tags Runtime observations +// @Produce json +// @Security BearerAuth +// @Param OpenAI-Beta header string true "agents=v1" +// @Param session_id path string true "Session ID" +// @Success 200 {object} v1.RuntimeObservation +// @Failure 400,401,404,500,503 {object} v1.ErrorResponse +// @Router /agents/sessions/{session_id}/runtime-observation [get] +func (h *Handler) getRuntimeObservation(w http.ResponseWriter, r *http.Request) { + if len(r.URL.Query()) != 0 { + writeError(w, http.StatusBadRequest, "unsupported_parameter", "Runtime observation retrieval does not accept query parameters.") + return + } + if h.runtimeObservations == nil { + writeError(w, http.StatusServiceUnavailable, "execution_unavailable", "Runtime observation is not configured on this service.") + return + } + ctx, cancel := context.WithTimeout(r.Context(), runtimeObservationSourceBudget) + defer cancel() + observation, err := h.runtimeObservations.ObserveSession(ctx, tenantID(r), chi.URLParam(r, "session_id")) + if err != nil { + writeStoreError(w, r, err) + return + } + response, err := runtimeObservationResponse(observation) + if err != nil { + writeStoreError(w, r, err) + return + } + writeJSON(w, http.StatusOK, response) +} + +// @Summary List current Runtime observations +// @Description Core extension listing one current Runtime context per tenant-owned Session in Session creation order. Each row has an independent resolved_at and optional provider observed_at; the page is not an atomic telemetry snapshot. +// @Tags Runtime observations +// @Produce json +// @Security BearerAuth +// @Param OpenAI-Beta header string true "agents=v1" +// @Param after query string false "Last observation ID from the previous page" +// @Param limit query int false "Page size" minimum(1) maximum(100) default(20) +// @Param order query string false "Session creation order" Enums(asc,desc) default(desc) +// @Success 200 {object} v1.RuntimeObservationList +// @Failure 400,401,404,500,503 {object} v1.ErrorResponse +// @Router /agents/runtime-observations [get] +func (h *Handler) listRuntimeObservations(w http.ResponseWriter, r *http.Request) { + if h.runtimeObservations == nil { + writeError(w, http.StatusServiceUnavailable, "execution_unavailable", "Runtime observation is not configured on this service.") + return + } + options, ok := readPage(w, r) + if !ok { + return + } + ctx, cancel := context.WithTimeout(r.Context(), runtimeObservationRequestBudget) + defer cancel() + page, err := h.store.ListSessions(ctx, tenantID(r), options.after, options.limit, options.ascending, nil) + if err != nil { + writeStoreError(w, r, err) + return + } + observations := make([]runtimeobs.Observation, len(page.Sessions)) + semaphore := make(chan struct{}, runtimeObservationConcurrency) + work, stop := context.WithCancel(ctx) + defer stop() + var wait sync.WaitGroup + var once sync.Once + var firstErr error + for index, session := range page.Sessions { + wait.Add(1) + go func(index int, sessionID string) { + defer wait.Done() + select { + case semaphore <- struct{}{}: + defer func() { <-semaphore }() + case <-work.Done(): + return + } + sampleCtx, sampleCancel := context.WithTimeout(work, runtimeObservationSourceBudget) + defer sampleCancel() + value, err := h.runtimeObservations.ObserveSession(sampleCtx, tenantID(r), sessionID) + if err != nil { + once.Do(func() { firstErr = err; stop() }) + return + } + observations[index] = value + }(index, session.ID) + } + wait.Wait() + if firstErr != nil { + writeStoreError(w, r, firstErr) + return + } + if err := ctx.Err(); err != nil { + writeError(w, http.StatusServiceUnavailable, "execution_unavailable", "Runtime observation collection exceeded its request budget.") + return + } + response := v1.RuntimeObservationList{Object: "list", Data: make([]v1.RuntimeObservation, 0, len(observations)), HasMore: page.NextCursor != ""} + for _, observation := range observations { + item, err := runtimeObservationResponse(observation) + if err != nil { + writeStoreError(w, r, err) + return + } + response.Data = append(response.Data, item) + } + if len(response.Data) > 0 { + response.FirstID = &response.Data[0].ID + response.LastID = &response.Data[len(response.Data)-1].ID + } + writeJSON(w, http.StatusOK, response) +} + +func runtimeObservationResponse(observation runtimeobs.Observation) (v1.RuntimeObservation, error) { + if observation.Target.SessionID == "" || observation.ResolvedAt.IsZero() || observation.ResolvedAt.Unix() < 0 { + return v1.RuntimeObservation{}, errors.New("invalid Runtime observation identity") + } + if !observation.Target.Instance.AllocationCreatedAt.IsZero() && + (observation.Target.Instance.AllocationCreatedAt.Unix() < 0 || observation.Target.Instance.AllocationCreatedAt.After(observation.ResolvedAt)) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime allocation creation time") + } + if observation.Sample != nil { + if observation.Sample.ObservedAt.IsZero() || observation.Sample.ObservedAt.Unix() < 0 || observation.Sample.ObservedAt.After(observation.ResolvedAt) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime sample time") + } + if observation.Sample.StartedAt != nil && + (observation.Sample.StartedAt.IsZero() || observation.Sample.StartedAt.Unix() < 0 || observation.Sample.StartedAt.After(observation.Sample.ObservedAt)) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime start time") + } + } + result := v1.RuntimeObservation{ + ID: observation.Target.SessionID, Object: "agent.runtime_observation", SessionID: observation.Target.SessionID, + Mode: string(observation.Target.Mode), Status: string(observation.Status), ResolvedAt: observation.ResolvedAt.Unix(), + } + if observation.Target.EnvironmentID != "" { + result.EnvironmentID = &observation.Target.EnvironmentID + } + if observation.ProviderType != "" { + if !runtimeProviderTypePattern.MatchString(observation.ProviderType) { + return v1.RuntimeObservation{}, errors.New("invalid Runtime observation provider type") + } + result.ProviderType = &observation.ProviderType + } + if observation.Reason != "" { + result.Reason = &observation.Reason + } + switch observation.Target.Mode { + case runtimeobs.ModeManaged: + result.Instance.Kind = "managed_allocation" + if observation.Target.Instance.AllocationID != "" { + result.Instance.AllocationID = &observation.Target.Instance.AllocationID + } + if observation.Target.Instance.DeviceID != "" { + result.Instance.DeviceID = &observation.Target.Instance.DeviceID + } + if !observation.Target.Instance.AllocationCreatedAt.IsZero() { + created := observation.Target.Instance.AllocationCreatedAt.Unix() + result.AllocationCreatedAt = &created + } + case runtimeobs.ModeSelfHosted: + result.Instance.Kind = "self_hosted_connection" + if observation.Target.Instance.DeviceID != "" { + result.Instance.DeviceID = &observation.Target.Instance.DeviceID + } + if observation.Target.Instance.ConnectionGeneration != "" { + result.Instance.ConnectionGeneration = &observation.Target.Instance.ConnectionGeneration + } + case runtimeobs.ModeNone: + result.Instance.Kind = "none" + default: + return v1.RuntimeObservation{}, errors.New("invalid Runtime observation mode") + } + if observation.Sample == nil { + return result, nil + } + observedAt := observation.Sample.ObservedAt.Unix() + result.ObservedAt = &observedAt + if observation.Sample.StartedAt != nil { + startedAt := observation.Sample.StartedAt.Unix() + result.StartedAt = &startedAt + } + if observation.Sample.CPUUsageSecondsTotal != nil || observation.Sample.CPUCapacityCores != nil { + result.CPU = &v1.RuntimeCPUObservation{UsageSecondsTotal: observation.Sample.CPUUsageSecondsTotal, CapacityCores: observation.Sample.CPUCapacityCores} + } + if observation.Sample.MemoryUsageBytes != nil || observation.Sample.MemoryLimitBytes != nil { + result.Memory = &v1.RuntimeMemoryObservation{UsageBytes: observation.Sample.MemoryUsageBytes, LimitBytes: observation.Sample.MemoryLimitBytes} + } + return result, nil +} diff --git a/services/agents-api/internal/api/runtime_observations_test.go b/services/agents-api/internal/api/runtime_observations_test.go new file mode 100644 index 000000000..29768be94 --- /dev/null +++ b/services/agents-api/internal/api/runtime_observations_test.go @@ -0,0 +1,224 @@ +package api + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + "time" + + v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/runtimeobs" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/google/uuid" +) + +type runtimeObservationFixture struct { + values map[string]runtimeobs.Observation +} + +func (f runtimeObservationFixture) ObserveSession(_ context.Context, tenant, session string) (runtimeobs.Observation, error) { + value := f.values[session] + value.Target.TenantID = tenant + return value, nil +} + +type runtimeObservationServiceFunc func(context.Context, string, string) (runtimeobs.Observation, error) + +func (f runtimeObservationServiceFunc) ObserveSession(ctx context.Context, tenant, session string) (runtimeobs.Observation, error) { + return f(ctx, tenant, session) +} + +func runtimeObservationRequest(handler http.Handler, path string) *httptest.ResponseRecorder { + request := httptest.NewRequest(http.MethodGet, path, nil) + request.Header.Set("Authorization", "Bearer test-api-key") + request.Header.Set("OpenAI-Beta", "agents=v1") + response := httptest.NewRecorder() + handler.ServeHTTP(response, request) + return response +} + +func TestRuntimeObservationRoutesUseSessionIdentityAndExactNullability(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + sessionID := uuid.NewString() + service := runtimeObservationFixture{values: map[string]runtimeobs.Observation{ + sessionID: {Target: runtimeobs.Target{SessionID: sessionID, Mode: runtimeobs.ModeNone}, Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now}, + }} + handler, saved, _ := testHandler(t, WithRuntimeObservations(service)) + saved.sessions = []store.Session{{ID: sessionID, CreatedAt: now}} + + for _, path := range []string{"/v1/agents/sessions/" + sessionID + "/runtime-observation", "/v1/agents/runtime-observations?limit=1"} { + response := runtimeObservationRequest(handler, path) + if response.Code != http.StatusOK { + t.Fatalf("%s returned %d: %s", path, response.Code, response.Body) + } + if path[len(path)-7:] == "limit=1" { + var page v1.RuntimeObservationList + if json.Unmarshal(response.Body.Bytes(), &page) != nil || len(page.Data) != 1 || page.FirstID == nil || *page.FirstID != sessionID || page.LastID == nil || *page.LastID != sessionID { + t.Fatalf("invalid observation page: %s", response.Body) + } + continue + } + var value v1.RuntimeObservation + if json.Unmarshal(response.Body.Bytes(), &value) != nil || value.ID != sessionID || value.SessionID != sessionID || value.Instance.Kind != "none" || value.EnvironmentID != nil || value.ProviderType != nil || value.ObservedAt != nil || value.CPU != nil || value.Memory != nil || value.Reason == nil || *value.Reason != "runtime_mode_not_observable" { + t.Fatalf("invalid unsupported observation: %s", response.Body) + } + } +} + +func TestRuntimeObservationRoutesRequireConfiguredServiceAndRejectQueries(t *testing.T) { + handler, _, _ := testHandler(t) + missing := runtimeObservationRequest(handler, "/v1/agents/runtime-observations") + if missing.Code != http.StatusServiceUnavailable { + t.Fatalf("unconfigured service returned %d: %s", missing.Code, missing.Body) + } + + sessionID := uuid.NewString() + now := time.Now().UTC() + service := runtimeObservationFixture{values: map[string]runtimeobs.Observation{ + sessionID: {Target: runtimeobs.Target{SessionID: sessionID, Mode: runtimeobs.ModeNone}, Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now}, + }} + handler, _, _ = testHandler(t, WithRuntimeObservations(service)) + invalid := runtimeObservationRequest(handler, "/v1/agents/sessions/"+sessionID+"/runtime-observation?provider=docker") + if invalid.Code != http.StatusBadRequest { + t.Fatalf("unsupported query returned %d: %s", invalid.Code, invalid.Body) + } +} + +func TestRuntimeObservationResponsePreservesObservedZero(t *testing.T) { + zeroCPU := float64(0) + zeroMemory := uint64(0) + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + sessionID, environmentID := uuid.NewString(), uuid.NewString() + value, err := runtimeObservationResponse(runtimeobs.Observation{ + Target: runtimeobs.Target{SessionID: sessionID, EnvironmentID: environmentID, Mode: runtimeobs.ModeManaged, Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), DeviceID: uuid.NewString(), AllocationCreatedAt: now.Add(-time.Hour)}}, + Status: runtimeobs.StatusObserved, ProviderType: "docker", ResolvedAt: now, + Sample: &runtimeobs.Sample{ObservedAt: now, CPUUsageSecondsTotal: &zeroCPU, MemoryUsageBytes: &zeroMemory}, + }) + if err != nil || value.CPU == nil || value.CPU.UsageSecondsTotal == nil || *value.CPU.UsageSecondsTotal != 0 || value.Memory == nil || value.Memory.UsageBytes == nil || *value.Memory.UsageBytes != 0 { + t.Fatalf("observed zero was lost: %+v %v", value, err) + } +} + +func TestRuntimeObservationResponseRejectsTimesOutsidePublicContract(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + preEpoch := time.Unix(-1, 0).UTC() + base := runtimeobs.Observation{ + Target: runtimeobs.Target{ + SessionID: uuid.NewString(), EnvironmentID: uuid.NewString(), Mode: runtimeobs.ModeManaged, + Instance: runtimeobs.Instance{AllocationID: uuid.NewString(), DeviceID: uuid.NewString(), AllocationCreatedAt: now.Add(-time.Hour)}, + }, + Status: runtimeobs.StatusObserved, ResolvedAt: now, + Sample: &runtimeobs.Sample{ObservedAt: now, StartedAt: timePointer(now.Add(-time.Minute))}, + } + for _, mutate := range []func(*runtimeobs.Observation){ + func(value *runtimeobs.Observation) { value.Target.Instance.AllocationCreatedAt = now.Add(time.Second) }, + func(value *runtimeobs.Observation) { value.Target.Instance.AllocationCreatedAt = preEpoch }, + func(value *runtimeobs.Observation) { value.Sample.ObservedAt = preEpoch }, + func(value *runtimeobs.Observation) { value.Sample.StartedAt = &preEpoch }, + } { + observation := base + sample := *base.Sample + observation.Sample = &sample + mutate(&observation) + if _, err := runtimeObservationResponse(observation); err == nil { + t.Fatalf("invalid Runtime time accepted: %+v", observation) + } + } +} + +func timePointer(value time.Time) *time.Time { return &value } + +func TestRuntimeObservationListPreservesStoreOrderAndTenantPagination(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + sessionIDs := []string{uuid.NewString(), uuid.NewString(), uuid.NewString()} + var expectedTenant string + service := runtimeObservationServiceFunc(func(_ context.Context, tenant, session string) (runtimeobs.Observation, error) { + if tenant != expectedTenant { + return runtimeobs.Observation{}, errors.New("unexpected tenant") + } + if session == sessionIDs[0] { + time.Sleep(20 * time.Millisecond) + } + return runtimeobs.Observation{ + Target: runtimeobs.Target{SessionID: session, Mode: runtimeobs.ModeNone}, + Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now, + }, nil + }) + handler, saved, tenant := testHandler(t, WithRuntimeObservations(service)) + expectedTenant = tenant + for _, id := range sessionIDs { + saved.sessions = append(saved.sessions, store.Session{ID: id, CreatedAt: now}) + } + saved.nextSessionCursor = "next" + + response := runtimeObservationRequest(handler, "/v1/agents/runtime-observations?after=cursor&limit=3&order=asc") + if response.Code != http.StatusOK { + t.Fatalf("list returned %d: %s", response.Code, response.Body) + } + var page v1.RuntimeObservationList + if err := json.Unmarshal(response.Body.Bytes(), &page); err != nil { + t.Fatal(err) + } + if len(page.Data) != len(sessionIDs) || !page.HasMore || saved.listTenant != tenant || saved.listAfter != "cursor" || saved.listLimit != 3 || !saved.listAscending { + t.Fatalf("pagination binding was not preserved: page=%+v store=%+v", page, saved) + } + for index, item := range page.Data { + if item.ID != sessionIDs[index] { + t.Fatalf("concurrent collection reordered page: %+v", page.Data) + } + } +} + +func TestRuntimeObservationListBoundsCollectionConcurrency(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + var active, maximum atomic.Int32 + service := runtimeObservationServiceFunc(func(_ context.Context, _, session string) (runtimeobs.Observation, error) { + current := active.Add(1) + defer active.Add(-1) + for current > maximum.Load() && !maximum.CompareAndSwap(maximum.Load(), current) { + } + time.Sleep(15 * time.Millisecond) + return runtimeobs.Observation{ + Target: runtimeobs.Target{SessionID: session, Mode: runtimeobs.ModeNone}, + Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now, + }, nil + }) + handler, saved, _ := testHandler(t, WithRuntimeObservations(service)) + for range 20 { + saved.sessions = append(saved.sessions, store.Session{ID: uuid.NewString(), CreatedAt: now}) + } + response := runtimeObservationRequest(handler, "/v1/agents/runtime-observations?limit=20") + if response.Code != http.StatusOK { + t.Fatalf("list returned %d: %s", response.Code, response.Body) + } + if got := maximum.Load(); got == 0 || got > runtimeObservationConcurrency { + t.Fatalf("collection concurrency = %d, want 1..%d", got, runtimeObservationConcurrency) + } +} + +func TestRuntimeObservationListRejectsWholePageOnIntegrityFailure(t *testing.T) { + now := time.Date(2026, 9, 22, 8, 0, 0, 0, time.UTC) + validID, invalidID := uuid.NewString(), uuid.NewString() + service := runtimeObservationFixture{values: map[string]runtimeobs.Observation{ + validID: { + Target: runtimeobs.Target{SessionID: validID, Mode: runtimeobs.ModeNone}, + Status: runtimeobs.StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: now, + }, + invalidID: {Target: runtimeobs.Target{Mode: runtimeobs.ModeNone}, Status: runtimeobs.StatusUnsupported, ResolvedAt: now}, + }} + handler, saved, _ := testHandler(t, WithRuntimeObservations(service)) + saved.sessions = []store.Session{{ID: validID, CreatedAt: now}, {ID: invalidID, CreatedAt: now}} + + response := runtimeObservationRequest(handler, "/v1/agents/runtime-observations?limit=2") + if response.Code != http.StatusInternalServerError { + t.Fatalf("integrity failure returned %d: %s", response.Code, response.Body) + } + var envelope v1.ErrorResponse + if err := json.Unmarshal(response.Body.Bytes(), &envelope); err != nil || envelope.Error.Code == "" { + t.Fatalf("integrity failure leaked a partial page: %s", response.Body) + } +} diff --git a/services/agents-api/internal/runtimeobs/identity.go b/services/agents-api/internal/runtimeobs/identity.go index 7e33f4bb9..17e38f7d4 100644 --- a/services/agents-api/internal/runtimeobs/identity.go +++ b/services/agents-api/internal/runtimeobs/identity.go @@ -1,5 +1,7 @@ package runtimeobs +import "time" + // Instance is one provider-owned Runtime incarnation. AllocationID is present // for managed compute. DeviceID and ConnectionGeneration are reserved for a // future authenticated self-hosted telemetry source. @@ -8,6 +10,8 @@ type Instance struct { ProviderKey string DeviceID string ConnectionGeneration string + AllocationState string + AllocationCreatedAt time.Time } // Target binds telemetry to durable Core identity. A Session is not itself a diff --git a/services/agents-api/internal/runtimeobs/resolver.go b/services/agents-api/internal/runtimeobs/resolver.go index 7b1047c0c..58a5f28df 100644 --- a/services/agents-api/internal/runtimeobs/resolver.go +++ b/services/agents-api/internal/runtimeobs/resolver.go @@ -9,7 +9,10 @@ import ( "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" ) -var ErrUnavailable = errors.New("Runtime observation unavailable") +var ( + ErrUnavailable = errors.New("Runtime observation unavailable") + ErrNotRunning = errors.New("Runtime is not running") +) type sessionStore interface { GetSession(context.Context, string, string) (store.Session, error) @@ -49,12 +52,18 @@ func (r *Resolver) Resolve(ctx context.Context, tenantID, sessionID string) (Tar if session.Environment == nil { return Target{}, errors.New("self-hosted Session is missing its Environment") } + if session.Environment.TenantID != session.TenantID || session.Environment.SessionID != session.ID { + return Target{}, errors.New("self-hosted Environment does not match resolved ownership") + } target.EnvironmentID = session.Environment.ID return target, nil case ModeManaged: if session.Environment == nil { return Target{}, errors.New("managed Session is missing its Environment") } + if session.Environment.TenantID != session.TenantID || session.Environment.SessionID != session.ID { + return Target{}, errors.New("managed Environment does not match resolved ownership") + } target.EnvironmentID = session.Environment.ID allocation, err := r.store.GetRuntimeAllocation(ctx, tenantID, target.EnvironmentID) if errors.Is(err, store.ErrNotFound) { @@ -63,7 +72,13 @@ func (r *Resolver) Resolve(ctx context.Context, tenantID, sessionID string) (Tar if err != nil { return Target{}, fmt.Errorf("resolve Runtime allocation: %w", err) } - target.Instance = Instance{AllocationID: allocation.ID, ProviderKey: allocation.ProviderKey, DeviceID: allocation.DeviceID} + if allocation.TenantID != tenantID || allocation.SessionID != session.ID || allocation.EnvironmentID != target.EnvironmentID { + return Target{}, errors.New("Runtime allocation does not match resolved ownership") + } + target.Instance = Instance{ + AllocationID: allocation.ID, ProviderKey: allocation.ProviderKey, DeviceID: allocation.DeviceID, + AllocationState: allocation.State, AllocationCreatedAt: allocation.CreatedAt, + } return target, nil default: return Target{}, errors.New("invalid stored Runtime environment type") diff --git a/services/agents-api/internal/runtimeobs/resolver_test.go b/services/agents-api/internal/runtimeobs/resolver_test.go index 738c79f2c..7edb4ba48 100644 --- a/services/agents-api/internal/runtimeobs/resolver_test.go +++ b/services/agents-api/internal/runtimeobs/resolver_test.go @@ -24,8 +24,11 @@ func (s resolverStore) GetRuntimeAllocation(context.Context, string, string) (st func TestResolverBindsManagedSessionEnvironmentAndAllocation(t *testing.T) { r, err := NewResolver(resolverStore{ - session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment"}}, - allocation: store.RuntimeAllocation{ID: "allocation", ProviderKey: "provider", DeviceID: "device"}, + session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}}, + allocation: store.RuntimeAllocation{ + ID: "allocation", TenantID: "tenant", SessionID: "session", EnvironmentID: "environment", + ProviderKey: "provider", DeviceID: "device", + }, }) if err != nil { t.Fatal(err) @@ -45,7 +48,7 @@ func TestResolverKeepsUnsupportedModesDistinct(t *testing.T) { environment *store.Environment }{ {mode: "none"}, - {mode: "self_hosted", environment: &store.Environment{ID: "environment"}}, + {mode: "self_hosted", environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}}, } { r, err := NewResolver(resolverStore{session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"` + tc.mode + `"}}`), Environment: tc.environment}}) if err != nil { @@ -60,7 +63,7 @@ func TestResolverKeepsUnsupportedModesDistinct(t *testing.T) { func TestResolverReportsManagedAllocationAsUnavailable(t *testing.T) { r, err := NewResolver(resolverStore{ - session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment"}}, + session: store.Session{ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), Environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}}, allocationErr: store.ErrNotFound, }) if err != nil { @@ -71,3 +74,51 @@ func TestResolverReportsManagedAllocationAsUnavailable(t *testing.T) { t.Fatalf("allocation absence was not preserved: %+v %v", target, err) } } + +func TestResolverRejectsMismatchedEnvironmentOwnership(t *testing.T) { + for _, mode := range []string{"self_hosted", "openai_hosted"} { + for _, environment := range []store.Environment{ + {ID: "environment", TenantID: "other", SessionID: "session"}, + {ID: "environment", TenantID: "tenant", SessionID: "other"}, + } { + resolver, err := NewResolver(resolverStore{session: store.Session{ + ID: "session", TenantID: "tenant", + Configuration: []byte(`{"environment":{"type":"` + mode + `"}}`), Environment: &environment, + }}) + if err != nil { + t.Fatal(err) + } + if _, err := resolver.Resolve(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("mismatched %s Environment accepted: %+v", mode, environment) + } + } + } +} + +func TestResolverRejectsMismatchedAllocationOwnership(t *testing.T) { + base := store.RuntimeAllocation{ + ID: "allocation", TenantID: "tenant", SessionID: "session", EnvironmentID: "environment", + ProviderKey: "provider", DeviceID: "device", + } + for _, mutate := range []func(*store.RuntimeAllocation){ + func(value *store.RuntimeAllocation) { value.TenantID = "other" }, + func(value *store.RuntimeAllocation) { value.SessionID = "other" }, + func(value *store.RuntimeAllocation) { value.EnvironmentID = "other" }, + } { + allocation := base + mutate(&allocation) + resolver, err := NewResolver(resolverStore{ + session: store.Session{ + ID: "session", TenantID: "tenant", Configuration: []byte(`{"environment":{"type":"openai_hosted"}}`), + Environment: &store.Environment{ID: "environment", TenantID: "tenant", SessionID: "session"}, + }, + allocation: allocation, + }) + if err != nil { + t.Fatal(err) + } + if _, err := resolver.Resolve(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("mismatched allocation accepted: %+v", allocation) + } + } +} diff --git a/services/agents-api/internal/runtimeobs/sample.go b/services/agents-api/internal/runtimeobs/sample.go index bf653c174..5b36e81f1 100644 --- a/services/agents-api/internal/runtimeobs/sample.go +++ b/services/agents-api/internal/runtimeobs/sample.go @@ -4,6 +4,7 @@ package runtimeobs import ( "errors" + "math" "time" ) @@ -36,19 +37,23 @@ type Sample struct { } func (s Sample) validate(now time.Time) error { - if s.ObservedAt.IsZero() || s.ObservedAt.After(now) { + if s.ObservedAt.IsZero() || s.ObservedAt.Unix() < 0 || s.ObservedAt.After(now) { return errors.New("invalid Runtime observation time") } - if s.StartedAt != nil && (s.StartedAt.IsZero() || s.StartedAt.After(s.ObservedAt)) { + if s.StartedAt != nil && (s.StartedAt.IsZero() || s.StartedAt.Unix() < 0 || s.StartedAt.After(s.ObservedAt)) { return errors.New("invalid Runtime start time") } - if s.CPUUsageSecondsTotal != nil && *s.CPUUsageSecondsTotal < 0 { + if s.CPUUsageSecondsTotal != nil && (*s.CPUUsageSecondsTotal < 0 || math.IsNaN(*s.CPUUsageSecondsTotal) || math.IsInf(*s.CPUUsageSecondsTotal, 0)) { return errors.New("invalid Runtime CPU usage") } - if s.CPUCapacityCores != nil && *s.CPUCapacityCores <= 0 { + if s.CPUCapacityCores != nil && (*s.CPUCapacityCores <= 0 || math.IsNaN(*s.CPUCapacityCores) || math.IsInf(*s.CPUCapacityCores, 0)) { return errors.New("invalid Runtime CPU capacity") } - if s.MemoryLimitBytes != nil && *s.MemoryLimitBytes == 0 { + const maxSafeJSONInteger = uint64(1<<53 - 1) + if s.MemoryUsageBytes != nil && *s.MemoryUsageBytes > maxSafeJSONInteger { + return errors.New("Runtime memory usage exceeds the public JSON integer range") + } + if s.MemoryLimitBytes != nil && (*s.MemoryLimitBytes == 0 || *s.MemoryLimitBytes > maxSafeJSONInteger) { return errors.New("invalid Runtime memory limit") } return nil diff --git a/services/agents-api/internal/runtimeobs/service.go b/services/agents-api/internal/runtimeobs/service.go index e05f7c678..e56d17348 100644 --- a/services/agents-api/internal/runtimeobs/service.go +++ b/services/agents-api/internal/runtimeobs/service.go @@ -8,10 +8,12 @@ import ( ) type Observation struct { - Target Target - Status Status - Sample *Sample - Reason string + Target Target + Status Status + Sample *Sample + Reason string + ProviderType string + ResolvedAt time.Time } type Service struct { @@ -36,25 +38,59 @@ func NewService(resolver TargetResolver, sources map[string]Source) (*Service, e func (s *Service) ObserveSession(ctx context.Context, tenantID, sessionID string) (Observation, error) { target, err := s.resolver.Resolve(ctx, tenantID, sessionID) + resolvedAt := s.now() if errors.Is(err, ErrUnavailable) { - return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_allocation_unavailable"}, nil + if target.TenantID != tenantID || target.SessionID != sessionID || target.Mode != ModeManaged || target.EnvironmentID == "" { + return Observation{}, errors.New("Runtime observation resolver returned invalid pending allocation identity") + } + return Observation{Target: target, Status: StatusUnavailable, Reason: "allocation_pending", ResolvedAt: resolvedAt}, nil } if err != nil { return Observation{}, err } + if target.TenantID != tenantID || target.SessionID != sessionID { + return Observation{}, errors.New("Runtime observation resolver returned mismatched ownership") + } + if (target.Mode == ModeNone && target.EnvironmentID != "") || + ((target.Mode == ModeSelfHosted || target.Mode == ModeManaged) && target.EnvironmentID == "") { + return Observation{}, errors.New("Runtime observation resolver returned mismatched Environment identity") + } if target.Mode == ModeNone || target.Mode == ModeSelfHosted { - return Observation{Target: target, Status: StatusUnsupported, Reason: "runtime_mode_not_observable"}, nil + return Observation{Target: target, Status: StatusUnsupported, Reason: "runtime_mode_not_observable", ResolvedAt: resolvedAt}, nil } if target.Mode != ModeManaged || target.Instance.AllocationID == "" || target.Instance.ProviderKey == "" { return Observation{}, errors.New("invalid managed Runtime observation target") } + if !target.Instance.AllocationCreatedAt.IsZero() && + (target.Instance.AllocationCreatedAt.Unix() < 0 || target.Instance.AllocationCreatedAt.After(resolvedAt)) { + return Observation{}, errors.New("invalid managed Runtime allocation creation time") + } + switch target.Instance.AllocationState { + case "creating": + return Observation{Target: target, Status: StatusUnavailable, Reason: "allocation_pending", ResolvedAt: resolvedAt}, nil + case "cleanup_pending", "released": + return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_not_running", ResolvedAt: resolvedAt}, nil + case "running": + default: + return Observation{}, errors.New("invalid managed Runtime allocation state") + } source, ok := s.sources[target.Instance.ProviderKey] if !ok { - return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_source_unavailable"}, nil + return Observation{Target: target, Status: StatusUnavailable, Reason: "source_not_configured", ResolvedAt: resolvedAt}, nil + } + providerType := "" + if typed, ok := source.(interface{ ObservationProviderType() string }); ok { + providerType = typed.ObservationProviderType() } sample, err := source.Observe(ctx, target) + if errors.Is(err, context.DeadlineExceeded) { + return Observation{Target: target, Status: StatusUnavailable, Reason: "sample_timeout", ProviderType: providerType, ResolvedAt: s.now()}, nil + } + if errors.Is(err, ErrNotRunning) { + return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_not_running", ProviderType: providerType, ResolvedAt: s.now()}, nil + } if errors.Is(err, ErrUnavailable) { - return Observation{Target: target, Status: StatusUnavailable, Reason: "runtime_sample_unavailable"}, nil + return Observation{Target: target, Status: StatusUnavailable, Reason: "sample_unavailable", ProviderType: providerType, ResolvedAt: s.now()}, nil } if err != nil { return Observation{}, fmt.Errorf("observe Runtime: %w", err) @@ -62,5 +98,5 @@ func (s *Service) ObserveSession(ctx context.Context, tenantID, sessionID string if err := sample.validate(s.now()); err != nil { return Observation{}, err } - return Observation{Target: target, Status: StatusObserved, Sample: &sample}, nil + return Observation{Target: target, Status: StatusObserved, Sample: &sample, ProviderType: providerType, ResolvedAt: s.now()}, nil } diff --git a/services/agents-api/internal/runtimeobs/service_test.go b/services/agents-api/internal/runtimeobs/service_test.go index bff6bf33d..6cb76682f 100644 --- a/services/agents-api/internal/runtimeobs/service_test.go +++ b/services/agents-api/internal/runtimeobs/service_test.go @@ -3,6 +3,7 @@ package runtimeobs import ( "context" "errors" + "math" "testing" "time" ) @@ -12,8 +13,15 @@ type fixedResolver struct { err error } -func (r fixedResolver) Resolve(context.Context, string, string) (Target, error) { - return r.target, r.err +func (r fixedResolver) Resolve(_ context.Context, tenant, session string) (Target, error) { + target := r.target + if target.TenantID == "" { + target.TenantID = tenant + } + if target.SessionID == "" { + target.SessionID = session + } + return target, r.err } type fixedSource struct { @@ -27,28 +35,48 @@ func (s *fixedSource) Observe(context.Context, Target) (Sample, error) { return s.sample, s.err } +type typedSource struct { + *fixedSource + providerType string +} + +func (s typedSource) ObservationProviderType() string { return s.providerType } + +type blockingSource struct{} + +func (blockingSource) Observe(ctx context.Context, _ Target) (Sample, error) { + <-ctx.Done() + return Sample{}, ctx.Err() +} + +func (blockingSource) ObservationProviderType() string { return "docker" } + func TestServiceDoesNotCallSourcesForUnsupportedModes(t *testing.T) { for _, mode := range []Mode{ModeNone, ModeSelfHosted} { source := &fixedSource{} - service, err := NewService(fixedResolver{target: Target{Mode: mode}}, map[string]Source{"provider": source}) + target := Target{Mode: mode} + if mode == ModeSelfHosted { + target.EnvironmentID = "environment" + } + service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) if err != nil { t.Fatal(err) } observation, err := service.ObserveSession(t.Context(), "tenant", "session") - if err != nil || observation.Status != StatusUnsupported || source.calls != 0 { + if err != nil || observation.Status != StatusUnsupported || observation.Reason != "runtime_mode_not_observable" || source.calls != 0 { t.Fatalf("unsupported mode touched a source: %+v %v calls=%d", observation, err, source.calls) } } } func TestServicePreservesUnavailableAndObservedZero(t *testing.T) { - target := Target{Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider"}} + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}} service, err := NewService(fixedResolver{target: target}, nil) if err != nil { t.Fatal(err) } observation, err := service.ObserveSession(t.Context(), "tenant", "session") - if err != nil || observation.Status != StatusUnavailable || observation.Sample != nil { + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "source_not_configured" || observation.Sample != nil { t.Fatalf("missing source was not unavailable: %+v %v", observation, err) } @@ -68,15 +96,20 @@ func TestServicePreservesUnavailableAndObservedZero(t *testing.T) { } func TestServiceMapsOnlyDeclaredUnavailability(t *testing.T) { - target := Target{Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider"}} + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}} for _, tc := range []struct { - err error - wantError bool + err error + wantReason string + wantError bool }{ - {err: ErrUnavailable}, + {err: ErrUnavailable, wantReason: "sample_unavailable"}, + {err: ErrNotRunning, wantReason: "runtime_not_running"}, + {err: context.DeadlineExceeded, wantReason: "sample_timeout"}, {err: errors.New("Docker permission denied"), wantError: true}, } { - service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": &fixedSource{err: tc.err}}) + service, err := NewService(fixedResolver{target: target}, map[string]Source{ + "provider": typedSource{fixedSource: &fixedSource{err: tc.err}, providerType: "docker"}, + }) if err != nil { t.Fatal(err) } @@ -84,8 +117,116 @@ func TestServiceMapsOnlyDeclaredUnavailability(t *testing.T) { if (err != nil) != tc.wantError { t.Fatalf("wrong error classification: %+v %v", observation, err) } - if !tc.wantError && observation.Status != StatusUnavailable { + if !tc.wantError && (observation.Status != StatusUnavailable || observation.Reason != tc.wantReason || observation.ProviderType != "docker") { t.Fatalf("declared unavailability was not mapped: %+v", observation) } } } + +func TestServiceMapsAnActualSourceDeadlineWithoutLeakingIt(t *testing.T) { + target := Target{ + EnvironmentID: "environment", Mode: ModeManaged, + Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}, + } + service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": blockingSource{}}) + if err != nil { + t.Fatal(err) + } + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Millisecond) + defer cancel() + observation, err := service.ObserveSession(ctx, "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "sample_timeout" || observation.ProviderType != "docker" { + t.Fatalf("source deadline was not safely classified: %+v %v", observation, err) + } +} + +func TestServiceClassifiesResolverAndTerminalAllocationUnavailability(t *testing.T) { + now := time.Date(2026, 9, 22, 1, 0, 0, 0, time.UTC) + service, err := NewService(fixedResolver{target: Target{SessionID: "session", EnvironmentID: "environment", Mode: ModeManaged}, err: ErrUnavailable}, nil) + if err != nil { + t.Fatal(err) + } + service.now = func() time.Time { return now } + observation, err := service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "allocation_pending" || !observation.ResolvedAt.Equal(now) { + t.Fatalf("pending allocation was not classified: %+v %v", observation, err) + } + + source := &fixedSource{} + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "creating"}} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + observation, err = service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "allocation_pending" || source.calls != 0 { + t.Fatalf("creating allocation reached its provider: %+v %v calls=%d", observation, err, source.calls) + } + + target = Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "released"}} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": &fixedSource{}}) + if err != nil { + t.Fatal(err) + } + observation, err = service.ObserveSession(t.Context(), "tenant", "session") + if err != nil || observation.Status != StatusUnavailable || observation.Reason != "runtime_not_running" { + t.Fatalf("released allocation was not classified: %+v %v", observation, err) + } + + source = &fixedSource{} + target = Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{ + AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running", AllocationCreatedAt: time.Now().Add(time.Hour), + }} + service, err = NewService(fixedResolver{target: target}, map[string]Source{"provider": source}) + if err != nil { + t.Fatal(err) + } + if _, err := service.ObserveSession(t.Context(), "tenant", "session"); err == nil || source.calls != 0 { + t.Fatalf("future allocation creation reached its provider: %v calls=%d", err, source.calls) + } +} + +func TestServiceRejectsUnsafeProviderSamples(t *testing.T) { + target := Target{EnvironmentID: "environment", Mode: ModeManaged, Instance: Instance{AllocationID: "allocation", ProviderKey: "provider", AllocationState: "running"}} + now := time.Date(2026, 9, 22, 1, 0, 0, 0, time.UTC) + preEpoch := time.Unix(-1, 0).UTC() + tooLarge := uint64(1 << 53) + for _, sample := range []Sample{ + {ObservedAt: preEpoch}, + {ObservedAt: now, StartedAt: &preEpoch}, + {ObservedAt: now, CPUUsageSecondsTotal: float64Pointer(-1)}, + {ObservedAt: now, CPUUsageSecondsTotal: float64Pointer(math.NaN())}, + {ObservedAt: now, CPUUsageSecondsTotal: float64Pointer(math.Inf(1))}, + {ObservedAt: now, CPUCapacityCores: float64Pointer(0)}, + {ObservedAt: now, MemoryUsageBytes: &tooLarge}, + {ObservedAt: now, MemoryLimitBytes: &tooLarge}, + } { + service, err := NewService(fixedResolver{target: target}, map[string]Source{"provider": &fixedSource{sample: sample}}) + if err != nil { + t.Fatal(err) + } + service.now = func() time.Time { return now } + if _, err := service.ObserveSession(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("unsafe sample accepted: %+v", sample) + } + } +} + +func TestServiceRejectsMismatchedResolvedOwnership(t *testing.T) { + for _, target := range []Target{ + {TenantID: "other", SessionID: "session", Mode: ModeNone}, + {TenantID: "tenant", SessionID: "other", Mode: ModeNone}, + {TenantID: "tenant", SessionID: "session", EnvironmentID: "unexpected", Mode: ModeNone}, + {TenantID: "tenant", SessionID: "session", Mode: ModeSelfHosted}, + } { + service, err := NewService(fixedResolver{target: target}, nil) + if err != nil { + t.Fatal(err) + } + if _, err := service.ObserveSession(t.Context(), "tenant", "session"); err == nil { + t.Fatalf("mismatched ownership accepted: %+v", target) + } + } +} + +func float64Pointer(value float64) *float64 { return &value } diff --git a/services/agents-api/internal/sandbox/docker/resources.go b/services/agents-api/internal/sandbox/docker/resources.go index 2f444c961..f0f057de0 100644 --- a/services/agents-api/internal/sandbox/docker/resources.go +++ b/services/agents-api/internal/sandbox/docker/resources.go @@ -15,6 +15,8 @@ import ( var _ runtimeobs.Source = (*Provider)(nil) +func (*Provider) ObservationProviderType() string { return "docker" } + // Observe is read-only. Inspect verifies allocation ownership before Docker // statistics are requested; it never renews or changes the container. func (p *Provider) Observe(ctx context.Context, target runtimeobs.Target) (runtimeobs.Sample, error) { @@ -27,13 +29,13 @@ func (p *Provider) Observe(ctx context.Context, target runtimeobs.Target) (runti reference := sandbox.Reference{TenantID: target.TenantID, EnvironmentID: target.EnvironmentID, AllocationID: target.Instance.AllocationID} inspected, err := p.inspect(ctx, reference) if errors.Is(err, sandbox.ErrNotFound) { - return runtimeobs.Sample{}, runtimeobs.ErrUnavailable + return runtimeobs.Sample{}, runtimeobs.ErrNotRunning } if err != nil { return runtimeobs.Sample{}, err } if inspected.Container.State == nil || !inspected.Container.State.Running { - return runtimeobs.Sample{}, runtimeobs.ErrUnavailable + return runtimeobs.Sample{}, runtimeobs.ErrNotRunning } result, err := p.client.ContainerStats(ctx, inspected.Container.ID, client.ContainerStatsOptions{Stream: false, IncludePreviousSample: false}) if err != nil { diff --git a/services/agents-api/internal/sandbox/docker/resources_test.go b/services/agents-api/internal/sandbox/docker/resources_test.go index ea4db1096..8a21dbff6 100644 --- a/services/agents-api/internal/sandbox/docker/resources_test.go +++ b/services/agents-api/internal/sandbox/docker/resources_test.go @@ -2,6 +2,7 @@ package docker import ( "encoding/json" + "errors" "net/http" "net/http/httptest" "strings" @@ -24,13 +25,14 @@ func TestObserveVerifiesOwnershipThenReadsOneShotStats(t *testing.T) { started := observed.Add(-time.Minute) statsRead := false omitMeasurements := false + running := true server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { w.Header().Set("Content-Type", "application/json") switch { case r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/json"): _ = json.NewEncoder(w).Encode(map[string]any{ "Id": "container-id", - "State": map[string]any{"Status": "running", "Running": true, "StartedAt": started.Format(time.RFC3339Nano)}, + "State": map[string]any{"Status": "running", "Running": running, "StartedAt": started.Format(time.RFC3339Nano)}, "HostConfig": map[string]any{"NanoCpus": 2_000_000_000, "Memory": 2048}, "Config": map[string]any{"Labels": map[string]string{ labelPrefix + "installation": installationID, @@ -76,6 +78,12 @@ func TestObserveVerifiesOwnershipThenReadsOneShotStats(t *testing.T) { if err != nil || missing.CPUUsageSecondsTotal != nil || missing.MemoryUsageBytes != nil || missing.CPUCapacityCores == nil || missing.MemoryLimitBytes == nil { t.Fatalf("missing Docker measurements became zero: %+v %v", missing, err) } + running = false + statsRead = false + if _, err := p.Observe(t.Context(), target); !errors.Is(err, runtimeobs.ErrNotRunning) || statsRead { + t.Fatalf("stopped Runtime was not classified before stats: %v stats=%v", err, statsRead) + } + running = true foreign := target foreign.Instance.ProviderKey = uuid.NewString() statsRead = false From 0a10d5b3ee8dbdb1c1a22a33eeda9e33021d423f Mon Sep 17 00:00:00 2001 From: sam Date: Tue, 22 Sep 2026 20:33:57 +0800 Subject: [PATCH 4/6] Add Runtime observability dashboard --- apps/web/e2e/agents-lifecycle.spec.ts | 17 +- apps/web/e2e/fixture-core.mjs | 9 + apps/web/src/App.tsx | 103 ++++++- .../src/features/dashboard/DashboardView.css | 289 ++++++++++++++++++ .../features/dashboard/DashboardView.test.tsx | 80 ++++- .../src/features/dashboard/DashboardView.tsx | 208 ++++++++++++- .../dashboard/dashboard-model.test.ts | 94 +++++- .../src/features/dashboard/dashboard-model.ts | 212 +++++++++++++ .../dashboard/runtime-snapshot.test.ts | 57 ++++ .../features/dashboard/runtime-snapshot.ts | 59 ++++ .../agents-api/runtime-observability-api.md | 8 +- .../runtime-observability-design.md | 19 +- docs/web/README.md | 11 +- docs/web/README.zh-CN.md | 12 +- 14 files changed, 1140 insertions(+), 38 deletions(-) create mode 100644 apps/web/src/features/dashboard/runtime-snapshot.test.ts create mode 100644 apps/web/src/features/dashboard/runtime-snapshot.ts diff --git a/apps/web/e2e/agents-lifecycle.spec.ts b/apps/web/e2e/agents-lifecycle.spec.ts index 640a954ef..9fd4d700a 100644 --- a/apps/web/e2e/agents-lifecycle.spec.ts +++ b/apps/web/e2e/agents-lifecycle.spec.ts @@ -2329,10 +2329,11 @@ test("presents Dashboard page-chain results and System boundaries without extra await expect.poll(async () => { const entries = await fixtureRequests(request); return [count(entries, "/v1/agents"), count(entries, "/v1/agents/sessions")]; - }).toEqual([count(before, "/v1/agents") + 1, count(before, "/v1/agents/sessions") + 1]); + }).toEqual([count(before, "/v1/agents") + 1, count(before, "/v1/agents/sessions") + 2]); const after = await fixtureRequests(request); expect(count(after, "/v1/agents")).toBe(count(before, "/v1/agents") + 1); - expect(count(after, "/v1/agents/sessions")).toBe(count(before, "/v1/agents/sessions") + 1); + expect(count(after, "/v1/agents/sessions")).toBe(count(before, "/v1/agents/sessions") + 2); + expect(count(after, "/v1/agents/runtime-observations")).toBe(count(before, "/v1/agents/runtime-observations") + 1); for (const path of detailPaths) expect(count(after, path)).toBe(count(before, path)); await attachScreenshot(page, testInfo, "desktop-dashboard-loaded-snapshot"); @@ -2387,11 +2388,12 @@ test("presents Dashboard page-chain results and System boundaries without extra return [count(entries, "/v1/agents"), count(entries, "/v1/agents/sessions")]; }).toEqual([ count(beforeSystemRefresh, "/v1/agents") + 1, - count(beforeSystemRefresh, "/v1/agents/sessions") + 1, + count(beforeSystemRefresh, "/v1/agents/sessions") + 2, ]); const afterSystemRefresh = await fixtureRequests(request); expect(count(afterSystemRefresh, "/v1/agents")).toBe(count(beforeSystemRefresh, "/v1/agents") + 1); - expect(count(afterSystemRefresh, "/v1/agents/sessions")).toBe(count(beforeSystemRefresh, "/v1/agents/sessions") + 1); + expect(count(afterSystemRefresh, "/v1/agents/sessions")).toBe(count(beforeSystemRefresh, "/v1/agents/sessions") + 2); + expect(count(afterSystemRefresh, "/v1/agents/runtime-observations")).toBe(count(beforeSystemRefresh, "/v1/agents/runtime-observations") + 1); for (const path of detailPaths) expect(count(afterSystemRefresh, path)).toBe(count(beforeSystemRefresh, path)); await attachScreenshot(page, testInfo, "desktop-system-contract-boundary"); @@ -2480,7 +2482,9 @@ test("publishes Dashboard counts only after every top-level Agent and Session pa await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Agents" })).toContainText("3"); await expect(dashboard.locator(".dashboard-summary > div").filter({ hasText: "Sessions" })).toContainText("2"); expect(agentAfters).toEqual([null, "agent_b"]); - expect(sessionAfters).toEqual([null, "session_snapshot"]); + expect(sessionAfters).toHaveLength(4); + expect(sessionAfters.filter((after) => after === null)).toHaveLength(2); + expect(sessionAfters.filter((after) => after === "session_snapshot")).toHaveLength(2); }); test("keeps the previous Dashboard result when pagination exceeds the safety limit", async ({ page, request }) => { @@ -2523,8 +2527,9 @@ test("keeps the previous Dashboard result when pagination exceeds the safety lim await refresh.click(); await expect.poll(() => reads).toBe(100); await expect(refresh).toBeEnabled(); - await expect(dashboard).toContainText("Using the last successful snapshot"); + await expect(dashboard).toContainText("Snapshot incomplete"); await expect(dashboard).toContainText("collection pagination exceeded the Web safety limit"); + await expect(dashboard).toContainText("Sessions changed while Runtime observations were loading"); await expect(loadedAgents).toContainText("3"); await dashboard.getByRole("button", { name: "Connection settings" }).click(); const connectionDialog = page.getByRole("dialog", { name: "Connect an Agent Core" }); diff --git a/apps/web/e2e/fixture-core.mjs b/apps/web/e2e/fixture-core.mjs index 5aebe44dd..640616184 100644 --- a/apps/web/e2e/fixture-core.mjs +++ b/apps/web/e2e/fixture-core.mjs @@ -995,6 +995,15 @@ const server = http.createServer(async (request, response) => { return sendJson(response, created, 201); } + // The legacy Web fixture uses human-readable Session IDs for interaction + // assertions. Runtime observation resources require canonical UUIDs, so this + // fixture advertises an empty, valid collection instead of inventing a false + // identity join. Positive Runtime rendering is covered by the typed component + // and coordinator tests with canonical identities. + if (request.method === "GET" && url.pathname === "/v1/agents/runtime-observations") { + return sendJson(response, page([])); + } + if (request.method === "GET" && url.pathname === "/v1/agents/sessions") { trackAbort(response, "sessionListReads"); const control = consumeControl("sessionList"); diff --git a/apps/web/src/App.tsx b/apps/web/src/App.tsx index e4a7d6f85..c0925e986 100644 --- a/apps/web/src/App.tsx +++ b/apps/web/src/App.tsx @@ -33,6 +33,12 @@ import { requestAgentUpdate, } from "./features/agents/agent-actions"; import { DashboardView } from "./features/dashboard/DashboardView"; +import { + loadRuntimeDashboardSnapshot, + RUNTIME_SNAPSHOT_REFRESH_MS, + RUNTIME_SNAPSHOT_TIMEOUT_MS, + type RuntimeDashboardSnapshot, +} from "./features/dashboard/runtime-snapshot"; import { SessionsView, type SessionDetailState, @@ -270,6 +276,10 @@ export function App() { const [sessionCollectionState, setSessionCollectionState] = useState("connecting"); const [sessionCollectionError, setSessionCollectionError] = useState(null); const [sessionCollectionHasSnapshot, setSessionCollectionHasSnapshot] = useState(false); + const [runtimeSnapshot, setRuntimeSnapshot] = useState(null); + const [runtimeCollectionState, setRuntimeCollectionState] = useState("connecting"); + const [runtimeCollectionError, setRuntimeCollectionError] = useState(null); + const [runtimeCollectionHasSnapshot, setRuntimeCollectionHasSnapshot] = useState(false); const [sessionAgentFilter, setSessionAgentFilter] = useState(null); const [filteredSessions, setFilteredSessions] = useState([]); const [filteredSessionCollectionState, setFilteredSessionCollectionState] = useState("connecting"); @@ -305,6 +315,8 @@ export function App() { const sessionCollectionRequestRef = useRef(0); const agentCollectionAbortRef = useRef(null); const sessionCollectionAbortRef = useRef(null); + const runtimeCollectionAbortRef = useRef(null); + const runtimeCollectionRequestRef = useRef(0); const filteredSessionCollectionAbortRef = useRef(null); const filteredSessionCollectionRequestRef = useRef(0); const sessionAgentFilterRef = useRef(sessionAgentFilter); @@ -516,6 +528,54 @@ export function App() { } }, [core, coreGeneration, notify]); + const refreshRuntimeSnapshot = useCallback(async () => { + if (coreGeneration !== connectionGenerationRef.current) return false; + runtimeCollectionAbortRef.current?.abort(); + const controller = new AbortController(); + runtimeCollectionAbortRef.current = controller; + const request = runtimeCollectionRequestRef.current + 1; + runtimeCollectionRequestRef.current = request; + let timedOut = false; + const timeout = window.setTimeout(() => { + timedOut = true; + controller.abort(); + }, RUNTIME_SNAPSHOT_TIMEOUT_MS); + setRuntimeCollectionState("connecting"); + setRuntimeCollectionError(null); + try { + const result = await settleCollection(() => loadRuntimeDashboardSnapshot( + core, + () => sessionCollectionRevisionRef.current, + controller.signal, + )); + if ( + coreGeneration !== connectionGenerationRef.current || + request !== runtimeCollectionRequestRef.current + ) return false; + if (result.status === "rejected") { + if (isAbort(result.reason) && !timedOut) return false; + const message = timedOut + ? "Runtime snapshot exceeded the 15 second Web refresh budget." + : errorMessage(result.reason); + setRuntimeCollectionState("failed"); + setRuntimeCollectionError(message); + return false; + } + if (result.value === null) { + setRuntimeCollectionState("failed"); + setRuntimeCollectionError("Session data changed while Runtime observations were loading. The previous complete snapshot was retained."); + return false; + } + setRuntimeSnapshot(result.value); + setRuntimeCollectionHasSnapshot(true); + setRuntimeCollectionState("ready"); + return true; + } finally { + window.clearTimeout(timeout); + if (runtimeCollectionAbortRef.current === controller) runtimeCollectionAbortRef.current = null; + } + }, [core, coreGeneration]); + const refreshFilteredSessions = useCallback(async (agentId: string) => { if ( coreGeneration !== connectionGenerationRef.current || @@ -855,9 +915,10 @@ export function App() { void refreshSessions(); void refreshVaults(); void refreshEnvironmentTemplates(); + void refreshRuntimeSnapshot(); const filter = sessionAgentFilterRef.current; if (filter) void refreshFilteredSessions(filter); - }, [refreshAgents, refreshEnvironmentTemplates, refreshFilteredSessions, refreshSessions, refreshVaults]); + }, [refreshAgents, refreshEnvironmentTemplates, refreshFilteredSessions, refreshRuntimeSnapshot, refreshSessions, refreshVaults]); const changeSessionAgentFilter = useCallback((agentId: string | null) => { if (sessionAgentFilterRef.current === agentId) return; @@ -894,6 +955,10 @@ export function App() { setSessions([]); setAgentCollectionHasSnapshot(false); setSessionCollectionHasSnapshot(false); + setRuntimeSnapshot(null); + setRuntimeCollectionState("connecting"); + setRuntimeCollectionError(null); + setRuntimeCollectionHasSnapshot(false); setItems([]); setTurns([]); setEnvironmentObservations(new Map()); @@ -908,7 +973,29 @@ export function App() { void refreshSessions(); void refreshVaults(); void refreshEnvironmentTemplates(); - }, [refreshAgents, refreshEnvironmentTemplates, refreshSessions, refreshVaults]); + void refreshRuntimeSnapshot(); + }, [refreshAgents, refreshEnvironmentTemplates, refreshRuntimeSnapshot, refreshSessions, refreshVaults]); + + useEffect(() => { + if (view !== "dashboard") return; + let timer: number | null = null; + const schedule = () => { + const jitter = Math.floor(Math.random() * 5_000); + timer = window.setTimeout(() => { + if (!document.hidden) void refreshRuntimeSnapshot(); + schedule(); + }, RUNTIME_SNAPSHOT_REFRESH_MS + jitter); + }; + const onVisibilityChange = () => { + if (!document.hidden) void refreshRuntimeSnapshot(); + }; + document.addEventListener("visibilitychange", onVisibilityChange); + schedule(); + return () => { + if (timer !== null) window.clearTimeout(timer); + document.removeEventListener("visibilitychange", onVisibilityChange); + }; + }, [refreshRuntimeSnapshot, view]); useEffect(() => { filteredSessionCollectionAbortRef.current?.abort(); @@ -1943,6 +2030,7 @@ export function App() { connectionGenerationRef.current += 1; agentCollectionRequestRef.current += 1; sessionCollectionRequestRef.current += 1; + runtimeCollectionRequestRef.current += 1; filteredSessionCollectionRequestRef.current += 1; vaultCollectionRequestRef.current += 1; agentCollectionRevisionRef.current = 0; @@ -1966,6 +2054,8 @@ export function App() { agentCollectionAbortRef.current = null; sessionCollectionAbortRef.current?.abort(); sessionCollectionAbortRef.current = null; + runtimeCollectionAbortRef.current?.abort(); + runtimeCollectionAbortRef.current = null; filteredSessionCollectionAbortRef.current?.abort(); filteredSessionCollectionAbortRef.current = null; vaultCollectionAbortRef.current?.abort(); @@ -1981,6 +2071,10 @@ export function App() { setSessionCollectionState("connecting"); setSessionCollectionError(null); setSessionCollectionHasSnapshot(false); + setRuntimeSnapshot(null); + setRuntimeCollectionState("connecting"); + setRuntimeCollectionError(null); + setRuntimeCollectionHasSnapshot(false); sessionAgentFilterRef.current = null; filteredSessionsRef.current = []; setSessionAgentFilter(null); @@ -2108,6 +2202,10 @@ export function App() { sessionCollectionState={sessionCollectionState} sessionCollectionError={sessionCollectionError} sessionCollectionHasSnapshot={sessionCollectionHasSnapshot} + runtimeSnapshot={runtimeSnapshot} + runtimeCollectionState={runtimeCollectionState} + runtimeCollectionError={runtimeCollectionError} + runtimeCollectionHasSnapshot={runtimeCollectionHasSnapshot} onRefresh={refreshDashboard} onCreateAgent={openAgentSetup} onStartSession={() => openSessionSetup()} @@ -2208,6 +2306,7 @@ export function App() { refreshing={ agentCollectionState === "connecting" || sessionCollectionState === "connecting" || + runtimeCollectionState === "connecting" || vaultCollectionState === "connecting" } onRefresh={refreshDashboard} diff --git a/apps/web/src/features/dashboard/DashboardView.css b/apps/web/src/features/dashboard/DashboardView.css index 999874c09..b76218653 100644 --- a/apps/web/src/features/dashboard/DashboardView.css +++ b/apps/web/src/features/dashboard/DashboardView.css @@ -613,10 +613,289 @@ border-radius: 50%; } +.dashboard-runtime-panel { + margin-top: 16px; + overflow: hidden; +} + +.dashboard-runtime-freshness { + color: var(--fg-muted); + font-family: var(--font-mono); + font-size: 10px; + font-variant-numeric: tabular-nums; + line-height: 15px; + text-align: right; +} + +.dashboard-runtime-summary { + display: grid; + grid-template-columns: repeat(4, minmax(0, 1fr)); + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-metric { + display: grid; + min-width: 0; + min-height: 106px; + padding: 16px; + grid-template-columns: 32px minmax(0, 1fr); + align-items: flex-start; + gap: 10px; +} + +.dashboard-runtime-metric + .dashboard-runtime-metric { + border-left: 1px solid var(--line); +} + +.dashboard-runtime-metric-icon { + display: inline-flex; + width: 32px; + height: 32px; + align-items: center; + justify-content: center; + color: var(--accent); + background: color-mix(in srgb, var(--accent) 8%, var(--surface)); + border: 1px solid color-mix(in srgb, var(--accent) 18%, var(--line)); + border-radius: 8px; +} + +.dashboard-runtime-metric > span:last-child { + display: grid; + min-width: 0; + gap: 2px; +} + +.dashboard-runtime-metric small, +.dashboard-runtime-metric > span:last-child > span { + color: var(--fg-muted); + font-size: 10px; + line-height: 14px; +} + +.dashboard-runtime-metric strong { + overflow: hidden; + font-family: var(--font-mono); + font-size: 17px; + font-variant-numeric: tabular-nums slashed-zero; + font-weight: 550; + line-height: 24px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-ledger { + min-width: 960px; +} + +.dashboard-runtime-panel:has(.dashboard-runtime-ledger) { + overflow-x: auto; +} + +.dashboard-runtime-header, +.dashboard-runtime-row { + display: grid; + grid-template-columns: minmax(190px, 1.45fr) minmax(140px, 1fr) minmax(130px, .9fr) minmax(150px, 1fr) minmax(140px, .9fr) minmax(110px, .75fr); + align-items: center; + column-gap: 14px; +} + +.dashboard-runtime-header { + min-height: 34px; + padding: 7px 14px; + color: var(--fg-muted); + font-size: 9px; + font-weight: 600; + letter-spacing: .04em; + line-height: 13px; + background: var(--surface-subtle); + border-bottom: 1px solid var(--line); + text-transform: uppercase; +} + +.dashboard-runtime-entry + .dashboard-runtime-entry { + border-top: 1px solid var(--line); +} + +.dashboard-runtime-row { + min-height: 70px; + padding: 11px 14px; + transition: background-color 150ms var(--ease-settle); +} + +.dashboard-runtime-row:hover, +.dashboard-runtime-row:focus-within { + background: var(--hover); +} + +.dashboard-runtime-row > span { + display: grid; + min-width: 0; + gap: 3px; +} + +.dashboard-runtime-row small { + overflow: hidden; + color: var(--fg-muted); + font-size: 9px; + line-height: 13px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-identity button { + width: fit-content; + max-width: 100%; + padding: 0; + overflow: hidden; + color: var(--fg); + font-size: 12px; + font-weight: 600; + line-height: 17px; + text-align: left; + text-overflow: ellipsis; + white-space: nowrap; + background: transparent; + border: 0; +} + +.dashboard-runtime-identity button:hover, +.dashboard-runtime-identity button:focus-visible { + color: var(--accent); +} + +.dashboard-runtime-status { + display: inline-flex; + width: fit-content; + align-items: center; + gap: 6px; + font-size: 10px; + font-weight: 550; + line-height: 15px; +} + +.dashboard-runtime-status > span { + width: 7px; + height: 7px; + background: var(--fg-muted); + border-radius: 50%; +} + +.dashboard-runtime-status-observed > span { + background: var(--success); + box-shadow: 0 0 0 3px color-mix(in srgb, var(--success) 11%, transparent); +} + +.dashboard-runtime-status-unavailable { + color: var(--warning); +} + +.dashboard-runtime-status-unavailable > span { + background: var(--warning); +} + +.dashboard-runtime-value strong { + overflow: hidden; + font-family: var(--font-mono); + font-size: 12px; + font-variant-numeric: tabular-nums slashed-zero; + font-weight: 550; + line-height: 17px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-bar { + display: block; + width: 100%; + max-width: 110px; + height: 3px; + margin-top: 2px; + overflow: hidden; + background: var(--surface-subtle); + border-radius: 999px; +} + +.dashboard-runtime-bar i { + display: block; + height: 100%; + background: var(--accent); + border-radius: inherit; +} + +.dashboard-runtime-detail { + padding: 0 14px 10px; + color: var(--fg-muted); + font-size: 10px; +} + +.dashboard-runtime-detail summary { + width: fit-content; + padding: 4px 0; + cursor: pointer; + user-select: none; +} + +.dashboard-runtime-detail[open] { + padding-top: 8px; + background: var(--surface-subtle); + border-top: 1px dashed var(--line); +} + +.dashboard-runtime-detail dl { + display: grid; + margin: 8px 0; + grid-template-columns: repeat(3, minmax(0, 1fr)); + gap: 8px 14px; +} + +.dashboard-runtime-detail dl > div { + min-width: 0; +} + +.dashboard-runtime-detail dt { + color: var(--fg-muted); + font-size: 9px; + text-transform: uppercase; +} + +.dashboard-runtime-detail dd { + margin: 2px 0 0; + overflow: hidden; + color: var(--fg); + font-family: var(--font-mono); + font-size: 10px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-detail button { + display: inline-flex; + align-items: center; + padding: 4px 7px; + gap: 4px; + color: var(--accent); + font-size: 10px; + background: transparent; + border: 1px solid color-mix(in srgb, var(--accent) 25%, var(--line)); + border-radius: 5px; +} + @media (max-width: 980px) { .dashboard-primary-grid { grid-template-columns: 1fr; } + + .dashboard-runtime-summary { + grid-template-columns: repeat(2, minmax(0, 1fr)); + } + + .dashboard-runtime-metric:nth-child(3) { + border-left: 0; + } + + .dashboard-runtime-metric:nth-child(n + 3) { + border-top: 1px solid var(--line); + } } @media (max-width: 760px) { @@ -701,6 +980,16 @@ align-items: flex-start; } + .dashboard-runtime-summary { + grid-template-columns: 1fr; + } + + .dashboard-runtime-metric + .dashboard-runtime-metric, + .dashboard-runtime-metric:nth-child(3) { + border-top: 1px solid var(--line); + border-left: 0; + } + .dashboard-panel header p { white-space: normal; } diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index f7c617ba4..5d70620aa 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -1,7 +1,7 @@ import { renderToStaticMarkup } from "react-dom/server"; import { describe, expect, it } from "vitest"; -import type { AgentSession, SavedAgent } from "@agents-core-web/agents-client"; +import type { AgentSession, RuntimeObservation, SavedAgent } from "@agents-core-web/agents-client"; import { DashboardView, type DashboardViewProps } from "./DashboardView"; @@ -74,6 +74,10 @@ function render(overrides: Partial = {}): string { sessionCollectionState="ready" sessionCollectionError={null} sessionCollectionHasSnapshot + runtimeSnapshot={{ sessions: [], observations: [], loadedAt: 1_700_000_000_000 }} + runtimeCollectionState="ready" + runtimeCollectionError={null} + runtimeCollectionHasSnapshot {...callbacks} {...overrides} />, @@ -121,6 +125,21 @@ describe("Dashboard loaded-result presentation", () => { expect(html).not.toContain("Sessions: Agent core request failed (502)"); }); + it("keeps a Runtime-only 503 scoped to the optional observation feature", () => { + const html = render({ + runtimeSnapshot: null, + runtimeCollectionState: "failed", + runtimeCollectionError: "Agent core request failed (503).", + runtimeCollectionHasSnapshot: false, + }); + + expect(html).toContain("Runtime: Agent core request failed (503)."); + expect(html).toContain("Runtime observations unavailable"); + expect(html).not.toContain("Agent Core backend is not ready"); + expect(html).not.toContain("Core backend is offline"); + expect(html).not.toContain("Open startup guide"); + }); + it("renders a compact actionable overview while preserving Environment qualifications", () => { const selfHosted: AgentSession["environment"] = { type: "self_hosted", @@ -174,6 +193,65 @@ describe("Dashboard loaded-result presentation", () => { expect(html).not.toContain("Execution ready"); }); + it("renders current Docker resources without inventing a CPU percentage or history", () => { + const hosted = session("11111111-1111-4111-8111-111111111111", { + metadata: { title: "Managed research" }, + environment: { + type: "openai_hosted", + id: "22222222-2222-4222-8222-222222222222", + capability_directories: [], + network: { access: "enabled", allowed_domains: [] }, + packages: { npm: [], python: [], system: [] }, + files: [], + plugins: [], + skills: [], + }, + usage: { + input_tokens: 30, + output_tokens: 12, + total_tokens: 42, + input_tokens_details: { cached_tokens: 7 }, + output_tokens_details: { reasoning_tokens: 3 }, + }, + }); + const observation: RuntimeObservation = { + id: hosted.id, + object: "agent.runtime_observation", + session_id: hosted.id, + environment_id: "22222222-2222-4222-8222-222222222222", + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: "33333333-3333-4333-8333-333333333333", + device_id: null, + connection_generation: null, + }, + status: "observed", + reason: null, + allocation_created_at: 1_700_000_000, + resolved_at: 1_700_000_100, + observed_at: 1_700_000_090, + started_at: 1_700_000_010, + cpu: { usage_seconds_total: 73.5, capacity_cores: 2, usage_cores: null, utilization_ratio: null }, + memory: { usage_bytes: 536_870_912, limit_bytes: 2_147_483_648 }, + }; + const html = render({ + runtimeSnapshot: { sessions: [hosted], observations: [observation], loadedAt: 1_700_000_100_000 }, + }); + + expect(html).toContain("Runtime monitoring"); + expect(html).toContain("1/1 managed observed"); + expect(html).toContain("CPU time / capacity"); + expect(html).toContain("1m 13s / 2 cores"); + expect(html).toContain("512 MiB / 2.00 GiB"); + expect(html).toContain("Current Runtime observations"); + expect(html).toContain("Managed research"); + expect(html).toContain("Identity and sample details"); + expect(html).not.toContain("CPU %"); + expect(html).not.toContain("historical chart"); + }); + it("keeps partial Usage out of the primary overview", () => { const html = render({ sessions: [session("usage-unknown")] }); diff --git a/apps/web/src/features/dashboard/DashboardView.tsx b/apps/web/src/features/dashboard/DashboardView.tsx index c2cd9786d..1ab5f2d87 100644 --- a/apps/web/src/features/dashboard/DashboardView.tsx +++ b/apps/web/src/features/dashboard/DashboardView.tsx @@ -2,9 +2,13 @@ import { AlertTriangle, ArrowRight, Bot, + Cpu, + Gauge, + MemoryStick, MessageSquare, RefreshCw, Rows3, + Server, } from "lucide-react"; import { useMemo, type ReactNode } from "react"; @@ -14,12 +18,18 @@ import { StatusIcon, type StatusKind } from "../../components/StatusIcon"; import { backendFailureStatus } from "../../lib/core-readiness"; import { buildDashboardSnapshot, + buildRuntimeDashboardModel, dashboardEnvironmentLabel, dashboardStatusLabel, + formatDashboardBytes, + formatDashboardDuration, + formatDashboardTokens, formatDashboardTimestamp, + runtimeObservationStatusLabel, type DashboardCollectionState, type DashboardSessionRow, } from "./dashboard-model"; +import type { RuntimeDashboardSnapshot } from "./runtime-snapshot"; import "./DashboardView.css"; export interface DashboardViewProps { @@ -31,6 +41,10 @@ export interface DashboardViewProps { sessionCollectionState: DashboardCollectionState; sessionCollectionError: string | null; sessionCollectionHasSnapshot: boolean; + runtimeSnapshot: RuntimeDashboardSnapshot | null; + runtimeCollectionState: DashboardCollectionState; + runtimeCollectionError: string | null; + runtimeCollectionHasSnapshot: boolean; onRefresh: () => void; onCreateAgent: () => void; onStartSession: () => void; @@ -40,6 +54,109 @@ export interface DashboardViewProps { onOpenSession: (sessionId: string) => void; } +function RuntimeMetric({ + icon, + label, + value, + detail, +}: { + icon: ReactNode; + label: string; + value: string; + detail: string; +}) { + return ( +
+ + + {label} + {value} + {detail} + +
+ ); +} + +function percent(usage: number | null | undefined, limit: number | null | undefined): number | null { + if (typeof usage !== "number" || typeof limit !== "number" || limit <= 0) return null; + return Math.min(100, Math.max(0, usage / limit * 100)); +} + +function RuntimeTable({ + snapshot, + onOpenSession, +}: { + snapshot: RuntimeDashboardSnapshot; + onOpenSession: (sessionId: string) => void; +}) { + const model = buildRuntimeDashboardModel(snapshot.sessions, snapshot.observations); + return ( +
+
+ Session / Runtime + Observation + CPU + Memory + Uptime + Tokens +
+ {model.rows.map((row) => { + const observation = row.observation; + const memoryPercent = observation.status === "observed" + ? percent(observation.memory?.usage_bytes, observation.memory?.limit_bytes) + : null; + return ( +
+
+ + + {row.session.agentLabel} · {dashboardStatusLabel(row.session.status)} + + + + + {observation.mode === "openai_hosted" ? observation.provider_type ?? "Managed" : dashboardEnvironmentLabel(row.session.environmentProfile)} + + + {observation.status === "observed" ? formatDashboardDuration(observation.cpu?.usage_seconds_total ?? null) : "—"} + {observation.status === "observed" && typeof observation.cpu?.capacity_cores === "number" + ? `${observation.cpu.capacity_cores.toLocaleString("en-US")} cores capacity` + : "No current sample"} + + + {observation.status === "observed" ? formatDashboardBytes(observation.memory?.usage_bytes ?? null) : "—"} + {observation.status === "observed" ? `of ${formatDashboardBytes(observation.memory?.limit_bytes ?? null)}` : "No current sample"} + {memoryPercent !== null ? : null} + + + {formatDashboardDuration(row.computeUptimeSeconds)} + {row.allocationAgeSeconds === null ? "Allocation age unavailable" : `${formatDashboardDuration(row.allocationAgeSeconds)} allocation age`} + + + {formatDashboardTokens(row.session.totalTokens)} + {row.session.totalTokens === null ? "Usage not reported" : "Session reported"} + +
+
+ Identity and sample details +
+
Session
{observation.session_id}
+
Environment
{observation.environment_id ?? "Not applicable"}
+
Allocation
{observation.instance.allocation_id ?? "Not available"}
+
Provider
{observation.provider_type ?? "Not available"}
+
Resolved
{formatDashboardTimestamp(observation.resolved_at)}
+
Observed
{formatDashboardTimestamp(observation.observed_at)}
+
+ +
+
+ ); + })} +
+ ); +} + function collectionHasSnapshot(state: DashboardCollectionState, hasSnapshot: boolean): boolean { return state === "ready" || hasSnapshot; } @@ -232,6 +349,10 @@ export function DashboardView({ sessionCollectionState, sessionCollectionError, sessionCollectionHasSnapshot, + runtimeSnapshot, + runtimeCollectionState, + runtimeCollectionError, + runtimeCollectionHasSnapshot, onRefresh, onCreateAgent, onStartSession, @@ -241,21 +362,28 @@ export function DashboardView({ onOpenSession, }: DashboardViewProps) { const snapshot = useMemo(() => buildDashboardSnapshot(agents, sessions, 6, 5), [agents, sessions]); + const runtimeModel = useMemo(() => runtimeSnapshot + ? buildRuntimeDashboardModel(runtimeSnapshot.sessions, runtimeSnapshot.observations) + : null, [runtimeSnapshot]); const agentsAvailable = collectionHasSnapshot(agentCollectionState, agentCollectionHasSnapshot); const sessionsAvailable = collectionHasSnapshot(sessionCollectionState, sessionCollectionHasSnapshot); - const refreshing = agentCollectionState === "connecting" || sessionCollectionState === "connecting"; + const runtimeAvailable = runtimeCollectionState === "ready" || runtimeCollectionHasSnapshot; + const refreshing = agentCollectionState === "connecting" || sessionCollectionState === "connecting" || runtimeCollectionState === "connecting"; const hasStaleSnapshot = ( (agentCollectionState === "failed" && agentCollectionHasSnapshot) || - (sessionCollectionState === "failed" && sessionCollectionHasSnapshot) + (sessionCollectionState === "failed" && sessionCollectionHasSnapshot) || + (runtimeCollectionState === "failed" && runtimeCollectionHasSnapshot) ); - const hasUnavailableSource = !agentsAvailable || !sessionsAvailable; + const hasUnavailableSource = !agentsAvailable || !sessionsAvailable || !runtimeAvailable; const attentionCount = snapshot.statusCounts.requires_action + snapshot.statusCounts.failed; - const sourceErrors: Array = [ + const sourceErrors: Array = [ agentCollectionState === "failed" ? ["Agents", agentCollectionError] as const : null, sessionCollectionState === "failed" ? ["Sessions", sessionCollectionError] as const : null, - ].filter((entry): entry is readonly ["Agents" | "Sessions", string | null] => entry !== null); - const backendFailureStatuses = sourceErrors.map(([, error]) => backendFailureStatus(error)); - const backendUnavailable = sourceErrors.length > 0 && backendFailureStatuses.every(Boolean); + runtimeCollectionState === "failed" ? ["Runtime", runtimeCollectionError] as const : null, + ].filter((entry): entry is readonly ["Agents" | "Sessions" | "Runtime", string | null] => entry !== null); + const coreSourceErrors = sourceErrors.filter(([label]) => label !== "Runtime"); + const backendFailureStatuses = coreSourceErrors.map(([, error]) => backendFailureStatus(error)); + const backendUnavailable = coreSourceErrors.length > 0 && backendFailureStatuses.every(Boolean); const backendFailureDetail = Array.from(new Set(backendFailureStatuses.filter(Boolean))).map((status) => ( status === "network" ? "network failure" : `HTTP ${status}` )).join(" / "); @@ -304,6 +432,7 @@ export function DashboardView({
+
@@ -359,6 +488,71 @@ export function DashboardView({ +
+
+
+

Runtime monitoring

+

Current provider samples joined to an exact complete Session snapshot · no historical series

+
+ {runtimeModel ? ( + + {runtimeModel.summary.observedRuntimeCount}/{runtimeModel.summary.managedRuntimeCount} managed observed + {runtimeModel.summary.newestResolvedAt === null ? "" : ` · ${formatDashboardTimestamp(runtimeModel.summary.newestResolvedAt)}`} + + ) : null} +
+ {!runtimeAvailable || !runtimeSnapshot || !runtimeModel ? ( +

+

+ ) : ( + <> +
+ } + label="Active Runtimes" + value={runtimeModel.summary.observedRuntimeCount.toLocaleString("en-US")} + detail={`${runtimeModel.summary.managedRuntimeCount} managed · ${runtimeModel.summary.unavailableRuntimeCount} unavailable`} + /> + } + label="CPU time / capacity" + value={runtimeModel.summary.cpuUsageSecondsTotal === null && runtimeModel.summary.cpuCapacityCores === null + ? "No current sample" + : `${formatDashboardDuration(runtimeModel.summary.cpuUsageSecondsTotal)} / ${runtimeModel.summary.cpuCapacityCores?.toLocaleString("en-US") ?? "—"} cores`} + detail={`${runtimeModel.summary.cpuCoverageCount}/${runtimeModel.summary.observedRuntimeCount} observed Runtimes report CPU`} + /> + } + label="Memory" + value={runtimeModel.summary.memoryUsageBytes === null && runtimeModel.summary.memoryLimitBytes === null + ? "No current sample" + : `${formatDashboardBytes(runtimeModel.summary.memoryUsageBytes)} / ${formatDashboardBytes(runtimeModel.summary.memoryLimitBytes)}`} + detail={`${runtimeModel.summary.memoryCoverageCount}/${runtimeModel.summary.observedRuntimeCount} observed Runtimes report memory`} + /> + } + label="Reported tokens" + value={runtimeModel.summary.totalTokens === null ? "Not reported" : formatDashboardTokens(runtimeModel.summary.totalTokens)} + detail={`${runtimeModel.summary.tokenCoverageCount}/${runtimeModel.summary.sessionCount} Sessions report usage`} + /> +
+ {runtimeModel.rows.length ? ( + + ) : ( +

No Session-owned Runtime contexts in this snapshot.

+ )} + + )} +
+
diff --git a/apps/web/src/features/dashboard/dashboard-model.test.ts b/apps/web/src/features/dashboard/dashboard-model.test.ts index e638c883f..b6f35bd0b 100644 --- a/apps/web/src/features/dashboard/dashboard-model.test.ts +++ b/apps/web/src/features/dashboard/dashboard-model.test.ts @@ -1,13 +1,17 @@ import { describe, expect, it } from "vitest"; -import type { AgentSession, SavedAgent, TokenUsage } from "@agents-core-web/agents-client"; +import type { AgentSession, RuntimeObservation, SavedAgent, TokenUsage } from "@agents-core-web/agents-client"; import { buildDashboardSnapshot, + buildRuntimeDashboardModel, dashboardEnvironmentLabel, dashboardEnvironmentProfile, dashboardStatusLabel, formatDashboardTimestamp, + formatDashboardBytes, + formatDashboardDuration, + runtimeObservationStatusLabel, } from "./dashboard-model"; function agent(id: string, overrides: Partial = {}): SavedAgent { @@ -228,4 +232,92 @@ describe("Dashboard loaded-snapshot model", () => { model: "fixture/fallback", }); }); + + it("aggregates only present Runtime measurements and preserves coverage", () => { + const managed = session("11111111-1111-4111-8111-111111111111", { usage: usage(21) }); + const unsupported = session("22222222-2222-4222-8222-222222222222"); + const observations: RuntimeObservation[] = [{ + id: managed.id, + object: "agent.runtime_observation", + session_id: managed.id, + environment_id: "33333333-3333-4333-8333-333333333333", + mode: "openai_hosted", + provider_type: "docker", + instance: { kind: "managed_allocation", allocation_id: "44444444-4444-4444-8444-444444444444", device_id: null, connection_generation: null }, + status: "observed", + reason: null, + allocation_created_at: 100, + resolved_at: 220, + observed_at: 210, + started_at: 150, + cpu: { usage_seconds_total: 3.5, capacity_cores: 2, usage_cores: null, utilization_ratio: null }, + memory: { usage_bytes: 512, limit_bytes: 2048 }, + }, { + id: unsupported.id, + object: "agent.runtime_observation", + session_id: unsupported.id, + environment_id: null, + mode: "none", + provider_type: null, + instance: { kind: "none", allocation_id: null, device_id: null, connection_generation: null }, + status: "unsupported", + reason: "runtime_mode_not_observable", + allocation_created_at: null, + resolved_at: 225, + observed_at: null, + started_at: null, + cpu: null, + memory: null, + }]; + const model = buildRuntimeDashboardModel([managed, unsupported], observations); + + expect(model.summary).toMatchObject({ + sessionCount: 2, + managedRuntimeCount: 1, + observedRuntimeCount: 1, + cpuUsageSecondsTotal: 3.5, + cpuCapacityCores: 2, + cpuCoverageCount: 1, + memoryUsageBytes: 512, + memoryLimitBytes: 2048, + memoryCoverageCount: 1, + totalTokens: 21, + tokenCoverageCount: 1, + oldestResolvedAt: 220, + newestResolvedAt: 225, + }); + expect(model.rows[0]?.computeUptimeSeconds).toBe(60); + expect(model.rows[0]?.allocationAgeSeconds).toBe(120); + expect(runtimeObservationStatusLabel(observations[0]!)).toBe("Observed"); + expect(formatDashboardBytes(2048)).toBe("2.00 KiB"); + expect(formatDashboardDuration(90)).toBe("1m 30s"); + }); + + it("does not infer a released allocation lifetime from the current resolution time", () => { + const stopped = session("11111111-1111-4111-8111-111111111111"); + const observation: RuntimeObservation = { + id: stopped.id, + object: "agent.runtime_observation", + session_id: stopped.id, + environment_id: "33333333-3333-4333-8333-333333333333", + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: "44444444-4444-4444-8444-444444444444", + device_id: null, + connection_generation: null, + }, + status: "unavailable", + reason: "runtime_not_running", + allocation_created_at: 100, + resolved_at: 10_000, + observed_at: null, + started_at: null, + cpu: null, + memory: null, + }; + + expect(buildRuntimeDashboardModel([stopped], [observation]).rows[0]?.allocationAgeSeconds).toBeNull(); + }); }); diff --git a/apps/web/src/features/dashboard/dashboard-model.ts b/apps/web/src/features/dashboard/dashboard-model.ts index 2239bce16..89e36e2a4 100644 --- a/apps/web/src/features/dashboard/dashboard-model.ts +++ b/apps/web/src/features/dashboard/dashboard-model.ts @@ -1,5 +1,6 @@ import type { AgentSession, + RuntimeObservation, SavedAgent, SessionStatus, TokenUsage, @@ -45,6 +46,35 @@ export interface DashboardSnapshot { recentSessions: DashboardSessionRow[]; } +export interface RuntimeDashboardRow { + session: DashboardSessionRow; + observation: RuntimeObservation; + computeUptimeSeconds: number | null; + allocationAgeSeconds: number | null; +} + +export interface RuntimeDashboardSummary { + sessionCount: number; + managedRuntimeCount: number; + observedRuntimeCount: number; + unavailableRuntimeCount: number; + cpuUsageSecondsTotal: number | null; + cpuCapacityCores: number | null; + cpuCoverageCount: number; + memoryUsageBytes: number | null; + memoryLimitBytes: number | null; + memoryCoverageCount: number; + totalTokens: number | null; + tokenCoverageCount: number; + oldestResolvedAt: number | null; + newestResolvedAt: number | null; +} + +export interface RuntimeDashboardModel { + summary: RuntimeDashboardSummary; + rows: RuntimeDashboardRow[]; +} + const sessionStatuses = new Set([ "idle", "in_progress", @@ -229,3 +259,185 @@ export function formatDashboardTimestamp(value: number | null): string { if (seconds === null) return "Unknown"; return `${new Date(seconds * 1_000).toISOString().slice(0, 16).replace("T", " ")} UTC`; } + +function safeFiniteNonNegative(value: unknown): number | null { + return typeof value === "number" && Number.isFinite(value) && value >= 0 ? value : null; +} + +function safeAdd(left: number, right: number): number | null { + const value = left + right; + return Number.isSafeInteger(left) && Number.isSafeInteger(right) && Number.isSafeInteger(value) ? value : null; +} + +function elapsedSeconds(start: number | null, end: number | null): number | null { + if (start === null || end === null || end < start) return null; + return end - start; +} + +export function buildRuntimeDashboardModel( + sessions: readonly AgentSession[], + observations: readonly RuntimeObservation[], +): RuntimeDashboardModel { + const sessionsById = new Map(sessions.map((session) => [session.id, session])); + const rows: RuntimeDashboardRow[] = []; + let managedRuntimeCount = 0; + let observedRuntimeCount = 0; + let unavailableRuntimeCount = 0; + let cpuUsageSecondsTotal = 0; + let cpuUsageKnown = false; + let cpuUsageSafe = true; + let cpuCapacityCores = 0; + let cpuCapacityKnown = false; + let cpuCapacitySafe = true; + let cpuCoverageCount = 0; + let memoryUsageBytes = 0; + let memoryUsageKnown = false; + let memoryUsageSafe = true; + let memoryLimitBytes = 0; + let memoryLimitKnown = false; + let memoryLimitSafe = true; + let memoryCoverageCount = 0; + let totalTokens = 0; + let tokensKnown = false; + let tokensSafe = true; + let tokenCoverageCount = 0; + let oldestResolvedAt: number | null = null; + let newestResolvedAt: number | null = null; + + for (const observation of observations) { + const session = sessionsById.get(observation.session_id); + if (!session) continue; + const sessionRow = toSessionRow(session); + const resolvedAt = canonicalTimestamp(observation.resolved_at); + if (resolvedAt !== null) { + oldestResolvedAt = oldestResolvedAt === null ? resolvedAt : Math.min(oldestResolvedAt, resolvedAt); + newestResolvedAt = newestResolvedAt === null ? resolvedAt : Math.max(newestResolvedAt, resolvedAt); + } + if (observation.mode === "openai_hosted") managedRuntimeCount += 1; + if (observation.status === "observed") { + observedRuntimeCount += 1; + const cpuUsage = safeFiniteNonNegative(observation.cpu?.usage_seconds_total); + const cpuCapacity = safeFiniteNonNegative(observation.cpu?.capacity_cores); + if (cpuUsage !== null) { + const next = cpuUsageSecondsTotal + cpuUsage; + if (Number.isFinite(next)) { + cpuUsageSecondsTotal = next; + cpuUsageKnown = true; + } else cpuUsageSafe = false; + } + if (cpuCapacity !== null) { + const next = cpuCapacityCores + cpuCapacity; + if (Number.isFinite(next)) { + cpuCapacityCores = next; + cpuCapacityKnown = true; + } else cpuCapacitySafe = false; + } + if (cpuUsage !== null || cpuCapacity !== null) cpuCoverageCount += 1; + + const memoryUsage = safeNonNegativeInteger(observation.memory?.usage_bytes); + const memoryLimit = safeNonNegativeInteger(observation.memory?.limit_bytes); + if (memoryUsage !== null) { + const next = safeAdd(memoryUsageBytes, memoryUsage); + if (next !== null) { + memoryUsageBytes = next; + memoryUsageKnown = true; + } else memoryUsageSafe = false; + } + if (memoryLimit !== null) { + const next = safeAdd(memoryLimitBytes, memoryLimit); + if (next !== null) { + memoryLimitBytes = next; + memoryLimitKnown = true; + } else memoryLimitSafe = false; + } + if (memoryUsage !== null || memoryLimit !== null) memoryCoverageCount += 1; + } else if (observation.status === "unavailable") { + unavailableRuntimeCount += 1; + } + + if (sessionRow.totalTokens !== null) { + const next = safeAdd(totalTokens, sessionRow.totalTokens); + if (next !== null) { + totalTokens = next; + tokensKnown = true; + } else tokensSafe = false; + tokenCoverageCount += 1; + } + rows.push({ + session: sessionRow, + observation, + computeUptimeSeconds: observation.status === "observed" + ? elapsedSeconds(canonicalTimestamp(observation.started_at), canonicalTimestamp(observation.observed_at)) + : null, + allocationAgeSeconds: observation.mode === "openai_hosted" && observation.reason !== "runtime_not_running" + ? elapsedSeconds(canonicalTimestamp(observation.allocation_created_at), resolvedAt) + : null, + }); + } + + const statusOrder = { observed: 0, unavailable: 1, unsupported: 2 } as const; + rows.sort((left, right) => ( + statusOrder[left.observation.status] - statusOrder[right.observation.status] + || compareSessionRows(left.session, right.session) + )); + return { + summary: { + sessionCount: rows.length, + managedRuntimeCount, + observedRuntimeCount, + unavailableRuntimeCount, + cpuUsageSecondsTotal: cpuUsageKnown && cpuUsageSafe ? cpuUsageSecondsTotal : null, + cpuCapacityCores: cpuCapacityKnown && cpuCapacitySafe ? cpuCapacityCores : null, + cpuCoverageCount, + memoryUsageBytes: memoryUsageKnown && memoryUsageSafe ? memoryUsageBytes : null, + memoryLimitBytes: memoryLimitKnown && memoryLimitSafe ? memoryLimitBytes : null, + memoryCoverageCount, + totalTokens: tokensKnown && tokensSafe ? totalTokens : null, + tokenCoverageCount, + oldestResolvedAt, + newestResolvedAt, + }, + rows, + }; +} + +export function formatDashboardBytes(value: number | null): string { + if (value === null) return "Unavailable"; + const units = ["B", "KiB", "MiB", "GiB", "TiB"]; + let amount = value; + let index = 0; + while (amount >= 1024 && index < units.length - 1) { + amount /= 1024; + index += 1; + } + const digits = amount >= 100 || index === 0 ? 0 : amount >= 10 ? 1 : 2; + return `${amount.toFixed(digits)} ${units[index]}`; +} + +export function formatDashboardDuration(value: number | null): string { + if (value === null || !Number.isFinite(value) || value < 0) return "Unavailable"; + const seconds = Math.floor(value); + if (seconds < 60) return `${seconds}s`; + const minutes = Math.floor(seconds / 60); + if (minutes < 60) return `${minutes}m ${seconds % 60}s`; + const hours = Math.floor(minutes / 60); + if (hours < 24) return `${hours}h ${minutes % 60}m`; + return `${Math.floor(hours / 24)}d ${hours % 24}h`; +} + +export function formatDashboardTokens(value: number | null): string { + if (value === null) return "Unavailable"; + return value.toLocaleString("en-US"); +} + +export function runtimeObservationStatusLabel(observation: RuntimeObservation): string { + if (observation.status === "observed") return "Observed"; + if (observation.status === "unsupported") return "Unsupported"; + switch (observation.reason) { + case "allocation_pending": return "Allocation pending"; + case "runtime_not_running": return "Not running"; + case "source_not_configured": return "Source unavailable"; + case "sample_timeout": return "Sample timeout"; + case "sample_unavailable": return "Sample unavailable"; + } +} diff --git a/apps/web/src/features/dashboard/runtime-snapshot.test.ts b/apps/web/src/features/dashboard/runtime-snapshot.test.ts new file mode 100644 index 000000000..4e852e59a --- /dev/null +++ b/apps/web/src/features/dashboard/runtime-snapshot.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from "vitest"; + +import type { + AgentCore, + AgentSession, + RuntimeObservation, +} from "@agents-core-web/agents-client"; + +import { + loadRuntimeDashboardSnapshot, + RuntimeSnapshotIncompleteError, +} from "./runtime-snapshot"; + +const session = { id: "11111111-1111-4111-8111-111111111111" } as AgentSession; +const observation = { + id: session.id, + session_id: session.id, +} as RuntimeObservation; + +function core( + sessions: AgentSession[], + observations: RuntimeObservation[], +): Pick { + return { + listSessions: async () => ({ object: "list", data: sessions, has_more: false, first_id: session.id, last_id: session.id }), + listRuntimeObservations: async () => ({ object: "list", data: observations, has_more: false, first_id: session.id, last_id: session.id }), + }; +} + +describe("Runtime Dashboard snapshot coordination", () => { + it("publishes only exact Session and observation identity sets", async () => { + const value = await loadRuntimeDashboardSnapshot(core([session], [observation]), () => 4); + expect(value?.sessions).toEqual([session]); + expect(value?.observations).toEqual([observation]); + + await expect(loadRuntimeDashboardSnapshot(core([session], []), () => 4)).rejects.toBeInstanceOf( + RuntimeSnapshotIncompleteError, + ); + }); + + it("discards a candidate when local Session state changes during collection", async () => { + let revision = 1; + const changing = core([session], [observation]); + changing.listRuntimeObservations = async () => { + revision += 1; + return { object: "list", data: [observation], has_more: false, first_id: session.id, last_id: session.id }; + }; + + await expect(loadRuntimeDashboardSnapshot(changing, () => revision)).resolves.toBeNull(); + }); + + it("fails closed when the target budget is exceeded", async () => { + await expect(loadRuntimeDashboardSnapshot(core([session], [observation]), () => 1, undefined, 0)).rejects.toThrow( + "target budget", + ); + }); +}); diff --git a/apps/web/src/features/dashboard/runtime-snapshot.ts b/apps/web/src/features/dashboard/runtime-snapshot.ts new file mode 100644 index 000000000..79d794406 --- /dev/null +++ b/apps/web/src/features/dashboard/runtime-snapshot.ts @@ -0,0 +1,59 @@ +import type { + AgentCore, + AgentSession, + RuntimeObservation, +} from "@agents-core-web/agents-client"; + +import { listAllCollectionPages } from "../../lib/collection-pagination"; + +export const RUNTIME_SNAPSHOT_TARGET_LIMIT = 10_000; +export const RUNTIME_SNAPSHOT_TIMEOUT_MS = 15_000; +export const RUNTIME_SNAPSHOT_REFRESH_MS = 30_000; + +export interface RuntimeDashboardSnapshot { + sessions: AgentSession[]; + observations: RuntimeObservation[]; + loadedAt: number; +} + +export class RuntimeSnapshotIncompleteError extends Error { + constructor(message: string) { + super(message); + this.name = "RuntimeSnapshotIncompleteError"; + } +} + +function identitySet(values: readonly { id: string }[]): Set { + return new Set(values.map((value) => value.id)); +} + +function setsEqual(left: ReadonlySet, right: ReadonlySet): boolean { + if (left.size !== right.size) return false; + for (const value of left) if (!right.has(value)) return false; + return true; +} + +export async function loadRuntimeDashboardSnapshot( + core: Pick, + readSessionRevision: () => number, + signal?: AbortSignal, + targetLimit = RUNTIME_SNAPSHOT_TARGET_LIMIT, +): Promise { + const revision = readSessionRevision(); + const [sessions, observations] = await Promise.all([ + listAllCollectionPages((options) => core.listSessions(options), signal), + listAllCollectionPages((options) => core.listRuntimeObservations(options), signal), + ]); + signal?.throwIfAborted(); + + if (revision !== readSessionRevision()) return null; + if (sessions.length > targetLimit || observations.length > targetLimit) { + throw new RuntimeSnapshotIncompleteError("Runtime snapshot exceeded the Web target budget."); + } + if (!setsEqual(identitySet(sessions), identitySet(observations))) { + throw new RuntimeSnapshotIncompleteError( + "Sessions changed while Runtime observations were loading. The previous complete snapshot was retained.", + ); + } + return { sessions, observations, loadedAt: Date.now() }; +} diff --git a/contracts/agents-api/runtime-observability-api.md b/contracts/agents-api/runtime-observability-api.md index 34394ffd3..4d6345a51 100644 --- a/contracts/agents-api/runtime-observability-api.md +++ b/contracts/agents-api/runtime-observability-api.md @@ -1,9 +1,9 @@ -# Runtime observation API proposal +# Runtime observation API -Status: Phase 2 implemented. The current-snapshot routes, strict +Status: Phase 2 and initial Core Web consumption implemented. The current-snapshot routes, strict `packages/agents-client` projection, and generated `openapi.yaml` contract are -implemented. Core Web integration, historical queries, and lifecycle controls -remain outside this phase. +implemented and consumed by the Dashboard through complete Session/observation +identity joins. Historical queries and lifecycle controls remain outside this phase. This is an Agents Core extension, not an upstream OpenAI Agents resource. The implementation must record that status in the coverage ledger and generated diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md index dd9dfa16d..d49bf3ed5 100644 --- a/contracts/agents-api/runtime-observability-design.md +++ b/contracts/agents-api/runtime-observability-design.md @@ -1,8 +1,9 @@ # Runtime observability and Dashboard design -Status: Phase 1 provider abstraction/Docker sampling and Phase 2 current-snapshot -API/client contract are implemented. Core Web integration, history backend, -additional providers, and lifecycle automation described below are not implemented. +Status: Phase 1 provider abstraction/Docker sampling, Phase 2 current-snapshot +API/client contract, and the initial Core Web current-snapshot Dashboard are +implemented. The history backend, additional providers, and lifecycle automation +described below are not implemented. ## 1. Problem statement @@ -267,7 +268,7 @@ paths, and raw labels are never displayed. ### 11.5 Web implementation shape -The Web change belongs in Core Web, not the Core service repository. It uses +The Web implementation belongs in Core Web, not the Core service layer. It uses `packages/agents-client` as the only Runtime-observation transport and keeps four seams separate: @@ -284,10 +285,12 @@ seams separate: Session detail surface. Trend components are absent unless a later history capability and contract are configured. -The initial refresh cadence is an operator-configured value, not an API guarantee. -Web pauses periodic reads when hidden, refreshes when visibility returns, and adds -jitter so multiple browsers do not synchronize. Filtering is local to the last -complete snapshot and never changes tenant authorization or provider selection. +The initial implementation uses a 30-second Web cadence plus up to five seconds +of jitter and a 15-second whole-refresh budget. These are Web configuration, not +API guarantees. Web pauses periodic reads when hidden, refreshes when visibility +returns, and adds jitter so multiple browsers do not synchronize. Filtering is +local to the last complete snapshot and never changes tenant authorization or +provider selection. ## 12. Token usage boundary diff --git a/docs/web/README.md b/docs/web/README.md index 3703cddeb..c94068ae1 100644 --- a/docs/web/README.md +++ b/docs/web/README.md @@ -11,7 +11,8 @@ credentials or execution into the browser. ## What you can do -- **Operate from one Dashboard** — see loaded Agents, active Sessions, work that needs +- **Operate from one Dashboard** — see loaded Agents, active Sessions, current Runtime + CPU/memory evidence, compute uptime, reported token coverage, work that needs attention, recent activity, and the two common create flows. - **Build reusable Agents** — start from a blank Agent or a practical template, then configure its model, instructions, text behavior, Functions, and HTTP MCP servers. @@ -29,8 +30,10 @@ credentials or execution into the browser. ### Dashboard Dashboard is the starting point. It summarizes the current Agent and Session results, -highlights Sessions that need attention, and links directly to Agent creation or a new -Session. +loads a complete tenant-scoped Runtime observation snapshot, shows current Docker +resource evidence and coverage without inventing missing values, highlights Sessions +that need attention, and links directly to Agent creation or a new Session. Historical +charts remain absent until an operator configures a separate history capability. ### Agents @@ -61,7 +64,7 @@ service is not mistaken for a ready model execution path. | Area | User experience | | --- | --- | -| Dashboard | Agent and Session overview, attention queue, recent activity, quick actions | +| Dashboard | Agent and Session overview, current Runtime CPU/memory/uptime and token coverage, attention queue, recent activity, quick actions | | Agents | Create, search, inspect, edit, delete, use templates, and start Sessions | | Sessions | Durable conversation history, Agent filtering, live events, cancellation, retry and continuation | | Trace | Turn history, usage when reported by Core, command output, Function and patch activity | diff --git a/docs/web/README.zh-CN.md b/docs/web/README.zh-CN.md index f444beddb..a1afb5249 100644 --- a/docs/web/README.zh-CN.md +++ b/docs/web/README.zh-CN.md @@ -10,8 +10,8 @@ Core 部署提供完整的产品界面,同时让凭据和执行能力始终留 ## 可以做什么 -- **通过 Dashboard 统一管理**:查看 Agent、活跃 Session、需要关注的工作、最近活动 - 和常用创建入口。 +- **通过 Dashboard 统一管理**:查看 Agent、活跃 Session、Runtime 当前 CPU/内存 + 证据、计算运行时长、Token 覆盖率、需要关注的工作、最近活动和常用创建入口。 - **创建可复用 Agent**:从空白配置或实用模板开始,设置模型、指令、文本行为、 Function 和 HTTP MCP 服务。 - **运行持久化对话**:创建 Session、发送消息、查看实时事件、重新打开历史工作、 @@ -27,8 +27,10 @@ Core 部署提供完整的产品界面,同时让凭据和执行能力始终留 ### Dashboard -Dashboard 是默认首页,集中展示当前 Agent 和 Session 结果、需要关注的 Session, -并可直接进入创建 Agent 或启动 Session 的流程。 +Dashboard 是默认首页,集中展示当前 Agent 和 Session 结果,加载完整的租户级 Runtime +观测快照,并在不把缺失值伪装成 0 的前提下展示当前 Docker 资源和数据覆盖率。页面也 +展示需要关注的 Session,并可直接进入创建 Agent 或启动 Session 的流程。在运维方配置 +独立历史能力之前,页面不会伪造历史趋势图。 ### Agents @@ -56,7 +58,7 @@ System 展示当前 Core 对 Web 暴露的能力,并区分 API 访问、Vault | 区域 | 用户可以完成的工作 | | --- | --- | -| Dashboard | 查看 Agent/Session 概览、关注队列、最近活动和快捷入口 | +| Dashboard | 查看 Agent/Session 概览、Runtime 当前 CPU/内存/运行时长与 Token 覆盖率、关注队列、最近活动和快捷入口 | | Agents | 创建、搜索、查看、编辑、删除、使用模板并启动 Session | | Sessions | 持久化对话、按 Agent 筛选、实时事件、取消、重试和继续执行 | | Trace | 查看 Turn 历史、Core 报告的 Usage、命令输出、Function 和 Patch 活动 | From 745f51adff0eb68e823353bb50b87ff8251be665 Mon Sep 17 00:00:00 2001 From: sam Date: Tue, 22 Sep 2026 21:49:03 +0800 Subject: [PATCH 5/6] Extend Runtime observability dashboard --- apps/web/package.json | 1 + .../src/features/dashboard/DashboardView.css | 467 +++++++++++++++--- .../features/dashboard/DashboardView.test.tsx | 19 +- .../src/features/dashboard/DashboardView.tsx | 148 +----- .../dashboard/RuntimeObservabilityContent.tsx | 349 +++++++++++++ .../dashboard/dashboard-model.test.ts | 48 ++ .../src/features/dashboard/dashboard-model.ts | 9 +- .../runtime-observability-design.md | 11 +- docs/web/README.md | 6 +- docs/web/README.zh-CN.md | 5 +- pnpm-lock.yaml | 22 + 11 files changed, 852 insertions(+), 233 deletions(-) create mode 100644 apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx diff --git a/apps/web/package.json b/apps/web/package.json index 98bf865e2..2a7f1da60 100644 --- a/apps/web/package.json +++ b/apps/web/package.json @@ -11,6 +11,7 @@ }, "dependencies": { "@agents-core-web/agents-client": "workspace:*", + "@tanstack/react-table": "^8.21.3", "lucide-react": "^1.22.0", "react": "^19.2.7", "react-dom": "^19.2.7", diff --git a/apps/web/src/features/dashboard/DashboardView.css b/apps/web/src/features/dashboard/DashboardView.css index b76218653..13778f0fb 100644 --- a/apps/web/src/features/dashboard/DashboardView.css +++ b/apps/web/src/features/dashboard/DashboardView.css @@ -683,74 +683,302 @@ white-space: nowrap; } -.dashboard-runtime-ledger { - min-width: 960px; +.dashboard-runtime-insights { + display: grid; + grid-template-columns: minmax(0, 1.15fr) minmax(0, .85fr); + border-bottom: 1px solid var(--line); } -.dashboard-runtime-panel:has(.dashboard-runtime-ledger) { - overflow-x: auto; +.dashboard-runtime-insight { + min-width: 0; + padding: 14px; } -.dashboard-runtime-header, -.dashboard-runtime-row { - display: grid; - grid-template-columns: minmax(190px, 1.45fr) minmax(140px, 1fr) minmax(130px, .9fr) minmax(150px, 1fr) minmax(140px, .9fr) minmax(110px, .75fr); - align-items: center; - column-gap: 14px; +.dashboard-runtime-insight + .dashboard-runtime-insight { + border-left: 1px solid var(--line); +} + +.dashboard-runtime-insight > header, +.dashboard-runtime-targets > header { + display: flex; + align-items: flex-start; + justify-content: space-between; + gap: 12px; +} + +.dashboard-runtime-insight h3, +.dashboard-runtime-targets h3, +.dashboard-runtime-insight p, +.dashboard-runtime-targets p { + margin: 0; +} + +.dashboard-runtime-insight h3, +.dashboard-runtime-targets h3 { + font-size: 12px; + font-weight: 600; + line-height: 17px; } -.dashboard-runtime-header { - min-height: 34px; - padding: 7px 14px; +.dashboard-runtime-insight header p, +.dashboard-runtime-targets header p, +.dashboard-runtime-insight header > span, +.dashboard-runtime-targets header > span { color: var(--fg-muted); font-size: 9px; - font-weight: 600; - letter-spacing: .04em; line-height: 13px; +} + +.dashboard-runtime-insight header > span, +.dashboard-runtime-targets header > span { + flex: 0 0 auto; + font-family: var(--font-mono); + font-variant-numeric: tabular-nums; +} + +.dashboard-runtime-health-grid { + display: grid; + margin-top: 12px; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 7px; +} + +.dashboard-runtime-health { + display: grid; + min-width: 0; + padding: 8px 9px 8px 19px; + position: relative; + gap: 1px; + color: var(--fg); + text-align: left; background: var(--surface-subtle); - border-bottom: 1px solid var(--line); - text-transform: uppercase; + border: 1px solid var(--line); + border-radius: 6px; } -.dashboard-runtime-entry + .dashboard-runtime-entry { - border-top: 1px solid var(--line); +.dashboard-runtime-health::before { + width: 6px; + height: 6px; + position: absolute; + top: 12px; + left: 8px; + content: ""; + background: var(--fg-muted); + border-radius: 50%; } -.dashboard-runtime-row { - min-height: 70px; - padding: 11px 14px; - transition: background-color 150ms var(--ease-settle); +.dashboard-runtime-health-observed::before { + background: var(--success); +} + +.dashboard-runtime-health-unavailable::before { + background: var(--warning); } -.dashboard-runtime-row:hover, -.dashboard-runtime-row:focus-within { +.dashboard-runtime-health:hover, +.dashboard-runtime-health:focus-visible { background: var(--hover); + border-color: color-mix(in srgb, var(--accent) 35%, var(--line)); +} + +.dashboard-runtime-health strong, +.dashboard-runtime-health span { + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-health strong { + font-size: 10px; + font-weight: 550; + line-height: 14px; +} + +.dashboard-runtime-health span, +.dashboard-runtime-health-more { + color: var(--fg-muted); + font-size: 9px; + line-height: 13px; +} + +.dashboard-runtime-health-more { + align-self: center; + padding-left: 4px; +} + +.dashboard-runtime-coverage-grid { + display: grid; + margin-top: 12px; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 7px; } -.dashboard-runtime-row > span { +.dashboard-runtime-coverage-grid > div { display: grid; min-width: 0; - gap: 3px; + padding: 9px 10px; + gap: 1px; + background: var(--surface-subtle); + border: 1px solid var(--line); + border-radius: 6px; } -.dashboard-runtime-row small { - overflow: hidden; +.dashboard-runtime-coverage-grid strong { + font-family: var(--font-mono); + font-size: 10px; + font-variant-numeric: tabular-nums; + font-weight: 600; + line-height: 14px; +} + +.dashboard-runtime-coverage-grid span { color: var(--fg-muted); font-size: 9px; line-height: 13px; - text-overflow: ellipsis; - white-space: nowrap; } -.dashboard-runtime-identity button { +.dashboard-runtime-coverage-warning strong { + color: var(--warning); +} + +.dashboard-runtime-coverage-unsupported strong { + color: var(--fg-muted); +} + +.dashboard-runtime-targets { + min-width: 0; +} + +.dashboard-runtime-targets > header { + min-height: 58px; + align-items: center; + padding: 11px 14px; + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-toolbar { + display: flex; + min-height: 48px; + align-items: center; + padding: 8px 14px; + gap: 8px; + background: var(--surface-subtle); + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-toolbar label { + display: flex; + align-items: center; + gap: 6px; + color: var(--fg-muted); + font-size: 9px; +} + +.dashboard-runtime-toolbar input, +.dashboard-runtime-toolbar select { + height: 30px; + color: var(--fg); + font: inherit; + background: var(--surface); + border: 1px solid var(--line); + border-radius: 5px; +} + +.dashboard-runtime-search { + width: min(360px, 100%); + height: 30px; + margin-right: auto; + padding: 0 8px; + background: var(--surface); + border: 1px solid var(--line); + border-radius: 5px; +} + +.dashboard-runtime-search input { + width: 100%; + min-width: 0; + height: auto; + padding: 0; + background: transparent; + border: 0; + outline: 0; +} + +.dashboard-runtime-toolbar select { + min-width: 112px; + padding: 0 24px 0 8px; +} + +.dashboard-runtime-table-scroll { + overflow-x: auto; +} + +.dashboard-runtime-table { + width: 100%; + min-width: 1080px; + border-collapse: collapse; + font-size: 10px; + line-height: 14px; +} + +.dashboard-runtime-table th { + height: 34px; + padding: 0 10px; + color: var(--fg-muted); + font-size: 9px; + font-weight: 600; + text-align: left; + background: var(--surface-subtle); + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-table th button { + display: inline-flex; + align-items: center; + padding: 0; + gap: 4px; + color: inherit; + font: inherit; + background: transparent; + border: 0; +} + +.dashboard-runtime-table td { + height: 68px; + padding: 9px 10px; + vertical-align: middle; + border-bottom: 1px solid var(--line); +} + +.dashboard-runtime-table tbody tr:hover { + background: var(--hover); +} + +.dashboard-runtime-table th:first-child, +.dashboard-runtime-table td:first-child { + width: 220px; + padding-left: 14px; +} + +.dashboard-runtime-target-identity, +.dashboard-runtime-table-value { + display: grid; + min-width: 0; + gap: 2px; +} + +.dashboard-runtime-target-identity { + position: relative; +} + +.dashboard-runtime-target-identity > button { width: fit-content; max-width: 100%; padding: 0; overflow: hidden; color: var(--fg); - font-size: 12px; + font: inherit; font-weight: 600; - line-height: 17px; text-align: left; text-overflow: ellipsis; white-space: nowrap; @@ -758,11 +986,63 @@ border: 0; } -.dashboard-runtime-identity button:hover, -.dashboard-runtime-identity button:focus-visible { +.dashboard-runtime-target-identity > button:hover, +.dashboard-runtime-target-identity > button:focus-visible { color: var(--accent); } +.dashboard-runtime-target-identity > small, +.dashboard-runtime-table-value small { + overflow: hidden; + color: var(--fg-muted); + font-size: 9px; + line-height: 13px; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-target-identity details { + color: var(--fg-muted); + font-size: 9px; +} + +.dashboard-runtime-target-identity summary { + width: fit-content; + cursor: pointer; +} + +.dashboard-runtime-target-identity dl { + display: grid; + width: 320px; + margin: 6px 0 0; + padding: 8px; + position: absolute; + z-index: 2; + gap: 4px; + background: var(--surface); + border: 1px solid var(--line); + border-radius: 6px; + box-shadow: var(--shadow-control); +} + +.dashboard-runtime-target-identity dl > div { + display: grid; + grid-template-columns: 72px minmax(0, 1fr); +} + +.dashboard-runtime-target-identity dt, +.dashboard-runtime-target-identity dd { + margin: 0; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.dashboard-runtime-target-identity dd { + color: var(--fg); + font-family: var(--font-mono); +} + .dashboard-runtime-status { display: inline-flex; width: fit-content; @@ -793,7 +1073,11 @@ background: var(--warning); } -.dashboard-runtime-value strong { +.dashboard-runtime-status-unsupported { + color: var(--fg-muted); +} + +.dashboard-runtime-table-value strong { overflow: hidden; font-family: var(--font-mono); font-size: 12px; @@ -822,64 +1106,49 @@ border-radius: inherit; } -.dashboard-runtime-detail { - padding: 0 14px 10px; +.dashboard-runtime-no-results { + margin: 0; + padding: 20px 14px; color: var(--fg-muted); font-size: 10px; + text-align: center; + border-bottom: 1px solid var(--line); } -.dashboard-runtime-detail summary { - width: fit-content; - padding: 4px 0; - cursor: pointer; - user-select: none; -} - -.dashboard-runtime-detail[open] { - padding-top: 8px; - background: var(--surface-subtle); - border-top: 1px dashed var(--line); -} - -.dashboard-runtime-detail dl { - display: grid; - margin: 8px 0; - grid-template-columns: repeat(3, minmax(0, 1fr)); - gap: 8px 14px; -} - -.dashboard-runtime-detail dl > div { - min-width: 0; -} - -.dashboard-runtime-detail dt { +.dashboard-runtime-pagination { + display: flex; + min-height: 46px; + align-items: center; + justify-content: space-between; + padding: 8px 14px; color: var(--fg-muted); + font-family: var(--font-mono); font-size: 9px; - text-transform: uppercase; } -.dashboard-runtime-detail dd { - margin: 2px 0 0; - overflow: hidden; - color: var(--fg); - font-family: var(--font-mono); - font-size: 10px; - text-overflow: ellipsis; - white-space: nowrap; +.dashboard-runtime-pagination > div { + display: flex; + gap: 6px; } -.dashboard-runtime-detail button { +.dashboard-runtime-pagination button { display: inline-flex; + min-height: 28px; align-items: center; - padding: 4px 7px; + padding: 4px 8px; gap: 4px; - color: var(--accent); - font-size: 10px; - background: transparent; - border: 1px solid color-mix(in srgb, var(--accent) 25%, var(--line)); + color: var(--fg); + font: inherit; + background: var(--surface-subtle); + border: 1px solid var(--line); border-radius: 5px; } +.dashboard-runtime-pagination button:disabled { + cursor: not-allowed; + opacity: .45; +} + @media (max-width: 980px) { .dashboard-primary-grid { grid-template-columns: 1fr; @@ -896,6 +1165,15 @@ .dashboard-runtime-metric:nth-child(n + 3) { border-top: 1px solid var(--line); } + + .dashboard-runtime-insights { + grid-template-columns: 1fr; + } + + .dashboard-runtime-insight + .dashboard-runtime-insight { + border-top: 1px solid var(--line); + border-left: 0; + } } @media (max-width: 760px) { @@ -949,6 +1227,26 @@ .dashboard-metric:nth-child(n + 3) { border-top: 1px solid var(--line); } + + .dashboard-runtime-toolbar { + align-items: stretch; + flex-wrap: wrap; + } + + .dashboard-runtime-search { + width: 100%; + flex-basis: 100%; + margin-right: 0; + } + + .dashboard-runtime-toolbar label:not(.dashboard-runtime-search) { + flex: 1; + } + + .dashboard-runtime-toolbar select { + min-width: 0; + flex: 1; + } } @media (max-width: 520px) { @@ -990,6 +1288,17 @@ border-left: 0; } + .dashboard-runtime-health-grid, + .dashboard-runtime-coverage-grid { + grid-template-columns: 1fr; + } + + .dashboard-runtime-targets > header { + align-items: flex-start; + flex-direction: column; + gap: 4px; + } + .dashboard-panel header p { white-space: normal; } diff --git a/apps/web/src/features/dashboard/DashboardView.test.tsx b/apps/web/src/features/dashboard/DashboardView.test.tsx index 5d70620aa..c0887eba7 100644 --- a/apps/web/src/features/dashboard/DashboardView.test.tsx +++ b/apps/web/src/features/dashboard/DashboardView.test.tsx @@ -245,10 +245,25 @@ describe("Dashboard loaded-result presentation", () => { expect(html).toContain("CPU time / capacity"); expect(html).toContain("1m 13s / 2 cores"); expect(html).toContain("512 MiB / 2.00 GiB"); - expect(html).toContain("Current Runtime observations"); + expect(html).toContain("Runtime health"); + expect(html).toContain("Observed · 2023-11-14 22:14 UTC"); + expect(html).not.toContain("Observed · 10s"); + expect(html).toContain("Usage coverage"); + expect(html).toContain("CPU 1/1"); + expect(html).toContain("Memory 1/1"); + expect(html).toContain("Tokens 1/1"); + expect(html).toContain("Unsupported 0"); + expect(html).toContain("Runtime targets"); + expect(html).toContain("Search Runtime targets"); + expect(html).toContain("All statuses"); + expect(html).toContain("All modes"); + expect(html).toContain(''); + expect(html).toContain("CPU time"); expect(html).toContain("Managed research"); - expect(html).toContain("Identity and sample details"); + expect(html).toContain("Identity"); + expect(html).toContain("Unknown remains unknown, never zero"); expect(html).not.toContain("CPU %"); + expect(html).not.toContain("CPU now"); expect(html).not.toContain("historical chart"); }); diff --git a/apps/web/src/features/dashboard/DashboardView.tsx b/apps/web/src/features/dashboard/DashboardView.tsx index 1ab5f2d87..4fd6cc68c 100644 --- a/apps/web/src/features/dashboard/DashboardView.tsx +++ b/apps/web/src/features/dashboard/DashboardView.tsx @@ -2,13 +2,9 @@ import { AlertTriangle, ArrowRight, Bot, - Cpu, - Gauge, - MemoryStick, MessageSquare, RefreshCw, Rows3, - Server, } from "lucide-react"; import { useMemo, type ReactNode } from "react"; @@ -21,14 +17,11 @@ import { buildRuntimeDashboardModel, dashboardEnvironmentLabel, dashboardStatusLabel, - formatDashboardBytes, - formatDashboardDuration, - formatDashboardTokens, formatDashboardTimestamp, - runtimeObservationStatusLabel, type DashboardCollectionState, type DashboardSessionRow, } from "./dashboard-model"; +import { RuntimeObservabilityContent } from "./RuntimeObservabilityContent"; import type { RuntimeDashboardSnapshot } from "./runtime-snapshot"; import "./DashboardView.css"; @@ -54,109 +47,6 @@ export interface DashboardViewProps { onOpenSession: (sessionId: string) => void; } -function RuntimeMetric({ - icon, - label, - value, - detail, -}: { - icon: ReactNode; - label: string; - value: string; - detail: string; -}) { - return ( -
- - - {label} - {value} - {detail} - -
- ); -} - -function percent(usage: number | null | undefined, limit: number | null | undefined): number | null { - if (typeof usage !== "number" || typeof limit !== "number" || limit <= 0) return null; - return Math.min(100, Math.max(0, usage / limit * 100)); -} - -function RuntimeTable({ - snapshot, - onOpenSession, -}: { - snapshot: RuntimeDashboardSnapshot; - onOpenSession: (sessionId: string) => void; -}) { - const model = buildRuntimeDashboardModel(snapshot.sessions, snapshot.observations); - return ( -
-
- Session / Runtime - Observation - CPU - Memory - Uptime - Tokens -
- {model.rows.map((row) => { - const observation = row.observation; - const memoryPercent = observation.status === "observed" - ? percent(observation.memory?.usage_bytes, observation.memory?.limit_bytes) - : null; - return ( -
-
- - - {row.session.agentLabel} · {dashboardStatusLabel(row.session.status)} - - - - - {observation.mode === "openai_hosted" ? observation.provider_type ?? "Managed" : dashboardEnvironmentLabel(row.session.environmentProfile)} - - - {observation.status === "observed" ? formatDashboardDuration(observation.cpu?.usage_seconds_total ?? null) : "—"} - {observation.status === "observed" && typeof observation.cpu?.capacity_cores === "number" - ? `${observation.cpu.capacity_cores.toLocaleString("en-US")} cores capacity` - : "No current sample"} - - - {observation.status === "observed" ? formatDashboardBytes(observation.memory?.usage_bytes ?? null) : "—"} - {observation.status === "observed" ? `of ${formatDashboardBytes(observation.memory?.limit_bytes ?? null)}` : "No current sample"} - {memoryPercent !== null ? : null} - - - {formatDashboardDuration(row.computeUptimeSeconds)} - {row.allocationAgeSeconds === null ? "Allocation age unavailable" : `${formatDashboardDuration(row.allocationAgeSeconds)} allocation age`} - - - {formatDashboardTokens(row.session.totalTokens)} - {row.session.totalTokens === null ? "Usage not reported" : "Session reported"} - -
-
- Identity and sample details -
-
Session
{observation.session_id}
-
Environment
{observation.environment_id ?? "Not applicable"}
-
Allocation
{observation.instance.allocation_id ?? "Not available"}
-
Provider
{observation.provider_type ?? "Not available"}
-
Resolved
{formatDashboardTimestamp(observation.resolved_at)}
-
Observed
{formatDashboardTimestamp(observation.observed_at)}
-
- -
-
- ); - })} -
- ); -} - function collectionHasSnapshot(state: DashboardCollectionState, hasSnapshot: boolean): boolean { return state === "ready" || hasSnapshot; } @@ -514,38 +404,12 @@ export function DashboardView({

) : ( <> -
- } - label="Active Runtimes" - value={runtimeModel.summary.observedRuntimeCount.toLocaleString("en-US")} - detail={`${runtimeModel.summary.managedRuntimeCount} managed · ${runtimeModel.summary.unavailableRuntimeCount} unavailable`} - /> - } - label="CPU time / capacity" - value={runtimeModel.summary.cpuUsageSecondsTotal === null && runtimeModel.summary.cpuCapacityCores === null - ? "No current sample" - : `${formatDashboardDuration(runtimeModel.summary.cpuUsageSecondsTotal)} / ${runtimeModel.summary.cpuCapacityCores?.toLocaleString("en-US") ?? "—"} cores`} - detail={`${runtimeModel.summary.cpuCoverageCount}/${runtimeModel.summary.observedRuntimeCount} observed Runtimes report CPU`} - /> - } - label="Memory" - value={runtimeModel.summary.memoryUsageBytes === null && runtimeModel.summary.memoryLimitBytes === null - ? "No current sample" - : `${formatDashboardBytes(runtimeModel.summary.memoryUsageBytes)} / ${formatDashboardBytes(runtimeModel.summary.memoryLimitBytes)}`} - detail={`${runtimeModel.summary.memoryCoverageCount}/${runtimeModel.summary.observedRuntimeCount} observed Runtimes report memory`} - /> - } - label="Reported tokens" - value={runtimeModel.summary.totalTokens === null ? "Not reported" : formatDashboardTokens(runtimeModel.summary.totalTokens)} - detail={`${runtimeModel.summary.tokenCoverageCount}/${runtimeModel.summary.sessionCount} Sessions report usage`} - /> -
{runtimeModel.rows.length ? ( - + ) : (

No Session-owned Runtime contexts in this snapshot.

)} diff --git a/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx new file mode 100644 index 000000000..c580856cc --- /dev/null +++ b/apps/web/src/features/dashboard/RuntimeObservabilityContent.tsx @@ -0,0 +1,349 @@ +import { + ChevronDown, + ChevronLeft, + ChevronRight, + ChevronsUpDown, + ChevronUp, + Cpu, + Gauge, + MemoryStick, + Search, + Server, +} from "lucide-react"; +import { useMemo, useState, type ReactNode } from "react"; +import { + flexRender, + getCoreRowModel, + getFilteredRowModel, + getPaginationRowModel, + getSortedRowModel, + useReactTable, + type ColumnDef, + type FilterFn, + type SortingState, +} from "@tanstack/react-table"; + +import { + buildRuntimeDashboardModel, + dashboardEnvironmentLabel, + dashboardStatusLabel, + formatDashboardBytes, + formatDashboardDuration, + formatDashboardTimestamp, + formatDashboardTokens, + runtimeObservationStatusLabel, + type RuntimeDashboardRow, +} from "./dashboard-model"; +import type { RuntimeDashboardSnapshot } from "./runtime-snapshot"; + +const PAGE_SIZE = 10; + +function RuntimeMetric({ + icon, + label, + value, + detail, +}: { + icon: ReactNode; + label: string; + value: string; + detail: string; +}) { + return ( +
+ + + {label} + {value} + {detail} + +
+ ); +} + +function percent(usage: number | null | undefined, limit: number | null | undefined): number | null { + if (typeof usage !== "number" || typeof limit !== "number" || limit <= 0) return null; + return Math.min(100, Math.max(0, usage / limit * 100)); +} + +function coverage(known: number, total: number): string { + if (total === 0) return "No observed Runtimes"; + return `${Math.round(known / total * 100)}% coverage`; +} + +function runtimeModeLabel(row: RuntimeDashboardRow): string { + if (row.observation.mode === "openai_hosted") { + const provider = row.observation.provider_type; + return provider ? `Managed ${provider === "docker" ? "Docker" : provider}` : "Managed"; + } + return dashboardEnvironmentLabel(row.session.environmentProfile); +} + +function observationTimestamp(row: RuntimeDashboardRow, loadedAt: number): number | null { + const timestamp = row.observation.status === "observed" + ? row.observation.observed_at + : row.observation.resolved_at; + if (!Number.isSafeInteger(timestamp) || timestamp < 0 || timestamp > Math.floor(loadedAt / 1_000)) return null; + return timestamp; +} + +function healthDetail(row: RuntimeDashboardRow, loadedAt: number): string { + const status = runtimeObservationStatusLabel(row.observation); + const timestamp = observationTimestamp(row, loadedAt); + return timestamp === null ? status : `${status} · ${formatDashboardTimestamp(timestamp)}`; +} + +function SortHeader({ + label, + sorted, + onClick, +}: { + label: string; + sorted: false | "asc" | "desc"; + onClick: (event: unknown) => void; +}) { + const Icon = sorted === "asc" ? ChevronUp : sorted === "desc" ? ChevronDown : ChevronsUpDown; + return ( + + ); +} + +const runtimeGlobalFilter: FilterFn = (row, _columnId, value) => { + const query = String(value).trim().toLocaleLowerCase(); + if (!query) return true; + const item = row.original; + return [ + item.session.title, + item.session.agentLabel, + item.observation.session_id, + item.observation.environment_id, + item.observation.instance.allocation_id, + item.observation.provider_type, + item.observation.status, + item.observation.reason, + runtimeModeLabel(item), + ].some((candidate) => typeof candidate === "string" && candidate.toLocaleLowerCase().includes(query)); +}; + +function RuntimeTargets({ + rows, + onOpenSession, +}: { + rows: RuntimeDashboardRow[]; + onOpenSession: (sessionId: string) => void; +}) { + const [sorting, setSorting] = useState([]); + const [globalFilter, setGlobalFilter] = useState(""); + const [statusFilter, setStatusFilter] = useState("all"); + const [modeFilter, setModeFilter] = useState("all"); + const filteredRows = useMemo(() => rows.filter((row) => ( + (statusFilter === "all" || row.observation.status === statusFilter) && + (modeFilter === "all" || row.observation.mode === modeFilter) + )), [modeFilter, rows, statusFilter]); + const columns = useMemo[]>(() => [{ + id: "session", + accessorFn: (row) => row.session.title, + header: ({ column }) => undefined)} />, + cell: ({ row }) => { + const item = row.original; + return ( +
+ + {item.session.agentLabel} +
+ Identity +
+
Session
{item.observation.session_id}
+
Environment
{item.observation.environment_id ?? "Not applicable"}
+
Allocation
{item.observation.instance.allocation_id ?? "Not available"}
+
Resolved
{formatDashboardTimestamp(item.observation.resolved_at)}
+
+
+
+ ); + }, + }, { + id: "mode", + accessorFn: runtimeModeLabel, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {runtimeModeLabel(row.original)}, + }, { + id: "status", + accessorFn: (row) => row.observation.status, + header: ({ column }) => undefined)} />, + cell: ({ row }) => ( + + + ), + }, { + id: "cpu", + accessorFn: (row) => row.observation.cpu?.usage_seconds_total ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => { + const cpu = row.original.observation.status === "observed" ? row.original.observation.cpu : null; + return {formatDashboardDuration(cpu?.usage_seconds_total ?? null)}{typeof cpu?.capacity_cores === "number" ? `${cpu.capacity_cores.toLocaleString("en-US")} cores` : "Capacity unknown"}; + }, + }, { + id: "memory", + accessorFn: (row) => row.observation.memory?.usage_bytes ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => { + const memory = row.original.observation.status === "observed" ? row.original.observation.memory : null; + const memoryPercent = percent(memory?.usage_bytes, memory?.limit_bytes); + return ( + + {formatDashboardBytes(memory?.usage_bytes ?? null)} + {memory?.limit_bytes == null ? "Limit unknown" : `of ${formatDashboardBytes(memory.limit_bytes)}`} + {memoryPercent !== null ? : null} + + ); + }, + }, { + id: "uptime", + accessorFn: (row) => row.computeUptimeSeconds ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {formatDashboardDuration(row.original.computeUptimeSeconds)}{row.original.allocationAgeSeconds === null ? "Allocation age unknown" : `${formatDashboardDuration(row.original.allocationAgeSeconds)} allocated`}, + }, { + id: "sessionStatus", + accessorFn: (row) => row.session.status, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {dashboardStatusLabel(row.original.session.status)}, + }, { + id: "tokens", + accessorFn: (row) => row.session.totalTokens ?? -1, + header: ({ column }) => undefined)} />, + cell: ({ row }) => {formatDashboardTokens(row.original.session.totalTokens)}{row.original.session.totalTokens === null ? "Not reported" : "Session reported"}, + }], [onOpenSession]); + const table = useReactTable({ + data: filteredRows, + columns, + state: { sorting, globalFilter }, + onSortingChange: setSorting, + onGlobalFilterChange: setGlobalFilter, + globalFilterFn: runtimeGlobalFilter, + getCoreRowModel: getCoreRowModel(), + getFilteredRowModel: getFilteredRowModel(), + getSortedRowModel: getSortedRowModel(), + getPaginationRowModel: getPaginationRowModel(), + initialState: { pagination: { pageIndex: 0, pageSize: PAGE_SIZE } }, + }); + const visibleRows = table.getFilteredRowModel().rows.length; + + return ( +
+
+
+

Runtime targets

+

Read-only Session navigation · missing measurements remain unknown

+
+ {visibleRows.toLocaleString("en-US")} visible +
+
+ + + +
+
+
+ + {table.getHeaderGroups().map((headerGroup) => ( + + {headerGroup.headers.map((header) => )} + + ))} + + + {table.getRowModel().rows.map((row) => ( + + {row.getVisibleCells().map((cell) => )} + + ))} + +
{flexRender(header.column.columnDef.header, header.getContext())}
{flexRender(cell.column.columnDef.cell, cell.getContext())}
+ {visibleRows === 0 ?

No Runtime targets match these filters.

: null} +
+ {table.getPageCount() > 1 ? ( +
+ Page {table.getState().pagination.pageIndex + 1} of {table.getPageCount()} +
+ + +
+
+ ) : null} + + ); +} + +export function RuntimeObservabilityContent({ + snapshot, + stale, + onOpenSession, +}: { + snapshot: RuntimeDashboardSnapshot; + stale: boolean; + onOpenSession: (sessionId: string) => void; +}) { + const model = useMemo(() => buildRuntimeDashboardModel(snapshot.sessions, snapshot.observations), [snapshot]); + const summary = model.summary; + + return ( + <> +
+ } label="Active Runtimes" value={summary.observedRuntimeCount.toLocaleString("en-US")} detail={`${summary.managedRuntimeCount} managed · ${summary.unavailableRuntimeCount} unavailable`} /> + } label="CPU time / capacity" value={summary.cpuUsageSecondsTotal === null && summary.cpuCapacityCores === null ? "No current sample" : `${formatDashboardDuration(summary.cpuUsageSecondsTotal)} / ${summary.cpuCapacityCores?.toLocaleString("en-US") ?? "—"} cores`} detail={`${summary.cpuCoverageCount}/${summary.observedRuntimeCount} observed Runtimes report CPU`} /> + } label="Memory" value={summary.memoryUsageBytes === null && summary.memoryLimitBytes === null ? "No current sample" : `${formatDashboardBytes(summary.memoryUsageBytes)} / ${formatDashboardBytes(summary.memoryLimitBytes)}`} detail={`${summary.memoryCoverageCount}/${summary.observedRuntimeCount} observed Runtimes report memory`} /> + } label="Reported tokens" value={formatDashboardTokens(summary.totalTokens)} detail={`${summary.tokenCoverageCount}/${summary.sessionCount} Sessions report usage`} /> +
+ +
+
+

Runtime health

Status and observation freshness by Session

{summary.sessionCount} contexts
+
+ {model.rows.slice(0, 8).map((row) => ( + + ))} + {model.rows.length > 8 ? +{model.rows.length - 8} more : null} +
+
+
+

Usage coverage

Unknown remains unknown, never zero

{stale ? "retained snapshot" : "current snapshot"}
+
+
CPU {summary.cpuCoverageCount}/{summary.observedRuntimeCount}{coverage(summary.cpuCoverageCount, summary.observedRuntimeCount)}
+
Memory {summary.memoryCoverageCount}/{summary.observedRuntimeCount}{coverage(summary.memoryCoverageCount, summary.observedRuntimeCount)}
+
Tokens {summary.tokenCoverageCount}/{summary.sessionCount}{coverage(summary.tokenCoverageCount, summary.sessionCount)}
+
Unavailable {summary.unavailableRuntimeCount}current observations
+
Unsupported {summary.unsupportedRuntimeCount}self-hosted or none
+
+
+
+ + + + ); +} diff --git a/apps/web/src/features/dashboard/dashboard-model.test.ts b/apps/web/src/features/dashboard/dashboard-model.test.ts index b6f35bd0b..03a167088 100644 --- a/apps/web/src/features/dashboard/dashboard-model.test.ts +++ b/apps/web/src/features/dashboard/dashboard-model.test.ts @@ -275,6 +275,8 @@ describe("Dashboard loaded-snapshot model", () => { sessionCount: 2, managedRuntimeCount: 1, observedRuntimeCount: 1, + unavailableRuntimeCount: 0, + unsupportedRuntimeCount: 1, cpuUsageSecondsTotal: 3.5, cpuCapacityCores: 2, cpuCoverageCount: 1, @@ -320,4 +322,50 @@ describe("Dashboard loaded-snapshot model", () => { expect(buildRuntimeDashboardModel([stopped], [observation]).rows[0]?.allocationAgeSeconds).toBeNull(); }); + + it("does not count capacity-only or limit-only samples as usage coverage", () => { + const managed = session("11111111-1111-4111-8111-111111111111", { + environment: { + type: "openai_hosted", + id: "33333333-3333-4333-8333-333333333333", + capability_directories: [], + network: { access: "disabled", allowed_domains: [] }, + packages: { npm: [], python: [], system: [] }, + files: [], + plugins: [], + skills: [], + }, + }); + const observation: RuntimeObservation = { + id: managed.id, + object: "agent.runtime_observation", + session_id: managed.id, + environment_id: "33333333-3333-4333-8333-333333333333", + mode: "openai_hosted", + provider_type: "docker", + instance: { + kind: "managed_allocation", + allocation_id: "44444444-4444-4444-8444-444444444444", + device_id: null, + connection_generation: null, + }, + status: "observed", + reason: null, + allocation_created_at: null, + resolved_at: 220, + observed_at: 210, + started_at: null, + cpu: { usage_seconds_total: null, capacity_cores: 2, usage_cores: null, utilization_ratio: null }, + memory: { usage_bytes: null, limit_bytes: 2048 }, + }; + + expect(buildRuntimeDashboardModel([managed], [observation]).summary).toMatchObject({ + cpuUsageSecondsTotal: null, + cpuCapacityCores: 2, + cpuCoverageCount: 0, + memoryUsageBytes: null, + memoryLimitBytes: 2048, + memoryCoverageCount: 0, + }); + }); }); diff --git a/apps/web/src/features/dashboard/dashboard-model.ts b/apps/web/src/features/dashboard/dashboard-model.ts index 89e36e2a4..5c31b55f1 100644 --- a/apps/web/src/features/dashboard/dashboard-model.ts +++ b/apps/web/src/features/dashboard/dashboard-model.ts @@ -58,6 +58,7 @@ export interface RuntimeDashboardSummary { managedRuntimeCount: number; observedRuntimeCount: number; unavailableRuntimeCount: number; + unsupportedRuntimeCount: number; cpuUsageSecondsTotal: number | null; cpuCapacityCores: number | null; cpuCoverageCount: number; @@ -283,6 +284,7 @@ export function buildRuntimeDashboardModel( let managedRuntimeCount = 0; let observedRuntimeCount = 0; let unavailableRuntimeCount = 0; + let unsupportedRuntimeCount = 0; let cpuUsageSecondsTotal = 0; let cpuUsageKnown = false; let cpuUsageSafe = true; @@ -332,7 +334,7 @@ export function buildRuntimeDashboardModel( cpuCapacityKnown = true; } else cpuCapacitySafe = false; } - if (cpuUsage !== null || cpuCapacity !== null) cpuCoverageCount += 1; + if (cpuUsage !== null) cpuCoverageCount += 1; const memoryUsage = safeNonNegativeInteger(observation.memory?.usage_bytes); const memoryLimit = safeNonNegativeInteger(observation.memory?.limit_bytes); @@ -350,9 +352,11 @@ export function buildRuntimeDashboardModel( memoryLimitKnown = true; } else memoryLimitSafe = false; } - if (memoryUsage !== null || memoryLimit !== null) memoryCoverageCount += 1; + if (memoryUsage !== null) memoryCoverageCount += 1; } else if (observation.status === "unavailable") { unavailableRuntimeCount += 1; + } else { + unsupportedRuntimeCount += 1; } if (sessionRow.totalTokens !== null) { @@ -386,6 +390,7 @@ export function buildRuntimeDashboardModel( managedRuntimeCount, observedRuntimeCount, unavailableRuntimeCount, + unsupportedRuntimeCount, cpuUsageSecondsTotal: cpuUsageKnown && cpuUsageSafe ? cpuUsageSecondsTotal : null, cpuCapacityCores: cpuCapacityKnown && cpuCapacitySafe ? cpuCapacityCores : null, cpuCoverageCount, diff --git a/contracts/agents-api/runtime-observability-design.md b/contracts/agents-api/runtime-observability-design.md index d49bf3ed5..f3522662d 100644 --- a/contracts/agents-api/runtime-observability-design.md +++ b/contracts/agents-api/runtime-observability-design.md @@ -244,8 +244,10 @@ as zero. ### 11.2 Runtime table Each row shows Session, Agent/harness when already available from the Session -snapshot, mode, observation status, CPU, memory, compute uptime, Turn state, and -reported tokens. Rows navigate to the existing Session view. No stop, restart, +snapshot, mode, observation status, cumulative CPU time, memory, compute uptime, +Session state, and reported tokens. The semantic table supports local search, +status and mode filters, sortable columns, and bounded pagination over the last +complete snapshot. Rows navigate to the existing Session view. No stop, restart, pause, or delete actions appear in the first release. ### 11.3 Detail view @@ -281,8 +283,9 @@ seams separate: 3. A feature-local state model retains `last_complete`, current refresh status, local filters, and the selected time range. It aborts an overlapping refresh and marks old data stale after a failed or incomplete refresh. -4. Presentational components render summary coverage, the Runtime table, and a - Session detail surface. Trend components are absent unless a later history +4. Presentational components render summary metrics, a compact health matrix, + explicit usage coverage, the Runtime table, and an identity detail surface. + Trend components and chart dependencies are absent unless a later history capability and contract are configured. The initial implementation uses a 30-second Web cadence plus up to five seconds diff --git a/docs/web/README.md b/docs/web/README.md index c94068ae1..398ccacc1 100644 --- a/docs/web/README.md +++ b/docs/web/README.md @@ -32,8 +32,10 @@ credentials or execution into the browser. Dashboard is the starting point. It summarizes the current Agent and Session results, loads a complete tenant-scoped Runtime observation snapshot, shows current Docker resource evidence and coverage without inventing missing values, highlights Sessions -that need attention, and links directly to Agent creation or a new Session. Historical -charts remain absent until an operator configures a separate history capability. +that need attention, and links directly to Agent creation or a new Session. Runtime +health and coverage cards sit above a searchable, filterable, sortable, paginated +semantic table. Historical charts remain absent until an operator configures a +separate history capability. ### Agents diff --git a/docs/web/README.zh-CN.md b/docs/web/README.zh-CN.md index a1afb5249..77b804710 100644 --- a/docs/web/README.zh-CN.md +++ b/docs/web/README.zh-CN.md @@ -29,8 +29,9 @@ Core 部署提供完整的产品界面,同时让凭据和执行能力始终留 Dashboard 是默认首页,集中展示当前 Agent 和 Session 结果,加载完整的租户级 Runtime 观测快照,并在不把缺失值伪装成 0 的前提下展示当前 Docker 资源和数据覆盖率。页面也 -展示需要关注的 Session,并可直接进入创建 Agent 或启动 Session 的流程。在运维方配置 -独立历史能力之前,页面不会伪造历史趋势图。 +展示 Runtime health 与 coverage 卡片,以及支持搜索、状态/模式筛选、排序、分页的语义 +表格;同时展示需要关注的 Session,并可直接进入创建 Agent 或启动 Session 的流程。 +在运维方配置独立历史能力之前,页面不会伪造历史趋势图,也不会引入趋势图组件。 ### Agents diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index dd71da556..d46c38103 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -23,6 +23,9 @@ importers: '@agents-core-web/agents-client': specifier: workspace:* version: link:../../packages/agents-client + '@tanstack/react-table': + specifier: ^8.21.3 + version: 8.21.3(react-dom@19.3.0(react@19.3.0))(react@19.3.0) lucide-react: specifier: ^1.22.0 version: 1.47.0(react@19.3.0) @@ -493,6 +496,17 @@ packages: '@standard-schema/spec@1.1.0': resolution: {integrity: sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==} + '@tanstack/react-table@8.21.3': + resolution: {integrity: sha512-5nNMTSETP4ykGegmVkhjcS8tTLW6Vl4axfEGQN3v0zdHYbK4UfoqfPChclTrJ4EoK9QynqAu9oUf8VEmrpZ5Ww==} + engines: {node: '>=12'} + peerDependencies: + react: '>=16.8' + react-dom: '>=16.8' + + '@tanstack/table-core@8.21.3': + resolution: {integrity: sha512-ldZXEhOBb8Is7xLs01fR3YEc3DERiz5silj8tnGkFZytt1abEvl/GhUmCE0PMLaMPTa3Jk4HbKmRlHmu+gCftg==} + engines: {node: '>=12'} + '@types/chai@5.2.3': resolution: {integrity: sha512-Mw558oeA9fFbv65/y4mHtXDs9bPnFMZAL/jxdPFUpOHHIXX91mcgEHbS5Lahr+pwZFR8A7GQleRWeI6cGFC2UA==} @@ -1763,6 +1777,14 @@ snapshots: '@standard-schema/spec@1.1.0': {} + '@tanstack/react-table@8.21.3(react-dom@19.3.0(react@19.3.0))(react@19.3.0)': + dependencies: + '@tanstack/table-core': 8.21.3 + react: 19.3.0 + react-dom: 19.3.0(react@19.3.0) + + '@tanstack/table-core@8.21.3': {} + '@types/chai@5.2.3': dependencies: '@types/deep-eql': 4.0.2 From 2b8dd07c7be88ac7fc077ddaedfd793150d404de Mon Sep 17 00:00:00 2001 From: sam Date: Tue, 22 Sep 2026 22:01:33 +0800 Subject: [PATCH 6/6] Adapt Runtime observations to deployment providers --- services/agents-api/cmd/server/main.go | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/services/agents-api/cmd/server/main.go b/services/agents-api/cmd/server/main.go index 6c97bf45c..46603fadc 100644 --- a/services/agents-api/cmd/server/main.go +++ b/services/agents-api/cmd/server/main.go @@ -96,12 +96,9 @@ func run() error { } observationSources := map[string]runtimeobs.Source{} if managed != nil { - for key, provider := range managed.Providers { - source, ok := provider.(runtimeobs.Source) - if !ok { - continue - } - observationSources[key] = source + source, ok := managed.Provider.(runtimeobs.Source) + if ok { + observationSources[managed.InstallationID] = source } } resolver, err := runtimeobs.NewResolver(executionStore)