diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index df13826fe..48b2d12f6 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1518,7 +1518,8 @@ replaced; do not carry obsolete compatibility code forward to satisfy this secti Unsupported `low`/`high` and unreadable catalogs fail before model execution; unsupported non-default levels remain an explicit implementation gap. Product requests that omit the native option retain their existing defaults. - Structured output formats remain a separate protocol gap. + Structured output has a separately qualified profile described in + [Structured output execution](#structured-output-execution). - `subagent_control` advertises native subagent tool control. Agents API requires it when resolved `multi_agent.enabled` is false and sends the typed internal `disable_subagents` policy on both new and resumed Turns. Native translation @@ -1907,6 +1908,36 @@ foreign, ambiguous or metadata-only history rejects before model input. The Runt volume and shared Environment/Session binding establish ownership; this lookup cannot select another Session's home or infer ownership from a model response. +### Structured output execution + +The public `text.format={type:"json_schema",schema:{...}}` is resolved with saved +Agent overrides and frozen in the existing Session configuration. Core transports +it in `ExecutionControls.OutputFormat`; it does not append prompt instructions, +validate/retry model answers, repair JSON or select native tool names. Public +admission requires the selected profile's structured-output qualification, and +only requests using this option require the Runtime's `structured_output` and +message-observation capabilities. A capability advertisement does not qualify a +new public combination. Claude advertises this operation only when the installed +SDK bridge reports its `structured_output` feature and the selected Runtime is +not a workspace profile. + +The current qualified path is Claude SDK, `environment:none`, medium verbosity, +single Agent, with optional ordinary function tools and text results. Workspace, +HTTP MCP, Subagent combinations and non-object root schemas remain unqualified. +The SDK uses binary64 JSON numbers: reject execution schemas whose numeric values +would change during that conversion, without narrowing saved Agent storage. +Codex and MiniMax structured output remain explicit execution gaps. + +The Claude adapter passes `outputFormat` to the maintained native SDK and allows +its native `StructuredOutput` terminal tool. A matching live root tool result and +an attributed successful SDK result confirm the final output. Publish the native +`result.result` string unchanged as a completed `final_answer` Message using the +native tool-use ID; the parent assistant ID can already own a prose Item. Do not +publish unvalidated retry candidates or serialize `structured_output` back to +JSON. Native retries remain harness-owned. Existing input receipts, usage, +cancellation, release and recovery rules apply unchanged. The implementation and +qualification limits are recorded in [the coverage note](contracts/agents-api/structured-output.md). + ### Claude SDK adapter foundation `packages/claude-sdk-adapter` privately owns the pinned official TypeScript SDK diff --git a/apps/parsar-daemon/internal/agent/claudesdk/execution_controls_test.go b/apps/parsar-daemon/internal/agent/claudesdk/execution_controls_test.go index b65e22933..f3b1c00b0 100644 --- a/apps/parsar-daemon/internal/agent/claudesdk/execution_controls_test.go +++ b/apps/parsar-daemon/internal/agent/claudesdk/execution_controls_test.go @@ -75,3 +75,27 @@ func TestMCPWithoutEnvironmentNoneRejectedBeforeSetup(t *testing.T) { t.Fatal("MCP reached native setup", err) } } + +func TestStructuredOutputConfigurationReachesNativeUnchanged(t *testing.T) { + root := t.TempDir() + t.Setenv("PARSAR_HOME", root) + config := Config{Entrypoint: filepath.Join(root, "worker"), StateDir: filepath.Join(root, "state")} + schema := json.RawMessage(`{"type":"object","properties":{"n":{"const":9007199254740992}}}`) + request := proto.PromptRequestPayload{RunID: "run", Prompt: "Original input.", ObserveMessages: true, DisableSubagents: true, AgentOptions: map[string]any{"model": "model", "system_prompt": "Original instructions."}, ExecutionControls: &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: "medium", OutputFormat: &proto.OutputFormat{Type: "json_schema", Schema: schema}}} + start, _, err := prepare(config, request) + if err != nil { + t.Fatal(err) + } + if start.OutputFormat == nil || string(start.OutputFormat.Schema) != string(schema) || start.Prompt != request.Prompt || start.SystemPrompt != "Original instructions." { + t.Fatal("native configuration changed") + } + request.ExecutionControls.OutputFormat.Schema = json.RawMessage(`{"type":"object","const":9007199254740993}`) + if _, _, err := prepare(config, request); err == nil { + t.Fatal("lossy schema accepted") + } + request.ExecutionControls.OutputFormat.Schema = schema + request.DisableSubagents = false + if _, _, err := prepare(config, request); err == nil { + t.Fatal("unqualified subagent combination accepted") + } +} diff --git a/apps/parsar-daemon/internal/agent/claudesdk/options.go b/apps/parsar-daemon/internal/agent/claudesdk/options.go index c1e27675f..5a551da65 100644 --- a/apps/parsar-daemon/internal/agent/claudesdk/options.go +++ b/apps/parsar-daemon/internal/agent/claudesdk/options.go @@ -24,6 +24,7 @@ type subagentOptions struct { type startRequest struct { Subagents *subagentOptions `json:"subagents,omitempty"` + OutputFormat *proto.OutputFormat `json:"output_format,omitempty"` Type string `json:"type"` Prompt string `json:"prompt,omitempty"` Model string `json:"model"` @@ -66,6 +67,16 @@ func prepareConfiguration(config Config, req proto.PromptRequestPayload) (startR if controls := req.ExecutionControls; controls != nil && (controls.WebSearch != "disabled" || controls.TextVerbosity != "medium") { return fail("execution controls require disabled web search and medium text verbosity") } + if req.ExecutionControls != nil && req.ExecutionControls.OutputFormat != nil { + format := req.ExecutionControls.OutputFormat + if format.Type != "json_schema" || !req.ObserveMessages || !req.DisableSubagents || config.Workspace != nil || req.MCPHTTPServers != nil { + return fail("structured output requires the qualified message-observing single-agent function profile") + } + if err := proto.ValidateBinary64Schema(format.Schema); err != nil { + return startRequest{}, nil, err + } + start.OutputFormat = format + } if req.ObserveSubagentIdentities { if req.DisableSubagents || len(req.FunctionTools) != 0 || req.MCPHTTPServers != nil || (req.LocalEnvironment != nil && len(req.LocalEnvironment.MCP) != 0) { return fail("subagent execution does not support this tool combination") diff --git a/apps/parsar-daemon/internal/agent/claudesdk/readiness.go b/apps/parsar-daemon/internal/agent/claudesdk/readiness.go index 3accf625a..3d01a10eb 100644 --- a/apps/parsar-daemon/internal/agent/claudesdk/readiness.go +++ b/apps/parsar-daemon/internal/agent/claudesdk/readiness.go @@ -25,6 +25,10 @@ type RuntimeInfo struct { Features []string `json:"features"` } +func (info RuntimeInfo) SupportsStructuredOutput() bool { + return slices.Contains(info.Features, "structured_output") +} + func (info RuntimeInfo) SupportsSubagents() bool { return slices.Contains(info.Features, "subagent_resources") } diff --git a/apps/parsar-daemon/internal/agent/codex/preparation.go b/apps/parsar-daemon/internal/agent/codex/preparation.go index 770f97127..14217bff5 100644 --- a/apps/parsar-daemon/internal/agent/codex/preparation.go +++ b/apps/parsar-daemon/internal/agent/codex/preparation.go @@ -36,6 +36,9 @@ func newSession(parent context.Context, req proto.PromptRequestPayload, out chan } func newPreparation(parent context.Context, req proto.PromptRequestPayload, cfg sessionConfig) (*Prepared, error) { + if req.ExecutionControls != nil && req.ExecutionControls.OutputFormat != nil { + return nil, errors.New("codex: structured output is not qualified") + } if req.WorkspaceReadOnly { return nil, errors.New("codex: workspace reads use the local Runtime interface") } diff --git a/apps/parsar-daemon/internal/agent/mcode/execution.go b/apps/parsar-daemon/internal/agent/mcode/execution.go index 22e77c1ab..d2fbd8765 100644 --- a/apps/parsar-daemon/internal/agent/mcode/execution.go +++ b/apps/parsar-daemon/internal/agent/mcode/execution.go @@ -16,7 +16,7 @@ func validateExecutionRequest(req proto.PromptRequestPayload) error { if !req.ReleaseOnCompletion || !req.DisableExecutionEnvironment || req.WorkDir != "" || req.AgentStateKey == "" || req.LocalEnvironment != nil || req.RequireExistingNativeSession || len(req.FunctionTools) != 0 || (req.MCPHTTPServers != nil && len(*req.MCPHTTPServers) != 0) { return fmt.Errorf("mcode: unsupported execution configuration") } - if req.ExecutionControls == nil || req.ExecutionControls.WebSearch != "disabled" || (req.ExecutionControls.TextVerbosity != "" && req.ExecutionControls.TextVerbosity != "medium") { + if req.ExecutionControls == nil || req.ExecutionControls.OutputFormat != nil || req.ExecutionControls.WebSearch != "disabled" || (req.ExecutionControls.TextVerbosity != "" && req.ExecutionControls.TextVerbosity != "medium") { return fmt.Errorf("mcode: unsupported execution controls") } if !req.DisableSubagents { diff --git a/apps/parsar-daemon/internal/cli/claude_sdk.go b/apps/parsar-daemon/internal/cli/claude_sdk.go index ad3f0a232..8587d23d3 100644 --- a/apps/parsar-daemon/internal/cli/claude_sdk.go +++ b/apps/parsar-daemon/internal/cli/claude_sdk.go @@ -97,6 +97,7 @@ func discoverClaudeSDK(rc *runContext, profile string, check func(context.Contex caps.WorkspaceReadPreparation, caps.NativeSessionRecovery = true, true } out.Info.Available, out.Info.Version = true, info.SDK + out.Info.Capabilities.StructuredOutput = out.Config.Workspace == nil && info.SupportsStructuredOutput() out.Info.Capabilities.SubagentObservations = info.SupportsSubagents() out.Info.Capabilities.MCPHTTPTools = info.SupportsHTTPMCP() out.Info.Capabilities.MCPHTTPBearerAuth = info.SupportsHTTPMCPBearer() diff --git a/apps/parsar-daemon/internal/cli/claude_sdk_test.go b/apps/parsar-daemon/internal/cli/claude_sdk_test.go index 5f16b35fc..a73da6a67 100644 --- a/apps/parsar-daemon/internal/cli/claude_sdk_test.go +++ b/apps/parsar-daemon/internal/cli/claude_sdk_test.go @@ -133,7 +133,7 @@ func TestClaudeSDKFeatureDiscovery(t *testing.T) { t.Fatal(err) } t.Setenv(claudeSDKNodeEnv, node) - for _, features := range [][]string{nil, {"mcp_http_tools"}, {"mcp_http_bearer_auth"}, {"mcp_http_tools", "mcp_http_bearer_auth"}, {"mcp_http_required"}, {"mcp_http_tools", "mcp_http_required"}, {"subagent_resources"}} { + for _, features := range [][]string{nil, {"mcp_http_tools"}, {"mcp_http_bearer_auth"}, {"mcp_http_tools", "mcp_http_bearer_auth"}, {"mcp_http_required"}, {"mcp_http_tools", "mcp_http_required"}, {"subagent_resources"}, {"structured_output"}} { out := discoverClaudeSDK(&runContext{stdout: &strings.Builder{}, stderr: &strings.Builder{}}, "default", func(context.Context, claudesdk.Config) (claudesdk.RuntimeInfo, error) { info := claudesdk.RuntimeInfo{SDK: "0.3.269", Native: "2.1.269 (Claude Code)", Features: features} return info, nil @@ -142,6 +142,9 @@ func TestClaudeSDKFeatureDiscovery(t *testing.T) { if out == nil || !out.Info.Available || out.Info.Capabilities.MCPHTTPTools != supported || out.Info.Capabilities.MCPHTTPBearerAuth != (supported && slices.Contains(features, "mcp_http_bearer_auth")) || out.Info.Capabilities.MCPHTTPRequired != (supported && slices.Contains(features, "mcp_http_required")) { t.Fatal("MCP feature discovery widened the runtime profile") } + if out.Info.Capabilities.StructuredOutput != slices.Contains(features, "structured_output") { + t.Fatal("structured output feature does not match the installed runtime") + } if out.Info.Capabilities.SubagentObservations != slices.Contains(features, "subagent_resources") { t.Fatal("Subagent feature discovery does not match the runtime contract") } diff --git a/contracts/agents-api/README.md b/contracts/agents-api/README.md index 3ec1d8241..c331ddb01 100644 --- a/contracts/agents-api/README.md +++ b/contracts/agents-api/README.md @@ -164,7 +164,7 @@ user-managed enrollment remain outside this qualification. | --- | --- | | Subagents / multi_agent | Six reads and same-child recovery have three-harness Docker evidence; optional native operations, live child progress, full lifecycle/interactions and tool combinations remain explicit gaps | | Environment Templates | Unsupported restricted hostname forms, unqualified installation overrides/null network and exact hosted errors remain gaps. CRUD/list, files, env/setup/system/npm/Python, inline/referenced Skills, Plugins, workspace capability directories and Session references have recorded coverage. Environment Plugin MCP transport and placement limits are [listed separately](environment-templates.md#environment-origin-mcp-plugins) | -| Input and configuration | Non-text initial input, broader content/configuration unions, structured output and reasoning/verbosity combinations | +| Input and configuration | Non-text initial input, broader content/configuration unions and reasoning/verbosity combinations; [structured output](structured-output.md) has a qualified Claude function profile, with other combinations remaining gaps | | Tools and interactions | Deferred functions, other tool types, effective tool-set enforcement and result/cancel publication ordering; MiniMax public functions and service-origin MCP remain unsupported | | Vault and Credentials | OAuth/refresh, archive semantics, revocation/concurrent mutation and exact hosted selection/error behavior; static bearer CRUD/token replacement is already present | | Existing resources | Full Item/SSE/Usage variants, omitted/null/default/error semantics, pagination and overlapping lifecycle behavior beyond recorded cases | @@ -453,7 +453,7 @@ extension separately from upstream fields and document it here when implemented. `openapi.yaml` is our generated supported surface; it is not the full upstream specification. The shared Go wire types are in `v1`. Physical Session cleanup, non-text -message input, structured output execution, broader options/tools, remaining Vault lifecycle, +message input, broader structured-output combinations, broader options/tools, remaining Vault lifecycle, Subagents and environment/provider resources remain incomplete. Reject unsupported requests explicitly; persisted saved configuration is not execution admission. diff --git a/contracts/agents-api/harness-onboarding.md b/contracts/agents-api/harness-onboarding.md index e65df5099..b0586a70f 100644 --- a/contracts/agents-api/harness-onboarding.md +++ b/contracts/agents-api/harness-onboarding.md @@ -81,10 +81,15 @@ An engine without native tools can guarantee their absence; an engine with tools must actually disable them when requested. Configuration acceptance is not proof of enforcement. -MCP, public function calls, image inputs, verbosity controls and other optional +MCP, public function calls, structured output, image inputs, verbosity controls and other optional operations do not need to match another engine. Reject unqualified combinations explicitly and record the gap. Never advertise a capability to bypass selection. +Structured-output adapters consume `ExecutionControls.OutputFormat` and publish +confirmed native output through the existing Message contract. Register public +qualification separately from the Runtime capability; see the +[structured-output boundary](structured-output.md). No Core engine-name branch is required. + Hosted workspace execution additionally requires verified preparation, workspace reads/output export, network behavior and credential/history isolation. Reuse the same dedicated Runtime binding and shared Files helpers. A native Bash sandbox diff --git a/contracts/agents-api/harnesses.md b/contracts/agents-api/harnesses.md index d3dc634f0..651baa4c6 100644 --- a/contracts/agents-api/harnesses.md +++ b/contracts/agents-api/harnesses.md @@ -80,7 +80,9 @@ syntactically or everything either upstream harness can theoretically perform. | Non-default verbosity | Native/model-dependent support | No equivalent qualified; medium only | | Public detailed Usage | Supported native counters | Native raw usage retained; public breakdown gap | | V1 `self_hosted` daemon enrollment at `/workspace` | [Qualified deployment scope](user-managed-runtime-v1.md) | [Qualified deployment scope](user-managed-runtime-v1.md) | -| Explicit reasoning, structured output, enabled `multi_agent`, message images | Shared service gaps | Shared service gaps | +| Structured output | Unqualified; explicit rejection | [Qualified single-agent function profile](structured-output.md) | +| Explicit reasoning, message images | Shared service gaps | Shared service gaps | +| Six Subagent reads | [Qualified scope](subagents.md) | [Qualified scope](subagents.md) | This inventory records supported combinations, not a feature-equality checklist. Do not silently drop options, fabricate measurements, weaken isolation or remove diff --git a/contracts/agents-api/openapi.yaml b/contracts/agents-api/openapi.yaml index 4070ca2c8..3578bf4ff 100644 --- a/contracts/agents-api/openapi.yaml +++ b/contracts/agents-api/openapi.yaml @@ -1598,9 +1598,12 @@ definitions: type: object v1.TextFormat: properties: + schema: + type: object type: enum: - text + - json_schema type: string required: - type @@ -2689,11 +2692,13 @@ paths: Unknown historical creators reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without - replay; retry with stream=false to retrieve the Session. Non-text initial - input remains unsupported. Basic Codex and Claude SDK openai_hosted creation - requires an explicitly configured managed provider. The Claude workspace profile - supports non-deferred function tools with text results alongside native workspace - tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; + replay; retry with stream=false to retrieve the Session. Claude SDK environment:none + supports qualified object-root json_schema output with medium verbosity, single-Agent + execution and ordinary functions; other combinations remain unsupported. Non-text + initial input remains unsupported. Basic Codex and Claude SDK openai_hosted + creation requires an explicitly configured managed provider. The Claude workspace + profile supports non-deferred function tools with text results alongside native + workspace tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; initial provisioning has no caller connection action. Network defaults to enabled; disabled and restricted exact ASCII hostnames are supported. Restricted policy requires 1–100 allowed domains. Unsupported hostname forms and startup diff --git a/contracts/agents-api/structured-output.md b/contracts/agents-api/structured-output.md new file mode 100644 index 000000000..5ec41f8d8 --- /dev/null +++ b/contracts/agents-api/structured-output.md @@ -0,0 +1,62 @@ +# Structured output + +The pinned Agents API accepts `text.format` with `type: "json_schema"` and a +`schema` object. This is not the Responses API wrapper: do not add `name`, `strict` +or an alternative public output field. The schema is saved and inherited through +the existing Agent/Session configuration resolution and immutable snapshot. + +## Qualified execution profile + +Claude SDK supports object-root schemas with `environment:none`, medium verbosity, +`multi_agent.enabled=false` and optional ordinary function tools returning text. +The native SDK remains responsible for its model/tool loop and schema validation. +Codex and MiniMax structured-output execution, workspace/HTTP MCP/Subagent +combinations and other root types are unqualified and explicitly rejected. These +are implementation gaps, not a redefinition of the official protocol. + +The Claude SDK consumes JSON numbers as binary64. Session admission rejects schema +numbers whose values cannot survive that conversion; saved Agent resources still +retain such schemas exactly. This does not claim that the upstream model/harness +preserves arbitrary numeric candidates internally. The adapter never reparses and +serializes a final answer to construct public text. + +## Core and Runtime boundary + +`ExecutionControls.OutputFormat` carries the schema through the shared execution +request. The selected service profile qualifies the operation; `structured_output` +and message observations are checked only for requests using the option. An +advertised capability alone never enables public support. Other harnesses can +implement this same request and existing Message observations without Core +engine-name branches or a second execution loop. + +The Claude adapter passes `outputFormat` to the fixed SDK. Its native +`StructuredOutput` tool is internal and is not an extra caller-defined function. +Successful live root calls are confirmed by their native tool-result receipt and +an attributed successful result containing `structured_output`. The adapter emits +a completed `final_answer` Message with the native tool-use ID and unchanged +`result.result` text. Parent assistant prose retains its own ID. Failed retries +and cancelled candidates cannot become a completed structured answer. No private +history read, output repair, schema coercion or prompt wrapper supplies the result. + +Recovery uses the existing Session/Turn/Items queries and native continuation. +SSE is still live-only. Frozen schemas apply to both initial and resumed execution; +ordinary text configuration retains its prior behavior. + +## Acceptance + +`TestNativeStructuredOutputPublicExecution` and +`services/agents-api/tests/official_structured_output.py` exercise the pinned SDK, +raw HTTP, actual PostgreSQL/Worker/gateway/daemon and real model APIs. They require +explicit private operator options and never supply model responses. The workflow +covers a function-only random value, unchanged saved configuration, native result +application receipts, ordered terminal SSE, persisted final JSON, daemon restart +and same-history continuation, cancellation, text override and tenant isolation. + +Focused tests cover native failure/retry projection, exact result bytes, schema +numeric admission, configuration transport and independent service qualification +for another harness. Running only these tests or importing the SDK is not a claim +of complete protocol compatibility. + +The retained Codex provider/tool-chain failure is not reopened by Claude's native +qualification. Its next attempt needs a concrete changed prerequisite; do not +weaken assertions, rewrite model output or repeatedly sample until one run passes. diff --git a/contracts/agents-api/v1/sessions.go b/contracts/agents-api/v1/sessions.go index 77fee1cfe..045fefde1 100644 --- a/contracts/agents-api/v1/sessions.go +++ b/contracts/agents-api/v1/sessions.go @@ -84,7 +84,8 @@ type TextConfig struct { } type TextFormat struct { - Type string `json:"type" enums:"text" binding:"required"` + Type string `json:"type" enums:"text,json_schema" binding:"required"` + Schema json.RawMessage `json:"schema,omitempty" swaggertype:"object"` } type Session struct { diff --git a/internal/agentdaemon/device/state.go b/internal/agentdaemon/device/state.go index cf510b4fb..7c8c1c8bb 100644 --- a/internal/agentdaemon/device/state.go +++ b/internal/agentdaemon/device/state.go @@ -80,6 +80,7 @@ type KindCapabilities struct { // ExecutionControls supports typed search and verbosity controls. ExecutionControls bool `json:"execution_controls,omitempty"` TextVerbosity bool `json:"text_verbosity,omitempty"` + StructuredOutput bool `json:"structured_output,omitempty"` SubagentControl bool `json:"subagent_control,omitempty"` FunctionTools bool `json:"function_tools,omitempty"` MCPHTTPTools bool `json:"mcp_http_tools,omitempty"` diff --git a/internal/agentdaemon/gateway/session.go b/internal/agentdaemon/gateway/session.go index 901b60014..6adcedf83 100644 --- a/internal/agentdaemon/gateway/session.go +++ b/internal/agentdaemon/gateway/session.go @@ -560,6 +560,7 @@ func deviceKindsFromHeartbeat(p proto.HeartbeatPayload) []device.SupportedAgentK WorkspaceOutputExport: info.Capabilities.WorkspaceOutputExport, WebSearchControl: info.Capabilities.WebSearchControl, TextVerbosity: info.Capabilities.TextVerbosity, + StructuredOutput: info.Capabilities.StructuredOutput, ExecutionControls: info.Capabilities.ExecutionControls, SubagentControl: info.Capabilities.SubagentControl, SubagentObservations: info.Capabilities.SubagentObservations, diff --git a/internal/agentdaemon/proto/inbound.go b/internal/agentdaemon/proto/inbound.go index a5b3cb3f4..1ea1623a8 100644 --- a/internal/agentdaemon/proto/inbound.go +++ b/internal/agentdaemon/proto/inbound.go @@ -272,6 +272,7 @@ type AgentKindCapabilities struct { // ExecutionControls supports typed search and verbosity controls. ExecutionControls bool `json:"execution_controls,omitempty"` TextVerbosity bool `json:"text_verbosity,omitempty"` + StructuredOutput bool `json:"structured_output,omitempty"` SubagentControl bool `json:"subagent_control,omitempty"` DurableInputReceipts bool `json:"durable_input_receipts,omitempty"` // DurableTurns includes strict resume, completion release and cancellation snapshots. diff --git a/internal/agentdaemon/proto/outbound.go b/internal/agentdaemon/proto/outbound.go index 71dee3142..a3edfeb72 100644 --- a/internal/agentdaemon/proto/outbound.go +++ b/internal/agentdaemon/proto/outbound.go @@ -1,5 +1,7 @@ package proto +import "encoding/json" + // Type constants for server → daemon frames. const ( // TypePromptRequest triggers one prompt cycle. Envelope.ID = RunID; @@ -169,6 +171,13 @@ type DeviceShutdownPayload struct { // ExecutionControls requires both values when supplied; omitting the block preserves agent options. // Send only to a peer advertising execution_controls, independently of older option capabilities. type ExecutionControls struct { - WebSearch string `json:"web_search"` - TextVerbosity string `json:"text_verbosity"` + WebSearch string `json:"web_search"` + TextVerbosity string `json:"text_verbosity"` + OutputFormat *OutputFormat `json:"output_format,omitempty"` +} + +// OutputFormat passes the public schema unchanged to a qualified native adapter. +type OutputFormat struct { + Type string `json:"type"` + Schema json.RawMessage `json:"schema"` } diff --git a/internal/agentdaemon/proto/output_format.go b/internal/agentdaemon/proto/output_format.go new file mode 100644 index 000000000..9a29291fd --- /dev/null +++ b/internal/agentdaemon/proto/output_format.go @@ -0,0 +1,54 @@ +package proto + +import ( + "bytes" + "encoding/json" + "errors" + "io" + "math/big" +) + +// ValidateBinary64Schema is an adapter restriction for runtimes that deserialize +// schemas through IEEE-754 JSON numbers. It never rewrites the caller's schema. +// Other adapters need not apply this restriction. +func ValidateBinary64Schema(raw json.RawMessage) error { + fail := errors.New("This runtime requires an object schema with lossless JSON numbers.") + var root struct { + Type string `json:"type"` + } + if json.Unmarshal(raw, &root) != nil || root.Type != "object" { + return fail + } + decoder := json.NewDecoder(bytes.NewReader(raw)) + decoder.UseNumber() + for { + token, err := decoder.Token() + if err == io.EOF { + break + } + if err != nil { + return fail + } + number, ok := token.(json.Number) + if !ok { + continue + } + value, err := number.Float64() + if err != nil { + return fail + } + encoded, err := json.Marshal(value) + if err != nil { + return fail + } + before, ok := new(big.Rat).SetString(number.String()) + if !ok { + return fail + } + after, ok := new(big.Rat).SetString(string(encoded)) + if !ok || before.Cmp(after) != 0 { + return fail + } + } + return nil +} diff --git a/internal/agentdaemon/proto/output_format_test.go b/internal/agentdaemon/proto/output_format_test.go new file mode 100644 index 000000000..3683a16b3 --- /dev/null +++ b/internal/agentdaemon/proto/output_format_test.go @@ -0,0 +1,23 @@ +package proto + +import "testing" + +func TestBinary64SchemaPreservesNumericMeaning(t *testing.T) { + for _, s := range []string{ + `{"type":"object","properties":{"n":{"const":9007199254740992}},"const":0.1}`, + `{"type":"object","const":1e3}`, `{"type":"object","enum":[{"n":-0.0}]}`, + } { + if err := ValidateBinary64Schema([]byte(s)); err != nil { + t.Fatalf("%s: %v", s, err) + } + } + for _, s := range []string{ + `{"type":"object","properties":{"n":{"const":0}},"const":9007199254740993}`, + `{"type":"object","const":1.0000000000000001}`, `{"type":"object","const":1e1000}`, + `{"type":"object","const":1e-1000}`, `{"type":"array"}`, `null`, `{}`, + } { + if err := ValidateBinary64Schema([]byte(s)); err == nil { + t.Fatalf("accepted %s", s) + } + } +} diff --git a/packages/claude-sdk-adapter/src/adapter.ts b/packages/claude-sdk-adapter/src/adapter.ts index 6570eafd3..f3229f3ef 100644 --- a/packages/claude-sdk-adapter/src/adapter.ts +++ b/packages/claude-sdk-adapter/src/adapter.ts @@ -1,3 +1,4 @@ +import { StructuredOutput } from "./structured_output.js"; import { Subagents } from "./subagents.js"; import type { Fact } from "./subagent_history.js"; import { WorkspaceDirectories, type WorkspaceDirectoryEvent } from "./workspace_directories.js"; @@ -65,6 +66,7 @@ export async function execute(request: Start | Prepare, emit: (event: Event) => const resultIDs = new Set(); let failed = false; let cancellationFactsFailed = false; + const structured = request.output_format ? new StructuredOutput() : undefined; const messages = request.observe_messages ? new MessageObserver() : undefined; let stream: ReturnType | undefined; let warm: WarmQuery | undefined; @@ -77,6 +79,7 @@ export async function execute(request: Start | Prepare, emit: (event: Event) => cwd: request.cwd, env: workspace?.options.env ?? { ...process.env, ...(subagents ? { CLAUDE_CODE_DISABLE_BACKGROUND_TASKS: "1" } : {}) }, model: request.model, + ...(request.output_format ? { outputFormat: request.output_format } : {}), systemPrompt: request.system_prompt, ...(request.resume ? { resume: request.resume } : {}), tools: subagents ? ["Agent", "SendMessage"] : [], allowedTools: profile?.allowed ?? names, strictMcpConfig: true, settingSources: [], @@ -128,6 +131,7 @@ export async function execute(request: Start | Prepare, emit: (event: Event) => } else stream = query({ prompt: inputs, options }); for await (const message of stream) { subagents?.consume(message); + structured?.consume(message, nativeID); await functions.consume(message, nativeID); if (mcp) for (const event of mcp.consume(message, nativeID)) await emit(event); if (commands) for (const event of commands.consume(message, nativeID, inputs.hasInput)) await emit(event); @@ -137,7 +141,7 @@ export async function execute(request: Start | Prepare, emit: (event: Event) => if (!nativeID || (request.resume && nativeID !== request.resume)) throw new Error("unexpected native session"); if (workspace) workspace.verify(message.tools, profile ? await stream.mcpServerStatus() : message.mcp_servers, nativeID); else if (profile) profile.verify(message.tools, await stream.mcpServerStatus(), nativeID); - else if (message.tools.length !== names.length + (subagents ? 2 : 0) || message.tools.some(name => ![...names, ...(subagents ? ["Task", "SendMessage"] : [])].includes(name)) || + else if (message.tools.length !== names.length + (subagents ? 2 : 0) + (structured ? 1 : 0) || message.tools.some(name => ![...names, ...(subagents ? ["Task", "SendMessage"] : []), ...(structured ? ["StructuredOutput"] : [])].includes(name)) || message.mcp_servers.length !== (definitions.length ? 1 : 0) || message.mcp_servers.some(server => server.name !== "functions" || server.status !== "connected")) { throw new Error("unexpected native configuration"); @@ -152,6 +156,7 @@ export async function execute(request: Start | Prepare, emit: (event: Event) => await emit({ type: "usage", session_id: nativeID, result_id: message.uuid, usage: resultUsage(message) }); for (const event of inputs.consume(message)) await emit(event); if (message.subtype !== "success" || message.is_error) throw new Error("unsuccessful native result"); + if (structured) await emit(structured.complete(message)); result = { type: "result", session_id: nativeID, text: message.result }; } if (message.type !== "result") for (const event of inputs.consume(message)) await emit(event); diff --git a/packages/claude-sdk-adapter/src/messages.ts b/packages/claude-sdk-adapter/src/messages.ts index 246b50875..7a47fc8bd 100644 --- a/packages/claude-sdk-adapter/src/messages.ts +++ b/packages/claude-sdk-adapter/src/messages.ts @@ -2,7 +2,7 @@ import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk"; export type MessageEvent = | { type: "delta"; delta: string; item_id: string } - | { type: "output_message"; message: { id: string; status: "in_progress" | "completed"; text?: string } }; + | { type: "output_message"; message: { id: string; status: "in_progress" | "completed"; phase?: "final_answer"; text?: string } }; type ActiveMessage = { id: string; diff --git a/packages/claude-sdk-adapter/src/request.ts b/packages/claude-sdk-adapter/src/request.ts index 3d0e6d331..3636db135 100644 --- a/packages/claude-sdk-adapter/src/request.ts +++ b/packages/claude-sdk-adapter/src/request.ts @@ -5,6 +5,7 @@ import { parseWorkspace, type Workspace } from "./workspace.js"; export type Start = { type: "start"; + output_format?: { type: "json_schema"; schema: Record }; prompt: string; model: string; system_prompt: string; @@ -28,7 +29,7 @@ export function parseRequest(line: string): Start | Prepare { const value: unknown = JSON.parse(line); if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("invalid_request"); const request = value as Record; - const allowed = new Set(["type", "prompt", "model", "system_prompt", "cwd", "resume", "require_history", "observe_messages", "subagents", "functions", "mcp_http_servers", "workspace"]); + const allowed = new Set(["type", "prompt", "model", "system_prompt", "cwd", "resume", "require_history", "observe_messages", "output_format", "subagents", "functions", "mcp_http_servers", "workspace"]); if (Object.keys(request).some(key => !allowed.has(key)) || (request.type !== "start" && request.type !== "prepare") || (request.type === "start" ? typeof request.prompt !== "string" || !request.prompt.trim() : "prompt" in request) || @@ -46,6 +47,12 @@ export function parseRequest(line: string): Start | Prepare { if (!value || typeof value !== "object" || Object.keys(value).length !== 1 || !Number.isSafeInteger(value.max_concurrent) || (value.max_concurrent as number) < 1 || (request.functions as unknown[] | undefined)?.length || request.mcp_http_servers !== undefined) throw new Error("invalid_request"); } + if (request.output_format !== undefined) { + const format = request.output_format as Start["output_format"]; + if (!format || format.type !== "json_schema" || Object.keys(format).some(key => !["type", "schema"].includes(key)) || + !format.schema || format.schema.type !== "object" || !request.observe_messages || request.subagents || + request.workspace || request.mcp_http_servers !== undefined) throw new Error("invalid_request"); + } parseHTTPServers(request.mcp_http_servers); const workspace = parseWorkspace(request.workspace, request.cwd); if (request.subagents && workspace?.mcp?.length) throw new Error("invalid_request"); diff --git a/packages/claude-sdk-adapter/src/runtime_check.ts b/packages/claude-sdk-adapter/src/runtime_check.ts index 9935c880b..6f969facc 100644 --- a/packages/claude-sdk-adapter/src/runtime_check.ts +++ b/packages/claude-sdk-adapter/src/runtime_check.ts @@ -45,7 +45,7 @@ try { assert.equal(smoke.error, undefined, "bridge_unavailable"); assert.equal(smoke.status, 0, "bridge_unavailable"); assert.deepEqual(JSON.parse(smoke.stdout), { type: "error", code: "invalid_request" }); - process.stdout.write(JSON.stringify({ type: "runtime_ready", protocol: 1, features: [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions"] : []), "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"], node: process.versions.node, sdk: sdk.version, mcp: mcp.version, native: nativeVersion }) + "\n"); + process.stdout.write(JSON.stringify({ type: "runtime_ready", protocol: 1, features: [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions"] : []), "structured_output", "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"], node: process.versions.node, sdk: sdk.version, mcp: mcp.version, native: nativeVersion }) + "\n"); } catch { // Native diagnostics can include operator environment; never forward them. process.stdout.write(JSON.stringify({ type: "runtime_unavailable" }) + "\n"); diff --git a/packages/claude-sdk-adapter/src/structured_output.ts b/packages/claude-sdk-adapter/src/structured_output.ts new file mode 100644 index 000000000..e39c28c1e --- /dev/null +++ b/packages/claude-sdk-adapter/src/structured_output.ts @@ -0,0 +1,39 @@ +import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk"; +import type { MessageEvent } from "./messages.js"; + +// StructuredOutput is the native terminal tool. Its acknowledged tool-use identity +// owns the final JSON message; the parent assistant may already own ordinary prose. +export class StructuredOutput { + private readonly calls = new Set(); + private accepted?: string; + + consume(message: SDKMessage, session: string): void { + if (!session || message.session_id !== session || + !("parent_tool_use_id" in message) || message.parent_tool_use_id !== null || + ("isReplay" in message && message.isReplay) || ("isSynthetic" in message && message.isSynthetic)) return; + if (message.type === "assistant") { + for (const block of message.message.content) { + if (block.type !== "tool_use" || block.name !== "StructuredOutput") continue; + if (!block.id || this.calls.has(block.id)) throw new Error("invalid structured output identity"); + this.calls.add(block.id); + } + } else if (message.type === "user" && Array.isArray(message.message.content)) { + for (const block of message.message.content) { + if (block.type !== "tool_result" || !this.calls.delete(block.tool_use_id)) continue; + if (!block.is_error) this.accepted = block.tool_use_id; + } + } + } + + complete(message: SDKMessage): MessageEvent { + if (message.type !== "result" || message.subtype !== "success" || message.is_error || + message.structured_output === undefined || !this.accepted || this.calls.size || + typeof message.result !== "string") throw new Error("unconfirmed structured output"); + // Preserve the SDK's final string. Reserializing structured_output can round + // JSON numbers and would replace the native validated result. + JSON.parse(message.result); + const id = this.accepted; + this.accepted = undefined; + return { type: "output_message", message: { id, status: "completed", phase: "final_answer", text: message.result } }; + } +} diff --git a/packages/claude-sdk-adapter/tests/structured_output.test.mjs b/packages/claude-sdk-adapter/tests/structured_output.test.mjs new file mode 100644 index 000000000..4d6d355af --- /dev/null +++ b/packages/claude-sdk-adapter/tests/structured_output.test.mjs @@ -0,0 +1,28 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { StructuredOutput } from "../dist/structured_output.js"; +import { parseStart } from "../dist/request.js"; +const call = (id, extra={}) => ({type:'assistant', session_id:'session', parent_tool_use_id:null, message:{content:[{type:'tool_use',id,name:'StructuredOutput',input:{}}]},...extra}); +const receipt = (id,error=false) => ({type:'user',session_id:'session',parent_tool_use_id:null,message:{content:[{type:'tool_result',tool_use_id:id,is_error:error}]}}); +const result = {type:'result',subtype:'success',is_error:false,structured_output:{number:9007199254740992},result:'{"number":9007199254740993}'}; +test('only the native confirmed terminal output is published, without JSON reserialization',()=>{ + const o=new StructuredOutput(); + o.consume(call('retry'),'session');o.consume(receipt('retry',true),'session'); + o.consume(call('final'),'session');assert.throws(()=>o.complete(result)); + o.consume(receipt('final'),'session'); + assert.deepEqual(o.complete(result),{type:'output_message',message:{id:'final',status:'completed',phase:'final_answer',text:result.result}}); + assert.throws(()=>o.complete(result)); +}); +test('unrelated or failed results cannot publish a candidate',()=>{ + for(const extra of [{isReplay:true},{isSynthetic:true},{parent_tool_use_id:'child'},{session_id:'other'}]){ + const o=new StructuredOutput();o.consume(call('id',extra),'session');o.consume(receipt('id'),'session');assert.throws(()=>o.complete(result)); + } + for(const r of [{...result,subtype:'error_max_structured_output_retries'},{...result,structured_output:undefined},{...result,is_error:true}]){ + const o=new StructuredOutput();o.consume(call('id'),'session');o.consume(receipt('id'),'session');assert.throws(()=>o.complete(r)); + } +}); +test('output configuration is restricted to the qualified profile',()=>{ + const r={type:'start',prompt:'hello',model:'model',system_prompt:'',cwd:'/tmp',observe_messages:true,output_format:{type:'json_schema',schema:{type:'object'}}}; + assert.deepEqual(parseStart(JSON.stringify(r)),r); + for(const change of [{observe_messages:false},{subagents:{max_concurrent:1}},{mcp_http_servers:[]},{output_format:{type:'text',schema:{}}}]) assert.throws(()=>parseStart(JSON.stringify({...r,...change}))); +}); diff --git a/scripts/check-claude-sdk-runtime.mjs b/scripts/check-claude-sdk-runtime.mjs index 6973ac16e..8c353f055 100644 --- a/scripts/check-claude-sdk-runtime.mjs +++ b/scripts/check-claude-sdk-runtime.mjs @@ -23,7 +23,7 @@ assert.equal(probe.status, 0, "Exported runtime is unavailable"); const report = JSON.parse(probe.stdout); assert.equal(report.type, "runtime_ready"); assert.equal(report.protocol, 1); -assert.deepEqual(report.features, [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions"] : []), "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"]); +assert.deepEqual(report.features, [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions"] : []), "structured_output", "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"]); assert.equal(report.sdk, source.dependencies["@anthropic-ai/claude-agent-sdk"]); assert.equal(report.mcp, source.dependencies["@modelcontextprotocol/sdk"]); console.log(`Verified exported SDK ${report.sdk}, MCP ${report.mcp}, ${report.native}`); diff --git a/services/agents-api/internal/api/claude_admission_test.go b/services/agents-api/internal/api/claude_admission_test.go index a04be6dd5..318887457 100644 --- a/services/agents-api/internal/api/claude_admission_test.go +++ b/services/agents-api/internal/api/claude_admission_test.go @@ -20,6 +20,11 @@ func TestClaudeSessionConfigurationAdmission(t *testing.T) { accepted bool }{ {"defaults", `,"instructions":null`, true}, + {"schema", `,"text":{"format":{"type":"json_schema","schema":{"type":"object","properties":{"value":{"type":"string"}}}}}`, true}, + {"lossy schema", `,"text":{"format":{"type":"json_schema","schema":{"type":"object","const":9007199254740993}}}`, false}, + {"array schema", `,"text":{"format":{"type":"json_schema","schema":{"type":"array"}}}`, false}, + {"schema and MCP", `,"text":{"format":{"type":"json_schema","schema":{"type":"object"}}},"tools":[` + publicMCP + `]`, false}, + {"schema and subagents", `,"text":{"format":{"type":"json_schema","schema":{"type":"object"}}},"multi_agent":{"enabled":true}`, false}, {"medium", `,"text":{"verbosity":"medium"}`, true}, {"low", `,"text":{"verbosity":"low"}`, false}, {"high", `,"text":{"verbosity":"high"}`, false}, diff --git a/services/agents-api/internal/api/handler.go b/services/agents-api/internal/api/handler.go index 7d0622ea0..4fa78bdbb 100644 --- a/services/agents-api/internal/api/handler.go +++ b/services/agents-api/internal/api/handler.go @@ -124,7 +124,7 @@ func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ... // createSession atomically reserves or admits initial text with the Session. // @Summary Create an execution Session -// @Description Supports inline configuration or a tenant-owned saved agent_id with per-Session field replacements. Execution supports model/instructions, text verbosity, non-deferred function tools, adapter-qualified multi_agent with persisted Subagent reads, implicit reasoning, service tier auto and environment type none, subject to the configured engine. Codex additionally supports HTTP MCP with explicit service origin, native allowed_tools and boolean required defaulting to false. Session vault_ids attach only project-owned Vaults; credential_id selects an attached static bearer credential for the exact HTTPS URL, while null/omission selects a unique match or remains anonymous. Ambiguous selection rejects creation. Frozen private selections never populate an omitted public credential_id; missing decryption configuration fails dispatch without anonymous fallback. Required initialization uses native startup before the first native Turn, including cold resume, and requires a separately advertised capability; exact hosted creation timing and error parity remain unverified. Other MCP origins and OAuth remain unsupported. The self_hosted profile requires Codex, an absolute workspace_directory and empty capability_directories, with optional non-deferred function tools and HTTP MCP using explicit service origin, optionally authenticated by the attached Vault rules. Remote MCP and remote Bearer authentication each require separately advertised combination support; old peers cannot receive unsupported work. Omitted/null capability_directories use the empty-list default; self_hosted requires configured execution plus executor registry. Claude SDK currently requires medium verbosity and object-root function schemas. It supports anonymous or attached static-bearer service-origin HTTP MCP on none with boolean required and separately advertised MCP/bearer/required runtime support. Required servers must be connected before the first native input is released; pending or failed startup rejects execution. The shared Vault selection and immutable binding rules apply; unsupported native labels/tool names reject before persistence. An attached Vault with no matching credential may remain anonymous; missing keys or failed credential lookup/decryption never fall back to anonymous execution. Omitted stream defaults to false; stream and agent_id cannot be null. Metadata may be null, but its values must be strings. Initial input accepts a string or user-message array containing text. None initial input atomically starts a Turn; self_hosted initial input is reserved while returning its Environment connection target, with execution deferred to native readiness and Session failure on initial timeout. Omitted or null input creates an idle Session. With stream=true, returns live Session events starting at creation; disconnect does not cancel execution. New Sessions retain their authenticated creator; all creation retries require the same typed subject, including across key rotation. Saved-Agent retries and inline requests using Vault attachments or credential references retain caller intent independently of later resource changes; unrelated inline retries preserve resolved/default equivalences. Unknown historical creators reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without replay; retry with stream=false to retrieve the Session. Non-text initial input remains unsupported. Basic Codex and Claude SDK openai_hosted creation requires an explicitly configured managed provider. The Claude workspace profile supports non-deferred function tools with text results alongside native workspace tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; initial provisioning has no caller connection action. Network defaults to enabled; disabled and restricted exact ASCII hostnames are supported. Restricted policy requires 1–100 allowed domains. Unsupported hostname forms and startup installations are rejected. Confidential env, system/npm/Python packages and ordered setup commands use the shared initialization lifecycle; requested network applies after setup. Initial inline and tenant-owned file_id files freeze encrypted bytes before provisioning, then install through the common Core lifecycle before native execution or live Files access. Referenced files/env/packages/setup overrides are rejected pending semantic verification. Tenant-owned environment_template_id references inherit omitted network and allow only narrowing overrides. Referenced network:null is explicitly unsupported pending semantic verification. Core freezes effective configuration; template updates/deletion do not alter Session snapshots or same-intent creation retries. Inline or tenant-owned skill_reference Skills share initialization. Templates preserve default/latest/explicit selectors; Session creation freezes concrete metadata and encrypted content atomically. Skill-list omission inherits and a supplied list replaces; null overrides and null version selectors remain unqualified and reject. Source deletion/default updates cannot change committed Session Skill contents. +// @Description Supports inline configuration or a tenant-owned saved agent_id with per-Session field replacements. Execution supports model/instructions, text verbosity, non-deferred function tools, adapter-qualified multi_agent with persisted Subagent reads, implicit reasoning, service tier auto and environment type none, subject to the configured engine. Codex additionally supports HTTP MCP with explicit service origin, native allowed_tools and boolean required defaulting to false. Session vault_ids attach only project-owned Vaults; credential_id selects an attached static bearer credential for the exact HTTPS URL, while null/omission selects a unique match or remains anonymous. Ambiguous selection rejects creation. Frozen private selections never populate an omitted public credential_id; missing decryption configuration fails dispatch without anonymous fallback. Required initialization uses native startup before the first native Turn, including cold resume, and requires a separately advertised capability; exact hosted creation timing and error parity remain unverified. Other MCP origins and OAuth remain unsupported. The self_hosted profile requires Codex, an absolute workspace_directory and empty capability_directories, with optional non-deferred function tools and HTTP MCP using explicit service origin, optionally authenticated by the attached Vault rules. Remote MCP and remote Bearer authentication each require separately advertised combination support; old peers cannot receive unsupported work. Omitted/null capability_directories use the empty-list default; self_hosted requires configured execution plus executor registry. Claude SDK currently requires medium verbosity and object-root function schemas. It supports anonymous or attached static-bearer service-origin HTTP MCP on none with boolean required and separately advertised MCP/bearer/required runtime support. Required servers must be connected before the first native input is released; pending or failed startup rejects execution. The shared Vault selection and immutable binding rules apply; unsupported native labels/tool names reject before persistence. An attached Vault with no matching credential may remain anonymous; missing keys or failed credential lookup/decryption never fall back to anonymous execution. Omitted stream defaults to false; stream and agent_id cannot be null. Metadata may be null, but its values must be strings. Initial input accepts a string or user-message array containing text. None initial input atomically starts a Turn; self_hosted initial input is reserved while returning its Environment connection target, with execution deferred to native readiness and Session failure on initial timeout. Omitted or null input creates an idle Session. With stream=true, returns live Session events starting at creation; disconnect does not cancel execution. New Sessions retain their authenticated creator; all creation retries require the same typed subject, including across key rotation. Saved-Agent retries and inline requests using Vault attachments or credential references retain caller intent independently of later resource changes; unrelated inline retries preserve resolved/default equivalences. Unknown historical creators reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without replay; retry with stream=false to retrieve the Session. Claude SDK environment:none supports qualified object-root json_schema output with medium verbosity, single-Agent execution and ordinary functions; other combinations remain unsupported. Non-text initial input remains unsupported. Basic Codex and Claude SDK openai_hosted creation requires an explicitly configured managed provider. The Claude workspace profile supports non-deferred function tools with text results alongside native workspace tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; initial provisioning has no caller connection action. Network defaults to enabled; disabled and restricted exact ASCII hostnames are supported. Restricted policy requires 1–100 allowed domains. Unsupported hostname forms and startup installations are rejected. Confidential env, system/npm/Python packages and ordered setup commands use the shared initialization lifecycle; requested network applies after setup. Initial inline and tenant-owned file_id files freeze encrypted bytes before provisioning, then install through the common Core lifecycle before native execution or live Files access. Referenced files/env/packages/setup overrides are rejected pending semantic verification. Tenant-owned environment_template_id references inherit omitted network and allow only narrowing overrides. Referenced network:null is explicitly unsupported pending semantic verification. Core freezes effective configuration; template updates/deletion do not alter Session snapshots or same-intent creation retries. Inline or tenant-owned skill_reference Skills share initialization. Templates preserve default/latest/explicit selectors; Session creation freezes concrete metadata and encrypted content atomically. Skill-list omission inherits and a supplied list replaces; null overrides and null version selectors remain unqualified and reject. Source deletion/default updates cannot change committed Session Skill contents. // @Tags Sessions // @Accept json // @Produce json,text/event-stream diff --git a/services/agents-api/internal/api/session_agent.go b/services/agents-api/internal/api/session_agent.go index dfdb76906..0f7a9ffae 100644 --- a/services/agents-api/internal/api/session_agent.go +++ b/services/agents-api/internal/api/session_agent.go @@ -70,13 +70,11 @@ func admitSessionAgent(cfg v1.SavedAgentConfiguration) (v1.Agent, error) { if cfg.ServiceTier != "auto" { return v1.Agent{}, errors.New("Execution currently supports service_tier=auto only.") } - if cfg.Text.Format.Type != "text" || len(cfg.Text.Format.Schema) > 0 { - return v1.Agent{}, errors.New("Execution currently supports text.format.type=text only.") - } text, err := resolveText(&v1.TextConfigInput{Verbosity: &cfg.Text.Verbosity}) if err != nil { return v1.Agent{}, err } + text.Format = v1.TextFormat{Type: cfg.Text.Format.Type, Schema: cfg.Text.Format.Schema} tools, err := resolveSessionTools(cfg.Tools) if err != nil { return v1.Agent{}, err diff --git a/services/agents-api/internal/api/text_configuration_test.go b/services/agents-api/internal/api/text_configuration_test.go index ad396026d..32b75849c 100644 --- a/services/agents-api/internal/api/text_configuration_test.go +++ b/services/agents-api/internal/api/text_configuration_test.go @@ -4,6 +4,7 @@ import ( "encoding/json" "net/http" "net/http/httptest" + "reflect" "strings" "testing" @@ -36,7 +37,7 @@ func TestTextConfigurationHTTP(t *testing.T) { t.Fatal(err) } want := v1.TextConfig{Format: v1.TextFormat{Type: "text"}, Verbosity: tc.want} - if got.Agent.Text != want || saved.Agent.Text != want { + if !reflect.DeepEqual(got.Agent.Text, want) || !reflect.DeepEqual(saved.Agent.Text, want) { t.Fatal(got.Agent.Text, saved.Agent.Text) } }) diff --git a/services/agents-api/internal/engine/claude.go b/services/agents-api/internal/engine/claude.go index 4698cc113..e6421b2ef 100644 --- a/services/agents-api/internal/engine/claude.go +++ b/services/agents-api/internal/engine/claude.go @@ -11,7 +11,8 @@ import ( func claudeProfile() Profile { return Profile{ - Placements: []string{"none", "openai_hosted", "self_hosted"}, MCPBearer: true, + StructuredOutput: true, + Placements: []string{"none", "openai_hosted", "self_hosted"}, MCPBearer: true, ValidateConfiguration: validateClaudeConfiguration, ValidateTools: validateClaudeTools, ValidateFunctionResult: func(content []proto.FunctionResultContent) error { @@ -32,7 +33,25 @@ func validateClaudeConfiguration(agent v1.Agent, environment *v1.Environment, ha if agent.Text.Verbosity != "" && agent.Text.Verbosity != "medium" { return errors.New("The configured engine currently supports medium text verbosity only.") } - if agent.Reasoning.Effort != nil || agent.Reasoning.Summary != nil || (agent.ServiceTier != "" && agent.ServiceTier != "auto") || (agent.Text.Format.Type != "" && agent.Text.Format.Type != "text") { + if agent.Reasoning.Effort != nil || agent.Reasoning.Summary != nil || (agent.ServiceTier != "" && agent.ServiceTier != "auto") { + return ErrInvalidInput + } + if agent.Text.Format.Type == "json_schema" { + if err := proto.ValidateBinary64Schema(agent.Text.Format.Schema); err != nil { + return err + } + if environment.Type != "none" || agent.MultiAgent.Enabled { + return errors.New("Structured output currently requires a single-agent environment:none profile.") + } + for _, raw := range agent.Tools { + var tool struct { + Type string `json:"type"` + } + if json.Unmarshal(raw, &tool) != nil || tool.Type != "function" { + return ErrInvalidInput + } + } + } else if agent.Text.Format.Type != "" && agent.Text.Format.Type != "text" { return ErrInvalidInput } return rejectSubagentTools(agent, "function", "mcp") diff --git a/services/agents-api/internal/engine/profile.go b/services/agents-api/internal/engine/profile.go index a8549a999..12f4a80ab 100644 --- a/services/agents-api/internal/engine/profile.go +++ b/services/agents-api/internal/engine/profile.go @@ -14,6 +14,7 @@ var ErrInvalidInput = errors.New("invalid engine configuration") type Profile struct { Placements []string WebSearchControl, TextVerbosity, MCPBearer bool + StructuredOutput bool ValidateConfiguration func(agent v1.Agent, environment *v1.Environment, hasDaemon bool) error ValidateTools func(environment *v1.Environment, hasDaemon bool, functions []proto.FunctionTool, mcp []proto.MCPHTTPServer) error ValidateFunctionResult func([]proto.FunctionResultContent) error diff --git a/services/agents-api/internal/execution/engine_profile.go b/services/agents-api/internal/execution/engine_profile.go index b48873889..a9c07b13e 100644 --- a/services/agents-api/internal/execution/engine_profile.go +++ b/services/agents-api/internal/execution/engine_profile.go @@ -16,6 +16,9 @@ func profileError(err error) error { } func validateProfileConfiguration(profile engine.Profile, snapshot Snapshot) error { + if snapshot.Agent.Text.Format.Type == "json_schema" && !profile.StructuredOutput { + return errors.New("Structured output is not qualified for this engine.") + } if profile.ValidateConfiguration != nil { if err := profile.ValidateConfiguration(snapshot.Agent, snapshot.Environment, snapshot.Daemon != nil); err != nil { return profileError(err) diff --git a/services/agents-api/internal/execution/request.go b/services/agents-api/internal/execution/request.go index 89e06e231..2c5ae7286 100644 --- a/services/agents-api/internal/execution/request.go +++ b/services/agents-api/internal/execution/request.go @@ -42,6 +42,9 @@ func (d *Dispatcher) executionRequest(ctx context.Context, session store.Session verbosity = "medium" } controls := &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: verbosity} + if snapshot.Agent.Text.Format.Type == "json_schema" { + controls.OutputFormat = &proto.OutputFormat{Type: "json_schema", Schema: snapshot.Agent.Text.Format.Schema} + } request := proto.PromptRequestPayload{AgentKind: session.Engine, FunctionTools: functions, AgentOptions: options, ExecutionControls: controls, AgentStateKey: "agents-api-" + session.ID, AgentSessionID: bound.NativeSessionID, ReleaseOnCompletion: true, StrictResume: true, diff --git a/services/agents-api/internal/execution/structured_output_test.go b/services/agents-api/internal/execution/structured_output_test.go new file mode 100644 index 000000000..b4b9fda20 --- /dev/null +++ b/services/agents-api/internal/execution/structured_output_test.go @@ -0,0 +1,48 @@ +package execution + +import ( + "context" + "encoding/json" + "testing" + + v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/device" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/engine" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +func TestStructuredOutputNeedsOperationQualification(t *testing.T) { + raw := json.RawMessage(`{"agent":{"model":"model","text":{"format":{"type":"json_schema","schema":{"type":"object"}}}},"environment":{"type":"none"}}`) + for _, kind := range []string{"codex", "claude_sdk", "mcode"} { + if err := (Policy{}).ValidateSessionConfiguration(kind, raw); (err == nil) != (kind == "claude_sdk") { + t.Fatalf("%s: %v", kind, err) + } + } + for _, qualified := range []bool{false, true} { + policy := Policy{Engines: engine.NewCatalog(map[string]engine.Profile{"new_harness": {Placements: []string{"none"}, StructuredOutput: qualified}})} + if err := policy.ValidateSessionConfiguration("new_harness", raw); (err == nil) != qualified { + t.Fatal("new harness did not use common qualification", err) + } + } +} + +func TestStructuredOutputRequestKeepsFrozenSchemaAndInstructions(t *testing.T) { + schema := json.RawMessage(`{"type":"object","properties":{"number":{"const":9007199254740992}}}`) + instructions := "Keep these original instructions." + snapshot := Snapshot{Agent: v1.Agent{Model: "model", Instructions: &instructions, Text: v1.TextConfig{Format: v1.TextFormat{Type: "json_schema", Schema: schema}}}} + request, err := (&Dispatcher{}).executionRequest(context.Background(), store.Session{}, snapshot, device.KindCapabilities{MessageItems: true}, store.SessionExecutionBinding{}) + if err != nil || request.ExecutionControls.OutputFormat == nil { + t.Fatal(err) + } + format := request.ExecutionControls.OutputFormat + if format.Type != "json_schema" || string(format.Schema) != string(schema) || request.AgentOptions["system_prompt"] != &instructions { + t.Fatal("configuration was rewritten") + } + encoded, err := json.Marshal(request) + if err != nil { + t.Fatal(err) + } + if !json.Valid(encoded) { + t.Fatal("invalid request") + } +} diff --git a/services/agents-api/internal/execution/support.go b/services/agents-api/internal/execution/support.go index 2c16aba54..5c187333d 100644 --- a/services/agents-api/internal/execution/support.go +++ b/services/agents-api/internal/execution/support.go @@ -69,10 +69,8 @@ func (p Policy) engineCapabilities(peer *gateway.Session, engine string, snapsho if !ok || (snapshot.Environment != nil && !profile.Accepts(snapshot.Environment.Type)) { return fail("execution engine placement is not supported") } - if profile.ValidateConfiguration != nil { - if err := validateProfileConfiguration(profile, snapshot); err != nil { - return device.KindCapabilities{}, err - } + if err := validateProfileConfiguration(profile, snapshot); err != nil { + return device.KindCapabilities{}, err } info, found, known := peer.AgentKindStatus(engine) caps := info.Capabilities @@ -85,6 +83,9 @@ func (p Policy) engineCapabilities(peer *gateway.Session, engine string, snapsho if profile.WebSearchControl && !caps.WebSearchControl { return fail("device must advertise web_search_control") } + if snapshot.Agent.Text.Format.Type == "json_schema" && (!caps.StructuredOutput || !caps.MessageItems) { + return fail("device must support structured output and message observations") + } if profile.TextVerbosity && !caps.TextVerbosity { return fail("device must advertise text_verbosity") } diff --git a/services/agents-api/internal/store/mcode_public_native_test.go b/services/agents-api/internal/store/mcode_public_native_test.go index 80fbabc4a..e86b4d3c7 100644 --- a/services/agents-api/internal/store/mcode_public_native_test.go +++ b/services/agents-api/internal/store/mcode_public_native_test.go @@ -73,7 +73,7 @@ func TestNativeMCodePublicExecution(t *testing.T) { } server := httptest.NewServer(handler) defer server.Close() - stop := startMCodeDaemon(t, h, home, binary) + stop := startNativeEngineDaemon(t, h, home, binary, "mcode") defer func() { stop() }() evidence := filepath.Join(home, "public.json") run := func(stage string) { @@ -128,7 +128,7 @@ func TestNativeMCodePublicExecution(t *testing.T) { t.Fatal("native binding missing", err) } stop() - stop = startMCodeDaemon(t, h, home, binary) + stop = startNativeEngineDaemon(t, h, home, binary, "mcode") run("resume") after, err := h.s.GetSessionExecutionBinding(ctx, h.tenant, proof.Session) if err != nil || before.NativeSessionID != after.NativeSessionID { @@ -140,7 +140,7 @@ func TestNativeMCodePublicExecution(t *testing.T) { t.Logf("Real mcode common-contract acceptance passed: %s", home) } -func startMCodeDaemon(t *testing.T, h *dispatchHarness, home, binary string) func() { +func startNativeEngineDaemon(t *testing.T, h *dispatchHarness, home, binary, engine string) func() { t.Helper() if h.conn != nil { _ = h.conn.Close() @@ -149,7 +149,7 @@ func startMCodeDaemon(t *testing.T, h *dispatchHarness, home, binary string) fun if err := os.MkdirAll(profile, 0700); err != nil { t.Fatal(err) } - auth, _ := json.Marshal(map[string]string{"server_url": h.url + "/api/v1", "runtime_id": h.device.ID, "runner_credential": h.credential, "device_name": "mcode native proof"}) + auth, _ := json.Marshal(map[string]string{"server_url": h.url + "/api/v1", "runtime_id": h.device.ID, "runner_credential": h.credential, "device_name": "native proof"}) if err := os.WriteFile(filepath.Join(profile, "auth.json"), auth, 0600); err != nil { t.Fatal(err) } @@ -186,7 +186,7 @@ func startMCodeDaemon(t *testing.T, h *dispatchHarness, home, binary string) fun deadline := time.Now().Add(45 * time.Second) for time.Now().Before(deadline) { if peer, err := h.registry.LookupDevice(h.device.ID); err == nil && peer != old { - if info, found, known := peer.AgentKindStatus("mcode"); found && known && info.Available && info.Capabilities.EnvironmentNone { + if info, found, known := peer.AgentKindStatus(engine); found && known && info.Available && info.Capabilities.EnvironmentNone { return stop } } @@ -197,6 +197,6 @@ func startMCodeDaemon(t *testing.T, h *dispatchHarness, home, binary string) fun } time.Sleep(100 * time.Millisecond) } - t.Fatalf("native mcode daemon not ready; evidence %s", home) + t.Fatalf("native daemon not ready; evidence %s", home) return stop } diff --git a/services/agents-api/internal/store/structured_output_dispatch_test.go b/services/agents-api/internal/store/structured_output_dispatch_test.go new file mode 100644 index 000000000..91f2c1109 --- /dev/null +++ b/services/agents-api/internal/store/structured_output_dispatch_test.go @@ -0,0 +1,58 @@ +package store_test + +import ( + "context" + "encoding/json" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/engine" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +func TestStructuredOutputDispatchRechecksOperationQualification(t *testing.T) { + h := newDispatchHarness(t) + configuration := json.RawMessage(`{"agent":{"model":"fixture","text":{"format":{"type":"json_schema","schema":{"type":"object"}}}},"environment":{"type":"none"}}`) + profile := engine.Profile{Placements: []string{"none"}, StructuredOutput: true} + policy := execution.Policy{Engines: engine.NewCatalog(map[string]engine.Profile{"fixture_harness": profile})} + if err := policy.ValidateSessionConfiguration("fixture_harness", configuration); err != nil { + t.Fatal(err) + } + var err error + h.session, err = h.s.CreateSession(t.Context(), h.tenant, store.CreateSessionInput{Creator: store.FixtureCreator(), Engine: "fixture_harness", IdempotencyKey: "structured", Configuration: configuration}) + if err != nil { + t.Fatal(err) + } + if err = h.s.BindSessionDevice(t.Context(), h.tenant, h.session.ID, h.device.ID); err != nil { + t.Fatal(err) + } + caps := proto.AgentKindCapabilities{Streaming: true, Steering: true, DurableTurns: true, DurableInputReceipts: true, ExecutionControls: true, EnvironmentNone: true, SubagentControl: true, ToolObservations: true, StructuredOutput: true, MessageItems: true} + h.write("", proto.TypeHeartbeat, proto.HeartbeatPayload{SupportedAgentKinds: []proto.SupportedAgentKind{{Kind: "fixture_harness", Available: true, Capabilities: caps}}}) + peer, _ := h.registry.LookupDevice(h.device.ID) + for deadline := time.Now().Add(3 * time.Second); ; { + info, found, known := peer.AgentKindStatus("fixture_harness") + if known && found && info.Capabilities.StructuredOutput { + break + } + if time.Now().After(deadline) { + t.Fatal("capability heartbeat missing") + } + time.Sleep(10 * time.Millisecond) + } + input := h.message("start", "Run") + // A restart can remove qualification while admitted work remains queued. + // Runtime advertisements cannot qualify an operation on their own. + profile.StructuredOutput = false + h.d.Policy = execution.Policy{Engines: engine.NewCatalog(map[string]engine.Profile{"fixture_harness": profile})} + ctx, cancel := context.WithTimeout(t.Context(), time.Second) + defer cancel() + if result := <-h.run(ctx, input.TurnID); result.err == nil { + t.Fatal("unqualified structured output dispatched") + } + turn, err := h.s.GetTurn(t.Context(), h.tenant, h.session.ID, input.TurnID) + if err != nil || turn.Status != store.TurnQueued { + t.Fatal("unqualified work was claimed", turn.Status, err) + } +} diff --git a/services/agents-api/internal/store/structured_output_native_test.go b/services/agents-api/internal/store/structured_output_native_test.go new file mode 100644 index 000000000..bd9c34e87 --- /dev/null +++ b/services/agents-api/internal/store/structured_output_native_test.go @@ -0,0 +1,116 @@ +package store_test + +import ( + "context" + "encoding/json" + "net/http/httptest" + "os" + "os/exec" + "path/filepath" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/device" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/api" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/google/uuid" +) + +func TestNativeStructuredOutputPublicExecution(t *testing.T) { + python, binary, root, optionsFile := os.Getenv("PARSAR_OFFICIAL_SDK_PYTHON"), os.Getenv("PARSAR_NATIVE_DAEMON_BIN"), os.Getenv("PARSAR_NATIVE_PROOF_DIR"), os.Getenv("PARSAR_STRUCTURED_OUTPUT_REAL_OPTIONS") + if python == "" || binary == "" || root == "" || optionsFile == "" { + t.Skip("native daemon, fixed SDK, real model options and evidence directory required") + } + raw, err := os.ReadFile(optionsFile) + if err != nil { + t.Fatal(err) + } + var options map[string]any + if json.Unmarshal(raw, &options) != nil { + t.Fatal("invalid private options") + } + model, _ := options["model"].(string) + if model == "" { + t.Fatal("real model required") + } + h := newDispatchHarness(t) + h.d.Options = func(context.Context, store.Session) (map[string]any, error) { return options, nil } + home, err := os.MkdirTemp(root, "structured-public-") + if err != nil { + t.Fatal(err) + } + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Minute) + defer cancel() + worker, err := execution.StartWorker(ctx, h.d) + if err != nil { + t.Fatal(err) + } + done := make(chan error, 1) + go func() { done <- worker.Run(ctx) }() + defer func() { + cancel() + select { + case <-done: + case <-time.After(20 * time.Second): + t.Error("worker did not stop") + } + }() + token, foreign := uuid.NewString(), uuid.NewString() + auth, err := api.NewAuthenticator([]api.APIKey{ + {OrganizationID: "test", ProjectID: h.tenant, SubjectKind: "service_account", SubjectID: "owner", TokenSHA256: device.HashCredential(token), TenantID: h.tenant}, + {OrganizationID: "test", ProjectID: uuid.NewString(), SubjectKind: "service_account", SubjectID: "other", TokenSHA256: device.HashCredential(foreign), TenantID: uuid.NewString()}, + }) + if err != nil { + t.Fatal(err) + } + handler, err := api.NewHandler(h.s, auth, "claude_sdk", api.WithExecution(worker)) + if err != nil { + t.Fatal(err) + } + server := httptest.NewServer(handler) + defer server.Close() + stop := startNativeEngineDaemon(t, h, home, binary, "claude_sdk") + defer func() { stop() }() + evidence := filepath.Join(home, "public.json") + run := func(stage string) { + cmd := exec.CommandContext(ctx, python, "../../tests/official_structured_output.py", server.URL, token, foreign, model, stage, evidence) + if output, err := cmd.CombinedOutput(); err != nil { + t.Fatalf("structured output %s: %v %s; evidence %s", stage, err, output, home) + } + } + run("initial") + var proof struct { + Session string `json:"session"` + Turn string `json:"turn"` + Call string `json:"call"` + } + raw, err = os.ReadFile(evidence) + if err != nil || json.Unmarshal(raw, &proof) != nil { + t.Fatal("invalid evidence", err) + } + call, err := h.s.GetFunctionCall(ctx, h.tenant, proof.Session, proof.Turn, proof.Call) + if err != nil || !call.Applied { + t.Fatal("function application receipt missing", err) + } + turn, err := h.s.GetTurn(ctx, h.tenant, proof.Session, proof.Turn) + if err != nil { + t.Fatal(err) + } + var outcome execution.Result + if json.Unmarshal(turn.Outcome, &outcome) != nil || outcome.Done.Usage.Raw["claude_sdk_result"] == nil || outcome.AppliedThrough < 1 { + t.Fatal("native usage or input receipt missing") + } + before, err := h.s.GetSessionExecutionBinding(ctx, h.tenant, proof.Session) + if err != nil || before.NativeSessionID == "" { + t.Fatal("native binding missing", err) + } + stop() + stop = startNativeEngineDaemon(t, h, home, binary, "claude_sdk") + run("resume") + after, err := h.s.GetSessionExecutionBinding(ctx, h.tenant, proof.Session) + if err != nil || before.NativeSessionID != after.NativeSessionID { + t.Fatal("native history changed", err) + } + t.Logf("Real structured output SDK/raw HTTP, function receipt, cold daemon recovery and isolation passed: %s", home) +} diff --git a/services/agents-api/tests/official_structured_output.py b/services/agents-api/tests/official_structured_output.py new file mode 100644 index 000000000..fe8fb90a5 --- /dev/null +++ b/services/agents-api/tests/official_structured_output.py @@ -0,0 +1,140 @@ +"""Real-model structured output through the pinned SDK and public HTTP surface.""" +import importlib.metadata +import json +import sys +import uuid +from pathlib import Path + +import httpx2 +from openai import BadRequestError, NotFoundError, OpenAI + +base, token, foreign, model, stage, evidence = sys.argv[1:] +pin = json.loads((Path(__file__).resolve().parents[3] / "contracts/agents-api/upstream.json").read_text()) +source = json.loads(importlib.metadata.distribution("openai").read_text("direct_url.json")) +assert source["vcs_info"]["commit_id"] == pin["commit"] +client = OpenAI(base_url=base + "/v1", api_key=token, max_retries=0, + _strict_response_validation=True, http_client=httpx2.Client(trust_env=False, timeout=150)) +other = client.with_options(api_key=foreign) +sessions = client.beta.agents.sessions +headers = {"Authorization": "Bearer " + token, "OpenAI-Beta": "agents=v1"} +proof = {} if stage == "initial" else json.loads(Path(evidence).read_text()) + + +def message(text): + return {"type": "agent.session.input.message", "input": [{"role": "user", "content": [{"type": "input_text", "text": text}]}]} + + +def run(session, prompt, respond=False, cancel=False): + events, handled = [], False + with sessions.events.stream(session, timeout=150) as stream: + sessions.events.create(session, events=[message(prompt)], idempotency_key=str(uuid.uuid4())) + for event in stream: + events.append(event.to_dict()) + if event.type == "agent.session.requires_action": + assert not handled and (respond or cancel), event.to_dict() + handled = True + action = event.session.required_actions[0] + assert action.name == "remember" + if cancel: + sessions.events.create(session, events=[{"type": "agent.session.input.cancel"}]) + else: + proof.update(turn=action.turn_id, call=action.call_id) + payload = {"type": "agent.session.input.tool_result", "turn_id": action.turn_id, + "call_id": action.call_id, "success": True, + "output": [{"type": "input_text", "text": proof["memory"]}]} + for _ in range(2): + sessions.events.create(session, events=[payload], idempotency_key="same-tool-result") + assert event.type not in {"agent.session.failed", "agent.session.turn.failed"}, event.to_dict() + if event.type == "agent.session.idle": + break + else: + raise AssertionError("stream closed without idle") + proof.setdefault("runs", []).append(events) + Path(evidence).write_text(json.dumps(proof, ensure_ascii=False, indent=2)) + assert len({e["event_id"] for e in events}) == len(events) + turns = sessions.turns.list(session, order="asc", limit=100).data + assert turns[-1].status == ("cancelled" if cancel else "completed"), turns[-1].to_dict() + # Claude retains native counters; the public breakdown remains unqualified. + assert turns[-1].usage is None + terminal = "agent.session.turn." + ("cancelled" if cancel else "completed") + assert [e["type"] for e in events].index(terminal) < [e["type"] for e in events].index("agent.session.idle") + return events, turns[-1] + + +def final(session, events=None): + items = sessions.items.list(session, order="asc", limit=100).data + answers = [item for item in items if item.type == "message" and item.role == "assistant" and item.phase == "final_answer"] + assert answers + text = answers[-1].content[0].text + assert json.loads(text) == {"memory": proof["memory"]}, text + with httpx2.Client(trust_env=False) as raw: + response = raw.get(base + "/v1/agents/sessions/" + session + "/items", headers=headers, params={"order": "asc", "limit": 100}) + assert response.status_code == 200 + assert any(i["id"] == answers[-1].id and i["content"][0]["text"] == text for i in response.json()["data"]) + if events is not None: + item_id = answers[-1].id + added = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.item.added" and e["item"]["id"] == item_id) + done = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.item.done" and e["item"]["id"] == item_id) + terminal = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.completed") + assert added < done < terminal + assert next(e["text"] for e in events if e["type"] == "agent.session.turn.output_text.done" and e["item_id"] == item_id) == text + return answers[-1].id + + +try: + if stage == "initial": + schema = {"type": "object", "properties": {"memory": {"type": "string"}}, "required": ["memory"], "additionalProperties": False} + config = {"model": model, "text": {"format": {"type": "json_schema", "schema": schema}}, + "tools": [{"type": "function", "name": "remember", "description": "Return a private memory value.", + "parameters": {"type": "object", "properties": {}, "additionalProperties": False}}]} + saved = client.beta.agents.create(**config) + session = sessions.create(agent_id=saved.id, environment={"type": "none"}) + proof.update(session=session.id, agent=saved.id, memory=str(uuid.uuid4()), format=config["text"]["format"]) + assert session.agent.text.format.to_dict() == proof["format"] + events, turn = run(session.id, "Call remember exactly once and return its exact memory value as the requested JSON. Do not invent it.", respond=True) + proof.update(first_events=events, first_output=final(session.id, events)) + assert len([i for i in sessions.items.list(session.id, limit=100).data if i.type == "function_call"]) == 1 + try: + other.beta.agents.sessions.retrieve(session.id) + raise AssertionError("cross-tenant read accepted") + except NotFoundError: + pass + try: + other.beta.agents.sessions.create(agent_id=saved.id, environment={"type": "none"}) + raise AssertionError("cross-tenant Agent reference accepted") + except NotFoundError: + pass + # Saving arbitrary schemas remains separate from runtime qualification. + huge = client.beta.agents.create(model=model, text={"format": {"type": "json_schema", "schema": {"type": "object", "const": 9007199254740993}}}) + assert huge.text.format.to_dict()["schema"]["const"] == 9007199254740993 + try: + sessions.create(agent_id=huge.id, environment={"type": "none"}) + raise AssertionError("lossy runtime schema accepted") + except BadRequestError: + pass + client.beta.agents.delete(huge.id) + else: + session = sessions.retrieve(proof["session"]) + assert session.agent.text.format.to_dict() == proof["format"] + assert final(session.id) == proof["first_output"] + events, turn = run(session.id, "Recall the exact memory from the previous turn and return it using the same JSON format. Do not call remember again.") + proof.update(resume_events=events, resumed_output=final(session.id, events), resumed_turn=turn.id) + assert proof["resumed_output"] != proof["first_output"] + assert len(sessions.turns.list(session.id).data) == 2 + assert len([i for i in sessions.items.list(session.id, limit=100).data if i.type == "function_call"]) == 1 + cancelled = sessions.create(agent_id=proof["agent"], environment={"type": "none"}) + events, _ = run(cancelled.id, "Call remember to obtain the memory, then return it as JSON.", cancel=True) + assert not any(i.type == "message" and i.role == "assistant" and i.phase == "final_answer" for i in sessions.items.list(cancelled.id, limit=100).data) + proof["cancel_events"] = events + plain = sessions.create(agent_id=proof["agent"], agent={"text": {"format": {"type": "text"}}}, environment={"type": "none"}) + events, _ = run(plain.id, "Do not use tools. Say PLAIN_OK in ordinary text.") + assert any(i.type == "message" and i.role == "assistant" and "PLAIN_OK" in i.content[0].text for i in sessions.items.list(plain.id, limit=100).data) + proof["plain_events"] = events + sessions.delete(cancelled.id) + sessions.delete(plain.id) + proof["passed"] = True + Path(evidence).write_text(json.dumps(proof, ensure_ascii=False, indent=2)) +finally: + Path(evidence).write_text(json.dumps(proof, ensure_ascii=False, indent=2)) + client.close() + other.close()