diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 0f2addedc..7f9154ecc 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -80,6 +80,14 @@ databases, credentials and migrations. The product uses Core exclusively; it has field presence, nullability, discriminators, defaults, status transitions, pagination, errors and streaming behavior. Engine limitations are implementation gaps to solve, not grounds for narrowing or redefining the upstream contract. +- Preserve qualified native capability differences across harnesses. If a material + difference from the official API has no clear mapping, pause that part and ask + the user before changing its semantics. Explicit unsupported enablement rejects; + ordinary requests retain native behavior with any official default discrepancy + recorded in the coverage ledger. In particular, native programmatic tool calling + is not currently qualified as the official default-on behavior. Do not build + a separate executor or model loop to fabricate parity. This does not relax + authentication, isolation, credential protection or data consistency. - Pin upstream source and SDK versions in `contracts/agents-api/upstream.json`. Use official SDKs for clients and reuse upstream types or schemas where suitable. SDK deserialization alone is not server validation or proof of compatibility: @@ -1531,6 +1539,18 @@ replaced; do not carry obsolete compatibility code forward to satisfy this secti both valid fields. This internal contract does not add public configuration or engine support. Future native adapters must verify the same semantics before advertising the capability. +- Explicit public `programmatic_tool_calling.enabled=false` uses the common + `ExecutionControls.DisableProgrammaticToolCalling` field and the operation-specific + `programmatic_tool_calling_disable` capability. Both public qualification and + Runtime support are required for that request; omission creates no prerequisite. + New and resumed executions retain the frozen setting. Codex disables native + code-mode features and checks managed requirements before starting/resuming a + thread, rejecting a conflicting requirement. Claude and MiniMax retain their + restricted native inventories, which exclude programmatic execution. This does + not remove unrelated native utilities or claim enabled programmatic support. + Explicit `web_search.mode=disabled` reuses the existing disabled search control. + Search remains off when omitted. Optional search settings are resource data and + do not cause execution while disabled. Enabled search remains unqualified here. - `web_search_control` advertises the Codex adapter's explicit `web_search` option (`disabled`, `cached`, or `live`). Agents API requires this capability before Codex dispatch; the typed execution controls force search off on new and resumed Turns. Native configuration translation stays in the diff --git a/apps/parsar-daemon/internal/agent/codex/preparation.go b/apps/parsar-daemon/internal/agent/codex/preparation.go index cf4bd4980..a46680c08 100644 --- a/apps/parsar-daemon/internal/agent/codex/preparation.go +++ b/apps/parsar-daemon/internal/agent/codex/preparation.go @@ -123,6 +123,11 @@ func newPreparation(parent context.Context, req proto.PromptRequestPayload, cfg if _, err := rpc.Start(cancelCtx, initParams); err != nil { return p.preparationFailed(fmt.Errorf("codex: rpc start: %w", err)) } + if req.ExecutionControls != nil && req.ExecutionControls.DisableProgrammaticToolCalling { + if err := verifyProgrammaticToolsDisabled(cancelCtx, rpc); err != nil { + return p.preparationFailed(err) + } + } if req.DisableExecutionEnvironment { if err := verifyNoExecutionEnvironment(cancelCtx, rpc); err != nil { cancelFn() diff --git a/apps/parsar-daemon/internal/agent/codex/programmatic_tools.go b/apps/parsar-daemon/internal/agent/codex/programmatic_tools.go new file mode 100644 index 000000000..4ed615b6c --- /dev/null +++ b/apps/parsar-daemon/internal/agent/codex/programmatic_tools.go @@ -0,0 +1,51 @@ +package codex + +import ( + "context" + "encoding/json" + "errors" + "slices" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" +) + +var programmaticFeatures = []string{"code_mode", "code_mode_only", "code_mode_prewarm"} + +func disableProgrammaticTools(plan *SessionPlan, controls *proto.ExecutionControls) { + if controls == nil || !controls.DisableProgrammaticToolCalling { + return + } + for _, feature := range programmaticFeatures { + plan.EnableFeatures = slices.DeleteFunc(plan.EnableFeatures, func(value string) bool { return value == feature }) + if !slices.Contains(plan.DisableFeatures, feature) { + plan.DisableFeatures = append(plan.DisableFeatures, feature) + } + plan.ExtraConfig = append(plan.ExtraConfig, [2]string{"features." + feature, "false"}) + } +} + +// Managed native requirements can override command-line feature settings. +// Check before starting or resuming a native Session, not after model execution. +func verifyProgrammaticToolsDisabled(ctx context.Context, rpc *JSONRPCClient) error { + raw, err := rpc.Request(ctx, "configRequirements/read", nil) + var response struct { + Requirements json.RawMessage `json:"requirements"` + } + if err != nil || json.Unmarshal(raw, &response) != nil || len(response.Requirements) == 0 { + return errors.New("codex: programmatic tool requirements unavailable") + } + if string(response.Requirements) != "null" { + var requirements struct { + Features map[string]bool `json:"featureRequirements"` + } + if json.Unmarshal(response.Requirements, &requirements) != nil { + return errors.New("codex: invalid programmatic tool requirements") + } + for _, feature := range programmaticFeatures { + if requirements.Features[feature] { + return errors.New("codex: required programmatic tools conflict with explicit disabling") + } + } + } + return nil +} diff --git a/apps/parsar-daemon/internal/agent/codex/programmatic_tools_test.go b/apps/parsar-daemon/internal/agent/codex/programmatic_tools_test.go new file mode 100644 index 000000000..8281ac9d6 --- /dev/null +++ b/apps/parsar-daemon/internal/agent/codex/programmatic_tools_test.go @@ -0,0 +1,75 @@ +package codex + +import ( + "context" + "encoding/json" + "reflect" + "slices" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" +) + +func TestProgrammaticToolsExplicitDisableOverridesNativeOptions(t *testing.T) { + plan := SessionPlan{EnableFeatures: []string{"code_mode", "unrelated", "code_mode_only", "code_mode_prewarm"}, ExtraConfig: [][2]string{{"features.code_mode", "true"}}} + before := slices.Clone(plan.EnableFeatures) + disableProgrammaticTools(&plan, nil) + if !reflect.DeepEqual(plan.EnableFeatures, before) { + t.Fatal("omission changed native configuration") + } + disableProgrammaticTools(&plan, &proto.ExecutionControls{DisableProgrammaticToolCalling: true}) + if !slices.Equal(plan.EnableFeatures, []string{"unrelated"}) { + t.Fatal(plan.EnableFeatures) + } + for _, feature := range programmaticFeatures { + if !slices.Contains(plan.DisableFeatures, feature) { + t.Fatal("missing disable", feature) + } + value := "" + for _, config := range plan.ExtraConfig { + if config[0] == "features."+feature { + value = config[1] + } + } + if value != "false" { + t.Fatal("final override missing", feature) + } + } +} + +func TestProgrammaticToolsRejectManagedOverridesBeforeExecution(t *testing.T) { + for _, tc := range []struct { + response string + accepted bool + }{ + {`{"requirements":null}`, true}, + {`{"requirements":{"featureRequirements":{"code_mode":false,"unrelated":true}}}`, true}, + {`{"requirements":{"featureRequirements":{"code_mode":true}}}`, false}, + {`{"requirements":{"featureRequirements":{"code_mode_only":true}}}`, false}, + {`{"requirements":{"featureRequirements":{"code_mode_prewarm":true}}}`, false}, + {`{}`, false}, {`null`, false}, {`{"requirements":[]}`, false}, + } { + t.Run(tc.response, func(t *testing.T) { + client, server, cleanup := NewTestClient() + defer cleanup() + ctx, cancel := context.WithTimeout(t.Context(), time.Second) + defer cancel() + done := make(chan error, 1) + go func() { done <- verifyProgrammaticToolsDisabled(ctx, client.JSONRPCClient) }() + var request struct{ ID, Method string } + if err := json.NewDecoder(server.FromClient).Decode(&request); err != nil { + t.Fatal(err) + } + if request.Method != "configRequirements/read" { + t.Fatal(request.Method) + } + if err := json.NewEncoder(server.ToClient).Encode(map[string]any{"id": request.ID, "result": json.RawMessage(tc.response)}); err != nil { + t.Fatal(err) + } + if err := <-done; (err == nil) != tc.accepted { + t.Fatal(err) + } + }) + } +} diff --git a/apps/parsar-daemon/internal/agent/codex/session_plan.go b/apps/parsar-daemon/internal/agent/codex/session_plan.go index 3fc8831a9..e02a69cff 100644 --- a/apps/parsar-daemon/internal/agent/codex/session_plan.go +++ b/apps/parsar-daemon/internal/agent/codex/session_plan.go @@ -28,6 +28,7 @@ func prepareSessionPlan(ctx context.Context, req proto.PromptRequestPayload, cfg plan.Cleanup() return SessionPlan{}, nil, err } + disableProgrammaticTools(&plan, req.ExecutionControls) if profile != "" { plan.Sandbox = "" plan.Permissions = profile diff --git a/apps/parsar-daemon/internal/cli/agent_discovery.go b/apps/parsar-daemon/internal/cli/agent_discovery.go index ee32b3e7d..5f0c4b925 100644 --- a/apps/parsar-daemon/internal/cli/agent_discovery.go +++ b/apps/parsar-daemon/internal/cli/agent_discovery.go @@ -81,26 +81,27 @@ func discoverAgentCLIs(rc *runContext, profile string, checks agentCLIChecks) (a Codex: proto.SupportedAgentKind{ Kind: "codex", Capabilities: proto.AgentKindCapabilities{ - Streaming: true, - Permissions: true, - Usage: true, - Resume: true, - Steering: true, - DurableTurns: true, - DurableInputReceipts: true, - MessageImages: true, - FunctionTools: true, - MCPHTTPTools: true, - MCPHTTPBearerAuth: true, - MessageItems: true, - ToolItems: true, - ToolObservations: true, - EnvironmentNone: true, - WebSearchControl: true, - TextVerbosity: codex.SupportsTextVerbosity, - ExecutionControls: codex.SupportsTextVerbosity, - SubagentControl: true, - SubagentObservations: true, + Streaming: true, + Permissions: true, + Usage: true, + Resume: true, + Steering: true, + DurableTurns: true, + DurableInputReceipts: true, + MessageImages: true, + FunctionTools: true, + MCPHTTPTools: true, + MCPHTTPBearerAuth: true, + MessageItems: true, + ToolItems: true, + ToolObservations: true, + EnvironmentNone: true, + WebSearchControl: true, + ProgrammaticToolCallingDisable: true, + TextVerbosity: codex.SupportsTextVerbosity, + ExecutionControls: codex.SupportsTextVerbosity, + SubagentControl: true, + SubagentObservations: true, }, }, Pi: proto.SupportedAgentKind{ diff --git a/apps/parsar-daemon/internal/cli/claude_sdk.go b/apps/parsar-daemon/internal/cli/claude_sdk.go index bbb9fe967..adc035a4f 100644 --- a/apps/parsar-daemon/internal/cli/claude_sdk.go +++ b/apps/parsar-daemon/internal/cli/claude_sdk.go @@ -31,6 +31,7 @@ func discoverClaudeSDK(rc *runContext, profile string, check func(context.Contex Streaming: true, Usage: true, Resume: true, Steering: true, MessageItems: true, ToolObservations: true, EnvironmentNone: true, SubagentControl: true, DurableTurns: true, DurableInputReceipts: true, FunctionTools: true, ExecutionControls: true, + ProgrammaticToolCallingDisable: true, }}} fail := func(err error) *claudeSDKDiscovery { fmt.Fprintf(rc.stderr, "parsar-daemon: configured Claude SDK runtime unavailable: %v\n", err) diff --git a/apps/parsar-daemon/internal/cli/mcode.go b/apps/parsar-daemon/internal/cli/mcode.go index c4a7fdcfd..62f276216 100644 --- a/apps/parsar-daemon/internal/cli/mcode.go +++ b/apps/parsar-daemon/internal/cli/mcode.go @@ -26,6 +26,7 @@ func discoverMCode(rc *runContext, check func(context.Context, string) (string, result.Capabilities.DurableTurns = true result.Capabilities.DurableInputReceipts = true result.Capabilities.ExecutionControls = true + result.Capabilities.ProgrammaticToolCallingDisable = true result.Capabilities.ToolObservations = true result.Capabilities.SubagentControl = true // Native preparation verifies the applied admission/tool profile before input. diff --git a/apps/parsar-daemon/internal/dispatch/environment.go b/apps/parsar-daemon/internal/dispatch/environment.go index 577d22b98..fef75ac38 100644 --- a/apps/parsar-daemon/internal/dispatch/environment.go +++ b/apps/parsar-daemon/internal/dispatch/environment.go @@ -7,6 +7,9 @@ import ( ) func validateExecutionEnvironment(req proto.PromptRequestPayload, caps proto.AgentKindCapabilities) error { + if err := req.ValidateProgrammaticToolCallingDisable(caps.ProgrammaticToolCallingDisable); err != nil { + return err + } if err := req.ValidateToolSearch(caps.ToolSearch); err != nil { return err } diff --git a/contracts/agents-api/README.md b/contracts/agents-api/README.md index f7938b3e0..75ad01bf8 100644 --- a/contracts/agents-api/README.md +++ b/contracts/agents-api/README.md @@ -249,6 +249,20 @@ including further deployment qualification; this inventory describes merged beha multi-agent settings default to six concurrent subagents. Function defer-loading defaults to false and programmatic tool calling to true. Saving these values does not itself admit a native execution. Session references are admitted separately. +- [Explicit disabled tools](tool-policy.md) can be saved, used inline or resolved from saved Agents: + `web_search.mode=disabled` and `programmatic_tool_calling.enabled=false`. Search + responses include `context_size=medium` for omitted/null size, nullable domains + and location; an empty domain list stays empty. Only explicit disabled mode is + qualified; omitted/null mode and enabled search remain gaps. Sessions reject + enabled programmatic execution, including the default true on a supplied PTC + declaration. An omitted PTC declaration preserves native behavior: this is an + approved difference from the official default-on behavior, not full compatibility. + Core carries the frozen disabled intent through the common Runtime contract; + native translation and inventory restrictions stay in adapters. Codex checks + managed requirements before new/resumed execution; conflicting forced features + reject before model input. Claude and MiniMax use their restricted tool profiles. + No independent executor or model/tool loop is introduced. Resource defaults and + hosted error parity beyond this supported subset remain unverified. - `POST /agents/{agent_id}` updates only supplied fields. Omitted fields remain unchanged; metadata replaces all pairs and null/empty clears it. Name/instructions null clears them. Concurrent updates preserve unrelated fields. Existing Session @@ -304,7 +318,7 @@ including further deployment qualification; this inventory describes merged beha rechecked before dispatch-only decryption; authenticated execution requires the separate bearer capability and never downgrades on failure. Exact URL/selection timing, implicit response population and hosted errors remain local or unverified. - Other MCP variants and web-search remain gaps, not changes to the pinned target + Other MCP variants and enabled web-search remain gaps, not changes to the pinned target or claims of complete resource coverage. - Use `/agents/sessions` beneath the configured API base URL, bearer authentication diff --git a/contracts/agents-api/openapi.yaml b/contracts/agents-api/openapi.yaml index 060e63ef4..1baa5cd1d 100644 --- a/contracts/agents-api/openapi.yaml +++ b/contracts/agents-api/openapi.yaml @@ -1937,8 +1937,10 @@ paths: required defaulting to false. Saving credential_id grants no access: Session admission checks attached Vault ownership and destination. MCP allowed_tools preserves null versus empty; saved HTTP transport includes empty headers. - Model-derived reasoning defaults, other MCP variants, web_search and public - retry conformance remain incomplete. Session execution admits only its supported + Model-derived reasoning defaults, other MCP variants, enabled web_search and + public retry conformance remain incomplete. Explicit disabled web_search can + be saved; Session execution also accepts explicit disabled programmatic_tool_calling + through qualified Runtime controls. Session execution admits only its supported configuration subset.' parameters: - description: agents=v1 @@ -2723,8 +2725,11 @@ paths: updates cannot change committed Session Skill contents. Deferred function discovery uses type-only tool_search and per-function defer_loading in the qualified single-agent Claude environment:none function profile, including - qualified inline image messages and text results. Other combinations remain - unqualified; see the operation coverage. + qualified inline image messages and text results. Explicit web_search mode + disabled and programmatic_tool_calling enabled false use frozen common Runtime + controls. Enabled forms remain unqualified. Omitted programmatic configuration + preserves native behavior, a documented difference from the official default-on + behavior. Other combinations remain unqualified; see the operation coverage. parameters: - description: agents=v1 in: header diff --git a/contracts/agents-api/tool-policy.md b/contracts/agents-api/tool-policy.md new file mode 100644 index 000000000..05fbd07f0 --- /dev/null +++ b/contracts/agents-api/tool-policy.md @@ -0,0 +1,70 @@ +# Explicit disabled tools + +Protocol baseline: `upstream.json` (OpenAI Python SDK 3.13.0). This covers a +bounded execution profile, not complete tool or Agents API compatibility. + +## Public behavior + +Saved Agents and inline Session configuration accept these declarations: + +```json +[ + {"type": "web_search", "mode": "disabled"}, + {"type": "programmatic_tool_calling", "enabled": false} +] +``` + +Saved references use the same parser and immutable Session snapshot. Disabled +search retains optional settings as resource data: omitted/null context size +resolves to `medium`; domain and location omission resolves to null; an empty +domain list remains empty. These settings cannot enable execution while disabled. +Web-search mode omission/null, enabled search and exact hosted errors remain +unqualified. Explicit enabled programmatic execution rejects at Session admission; +saving that intent remains separate from execution qualification. + +Omitting programmatic configuration preserves each harness's native behavior. +The user approved this difference from the official default-on behavior. Native +feature differences stay in adapters; Core does not supply another executor or +model loop. Unrelated native utility tools are not implicitly removed. + +## Runtime boundary + +The common `DisableProgrammaticToolCalling` control carries explicit disabled +intent on initial execution and cold continuation. Public qualification and the +operation-specific Runtime capability must both permit this request. Omission +does not require the new capability. Search uses the existing disabled control. + +| Adapter | Native enforcement | +| --- | --- | +| Codex | Disable code-mode features; check native managed requirements before thread start/resume and reject a forced conflicting feature. | +| Claude Code | Retain the restricted built-in inventory and verify native initialization against that inventory. | +| MiniMax Code | Retain the protected native tool profile, empty text-execution inventory and disabled web-search feature. | + +## Acceptance + +The opt-in `TestNativeToolPolicyPublicExecution` fixture and +`services/agents-api/tests/official_tool_policy.py` exercise a real PostgreSQL +database, independent API, Docker daemon and native harness using the pinned +official SDK with strict response validation and raw HTTP. Supply the existing +native-test environment variables plus `PARSAR_TOOL_POLICY_ENGINE` and a private +`PARSAR_TOOL_POLICY_REAL_OPTIONS` file. Each concurrent execution worker requires +its own dedicated test database. + +On 2026-09-22, Codex and Claude with Kimi K3, and MiniMax Code with MiniMax-M2.7, +passed SDK/raw HTTP through both saved and inline configurations. Each of four +Sessions completed a first Turn and continued its random marker after a full +daemon restart with the same native Session. Checks include public configuration, +SSE order, Items, native input receipts, unsupported enablement without Session +persistence, omitted configuration admission and tenant isolation. + +Native evidence separately records Codex's disabled feature arguments and search +setting, Claude's empty built-in tool argument and initialization validation, and +MiniMax's restricted native configuration. Successful model text alone does not +prove disabled tools. Controlled tests cover managed requirement conflicts, +invalid fields and the common qualification contract. + +Evidence is retained privately under `~/.parsar/remediation/20260922/tool-policy` +on the development host and test server. Live qualification in this batch uses +`environment:none`; workspace provisioning, provider matrices, enabled tools, +native question extensions and complete hosted default/error semantics were not +requalified. Existing workspace mechanisms are unchanged. diff --git a/internal/agentdaemon/device/state.go b/internal/agentdaemon/device/state.go index 6fa46797c..4cd991e85 100644 --- a/internal/agentdaemon/device/state.go +++ b/internal/agentdaemon/device/state.go @@ -60,23 +60,24 @@ type HeartbeatStatus struct { // KindCapabilities mirrors the daemon heartbeat capability // shape after gateway-level normalization. Persistence stays separate from wire protocol structs. type KindCapabilities struct { - SubagentObservations bool `json:"subagent_observations,omitempty"` - Streaming bool `json:"streaming,omitempty"` - Permissions bool `json:"permissions,omitempty"` - Usage bool `json:"usage,omitempty"` - Resume bool `json:"resume,omitempty"` - NativeSessionRecovery bool `json:"native_session_recovery,omitempty"` - Steering bool `json:"steering,omitempty"` - MessageItems bool `json:"message_items,omitempty"` - ToolItems bool `json:"tool_items,omitempty"` - ToolObservations bool `json:"tool_observations,omitempty"` - EnvironmentNone bool `json:"environment_none,omitempty"` - LocalEnvironment bool `json:"local_environment,omitempty"` - LocalEnvironmentNetworkPolicy bool `json:"local_environment_network_policy,omitempty"` - Preparation bool `json:"preparation,omitempty"` - WorkspaceReadPreparation bool `json:"workspace_read_preparation,omitempty"` - WorkspaceOutputExport bool `json:"workspace_output_export,omitempty"` - WebSearchControl bool `json:"web_search_control,omitempty"` + SubagentObservations bool `json:"subagent_observations,omitempty"` + Streaming bool `json:"streaming,omitempty"` + Permissions bool `json:"permissions,omitempty"` + Usage bool `json:"usage,omitempty"` + Resume bool `json:"resume,omitempty"` + NativeSessionRecovery bool `json:"native_session_recovery,omitempty"` + Steering bool `json:"steering,omitempty"` + MessageItems bool `json:"message_items,omitempty"` + ToolItems bool `json:"tool_items,omitempty"` + ToolObservations bool `json:"tool_observations,omitempty"` + EnvironmentNone bool `json:"environment_none,omitempty"` + LocalEnvironment bool `json:"local_environment,omitempty"` + LocalEnvironmentNetworkPolicy bool `json:"local_environment_network_policy,omitempty"` + Preparation bool `json:"preparation,omitempty"` + WorkspaceReadPreparation bool `json:"workspace_read_preparation,omitempty"` + WorkspaceOutputExport bool `json:"workspace_output_export,omitempty"` + ProgrammaticToolCallingDisable bool `json:"programmatic_tool_calling_disable,omitempty"` + WebSearchControl bool `json:"web_search_control,omitempty"` // ExecutionControls supports typed search and verbosity controls. ExecutionControls bool `json:"execution_controls,omitempty"` TextVerbosity bool `json:"text_verbosity,omitempty"` diff --git a/internal/agentdaemon/gateway/session.go b/internal/agentdaemon/gateway/session.go index d8b69adc9..4188fa311 100644 --- a/internal/agentdaemon/gateway/session.go +++ b/internal/agentdaemon/gateway/session.go @@ -541,36 +541,37 @@ func deviceKindsFromHeartbeat(p proto.HeartbeatPayload) []device.SupportedAgentK Available: info.Available, Version: info.Version, Capabilities: device.KindCapabilities{ - Streaming: info.Capabilities.Streaming, - Permissions: info.Capabilities.Permissions, - Usage: info.Capabilities.Usage, - Resume: info.Capabilities.Resume, - Steering: info.Capabilities.Steering, - DurableTurns: info.Capabilities.DurableTurns, - DurableInputReceipts: info.Capabilities.DurableInputReceipts, - NativeSessionRecovery: info.Capabilities.NativeSessionRecovery, - MessageItems: info.Capabilities.MessageItems, - ToolItems: info.Capabilities.ToolItems, - ToolObservations: info.Capabilities.ToolObservations, - EnvironmentNone: info.Capabilities.EnvironmentNone, - LocalEnvironment: info.Capabilities.LocalEnvironment, - LocalEnvironmentNetworkPolicy: info.Capabilities.LocalEnvironmentNetworkPolicy, - Preparation: info.Capabilities.Preparation, - WorkspaceReadPreparation: info.Capabilities.WorkspaceReadPreparation, - WorkspaceOutputExport: info.Capabilities.WorkspaceOutputExport, - WebSearchControl: info.Capabilities.WebSearchControl, - TextVerbosity: info.Capabilities.TextVerbosity, - StructuredOutput: info.Capabilities.StructuredOutput, - ToolSearch: info.Capabilities.ToolSearch, - MessageImages: info.Capabilities.MessageImages, - ExecutionControls: info.Capabilities.ExecutionControls, - SubagentControl: info.Capabilities.SubagentControl, - SubagentObservations: info.Capabilities.SubagentObservations, - FunctionTools: info.Capabilities.FunctionTools, - MCPHTTPTools: info.Capabilities.MCPHTTPTools, - MCPHTTPRequired: info.Capabilities.MCPHTTPRequired, - MCPHTTPBearerAuth: info.Capabilities.MCPHTTPBearerAuth, - WorkspaceAuthoring: info.Capabilities.WorkspaceAuthoring, + Streaming: info.Capabilities.Streaming, + Permissions: info.Capabilities.Permissions, + Usage: info.Capabilities.Usage, + Resume: info.Capabilities.Resume, + Steering: info.Capabilities.Steering, + DurableTurns: info.Capabilities.DurableTurns, + DurableInputReceipts: info.Capabilities.DurableInputReceipts, + NativeSessionRecovery: info.Capabilities.NativeSessionRecovery, + MessageItems: info.Capabilities.MessageItems, + ToolItems: info.Capabilities.ToolItems, + ToolObservations: info.Capabilities.ToolObservations, + EnvironmentNone: info.Capabilities.EnvironmentNone, + LocalEnvironment: info.Capabilities.LocalEnvironment, + LocalEnvironmentNetworkPolicy: info.Capabilities.LocalEnvironmentNetworkPolicy, + Preparation: info.Capabilities.Preparation, + WorkspaceReadPreparation: info.Capabilities.WorkspaceReadPreparation, + WorkspaceOutputExport: info.Capabilities.WorkspaceOutputExport, + WebSearchControl: info.Capabilities.WebSearchControl, + ProgrammaticToolCallingDisable: info.Capabilities.ProgrammaticToolCallingDisable, + TextVerbosity: info.Capabilities.TextVerbosity, + StructuredOutput: info.Capabilities.StructuredOutput, + ToolSearch: info.Capabilities.ToolSearch, + MessageImages: info.Capabilities.MessageImages, + ExecutionControls: info.Capabilities.ExecutionControls, + SubagentControl: info.Capabilities.SubagentControl, + SubagentObservations: info.Capabilities.SubagentObservations, + FunctionTools: info.Capabilities.FunctionTools, + MCPHTTPTools: info.Capabilities.MCPHTTPTools, + MCPHTTPRequired: info.Capabilities.MCPHTTPRequired, + MCPHTTPBearerAuth: info.Capabilities.MCPHTTPBearerAuth, + WorkspaceAuthoring: info.Capabilities.WorkspaceAuthoring, }, }) } diff --git a/internal/agentdaemon/proto/execution_controls.go b/internal/agentdaemon/proto/execution_controls.go new file mode 100644 index 000000000..5ccd11220 --- /dev/null +++ b/internal/agentdaemon/proto/execution_controls.go @@ -0,0 +1,12 @@ +package proto + +import "errors" + +// ValidateProgrammaticToolCallingDisable applies only to explicit tool disabling. +// Omission preserves the adapter's ordinary native behavior. +func (r PromptRequestPayload) ValidateProgrammaticToolCallingDisable(supported bool) error { + if r.ExecutionControls != nil && r.ExecutionControls.DisableProgrammaticToolCalling && !supported { + return errors.New("engine does not support disabling programmatic tool calling") + } + return nil +} diff --git a/internal/agentdaemon/proto/inbound.go b/internal/agentdaemon/proto/inbound.go index af6a8ab5b..b88ae9052 100644 --- a/internal/agentdaemon/proto/inbound.go +++ b/internal/agentdaemon/proto/inbound.go @@ -251,24 +251,25 @@ const ( // cancellation belong to the daemon connector itself; these bits are // the engine-specific surface the UI uses for filtering and copy. type AgentKindCapabilities struct { - SubagentObservations bool `json:"subagent_observations,omitempty"` - Streaming bool `json:"streaming,omitempty"` - Permissions bool `json:"permissions,omitempty"` - Usage bool `json:"usage,omitempty"` - Resume bool `json:"resume,omitempty"` - NativeSessionRecovery bool `json:"native_session_recovery,omitempty"` - WorkspaceAuthoring bool `json:"workspace_authoring,omitempty"` - Steering bool `json:"steering,omitempty"` - MessageItems bool `json:"message_items,omitempty"` - ToolItems bool `json:"tool_items,omitempty"` - ToolObservations bool `json:"tool_observations,omitempty"` - EnvironmentNone bool `json:"environment_none,omitempty"` - LocalEnvironment bool `json:"local_environment,omitempty"` - LocalEnvironmentNetworkPolicy bool `json:"local_environment_network_policy,omitempty"` - Preparation bool `json:"preparation,omitempty"` - WorkspaceReadPreparation bool `json:"workspace_read_preparation,omitempty"` - WorkspaceOutputExport bool `json:"workspace_output_export,omitempty"` - WebSearchControl bool `json:"web_search_control,omitempty"` + SubagentObservations bool `json:"subagent_observations,omitempty"` + Streaming bool `json:"streaming,omitempty"` + Permissions bool `json:"permissions,omitempty"` + Usage bool `json:"usage,omitempty"` + Resume bool `json:"resume,omitempty"` + NativeSessionRecovery bool `json:"native_session_recovery,omitempty"` + WorkspaceAuthoring bool `json:"workspace_authoring,omitempty"` + Steering bool `json:"steering,omitempty"` + MessageItems bool `json:"message_items,omitempty"` + ToolItems bool `json:"tool_items,omitempty"` + ToolObservations bool `json:"tool_observations,omitempty"` + EnvironmentNone bool `json:"environment_none,omitempty"` + LocalEnvironment bool `json:"local_environment,omitempty"` + LocalEnvironmentNetworkPolicy bool `json:"local_environment_network_policy,omitempty"` + Preparation bool `json:"preparation,omitempty"` + WorkspaceReadPreparation bool `json:"workspace_read_preparation,omitempty"` + WorkspaceOutputExport bool `json:"workspace_output_export,omitempty"` + ProgrammaticToolCallingDisable bool `json:"programmatic_tool_calling_disable,omitempty"` + WebSearchControl bool `json:"web_search_control,omitempty"` // ExecutionControls supports typed search and verbosity controls. ExecutionControls bool `json:"execution_controls,omitempty"` TextVerbosity bool `json:"text_verbosity,omitempty"` diff --git a/internal/agentdaemon/proto/outbound.go b/internal/agentdaemon/proto/outbound.go index b60826df3..3263f5afa 100644 --- a/internal/agentdaemon/proto/outbound.go +++ b/internal/agentdaemon/proto/outbound.go @@ -149,9 +149,10 @@ type DeviceShutdownPayload struct { // ExecutionControls requires both values when supplied; omitting the block preserves agent options. // Send only to a peer advertising execution_controls, independently of older option capabilities. type ExecutionControls struct { - WebSearch string `json:"web_search"` - TextVerbosity string `json:"text_verbosity"` - OutputFormat *OutputFormat `json:"output_format,omitempty"` + DisableProgrammaticToolCalling bool `json:"disable_programmatic_tool_calling,omitempty"` + WebSearch string `json:"web_search"` + TextVerbosity string `json:"text_verbosity"` + OutputFormat *OutputFormat `json:"output_format,omitempty"` } // OutputFormat passes the public schema unchanged to a qualified native adapter. diff --git a/services/agents-api/internal/api/agents.go b/services/agents-api/internal/api/agents.go index 157f1c963..0144aeec8 100644 --- a/services/agents-api/internal/api/agents.go +++ b/services/agents-api/internal/api/agents.go @@ -20,7 +20,7 @@ type AgentStore interface { } // @Summary Create a reusable Agent -// @Description Persists configuration independently of execution. Supports model/name/instructions/metadata, explicit reasoning and service tiers, multi_agent, text/json_schema, function/tool_search/programmatic_tool_calling and HTTP MCP with nullable credential_id and explicit service origin and boolean required defaulting to false. Saving credential_id grants no access: Session admission checks attached Vault ownership and destination. MCP allowed_tools preserves null versus empty; saved HTTP transport includes empty headers. Model-derived reasoning defaults, other MCP variants, web_search and public retry conformance remain incomplete. Session execution admits only its supported configuration subset. +// @Description Persists configuration independently of execution. Supports model/name/instructions/metadata, explicit reasoning and service tiers, multi_agent, text/json_schema, function/tool_search/programmatic_tool_calling and HTTP MCP with nullable credential_id and explicit service origin and boolean required defaulting to false. Saving credential_id grants no access: Session admission checks attached Vault ownership and destination. MCP allowed_tools preserves null versus empty; saved HTTP transport includes empty headers. Model-derived reasoning defaults, other MCP variants, enabled web_search and public retry conformance remain incomplete. Explicit disabled web_search can be saved; Session execution also accepts explicit disabled programmatic_tool_calling through qualified Runtime controls. Session execution admits only its supported configuration subset. // @Tags Agents // @Accept json // @Produce json diff --git a/services/agents-api/internal/api/disabled_tools.go b/services/agents-api/internal/api/disabled_tools.go new file mode 100644 index 000000000..f6f89133f --- /dev/null +++ b/services/agents-api/internal/api/disabled_tools.go @@ -0,0 +1,58 @@ +package api + +import ( + "encoding/json" + "errors" +) + +func resolveProgrammaticTool(raw json.RawMessage) (json.RawMessage, error) { + var input struct { + Type string `json:"type"` + Enabled json.RawMessage `json:"enabled"` + } + if decodeInputObject(raw, &input, "type", "enabled") != nil { + return nil, errors.New("Invalid programmatic_tool_calling fields.") + } + enabled, err := optionalBoolean(input.Enabled, true) + if err != nil { + return nil, errors.New("programmatic_tool_calling.enabled must be a boolean.") + } + return json.Marshal(struct { + Type string `json:"type"` + Enabled bool `json:"enabled"` + }{input.Type, enabled}) +} + +// Only disabled search is qualified; its optional settings remain resource data. +func resolveDisabledWebSearch(raw json.RawMessage) (json.RawMessage, error) { + var input struct { + Type string `json:"type"` + Mode string `json:"mode"` + ContextSize *string `json:"context_size"` + AllowedDomains []*string `json:"allowed_domains"` + Location *struct { + City *string `json:"city"` + Country *string `json:"country"` + Region *string `json:"region"` + Timezone *string `json:"timezone"` + } `json:"location"` + } + if decodeInputObject(raw, &input, "type", "mode", "context_size", "allowed_domains", "location") != nil || input.Mode != "disabled" { + return nil, errors.New("Only disabled web_search is qualified for execution.") + } + if input.ContextSize == nil { + value := "medium" + input.ContextSize = &value + } + for _, domain := range input.AllowedDomains { + if domain == nil { + return nil, errors.New("web_search.allowed_domains must contain strings.") + } + } + switch *input.ContextSize { + case "low", "medium", "high": + default: + return nil, errors.New("Invalid web_search.context_size.") + } + return json.Marshal(input) +} diff --git a/services/agents-api/internal/api/disabled_tools_test.go b/services/agents-api/internal/api/disabled_tools_test.go new file mode 100644 index 000000000..ff785affe --- /dev/null +++ b/services/agents-api/internal/api/disabled_tools_test.go @@ -0,0 +1,64 @@ +package api + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" +) + +func TestDisabledToolConfigurationRoundTrip(t *testing.T) { + for _, raw := range []string{ + `{"type":"programmatic_tool_calling","enabled":false}`, + `{"type":"web_search","mode":"disabled"}`, + `{"type":"web_search","mode":"disabled","context_size":"high","allowed_domains":[],"location":{"city":"Paris","country":null}}`, + `{"type":"web_search","mode":"disabled","context_size":null,"allowed_domains":["example.com"],"location":null}`, + } { + saved, err := resolveSavedTools([]json.RawMessage{json.RawMessage(raw)}) + if err != nil { + t.Fatal(raw, err) + } + inline, err := resolveSessionTools([]json.RawMessage{json.RawMessage(raw)}) + if err != nil || string(saved[0]) != string(inline[0]) { + t.Fatal(raw, saved, inline, err) + } + referenced, err := resolveSessionTools(saved) + if err != nil || string(referenced[0]) != string(saved[0]) { + t.Fatal("saved configuration changed on resolution", err) + } + if strings.Contains(raw, "web_search") && !strings.Contains(string(saved[0]), `"context_size":`) { + t.Fatal("missing resource default") + } + if strings.Contains(raw, `"allowed_domains":[]`) && !strings.Contains(string(saved[0]), `"allowed_domains":[]`) { + t.Fatal("empty domain list lost") + } + } +} + +func TestDisabledToolAdmissionPrecedesPersistence(t *testing.T) { + for _, tools := range []string{ + `[{"type":"programmatic_tool_calling"}]`, + `[{"type":"programmatic_tool_calling","enabled":true}]`, + `[{"type":"programmatic_tool_calling","enabled":null}]`, + `[{"type":"programmatic_tool_calling","enabled":"false"}]`, + `[{"type":"programmatic_tool_calling","enabled":false,"extra":true}]`, + `[{"type":"programmatic_tool_calling","enabled":false},{"type":"programmatic_tool_calling","enabled":false}]`, + `[{"type":"web_search","mode":"live"}]`, + `[{"type":"web_search","mode":"cached"}]`, + `[{"type":"web_search","mode":"disabled","allowed_domains":[null]}]`, + `[{"type":"web_search","mode":"disabled","context_size":"enormous"}]`, + `[{"type":"web_search","mode":"disabled","location":{"extra":true}}]`, + `[{"type":"web_search","mode":"disabled"},{"type":"web_search","mode":"disabled"}]`, + } { + h, store, _ := testHandler(t) + req := httptest.NewRequest(http.MethodPost, "/v1/agents/sessions", strings.NewReader(`{"agent":{"model":"model","tools":`+tools+`},"environment":{"type":"none"}}`)) + req.Header.Set("Authorization", "Bearer test-api-key") + req.Header.Set("OpenAI-Beta", "agents=v1") + response := httptest.NewRecorder() + h.ServeHTTP(response, req) + if response.Code != 400 || store.tenant != "" { + t.Fatal(tools, response.Code, response.Body) + } + } +} diff --git a/services/agents-api/internal/api/handler.go b/services/agents-api/internal/api/handler.go index 6ff481e6b..3aa075e3e 100644 --- a/services/agents-api/internal/api/handler.go +++ b/services/agents-api/internal/api/handler.go @@ -124,7 +124,7 @@ func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ... // createSession atomically reserves or admits initial text with the Session. // @Summary Create an execution Session -// @Description Supports inline configuration or a tenant-owned saved agent_id with per-Session field replacements. Execution supports model/instructions, text verbosity, non-deferred function tools, adapter-qualified multi_agent with persisted Subagent reads, implicit reasoning, service tier auto and environment type none, subject to the configured engine. Codex additionally supports HTTP MCP with explicit service origin, native allowed_tools and boolean required defaulting to false. Session vault_ids attach only project-owned Vaults; credential_id selects an attached static bearer credential for the exact HTTPS URL, while null/omission selects a unique match or remains anonymous. Ambiguous selection rejects creation. Frozen private selections never populate an omitted public credential_id; missing decryption configuration fails dispatch without anonymous fallback. Required initialization uses native startup before the first native Turn, including cold resume, and requires a separately advertised capability; exact hosted creation timing and error parity remain unverified. Other MCP origins and OAuth remain unsupported. The self_hosted profile requires Codex, an absolute workspace_directory and empty capability_directories, with optional non-deferred function tools and HTTP MCP using explicit service origin, optionally authenticated by the attached Vault rules. Remote MCP and remote Bearer authentication each require separately advertised combination support; old peers cannot receive unsupported work. Omitted/null capability_directories use the empty-list default; self_hosted requires configured execution plus executor registry. Claude SDK currently requires medium verbosity and object-root function schemas. It supports anonymous or attached static-bearer service-origin HTTP MCP on none with boolean required and separately advertised MCP/bearer/required runtime support. Required servers must be connected before the first native input is released; pending or failed startup rejects execution. The shared Vault selection and immutable binding rules apply; unsupported native labels/tool names reject before persistence. An attached Vault with no matching credential may remain anonymous; missing keys or failed credential lookup/decryption never fall back to anonymous execution. Omitted stream defaults to false; stream and agent_id cannot be null. Metadata may be null, but its values must be strings. Initial input accepts a string or ordered user-message array. Codex and Claude SDK on none also accept inline PNG/JPEG image content; other image combinations and remote URLs are unsupported. None initial input atomically starts a Turn; self_hosted initial input is reserved while returning its Environment connection target, with execution deferred to native readiness and Session failure on initial timeout. Omitted or null input creates an idle Session. With stream=true, returns live Session events starting at creation; disconnect does not cancel execution. New Sessions retain their authenticated creator; all creation retries require the same typed subject, including across key rotation. Saved-Agent retries and inline requests using Vault attachments or credential references retain caller intent independently of later resource changes; unrelated inline retries preserve resolved/default equivalences. Unknown historical creators reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without replay; retry with stream=false to retrieve the Session. Claude SDK environment:none supports qualified object-root json_schema output with medium verbosity, single-Agent execution and ordinary functions; other combinations remain unsupported. Non-text initial input remains unsupported. Basic Codex and Claude SDK openai_hosted creation requires an explicitly configured managed provider. The Claude workspace profile supports non-deferred function tools with text results alongside native workspace tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; initial provisioning has no caller connection action. Network defaults to enabled; disabled and restricted exact ASCII hostnames are supported. Restricted policy requires 1–100 allowed domains. Unsupported hostname forms and startup installations are rejected. Confidential env, system/npm/Python packages and ordered setup commands use the shared initialization lifecycle; requested network applies after setup. Initial inline and tenant-owned file_id files freeze encrypted bytes before provisioning, then install through the common Core lifecycle before native execution or live Files access. Referenced files/env/packages/setup overrides are rejected pending semantic verification. Tenant-owned environment_template_id references inherit omitted network and allow only narrowing overrides. Referenced network:null is explicitly unsupported pending semantic verification. Core freezes effective configuration; template updates/deletion do not alter Session snapshots or same-intent creation retries. Inline or tenant-owned skill_reference Skills share initialization. Templates preserve default/latest/explicit selectors; Session creation freezes concrete metadata and encrypted content atomically. Skill-list omission inherits and a supplied list replaces; null overrides and null version selectors remain unqualified and reject. Source deletion/default updates cannot change committed Session Skill contents. Deferred function discovery uses type-only tool_search and per-function defer_loading in the qualified single-agent Claude environment:none function profile, including qualified inline image messages and text results. Other combinations remain unqualified; see the operation coverage. +// @Description Supports inline configuration or a tenant-owned saved agent_id with per-Session field replacements. Execution supports model/instructions, text verbosity, non-deferred function tools, adapter-qualified multi_agent with persisted Subagent reads, implicit reasoning, service tier auto and environment type none, subject to the configured engine. Codex additionally supports HTTP MCP with explicit service origin, native allowed_tools and boolean required defaulting to false. Session vault_ids attach only project-owned Vaults; credential_id selects an attached static bearer credential for the exact HTTPS URL, while null/omission selects a unique match or remains anonymous. Ambiguous selection rejects creation. Frozen private selections never populate an omitted public credential_id; missing decryption configuration fails dispatch without anonymous fallback. Required initialization uses native startup before the first native Turn, including cold resume, and requires a separately advertised capability; exact hosted creation timing and error parity remain unverified. Other MCP origins and OAuth remain unsupported. The self_hosted profile requires Codex, an absolute workspace_directory and empty capability_directories, with optional non-deferred function tools and HTTP MCP using explicit service origin, optionally authenticated by the attached Vault rules. Remote MCP and remote Bearer authentication each require separately advertised combination support; old peers cannot receive unsupported work. Omitted/null capability_directories use the empty-list default; self_hosted requires configured execution plus executor registry. Claude SDK currently requires medium verbosity and object-root function schemas. It supports anonymous or attached static-bearer service-origin HTTP MCP on none with boolean required and separately advertised MCP/bearer/required runtime support. Required servers must be connected before the first native input is released; pending or failed startup rejects execution. The shared Vault selection and immutable binding rules apply; unsupported native labels/tool names reject before persistence. An attached Vault with no matching credential may remain anonymous; missing keys or failed credential lookup/decryption never fall back to anonymous execution. Omitted stream defaults to false; stream and agent_id cannot be null. Metadata may be null, but its values must be strings. Initial input accepts a string or ordered user-message array. Codex and Claude SDK on none also accept inline PNG/JPEG image content; other image combinations and remote URLs are unsupported. None initial input atomically starts a Turn; self_hosted initial input is reserved while returning its Environment connection target, with execution deferred to native readiness and Session failure on initial timeout. Omitted or null input creates an idle Session. With stream=true, returns live Session events starting at creation; disconnect does not cancel execution. New Sessions retain their authenticated creator; all creation retries require the same typed subject, including across key rotation. Saved-Agent retries and inline requests using Vault attachments or credential references retain caller intent independently of later resource changes; unrelated inline retries preserve resolved/default equivalences. Unknown historical creators reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without replay; retry with stream=false to retrieve the Session. Claude SDK environment:none supports qualified object-root json_schema output with medium verbosity, single-Agent execution and ordinary functions; other combinations remain unsupported. Non-text initial input remains unsupported. Basic Codex and Claude SDK openai_hosted creation requires an explicitly configured managed provider. The Claude workspace profile supports non-deferred function tools with text results alongside native workspace tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; initial provisioning has no caller connection action. Network defaults to enabled; disabled and restricted exact ASCII hostnames are supported. Restricted policy requires 1–100 allowed domains. Unsupported hostname forms and startup installations are rejected. Confidential env, system/npm/Python packages and ordered setup commands use the shared initialization lifecycle; requested network applies after setup. Initial inline and tenant-owned file_id files freeze encrypted bytes before provisioning, then install through the common Core lifecycle before native execution or live Files access. Referenced files/env/packages/setup overrides are rejected pending semantic verification. Tenant-owned environment_template_id references inherit omitted network and allow only narrowing overrides. Referenced network:null is explicitly unsupported pending semantic verification. Core freezes effective configuration; template updates/deletion do not alter Session snapshots or same-intent creation retries. Inline or tenant-owned skill_reference Skills share initialization. Templates preserve default/latest/explicit selectors; Session creation freezes concrete metadata and encrypted content atomically. Skill-list omission inherits and a supplied list replaces; null overrides and null version selectors remain unqualified and reject. Source deletion/default updates cannot change committed Session Skill contents. Deferred function discovery uses type-only tool_search and per-function defer_loading in the qualified single-agent Claude environment:none function profile, including qualified inline image messages and text results. Explicit web_search mode disabled and programmatic_tool_calling enabled false use frozen common Runtime controls. Enabled forms remain unqualified. Omitted programmatic configuration preserves native behavior, a documented difference from the official default-on behavior. Other combinations remain unqualified; see the operation coverage. // @Tags Sessions // @Accept json // @Produce json,text/event-stream diff --git a/services/agents-api/internal/api/saved_tools.go b/services/agents-api/internal/api/saved_tools.go index f00f9d21d..9a4a2f2c8 100644 --- a/services/agents-api/internal/api/saved_tools.go +++ b/services/agents-api/internal/api/saved_tools.go @@ -34,21 +34,11 @@ func resolveSavedTools(input []json.RawMessage) ([]json.RawMessage, error) { } value, _ = json.Marshal(kind) case "programmatic_tool_calling": - var input struct { - Type string `json:"type"` - Enabled json.RawMessage `json:"enabled"` - } - if decodeInputObject(raw, &input, "type", "enabled") != nil { - return nil, errors.New("Invalid programmatic_tool_calling fields.") - } - enabled, err := optionalBoolean(input.Enabled, true) + resolved, err := resolveProgrammaticTool(raw) if err != nil { - return nil, errors.New("programmatic_tool_calling.enabled must be a boolean.") + return nil, err } - value, _ = json.Marshal(struct { - Type string `json:"type"` - Enabled bool `json:"enabled"` - }{kind.Type, enabled}) + value = resolved case "mcp": resolved, err := resolveMCPTool(raw, true) if err != nil { @@ -56,7 +46,11 @@ func resolveSavedTools(input []json.RawMessage) ([]json.RawMessage, error) { } value = resolved case "web_search": - return nil, errors.New("Persisted web_search configuration is not implemented yet.") + resolved, err := resolveDisabledWebSearch(raw) + if err != nil { + return nil, err + } + value = resolved default: return nil, errors.New("Unknown persisted tool type.") } diff --git a/services/agents-api/internal/api/session_tools.go b/services/agents-api/internal/api/session_tools.go index 7a5fade4f..912f99508 100644 --- a/services/agents-api/internal/api/session_tools.go +++ b/services/agents-api/internal/api/session_tools.go @@ -13,6 +13,7 @@ func resolveSessionTools(input []json.RawMessage) ([]json.RawMessage, error) { positions := make([]int, 0, len(input)) servers := map[string]bool{} search := false + controls := map[string]bool{} for i, raw := range input { var kind struct { Type string `json:"type"` @@ -21,6 +22,31 @@ func resolveSessionTools(input []json.RawMessage) ([]json.RawMessage, error) { return nil, errors.New("Invalid execution tool configuration.") } switch kind.Type { + case "programmatic_tool_calling", "web_search": + if controls[kind.Type] { + return nil, errors.New("Execution requires distinct tool controls.") + } + controls[kind.Type] = true + var resolved json.RawMessage + var err error + if kind.Type == "web_search" { + resolved, err = resolveDisabledWebSearch(raw) + } else { + resolved, err = resolveProgrammaticTool(raw) + var value struct { + Enabled bool `json:"enabled"` + } + if err == nil { + _ = json.Unmarshal(resolved, &value) + if value.Enabled { + err = errors.New("Programmatic tool calling is not qualified for execution.") + } + } + } + if err != nil { + return nil, err + } + tools[i] = resolved case "tool_search": if search || decodeInputObject(raw, &kind, "type") != nil { return nil, errors.New("Execution requires one type-only tool_search declaration.") @@ -46,7 +72,7 @@ func resolveSessionTools(input []json.RawMessage) ([]json.RawMessage, error) { functions = append(functions, function) positions = append(positions, i) default: - return nil, errors.New("Execution currently supports functions, tool_search and the service-origin HTTP MCP profile only.") + return nil, errors.New("Unsupported execution tool; supported tools include functions, qualified tool_search, service-origin HTTP MCP and explicit disabled controls.") } } resolved, err := resolveFunctions(functions) diff --git a/services/agents-api/internal/engine/claude.go b/services/agents-api/internal/engine/claude.go index d9f474e46..aae4ac8ce 100644 --- a/services/agents-api/internal/engine/claude.go +++ b/services/agents-api/internal/engine/claude.go @@ -11,10 +11,11 @@ import ( func claudeProfile() Profile { return Profile{ - StructuredOutput: true, - ToolSearch: true, - MessageImagePlacements: []string{"none"}, - Placements: []string{"none", "openai_hosted", "self_hosted"}, MCPBearer: true, + ProgrammaticToolCallingDisable: true, + StructuredOutput: true, + ToolSearch: true, + MessageImagePlacements: []string{"none"}, + Placements: []string{"none", "openai_hosted", "self_hosted"}, MCPBearer: true, ValidateConfiguration: validateClaudeConfiguration, ValidateTools: validateClaudeTools, ValidateFunctionResult: func(content []proto.InputContent) error { @@ -49,7 +50,7 @@ func validateClaudeConfiguration(agent v1.Agent, environment *v1.Environment, ha var tool struct { Type string `json:"type"` } - if json.Unmarshal(raw, &tool) != nil || tool.Type != "function" { + if json.Unmarshal(raw, &tool) != nil || (tool.Type != "function" && tool.Type != "web_search" && tool.Type != "programmatic_tool_calling") { return ErrInvalidInput } } @@ -65,7 +66,7 @@ func validateClaudeConfiguration(agent v1.Agent, environment *v1.Environment, ha return ErrInvalidInput } search = search || tool.Type == "tool_search" - otherTools = otherTools || (tool.Type != "function" && tool.Type != "tool_search") + otherTools = otherTools || (tool.Type != "function" && tool.Type != "tool_search" && tool.Type != "web_search" && tool.Type != "programmatic_tool_calling") } if search && (environment.Type != "none" || agent.MultiAgent.Enabled || otherTools || agent.Text.Format.Type == "json_schema") { return errors.New("Tool discovery currently requires a single-agent environment:none function profile.") diff --git a/services/agents-api/internal/engine/codex.go b/services/agents-api/internal/engine/codex.go index 48a1e7212..ff5def2ee 100644 --- a/services/agents-api/internal/engine/codex.go +++ b/services/agents-api/internal/engine/codex.go @@ -9,9 +9,10 @@ import ( func codexProfile() Profile { return Profile{ - Placements: []string{"none", "self_hosted", "openai_hosted"}, - MessageImagePlacements: []string{"none"}, - WebSearchControl: true, TextVerbosity: true, MCPBearer: true, + ProgrammaticToolCallingDisable: true, + Placements: []string{"none", "self_hosted", "openai_hosted"}, + MessageImagePlacements: []string{"none"}, + WebSearchControl: true, TextVerbosity: true, MCPBearer: true, ValidateConfiguration: func(agent v1.Agent, _ *v1.Environment, _ bool) error { return rejectSubagentTools(agent, "function", "mcp") }, diff --git a/services/agents-api/internal/engine/mcode.go b/services/agents-api/internal/engine/mcode.go index d664494c6..79e0699dc 100644 --- a/services/agents-api/internal/engine/mcode.go +++ b/services/agents-api/internal/engine/mcode.go @@ -9,7 +9,7 @@ import ( ) func mcodeProfile() Profile { - return Profile{Placements: []string{"none", "openai_hosted", "self_hosted"}, ValidateConfiguration: func(a v1.Agent, e *v1.Environment, daemon bool) error { + return Profile{ProgrammaticToolCallingDisable: true, Placements: []string{"none", "openai_hosted", "self_hosted"}, ValidateConfiguration: func(a v1.Agent, e *v1.Environment, daemon bool) error { if e == nil || (e.Type != "none" && e.Type != "openai_hosted" && e.Type != "self_hosted") || daemon || strings.TrimSpace(a.Model) == "" || a.Reasoning.Effort != nil || a.Reasoning.Summary != nil || (a.ServiceTier != "" && a.ServiceTier != "auto") || (a.Text.Format.Type != "" && a.Text.Format.Type != "text") || (a.Text.Verbosity != "" && a.Text.Verbosity != "medium") { return ErrInvalidInput } diff --git a/services/agents-api/internal/engine/profile.go b/services/agents-api/internal/engine/profile.go index 873012653..d4d290198 100644 --- a/services/agents-api/internal/engine/profile.go +++ b/services/agents-api/internal/engine/profile.go @@ -12,6 +12,7 @@ var ErrInvalidInput = errors.New("invalid engine configuration") // Profile records qualified public behavior, independently of Runtime advertisements. type Profile struct { + ProgrammaticToolCallingDisable bool Placements []string WebSearchControl, TextVerbosity, MCPBearer bool StructuredOutput bool diff --git a/services/agents-api/internal/execution/disabled_tools_test.go b/services/agents-api/internal/execution/disabled_tools_test.go new file mode 100644 index 000000000..308700237 --- /dev/null +++ b/services/agents-api/internal/execution/disabled_tools_test.go @@ -0,0 +1,52 @@ +package execution + +import ( + "encoding/json" + "testing" + + v1 "github.com/MiniMax-AI-Dev/parsar/contracts/agents-api/v1" + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/device" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/engine" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" +) + +func TestDisabledToolsUseCommonOperationQualification(t *testing.T) { + raw := json.RawMessage(`{"agent":{"model":"model","tools":[{"type":"web_search","mode":"disabled"},{"type":"programmatic_tool_calling","enabled":false}]},"environment":{"type":"none"}}`) + for _, kind := range []string{"codex", "claude_sdk", "mcode"} { + if err := (Policy{}).ValidateSessionConfiguration(kind, raw); err != nil { + t.Fatal(kind, err) + } + } + for _, qualified := range []bool{false, true} { + policy := Policy{Engines: engine.NewCatalog(map[string]engine.Profile{"new_harness": {Placements: []string{"none"}, ProgrammaticToolCallingDisable: qualified}})} + if err := policy.ValidateSessionConfiguration("new_harness", raw); (err == nil) != qualified { + t.Fatal("qualification differs", qualified, err) + } + if err := policy.ValidateSessionConfiguration("new_harness", json.RawMessage(`{"agent":{"model":"model"},"environment":{"type":"none"}}`)); err != nil { + t.Fatal("omission acquired a new prerequisite", err) + } + } +} + +func TestDisabledToolRequestPreservesIntentOnResume(t *testing.T) { + for _, disabled := range []bool{false, true} { + snapshot := Snapshot{Agent: v1.Agent{Model: "model"}} + if disabled { + snapshot.Agent.Tools = []json.RawMessage{json.RawMessage(`{"type":"programmatic_tool_calling","enabled":false}`)} + } + before, _ := json.Marshal(snapshot) + for _, nativeID := range []string{"", "native-session"} { + request, err := (&Dispatcher{}).executionRequest(t.Context(), store.Session{ID: "session"}, snapshot, device.KindCapabilities{}, store.SessionExecutionBinding{NativeSessionID: nativeID}) + if err != nil || request.ExecutionControls.DisableProgrammaticToolCalling != disabled || request.ExecutionControls.WebSearch != "disabled" || request.AgentSessionID != nativeID { + t.Fatal(request, err) + } + if request.ValidateProgrammaticToolCallingDisable(true) != nil || (request.ValidateProgrammaticToolCallingDisable(false) != nil) != disabled { + t.Fatal("runtime operation qualification differs") + } + } + after, _ := json.Marshal(snapshot) + if string(before) != string(after) { + t.Fatal("snapshot mutated") + } + } +} diff --git a/services/agents-api/internal/execution/engine_profile.go b/services/agents-api/internal/execution/engine_profile.go index 7204fd995..d29865cd6 100644 --- a/services/agents-api/internal/execution/engine_profile.go +++ b/services/agents-api/internal/execution/engine_profile.go @@ -26,19 +26,22 @@ func validateProfileConfiguration(profile engine.Profile, snapshot Snapshot) err return profileError(err) } } - functions, mcp, search, err := executionTools(snapshot.Agent.Tools) + tools, err := executionTools(snapshot.Agent.Tools) // Preserve each profile's admission error precedence when tool decoding fails. if profile.ValidateConfiguration != nil && err != nil { return err } if err == nil { - request := proto.PromptRequestPayload{ToolSearch: search, FunctionTools: functions} + if tools.DisableProgrammatic && !profile.ProgrammaticToolCallingDisable { + return errors.New("Disabling programmatic tool calling is not qualified for this engine.") + } + request := proto.PromptRequestPayload{ToolSearch: tools.Search, FunctionTools: tools.Functions} if err := request.ValidateToolSearch(profile.ToolSearch); err != nil { return err } } if profile.ValidateTools != nil { - if validationErr := profile.ValidateTools(snapshot.Environment, snapshot.Daemon != nil, functions, mcp); validationErr != nil { + if validationErr := profile.ValidateTools(snapshot.Environment, snapshot.Daemon != nil, tools.Functions, tools.MCP); validationErr != nil { return profileError(validationErr) } } diff --git a/services/agents-api/internal/execution/mcp.go b/services/agents-api/internal/execution/mcp.go index 752e0b9aa..4fc1a481c 100644 --- a/services/agents-api/internal/execution/mcp.go +++ b/services/agents-api/internal/execution/mcp.go @@ -11,23 +11,58 @@ import ( "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" ) -func executionTools(raw []json.RawMessage) ([]proto.FunctionTool, []proto.MCPHTTPServer, bool, error) { +type executionToolSet struct { + Functions []proto.FunctionTool + MCP []proto.MCPHTTPServer + Search bool + DisableProgrammatic bool +} + +func executionTools(raw []json.RawMessage) (executionToolSet, error) { functions := make([]json.RawMessage, 0, len(raw)) var servers []proto.MCPHTTPServer search := false + disableProgrammatic := false + controls := map[string]bool{} names := map[string]bool{} for _, value := range raw { var kind struct { Type string `json:"type"` } if json.Unmarshal(value, &kind) != nil { - return nil, nil, false, errors.New("invalid execution tool") + return executionToolSet{}, errors.New("invalid execution tool") + } + if kind.Type == "programmatic_tool_calling" || kind.Type == "web_search" { + if controls[kind.Type] { + return executionToolSet{}, errors.New("execution requires distinct tool controls") + } + controls[kind.Type] = true + if kind.Type == "programmatic_tool_calling" { + var control struct { + Type string `json:"type"` + Enabled *bool `json:"enabled"` + } + decoder := json.NewDecoder(bytes.NewReader(value)) + decoder.DisallowUnknownFields() + if decoder.Decode(&control) != nil || control.Enabled == nil || *control.Enabled { + return executionToolSet{}, errors.New("programmatic tool calling is not qualified for execution") + } + disableProgrammatic = true + } else { + var control struct { + Mode string `json:"mode"` + } + if json.Unmarshal(value, &control) != nil || control.Mode != "disabled" { + return executionToolSet{}, errors.New("only disabled web search is qualified for execution") + } + } + continue } if kind.Type == "tool_search" { decoder := json.NewDecoder(bytes.NewReader(value)) decoder.DisallowUnknownFields() if decoder.Decode(&kind) != nil || search { - return nil, nil, false, errors.New("execution requires one type-only tool_search declaration") + return executionToolSet{}, errors.New("execution requires one type-only tool_search declaration") } search = true continue @@ -40,16 +75,16 @@ func executionTools(raw []json.RawMessage) ([]proto.FunctionTool, []proto.MCPHTT decoder := json.NewDecoder(bytes.NewReader(value)) decoder.DisallowUnknownFields() if decoder.Decode(&tool) != nil || strings.TrimSpace(tool.ServerLabel) == "" || names[tool.ServerLabel] || tool.ConnectionOrigin != "service" || len(tool.RequestMetadata) != 0 || tool.Transport.Type != "http" || tool.Transport.Headers != nil { - return nil, nil, false, errors.New("unsupported execution MCP configuration") + return executionToolSet{}, errors.New("unsupported execution MCP configuration") } u, err := url.Parse(tool.Transport.ServerURL) if err != nil || u.Hostname() == "" || (u.Scheme != "http" && u.Scheme != "https") || u.User != nil || u.Fragment != "" || u.RawQuery != "" || u.ForceQuery { - return nil, nil, false, errors.New("unsupported execution MCP URL") + return executionToolSet{}, errors.New("unsupported execution MCP URL") } if tool.AllowedTools != nil { for _, name := range *tool.AllowedTools { if name == "" { - return nil, nil, false, errors.New("invalid execution MCP tool name") + return executionToolSet{}, errors.New("invalid execution MCP tool name") } } } @@ -58,5 +93,5 @@ func executionTools(raw []json.RawMessage) ([]proto.FunctionTool, []proto.MCPHTT ServerURL: tool.Transport.ServerURL, AllowedTools: tool.AllowedTools, Required: tool.Required}) } resolved, err := functionTools(functions) - return resolved, servers, search, err + return executionToolSet{Functions: resolved, MCP: servers, Search: search, DisableProgrammatic: disableProgrammatic}, err } diff --git a/services/agents-api/internal/execution/mcp_support_test.go b/services/agents-api/internal/execution/mcp_support_test.go index 7cf9af17f..bae016a74 100644 --- a/services/agents-api/internal/execution/mcp_support_test.go +++ b/services/agents-api/internal/execution/mcp_support_test.go @@ -17,13 +17,13 @@ func mcpSupportFixture(t *testing.T) (Snapshot, []proto.MCPHTTPServer, device.Ki tool := json.RawMessage(`{"type":"mcp","server_label":"tickets","connection_origin":"service","transport":{"type":"http","server_url":"https://mcp.example/tools"}}`) snapshot := Snapshot{Agent: v1.Agent{Model: "model", Tools: []json.RawMessage{tool}}, Environment: &v1.Environment{Type: "none"}, VaultIDs: []string{vault}, MCPCredentials: []store.MCPCredentialBinding{{ServerLabel: "tickets", ServerURL: "https://mcp.example/tools", VaultID: vault, CredentialID: credential, AuthType: "static_bearer"}}} - _, servers, _, err := executionTools(snapshot.Agent.Tools) + tools, err := executionTools(snapshot.Agent.Tools) if err != nil { t.Fatal(err) } caps := device.KindCapabilities{EnvironmentNone: true, MCPHTTPTools: true, MCPHTTPBearerAuth: true, MCPHTTPRequired: true, Preparation: true} - return snapshot, servers, caps + return snapshot, tools.MCP, caps } func TestMCPPublicBearerPolicyIsIndependentOfRuntimeCapabilities(t *testing.T) { diff --git a/services/agents-api/internal/execution/mcp_test.go b/services/agents-api/internal/execution/mcp_test.go index aba22efeb..d4bd63d6c 100644 --- a/services/agents-api/internal/execution/mcp_test.go +++ b/services/agents-api/internal/execution/mcp_test.go @@ -27,7 +27,8 @@ func TestMCPRequiresSupportedServicePlacement(t *testing.T) { } } function := json.RawMessage(`{"type":"function","name":"lookup","description":"Read","parameters":{"type":"object"},"defer_loading":false}`) - functions, servers, _, err := executionTools([]json.RawMessage{tool, function}) + tools, err := executionTools([]json.RawMessage{tool, function}) + functions, servers := tools.Functions, tools.MCP if err != nil || len(functions) != 1 || functions[0].Name != "lookup" || len(servers) != 1 || servers[0].AllowedTools == nil || len(*servers[0].AllowedTools) != 0 { t.Fatal("mixed tool configuration lost the deny-all declaration", functions, servers, err) } diff --git a/services/agents-api/internal/execution/request.go b/services/agents-api/internal/execution/request.go index e12b7d1f4..cab8e3759 100644 --- a/services/agents-api/internal/execution/request.go +++ b/services/agents-api/internal/execution/request.go @@ -15,7 +15,7 @@ func (d *Dispatcher) executionRequest(ctx context.Context, session store.Session if recoverNativeSession && !caps.NativeSessionRecovery { return proto.PromptRequestPayload{}, errors.New("native session recovery is unavailable") } - functions, mcp, search, err := executionTools(snapshot.Agent.Tools) + tools, err := executionTools(snapshot.Agent.Tools) if err != nil { return proto.PromptRequestPayload{}, err } @@ -41,11 +41,11 @@ func (d *Dispatcher) executionRequest(ctx context.Context, session store.Session if verbosity == "" { verbosity = "medium" } - controls := &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: verbosity} + controls := &proto.ExecutionControls{DisableProgrammaticToolCalling: tools.DisableProgrammatic, WebSearch: "disabled", TextVerbosity: verbosity} if snapshot.Agent.Text.Format.Type == "json_schema" { controls.OutputFormat = &proto.OutputFormat{Type: "json_schema", Schema: snapshot.Agent.Text.Format.Schema} } - request := proto.PromptRequestPayload{AgentKind: session.Engine, FunctionTools: functions, ToolSearch: search, + request := proto.PromptRequestPayload{AgentKind: session.Engine, FunctionTools: tools.Functions, ToolSearch: tools.Search, AgentOptions: options, ExecutionControls: controls, AgentStateKey: "agents-api-" + session.ID, AgentSessionID: bound.NativeSessionID, ReleaseOnCompletion: true, StrictResume: true, RequireExistingNativeSession: recoverNativeSession, @@ -53,24 +53,24 @@ func (d *Dispatcher) executionRequest(ctx context.Context, session store.Session ObserveSubagentIdentities: snapshot.Agent.MultiAgent.Enabled, MaxConcurrentSubagents: snapshot.Agent.MultiAgent.MaxConcurrentSubagents, DisableSubagents: !snapshot.Agent.MultiAgent.Enabled} - if len(mcp) != 0 { - selected, err := d.mcpExecutionCredentials(session.Engine, snapshot, mcp, caps) + if len(tools.MCP) != 0 { + selected, err := d.mcpExecutionCredentials(session.Engine, snapshot, tools.MCP, caps) if err != nil { return proto.PromptRequestPayload{}, err } if len(selected) > 0 && d.Store == nil { return proto.PromptRequestPayload{}, errors.New("authenticated MCP execution is unavailable") } - for i := range mcp { - if binding, ok := selected[mcp[i].ServerLabel]; ok { + for i := range tools.MCP { + if binding, ok := selected[tools.MCP[i].ServerLabel]; ok { token, err := d.Store.MCPBearerToken(ctx, session.TenantID, snapshot.VaultIDs, binding) if err != nil { return proto.PromptRequestPayload{}, err } - mcp[i].BearerToken = &token + tools.MCP[i].BearerToken = &token } } - request.MCPHTTPServers = &mcp + request.MCPHTTPServers = &tools.MCP } return request, nil } diff --git a/services/agents-api/internal/execution/support.go b/services/agents-api/internal/execution/support.go index f8ec22604..c00dbc5cf 100644 --- a/services/agents-api/internal/execution/support.go +++ b/services/agents-api/internal/execution/support.go @@ -107,17 +107,20 @@ func (p Policy) engineCapabilities(peer *gateway.Session, engine string, snapsho if !snapshot.Agent.MultiAgent.Enabled && !caps.SubagentControl { return fail("device must advertise subagent_control") } - functions, mcp, search, err := executionTools(snapshot.Agent.Tools) + tools, err := executionTools(snapshot.Agent.Tools) if err != nil { return fail("invalid execution tool configuration") } - if err := (proto.PromptRequestPayload{ToolSearch: search, FunctionTools: functions}).ValidateToolSearch(caps.ToolSearch); err != nil { + if err := (proto.PromptRequestPayload{ToolSearch: tools.Search, FunctionTools: tools.Functions}).ValidateToolSearch(caps.ToolSearch); err != nil { return fail(err.Error()) } - if len(functions) > 0 && !caps.FunctionTools { + if tools.DisableProgrammatic && !caps.ProgrammaticToolCallingDisable { + return fail("device must support disabling programmatic tool calling") + } + if len(tools.Functions) > 0 && !caps.FunctionTools { return fail("device must advertise function_tools") } - if _, err := p.mcpExecutionCredentials(engine, snapshot, mcp, caps); err != nil { + if _, err := p.mcpExecutionCredentials(engine, snapshot, tools.MCP, caps); err != nil { return device.KindCapabilities{}, err } if snapshot.Environment != nil && (snapshot.Environment.Type == "openai_hosted" || snapshot.Environment.Type == "self_hosted") { diff --git a/services/agents-api/internal/execution/tool_search_test.go b/services/agents-api/internal/execution/tool_search_test.go index 1a0ac6d4c..eed3015f1 100644 --- a/services/agents-api/internal/execution/tool_search_test.go +++ b/services/agents-api/internal/execution/tool_search_test.go @@ -35,11 +35,11 @@ func TestDiscoveryKeepsMixedFunctionDefinitions(t *testing.T) { if err := json.Unmarshal([]byte(discoveryConfiguration), &snapshot); err != nil { t.Fatal(err) } - functions, mcp, search, err := executionTools(snapshot.Agent.Tools) - if err != nil || len(mcp) != 0 || !search || len(functions) != 2 || !functions[0].DeferLoading || functions[1].DeferLoading { - t.Fatal(functions, mcp, search, err) + tools, err := executionTools(snapshot.Agent.Tools) + if err != nil || len(tools.MCP) != 0 || !tools.Search || len(tools.Functions) != 2 || !tools.Functions[0].DeferLoading || tools.Functions[1].DeferLoading { + t.Fatal(tools, err) } - request := proto.PromptRequestPayload{ToolSearch: search, FunctionTools: functions} + request := proto.PromptRequestPayload{ToolSearch: tools.Search, FunctionTools: tools.Functions} if request.ValidateToolSearch(true) != nil || request.ValidateToolSearch(false) == nil { t.Fatal("Runtime support is not operation-specific") } diff --git a/services/agents-api/internal/store/tool_policy_native_test.go b/services/agents-api/internal/store/tool_policy_native_test.go new file mode 100644 index 000000000..eef1559c8 --- /dev/null +++ b/services/agents-api/internal/store/tool_policy_native_test.go @@ -0,0 +1,139 @@ +package store_test + +import ( + "context" + "encoding/json" + "net/http/httptest" + "os" + "os/exec" + "path/filepath" + "testing" + "time" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/device" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/api" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/store" + "github.com/google/uuid" +) + +// Run once per engine with a real provider and a daemon containing that adapter. +// Native tool inventory qualification is separate from these public API checks. +func TestNativeToolPolicyPublicExecution(t *testing.T) { + python, binary, root, optionsFile := os.Getenv("PARSAR_OFFICIAL_SDK_PYTHON"), os.Getenv("PARSAR_NATIVE_DAEMON_BIN"), os.Getenv("PARSAR_NATIVE_PROOF_DIR"), os.Getenv("PARSAR_TOOL_POLICY_REAL_OPTIONS") + if python == "" || binary == "" || root == "" || optionsFile == "" { + t.Skip("native daemon, pinned SDK, private real-model options and evidence directory required") + } + kind := os.Getenv("PARSAR_TOOL_POLICY_ENGINE") + if kind != "codex" && kind != "claude_sdk" && kind != "mcode" { + t.Fatal("tool policy acceptance requires codex, claude_sdk or mcode") + } + raw, err := os.ReadFile(optionsFile) + if err != nil { + t.Fatal(err) + } + var options map[string]any + if json.Unmarshal(raw, &options) != nil { + t.Fatal("invalid private options") + } + model, _ := options["model"].(string) + if model == "" { + t.Fatal("real model required") + } + h := newDispatchHarness(t) + h.d.Options = func(context.Context, store.Session) (map[string]any, error) { return options, nil } + home, err := os.MkdirTemp(root, "tool-policy-"+kind+"-") + if err != nil { + t.Fatal(err) + } + ctx, cancel := context.WithTimeout(t.Context(), 15*time.Minute) + defer cancel() + worker, err := execution.StartWorker(ctx, h.d) + if err != nil { + t.Fatal(err) + } + done := make(chan error, 1) + go func() { done <- worker.Run(ctx) }() + defer func() { + cancel() + select { + case <-done: + case <-time.After(20 * time.Second): + t.Error("worker did not stop") + } + }() + token, foreign, foreignTenant := uuid.NewString(), uuid.NewString(), uuid.NewString() + auth, err := api.NewAuthenticator([]api.APIKey{ + {OrganizationID: "test", ProjectID: h.tenant, SubjectKind: "service_account", SubjectID: "owner", TokenSHA256: device.HashCredential(token), TenantID: h.tenant}, + {OrganizationID: "test", ProjectID: foreignTenant, SubjectKind: "service_account", SubjectID: "other", TokenSHA256: device.HashCredential(foreign), TenantID: foreignTenant}, + }) + if err != nil { + t.Fatal(err) + } + handler, err := api.NewHandler(h.s, auth, kind, api.WithExecution(worker), api.WithExecutionPolicy(h.d.Policy)) + if err != nil { + t.Fatal(err) + } + server := httptest.NewServer(handler) + defer server.Close() + stop := startNativeEngineDaemon(t, h, home, binary, kind) + defer func() { stop() }() + evidence := filepath.Join(home, "public.json") + run := func(stage string) { + cmd := exec.CommandContext(ctx, python, "../../tests/official_tool_policy.py", server.URL, token, foreign, model, stage, evidence) + if output, err := cmd.CombinedOutput(); err != nil { + t.Fatalf("tool policy %s: %v %s; evidence %s", stage, err, output, home) + } + } + run("initial") + var proof struct { + Sessions []struct { + ID string `json:"id"` + FirstTurn string `json:"first_turn"` + } `json:"sessions"` + Omitted []string `json:"omitted"` + } + raw, err = os.ReadFile(evidence) + if err != nil || json.Unmarshal(raw, &proof) != nil || len(proof.Sessions) != 4 || len(proof.Omitted) != 2 { + t.Fatal("invalid public evidence", err) + } + page, err := h.s.ListSessions(ctx, h.tenant, "", 100, true, nil) + if err != nil || len(page.Sessions) != 1+len(proof.Sessions)+len(proof.Omitted) { + t.Fatal("rejected configuration persisted a Session", err) + } + foreignPage, err := h.s.ListSessions(ctx, foreignTenant, "", 100, true, nil) + if err != nil || len(foreignPage.Sessions) != 0 { + t.Fatal("foreign Agent reference persisted a Session", err) + } + nativeIDs := make(map[string]string, len(proof.Sessions)) + for _, item := range proof.Sessions { + session, err := h.s.GetSession(ctx, h.tenant, item.ID) + if err != nil || session.Engine != kind { + t.Fatal("selected engine was not persisted", err) + } + turn, err := h.s.GetTurn(ctx, h.tenant, item.ID, item.FirstTurn) + if err != nil { + t.Fatal(err) + } + inputs, err := h.s.ListTurnInputs(ctx, h.tenant, item.ID, item.FirstTurn, 0, 100) + var outcome execution.Result + if err != nil || len(inputs) != 1 || json.Unmarshal(turn.Outcome, &outcome) != nil || outcome.AppliedThrough != inputs[0].Sequence { + t.Fatal("native text input receipt missing", err) + } + binding, err := h.s.GetSessionExecutionBinding(ctx, h.tenant, item.ID) + if err != nil || binding.NativeSessionID == "" { + t.Fatal("native binding missing", err) + } + nativeIDs[item.ID] = binding.NativeSessionID + } + stop() + stop = startNativeEngineDaemon(t, h, home, binary, kind) + run("resume") + for id, before := range nativeIDs { + after, err := h.s.GetSessionExecutionBinding(ctx, h.tenant, id) + if err != nil || after.NativeSessionID != before { + t.Fatal("cold continuation changed native history", err) + } + } + t.Logf("Real %s disabled tool policy SDK/raw HTTP, native receipts, cold continuation and tenant isolation passed: %s", kind, home) +} diff --git a/services/agents-api/tests/official_tool_policy.py b/services/agents-api/tests/official_tool_policy.py new file mode 100644 index 000000000..5859afb72 --- /dev/null +++ b/services/agents-api/tests/official_tool_policy.py @@ -0,0 +1,159 @@ +"""Opt-in disabled tool policy acceptance through the pinned SDK and raw HTTP.""" +import importlib.metadata +import json +import sys +import uuid +from pathlib import Path + +import httpx2 +from openai import BadRequestError, NotFoundError, OpenAI + + +def main(): + base, token, foreign, model, stage, evidence = sys.argv[1:] + assert stage in {"initial", "resume"} + pin = json.loads((Path(__file__).resolve().parents[3] / "contracts/agents-api/upstream.json").read_text()) + distribution = importlib.metadata.distribution("openai") + assert distribution.version == pin["sdk_version"] + assert json.loads(distribution.read_text("direct_url.json"))["vcs_info"]["commit_id"] == pin["commit"] + client = OpenAI(base_url=base + "/v1", api_key=token, max_retries=0, + _strict_response_validation=True, + http_client=httpx2.Client(trust_env=False, timeout=150)) + other = client.with_options(api_key=foreign) + raw = httpx2.Client(trust_env=False, timeout=150) + sessions = client.beta.agents.sessions + headers = {"Authorization": "Bearer " + token, "OpenAI-Beta": "agents=v1"} + foreign_headers = {**headers, "Authorization": "Bearer " + foreign} + proof = {"sessions": [], "omitted": []} if stage == "initial" else json.loads(Path(evidence).read_text()) + tools = [{"type": "web_search", "mode": "disabled"}, + {"type": "programmatic_tool_calling", "enabled": False}] + # These are pinned response defaults, including explicit nullable fields. + expected = [{"type": "web_search", "mode": "disabled", "context_size": "medium", + "allowed_domains": None, "location": None}, tools[1]] + config = {"model": model, "instructions": "Follow user instructions and remember supplied markers.", "tools": tools} + + def request(method, path, *, status=200, foreign_tenant=False, **kwargs): + response = raw.request(method, base + "/v1/agents" + path, + headers=foreign_headers if foreign_tenant else headers, **kwargs) + assert response.status_code == status, (method, path, response.status_code) + return response.json() + + def save(): + Path(evidence).write_text(json.dumps(proof, indent=2)) + + def check_config(sid): + assert [tool.to_dict() for tool in sessions.retrieve(sid).agent.tools] == expected + assert request("GET", "/sessions/" + sid)["agent"]["tools"] == expected + + def execute(entry, prompt): + sid = entry["id"] + events = [] + with sessions.events.stream(sid, timeout=150) as stream: + sessions.events.create(sid, events=[{"type": "agent.session.input.message", "input": [ + {"role": "user", "content": [{"type": "input_text", "text": prompt}]}]}], + idempotency_key=str(uuid.uuid4())) + for event in stream: + events.append(event.type) + assert event.type not in {"agent.session.failed", "agent.session.turn.failed", "agent.session.requires_action"}, event.type + if event.type == "agent.session.idle": + break + else: + raise AssertionError("stream ended without idle") + assert events.index("agent.session.turn.created") < events.index("agent.session.turn.completed") < events.index("agent.session.idle") + turns = sessions.turns.list(sid, order="asc", limit=100).data + assert turns[-1].status == "completed", turns[-1].status + assert len(turns) == (1 if stage == "initial" else 2) + items = sessions.items.list(sid, order="asc", limit=100).data + answers = [item for item in items if item.type == "message" and item.role == "assistant" and item.turn_id == turns[-1].id] + text = "\n".join(content.text for item in answers for content in item.content if content.type == "output_text") + assert entry["marker"] in text, "Native answer did not retain the marker" + assert not any(item.type in {"web_search_call", "function_call", "mcp_call", "command_execution"} for item in items) + public = request("GET", "/sessions/" + sid + "/items", params={"order": "asc", "limit": 100}) + assert public["data"] == [item.to_dict() for item in items] + entry["first_turn" if stage == "initial" else "resumed_turn"] = turns[-1].id + entry[stage + "_events"] = events + save() + + def reject_configuration(payload): + request("POST", "/sessions", status=400, json=payload) + try: + sessions.create(**payload) + raise AssertionError("Unsupported enabled tool configuration was admitted") + except BadRequestError: + pass + + try: + if stage == "initial": + sdk_agent = client.beta.agents.create(**config) + assert [tool.to_dict() for tool in sdk_agent.tools] == expected + raw_agent = request("POST", "", json=config) + assert raw_agent["tools"] == expected + proof["agents"] = [sdk_agent.id, raw_agent["id"]] + for aid in proof["agents"]: + assert [tool.to_dict() for tool in client.beta.agents.retrieve(aid).tools] == expected + assert request("GET", "/" + aid)["tools"] == expected + payloads = [ + ("sdk_saved", {"agent_id": sdk_agent.id}), + ("sdk_inline", {"agent": config}), + ("raw_saved", {"agent_id": raw_agent["id"]}), + ("raw_inline", {"agent": config}), + ] + for name, payload in payloads: + payload["environment"] = {"type": "none"} + if name.startswith("sdk"): + sid = sessions.create(**payload).id + else: + sid = request("POST", "/sessions", json=payload)["id"] + entry = {"id": sid, "path": name, "marker": "TOOL-POLICY-" + uuid.uuid4().hex} + proof["sessions"].append(entry) + save() + check_config(sid) + execute(entry, "Remember this exact marker for later: " + entry["marker"] + ". Reply with that marker only.") + for enabled in [{"type": "web_search", "mode": "cached"}, + {"type": "web_search", "mode": "live"}, + {"type": "programmatic_tool_calling", "enabled": True}]: + reject_configuration({"agent": {"model": model, "tools": [enabled]}, "environment": {"type": "none"}}) + reject_configuration({"agent_id": sdk_agent.id, "agent": {"tools": [enabled]}, "environment": {"type": "none"}}) + # Saving PTC intent is independent of Session execution qualification. + enabled_agent = client.beta.agents.create(model=model, tools=[{"type": "programmatic_tool_calling", "enabled": True}]) + reject_configuration({"agent_id": enabled_agent.id, "environment": {"type": "none"}}) + payload = {"agent": {"model": model}, "environment": {"type": "none"}} + omitted = sessions.create(**payload) + assert omitted.agent.tools == [] + proof["omitted"].append(omitted.id) + omitted_raw = request("POST", "/sessions", json=payload) + assert omitted_raw["agent"]["tools"] == [] + proof["omitted"].append(omitted_raw["id"]) + for aid in proof["agents"]: + request("GET", "/" + aid, status=404, foreign_tenant=True) + request("POST", "/sessions", status=404, foreign_tenant=True, + json={"agent_id": aid, "environment": {"type": "none"}}) + try: + other.beta.agents.sessions.create(agent_id=aid, environment={"type": "none"}) + raise AssertionError("Foreign tenant used a saved Agent") + except NotFoundError: + pass + else: + for entry in proof["sessions"]: + check_config(entry["id"]) + # The marker is absent from this input: success requires native history. + execute(entry, "Reply with the exact TOOL-POLICY marker I asked you to remember earlier, and nothing else.") + proof["passed"] = True + for entry in proof["sessions"]: + sid = entry["id"] + for suffix in ["", "/items", "/turns"]: + request("GET", "/sessions/" + sid + suffix, status=404, foreign_tenant=True) + try: + other.beta.agents.sessions.retrieve(sid) + raise AssertionError("Foreign tenant retrieved a Session") + except NotFoundError: + pass + finally: + save() + raw.close() + other.close() + client.close() + + +if __name__ == "__main__": + main()