From f6d2c3f2a1dc2ff5f177213bd2b20d78496011d0 Mon Sep 17 00:00:00 2001 From: saladday <1203511142@qq.com> Date: Tue, 22 Sep 2026 16:52:08 +0800 Subject: [PATCH] Qualify Claude structured output in hosted workspaces --- CONTRIBUTING.md | 20 +- .../internal/agent/claudesdk/options.go | 2 +- .../internal/agent/claudesdk/preparation.go | 3 + .../claudesdk/preparation_fixture_test.go | 13 +- .../internal/agent/claudesdk/readiness.go | 4 + .../claudesdk/workspace_structured_test.go | 75 +++++++ apps/parsar-daemon/internal/cli/claude_sdk.go | 5 +- contracts/agents-api/README.md | 2 +- contracts/agents-api/openapi.yaml | 68 +++--- contracts/agents-api/structured-output.md | 27 ++- packages/claude-sdk-adapter/src/adapter.ts | 2 +- packages/claude-sdk-adapter/src/request.ts | 3 +- .../claude-sdk-adapter/src/runtime_check.ts | 2 +- packages/claude-sdk-adapter/src/workspace.ts | 5 +- .../tests/mcp_workspace.test.mjs | 2 + .../tests/workspace.test.mjs | 28 +++ scripts/check-claude-sdk-runtime.mjs | 2 +- services/agents-api/internal/api/handler.go | 2 +- .../internal/api/hosted_structured_test.go | 60 +++++ services/agents-api/internal/engine/claude.go | 7 +- .../official_hosted_structured_native.py | 208 ++++++++++++++++++ 21 files changed, 484 insertions(+), 56 deletions(-) create mode 100644 apps/parsar-daemon/internal/agent/claudesdk/workspace_structured_test.go create mode 100644 services/agents-api/internal/api/hosted_structured_test.go create mode 100644 services/agents-api/tests/official_hosted_structured_native.py diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 8e0bc281d..9872170c0 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1999,12 +1999,20 @@ admission requires the selected profile's structured-output qualification, and only requests using this option require the Runtime's `structured_output` and message-observation capabilities. A capability advertisement does not qualify a new public combination. Claude advertises this operation only when the installed -SDK bridge reports its `structured_output` feature and the selected Runtime is -not a workspace profile. - -The current qualified path is Claude SDK, `environment:none`, medium verbosity, -single Agent, with optional ordinary function tools and text results. Workspace, -HTTP MCP, Subagent combinations and non-object root schemas remain unqualified. +SDK bridge reports its `structured_output` feature. A workspace Runtime also +requires the complete local Runtime contract and `workspace_structured_output`; +preparation checks that bundle before native launch. These remain adapter readiness +features, not new Core lifecycle or public protocol variants. + +The current qualified path is Claude SDK, `environment:none` or Core-managed +Docker `openai_hosted`, medium verbosity, single Agent, with optional ordinary +function tools and text results. The workspace uses its existing preparation and +native sandbox with only the SDK's configured `StructuredOutput` tool added to +inventory and permission checks. Frozen schemas reach preparation before the +input handoff; Start cannot replace them. Skills, Plugins, capability directories, +HTTP MCP, Subagent/tool-discovery combinations and non-object root schemas remain +unqualified. Check resolved template contents as well as inline configuration; +ordinary text requests retain their existing qualifications. The SDK uses binary64 JSON numbers: reject execution schemas whose numeric values would change during that conversion, without narrowing saved Agent storage. Codex and MiniMax structured output remain explicit execution gaps. diff --git a/apps/parsar-daemon/internal/agent/claudesdk/options.go b/apps/parsar-daemon/internal/agent/claudesdk/options.go index b0ad01366..a9ecb8fee 100644 --- a/apps/parsar-daemon/internal/agent/claudesdk/options.go +++ b/apps/parsar-daemon/internal/agent/claudesdk/options.go @@ -79,7 +79,7 @@ func prepareConfiguration(config Config, req proto.PromptRequestPayload) (startR } if req.ExecutionControls != nil && req.ExecutionControls.OutputFormat != nil { format := req.ExecutionControls.OutputFormat - if format.Type != "json_schema" || !req.ObserveMessages || !req.DisableSubagents || config.Workspace != nil || req.MCPHTTPServers != nil { + if format.Type != "json_schema" || !req.ObserveMessages || !req.DisableSubagents || req.MCPHTTPServers != nil || (req.LocalEnvironment != nil && (len(req.LocalEnvironment.MCP) != 0 || len(req.LocalEnvironment.Skills) != 0)) { return fail("structured output requires the qualified message-observing single-agent function profile") } if err := proto.ValidateBinary64Schema(format.Schema); err != nil { diff --git a/apps/parsar-daemon/internal/agent/claudesdk/preparation.go b/apps/parsar-daemon/internal/agent/claudesdk/preparation.go index 91166a81e..cdd683865 100644 --- a/apps/parsar-daemon/internal/agent/claudesdk/preparation.go +++ b/apps/parsar-daemon/internal/agent/claudesdk/preparation.go @@ -53,6 +53,9 @@ func NewPreparationFactory(config Config) agent.PreparationFactory { if err != nil || !info.supportsWorkspacePreparation() { return nil, fmt.Errorf("claudesdk: packaged runtime does not support workspace preparation") } + if start.OutputFormat != nil && !info.SupportsWorkspaceStructuredOutput() { + return nil, fmt.Errorf("claudesdk: packaged runtime does not support workspace structured output") + } if start.Subagents != nil && !info.SupportsSubagents() { return nil, fmt.Errorf("claudesdk: packaged runtime does not support subagent resources") } diff --git a/apps/parsar-daemon/internal/agent/claudesdk/preparation_fixture_test.go b/apps/parsar-daemon/internal/agent/claudesdk/preparation_fixture_test.go index 3986d6ae2..2c02bd316 100644 --- a/apps/parsar-daemon/internal/agent/claudesdk/preparation_fixture_test.go +++ b/apps/parsar-daemon/internal/agent/claudesdk/preparation_fixture_test.go @@ -44,6 +44,12 @@ func runPreparationHelper() { } else if mode == "old-command-runtime" { features = []string{"workspace_tools", "workspace_prepare"} } + if strings.HasPrefix(mode, "structured-") { + features = append(features, "local_runtime_v1", "structured_output") + if mode == "structured-ready" { + features = append(features, "workspace_structured_output") + } + } _ = json.NewEncoder(os.Stdout).Encode(RuntimeInfo{Type: "runtime_ready", Protocol: 2, Node: "fixture", SDK: "fixture", MCP: "fixture", Native: "fixture", Features: features}) return } @@ -105,7 +111,12 @@ func runPreparationHelper() { return } emit(bridgeEvent{Type: "input_ready", SessionID: request.Resume}) - emit(bridgeEvent{Type: "delta", Delta: "partial"}) + if request.ObserveMessages { + text := "completed" + emit(bridgeEvent{Type: "output_message", Message: &proto.OutputMessagePayload{ID: "native-message", Status: "completed", Text: &text}}) + } else { + emit(bridgeEvent{Type: "delta", Delta: "partial"}) + } emit(bridgeEvent{Type: "usage", ResultID: "native-result", SessionID: request.Resume, Usage: json.RawMessage(usageFixture)}) emit(bridgeEvent{Type: "input_closed", SessionID: request.Resume}) emit(bridgeEvent{Type: "result", SessionID: request.Resume, Text: "completed"}) diff --git a/apps/parsar-daemon/internal/agent/claudesdk/readiness.go b/apps/parsar-daemon/internal/agent/claudesdk/readiness.go index b485c4a6a..878f3ed16 100644 --- a/apps/parsar-daemon/internal/agent/claudesdk/readiness.go +++ b/apps/parsar-daemon/internal/agent/claudesdk/readiness.go @@ -41,6 +41,10 @@ func (info RuntimeInfo) SupportsStructuredOutput() bool { return slices.Contains(info.Features, "structured_output") } +func (info RuntimeInfo) SupportsWorkspaceStructuredOutput() bool { + return info.SupportsStructuredOutput() && info.SupportsLocalRuntime() && slices.Contains(info.Features, "workspace_structured_output") +} + func (info RuntimeInfo) SupportsSubagents() bool { return slices.Contains(info.Features, "subagent_resources") } diff --git a/apps/parsar-daemon/internal/agent/claudesdk/workspace_structured_test.go b/apps/parsar-daemon/internal/agent/claudesdk/workspace_structured_test.go new file mode 100644 index 000000000..a9f7a8ebc --- /dev/null +++ b/apps/parsar-daemon/internal/agent/claudesdk/workspace_structured_test.go @@ -0,0 +1,75 @@ +//go:build unix + +package claudesdk + +import ( + "encoding/json" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/MiniMax-AI-Dev/parsar/internal/agentdaemon/proto" +) + +func TestWorkspaceStructuredPreparationQualificationAndFrozenSchema(t *testing.T) { + for _, mode := range []string{"structured-missing", "structured-ready"} { + t.Run(mode, func(t *testing.T) { + config := preparationFixture(t, mode) + req := preparationRequest() + req.ObserveMessages = true + schema := `{"type":"object","properties":{"n":{"const":9007199254740992}}}` + req.ExecutionControls = &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: "medium", OutputFormat: &proto.OutputFormat{Type: "json_schema", Schema: json.RawMessage(schema)}} + p, err := NewPreparationFactory(config)(t.Context(), req) + if mode == "structured-missing" { + if err == nil || !strings.Contains(err.Error(), "workspace structured output") { + t.Fatal("unqualified bundle admitted", err) + } + if _, err := os.Stat(filepath.Join(config.StateDir, "launched")); !os.IsNotExist(err) { + t.Fatal("unqualified request reached native launch", err) + } + return + } + if err != nil { + t.Fatal(err) + } + defer p.Close() + req.ExecutionControls.OutputFormat.Schema[0] = ' ' + var frozen startRequest + if err := json.Unmarshal(waitPreparationFile(t, filepath.Join(config.StateDir, "prepare.json")), &frozen); err != nil { + t.Fatal(err) + } + if frozen.OutputFormat == nil || string(frozen.OutputFormat.Schema) != schema { + t.Fatal("prepared native schema changed with caller memory") + } + out := make(chan proto.Envelope, 16) + if _, err := p.Start(t.Context(), "run", proto.TextInput("hello"), out); err != nil { + t.Fatal(err) + } + for event := range out { + if event.Type == proto.TypeError { + t.Fatal("prepared execution failed", string(event.Payload)) + } + } + var started map[string]json.RawMessage + if err := json.Unmarshal(waitPreparationFile(t, filepath.Join(config.StateDir, "start.json")), &started); err != nil || len(started) != 2 || started["output_format"] != nil { + t.Fatal("Start replaced the prepared configuration", err) + } + }) + } +} + +func TestWorkspaceStructuredReadinessRequiresCompleteLocalContract(t *testing.T) { + features := []string{"workspace_tools", "workspace_prepare", "workspace_command_observations", "local_runtime_v1", "structured_output", "workspace_structured_output"} + if !(RuntimeInfo{Features: features}).SupportsWorkspaceStructuredOutput() { + t.Fatal("qualified Runtime unavailable") + } + for i, missing := range features { + t.Run(missing, func(t *testing.T) { + partial := append(append([]string{}, features[:i]...), features[i+1:]...) + if (RuntimeInfo{Features: partial}).SupportsWorkspaceStructuredOutput() { + t.Fatal("incomplete workspace bundle admitted") + } + }) + } +} diff --git a/apps/parsar-daemon/internal/cli/claude_sdk.go b/apps/parsar-daemon/internal/cli/claude_sdk.go index 1cfb22748..6c5224542 100644 --- a/apps/parsar-daemon/internal/cli/claude_sdk.go +++ b/apps/parsar-daemon/internal/cli/claude_sdk.go @@ -101,7 +101,10 @@ func discoverClaudeSDK(rc *runContext, profile string, check func(context.Contex out.Info.Capabilities.MessageImages = info.SupportsMessageImages() out.Info.Capabilities.FunctionResultImages = info.SupportsFunctionResultImages() out.Info.Capabilities.ToolSearch = out.Config.Workspace == nil && info.SupportsToolSearch() - out.Info.Capabilities.StructuredOutput = out.Config.Workspace == nil && info.SupportsStructuredOutput() + out.Info.Capabilities.StructuredOutput = info.SupportsStructuredOutput() + if out.Config.Workspace != nil { + out.Info.Capabilities.StructuredOutput = info.SupportsWorkspaceStructuredOutput() + } out.Info.Capabilities.SubagentObservations = info.SupportsSubagents() out.Info.Capabilities.MCPHTTPTools = info.SupportsHTTPMCP() out.Info.Capabilities.MCPHTTPBearerAuth = info.SupportsHTTPMCPBearer() diff --git a/contracts/agents-api/README.md b/contracts/agents-api/README.md index d0d352226..d6843fcf5 100644 --- a/contracts/agents-api/README.md +++ b/contracts/agents-api/README.md @@ -164,7 +164,7 @@ user-managed enrollment remain outside this qualification. | --- | --- | | Subagents / multi_agent | Six reads and same-child recovery have three-harness Docker evidence; optional native operations, live child progress, full lifecycle/interactions and tool combinations remain explicit gaps | | Environment Templates | Unsupported restricted hostname forms, unqualified installation overrides/null network and exact hosted errors remain gaps. CRUD/list, files, env/setup/system/npm/Python, inline/referenced Skills, Plugins, workspace capability directories and Session references have recorded coverage. Environment Plugin MCP transport and placement limits are [listed separately](environment-templates.md#environment-origin-mcp-plugins) | -| Input and configuration | Non-text initial input, broader content/configuration unions and reasoning/verbosity combinations; [structured output](structured-output.md) has a qualified Claude function profile, with other combinations remaining gaps | +| Input and configuration | Non-text initial input, broader content/configuration unions and reasoning/verbosity combinations; [structured output](structured-output.md) has qualified Claude function profiles on none and Core-managed Docker openai_hosted, with other combinations remaining gaps | | Tools and interactions | [Deferred discovery qualification](tool-search.md), other tool types, effective tool-set enforcement and result/cancel publication ordering; MiniMax public functions and service-origin MCP remain unsupported | | Vault and Credentials | OAuth/refresh, archive semantics, revocation/concurrent mutation and exact hosted selection/error behavior; static bearer CRUD/token replacement is already present | | Existing resources | Full Item/SSE/Usage variants, omitted/null/default/error semantics, pagination and overlapping lifecycle behavior beyond recorded cases | diff --git a/contracts/agents-api/openapi.yaml b/contracts/agents-api/openapi.yaml index be0a2b2a5..83fe97568 100644 --- a/contracts/agents-api/openapi.yaml +++ b/contracts/agents-api/openapi.yaml @@ -2698,39 +2698,41 @@ paths: reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without replay; retry with stream=false - to retrieve the Session. Claude SDK environment:none supports qualified object-root - json_schema output with medium verbosity, single-Agent execution and ordinary - functions; other combinations remain unsupported. Other non-text initial input - remains unsupported. Basic Codex and Claude SDK openai_hosted creation requires - an explicitly configured managed provider. The Claude workspace profile supports - non-deferred function tools with text or successful inline PNG/JPEG results - alongside native workspace tools; HTTP MCP remains unsupported. Idle Sessions - provision automatically; initial provisioning has no caller connection action. - Network defaults to enabled; disabled and restricted exact ASCII hostnames - are supported. Restricted policy requires 1–100 allowed domains. Unsupported - hostname forms and startup installations are rejected. Confidential env, system/npm/Python - packages and ordered setup commands use the shared initialization lifecycle; - requested network applies after setup. Initial inline and tenant-owned file_id - files freeze encrypted bytes before provisioning, then install through the - common Core lifecycle before native execution or live Files access. Referenced - files/env/packages/setup overrides are rejected pending semantic verification. - Tenant-owned environment_template_id references inherit omitted network and - allow only narrowing overrides. Referenced network:null is explicitly unsupported - pending semantic verification. Core freezes effective configuration; template - updates/deletion do not alter Session snapshots or same-intent creation retries. - Inline or tenant-owned skill_reference Skills share initialization. Templates - preserve default/latest/explicit selectors; Session creation freezes concrete - metadata and encrypted content atomically. Skill-list omission inherits and - a supplied list replaces; null overrides and null version selectors remain - unqualified and reject. Source deletion/default updates cannot change committed - Session Skill contents. Deferred function discovery uses type-only tool_search - and per-function defer_loading in the qualified single-agent Claude environment:none - function profile, including qualified inline image messages and text results. - Explicit web_search mode disabled and programmatic_tool_calling enabled false - use frozen common Runtime controls. Enabled forms remain unqualified. Omitted - programmatic configuration preserves native behavior, a documented difference - from the official default-on behavior. Other combinations remain unqualified; - see the operation coverage. + to retrieve the Session. Claude SDK on none and Core-managed Docker openai_hosted + supports qualified object-root json_schema output with medium verbosity, single-Agent + execution and ordinary functions. Hosted execution reuses native workspace + tools and Files/Artifacts; Skills, Plugins, capability directories, HTTP MCP, + Subagent and tool_search combinations remain unqualified, including inherited + template contents. Other non-text initial input remains unsupported. Basic + Codex and Claude SDK openai_hosted creation requires an explicitly configured + managed provider. The Claude workspace profile supports non-deferred function + tools with text or successful inline PNG/JPEG results alongside native workspace + tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; + initial provisioning has no caller connection action. Network defaults to + enabled; disabled and restricted exact ASCII hostnames are supported. Restricted + policy requires 1–100 allowed domains. Unsupported hostname forms and startup + installations are rejected. Confidential env, system/npm/Python packages and + ordered setup commands use the shared initialization lifecycle; requested + network applies after setup. Initial inline and tenant-owned file_id files + freeze encrypted bytes before provisioning, then install through the common + Core lifecycle before native execution or live Files access. Referenced files/env/packages/setup + overrides are rejected pending semantic verification. Tenant-owned environment_template_id + references inherit omitted network and allow only narrowing overrides. Referenced + network:null is explicitly unsupported pending semantic verification. Core + freezes effective configuration; template updates/deletion do not alter Session + snapshots or same-intent creation retries. Inline or tenant-owned skill_reference + Skills share initialization. Templates preserve default/latest/explicit selectors; + Session creation freezes concrete metadata and encrypted content atomically. + Skill-list omission inherits and a supplied list replaces; null overrides + and null version selectors remain unqualified and reject. Source deletion/default + updates cannot change committed Session Skill contents. Deferred function + discovery uses type-only tool_search and per-function defer_loading in the + qualified single-agent Claude environment:none function profile, including + qualified inline image messages and text results. Explicit web_search mode + disabled and programmatic_tool_calling enabled false use frozen common Runtime + controls. Enabled forms remain unqualified. Omitted programmatic configuration + preserves native behavior, a documented difference from the official default-on + behavior. Other combinations remain unqualified; see the operation coverage. parameters: - description: agents=v1 in: header diff --git a/contracts/agents-api/structured-output.md b/contracts/agents-api/structured-output.md index 5ec41f8d8..74572f7ad 100644 --- a/contracts/agents-api/structured-output.md +++ b/contracts/agents-api/structured-output.md @@ -7,10 +7,13 @@ the existing Agent/Session configuration resolution and immutable snapshot. ## Qualified execution profile -Claude SDK supports object-root schemas with `environment:none`, medium verbosity, -`multi_agent.enabled=false` and optional ordinary function tools returning text. -The native SDK remains responsible for its model/tool loop and schema validation. -Codex and MiniMax structured-output execution, workspace/HTTP MCP/Subagent +Claude SDK supports object-root schemas with `environment:none` or Core-managed +Docker `openai_hosted`, medium verbosity, `multi_agent.enabled=false` and optional +ordinary function tools returning text. Hosted execution uses the existing native +workspace tools, preparation, Files and Artifacts. The native SDK remains +responsible for its model/tool loop and schema validation. Codex and MiniMax +structured-output execution, self-hosted/E2B execution, Skills/Plugins/capability +directories (including inherited template contents), HTTP MCP, Subagent/tool-search combinations and other root types are unqualified and explicitly rejected. These are implementation gaps, not a redefinition of the official protocol. @@ -38,6 +41,12 @@ a completed `final_answer` Message with the native tool-use ID and unchanged and cancelled candidates cannot become a completed structured answer. No private history read, output repair, schema coercion or prompt wrapper supplies the result. +Workspace preparation additionally verifies the installed bridge's +`workspace_structured_output` feature and complete local Runtime contract. Native +inventory includes `StructuredOutput` only when output configuration requests it; +root tool identity, abort checks, filesystem and credential protections remain +unchanged. A bundle feature alone does not qualify another public combination. + Recovery uses the existing Session/Turn/Items queries and native continuation. SSE is still live-only. Frozen schemas apply to both initial and resumed execution; ordinary text configuration retains its prior behavior. @@ -52,6 +61,16 @@ covers a function-only random value, unchanged saved configuration, native resul application receipts, ordered terminal SSE, persisted final JSON, daemon restart and same-history continuation, cancellation, text override and tenant isolation. +`services/agents-api/tests/official_hosted_structured_native.py` extends public +acceptance to an independently deployed Core, dedicated PostgreSQL and Docker +Runtime using the real Kimi API. It covers initial saved configuration and inline +prepared configuration, function-only random values, active input receipts, +native file writes consistent with final JSON, Files/Artifact reads, unchanged +terminal SSE, result retries/conflicts, cold Core/Runtime continuation, pending +cancellation without a fabricated final, ordinary text and tenant/Session isolation. +The accepted image retains the pinned SDK 0.3.269 and Claude Code 2.1.269. This +qualifies that native/provider combination, not every model or schema dialect. + Focused tests cover native failure/retry projection, exact result bytes, schema numeric admission, configuration transport and independent service qualification for another harness. Running only these tests or importing the SDK is not a claim diff --git a/packages/claude-sdk-adapter/src/adapter.ts b/packages/claude-sdk-adapter/src/adapter.ts index 0af8d9884..5f99a4e16 100644 --- a/packages/claude-sdk-adapter/src/adapter.ts +++ b/packages/claude-sdk-adapter/src/adapter.ts @@ -41,7 +41,7 @@ export async function execute(request: Start | Prepare, emit: (event: Event) => const declarations = request.workspace?.mcp ?? request.mcp_http_servers; const profile = declarations === undefined ? undefined : new MCPProfile(declarations, names); const subagents = request.subagents ? new Subagents(request.cwd, request.subagents.max_concurrent, request.resume) : undefined; - const workspace = request.workspace === undefined ? undefined : new WorkspaceProfile(request.cwd, request.workspace, names, profile, subagents); + const workspace = request.workspace === undefined ? undefined : new WorkspaceProfile(request.cwd, request.workspace, names, profile, subagents, !!request.output_format); const commands = workspace ? new CommandObserver() : undefined; if (request.type === "prepare" && !workspace) throw new Error("invalid_request"); if (workspace && "mcp_http_servers" in request) throw new Error("invalid_request"); diff --git a/packages/claude-sdk-adapter/src/request.ts b/packages/claude-sdk-adapter/src/request.ts index f8105a391..dbf353008 100644 --- a/packages/claude-sdk-adapter/src/request.ts +++ b/packages/claude-sdk-adapter/src/request.ts @@ -58,12 +58,13 @@ export function parseRequest(line: string): Start | Prepare { const format = request.output_format as Start["output_format"]; if (!format || format.type !== "json_schema" || Object.keys(format).some(key => !["type", "schema"].includes(key)) || !format.schema || format.schema.type !== "object" || !request.observe_messages || request.subagents || - request.workspace || request.mcp_http_servers !== undefined) throw new Error("invalid_request"); + request.mcp_http_servers !== undefined) throw new Error("invalid_request"); } if (request.type === "start") requestInput(request.input); parseHTTPServers(request.mcp_http_servers); const workspace = parseWorkspace(request.workspace, request.cwd); if (request.subagents && workspace?.mcp?.length) throw new Error("invalid_request"); + if (request.output_format && (workspace?.mcp?.length || workspace?.skills?.length)) throw new Error("invalid_request"); if (request.require_history && !workspace) throw new Error("invalid_request"); if ((workspace && "mcp_http_servers" in request) || (request.type === "prepare" && !workspace)) throw new Error("invalid_request"); diff --git a/packages/claude-sdk-adapter/src/runtime_check.ts b/packages/claude-sdk-adapter/src/runtime_check.ts index 5db1b0a51..e221c67fd 100644 --- a/packages/claude-sdk-adapter/src/runtime_check.ts +++ b/packages/claude-sdk-adapter/src/runtime_check.ts @@ -45,7 +45,7 @@ try { assert.equal(smoke.error, undefined, "bridge_unavailable"); assert.equal(smoke.status, 0, "bridge_unavailable"); assert.deepEqual(JSON.parse(smoke.stdout), { type: "error", code: "invalid_request" }); - process.stdout.write(JSON.stringify({ type: "runtime_ready", protocol: 2, features: [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions"] : []), "message_images", "function_result_images", "tool_search", "structured_output", "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"], node: process.versions.node, sdk: sdk.version, mcp: mcp.version, native: nativeVersion }) + "\n"); + process.stdout.write(JSON.stringify({ type: "runtime_ready", protocol: 2, features: [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions", "workspace_structured_output"] : []), "message_images", "function_result_images", "tool_search", "structured_output", "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"], node: process.versions.node, sdk: sdk.version, mcp: mcp.version, native: nativeVersion }) + "\n"); } catch { // Native diagnostics can include operator environment; never forward them. process.stdout.write(JSON.stringify({ type: "runtime_unavailable" }) + "\n"); diff --git a/packages/claude-sdk-adapter/src/workspace.ts b/packages/claude-sdk-adapter/src/workspace.ts index 8e5708379..c4a6772ef 100644 --- a/packages/claude-sdk-adapter/src/workspace.ts +++ b/packages/claude-sdk-adapter/src/workspace.ts @@ -82,7 +82,7 @@ export class WorkspaceProfile { readonly options: Options; private readonly skillNames: readonly string[]; - constructor(private readonly cwd: string, private readonly config: Workspace, private readonly functions: readonly string[] = [], private readonly mcp?: MCPProfile, private readonly subagents?: Subagents) { + constructor(private readonly cwd: string, private readonly config: Workspace, private readonly functions: readonly string[] = [], private readonly mcp?: MCPProfile, private readonly subagents?: Subagents, private readonly structuredOutput = false) { config = parseWorkspace(config, cwd)!; // SDK history lookup reads the bridge environment, independently of query.env. if (process.env.HOME !== config.home || process.env.CLAUDE_CONFIG_DIR !== config.state || @@ -144,7 +144,7 @@ export class WorkspaceProfile { this.mcp.verify(tools, servers as Parameters[1], sessionID, [...nativeTools, ...(this.skillNames.length ? ["Skill"] : []), ...(this.subagents ? ["Task", "SendMessage"] : [])]); return; } - const expected = [...nativeTools, ...this.functions, ...(this.skillNames.length ? ["Skill"] : []), ...(this.subagents ? ["Task", "SendMessage"] : [])]; + const expected = [...nativeTools, ...(this.structuredOutput ? ["StructuredOutput"] : []), ...this.functions, ...(this.skillNames.length ? ["Skill"] : []), ...(this.subagents ? ["Task", "SendMessage"] : [])]; if (servers.length !== (this.functions.length ? 1 : 0) || servers.some(server => server.name !== "functions" || server.status !== "connected") || tools.length !== expected.length || new Set(tools).size !== tools.length || @@ -186,6 +186,7 @@ export class WorkspaceProfile { private permits(name: string, value: unknown): boolean { if (!value || typeof value !== "object" || Array.isArray(value)) return false; const input = value as Record; + if (this.structuredOutput && name === "StructuredOutput") return true; if (this.functions.includes(name) || this.mcp?.permits(name)) return true; if (name === "Skill") return typeof input.skill === "string" && this.skillNames.includes(input.skill); if (name === "Bash") return typeof input.command === "string" && !!input.command.trim() && diff --git a/packages/claude-sdk-adapter/tests/mcp_workspace.test.mjs b/packages/claude-sdk-adapter/tests/mcp_workspace.test.mjs index 03eaac824..13fce9f8e 100644 --- a/packages/claude-sdk-adapter/tests/mcp_workspace.test.mjs +++ b/packages/claude-sdk-adapter/tests/mcp_workspace.test.mjs @@ -31,6 +31,8 @@ function fixture(t, declarations = [stdio]) { test("installed MCP private projection cannot launch arbitrary unsandboxed commands", t => { const { request } = fixture(t); assert.deepEqual(parseStart(JSON.stringify(request)), request); + assert.throws(() => parseStart(JSON.stringify({ ...request, observe_messages: true, + output_format: { type: "json_schema", schema: { type: "object" } } })), /invalid_request/); assert.equal(immediateInput(request), undefined); assert.deepEqual(parseEnvironmentMCP([stdio]), [stdio]); for (const value of [[stdio, stdio], [{ ...stdio, command: "/bin/sh" }], [{ ...stdio, env: { TOKEN: "secret" } }], diff --git a/packages/claude-sdk-adapter/tests/workspace.test.mjs b/packages/claude-sdk-adapter/tests/workspace.test.mjs index 12e68c4cd..31f0576fb 100644 --- a/packages/claude-sdk-adapter/tests/workspace.test.mjs +++ b/packages/claude-sdk-adapter/tests/workspace.test.mjs @@ -128,6 +128,34 @@ test("workspace native options keep the strict sandbox and exact native inventor assert.throws(() => profile.verify(options.tools, [{ name: "untrusted", status: "connected" }]), /unexpected native workspace inventory/); }); +test("structured workspace admits only its configured native terminal tool", async t => { + const { dirs, config, request } = fixture(t); + const output_format = { type: "json_schema", schema: { type: "object" } }; + const configured = { ...request, observe_messages: true, output_format }; + assert.deepEqual(parseStart(JSON.stringify(configured)), configured); + const ordinary = new WorkspaceProfile(dirs.workspace, config); + const structured = new WorkspaceProfile(dirs.workspace, config, [], undefined, undefined, true); + const { canUseTool: ordinaryPermission, hooks: ordinaryHooks, ...ordinaryOptions } = ordinary.options; + const { canUseTool: structuredPermission, hooks: structuredHooks, ...structuredOptions } = structured.options; + assert.deepEqual(structuredOptions, ordinaryOptions); + const inventory = ["Bash", "Read", "Edit", "StructuredOutput"]; + structured.verify(inventory, []); + assert.throws(() => ordinary.verify(inventory, [])); + assert.throws(() => structured.verify(["Bash", "Read", "Edit"], [])); + assert.throws(() => structured.verify([...inventory, "Write"], [])); + const context = { signal: new AbortController().signal, toolUseID: "terminal", requestId: "request" }; + assert.equal((await ordinary.canUseTool("StructuredOutput", {}, context)).behavior, "deny"); + assert.equal((await structured.canUseTool("StructuredOutput", {}, context)).behavior, "allow"); + for (const extra of [{ agentID: "child" }, { signal: AbortSignal.abort() }]) { + assert.equal((await structured.canUseTool("StructuredOutput", {}, { ...context, ...extra })).behavior, "deny"); + } + const hook = { hook_event_name: "PreToolUse", tool_name: "StructuredOutput", tool_input: {}, tool_use_id: "terminal" }; + assert.deepEqual(await structured.beforeTool(hook, "terminal", context), {}); + for (const [input, id, ctx] of [[{ ...hook, agent_id: "child" }, "terminal", context], [hook, "wrong", context], [hook, "terminal", { ...context, signal: AbortSignal.abort() }]]) { + assert.equal((await structured.beforeTool(input, id, ctx)).hookSpecificOutput.permissionDecision, "deny"); + } +}); + test("workspace permissions and pre-tool hook reject outside paths and unsafe Bash flags", async t => { const { dirs, config } = fixture(t); writeFileSync(join(dirs.workspace, "file.txt"), "fixture"); diff --git a/scripts/check-claude-sdk-runtime.mjs b/scripts/check-claude-sdk-runtime.mjs index 716c4c226..156f79983 100644 --- a/scripts/check-claude-sdk-runtime.mjs +++ b/scripts/check-claude-sdk-runtime.mjs @@ -23,7 +23,7 @@ assert.equal(probe.status, 0, "Exported runtime is unavailable"); const report = JSON.parse(probe.stdout); assert.equal(report.type, "runtime_ready"); assert.equal(report.protocol, 2); -assert.deepEqual(report.features, [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions"] : []), "message_images", "function_result_images", "tool_search", "structured_output", "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"]); +assert.deepEqual(report.features, [...(process.platform === "linux" ? ["workspace_directory", "local_runtime_v1", "workspace_functions", "workspace_structured_output"] : []), "message_images", "function_result_images", "tool_search", "structured_output", "subagent_resources", "mcp_http_tools", "mcp_http_bearer_auth", "mcp_http_required", "workspace_tools", "workspace_prepare", "workspace_read", "workspace_command_observations"]); assert.equal(report.sdk, source.dependencies["@anthropic-ai/claude-agent-sdk"]); assert.equal(report.mcp, source.dependencies["@modelcontextprotocol/sdk"]); console.log(`Verified exported SDK ${report.sdk}, MCP ${report.mcp}, ${report.native}`); diff --git a/services/agents-api/internal/api/handler.go b/services/agents-api/internal/api/handler.go index 24255c410..bcbb51fc9 100644 --- a/services/agents-api/internal/api/handler.go +++ b/services/agents-api/internal/api/handler.go @@ -124,7 +124,7 @@ func NewHandler(s ResourceStore, auth *Authenticator, engine string, options ... // createSession atomically reserves or admits initial text with the Session. // @Summary Create an execution Session -// @Description Supports inline configuration or a tenant-owned saved agent_id with per-Session field replacements. Execution supports model/instructions, text verbosity, non-deferred function tools, adapter-qualified multi_agent with persisted Subagent reads, implicit reasoning, service tier auto and environment type none, subject to the configured engine. Codex additionally supports HTTP MCP with explicit service origin, native allowed_tools and boolean required defaulting to false. Session vault_ids attach only project-owned Vaults; credential_id selects an attached static bearer credential for the exact HTTPS URL, while null/omission selects a unique match or remains anonymous. Ambiguous selection rejects creation. Frozen private selections never populate an omitted public credential_id; missing decryption configuration fails dispatch without anonymous fallback. Required initialization uses native startup before the first native Turn, including cold resume, and requires a separately advertised capability; exact hosted creation timing and error parity remain unverified. Other MCP origins and OAuth remain unsupported. The self_hosted profile requires Codex, an absolute workspace_directory and empty capability_directories, with optional non-deferred function tools and HTTP MCP using explicit service origin, optionally authenticated by the attached Vault rules. Remote MCP and remote Bearer authentication each require separately advertised combination support; old peers cannot receive unsupported work. Omitted/null capability_directories use the empty-list default; self_hosted requires configured execution plus executor registry. Claude SDK currently requires medium verbosity and object-root function schemas. It supports anonymous or attached static-bearer service-origin HTTP MCP on none with boolean required and separately advertised MCP/bearer/required runtime support. Required servers must be connected before the first native input is released; pending or failed startup rejects execution. The shared Vault selection and immutable binding rules apply; unsupported native labels/tool names reject before persistence. An attached Vault with no matching credential may remain anonymous; missing keys or failed credential lookup/decryption never fall back to anonymous execution. Omitted stream defaults to false; stream and agent_id cannot be null. Metadata may be null, but its values must be strings. Initial input accepts a string or ordered user-message array. Codex and Claude SDK on none and qualified openai_hosted also accept inline PNG/JPEG image content; other image combinations and remote URLs are unsupported. None initial input atomically starts a Turn; self_hosted initial input is reserved while returning its Environment connection target, with execution deferred to native readiness and Session failure on initial timeout. Omitted or null input creates an idle Session. With stream=true, returns live Session events starting at creation; disconnect does not cancel execution. New Sessions retain their authenticated creator; all creation retries require the same typed subject, including across key rotation. Saved-Agent retries and inline requests using Vault attachments or credential references retain caller intent independently of later resource changes; unrelated inline retries preserve resolved/default equivalences. Unknown historical creators reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without replay; retry with stream=false to retrieve the Session. Claude SDK environment:none supports qualified object-root json_schema output with medium verbosity, single-Agent execution and ordinary functions; other combinations remain unsupported. Other non-text initial input remains unsupported. Basic Codex and Claude SDK openai_hosted creation requires an explicitly configured managed provider. The Claude workspace profile supports non-deferred function tools with text or successful inline PNG/JPEG results alongside native workspace tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; initial provisioning has no caller connection action. Network defaults to enabled; disabled and restricted exact ASCII hostnames are supported. Restricted policy requires 1–100 allowed domains. Unsupported hostname forms and startup installations are rejected. Confidential env, system/npm/Python packages and ordered setup commands use the shared initialization lifecycle; requested network applies after setup. Initial inline and tenant-owned file_id files freeze encrypted bytes before provisioning, then install through the common Core lifecycle before native execution or live Files access. Referenced files/env/packages/setup overrides are rejected pending semantic verification. Tenant-owned environment_template_id references inherit omitted network and allow only narrowing overrides. Referenced network:null is explicitly unsupported pending semantic verification. Core freezes effective configuration; template updates/deletion do not alter Session snapshots or same-intent creation retries. Inline or tenant-owned skill_reference Skills share initialization. Templates preserve default/latest/explicit selectors; Session creation freezes concrete metadata and encrypted content atomically. Skill-list omission inherits and a supplied list replaces; null overrides and null version selectors remain unqualified and reject. Source deletion/default updates cannot change committed Session Skill contents. Deferred function discovery uses type-only tool_search and per-function defer_loading in the qualified single-agent Claude environment:none function profile, including qualified inline image messages and text results. Explicit web_search mode disabled and programmatic_tool_calling enabled false use frozen common Runtime controls. Enabled forms remain unqualified. Omitted programmatic configuration preserves native behavior, a documented difference from the official default-on behavior. Other combinations remain unqualified; see the operation coverage. +// @Description Supports inline configuration or a tenant-owned saved agent_id with per-Session field replacements. Execution supports model/instructions, text verbosity, non-deferred function tools, adapter-qualified multi_agent with persisted Subagent reads, implicit reasoning, service tier auto and environment type none, subject to the configured engine. Codex additionally supports HTTP MCP with explicit service origin, native allowed_tools and boolean required defaulting to false. Session vault_ids attach only project-owned Vaults; credential_id selects an attached static bearer credential for the exact HTTPS URL, while null/omission selects a unique match or remains anonymous. Ambiguous selection rejects creation. Frozen private selections never populate an omitted public credential_id; missing decryption configuration fails dispatch without anonymous fallback. Required initialization uses native startup before the first native Turn, including cold resume, and requires a separately advertised capability; exact hosted creation timing and error parity remain unverified. Other MCP origins and OAuth remain unsupported. The self_hosted profile requires Codex, an absolute workspace_directory and empty capability_directories, with optional non-deferred function tools and HTTP MCP using explicit service origin, optionally authenticated by the attached Vault rules. Remote MCP and remote Bearer authentication each require separately advertised combination support; old peers cannot receive unsupported work. Omitted/null capability_directories use the empty-list default; self_hosted requires configured execution plus executor registry. Claude SDK currently requires medium verbosity and object-root function schemas. It supports anonymous or attached static-bearer service-origin HTTP MCP on none with boolean required and separately advertised MCP/bearer/required runtime support. Required servers must be connected before the first native input is released; pending or failed startup rejects execution. The shared Vault selection and immutable binding rules apply; unsupported native labels/tool names reject before persistence. An attached Vault with no matching credential may remain anonymous; missing keys or failed credential lookup/decryption never fall back to anonymous execution. Omitted stream defaults to false; stream and agent_id cannot be null. Metadata may be null, but its values must be strings. Initial input accepts a string or ordered user-message array. Codex and Claude SDK on none and qualified openai_hosted also accept inline PNG/JPEG image content; other image combinations and remote URLs are unsupported. None initial input atomically starts a Turn; self_hosted initial input is reserved while returning its Environment connection target, with execution deferred to native readiness and Session failure on initial timeout. Omitted or null input creates an idle Session. With stream=true, returns live Session events starting at creation; disconnect does not cancel execution. New Sessions retain their authenticated creator; all creation retries require the same typed subject, including across key rotation. Saved-Agent retries and inline requests using Vault attachments or credential references retain caller intent independently of later resource changes; unrelated inline retries preserve resolved/default equivalences. Unknown historical creators reject retries; known creators without recorded intent retain resolved-snapshot retry rules. These conflict policies are local and not verified hosted parity. Creation retries observe future events without replay; retry with stream=false to retrieve the Session. Claude SDK on none and Core-managed Docker openai_hosted supports qualified object-root json_schema output with medium verbosity, single-Agent execution and ordinary functions. Hosted execution reuses native workspace tools and Files/Artifacts; Skills, Plugins, capability directories, HTTP MCP, Subagent and tool_search combinations remain unqualified, including inherited template contents. Other non-text initial input remains unsupported. Basic Codex and Claude SDK openai_hosted creation requires an explicitly configured managed provider. The Claude workspace profile supports non-deferred function tools with text or successful inline PNG/JPEG results alongside native workspace tools; HTTP MCP remains unsupported. Idle Sessions provision automatically; initial provisioning has no caller connection action. Network defaults to enabled; disabled and restricted exact ASCII hostnames are supported. Restricted policy requires 1–100 allowed domains. Unsupported hostname forms and startup installations are rejected. Confidential env, system/npm/Python packages and ordered setup commands use the shared initialization lifecycle; requested network applies after setup. Initial inline and tenant-owned file_id files freeze encrypted bytes before provisioning, then install through the common Core lifecycle before native execution or live Files access. Referenced files/env/packages/setup overrides are rejected pending semantic verification. Tenant-owned environment_template_id references inherit omitted network and allow only narrowing overrides. Referenced network:null is explicitly unsupported pending semantic verification. Core freezes effective configuration; template updates/deletion do not alter Session snapshots or same-intent creation retries. Inline or tenant-owned skill_reference Skills share initialization. Templates preserve default/latest/explicit selectors; Session creation freezes concrete metadata and encrypted content atomically. Skill-list omission inherits and a supplied list replaces; null overrides and null version selectors remain unqualified and reject. Source deletion/default updates cannot change committed Session Skill contents. Deferred function discovery uses type-only tool_search and per-function defer_loading in the qualified single-agent Claude environment:none function profile, including qualified inline image messages and text results. Explicit web_search mode disabled and programmatic_tool_calling enabled false use frozen common Runtime controls. Enabled forms remain unqualified. Omitted programmatic configuration preserves native behavior, a documented difference from the official default-on behavior. Other combinations remain unqualified; see the operation coverage. // @Tags Sessions // @Accept json // @Produce json,text/event-stream diff --git a/services/agents-api/internal/api/hosted_structured_test.go b/services/agents-api/internal/api/hosted_structured_test.go new file mode 100644 index 000000000..44bb115c7 --- /dev/null +++ b/services/agents-api/internal/api/hosted_structured_test.go @@ -0,0 +1,60 @@ +package api + +import ( + "encoding/json" + "testing" + + "github.com/MiniMax-AI-Dev/parsar/services/agents-api/internal/execution" +) + +func TestHostedStructuredConfigurationQualification(t *testing.T) { + format := `"text":{"format":{"type":"json_schema","schema":{"type":"object"}}}` + for _, fields := range []string{ + `"skills":[` + string(skillInput(t, "Use native tools.")) + `]`, + `"plugins":[` + string(pluginInput(t)) + `]`, + `"capability_directories":["/workspace/capabilities"]`, + } { + template, err := decodeTemplateInput(json.RawMessage(`{` + fields + `}`)) + if err != nil { + t.Fatal(err) + } + lookup := &templateLookupStore{network: "enabled", skills: template.Initialization.Skills, + plugins: template.Initialization.Plugins, directories: template.Initialization.CapabilityDirectories} + h := Handler{store: lookup} + for _, environment := range []string{ + `{"type":"openai_hosted",` + fields + `}`, + `{"type":"openai_hosted","environment_template_id":"saved"}`, + `{"type":"openai_hosted","environment_template_id":"saved","skills":[],"plugins":[],"capability_directories":[]}`, + } { + var decoded decodedSessionRequest + if err := json.Unmarshal([]byte(`{"agent":{"model":"model",`+format+`},"environment":`+environment+`}`), &decoded); err != nil { + t.Fatal(err) + } + input, err := decoded.validated() + if err != nil { + t.Fatal(err) + } + if err := h.resolveTemplateEnvironment(t.Context(), "tenant", &input); err != nil { + t.Fatal(err) + } + configuration, err := resolve(input, "tenant", "key", nil) + if err != nil { + t.Fatal(err) + } + wantAccepted := len(input.Environment.Skills)+len(input.Environment.Plugins)+len(input.Environment.CapabilityDirectories) == 0 + if err := (execution.Policy{}).ValidateSessionConfiguration("claude_sdk", configuration); (err == nil) != wantAccepted { + t.Fatal("incorrect inline/template qualification", environment, err) + } + if !wantAccepted { + input.Agent.Text.Format = json.RawMessage(`{"type":"text"}`) + configuration, err = resolve(input, "tenant", "key", nil) + if err != nil { + t.Fatal(err) + } + if err := (execution.Policy{}).ValidateSessionConfiguration("claude_sdk", configuration); err != nil { + t.Fatal("ordinary workspace profile changed", err) + } + } + } + } +} diff --git a/services/agents-api/internal/engine/claude.go b/services/agents-api/internal/engine/claude.go index 962c95a5b..fa94eaf5e 100644 --- a/services/agents-api/internal/engine/claude.go +++ b/services/agents-api/internal/engine/claude.go @@ -46,8 +46,11 @@ func validateClaudeConfiguration(agent v1.Agent, environment *v1.Environment, ha if err := proto.ValidateBinary64Schema(agent.Text.Format.Schema); err != nil { return err } - if environment.Type != "none" || agent.MultiAgent.Enabled { - return errors.New("Structured output currently requires a single-agent environment:none profile.") + if (environment.Type != "none" && environment.Type != "openai_hosted") || agent.MultiAgent.Enabled { + return errors.New("Structured output requires a qualified single-agent placement.") + } + if len(environment.Skills) != 0 || len(environment.Plugins) != 0 || len(environment.CapabilityDirectories) != 0 { + return errors.New("Structured output with environment Skills or Plugins is not qualified.") } for _, raw := range agent.Tools { var tool struct { diff --git a/services/agents-api/tests/official_hosted_structured_native.py b/services/agents-api/tests/official_hosted_structured_native.py new file mode 100644 index 000000000..31d4e77c8 --- /dev/null +++ b/services/agents-api/tests/official_hosted_structured_native.py @@ -0,0 +1,208 @@ +"""Structured final answers after real native Docker workspace/tool execution.""" +import importlib.metadata +import json +from pathlib import Path +import time +import uuid + + +def verify_hosted_structured(client, foreign, http, model, kind, restart, evidence): + pin = json.loads((Path(__file__).resolve().parents[3] / "contracts/agents-api/upstream.json").read_text()) + dist = importlib.metadata.distribution("openai") + assert dist.version == pin["sdk_version"] + assert json.loads(dist.read_text("direct_url.json"))["vcs_info"]["commit_id"] == pin["commit"] + sessions = client.beta.agents.sessions + root = str(client.base_url).rstrip("/") + "/agents" + headers = {"Authorization": "Bearer " + client.api_key, "OpenAI-Beta": "agents=v1"} + other_headers = {**headers, "Authorization": "Bearer " + foreign.api_key} + proof = {"engine": kind, "checks": [], "runs": [], "calls": []} + owned = [] + saved = None + schema = {"type": "object", "properties": {name: {"type": "string"} for name in ("memory", "path", "marker")}, + "required": ["memory", "path", "marker"], "additionalProperties": False} + output_format = {"type": "json_schema", "schema": schema} + tool = {"type": "function", "name": "remember", "description": "Return the memory value for this task.", + "parameters": {"type": "object", "properties": {}, "additionalProperties": False}} + + def save(): + Path(evidence).write_text(json.dumps(proof, indent=2)) + + def check(name): + proof["checks"].append(name) + save() + print(name, flush=True) + + def until(fn, timeout=240): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + value = fn() + if value: + return value + time.sleep(0.2) + raise AssertionError("Public workflow did not reach expected state") + + def message(text): + return {"type": "agent.session.input.message", "input": [{"role": "user", "content": [{"type": "input_text", "text": text}]}]} + + def prompt(path, marker, recall=False): + source = "Recall the last memory from conversation history without calling remember or reading files." if recall else "Call remember exactly once to obtain the memory." + return source + " Use native workspace tools to write that exact memory with no newline to " + path + ". Create the parent directory if needed. Return memory, that path, and marker '" + marker + "' using the requested output format." + + def items(sid): + response = http.get(root + "/sessions/" + sid + "/items", headers=headers, params={"order": "asc", "limit": 100}) + assert response.status_code == 200 and not response.json()["has_more"] + raw = response.json()["data"] + assert raw == [i.to_dict() for i in sessions.items.list(sid, order="asc", limit=100).data] + return raw + + def respond(sid, action, memory): + assert action.name == "remember" + result = {"type": "agent.session.input.tool_result", "turn_id": action.turn_id, "call_id": action.call_id, "success": True, "output": memory} + key = "result-" + action.call_id + assert http.post(root + "/sessions/" + sid + "/events", headers=other_headers, json={"events": [result]}).status_code == 404 + sessions.events.create(sid, events=[result], idempotency_key=key) + sessions.events.create(sid, events=[result], idempotency_key=key) + response = http.post(root + "/sessions/" + sid + "/events", headers={**headers, "Idempotency-Key": key}, json={"events": [{**result, "output": "conflict"}]}) + assert response.status_code == 409 + proof["calls"].append(result) + + def terminal(sid, count, status="completed"): + def finished(): + turns = sessions.turns.list(sid, order="asc", limit=100).data + if len(turns) < count or turns[-1].status not in {"completed", "failed", "cancelled"}: + return False + assert len(turns) == count and turns[-1].status == status, [t.to_dict() for t in turns] + return turns[-1] + return until(finished) + + def final(sid, eid, turn, expected, events=None): + stored = items(sid) + answers = [i for i in stored if i["type"] == "message" and i.get("role") == "assistant" and i.get("phase") == "final_answer" and i["turn_id"] == turn.id] + assert len(answers) == 1, answers + answer = answers[0] + raw = answer["content"][0]["text"] + assert json.loads(raw) == expected, raw + if events: + added = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.item.added" and e["item"]["id"] == answer["id"]) + done = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.item.done" and e["item"]["id"] == answer["id"]) + complete = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.completed") + assert added < done < complete + assert next(e["text"] for e in events if e["type"] == "agent.session.turn.output_text.done" and e["item_id"] == answer["id"]) == raw + artifacts = [a for a in sessions.artifacts.list(sid, limit=100) if a.path == expected["path"] and a.turn_id == turn.id] + assert len(artifacts) == 1 + with sessions.artifacts.with_streaming_response.content(artifacts[0].id, session_id=sid) as response: + assert response.read() == expected["memory"].encode() + assert any(f.path == expected["path"] for f in client.beta.agents.environments.files.list(eid, path="/workspace/outputs")) + assert http.get(root + "/sessions/" + sid + "/artifacts/" + artifacts[0].id + "/content", headers=other_headers).status_code == 404 + proof.setdefault("answers", []).append(answer) + + def run(sid, text, handler=None): + events = [] + handled = False + try: + with sessions.events.stream(sid, timeout=300) as stream: + sessions.events.create(sid, events=[message(text)]) + for event in stream: + events.append(event.to_dict()) + if event.type == "agent.session.requires_action": + assert handler is not None and not handled and len(event.session.required_actions) == 1 + handled = True + handler(event.session.required_actions[0]) + assert event.type not in {"agent.session.failed", "agent.session.turn.failed"}, event.to_dict() + if event.type == "agent.session.idle" and any(e["type"] == "agent.session.turn.created" for e in events): + break + else: + raise AssertionError("SSE ended without idle") + finally: + proof["runs"].append(events) + save() + assert handled == (handler is not None) + types = [e["type"] for e in events] + terminals = [t for t in types if t in {"agent.session.turn.completed", "agent.session.turn.cancelled"}] + assert len(terminals) == 1 and types[-1] == "agent.session.idle" + assert types.index("agent.session.turn.created") < types.index(terminals[0]) < len(types) - 1 + assert len({e["event_id"] for e in events}) == len(events) + return events + + try: + saved = client.beta.agents.create(model=model, text={"format": output_format}, tools=[tool]) + first_memory = uuid.uuid4().hex + path = "/workspace/outputs/initial.txt" + session = sessions.create(agent_id=saved.id, environment={"type": "openai_hosted"}, input=prompt(path, "initial")) + sid, eid = session.id, session.environment.id + owned.append(sid) + proof.update(session=sid, environment=eid, agent=saved.id, format=output_format) + assert session.agent.text.format.to_dict() == output_format + def pending(): + current = sessions.retrieve(sid) + assert current.status != "failed", current.to_dict() + return current.required_actions + action = until(pending)[0] + respond(sid, action, first_memory) + first = terminal(sid, 1) + final(sid, eid, first, {"memory": first_memory, "path": path, "marker": "initial"}) + check("saved_schema_initial_function_native_file_and_final_json") + + memory = uuid.uuid4().hex + path = "/workspace/outputs/active.txt" + def active(action): + incoming = [message("Use marker 'active' for the final result, replacing the earlier marker. Keep the requested file path and memory task.")] + sessions.events.create(sid, events=incoming, idempotency_key="active-marker") + sessions.events.create(sid, events=incoming, idempotency_key="active-marker") + respond(sid, action, memory) + events = run(sid, prompt(path, "obsolete"), active) + second = terminal(sid, 2) + final(sid, eid, second, {"memory": memory, "path": path, "marker": "active"}, events) + outputs = [i for i in items(sid) if i["type"] == "function_call_output"] + assert len(outputs) == 2 and [i["output"] for i in outputs] == [first_memory, memory] + check("prepared_schema_active_receipt_function_retry_and_exact_sse") + + for suffix in ("", "/items", "/turns", "/artifacts"): + assert http.get(root + "/sessions/" + sid + suffix, headers=other_headers).status_code == 404 + assert http.get(root + "/environments/" + eid + "/files", headers=other_headers).status_code == 404 + assert http.post(root + "/sessions", headers=other_headers, json={"agent_id": saved.id, "environment": {"type": "openai_hosted"}}).status_code == 404 + before = items(sid) + restart(sid, eid) + assert items(sid) == before and sessions.retrieve(sid).agent.text.format.to_dict() == output_format + path = "/workspace/outputs/resumed.txt" + events = run(sid, prompt(path, "resumed", recall=True)) + third = terminal(sid, 3) + final(sid, eid, third, {"memory": memory, "path": path, "marker": "resumed"}, events) + assert len([i for i in items(sid) if i["type"] == "function_call"]) == 2 + check("cold_core_runtime_continuation_without_replay_and_tenant_isolation") + + def cancel(action): + for _ in range(2): + sessions.events.create(sid, events=[{"type": "agent.session.input.cancel"}], idempotency_key="cancel-pending") + run(sid, prompt("/workspace/outputs/cancelled.txt", "cancelled"), cancel) + cancelled = terminal(sid, 4, "cancelled") + assert not any(i.get("turn_id") == cancelled.id and i["type"] == "message" and i.get("phase") == "final_answer" for i in items(sid)) + assert not sessions.retrieve(sid).required_actions + check("pending_cancellation_has_no_structured_final") + + plain = sessions.create(agent_id=saved.id, agent={"text": {"format": {"type": "text"}}}, environment={"type": "openai_hosted"}) + owned.append(plain.id) + run(plain.id, "Do not call remember. Use native tools to write exactly PLAIN_OK to /workspace/plain.txt with no newline, then reply only PLAIN_OK.") + assert any(i["type"] == "message" and i.get("role") == "assistant" and "PLAIN_OK" in i["content"][0].get("text", "") for i in items(plain.id)) + foreign_result = proof["calls"][0] + assert http.post(root + "/sessions/" + plain.id + "/events", headers=headers, json={"events": [foreign_result]}).status_code in {400, 404, 409} + artifact = list(sessions.artifacts.list(sid, limit=100))[0] + assert http.get(root + "/sessions/" + plain.id + "/artifacts/" + artifact.id + "/content", headers=headers).status_code == 404 + check("text_override_and_same_tenant_session_isolation") + + inline = sessions.create(agent={"model": model, "text": {"format": output_format}}, environment={"type": "openai_hosted"}) + owned.append(inline.id) + inline_memory = uuid.uuid4().hex + path = "/workspace/outputs/inline.txt" + events = run(inline.id, "Use native tools to create the parent directory and write exactly " + inline_memory + " with no newline to " + path + ". Return that memory, path, and marker 'inline' using the requested output format.") + final(inline.id, inline.environment.id, terminal(inline.id, 1), {"memory": inline_memory, "path": path, "marker": "inline"}, events) + assert inline.agent.text.format.to_dict() == output_format + check("inline_schema_prepared_execution_native_file_and_final_json") + proof.update(passed=True, items=items(sid), turns=[t.to_dict() for t in sessions.turns.list(sid, order="asc", limit=100).data]) + return proof["checks"] + finally: + save() + for sid in reversed(owned): + sessions.delete(sid) + if saved: + client.beta.agents.delete(saved.id)