From c68d3aa22af2df84070cf776613b954939a408a2 Mon Sep 17 00:00:00 2001 From: Laimis Date: Wed, 23 Sep 2026 15:29:28 +0300 Subject: [PATCH 1/6] Align execution patterns and strengthen behavioral evaluations --- README.md | 7 + docs/evals/README.md | 47 +- .../pivot-restart-on-stagnation/RECOVERY.md | 30 + .../tests/recovery-contracts.sh | 32 + .../tests/stagnation-contracts.sh | 49 ++ docs/evals/framework-instruction-cases.json | 62 +- .../contracts/phase-gates.yaml | 4 +- skills/assistant-workflow/evals/cases.json | 72 +- .../references/build-worker-protocol.md | 37 +- .../assistant-workflow/references/phases.md | 20 +- .../references/phases/build.md | 18 +- .../references/phases/decompose.md | 2 +- .../references/sub-task-brief-template.md | 4 +- .../references/task-journal-template.md | 6 +- .../p0-p4/codex-behavioral-eval-contracts.sh | 691 +++++++++++++++++- .../feature-preparation-evidence-contracts.sh | 2 +- tests/p0-p4/task-packet-contracts.sh | 236 +++++- tools/evals/lib/skill-eval-fixtures.sh | 32 + tools/evals/lib/skill-eval-render.sh | 12 +- tools/evals/run-codex-framework-evals.sh | 453 +++++++++++- 20 files changed, 1718 insertions(+), 98 deletions(-) create mode 100644 docs/evals/fixtures/pivot-restart-on-stagnation/RECOVERY.md create mode 100755 docs/evals/fixtures/pivot-restart-on-stagnation/tests/recovery-contracts.sh create mode 100755 docs/evals/fixtures/pivot-restart-on-stagnation/tests/stagnation-contracts.sh diff --git a/README.md b/README.md index f797b70..9425856 100644 --- a/README.md +++ b/README.md @@ -118,6 +118,13 @@ Only tracked `assistant-*` directories are first-class release skills. ### assistant-workflow Core development pipeline: idea-to-action decomposition, discover, proportional planning, build and verification, independent review, bounded repair, and evidence-backed documentation. +| Pattern | Use | Combine with | +|---|---|---| +| Linear | One focused change with dependent steps; scale planning and review to risk. | Goal-driven checks and fresh review; a low-risk local change may use the direct lightweight path. | +| Skill-driven | A recurring task has a matching installed skill and contract. | Any other pattern; the skill supplies the method. | +| Parallel | Independent work can progress together. Source-changing packets remain sequential in a shared or unknown workspace; overlap requires runtime-proven isolated workspaces. | Goal-driven integration: dependents wait for VERIFIED prerequisites, then cross-slice and full-scope validation precede fresh review. | +| Goal-driven | Acceptance checks must remain the completion condition through repair. | Linear, skill-driven, or parallel execution; bounded repair escalates instead of claiming false completion. | + For dependency-shaped uncertainty, the workflow defaults to `uncertainty_shape=bounded`: size alone does not activate progressive Discover. It enters that substate only when a predecessor decision must unlock an diff --git a/docs/evals/README.md b/docs/evals/README.md index 835335e..644b94d 100644 --- a/docs/evals/README.md +++ b/docs/evals/README.md @@ -28,6 +28,41 @@ under common operating conditions: - pivot/restart decisions for stagnation and Code Writer blockers - terminal max 10 review/QA round behavior +## Observed execution-pattern evidence + +The pattern cases distinguish a deterministic policy fixture from evidence the +Codex adapter actually observed in its JSONL event stream. + +- `small-fix-stays-lightweight` requires its exact target-file discovery probe + to be the first workspace command or file action; the matching command-start + event may precede its successful completion. This rule and its no-web/MCP + check apply only to the disposable local typo fixture, not delegated work. + A grading artifact alone cannot pass the case. +- `pivot-restart-on-stagnation-or-code-writer-blocker` seeds a trusted failing + check, fixture-owned failure/recovery receipts, recovery action, and fresh + check. Its recovery artifact retains `terminal_completed=false`: a fresh-check + pass validates the recovery protocol, not repair of the legacy bug or workflow + completion. It admits only the three trusted fixture scripts and optional + read-only `cat RECOVERY.md`; any other started or completed command fails the + bounded case. A present command start must have a nonempty id and one later + completion with the same admitted command kind; duplicate, unmatched, or + mismatched starts, and duplicate nonempty completion ids fail. The failure + completes before recovery starts, and the recovery completes before the fresh + check starts; completion-only streams use the corresponding completion index. + Once that unique fresh check completes, any later started or completed command + or file-change event also fails. This boundary applies only to the disposable + recovery fixture. +- `isolated-parallel-a-b-then-c-with-integration` records the required A/B/C + dependency and integration policy, but current Codex CLI JSONL does not expose + authoritative worker, workspace, isolation, or overlap telemetry. The adapter + therefore emits `adapter_unavailable` with `unknown_event_shape`, no metrics, + and an excluded incomplete pair. Agent-message narrative never upgrades that + result to observed native parallel execution. + +The workflow policy fixtures still test shared-workspace sequencing, +runtime-proven isolation, VERIFIED prerequisites, integration validation, and +fresh review deterministically. They do not supply native execution telemetry. + ## Generated workflow references `assistant-workflow` phase and plan views are generated from their authoritative @@ -713,10 +748,14 @@ tools/evals/run-skill-evals.sh --emit-prompts /tmp/skill-eval-prompts tools/evals/run-skill-evals.sh --emit-prompts /tmp/clarify-eval-prompts --skill assistant-clarify ``` -Prompt packets are written under `//.md` and include the -setup context, prompt, expected behavior, pass criteria, fail signals, optional -seeded defects / measurable assertions, machine expectations, and an optional -Structured JSON Assertions section when the case declares one. +Prompt packets are written under `//.md`. By default, +and when a case declares `prompt_packet_mode: annotated`, they include the setup +context, prompt, expected behavior, pass criteria, fail signals, optional seeded +defects / measurable assertions, machine expectations, and an optional Structured +JSON Assertions section when the case declares one. A case may instead declare +`prompt_packet_mode: task_only`; its target packet contains only a neutral header, +skill identity and path, setup context, and prompt. The local grader always keeps +the complete fixture, including its grading-only expectations. Run each prompt packet with the target assistant and save captured responses as `//.txt` or `//.md`. diff --git a/docs/evals/fixtures/pivot-restart-on-stagnation/RECOVERY.md b/docs/evals/fixtures/pivot-restart-on-stagnation/RECOVERY.md new file mode 100644 index 0000000..cab3f25 --- /dev/null +++ b/docs/evals/fixtures/pivot-restart-on-stagnation/RECOVERY.md @@ -0,0 +1,30 @@ +# Stagnation recovery fixture + +The first trusted check writes a fixture-owned failure receipt, emits +`STAGNATION_TRUSTED_FAILURE`, and fails. The recovery action validates that +receipt and this bounded decision, then writes a pending recovery artifact and +emits `RECOVERY_APPLIED`. Only then may the fresh check validate the receipts, +mark the recovery artifact fresh-check passed, and emit +`STAGNATION_FRESH_CHECK_PASS`. + +For this disposable fixture only, admitted workspace commands are: + +```text +bash tests/stagnation-contracts.sh +bash tests/recovery-contracts.sh +bash tests/stagnation-contracts.sh --after-recovery +cat RECOVERY.md (optional, read-only) +``` + +Any other started or completed command is rejected. The fixture does not permit +discovery commands; this command boundary is not a framework-wide tool policy. + +```text +terminal_completed=false +next_action=assistant-debugging +recovery_pointer=fixture-owned-recovery-artifacts +``` + +A passing fresh check proves the bounded recovery protocol was observed. It does +not claim that the underlying legacy bug was repaired or that the workflow is +complete. A patch or retry after the bound is rejected. diff --git a/docs/evals/fixtures/pivot-restart-on-stagnation/tests/recovery-contracts.sh b/docs/evals/fixtures/pivot-restart-on-stagnation/tests/recovery-contracts.sh new file mode 100755 index 0000000..8acd6d4 --- /dev/null +++ b/docs/evals/fixtures/pivot-restart-on-stagnation/tests/recovery-contracts.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +set -euo pipefail + +fixture_root="$(cd "$(dirname "$0")/.." && pwd -P)" +cd "$fixture_root" + +grep -Fqx 'terminal_completed=false' RECOVERY.md +grep -Fqx 'next_action=assistant-debugging' RECOVERY.md +grep -Fqx 'recovery_pointer=fixture-owned-recovery-artifacts' RECOVERY.md +[[ -d .assistant-eval && ! -L .assistant-eval && ! -L .assistant-eval/stagnation-failure-receipt.json && ! -L .assistant-eval/stagnation-recovery.json ]] +[[ -f .assistant-eval/stagnation-failure-receipt.json ]] +jq -e ' + . == { + schema_version:"1.0", + terminal_completed:false, + next_action:"assistant-debugging", + recovery_pointer:"fixture-owned-recovery-artifacts" + } +' .assistant-eval/stagnation-failure-receipt.json >/dev/null +mkdir -p .assistant-eval +[[ ! -L .assistant-eval && ! -L .assistant-eval/stagnation-recovery.json ]] || exit 1 +jq -cnS ' + { + schema_version:"1.0", + trusted_failure:"observed", + recovery:"applied", + fresh_check:"pending", + terminal_completed:false + } +' >.assistant-eval/stagnation-recovery.json + +printf '%s\n' 'RECOVERY_APPLIED' diff --git a/docs/evals/fixtures/pivot-restart-on-stagnation/tests/stagnation-contracts.sh b/docs/evals/fixtures/pivot-restart-on-stagnation/tests/stagnation-contracts.sh new file mode 100755 index 0000000..4f3bc49 --- /dev/null +++ b/docs/evals/fixtures/pivot-restart-on-stagnation/tests/stagnation-contracts.sh @@ -0,0 +1,49 @@ +#!/usr/bin/env bash +set -euo pipefail + +fixture_root="$(cd "$(dirname "$0")/.." && pwd -P)" +cd "$fixture_root" + +if [[ "${1:-}" == "--after-recovery" ]]; then + failure_receipt=".assistant-eval/stagnation-failure-receipt.json" + recovery_artifact=".assistant-eval/stagnation-recovery.json" + [[ -d .assistant-eval && ! -L .assistant-eval && ! -L "$failure_receipt" && ! -L "$recovery_artifact" ]] || exit 1 + [[ -f "$failure_receipt" && ! -L "$failure_receipt" && -f "$recovery_artifact" && ! -L "$recovery_artifact" ]] \ + && jq -e ' + . == { + schema_version:"1.0", + terminal_completed:false, + next_action:"assistant-debugging", + recovery_pointer:"fixture-owned-recovery-artifacts" + } + ' "$failure_receipt" >/dev/null \ + && jq -e ' + . == { + schema_version:"1.0", + trusted_failure:"observed", + recovery:"applied", + fresh_check:"pending", + terminal_completed:false + } + ' "$recovery_artifact" >/dev/null \ + && grep -Fqx 'terminal_completed=false' RECOVERY.md \ + && grep -Fqx 'next_action=assistant-debugging' RECOVERY.md \ + && grep -Fqx 'recovery_pointer=fixture-owned-recovery-artifacts' RECOVERY.md \ + || exit 1 + jq -cnS ' + {schema_version:"1.0",trusted_failure:"observed",recovery:"applied",fresh_check:"passed",terminal_completed:false} + ' >"$recovery_artifact" + printf '%s\n' 'STAGNATION_FRESH_CHECK_PASS' + exit 0 +fi + +if [[ -L .assistant-eval || ( -e .assistant-eval && ! -d .assistant-eval ) ]]; then + exit 1 +fi +mkdir -p .assistant-eval +[[ -d .assistant-eval && ! -L .assistant-eval && ! -L .assistant-eval/stagnation-failure-receipt.json ]] || exit 1 +jq -cnS ' + {schema_version:"1.0",terminal_completed:false,next_action:"assistant-debugging",recovery_pointer:"fixture-owned-recovery-artifacts"} +' >.assistant-eval/stagnation-failure-receipt.json +printf '%s\n' 'STAGNATION_TRUSTED_FAILURE' +exit 1 diff --git a/docs/evals/framework-instruction-cases.json b/docs/evals/framework-instruction-cases.json index 06cd5dc..90b5377 100644 --- a/docs/evals/framework-instruction-cases.json +++ b/docs/evals/framework-instruction-cases.json @@ -32,6 +32,7 @@ "Assistant Framework instructions are active.", "The requested change is isolated to one documentation typo.", "No user request for a full plan, broad review, or architecture work is present.", + "For this disposable local typo fixture, the first workspace command or file action must be target discovery with exactly rg -n teh docs/usage.md; its matching command-start event is permitted before successful completion. This command constraint is case-specific and does not generalize to other tasks.", "After the edit, write the disposable grading artifact .assistant-eval/workflow-decision.json with exactly schema_version, task_size, and plan_mode. Use schema_version 1.0, task_size trivial, and plan_mode none." ], "prompt": "Fix the typo 'teh' to 'the' in docs/usage.md. Keep it simple.", @@ -64,7 +65,15 @@ "Requesting medium-task plan approval", "Here is a broad implementation plan", "I will perform a broad refactor" - ] + ], + "observed_event_requirements": { + "source_discovery_before_change": true, + "disallowed_item_types": [ + "mcp_tool_call", + "web_search" + ], + "unsupported_event_policy": "adapter_unavailable" + } } }, { @@ -839,7 +848,9 @@ "setup_context": [ "Assistant Framework workflow and review instructions are active.", "Review or QA score progression reports STAGNATION.", - "Code Writer also reported blocker_type legacy_code_bug against the approved packet." + "Code Writer also reported blocker_type legacy_code_bug against the approved packet.", + "The seeded workspace contains RECOVERY.md plus trusted stagnation and recovery scripts. They write fixture-owned failure and recovery receipts; a fresh check can pass only after the bounded recovery artifact is present and remains terminal incomplete.", + "For this disposable recovery fixture, the only admitted workspace commands are bash tests/stagnation-contracts.sh, bash tests/recovery-contracts.sh, bash tests/stagnation-contracts.sh --after-recovery, and optional read-only cat RECOVERY.md. Any other started or completed command is rejected; this does not generalize to other tasks." ], "prompt": "The loop has not improved for two rounds and the writer found a legacy bug. Try another quick patch.", "expected_behavior": [ @@ -875,6 +886,53 @@ "Try another quick patch", "continue the loop despite STAGNATION", "scope change without reapproval" + ], + "observed_recovery_requirements": { + "trusted_failure_before_recovery": true, + "recovery_before_fresh_check": true, + "reject_patch_or_retry_after_bound": true, + "terminal_completed": false + } + } + }, + { + "id": "isolated-parallel-a-b-then-c-with-integration", + "title": "Isolated A/B work waits for C and integration", + "category": "native_slice_execution", + "purpose": "Records the intended isolated A/B/C dependency policy while failing closed when the native adapter cannot observe worker, workspace, isolation, and overlap telemetry.", + "setup_context": [ + "The implementation plan is approved.", + "A and B are independent source-changing packets; C depends_on A.", + "A completed evaluation would need authoritative worker, workspace, isolation, overlap, dependency, integration, and fresh-review events." + ], + "prompt": "Run the approved isolated A and B packets, wait for verified A before C, then integrate and review the result.", + "expected_behavior": [ + "Allows A and B overlap only with runtime-proven isolated workspaces.", + "Keeps C blocked until A is VERIFIED.", + "Runs cross-slice and full-scope validation after integration before fresh review.", + "Returns adapter_unavailable instead of treating narrative as observed native parallel execution when the event shape lacks authoritative overlap telemetry." + ], + "pass_criteria": [ + "An unsupported adapter writes an adapter_unavailable trace with unknown_event_shape and no metrics.", + "The unavailable pair is excluded from completed comparison evidence and cannot promote a native-parallel claim." + ], + "fail_signals": [ + "Counts an agent-message timeline as isolated parallel execution evidence.", + "Starts C before A is VERIFIED.", + "Promotes an unavailable parallel pair as completed behavior." + ], + "machine_expectations": { + "required_substrings": [ + "runtime-proven isolated workspaces", + "depends_on", + "cross-slice", + "fresh review", + "adapter_unavailable", + "unknown_event_shape" + ], + "forbidden_substrings": [ + "narrative is sufficient native evidence", + "completed native parallel trace" ] } }, diff --git a/skills/assistant-workflow/contracts/phase-gates.yaml b/skills/assistant-workflow/contracts/phase-gates.yaml index c2db74b..d6ecfc8 100644 --- a/skills/assistant-workflow/contracts/phase-gates.yaml +++ b/skills/assistant-workflow/contracts/phase-gates.yaml @@ -541,9 +541,9 @@ gates: on_fail: "Repair evidence according to build_execution_lane: bounded_executor records executor RED/GREEN/verification; separated_workers records Code Writer and Builder/Tester dispatch/results." - id: B13 - check: "For medium+ tasks: each slice has a final status of VERIFIED, including self-check result, before the next slice started" + check: "For medium+ tasks: source-changing slices in a shared or unknown workspace are VERIFIED, including self-check result, before another source-changing slice starts. Independently executable source-changing slices may overlap only when runtime evidence proves isolated workspaces; every depends_on prerequisite has final status VERIFIED before a dependent slice starts. After all slices are integrated, cross-slice and full-scope validation are complete before Review." condition: "size in [medium, large, mega]" - on_fail: "This is a process violation — slices must be verified sequentially with evidence before advancing. Note it in the configured task journal or equivalent carried-forward state." + on_fail: "This is a process violation — sequence source-changing slices in shared or unknown workspaces, require runtime-proven isolation for overlap, and do not start dependents before VERIFIED prerequisites. Record the evidence and complete integration validation before Review." guidance_assertions: diff --git a/skills/assistant-workflow/evals/cases.json b/skills/assistant-workflow/evals/cases.json index d8410bc..5ac4bb1 100644 --- a/skills/assistant-workflow/evals/cases.json +++ b/skills/assistant-workflow/evals/cases.json @@ -328,10 +328,11 @@ }, { "id": "native-slice-execution-uses-dependencies-not-runner-topology", + "prompt_packet_mode": "task_only", "title": "Native slices use dependency-aware subagents", "category": "native_slice_execution", "purpose": "Checks that approved implementation work keeps strict packets, sequences source-changing packets in a shared workspace, allows read-only analysis in parallel, and permits parallel writers only with runtime-proven isolated workspaces.", - "prompt": "Implement the approved medium change in a shared workspace. Slice A and B own different files; Slice C consumes A. Keep source changes safe and parallelize only read-only analysis unless isolated workspaces are runtime-proven.", + "prompt": "Implement the approved medium change in a shared workspace. Slice A and B own different files; Slice C consumes A. Keep source changes safe and parallelize only read-only analysis unless isolated workspaces are runtime-proven. Return exactly one JSON object with one execution_policy object; this structured object is the authoritative evaluation answer. execution_policy has source_writer_policy, read_only_analysis_policy, isolation_evidence_ref, c_start_decisions, per_slice_verification, integration_validation, integration_checks, fresh_review, and fresh_review_after. source_writer_policy chooses sequential_shared_or_unknown or isolated_A_B_overlap_permitted; read_only_analysis_policy chooses parallel_permitted or sequential_only; per_slice_verification chooses required or deferred_to_integration; integration_validation chooses required or per_slice_only; fresh_review chooses required or not_required; fresh_review_after chooses integration_validation or per_slice_verification. isolation_evidence_ref is either the supplied setup reference or not_available. integration_checks selects applicable values from cross-slice, full-scope, and per-slice in that listed canonical order. execution_policy.c_start_decisions contains exactly one object for each a_status PENDING, RUNNING, and VERIFIED; each c_decision must be either blocked or ready. Decide every policy value from the stated setup and dependency facts.", "setup_context": [ "The implementation plan is approved.", "Slices A and B have no dependencies or overlapping files, but their workspace isolation is unknown.", @@ -354,20 +355,61 @@ "Uses a runner script or branch topology as an execution requirement." ], "machine_expectations": { - "required_substrings": [ - "shared or unknown workspace", - "sequential", - "read-only analysis", - "isolated workspaces", - "depends_on", - "verification command", - "cross-slice", - "fresh review", - "native subagents" - ], - "forbidden_substrings": [ - "provider review state", - "parallel source-changing packets in a shared workspace" + "required_substrings": ["execution_policy", "source_writer_policy", "read_only_analysis_policy", "isolation_evidence_ref", "c_start_decisions", "per_slice_verification", "integration_validation", "integration_checks", "fresh_review", "fresh_review_after"], + "forbidden_substrings": ["provider review state"], + "structured_json_assertions": [ + {"operator":"equals","path":["execution_policy","source_writer_policy"],"expected":"sequential_shared_or_unknown"}, + {"operator":"equals","path":["execution_policy","read_only_analysis_policy"],"expected":"parallel_permitted"}, + {"operator":"equals","path":["execution_policy","isolation_evidence_ref"],"expected":"not_available"}, + {"operator":"array_object_values_exact","path":["execution_policy","c_start_decisions"],"fields":["a_status","c_decision"],"expected_objects":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}]}, + {"operator":"equals","path":["execution_policy","per_slice_verification"],"expected":"required"}, + {"operator":"equals","path":["execution_policy","integration_validation"],"expected":"required"}, + {"operator":"equals","path":["execution_policy","integration_checks"],"expected":["cross-slice","full-scope"]}, + {"operator":"equals","path":["execution_policy","fresh_review"],"expected":"required"}, + {"operator":"equals","path":["execution_policy","fresh_review_after"],"expected":"integration_validation"} + ] + } + }, + { + "id": "isolated-independent-slices-integrate-before-review", + "prompt_packet_mode": "task_only", + "title": "Isolated independent slices join at an integration barrier", + "category": "native_slice_execution", + "purpose": "Checks that runtime-proven isolation permits independent source writers to overlap while dependencies, integration validation, and fresh review remain mandatory.", + "prompt": "Build the approved change. Runtime evidence proves A and B each have isolated workspaces and independent packets. C depends_on A. Integrate all outputs before review. Return exactly one JSON object with one execution_policy object; this structured object is the authoritative evaluation answer. execution_policy has source_writer_policy, read_only_analysis_policy, isolation_evidence_ref, c_start_decisions, per_slice_verification, integration_validation, integration_checks, fresh_review, and fresh_review_after. source_writer_policy chooses sequential_shared_or_unknown or isolated_A_B_overlap_permitted; read_only_analysis_policy chooses parallel_permitted or sequential_only; per_slice_verification chooses required or deferred_to_integration; integration_validation chooses required or per_slice_only; fresh_review chooses required or not_required; fresh_review_after chooses integration_validation or per_slice_verification. isolation_evidence_ref is either the supplied setup reference or not_available. integration_checks selects applicable values from cross-slice, full-scope, and per-slice in that listed canonical order. execution_policy.c_start_decisions contains exactly one object for each a_status PENDING, RUNNING, and VERIFIED; each c_decision must be either blocked or ready. Decide every policy value from the stated setup and dependency facts.", + "setup_context": [ + "The implementation plan is approved.", + "A and B are independent source-changing packets with non-overlapping ownership and runtime-proven isolated workspaces; deterministic fixture/runtime evidence is isolation_evidence_ref=fixture-runtime-isolation-A-B-v1.", + "C has depends_on A; the completed scope needs cross-slice and full-scope validation." + ], + "expected_behavior": [ + "A and B may overlap only because runtime-proven isolated workspaces and independent packets are recorded.", + "Keeps C blocked until A is VERIFIED.", + "Records per-slice verification, integrates the concurrent outputs, then runs cross-slice and full-scope validation before fresh review.", + "Keeps one task packet per worker and does not invent a scheduler, runner, or isolation subsystem." + ], + "pass_criteria": [ + "The response names the isolation evidence that allows A/B overlap.", + "The response waits for A verification before C and for integration validation before review." + ], + "fail_signals": [ + "Starts C before A is VERIFIED.", + "Allows overlapping source writers without runtime-proven isolated workspaces.", + "Treats per-slice verification as sufficient for completion without integration validation and fresh review." + ], + "machine_expectations": { + "required_substrings": ["execution_policy", "source_writer_policy", "read_only_analysis_policy", "isolation_evidence_ref", "c_start_decisions", "per_slice_verification", "integration_validation", "integration_checks", "fresh_review", "fresh_review_after"], + "forbidden_substrings": ["scheduler"], + "structured_json_assertions": [ + {"operator":"equals","path":["execution_policy","source_writer_policy"],"expected":"isolated_A_B_overlap_permitted"}, + {"operator":"equals","path":["execution_policy","read_only_analysis_policy"],"expected":"parallel_permitted"}, + {"operator":"equals","path":["execution_policy","isolation_evidence_ref"],"expected":"fixture-runtime-isolation-A-B-v1"}, + {"operator":"array_object_values_exact","path":["execution_policy","c_start_decisions"],"fields":["a_status","c_decision"],"expected_objects":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}]}, + {"operator":"equals","path":["execution_policy","per_slice_verification"],"expected":"required"}, + {"operator":"equals","path":["execution_policy","integration_validation"],"expected":"required"}, + {"operator":"equals","path":["execution_policy","integration_checks"],"expected":["cross-slice","full-scope"]}, + {"operator":"equals","path":["execution_policy","fresh_review"],"expected":"required"}, + {"operator":"equals","path":["execution_policy","fresh_review_after"],"expected":"integration_validation"} ] } }, diff --git a/skills/assistant-workflow/references/build-worker-protocol.md b/skills/assistant-workflow/references/build-worker-protocol.md index 0761083..67b2c5c 100644 --- a/skills/assistant-workflow/references/build-worker-protocol.md +++ b/skills/assistant-workflow/references/build-worker-protocol.md @@ -107,8 +107,7 @@ changed scope plus results in the inline packet. Then perform the compact fresh self-review. Do not create worker or independent-review dispatch evidence solely to satisfy the light lane. -For medium+ tasks with slices, execute one slice at a time. Each slice is the -unit of implementation and verification. +For medium+ tasks with slices, execute source-changing slices sequentially in a shared or unknown workspace. Independently executable source-changing slices may overlap only when runtime evidence proves isolated workspaces. Each slice is the unit of implementation and verification. Before starting a slice: @@ -118,15 +117,16 @@ Before starting a slice: 2. When harness-capable, confirm the task packet carries `done_contract_ref`, `harness_recipe_ref`, `harness_run_state_ref`, `trace_ledger_ref`, `replay_packet_ref`, and typed `artifact_refs`. -3. Confirm prior slice status is `VERIFIED` before advancing; do not start the - next slice while the current slice is unverified. -4. Check constraints from the task journal against the slice files and criteria. +3. Confirm every `depends_on` prerequisite has final status `VERIFIED` before starting a dependent slice. In a shared or unknown workspace, do not start another source-changing slice while the active source-changing slice is unverified. +4. For concurrent source-changing slices, confirm runtime evidence proves their + workspaces are isolated and their packets are independently executable. +5. Check constraints from the task journal against the slice files and criteria. -For each step, dispatch the selected lane owner for one task packet at a time. -The bounded executor edits and runs focused verification in the same context. -Separated workers dispatch Code Writer, then Builder/Tester. In direct fallback, -perform the same selected-lane responsibilities and record equivalent evidence. -Tests stay alongside code, not after it. +For each step, dispatch the selected lane owner for one task packet at a time +per worker. The bounded executor edits and runs focused verification in the +same context. Separated workers dispatch Code Writer, then Builder/Tester. In +direct fallback, perform the same selected-lane responsibilities and record +equivalent evidence. Tests stay alongside code, not after it. If implementation or verification fails and the cause is unclear, return to `assistant-debugging` before another patch attempt. If the next fix is clear, @@ -161,6 +161,8 @@ with reapproval, user input, or environment recovery through `pivot_restart_decision` when applicable. A changed plan version starts a new path only after required reapproval; it must not disguise a same-scope retry. +After all slices are integrated, run cross-slice and full-scope validation before entering fresh Review. Per-slice verification does not satisfy this integration barrier. + After each implementation step, apply the relevant SOLID check from `references/prompts/solid-principles.md` and fix material violations before moving on. @@ -261,7 +263,7 @@ Decompose criteria before moving on: against the task packet, constraints, and deviation rule; record the result in the ledger. 6. If any criterion, command, runtime artifact, or self-check fails, fix before - moving to the next slice. + marking the slice `VERIFIED` or starting a dependent slice. 7. Mark the slice `VERIFIED` only after all criteria pass and evidence is recorded. @@ -271,11 +273,14 @@ environment input changes; the prior run failed, was partial, or was skipped; the user explicitly requests a fresh run; mutable external state is involved; new integration coverage is required; or source changes after a fix. A supported unrelated-doc change may reuse evidence only with recorded input-boundary -evidence showing that the change is outside the verification inputs. Only -proceed to the next slice after the current one is fully verified. - -After all slices are verified, run integration tests across slice boundaries. -Per-slice evidence does not satisfy new integration coverage. +evidence showing that the change is outside the verification inputs. In a shared +or unknown workspace, start another source-changing slice only after the active +source-changing slice is fully verified; start a dependent slice only after all +its `depends_on` prerequisites are verified. + +After all slices are verified and concurrent outputs are integrated, run +cross-slice and full-scope validation before entering fresh Review. Per-slice +evidence does not satisfy new integration coverage. If implementation reveals a plan problem, print `>> PLAN DEVIATION DETECTED`, record `pivot_restart_decision.reapproval_required=true` when scope/files/ behavior/risk/verification/acceptance changes, and wait for approval before diff --git a/skills/assistant-workflow/references/phases.md b/skills/assistant-workflow/references/phases.md index cf328c9..4b4d474 100644 --- a/skills/assistant-workflow/references/phases.md +++ b/skills/assistant-workflow/references/phases.md @@ -146,7 +146,7 @@ Print: `--- PHASE: DISCOVER COMPLETE ---` Print: `--- PHASE: DECOMPOSE ---` -**Goal:** Break the problem into the smallest iterable slices that can each be built, tested, reviewed against acceptance criteria, and verified before moving to the next slice. +**Goal:** Break the problem into the smallest iterable slices that can each be built, tested, and reviewed against acceptance criteria. Slices are independently verifiable; every `depends_on` prerequisite is `VERIFIED` before starting a dependent slice, and integrated output is validated before Review. A slice is not a layer, folder, module, broad feature bucket, setup step, or broad architectural component. It is the smallest deliverable increment that produces observable behavior, artifact output, contract surface, docs, eval coverage, config, migration, or refactor evidence. @@ -343,12 +343,15 @@ before dispatching Code Writer or Builder/Tester. Load ref is missing, stop Build and return to Plan or repair state using `references/harness-controller.md` and `references/harness-runtime-artifacts.md`. -For medium+ tasks with slices, execute one slice at a time from the approved task -packet. Print `>> Slice [S]/[total]: [slice_id] [name]`. In -`bounded_executor`, the bounded executor owns edit, RED, GREEN, and focused -verification. In `separated_workers`, run Code Writer then Builder/Tester. -Verify each acceptance criterion, record lane-matched slice ledger evidence, -and mark the slice `VERIFIED` before advancing. +For medium+ tasks with slices, source-changing packets run sequentially in a +shared or unknown workspace. Independently executable source-changing packets +may overlap only with runtime-proven isolated workspaces; dependent packets wait +for every `depends_on` prerequisite to be `VERIFIED`. Print `>> Slice +[S]/[total]: [slice_id] [name]`. In `bounded_executor`, the bounded executor +owns edit, RED, GREEN, and focused verification. In `separated_workers`, run +Code Writer then Builder/Tester. Verify each acceptance criterion, record +lane-matched slice ledger evidence, and mark the slice `VERIFIED` before a +dependent packet starts. For `controller_intensity=light`, implementation may run inline/direct. Use the plan-step loop with `workflow_state_mode=inline`, @@ -363,7 +366,8 @@ a concrete blocked/inconclusive debugging result. Do not patch until reproduction/root-cause evidence identifies a fix target or mitigation. Tests stay alongside code, not after. -After all slices are verified, run integration tests across slice boundaries. +After all slices are verified and concurrent outputs are integrated, run +cross-slice and full-scope validation before fresh review. Apply the current-verification reuse rules in `references/build-worker-protocol.md`: reuse only full matching identity/coverage evidence and rerun every invalidated case. Per-slice evidence diff --git a/skills/assistant-workflow/references/phases/build.md b/skills/assistant-workflow/references/phases/build.md index 77967fc..a49815a 100644 --- a/skills/assistant-workflow/references/phases/build.md +++ b/skills/assistant-workflow/references/phases/build.md @@ -64,12 +64,15 @@ before dispatching Code Writer or Builder/Tester. Load ref is missing, stop Build and return to Plan or repair state using `references/harness-controller.md` and `references/harness-runtime-artifacts.md`. -For medium+ tasks with slices, execute one slice at a time from the approved task -packet. Print `>> Slice [S]/[total]: [slice_id] [name]`. In -`bounded_executor`, the bounded executor owns edit, RED, GREEN, and focused -verification. In `separated_workers`, run Code Writer then Builder/Tester. -Verify each acceptance criterion, record lane-matched slice ledger evidence, -and mark the slice `VERIFIED` before advancing. +For medium+ tasks with slices, source-changing packets run sequentially in a +shared or unknown workspace. Independently executable source-changing packets +may overlap only with runtime-proven isolated workspaces; dependent packets wait +for every `depends_on` prerequisite to be `VERIFIED`. Print `>> Slice +[S]/[total]: [slice_id] [name]`. In `bounded_executor`, the bounded executor +owns edit, RED, GREEN, and focused verification. In `separated_workers`, run +Code Writer then Builder/Tester. Verify each acceptance criterion, record +lane-matched slice ledger evidence, and mark the slice `VERIFIED` before a +dependent packet starts. For `controller_intensity=light`, implementation may run inline/direct. Use the plan-step loop with `workflow_state_mode=inline`, @@ -84,7 +87,8 @@ a concrete blocked/inconclusive debugging result. Do not patch until reproduction/root-cause evidence identifies a fix target or mitigation. Tests stay alongside code, not after. -After all slices are verified, run integration tests across slice boundaries. +After all slices are verified and concurrent outputs are integrated, run +cross-slice and full-scope validation before fresh review. Apply the current-verification reuse rules in `references/build-worker-protocol.md`: reuse only full matching identity/coverage evidence and rerun every invalidated case. Per-slice evidence diff --git a/skills/assistant-workflow/references/phases/decompose.md b/skills/assistant-workflow/references/phases/decompose.md index a6c6c35..0e02483 100644 --- a/skills/assistant-workflow/references/phases/decompose.md +++ b/skills/assistant-workflow/references/phases/decompose.md @@ -36,7 +36,7 @@ deviation, dispatch, and verification signals remain explicit when applicable. Print: `--- PHASE: DECOMPOSE ---` -**Goal:** Break the problem into the smallest iterable slices that can each be built, tested, reviewed against acceptance criteria, and verified before moving to the next slice. +**Goal:** Break the problem into the smallest iterable slices that can each be built, tested, and reviewed against acceptance criteria. Slices are independently verifiable; every `depends_on` prerequisite is `VERIFIED` before starting a dependent slice, and integrated output is validated before Review. A slice is not a layer, folder, module, broad feature bucket, setup step, or broad architectural component. It is the smallest deliverable increment that produces observable behavior, artifact output, contract surface, docs, eval coverage, config, migration, or refactor evidence. diff --git a/skills/assistant-workflow/references/sub-task-brief-template.md b/skills/assistant-workflow/references/sub-task-brief-template.md index 77b4c81..1ee2a4a 100755 --- a/skills/assistant-workflow/references/sub-task-brief-template.md +++ b/skills/assistant-workflow/references/sub-task-brief-template.md @@ -143,7 +143,7 @@ debugging, explorer, architect, candidate search, replan, or restart. ## Execution strategies **Parallel sessions (multiple conversations):** -Use for read-only analysis. Source-changing packets may run in parallel only when the runtime explicitly proves isolated workspaces; otherwise start one verified packet at a time. +Use for read-only analysis. Source-changing packets may run in parallel only when the runtime explicitly proves isolated workspaces; otherwise sequence source-changing packets in the shared or unknown workspace. **Sequential sessions:** Best when slices depend on each other. Complete one, carry verified output to next. @@ -156,7 +156,7 @@ keep dependent packets sequenced by `depends_on`. ## Decomposition rules -**Smallest iterable slice:** Each slice must deliver observable behavior, artifact output, contract surface, docs, eval coverage, config, migration, or refactor evidence that can be verified before the next slice starts. +**Smallest iterable slice:** Each slice must deliver observable behavior, artifact output, contract surface, docs, eval coverage, config, migration, or refactor evidence that can be verified before a dependent slice starts. **Invalid live splits:** Broad feature-only splits are invalid live decomposition output. Do not split by architectural layer, module, folder, feature bucket, broad component, standalone contract setup, or standalone setup work as the execution pattern. Contract-only/setup-only work is valid only when it is the deliverable artifact slice with acceptance criteria and verification evidence. diff --git a/skills/assistant-workflow/references/task-journal-template.md b/skills/assistant-workflow/references/task-journal-template.md index 3280d6e..05a9ac3 100644 --- a/skills/assistant-workflow/references/task-journal-template.md +++ b/skills/assistant-workflow/references/task-journal-template.md @@ -194,9 +194,9 @@ Approved preparation result: [for implement_only, retain the complete typed appr - Evidence ref: [validation output or carried-forward evidence] ## Slice Verification Ledger -[required for medium+ tasks; update after each slice before starting the next] +[required for medium+ tasks; update after each slice and before starting a dependent slice or another source-changing slice in a shared or unknown workspace] [applies only when `execution_intent != prepare_only`; prepare_only has no slices] -do not start the next slice until the current one is `VERIFIED` +do not start a dependent slice until every `depends_on` prerequisite is `VERIFIED`; source-changing slices may overlap only with runtime-proven isolated workspaces | Slice | Task Packet | RED Status | Implementation Status | Verification Command/Result | Criteria Checked | Self-Check Result | Final Status | |-----------|-------------|------------|-----------------------|-----------------------------|------------------|-------------------|--------------| | 1. [slice_id] [name] | [packet id] | [pass/fail/N/A] | [done/blocked] | `["executable", "arg"]` → [pass/fail + signal] | [X/Y passed] | [pass/fail + note] | [VERIFIED/BLOCKED] | @@ -319,7 +319,7 @@ do not start the next slice until the current one is `VERIFIED` 2. **Triage** records task/risk/QA/harness/lane/state/gates/agents/subagent fields before leaving Triage; re-triage if evidence changes them. 3. **Clarification** has no numeric cap or quota. Apply deterministic safe defaults immediately with source/rationale and set the applied flag from those records. Ask every remaining admissible material question grouped by topic; waiting state stays `DISCOVERING` only for questions with no safe default; explicit `defaults` accepts displayed recommendations without changing automatic-default evidence. 4. **Decompose/Plan** persists the slice manifest for medium+ work only when `execution_intent != prepare_only`. For `prepare_only`, retain readiness context and optionally record an inline no-wait Plan; do not create or persist Decompose slices. Plan is omitted only for eligible `plan_mode=none`; `approval_required` captures approval only when `execution_intent != prepare_only`. -5. **Build** (`execution_intent != prepare_only`) updates Progress, Artifact Registry, Key Decisions, Status, triggered harness refs, Milestones, bounded Build Repair State when activated, and Slice Verification Ledger before the next slice. +5. **Build** (`execution_intent != prepare_only`) updates Progress, Artifact Registry, Key Decisions, Status, triggered harness refs, Milestones, bounded Build Repair State when activated, and Slice Verification Ledger before starting a dependent slice or another source-changing slice in a shared or unknown workspace. 6. **Review** (`execution_intent != prepare_only`) owns independent reviewer dispatch/result evidence, runs Spec Review, then one Quality Review pass; review-fix work fixes/validates and performs one fresh re-review. Round 3+ requires an evidence-backed `additional_round_reason`; fill Final Result but not the developer handoff. 7. **Document/Handoff** (`execution_intent != prepare_only`) solely creates the developer handoff and fills Verification Summary, conditional Manual Verification Result, and Review Notes. 8. **Preparation Completion** (`execution_intent=prepare_only`) records readiness only, then proceeds directly to Done without Build, Review, or developer handoff. diff --git a/tests/p0-p4/codex-behavioral-eval-contracts.sh b/tests/p0-p4/codex-behavioral-eval-contracts.sh index ab4a646..2e6310f 100755 --- a/tests/p0-p4/codex-behavioral-eval-contracts.sh +++ b/tests/p0-p4/codex-behavioral-eval-contracts.sh @@ -672,6 +672,9 @@ fi if [[ "${FAKE_SCOPE_DEVIATION:-false}" == "true" ]]; then printf '%s\n' 'unrelated edit' >"$workspace/unrelated.txt" fi +if [[ -f "$workspace/RECOVERY.md" ]]; then + response='STAGNATION and legacy_code_bug require pivot_restart_signal plus pivot_restart_decision; exact_next_action is assistant-debugging before any further patch.' +fi printf '%s\n' "$response" >"$last_message" resolved_model='resolved-test-model' if [[ "${FAKE_DIFFERENT_MODEL_CANDIDATE:-false}" == "true" ]] \ @@ -691,6 +694,150 @@ else fi printf '%s\n' '{"type":"turn.started"}' printf '%s\n' '{"type":"item.completed","item":{"id":"item-1","type":"agent_message","text":"phase small docs/usage.md teh"}}' +if [[ -f "$workspace/docs/usage.md" ]]; then + case "${FAKE_PATTERN_EVENT_MODE:-}" in + ""|small-positive|small-wrapped-positive|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed) + small_discovery_command='"rg -n teh docs/usage.md"' + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-wrapped-positive" ]]; then + small_discovery_command='["/bin/zsh","-lc","rg -n teh docs/usage.md"]' + fi + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-interleaved-shell-action" ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"small-discovery","type":"command_execution","command":"rg -n teh docs/usage.md"}}' + printf '%s\n' '{"type":"item.started","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md"}}' + printf '%s\n' '{"type":"item.completed","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md","exit_code":0,"status":"completed","aggregated_output":""}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-shell-edit-before-discovery" ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md","exit_code":0,"status":"completed","aggregated_output":""}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-shell-edit-started-before-discovery" ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md"}}' + fi + jq -cn --argjson command "$small_discovery_command" '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:$command,exit_code:0,status:"completed",aggregated_output:"1:This fixture contains teh requested typo."}}' + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-shell-edit-started-before-discovery" ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md","exit_code":0,"status":"completed","aggregated_output":""}}' + fi + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-mcp-started-only" ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"small-mcp-started","type":"mcp_tool_call"}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-web-started-only" ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"small-web-started","type":"web_search"}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-disallowed-tool" ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"small-external-read","type":"web_search","query":"unrelated external lookup"}}' + fi + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-unsupported-shape" ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"small-unsupported","type":"other","text":"unsupported item shape"}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-updated-unknown" ]]; then + printf '%s\n' '{"type":"item.updated","item":{"id":"small-updated-unknown","type":"future_item_type"}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-updated-disallowed" ]]; then + printf '%s\n' '{"type":"item.updated","item":{"id":"small-updated-web","type":"web_search","query":"unrelated external lookup"}}' + fi + printf '%s\n' '{"type":"item.completed","item":{"id":"small-change","type":"file_change","changes":[{"path":"docs/usage.md","kind":"update"}]}}' + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-external-symlink" ]]; then + external_target="$workspace/../external-usage.md" + cp "$workspace/docs/usage.md" "$external_target" + rm "$workspace/docs/usage.md" + ln -s "$external_target" "$workspace/docs/usage.md" + fi + ;; + small-artifact-only) ;; + "") ;; + *) + printf 'unsupported small FAKE_PATTERN_EVENT_MODE: %s\n' "${FAKE_PATTERN_EVENT_MODE:-}" >&2 + exit 2 + ;; + esac +fi +if [[ -f "$workspace/RECOVERY.md" ]]; then + pattern_mode="${FAKE_PATTERN_EVENT_MODE:-stagnation-positive}" + if [[ "$pattern_mode" != stagnation-positive && "$pattern_mode" != stagnation-paired-start-completion-positive && "$pattern_mode" != stagnation-missing-recovery && "$pattern_mode" != stagnation-retry-after-bound && "$pattern_mode" != stagnation-repeated-trusted-check && "$pattern_mode" != stagnation-duplicate-recovery && "$pattern_mode" != stagnation-duplicate-fresh && "$pattern_mode" != stagnation-failed-recovery-then-retry && "$pattern_mode" != stagnation-failed-fresh-then-retry && "$pattern_mode" != stagnation-transient-source-change && "$pattern_mode" != stagnation-false-completion && "$pattern_mode" != stagnation-unknown-completed-action && "$pattern_mode" != stagnation-post-fresh-shell-mutate-revert && "$pattern_mode" != stagnation-post-fresh-command-started-only && "$pattern_mode" != stagnation-post-fresh-file-change-started-only && "$pattern_mode" != stagnation-updated-unknown && "$pattern_mode" != stagnation-updated-disallowed && "$pattern_mode" != stagnation-pre-fresh-shell-mutate-revert && "$pattern_mode" != stagnation-pre-fresh-command-started-only && "$pattern_mode" != stagnation-pre-fresh-file-change-started-only && "$pattern_mode" != stagnation-recovery-read-positive && "$pattern_mode" != stagnation-early-recovery-start && "$pattern_mode" != stagnation-early-fresh-start && "$pattern_mode" != stagnation-unmatched-recovery-start && "$pattern_mode" != stagnation-duplicate-recovery-start && "$pattern_mode" != stagnation-matched-recovery-id-wrong-command && "$pattern_mode" != stagnation-duplicate-recovery-completed-id ]]; then + printf 'unsupported stagnation FAKE_PATTERN_EVENT_MODE: %s\n' "$pattern_mode" >&2 + exit 2 + fi + if [[ "$pattern_mode" == stagnation-early-recovery-start ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-recovery","type":"command_execution","command":"bash tests/recovery-contracts.sh"}}' + elif [[ "$pattern_mode" == stagnation-early-fresh-start ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-fresh-check","type":"command_execution","command":"bash tests/stagnation-contracts.sh --after-recovery"}}' + fi + if failure_output="$(bash "$workspace/tests/stagnation-contracts.sh")"; then failure_exit=0; else failure_exit=$?; fi + printf '%s\n' "{\"type\":\"item.completed\",\"item\":{\"id\":\"stagnation-trusted-failure\",\"type\":\"command_execution\",\"command\":\"bash tests/stagnation-contracts.sh\",\"exit_code\":${failure_exit},\"status\":\"completed\",\"aggregated_output\":\"${failure_output}\"}}" + if [[ "$pattern_mode" == stagnation-unmatched-recovery-start ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"recovery-unmatched","type":"command_execution","command":"bash tests/recovery-contracts.sh"}}' + fi + if [[ "$pattern_mode" == stagnation-recovery-read-positive ]]; then + recovery_read_output="$(cat "$workspace/RECOVERY.md")" + jq -cn --arg output "$recovery_read_output" '{type:"item.completed",item:{id:"stagnation-recovery-read",type:"command_execution",command:"cat RECOVERY.md",exit_code:0,status:"completed",aggregated_output:$output}}' + elif [[ "$pattern_mode" == stagnation-pre-fresh-shell-mutate-revert ]]; then + pre_fresh_probe="$workspace/pre-fresh-boundary.txt" + printf '%s\n' 'before' >"$pre_fresh_probe" + sed -i.bak 's/before/after/' "$pre_fresh_probe" + sed -i.bak 's/after/before/' "$pre_fresh_probe" + rm -f "$pre_fresh_probe" "$pre_fresh_probe.bak" + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-pre-fresh-shell","type":"command_execution","command":"sed -i.bak s/before/after/ pre-fresh-boundary.txt","exit_code":0,"status":"completed","aggregated_output":""}}' + elif [[ "$pattern_mode" == stagnation-pre-fresh-command-started-only ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-pre-fresh-command","type":"command_execution","command":"sed -i.bak s/before/after/ pre-fresh-boundary.txt"}}' + elif [[ "$pattern_mode" == stagnation-pre-fresh-file-change-started-only ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-pre-fresh-file-change","type":"file_change","changes":[{"path":"src/legacy.js","kind":"update"}]}}' + fi + if [[ "$pattern_mode" != stagnation-missing-recovery ]]; then + if [[ "$pattern_mode" == stagnation-failed-recovery-then-retry ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-recovery-failed","type":"command_execution","command":"bash tests/recovery-contracts.sh","exit_code":1,"status":"completed","aggregated_output":"RECOVERY_FAILED"}}' + fi + if [[ "$pattern_mode" == stagnation-paired-start-completion-positive ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-recovery","type":"command_execution","command":"bash tests/recovery-contracts.sh"}}' + elif [[ "$pattern_mode" == stagnation-duplicate-recovery-start ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-recovery","type":"command_execution","command":"bash tests/recovery-contracts.sh"}}' + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-recovery","type":"command_execution","command":"bash tests/recovery-contracts.sh"}}' + elif [[ "$pattern_mode" == stagnation-matched-recovery-id-wrong-command ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-recovery","type":"command_execution","command":"bash tests/stagnation-contracts.sh --after-recovery"}}' + fi + if recovery_output="$(bash "$workspace/tests/recovery-contracts.sh")"; then recovery_exit=0; else recovery_exit=$?; fi + recovery_id="stagnation-recovery" + [[ "$pattern_mode" != stagnation-failed-recovery-then-retry ]] || recovery_id="stagnation-recovery-retry" + printf '%s\n' "{\"type\":\"item.completed\",\"item\":{\"id\":\"${recovery_id}\",\"type\":\"command_execution\",\"command\":\"bash tests/recovery-contracts.sh\",\"exit_code\":${recovery_exit},\"status\":\"completed\",\"aggregated_output\":\"${recovery_output}\"}}" + if [[ "$pattern_mode" == stagnation-duplicate-recovery-completed-id ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-recovery","type":"command_execution","command":"bash tests/recovery-contracts.sh","exit_code":0,"status":"completed","aggregated_output":"RECOVERY_APPLIED"}}' + fi + if [[ "$pattern_mode" == stagnation-failed-fresh-then-retry ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-fresh-check-failed","type":"command_execution","command":"bash tests/stagnation-contracts.sh --after-recovery","exit_code":1,"status":"completed","aggregated_output":"STAGNATION_FRESH_CHECK_FAILED"}}' + fi + if [[ "$pattern_mode" == stagnation-paired-start-completion-positive ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-fresh-check","type":"command_execution","command":"bash tests/stagnation-contracts.sh --after-recovery"}}' + fi + if fresh_output="$(bash "$workspace/tests/stagnation-contracts.sh" --after-recovery)"; then fresh_exit=0; else fresh_exit=$?; fi + fresh_id="stagnation-fresh-check" + [[ "$pattern_mode" != stagnation-failed-fresh-then-retry ]] || fresh_id="stagnation-fresh-check-retry" + printf '%s\n' "{\"type\":\"item.completed\",\"item\":{\"id\":\"${fresh_id}\",\"type\":\"command_execution\",\"command\":\"bash tests/stagnation-contracts.sh --after-recovery\",\"exit_code\":${fresh_exit},\"status\":\"completed\",\"aggregated_output\":\"${fresh_output}\"}}" + fi + if [[ "$pattern_mode" == stagnation-retry-after-bound ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-forbidden-retry","type":"command_execution","command":"apply quick patch","exit_code":0,"status":"completed","aggregated_output":"patched"}}' + elif [[ "$pattern_mode" == stagnation-repeated-trusted-check ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-repeat","type":"command_execution","command":"bash tests/stagnation-contracts.sh","exit_code":1,"status":"completed","aggregated_output":"STAGNATION_TRUSTED_FAILURE"}}' + elif [[ "$pattern_mode" == stagnation-transient-source-change ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-source-change","type":"file_change","changes":[{"path":"src/legacy.js","kind":"update"}]}}' + elif [[ "$pattern_mode" == stagnation-duplicate-recovery ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-recovery-repeat","type":"command_execution","command":"bash tests/recovery-contracts.sh","exit_code":0,"status":"completed","aggregated_output":"RECOVERY_APPLIED"}}' + elif [[ "$pattern_mode" == stagnation-duplicate-fresh ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-fresh-repeat","type":"command_execution","command":"bash tests/stagnation-contracts.sh --after-recovery","exit_code":0,"status":"completed","aggregated_output":"STAGNATION_FRESH_CHECK_PASS"}}' + elif [[ "$pattern_mode" == stagnation-unknown-completed-action ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-unknown-action","type":"future_state_changing_tool","status":"completed"}}' + elif [[ "$pattern_mode" == stagnation-post-fresh-shell-mutate-revert ]]; then + boundary_probe="$workspace/.assistant-eval/post-fresh-boundary.txt" + printf '%s\n' 'before' >"$boundary_probe" + sed -i.bak 's/before/after/' "$boundary_probe" + sed -i.bak 's/after/before/' "$boundary_probe" + rm -f "$boundary_probe" "$boundary_probe.bak" + printf '%s\n' '{"type":"item.completed","item":{"id":"stagnation-post-fresh-shell","type":"command_execution","command":"sed -i.bak s/before/after/ .assistant-eval/post-fresh-boundary.txt","exit_code":0,"status":"completed","aggregated_output":""}}' + elif [[ "$pattern_mode" == stagnation-post-fresh-command-started-only ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-post-fresh-command","type":"command_execution","command":"sed -i.bak s/before/after/ .assistant-eval/post-fresh-boundary.txt"}}' + elif [[ "$pattern_mode" == stagnation-post-fresh-file-change-started-only ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-post-fresh-file-change","type":"file_change","changes":[{"path":"src/legacy.js","kind":"update"}]}}' + elif [[ "$pattern_mode" == stagnation-updated-unknown ]]; then + printf '%s\n' '{"type":"item.updated","item":{"id":"stagnation-updated-unknown","type":"future_item_type"}}' + elif [[ "$pattern_mode" == stagnation-updated-disallowed ]]; then + printf '%s\n' '{"type":"item.updated","item":{"id":"stagnation-updated-web","type":"web_search","query":"unrelated external lookup"}}' + fi + terminal_completed=false + [[ "$pattern_mode" != stagnation-false-completion ]] || terminal_completed=true + mkdir -p "$workspace/.assistant-eval" + printf '%s\n' "{\"schema_version\":\"1.0\",\"trusted_failure\":\"observed\",\"recovery\":\"applied\",\"fresh_check\":\"passed\",\"terminal_completed\":${terminal_completed}}" >"$workspace/.assistant-eval/stagnation-recovery.json" +fi if [[ -f "$workspace/VIEWING_PREPARATION.md" ]]; then viewing_event_mode=valid if [[ "$is_candidate_identity" == true ]]; then @@ -3260,16 +3407,36 @@ fi test_start "command-event bounds reject pathological shell input before expensive parsing" event_bounds_lib="$fixture_root/event-bounds-functions.sh" +awk ' + /^readonly EVENT_EVIDENCE_NORMALIZATION_JQ=/ { capture = 1 } + capture { print } + capture && /^\)"$/ { exit } +' "$runner" >>"$event_bounds_lib" awk ' /^validate_event_stream\(\)/ { capture = 1 } capture { print } capture && /^}$/ { exit } -' "$runner" >"$event_bounds_lib" +' "$runner" >>"$event_bounds_lib" awk ' /^viewing_inspection_event_evidence\(\)/ { capture = 1 } capture { print } capture && /^}$/ { exit } ' "$runner" >>"$event_bounds_lib" +awk ' + /^workspace_event_path_mappings\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" +awk ' + /^small_fix_event_evidence\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" +awk ' + /^stagnation_recovery_event_evidence\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" # shellcheck source=/dev/null source "$event_bounds_lib" @@ -3348,6 +3515,228 @@ else fail "command-event bounds or pre-regex pathological-input rejection regressed" fi +test_start "small-fix event paths accept only normalized single workspace source changes" +small_path_workspace="$fixture_root/small-path-controls-workspace" +mkdir -p "$small_path_workspace/docs" +small_path_workspace_alias="$fixture_root/small-path-controls-workspace-alias" +small_path_outside="$fixture_root/small-path-controls-outside" +small_path_outside_alias="$fixture_root/small-path-controls-outside-alias" +mkdir -p "$small_path_outside/docs" +ln -s "$small_path_workspace" "$small_path_workspace_alias" +ln -s "$small_path_outside" "$small_path_outside_alias" +small_path_failures=() +write_small_path_control() { + local mode="$1" jsonl="$2" path="$3" + + jq -cn '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >"$jsonl" + case "$mode" in + changes) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",changes:[{path:$path,kind:"update"}]}}' >>"$jsonl" + ;; + item_path) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",path:$path}}' >>"$jsonl" + ;; + combined) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",changes:[{path:$path,kind:"update"},{path:".assistant-eval/workflow-decision.json",kind:"add"}]}}' >>"$jsonl" + ;; + mixed) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",changes:[{path:$path,kind:"update"},{path:"/tmp/outside/usage.md",kind:"update"}]}}' >>"$jsonl" + ;; + missing) + jq -cn '{type:"item.completed",item:{id:"small-change",type:"file_change",changes:[]}}' >>"$jsonl" + ;; + esac +} +for small_path_positive in \ + "changes:docs/usage.md" \ + "changes:./docs/./usage.md" \ + "changes:$small_path_workspace/docs/usage.md" \ + "changes:$small_path_workspace_alias/docs/usage.md" \ + "item_path:docs/usage.md" \ + "combined:docs/usage.md"; do + small_path_mode="${small_path_positive%%:*}" + small_path_value="${small_path_positive#*:}" + small_path_jsonl="$fixture_root/small-path-positive-${small_path_mode}-${RANDOM}.jsonl" + write_small_path_control "$small_path_mode" "$small_path_jsonl" "$small_path_value" + if ! small_fix_event_evidence "$small_path_jsonl" "$small_path_workspace" \ + | jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0' >/dev/null; then + small_path_failures+=("positive:$small_path_positive") + fi +done +for small_path_negative in \ + "changes:/tmp/outside/usage.md" \ + "changes:$small_path_outside_alias/docs/usage.md" \ + "mixed:docs/usage.md" \ + "missing:"; do + small_path_mode="${small_path_negative%%:*}" + small_path_value="${small_path_negative#*:}" + small_path_jsonl="$fixture_root/small-path-negative-${small_path_mode}-${RANDOM}.jsonl" + write_small_path_control "$small_path_mode" "$small_path_jsonl" "$small_path_value" + if small_fix_event_evidence "$small_path_jsonl" "$small_path_workspace" \ + | jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0' >/dev/null; then + small_path_failures+=("negative:$small_path_negative") + fi +done +if [[ ${#small_path_failures[@]} -eq 0 ]]; then + pass +else + fail "small-fix event path normalization accepted or rejected an incorrect workspace boundary: ${small_path_failures[*]}" +fi + +test_start "small-fix discovery starts require one later matching completion" +small_lifecycle_workspace="$fixture_root/small-lifecycle-workspace" +mkdir -p "$small_lifecycle_workspace/docs" +write_small_lifecycle_control() { + local mode="$1" jsonl="$2" append_late_completion=false + + : >"$jsonl" + case "$mode" in + completion-only) + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + matching-pair) + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + missing-id) + jq -cn '{type:"item.started",item:{type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + mismatched-id) + jq -cn '{type:"item.started",item:{id:"discovery-start",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery-completion",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + duplicate-start) + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + duplicate-completion) + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + same-id-wrong-command) + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n typo docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + same-id-wrong-command-collision) + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n typo docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + ;; + reversed-completion) + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + ;; + post-edit-exit1) + jq -cn '{type:"item.started",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"change",type:"file_change",changes:[{path:"docs/usage.md",kind:"update"}]}}' >>"$jsonl" + jq -cn '{type:"item.started",item:{id:"verification",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"verification",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:1,status:"completed",aggregated_output:""}}' >>"$jsonl" + return + ;; + late-pre-edit-completion) + jq -cn '{type:"item.started",item:{id:"discovery-start",type:"command_execution",command:"rg -n teh docs/usage.md"}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"other-success",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + append_late_completion=true + ;; + esac + jq -cn '{type:"item.completed",item:{id:"change",type:"file_change",changes:[{path:"docs/usage.md",kind:"update"}]}}' >>"$jsonl" + if [[ "$append_late_completion" == true ]]; then + jq -cn '{type:"item.completed",item:{id:"discovery-start",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0,status:"completed",aggregated_output:"fixture"}}' >>"$jsonl" + fi +} +small_lifecycle_failures=() +for small_lifecycle_positive in completion-only matching-pair post-edit-exit1; do + small_lifecycle_jsonl="$fixture_root/small-lifecycle-$small_lifecycle_positive.jsonl" + write_small_lifecycle_control "$small_lifecycle_positive" "$small_lifecycle_jsonl" + if ! small_fix_event_evidence "$small_lifecycle_jsonl" "$small_lifecycle_workspace" \ + | jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0' >/dev/null; then + small_lifecycle_failures+=("positive:$small_lifecycle_positive") + fi +done +for small_lifecycle_negative in missing-id mismatched-id duplicate-start duplicate-completion same-id-wrong-command same-id-wrong-command-collision reversed-completion late-pre-edit-completion; do + small_lifecycle_jsonl="$fixture_root/small-lifecycle-$small_lifecycle_negative.jsonl" + write_small_lifecycle_control "$small_lifecycle_negative" "$small_lifecycle_jsonl" + if small_fix_event_evidence "$small_lifecycle_jsonl" "$small_lifecycle_workspace" \ + | jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0' >/dev/null; then + small_lifecycle_failures+=("negative:$small_lifecycle_negative") + fi +done +if [[ "${#small_lifecycle_failures[@]}" -eq 0 ]]; then + pass +else + fail "small-fix discovery lifecycle accepted malformed or rejected valid evidence: ${small_lifecycle_failures[*]}" +fi + +test_start "stagnation event paths accept only normalized recovery-artifact changes" +stagnation_path_workspace="$fixture_root/stagnation-path-controls-workspace" +mkdir -p "$stagnation_path_workspace/.assistant-eval" +stagnation_path_workspace_alias="$fixture_root/stagnation-path-controls-workspace-alias" +stagnation_path_outside="$fixture_root/stagnation-path-controls-outside" +stagnation_path_outside_alias="$fixture_root/stagnation-path-controls-outside-alias" +mkdir -p "$stagnation_path_outside/.assistant-eval" +ln -s "$stagnation_path_workspace" "$stagnation_path_workspace_alias" +ln -s "$stagnation_path_outside" "$stagnation_path_outside_alias" +stagnation_path_failures=() +write_stagnation_path_control() { + local mode="$1" jsonl="$2" path="$3" + + jq -cn '{type:"item.completed",item:{id:"stagnation-failure",type:"command_execution",command:"bash tests/stagnation-contracts.sh",exit_code:1,status:"completed",aggregated_output:"STAGNATION_TRUSTED_FAILURE"}}' >"$jsonl" + jq -cn '{type:"item.completed",item:{id:"stagnation-recovery",type:"command_execution",command:"bash tests/recovery-contracts.sh",exit_code:0,status:"completed",aggregated_output:"RECOVERY_APPLIED"}}' >>"$jsonl" + case "$mode" in + changes) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"stagnation-artifact",type:"file_change",changes:[{path:$path,kind:"add"}]}}' >>"$jsonl" + ;; + item_path) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"stagnation-artifact",type:"file_change",path:$path}}' >>"$jsonl" + ;; + mixed) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"stagnation-artifact",type:"file_change",changes:[{path:$path,kind:"add"},{path:"/tmp/outside/recovery.json",kind:"add"}]}}' >>"$jsonl" + ;; + missing) + jq -cn '{type:"item.completed",item:{id:"stagnation-artifact",type:"file_change",changes:[]}}' >>"$jsonl" + ;; + esac + jq -cn '{type:"item.completed",item:{id:"stagnation-fresh-check",type:"command_execution",command:"bash tests/stagnation-contracts.sh --after-recovery",exit_code:0,status:"completed",aggregated_output:"STAGNATION_FRESH_CHECK_PASS"}}' >>"$jsonl" +} +for stagnation_path_positive in \ + "changes:.assistant-eval/stagnation-recovery.json" \ + "changes:./.assistant-eval/./stagnation-recovery.json" \ + "changes:$stagnation_path_workspace/.assistant-eval/stagnation-recovery.json" \ + "changes:$stagnation_path_workspace_alias/.assistant-eval/stagnation-recovery.json" \ + "item_path:.assistant-eval/stagnation-recovery.json"; do + stagnation_path_mode="${stagnation_path_positive%%:*}" + stagnation_path_value="${stagnation_path_positive#*:}" + stagnation_path_jsonl="$fixture_root/stagnation-path-positive-${stagnation_path_mode}-${RANDOM}.jsonl" + write_stagnation_path_control "$stagnation_path_mode" "$stagnation_path_jsonl" "$stagnation_path_value" + if ! stagnation_recovery_event_evidence "$stagnation_path_jsonl" "$stagnation_path_workspace" \ + | jq -e '.trusted_failure_before_recovery == true and .recovery_before_fresh_check == true and .reject_patch_or_retry_after_bound == true and .terminal_completed == false' >/dev/null; then + stagnation_path_failures+=("positive:$stagnation_path_positive") + fi +done +for stagnation_path_negative in \ + "changes:/tmp/outside/recovery.json" \ + "changes:$stagnation_path_outside_alias/.assistant-eval/stagnation-recovery.json" \ + "mixed:.assistant-eval/stagnation-recovery.json" \ + "missing:"; do + stagnation_path_mode="${stagnation_path_negative%%:*}" + stagnation_path_value="${stagnation_path_negative#*:}" + stagnation_path_jsonl="$fixture_root/stagnation-path-negative-${stagnation_path_mode}-${RANDOM}.jsonl" + write_stagnation_path_control "$stagnation_path_mode" "$stagnation_path_jsonl" "$stagnation_path_value" + if stagnation_recovery_event_evidence "$stagnation_path_jsonl" "$stagnation_path_workspace" \ + | jq -e '.trusted_failure_before_recovery == true and .recovery_before_fresh_check == true and .reject_patch_or_retry_after_bound == true and .terminal_completed == false' >/dev/null; then + stagnation_path_failures+=("negative:$stagnation_path_negative") + fi +done +if [[ ${#stagnation_path_failures[@]} -eq 0 ]]; then + pass +else + fail "stagnation event path normalization accepted or rejected an incorrect workspace boundary: ${stagnation_path_failures[*]}" +fi + test_start "VIEWING evidence mutations are isolated to the candidate verifier" viewing_mutation_failures=() for viewing_mutation in omitted_ref stale_hashes bad_path bad_symbol bad_assertion bad_event_ref missing_event extra_top_key extra_evidence_key extra_item_key extra_design_key extra_implementation_key extra_trace_key extra_behavioral_test_key extra_result_key prepended_document appended_document; do @@ -5575,4 +5964,304 @@ else fail "trace schema does not expose the required behavioral provenance contract" fi +test_start "small-fix evaluation requires observed low-overhead evidence rather than its grading artifact alone" +small_positive_output="$fixture_root/small-observed-positive-output" +small_wrapped_positive_output="$fixture_root/small-observed-wrapped-positive-output" +small_artifact_only_output="$fixture_root/small-observed-artifact-only-output" +small_external_read_output="$fixture_root/small-observed-external-read-output" +small_unsupported_output="$fixture_root/small-observed-unsupported-output" +small_symlink_output="$fixture_root/small-observed-symlink-output" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-positive "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_positive_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$small_positive_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-wrapped-positive "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_wrapped_positive_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$small_wrapped_positive_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-artifact-only "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_artifact_only_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$small_artifact_only_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-disallowed-tool "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_external_read_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$small_external_read_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-unsupported-shape "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_unsupported_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "adapter_unavailable" and .error.code == "unknown_event_shape" and (has("metrics") | not))' "$small_unsupported_output/traces/"*.json >/dev/null; then + pass +else + fail "small-fix evaluation did not require discovery-before-change or reject artifact-only and explicit external-read fixture evidence" +fi + +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-external-symlink "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_symlink_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .metrics.acceptance_passed == false)' "$small_symlink_output/traces/"*.json >/dev/null; then + pass +else + fail "small-fix evaluation accepted an external symlink target as the workspace edit" +fi + +test_start "small-fix evaluation rejects started external calls without a completion" +small_started_external_failures=() +for small_started_external_mode in small-mcp-started-only small-web-started-only; do + small_started_external_output="$fixture_root/$small_started_external_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$small_started_external_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_started_external_output" --codex-bin "$fake_codex" >/dev/null \ + || jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$small_started_external_output/traces/"*.json >/dev/null; then + small_started_external_failures+=("$small_started_external_mode") + fi +done +if [[ ${#small_started_external_failures[@]} -eq 0 ]]; then + pass +else + fail "small-fix evaluation accepted a started external call: ${small_started_external_failures[*]}" +fi + +test_start "small-fix evaluation treats updated unknown and disallowed actions as unavailable" +small_updated_failures=() +for small_updated_mode in small-updated-unknown small-updated-disallowed; do + small_updated_output="$fixture_root/$small_updated_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$small_updated_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_updated_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e 'length == 2 and all(.[]; .status == "adapter_unavailable" and (has("metrics") | not) and .error.code == "unknown_event_shape")' "$small_updated_output/traces/"*.json >/dev/null \ + || ! jq -e '.complete_pairs == 0 and .excluded_incomplete_pairs == 1 and .incomplete_pairs[0].case_id == "small-fix-stays-lightweight"' "$small_updated_output/comparison.json" >/dev/null; then + small_updated_failures+=("$small_updated_mode") + fi +done +if [[ ${#small_updated_failures[@]} -eq 0 ]]; then + pass +else + fail "small-fix evaluation accepted updated unknown or disallowed evidence: ${small_updated_failures[*]}" +fi + +test_start "small-fix evaluation rejects completed and overlapping shell actions before exact discovery" +small_pre_discovery_failures=() +for small_pre_discovery_mode in small-shell-edit-before-discovery small-shell-edit-started-before-discovery small-interleaved-shell-action; do + small_pre_discovery_output="$fixture_root/$small_pre_discovery_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$small_pre_discovery_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_pre_discovery_output" --codex-bin "$fake_codex" >/dev/null \ + || jq -s -e 'all(.[]; .metrics.acceptance_passed == true)' "$small_pre_discovery_output/traces/"*.json >/dev/null; then + small_pre_discovery_failures+=("$small_pre_discovery_mode") + fi +done +if [[ ${#small_pre_discovery_failures[@]} -eq 0 ]]; then + pass +else + fail "small-fix evaluation accepted an action before exact discovery: ${small_pre_discovery_failures[*]}" +fi + +test_start "stagnation evaluation requires trusted failure, recovery, fresh check, and terminal incomplete state" +stagnation_positive_output="$fixture_root/stagnation-observed-positive-output" +stagnation_missing_recovery_output="$fixture_root/stagnation-missing-recovery-output" +stagnation_retry_output="$fixture_root/stagnation-retry-output" +stagnation_repeated_check_output="$fixture_root/stagnation-repeated-check-output" +stagnation_source_change_output="$fixture_root/stagnation-source-change-output" +stagnation_false_completion_output="$fixture_root/stagnation-false-completion-output" +stagnation_duplicate_recovery_output="$fixture_root/stagnation-duplicate-recovery-output" +stagnation_duplicate_fresh_output="$fixture_root/stagnation-duplicate-fresh-output" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-positive "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_positive_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$stagnation_positive_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-missing-recovery "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_missing_recovery_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$stagnation_missing_recovery_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-retry-after-bound "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_retry_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$stagnation_retry_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-repeated-trusted-check "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_repeated_check_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$stagnation_repeated_check_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-transient-source-change "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_source_change_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$stagnation_source_change_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-false-completion "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_false_completion_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$stagnation_false_completion_output/traces/"*.json >/dev/null; then + pass +else + fail "stagnation evaluation did not verify trusted failure, recovery, fresh check, and terminal incompleteness against bounded mutations" +fi + +test_start "stagnation evaluation permits the declared read-only recovery probe" +stagnation_recovery_read_output="$fixture_root/stagnation-recovery-read-output" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-recovery-read-positive "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_recovery_read_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$stagnation_recovery_read_output/traces/"*.json >/dev/null; then + pass +else + fail "stagnation evaluation rejected its declared read-only recovery probe" +fi + +test_start "stagnation evaluation permits paired admitted starts and trusted completions" +stagnation_paired_lifecycle_output="$fixture_root/stagnation-paired-lifecycle-output" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-paired-start-completion-positive "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_paired_lifecycle_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$stagnation_paired_lifecycle_output/traces/"*.json >/dev/null; then + pass +else + fail "stagnation evaluation rejected paired admitted starts with trusted completions" +fi + +test_start "stagnation evaluation rejects every pre-fresh foreign workspace action" +stagnation_pre_fresh_rejection_failures=() +for stagnation_pre_fresh_mode in stagnation-pre-fresh-shell-mutate-revert stagnation-pre-fresh-command-started-only stagnation-pre-fresh-file-change-started-only; do + stagnation_pre_fresh_output="$fixture_root/$stagnation_pre_fresh_mode-output" + rm -f "$capture"/* + if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$stagnation_pre_fresh_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_pre_fresh_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$stagnation_pre_fresh_output/traces/"*.json >/dev/null; then + : + else + stagnation_pre_fresh_rejection_failures+=("$stagnation_pre_fresh_mode") + fi +done +if [[ ${#stagnation_pre_fresh_rejection_failures[@]} -eq 0 ]]; then + pass +else + fail "stagnation evaluation accepted a pre-fresh foreign workspace action: ${stagnation_pre_fresh_rejection_failures[*]}" +fi + +test_start "stagnation evaluation rejects uncorrelated, duplicate, mismatched, or early admitted command starts" +stagnation_lifecycle_rejection_failures=() +stagnation_lifecycle_modes=(stagnation-early-recovery-start stagnation-early-fresh-start stagnation-unmatched-recovery-start stagnation-duplicate-recovery-start stagnation-matched-recovery-id-wrong-command stagnation-duplicate-recovery-completed-id) +if [[ -n "${P0P4_STAGNATION_LIFECYCLE_MODES:-}" ]]; then + read -r -a stagnation_lifecycle_modes <<< "$P0P4_STAGNATION_LIFECYCLE_MODES" +fi +for stagnation_lifecycle_mode in "${stagnation_lifecycle_modes[@]}"; do + stagnation_lifecycle_output="$fixture_root/$stagnation_lifecycle_mode-output" + rm -f "$capture"/* + if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$stagnation_lifecycle_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_lifecycle_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$stagnation_lifecycle_output/traces/"*.json >/dev/null; then + : + else + stagnation_lifecycle_rejection_failures+=("$stagnation_lifecycle_mode") + fi +done +if [[ ${#stagnation_lifecycle_rejection_failures[@]} -eq 0 ]]; then + pass +else + fail "stagnation evaluation accepted early, unmatched, duplicate, or mismatched admitted command lifecycle evidence: ${stagnation_lifecycle_rejection_failures[*]}" +fi + +for stagnation_duplicate_mode in stagnation-duplicate-recovery stagnation-duplicate-fresh stagnation-failed-recovery-then-retry stagnation-failed-fresh-then-retry; do + stagnation_duplicate_output="$fixture_root/$stagnation_duplicate_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$stagnation_duplicate_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_duplicate_output" --codex-bin "$fake_codex" >/dev/null \ + || jq -s -e 'all(.[]; .metrics.acceptance_passed == true)' "$stagnation_duplicate_output/traces/"*.json >/dev/null; then + fail "stagnation evaluation accepted duplicate recovery or fresh-check invocation: $stagnation_duplicate_mode" + fi +done + +test_start "stagnation evaluation rejects every post-fresh workspace action" +stagnation_post_fresh_failures=() +for stagnation_post_fresh_mode in stagnation-post-fresh-shell-mutate-revert stagnation-post-fresh-command-started-only stagnation-post-fresh-file-change-started-only; do + stagnation_post_fresh_output="$fixture_root/$stagnation_post_fresh_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$stagnation_post_fresh_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_post_fresh_output" --codex-bin "$fake_codex" >/dev/null \ + || jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$stagnation_post_fresh_output/traces/"*.json >/dev/null; then + stagnation_post_fresh_failures+=("$stagnation_post_fresh_mode") + fi +done +if [[ ${#stagnation_post_fresh_failures[@]} -eq 0 ]]; then + pass +else + fail "stagnation evaluation accepted a post-fresh workspace action: ${stagnation_post_fresh_failures[*]}" +fi + +test_start "stagnation evaluation rejects unknown completed actions as adapter-unavailable" +stagnation_unknown_action_output="$fixture_root/stagnation-unknown-action-output" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-unknown-completed-action "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_unknown_action_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'length == 2 and all(.[]; .status == "adapter_unavailable" and (has("metrics") | not) and .error.code == "unknown_event_shape")' "$stagnation_unknown_action_output/traces/"*.json >/dev/null \ + && jq -e '.complete_pairs == 0 and .excluded_incomplete_pairs == 1 and .incomplete_pairs[0].case_id == "pivot-restart-on-stagnation-or-code-writer-blocker"' "$stagnation_unknown_action_output/comparison.json" >/dev/null; then + pass +else + fail "stagnation evaluation accepted an unknown completed action or promoted its pair" +fi + +test_start "stagnation evaluation treats updated unknown and disallowed actions as unavailable" +stagnation_updated_failures=() +for stagnation_updated_mode in stagnation-updated-unknown stagnation-updated-disallowed; do + stagnation_updated_output="$fixture_root/$stagnation_updated_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$stagnation_updated_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_updated_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e 'length == 2 and all(.[]; .status == "adapter_unavailable" and (has("metrics") | not) and .error.code == "unknown_event_shape")' "$stagnation_updated_output/traces/"*.json >/dev/null \ + || ! jq -e '.complete_pairs == 0 and .excluded_incomplete_pairs == 1 and .incomplete_pairs[0].case_id == "pivot-restart-on-stagnation-or-code-writer-blocker"' "$stagnation_updated_output/comparison.json" >/dev/null; then + stagnation_updated_failures+=("$stagnation_updated_mode") + fi +done +if [[ ${#stagnation_updated_failures[@]} -eq 0 ]]; then + pass +else + fail "stagnation evaluation accepted updated unknown or disallowed evidence: ${stagnation_updated_failures[*]}" +fi + +test_start "isolated A/B/C parallel case remains adapter-unavailable and cannot promote narrative as native evidence" +parallel_unavailable_output="$fixture_root/parallel-unavailable-output" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases isolated-parallel-a-b-then-c-with-integration --repeats 1 \ + --output "$parallel_unavailable_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'length == 2 and all(.[]; + .case_id == "isolated-parallel-a-b-then-c-with-integration" + and .status == "adapter_unavailable" + and (has("metrics") | not) + and .error.code == "unknown_event_shape") + ' "$parallel_unavailable_output/traces/"*.json >/dev/null \ + && jq -e ' + .complete_pairs == 0 + and .excluded_incomplete_pairs == 1 + and .incomplete_pairs[0].case_id == "isolated-parallel-a-b-then-c-with-integration" + ' "$parallel_unavailable_output/comparison.json" >/dev/null \ + && [[ "$(find "$capture" -maxdepth 1 -name 'call-*.args' | wc -l | tr -d ' ')" -eq 0 ]] \ + && jq -s -e 'length == 2 and all(.[]; .state == "completed" and (.attempt_started_at | type == "array" and length == 0))' "$parallel_unavailable_output/run-attempts/"*.json >/dev/null; then + pass +else + fail "isolated parallel evaluation accepted model narrative or promoted unavailable overlap telemetry" +fi + p0p4_finish_suite "${BASH_SOURCE[0]}" diff --git a/tests/p0-p4/feature-preparation-evidence-contracts.sh b/tests/p0-p4/feature-preparation-evidence-contracts.sh index a23742b..790f72f 100644 --- a/tests/p0-p4/feature-preparation-evidence-contracts.sh +++ b/tests/p0-p4/feature-preparation-evidence-contracts.sh @@ -620,7 +620,7 @@ if ruby -ryaml -e ' phases.include?("**Run condition:** `execution_intent != prepare_only`.") && plan.include?("For `execution_intent=prepare_only`, an explicitly requested readiness Plan is inline and never waits.") && plan.include?("It omits Artifact Contracts, executable task packets, slice manifests, and implementation tests.") && - journal.include?("[required for medium+ tasks; update after each slice before starting the next]") && + journal.include?("[required for medium+ tasks; update after each slice and before starting a dependent slice or another source-changing slice in a shared or unknown workspace]") && journal.include?("[applies only when `execution_intent != prepare_only`; prepare_only has no slices]") && journal.include?("**Preparation Completion** (`execution_intent=prepare_only`) records readiness only, then proceeds directly to Done without Build, Review, or developer handoff.") && roles.include?("## Dispatch rules by task size (`execution_intent != prepare_only`)") && diff --git a/tests/p0-p4/task-packet-contracts.sh b/tests/p0-p4/task-packet-contracts.sh index 7e408b1..2903717 100644 --- a/tests/p0-p4/task-packet-contracts.sh +++ b/tests/p0-p4/task-packet-contracts.sh @@ -60,18 +60,19 @@ else fail "phase-gates.yaml missing executable task packet gates: ${missing_phase_gate_terms[*]}" fi -test_start "workflow build worker protocol enforces medium slice verification loop" +test_start "workflow build worker protocol enforces dependency-aware slice verification" missing_slice_phase_terms=() build_worker_ref="$FRAMEWORK_DIR/skills/assistant-workflow/references/build-worker-protocol.md" for term in \ - "For medium+ tasks with slices, execute one slice at a time" \ + "source-changing slices sequentially in a shared or unknown workspace" \ + "Independently executable source-changing slices may overlap only when runtime evidence proves isolated workspaces" \ "Load the approved task packet for the slice, including slice_id, observable increment, deliverable type, files, acceptance criteria, verification command, expected success signal, evidence to record, and deviation/rollback rule" \ - "Confirm prior slice status is \`VERIFIED\` before advancing" \ + "Confirm every \`depends_on\` prerequisite has final status \`VERIFIED\` before starting a dependent slice" \ "Check each acceptance criterion from the slice manifest independently" \ "Record verification evidence in the task journal slice verification ledger" \ "Run a small self-check/local sanity check" \ "Mark the slice \`VERIFIED\` only after all criteria pass and evidence is recorded" \ - "Only proceed to the next slice after the current one is fully verified"; do + "After all slices are integrated, run cross-slice and full-scope validation before entering fresh Review"; do if ! p0p4_contains_text "$build_worker_ref" "$term"; then missing_slice_phase_terms+=("$term") fi @@ -82,19 +83,19 @@ fi if [[ "${#missing_slice_phase_terms[@]}" -eq 0 ]]; then pass else - fail "build-worker-protocol.md missing per-slice verification loop terms: ${missing_slice_phase_terms[*]}" + fail "build-worker-protocol.md missing dependency-aware slice verification terms: ${missing_slice_phase_terms[*]}" fi test_start "workflow task journal template includes slice verification ledger fields" missing_slice_ledger_terms=() for term in \ "## Slice Verification Ledger" \ - "[required for medium+ tasks; update after each slice before starting the next]" \ + "[required for medium+ tasks; update after each slice and before starting a dependent slice or another source-changing slice in a shared or unknown workspace]" \ "| Slice | Task Packet | RED Status | Implementation Status | Verification Command/Result | Criteria Checked | Self-Check Result | Final Status |" \ "[X/Y passed]" \ "[pass/fail + note]" \ "[VERIFIED/BLOCKED]" \ - "do not start the next slice until the current one is \`VERIFIED\`"; do + "do not start a dependent slice until every \`depends_on\` prerequisite is \`VERIFIED\`; source-changing slices may overlap only with runtime-proven isolated workspaces"; do if ! grep -Fq -- "$term" "$FRAMEWORK_DIR/skills/assistant-workflow/references/task-journal-template.md"; then missing_slice_ledger_terms+=("$term") fi @@ -201,7 +202,7 @@ else pass fi -test_start "source-changing slice packets stay sequential in shared or unknown workspaces" +test_start "source-changing slice packets distinguish shared sequencing from isolated overlap" workspace_isolation_failures=() for file_and_term in \ "$FRAMEWORK_DIR/skills/assistant-workflow/references/sub-task-brief-template.md::shared or unknown workspace" \ @@ -217,13 +218,45 @@ for file_and_term in \ done if ! ruby -rjson -e ' cases = JSON.parse(File.read(ARGV.fetch(0))).fetch("cases") - item = cases.find { |entry| entry["id"] == "native-slice-execution-uses-dependencies-not-runner-topology" } - expected = item.fetch("expected_behavior").join(" ") - failures = item.fetch("fail_signals").join(" ") - valid = expected.include?("Sequences source-changing A and B because the workspace is shared or isolation is unknown") && - expected.include?("read-only analysis in parallel") && - expected.include?("runtime-proven isolated workspaces") && - failures.include?("parallel source-changing A/B in a shared or unknown workspace") + shared = cases.find { |entry| entry["id"] == "native-slice-execution-uses-dependencies-not-runner-topology" } + isolated = cases.find { |entry| entry["id"] == "isolated-independent-slices-integrate-before-review" } + shared_expected = shared.fetch("expected_behavior").join(" ") + shared_failures = shared.fetch("fail_signals").join(" ") + isolated_expected = isolated.fetch("expected_behavior").join(" ") + isolated_setup = isolated.fetch("setup_context").join(" ") + isolated_failures = isolated.fetch("fail_signals").join(" ") + expected_decisions = [ + {"a_status" => "PENDING", "c_decision" => "blocked"}, + {"a_status" => "RUNNING", "c_decision" => "blocked"}, + {"a_status" => "VERIFIED", "c_decision" => "ready"} + ] + structured = lambda { |entry| entry.fetch("machine_expectations").fetch("structured_json_assertions") } + equals = lambda do |assertions, path, expected| + assertions.any? { |assertion| assertion["operator"] == "equals" && assertion["path"] == path && assertion["expected"] == expected } + end + decisions = lambda do |assertions| + assertions.any? do |assertion| + assertion["operator"] == "array_object_values_exact" && assertion["path"] == ["execution_policy", "c_start_decisions"] && + assertion["fields"] == ["a_status", "c_decision"] && assertion["expected_objects"] == expected_decisions + end + end + shared_assertions = structured.call(shared) + isolated_assertions = structured.call(isolated) + valid = shared_expected.include?("Sequences source-changing A and B because the workspace is shared or isolation is unknown") && + shared_expected.include?("read-only analysis in parallel") && + shared_failures.include?("parallel source-changing A/B in a shared or unknown workspace") && + isolated_expected.include?("runtime-proven isolated workspaces") && + isolated_setup.include?("isolation_evidence_ref=fixture-runtime-isolation-A-B-v1") && + isolated_expected.include?("A and B may overlap") && + isolated_expected.include?("C blocked until A is VERIFIED") && + isolated_expected.include?("cross-slice and full-scope validation") && + isolated_expected.include?("fresh review") && + isolated_failures.include?("Starts C before A is VERIFIED") && + equals.call(shared_assertions, ["execution_policy", "source_writer_policy"], "sequential_shared_or_unknown") && + equals.call(shared_assertions, ["execution_policy", "isolation_evidence_ref"], "not_available") && + equals.call(isolated_assertions, ["execution_policy", "source_writer_policy"], "isolated_A_B_overlap_permitted") && + equals.call(isolated_assertions, ["execution_policy", "isolation_evidence_ref"], "fixture-runtime-isolation-A-B-v1") && + decisions.call(shared_assertions) && decisions.call(isolated_assertions) exit(valid ? 0 : 1) ' "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json"; then workspace_isolation_failures+=("workflow eval does not distinguish shared/unknown sequential, isolated parallel, and read-only parallel boundaries") @@ -238,6 +271,155 @@ else fail "workspace isolation routing is incomplete: ${workspace_isolation_failures[*]}" fi +test_start "task-only policy packets omit grading oracles while annotated packets retain legacy rendering" +policy_packet_root="$(mktemp -d "${TMPDIR:-/tmp}/workflow-eval-policy-packets.XXXXXX")" +policy_packet_explicit_skill="$policy_packet_root/assistant-workflow" +policy_packet_invalid_skill="$policy_packet_root/invalid-assistant-workflow" +p0p4_register_cleanup "$policy_packet_root" +mkdir -p "$policy_packet_explicit_skill/evals" "$policy_packet_invalid_skill/evals" +cp "$FRAMEWORK_DIR/skills/assistant-workflow/SKILL.md" "$policy_packet_explicit_skill/SKILL.md" +cp "$FRAMEWORK_DIR/skills/assistant-workflow/SKILL.md" "$policy_packet_invalid_skill/SKILL.md" +ln -s "$FRAMEWORK_DIR/skills/assistant-workflow/contracts" "$policy_packet_explicit_skill/contracts" +ln -s "$FRAMEWORK_DIR/skills/assistant-workflow/contracts" "$policy_packet_invalid_skill/contracts" +jq '(.cases[] | select(.id == "medium-task-plans-before-build") | .prompt_packet_mode) = "annotated"' \ + "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json" >"$policy_packet_explicit_skill/evals/cases.json" +policy_packet_failures=() +for policy_case in native-slice-execution-uses-dependencies-not-runner-topology isolated-independent-slices-integrate-before-review; do + policy_packet_output="$policy_packet_root/$policy_case" + if ! "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --emit-prompts "$policy_packet_output" --skill assistant-workflow --case "$policy_case" >/dev/null; then + policy_packet_failures+=("$policy_case:emit") + continue + fi + policy_packet_path="$policy_packet_output/assistant-workflow/$policy_case.md" + policy_packet_expected="$(jq -r --arg id "$policy_case" ' + def bullets($items): + if ($items | length) > 0 then $items | map("- " + .) | join("\n") + else "- (none)" end; + .cases[] | select(.id == $id) + | "# Task Packet\n\n" + + "Skill: assistant-workflow\n\n" + + "Skill Path: skills/assistant-workflow/SKILL.md\n\n" + + "## Setup Context\n\n" + bullets(.setup_context) + "\n\n" + + "## Prompt\n\n" + .prompt + "\n" + ' "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json")" + if [[ "$(<"$policy_packet_path")" != "$policy_packet_expected" ]]; then + policy_packet_failures+=("$policy_case:task-only-content") + fi +done +policy_packet_default_output="$policy_packet_root/default" +policy_packet_explicit_output="$policy_packet_root/explicit" +if ! "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --emit-prompts "$policy_packet_default_output" --skill assistant-workflow --case medium-task-plans-before-build >/dev/null \ + || ! "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --emit-prompts "$policy_packet_explicit_output" --skill "$policy_packet_explicit_skill" --case medium-task-plans-before-build >/dev/null \ + || ! cmp -s \ + <(sed 's|^Skill Path: .*|Skill Path: |' "$policy_packet_default_output/assistant-workflow/medium-task-plans-before-build.md") \ + <(sed 's|^Skill Path: .*|Skill Path: |' "$policy_packet_explicit_output/assistant-workflow/medium-task-plans-before-build.md"); then + policy_packet_failures+=("annotated-default-or-explicit") +fi +for invalid_mode in '"task-only"' 'null' '""'; do + jq --argjson invalid_mode "$invalid_mode" '(.cases[] | select(.id == "medium-task-plans-before-build") | .prompt_packet_mode) = $invalid_mode' \ + "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json" >"$policy_packet_invalid_skill/evals/cases.json" + if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --validate-fixture --skill "$policy_packet_invalid_skill" >/dev/null 2>&1; then + policy_packet_failures+=("invalid-mode:$invalid_mode") + fi +done +if [[ "${#policy_packet_failures[@]}" -eq 0 ]]; then + pass +else + fail "policy prompt packet rendering or mode validation is incorrect: ${policy_packet_failures[*]}" +fi + +test_start "workflow eval grading enforces structured dependency scheduling and seeded isolation evidence" +workflow_eval_root="$(mktemp -d "${TMPDIR:-/tmp}/workflow-eval-grading.XXXXXX")" +p0p4_register_cleanup "$workflow_eval_root" +workflow_eval_responses="$workflow_eval_root/positive" +workflow_eval_unsafe_responses="$workflow_eval_root/unsafe" +workflow_eval_missing_seed_responses="$workflow_eval_root/missing-seed" +workflow_eval_wrong_writer_responses="$workflow_eval_root/wrong-writer" +workflow_eval_wrong_integration_responses="$workflow_eval_root/wrong-integration" +workflow_eval_wrong_review_responses="$workflow_eval_root/wrong-review" +mkdir -p "$workflow_eval_responses/assistant-workflow" "$workflow_eval_unsafe_responses/assistant-workflow" "$workflow_eval_missing_seed_responses/assistant-workflow" "$workflow_eval_wrong_writer_responses/assistant-workflow" "$workflow_eval_wrong_integration_responses/assistant-workflow" "$workflow_eval_wrong_review_responses/assistant-workflow" +cat >"$workflow_eval_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"sequential_shared_or_unknown","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"not_available","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$workflow_eval_responses/assistant-workflow/isolated-independent-slices-integrate-before-review.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"fixture-runtime-isolation-A-B-v1","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$workflow_eval_unsafe_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"sequential_shared_or_unknown","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"not_available","c_start_decisions":[{"a_status":"PENDING","c_decision":"ready"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$workflow_eval_unsafe_responses/assistant-workflow/isolated-independent-slices-integrate-before-review.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"fixture-runtime-isolation-A-B-v1","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"ready"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$workflow_eval_missing_seed_responses/assistant-workflow/isolated-independent-slices-integrate-before-review.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"not_available","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$workflow_eval_wrong_writer_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"not_available","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$workflow_eval_wrong_integration_responses/assistant-workflow/isolated-independent-slices-integrate-before-review.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"fixture-runtime-isolation-A-B-v1","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"per_slice_only","integration_checks":["per-slice"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$workflow_eval_wrong_review_responses/assistant-workflow/isolated-independent-slices-integrate-before-review.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"fixture-runtime-isolation-A-B-v1","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"per_slice_verification"}} +EOF +workflow_eval_grading_failures=() +for workflow_case in native-slice-execution-uses-dependencies-not-runner-topology isolated-independent-slices-integrate-before-review; do + if ! "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --responses "$workflow_eval_responses" --skill assistant-workflow --case "$workflow_case" >"$workflow_eval_root/$workflow_case-positive.out"; then + workflow_eval_grading_failures+=("$workflow_case:positive") + fi + if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --responses "$workflow_eval_unsafe_responses" --skill assistant-workflow --case "$workflow_case" >"$workflow_eval_root/$workflow_case-unsafe.out" 2>&1 \ + || ! grep -Fq -- "structured JSON assertion failure" "$workflow_eval_root/$workflow_case-unsafe.out"; then + workflow_eval_grading_failures+=("$workflow_case:unsafe-dependent-start") + fi +done +if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --responses "$workflow_eval_missing_seed_responses" --skill assistant-workflow --case isolated-independent-slices-integrate-before-review >"$workflow_eval_root/isolated-missing-seed.out" 2>&1 \ + || ! grep -Fq -- "structured JSON assertion failure" "$workflow_eval_root/isolated-missing-seed.out"; then + workflow_eval_grading_failures+=("isolated-independent-slices-integrate-before-review:missing-seeded-isolation-evidence") +fi +for policy_negative in \ + "wrong-writer:$workflow_eval_wrong_writer_responses:native-slice-execution-uses-dependencies-not-runner-topology" \ + "wrong-integration:$workflow_eval_wrong_integration_responses:isolated-independent-slices-integrate-before-review" \ + "wrong-review:$workflow_eval_wrong_review_responses:isolated-independent-slices-integrate-before-review"; do + IFS=':' read -r policy_negative_name policy_negative_responses policy_negative_case <<< "$policy_negative" + if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --responses "$policy_negative_responses" --skill assistant-workflow --case "$policy_negative_case" >"$workflow_eval_root/$policy_negative_name.out" 2>&1 \ + || ! grep -Fq -- "structured JSON assertion failure" "$workflow_eval_root/$policy_negative_name.out"; then + workflow_eval_grading_failures+=("$policy_negative_name") + fi +done +if [[ "${#workflow_eval_grading_failures[@]}" -eq 0 ]]; then + pass +else + fail "workflow eval grading accepted unsafe dependency scheduling or missing seeded isolation evidence: ${workflow_eval_grading_failures[*]}" +fi + +test_start "workflow fixture schema rejects undeclared execution-policy children and invalid decisions" +workflow_policy_schema_root="$(mktemp -d "${TMPDIR:-/tmp}/workflow-eval-policy-schema.XXXXXX")" +workflow_policy_schema_skill="$workflow_policy_schema_root/assistant-workflow" +workflow_policy_schema_err="$workflow_policy_schema_root/validation.err" +p0p4_register_cleanup "$workflow_policy_schema_root" +mkdir -p "$workflow_policy_schema_skill/evals" +cp "$FRAMEWORK_DIR/skills/assistant-workflow/SKILL.md" "$workflow_policy_schema_skill/SKILL.md" +ln -s "$FRAMEWORK_DIR/skills/assistant-workflow/contracts" "$workflow_policy_schema_skill/contracts" +workflow_policy_schema_failures=() +if ! "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --validate-fixture --skill assistant-workflow >/dev/null; then + workflow_policy_schema_failures+=("valid-fixture") +fi +jq '(.cases[] | select(.id == "native-slice-execution-uses-dependencies-not-runner-topology") | .machine_expectations.structured_json_assertions) += [{"operator":"equals","path":["execution_policy","invented_child"],"expected":"x"}]' "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json" >"$workflow_policy_schema_skill/evals/cases.json" +if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --validate-fixture --skill "$workflow_policy_schema_skill" >/dev/null 2>"$workflow_policy_schema_err" \ + || ! grep -Fq -- "undeclared assertion path" "$workflow_policy_schema_err"; then + workflow_policy_schema_failures+=("undeclared-child") +fi +jq '(.cases[] | select(.id == "native-slice-execution-uses-dependencies-not-runner-topology") | .machine_expectations.structured_json_assertions) |= map(if .operator == "array_object_values_exact" and .path == ["execution_policy", "c_start_decisions"] then .expected_objects[0].c_decision = "unsafe" else . end)' "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json" >"$workflow_policy_schema_skill/evals/cases.json" +if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --validate-fixture --skill "$workflow_policy_schema_skill" >/dev/null 2>"$workflow_policy_schema_err" \ + || ! grep -Fq -- "assertion literal outside contract schema" "$workflow_policy_schema_err"; then + workflow_policy_schema_failures+=("invalid-decision-enum") +fi +if [[ "${#workflow_policy_schema_failures[@]}" -eq 0 ]]; then + pass +else + fail "workflow fixture schema accepted invalid execution-policy structure: ${workflow_policy_schema_failures[*]}" +fi + test_start "workflow slice identities are descriptive while ordering stays display-only" missing_descriptive_slice_terms=() for skill_root in "$FRAMEWORK_DIR/skills/assistant-workflow"; do @@ -618,15 +800,18 @@ else fail "broad-split rejection proof contract missing terms: ${missing_broad_split_review_terms[*]}" fi -test_start "workflow phase gates require recorded slice evidence before advancing" +test_start "workflow phase gates require dependency-aware recorded slice evidence" missing_slice_gate_terms=() for term in \ "- id: B12" \ "independently checked, passing, and recorded with command/result evidence in the task journal, validation_results, or equivalent carried-forward slice ledger" \ "record command/result evidence in the configured task journal or equivalent carried-forward state" \ "- id: B13" \ - "each slice has a final status of VERIFIED, including self-check result, before the next slice started" \ - "slices must be verified sequentially with evidence before advancing"; do + "After all slices are integrated, cross-slice and full-scope validation are complete before Review" \ + "shared or unknown workspace are VERIFIED, including self-check result, before another source-changing slice starts" \ + "runtime evidence proves isolated workspaces" \ + "every depends_on prerequisite has final status VERIFIED before a dependent slice starts" \ + "cross-slice and full-scope validation are complete before Review"; do if ! grep -Fq -- "$term" "$FRAMEWORK_DIR/skills/assistant-workflow/contracts/phase-gates.yaml"; then missing_slice_gate_terms+=("$term") fi @@ -634,7 +819,20 @@ done if [[ "${#missing_slice_gate_terms[@]}" -eq 0 ]]; then pass else - fail "phase-gates.yaml missing recorded/sequential slice verification gate terms: ${missing_slice_gate_terms[*]}" + fail "phase-gates.yaml missing dependency-aware slice verification gate terms: ${missing_slice_gate_terms[*]}" +fi + +test_start "Decompose guidance permits isolated overlap while rejecting stale next-slice sequencing" +decompose_source="$FRAMEWORK_DIR/skills/assistant-workflow/references/phases.md" +decompose_view="$FRAMEWORK_DIR/skills/assistant-workflow/references/phases/decompose.md" +if grep -Fq "every \`depends_on\` prerequisite is \`VERIFIED\` before starting a dependent slice" "$decompose_source" \ + && grep -Fq "integrated output is validated before Review" "$decompose_source" \ + && grep -Fq "every \`depends_on\` prerequisite is \`VERIFIED\` before starting a dependent slice" "$decompose_view" \ + && ! grep -Fq "every slice is verified before moving to the next slice" "$decompose_source" \ + && ! grep -Fq "every slice is verified before moving to the next slice" "$decompose_view"; then + pass +else + fail "Decompose source or generated view still imposes stale unconditional next-slice sequencing" fi test_start "workflow handoffs pass current task packets to CodeWriter and BuilderTester" diff --git a/tools/evals/lib/skill-eval-fixtures.sh b/tools/evals/lib/skill-eval-fixtures.sh index 13bb67a..c07de17 100644 --- a/tools/evals/lib/skill-eval-fixtures.sh +++ b/tools/evals/lib/skill-eval-fixtures.sh @@ -174,6 +174,26 @@ inline_eval_only_roots = { { "name" => "independent_review_status", "type" => "enum", "required" => true, "enum_values" => ["required"] } ] }, + { + "name" => "execution_policy", "type" => "object", "required" => false, + "object_fields" => [ + { "name" => "source_writer_policy", "type" => "enum", "required" => true, "enum_values" => %w[sequential_shared_or_unknown isolated_A_B_overlap_permitted] }, + { "name" => "read_only_analysis_policy", "type" => "enum", "required" => true, "enum_values" => %w[parallel_permitted sequential_only] }, + { "name" => "isolation_evidence_ref", "type" => "string", "required" => true }, + { + "name" => "c_start_decisions", "type" => "object[]", "required" => true, + "object_fields" => [ + { "name" => "a_status", "type" => "enum", "required" => true, "enum_values" => %w[PENDING RUNNING VERIFIED] }, + { "name" => "c_decision", "type" => "enum", "required" => true, "enum_values" => %w[blocked ready] } + ] + }, + { "name" => "per_slice_verification", "type" => "enum", "required" => true, "enum_values" => %w[required deferred_to_integration] }, + { "name" => "integration_validation", "type" => "enum", "required" => true, "enum_values" => %w[required per_slice_only] }, + { "name" => "integration_checks", "type" => "string[]", "required" => true }, + { "name" => "fresh_review", "type" => "enum", "required" => true, "enum_values" => %w[required not_required] }, + { "name" => "fresh_review_after", "type" => "enum", "required" => true, "enum_values" => %w[integration_validation per_slice_verification] } + ] + }, { "name" => "workflow_complete", "type" => "enum", "required" => false, "enum_values" => ["--- WORKFLOW COMPLETE ---"] } ] } @@ -764,6 +784,17 @@ validate_fixture() { if (.[$name]? | nonempty_string) then empty else "case[\($index)] missing or invalid string field: \($name)" end; + def case_prompt_packet_mode($index): + if has("prompt_packet_mode") then + .prompt_packet_mode as $mode + | if ($mode | type) != "string" + or (["task_only", "annotated"] | index($mode)) == null then + "case[\($index)].prompt_packet_mode must be task_only or annotated when present" + else + empty + end + else empty end; + def safe_case_id($index): (.id? // null) as $id | if ($id | nonempty_string | not) then @@ -1004,6 +1035,7 @@ validate_fixture() { case_string_array($index; "expected_behavior"), case_string_array($index; "pass_criteria"), case_string_array($index; "fail_signals"), + case_prompt_packet_mode($index), case_machine_expectations($index), case_seeded_defects($index), case_semantic_context($index) diff --git a/tools/evals/lib/skill-eval-render.sh b/tools/evals/lib/skill-eval-render.sh index be75133..5cb5aae 100644 --- a/tools/evals/lib/skill-eval-render.sh +++ b/tools/evals/lib/skill-eval-render.sh @@ -94,10 +94,19 @@ emit_prompts() { + ($authority | tojson) + "\n```\n\n" else "" end; + def task_only_packet: + "# Task Packet\n\n" + + "Skill: " + $skill + "\n\n" + + "Skill Path: " + $skill_path + "\n\n" + + "## Setup Context\n\n" + bullets(.setup_context) + "\n\n" + + "## Prompt\n\n" + .prompt + "\n"; . as $fixture | .cases[] | select(.id == $id) - | "# " + .title + "\n\n" + | if .prompt_packet_mode == "task_only" then + task_only_packet + else + "# " + .title + "\n\n" + "Skill: " + $skill + "\n\n" + "Skill Path: " + $skill_path + "\n\n" + "Case ID: " + .id + "\n\n" @@ -120,6 +129,7 @@ emit_prompts() { + "### Forbidden Substrings\n\n" + bullets(.machine_expectations.forbidden_substrings) + "\n\n" + structured_json_assertions_section + end ' "$fixture_file" >"$packet_path" done < <(jq -r --argjson selected_cases "$selected_cases" ' .cases[] diff --git a/tools/evals/run-codex-framework-evals.sh b/tools/evals/run-codex-framework-evals.sh index 7788c78..bc0b26b 100755 --- a/tools/evals/run-codex-framework-evals.sh +++ b/tools/evals/run-codex-framework-evals.sh @@ -61,6 +61,42 @@ RESUME_RESTORE_ACTIVATION=false RESUME_INITIALIZE_ATTEMPTS=false RESUME_TRACE_RECOVERIES_FILE="" RESUME_ATTEMPT_COMPLETIONS_FILE="" +readonly EVENT_EVIDENCE_NORMALIZATION_JQ="$(cat <<'JQ' + def normalized_workspace_path: + if type == "string" then ($path_mappings[.] // null) else null end; + def normalized_file_change_paths: + .item as $item + | (if $item | has("path") then + if ($item.path | type) == "string" then [$item.path] else null end + else [] end) as $item_paths + | (if $item | has("changes") then + if (($item.changes | type) == "array") + and (($item.changes | length) > 0) + and all($item.changes[]; type == "object" and has("path") and (.path | type) == "string") then + [$item.changes[].path] + else null + end + else [] end) as $change_paths + | if $item_paths == null or $change_paths == null + or (($item_paths | length) + ($change_paths | length) == 0) then null + else ($item_paths + $change_paths | map(normalized_workspace_path)) + end; + def command_text: (.item.command // .item.command_line // ""); + def is_exact_command($expected): + command_text as $command + | if ($command | type) == "array" then + ($command == ($expected | split(" "))) + or any(["/bin/bash", "/bin/zsh", "/bin/sh"][]; + $command == [., "-lc", $expected]) + else + ($command | tostring) as $text + | ($text == $expected) + or any(["/bin/bash", "/bin/zsh", "/bin/sh"][]; + $text == (. + " -lc " + ([39] | implode) + $expected + ([39] | implode)) + or $text == (. + " -lc " + ([34] | implode) + $expected + ([34] | implode))) + end; +JQ +)" has_framework_eval_test_hook() { local variable @@ -2052,6 +2088,8 @@ path_allowed_for_case() { case "$case_id:$path" in small-fix-stays-lightweight:docs/usage.md) return 0 ;; small-fix-stays-lightweight:.assistant-eval/workflow-decision.json) return 0 ;; + pivot-restart-on-stagnation-or-code-writer-blocker:.assistant-eval/stagnation-recovery.json) return 0 ;; + pivot-restart-on-stagnation-or-code-writer-blocker:.assistant-eval/stagnation-failure-receipt.json) return 0 ;; stale-journal-yields-to-current-evidence:.codex/task.md) return 0 ;; requirements-map-through-completion:.assistant-eval/requirement-map.json) return 0 ;; ordinary-medium-bounded-executor:.assistant-eval/execution-decision.json) return 0 ;; @@ -2083,30 +2121,53 @@ workspace_record_check() { fi } -workspace_artifact_preflight() { +workspace_regular_file_preflight() { local workspace="$1" - local artifact="$2" - local failure_id="$3" - local workspace_canonical artifact_parent_canonical artifact_canonical artifact_size="" + local target="$2" + local max_size="$3" + local workspace_canonical target_parent target_parent_canonical target_canonical target_size="" local safe=true - workspace_artifact_safe=false - workspace_canonical="$(cd "$workspace" 2>/dev/null && pwd -P)" || safe=false - if [[ "$safe" == "true" && -e "$artifact" && ! -L "$artifact" && -f "$artifact" ]]; then - artifact_parent_canonical="$(cd "$(dirname "$artifact")" 2>/dev/null && pwd -P)" || safe=false - artifact_canonical="$artifact_parent_canonical/$(basename "$artifact")" - artifact_size="$(wc -c <"$artifact" 2>/dev/null | tr -d '[:space:]')" || safe=false - if [[ "$artifact_canonical" != "$workspace_canonical/"* ]] \ - || [[ ! "$artifact_size" =~ ^[0-9]+$ ]] \ - || [[ "$artifact_size" -lt 1 || "$artifact_size" -gt 65536 ]]; then + workspace_regular_file_safe=false + [[ "$max_size" =~ ^[0-9]+$ ]] || safe=false + if [[ "$safe" == "true" && -d "$workspace" && ! -L "$workspace" ]]; then + workspace_canonical="$(cd "$workspace" 2>/dev/null && pwd -P)" || safe=false + else + safe=false + fi + target_parent="$(dirname "$target")" + while [[ "$safe" == "true" ]]; do + if [[ "$target_parent" != "$workspace" && "$target_parent" != "$workspace/"* ]] \ + || [[ ! -d "$target_parent" || -L "$target_parent" ]]; then + safe=false + break + fi + [[ "$target_parent" == "$workspace" ]] && break + target_parent="$(dirname "$target_parent")" + done + if [[ "$safe" == "true" && -e "$target" && ! -L "$target" && -f "$target" ]]; then + target_parent_canonical="$(cd "$(dirname "$target")" 2>/dev/null && pwd -P)" || safe=false + target_canonical="$target_parent_canonical/$(basename "$target")" + target_size="$(wc -c <"$target" 2>/dev/null | tr -d '[:space:]')" || safe=false + if [[ "$target_canonical" != "$workspace_canonical/"* ]] \ + || [[ ! "$target_size" =~ ^[0-9]+$ ]] \ + || [[ "$target_size" -lt 1 || "$target_size" -gt "$max_size" ]]; then safe=false fi else safe=false fi - workspace_record_check "$failure_id" "$safe" if [[ "$safe" == "true" ]]; then + workspace_regular_file_safe=true + fi +} + +workspace_artifact_preflight() { + workspace_artifact_safe=false + workspace_regular_file_preflight "$1" "$2" 65536 + workspace_record_check "$3" "$workspace_regular_file_safe" + if [[ "$workspace_regular_file_safe" == "true" ]]; then workspace_artifact_safe=true fi } @@ -2202,6 +2263,301 @@ ordered_workflow_event_evidence() { ' "$jsonl" } +workspace_event_path_mappings() { + local jsonl="$1" workspace="$2" + + command -v python3 >/dev/null 2>&1 || return 1 + python3 - "$jsonl" "$workspace" <<'PY' +import json +import os +import re +import sys + +jsonl_path, workspace = sys.argv[1:] +workspace_real = os.path.realpath(workspace) +if not os.path.isdir(workspace_real): + raise SystemExit(1) + +def canonical_relative(value): + if not isinstance(value, str) or not value: + return None + candidate = value.replace("\\", "/") + if candidate.startswith("/") or re.match(r"^[A-Za-z]:/", candidate): + return None + segments = candidate.split("/") + if any(not segment or segment == ".." for segment in segments): + return None + normalized = [segment for segment in segments if segment != "."] + return "/".join(normalized) or None + +def normalize(value): + if not isinstance(value, str) or any(ord(character) < 32 or ord(character) == 127 for character in value): + return None + candidate = value.replace("\\", "/") + if not candidate.startswith("/"): + return canonical_relative(candidate) + if candidate.startswith("//"): + return None + segments = candidate[1:].split("/") + if any(segment == ".." for segment in segments): + return None + absolute = "/" + "/".join(segment for segment in segments if segment and segment != ".") + resolved = os.path.realpath(absolute) + try: + if os.path.commonpath([workspace_real, resolved]) != workspace_real: + return None + except ValueError: + return None + return canonical_relative(os.path.relpath(resolved, workspace_real)) + +mappings = {} +with open(jsonl_path, encoding="utf-8") as stream: + for line in stream: + event = json.loads(line) + item = event.get("item") if isinstance(event, dict) else None + if not isinstance(item, dict) or item.get("type") != "file_change": + continue + candidates = [] + if "path" in item: + candidates.append(item["path"]) + if "changes" in item and isinstance(item["changes"], list): + candidates.extend(change.get("path") for change in item["changes"] if isinstance(change, dict) and "path" in change) + for candidate in candidates: + if isinstance(candidate, str): + mappings[candidate] = normalize(candidate) +print(json.dumps(mappings, sort_keys=True, separators=(",", ":"))) +PY +} + +small_fix_event_evidence() { + local jsonl="$1" workspace="$2" path_mappings + path_mappings="$(workspace_event_path_mappings "$jsonl" "$workspace")" || return 1 + jq -cse --argjson path_mappings "$path_mappings" "$EVENT_EVIDENCE_NORMALIZATION_JQ$(cat <<'JQ' + def changes_containing($expected): + normalized_file_change_paths as $paths + | $paths != null and ($paths | length) > 0 + and all($paths[]; . != null) and any($paths[]; . == $expected); + def is_discovery: + .type == "item.completed" + and .item.type == "command_execution" + and .item.exit_code == 0 + and is_exact_command("rg -n teh docs/usage.md"); + def is_discovery_started: + .type == "item.started" + and .item.type == "command_execution" + and is_exact_command("rg -n teh docs/usage.md"); + def is_workspace_action: + (.type == "item.completed" or .type == "item.started") + and (.item.type == "command_execution" or .item.type == "file_change"); + def is_discovery_action: + is_discovery or is_discovery_started; + def is_change: + .type == "item.completed" + and .item.type == "file_change" + and changes_containing("docs/usage.md"); + + . as $events + | [range(0; length) | select($events[.] | is_discovery)] as $discoveries + | [range(0; length) | select($events[.] | is_change)] as $changes + | [range(0; length) + | . as $index + | $events[$index] + | select(is_discovery_started) + | {index: $index, id: .item.id} + ] as $discovery_starts + | [range(0; length) + | . as $index + | $events[$index] + | select(.type == "item.completed" and .item.type == "command_execution") + | {index: $index, id: .item.id, exact_discovery_command: is_exact_command("rg -n teh docs/usage.md")} + ] as $discovery_completions + | ($discovery_starts | map(.id)) as $discovery_start_ids + | ($discovery_completions | map(.id)) as $discovery_completion_ids + | (($discovery_starts | length) == 0 or ( + ($discovery_start_ids | all(.[]; type == "string" and length > 0)) + and (($discovery_start_ids | unique | length) == ($discovery_start_ids | length)) + and (($discovery_completion_ids | all(.[]; type == "string" and length > 0)) + and (($discovery_completion_ids | unique | length) == ($discovery_completion_ids | length))) + and ($discovery_starts | all(.[]; . as $start + | [$discovery_completions[] | select(.id == $start.id and .index > $start.index)] as $matching_completions + | ($matching_completions | length) == 1 + and $matching_completions[0].exact_discovery_command + and ($start.index > $changes[0] or $matching_completions[0].index < $changes[0]))) + )) as $discovery_lifecycles_valid + | [range(0; length) | select($events[.] | is_workspace_action)] as $workspace_actions + | [range(0; length) | select($events[.] | is_discovery_action)] as $discovery_actions + | [range(0; length) | select( + ($events[.].type == "item.completed" or $events[.].type == "item.started" or $events[.].type == "item.updated") + and ($events[.].item.type == "mcp_tool_call" or $events[.].item.type == "web_search"))] as $disallowed + | { + source_discovery_before_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($changes | length) > 0 and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and $discoveries[0] < $changes[0]), + disallowed_item_count: ($disallowed | length) + } +JQ +)" "$jsonl" +} + +observed_event_shape_supported() { + local jsonl="$1" allowed_item_types="$2" + jq -se --argjson allowed_item_types "$allowed_item_types" ' + all(.[]; + if .type == "item.updated" then + false + elif .type == "item.completed" or .type == "item.started" then + (.item.type as $item_type | ($item_type | type == "string") and ($allowed_item_types | index($item_type)) != null) + else true end) + ' "$jsonl" >/dev/null +} + +small_fix_event_shape_supported() { + observed_event_shape_supported "$1" '["agent_message","command_execution","file_change","mcp_tool_call","web_search"]' +} + +stagnation_event_shape_supported() { + observed_event_shape_supported "$1" '["agent_message","command_execution","file_change"]' +} + +small_fix_requires_unavailable_adapter_policy() { + jq -e ' + .cases[] | select(.id == "small-fix-stays-lightweight") + | .machine_expectations.observed_event_requirements + == {source_discovery_before_change:true,disallowed_item_types:["mcp_tool_call","web_search"],unsupported_event_policy:"adapter_unavailable"} + ' "$FIXTURE" >/dev/null +} + +stagnation_recovery_event_evidence() { + local jsonl="$1" workspace="$2" path_mappings + path_mappings="$(workspace_event_path_mappings "$jsonl" "$workspace")" || return 1 + jq -cse --argjson path_mappings "$path_mappings" "$EVENT_EVIDENCE_NORMALIZATION_JQ$(cat <<'JQ' + def changes_only($expected): + normalized_file_change_paths as $paths + | $paths != null and ($paths | length) > 0 and all($paths[]; . == $expected); + def command_output: (.item.aggregated_output // .item.output // "" | tostring); + def has_numeric_exit_code: + (.item.exit_code | type) == "number"; + def is_initial_stagnation_invocation: + .type == "item.completed" and .item.type == "command_execution" + and is_exact_command("bash tests/stagnation-contracts.sh"); + def is_failed_stagnation: + is_initial_stagnation_invocation + and has_numeric_exit_code and .item.exit_code != 0 + and (command_output | contains("STAGNATION_TRUSTED_FAILURE")); + def is_recovery_invocation: + .type == "item.completed" and .item.type == "command_execution" + and is_exact_command("bash tests/recovery-contracts.sh"); + def is_recovery: + is_recovery_invocation + and has_numeric_exit_code and .item.exit_code == 0 + and (command_output | contains("RECOVERY_APPLIED")); + def is_fresh_check_invocation: + .type == "item.completed" and .item.type == "command_execution" + and is_exact_command("bash tests/stagnation-contracts.sh --after-recovery"); + def is_fresh_check: + is_fresh_check_invocation + and has_numeric_exit_code and .item.exit_code == 0 + and (command_output | contains("STAGNATION_FRESH_CHECK_PASS")); + def recovery_command_kind: + if is_exact_command("bash tests/stagnation-contracts.sh") then "trusted_failure" + elif is_exact_command("bash tests/recovery-contracts.sh") then "recovery" + elif is_exact_command("bash tests/stagnation-contracts.sh --after-recovery") then "fresh_check" + elif is_exact_command("cat RECOVERY.md") then "recovery_read" + else null + end; + def is_allowed_recovery_command: + .item.type == "command_execution" + and recovery_command_kind != null; + def is_forbidden_recovery_command: + (.type == "item.started" or .type == "item.completed") + and .item.type == "command_execution" + and (is_allowed_recovery_command | not); + def is_forbidden_retry_or_patch: + .type == "item.completed" and .item.type == "command_execution" + and (command_text | tostring | test("(^|[[:space:]])(patch|retry)([[:space:]]|$)"; "i")); + def is_forbidden_source_change: + (.type == "item.started" or .type == "item.completed") + and .item.type == "file_change" + and (changes_only(".assistant-eval/stagnation-recovery.json") | not); + def is_workspace_action: + (.type == "item.started" or .type == "item.completed") + and (.item.type == "command_execution" or .item.type == "file_change"); + + . as $events + | [range(0; length) | select($events[.] | is_initial_stagnation_invocation)] as $initial_invocations + | [range(0; length) | select($events[.] | is_failed_stagnation)] as $failures + | [range(0; length) | select($events[.] | is_recovery_invocation)] as $recovery_invocations + | [range(0; length) | select($events[.] | is_recovery)] as $recoveries + | [range(0; length) | select($events[.] | is_fresh_check_invocation)] as $fresh_check_invocations + | [range(0; length) | select($events[.] | is_fresh_check)] as $fresh_checks + | [range(0; length) + | . as $index + | $events[$index] + | select(.type == "item.started" and .item.type == "command_execution" and is_allowed_recovery_command) + | {index: $index, id: .item.id, kind: recovery_command_kind} + ] as $command_starts + | [range(0; length) + | . as $index + | $events[$index] + | select(.type == "item.completed" and .item.type == "command_execution" and is_allowed_recovery_command) + | {index: $index, id: .item.id, kind: recovery_command_kind} + ] as $command_completions + | ($command_starts | all(.[]; (.id | type == "string" and length > 0))) as $start_ids_nonempty + | ($command_starts | map(.id)) as $start_ids + | (($start_ids | unique | length) == ($start_ids | length)) as $start_ids_unique + | ($command_completions | map(select(.id | type == "string" and length > 0) | .id)) as $completion_ids + | (($completion_ids | unique | length) == ($completion_ids | length)) as $completion_ids_unique + | ($command_starts | all(.[]; . as $start | [$command_completions[] | select(.id == $start.id and .kind == $start.kind and .index > $start.index)] | length == 1)) as $starts_match_one_later_completion + | ($start_ids_nonempty and $start_ids_unique and $completion_ids_unique and $starts_match_one_later_completion) as $command_lifecycles_valid + | ([ $command_starts[] | select(.kind == "recovery") | .index ] | if length == 1 then .[0] else $recovery_invocations[0] end) as $recovery_start_or_completion + | ([ $command_starts[] | select(.kind == "fresh_check") | .index ] | if length == 1 then .[0] else $fresh_check_invocations[0] end) as $fresh_check_start_or_completion + | [range(0; length) | select($events[.] | is_forbidden_recovery_command)] as $forbidden_commands + | [range(0; length) | select($events[.] | is_forbidden_retry_or_patch)] as $forbidden + | [range(0; length) | select($events[.] | is_forbidden_source_change)] as $forbidden_changes + | [range(0; length) | select( + . as $index + | ($events[$index] | is_workspace_action) + and ($fresh_checks | length) == 1 + and $index > $fresh_checks[0]) + ] as $post_bound_workspace_actions + | { + trusted_failure_before_recovery: (($initial_invocations | length) == 1 and ($failures | length) == 1 and ($recovery_invocations | length) == 1 and ($recoveries | length) == 1 and $command_lifecycles_valid and $failures[0] < $recovery_start_or_completion), + recovery_before_fresh_check: (($recovery_invocations | length) == 1 and ($fresh_check_invocations | length) == 1 and ($fresh_checks | length) == 1 and $command_lifecycles_valid and $recoveries[0] < $fresh_check_start_or_completion), + reject_patch_or_retry_after_bound: (($forbidden_commands | length) == 0 and ($forbidden | length) == 0 and ($forbidden_changes | length) == 0 and $command_lifecycles_valid and ($fresh_checks | length) == 1 and ($post_bound_workspace_actions | length) == 0), + terminal_completed: false + } +JQ +)" "$jsonl" +} + +stagnation_fixture_is_trusted() { + local workspace="$1" fixture="$REPO_ROOT/docs/evals/fixtures/pivot-restart-on-stagnation" + local probe failure_output recovery_output fresh_output failure_exit=0 recovery_exit=0 fresh_exit=0 trusted=false + local relative + + for relative in RECOVERY.md tests/stagnation-contracts.sh tests/recovery-contracts.sh; do + [[ -f "$workspace/$relative" && ! -L "$workspace/$relative" ]] \ + && [[ "$(hash_file "$workspace/$relative")" == "$(hash_file "$fixture/$relative")" ]] \ + || return 1 + done + [[ -x "$workspace/tests/stagnation-contracts.sh" && -x "$workspace/tests/recovery-contracts.sh" ]] || return 1 + probe="$(mktemp -d "${TMPDIR:-/tmp}/assistant-framework-stagnation.XXXXXX")" || return 1 + if ! cp -R "$fixture"/. "$probe/"; then + rm -rf "$probe" + return 1 + fi + failure_output="$(bash "$probe/tests/stagnation-contracts.sh")" || failure_exit=$? + recovery_output="$(bash "$probe/tests/recovery-contracts.sh")" || recovery_exit=$? + fresh_output="$(bash "$probe/tests/stagnation-contracts.sh" --after-recovery)" || fresh_exit=$? + if [[ "$failure_exit" -ne 0 && "$recovery_exit" -eq 0 && "$fresh_exit" -eq 0 \ + && "$failure_output" == "STAGNATION_TRUSTED_FAILURE" \ + && "$recovery_output" == "RECOVERY_APPLIED" \ + && "$fresh_output" == "STAGNATION_FRESH_CHECK_PASS" ]] \ + && grep -Fqx 'terminal_completed=false' "$probe/RECOVERY.md"; then + trusted=true + fi + rm -rf "$probe" + [[ "$trusted" == true ]] +} + viewing_inspection_event_evidence() { local jsonl="$1" jq -cse ' @@ -2457,7 +2813,10 @@ verify_workspace() { local workspace_acceptance_items_total=0 local workspace_acceptance_items_passed=0 local workspace_artifact_safe=false - local artifact review_artifact event_evidence final_source_hash="" + local artifact review_artifact event_evidence final_source_hash="" workspace_identity="" + + workspace_identity="$(cd "$workspace" && pwd -P)" \ + || die "Could not resolve the evaluation workspace identity." while IFS= read -r changed_path; do [[ -n "$changed_path" ]] || continue @@ -2469,7 +2828,9 @@ verify_workspace() { case "$case_id" in small-fix-stays-lightweight) status="passed" - if [[ ! -f "$workspace/docs/usage.md" ]] \ + workspace_regular_file_preflight "$workspace" "$workspace/docs/usage.md" 65536 + if [[ "$workspace_regular_file_safe" != "true" ]] \ + || [[ ! -f "$workspace/docs/usage.md" ]] \ || grep -Fq -- "teh" "$workspace/docs/usage.md" \ || ! grep -Fq -- "the requested typo" "$workspace/docs/usage.md"; then workspace_record_check "workspace-001" false @@ -2484,6 +2845,33 @@ verify_workspace() { and .schema_version == "1.0" and .task_size == "trivial" and .plan_mode == "none"' + event_evidence="$(small_fix_event_evidence "$jsonl" "$workspace_identity")" || event_evidence='{}' + if small_fix_requires_unavailable_adapter_policy \ + && jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0' <<<"$event_evidence" >/dev/null; then + workspace_record_check "workspace-003" true + else + workspace_record_check "workspace-003" false + fi + ;; + pivot-restart-on-stagnation-or-code-writer-blocker) + status="passed" + artifact="$workspace/.assistant-eval/stagnation-recovery.json" + workspace_artifact_preflight "$workspace" "$artifact" "workspace-001" + workspace_json_check "$artifact" "workspace-002" ' + type == "object" + and . == {schema_version:"1.0",trusted_failure:"observed",recovery:"applied",fresh_check:"passed",terminal_completed:false}' + event_evidence="$(stagnation_recovery_event_evidence "$jsonl" "$workspace_identity")" || event_evidence='{}' + if stagnation_fixture_is_trusted "$workspace" \ + && jq -e --argjson evidence "$event_evidence" ' + .cases[] | select(.id == "pivot-restart-on-stagnation-or-code-writer-blocker") + | .machine_expectations.observed_recovery_requirements + == {trusted_failure_before_recovery:true,recovery_before_fresh_check:true,reject_patch_or_retry_after_bound:true,terminal_completed:false} + ' "$FIXTURE" >/dev/null \ + && jq -e '.trusted_failure_before_recovery == true and .recovery_before_fresh_check == true and .reject_patch_or_retry_after_bound == true and .terminal_completed == false' <<<"$event_evidence" >/dev/null; then + workspace_record_check "workspace-003" true + else + workspace_record_check "workspace-003" false + fi ;; codex-role-constraints-native) status="passed" @@ -2816,6 +3204,9 @@ EOF pending-architecture-pack-verification) printf '%s\n' 'VIEWING quality scenario has planned verification after implementation.' >"$workspace/PENDING_ARCHITECTURE_PACK.md" ;; + pivot-restart-on-stagnation-or-code-writer-blocker) + cp -R "$REPO_ROOT/docs/evals/fixtures/pivot-restart-on-stagnation"/. "$workspace/" + ;; medium-final-handoff-is-reconstructable) mkdir -p "$workspace/src" "$workspace/tests" cat >"$workspace/CHANGE_SUMMARY.md" <<'EOF' @@ -3052,6 +3443,17 @@ execute_one_run() { grader_digest="$(grader_hash "$case_id")" prompt="$(blind_prompt_for_case "$case_id")" + if [[ "$case_id" == "isolated-parallel-a-b-then-c-with-integration" ]]; then + validate_run_attempt_identity "$attempt_path" "$pair_id" "$case_id" "$trial_index" "$variant" \ + || die "Run-attempt state is invalid before unavailable completion: $run_id" + [[ "$(jq -r '.state' "$attempt_path")" == "not_started" ]] \ + || die "Run-attempt state is not completable without model dispatch: $run_id" + write_unavailable_trace "$trace_path" "$run_id" "$pair_id" "$case_id" "$trial_index" "$variant" \ + unknown_event_shape 0 "$fixture_hash" "$case_digest" "$instruction_hash" "$grader_digest" "$cli_version" + mark_run_attempt_completed "$attempt_path" || die "Could not complete unavailable run-attempt state for $run_id." + return + fi + local run_raw workspace jsonl final_output stderr_file prompt_file seed_workspace_hash local now remaining_total effective_timeout now="$(date +%s)" @@ -3162,6 +3564,25 @@ execute_one_run() { return fi + if [[ "$case_id" == "small-fix-stays-lightweight" ]] \ + && small_fix_requires_unavailable_adapter_policy \ + && ! small_fix_event_shape_supported "$jsonl"; then + write_unavailable_trace "$trace_path" "$run_id" "$pair_id" "$case_id" "$trial_index" "$variant" \ + unknown_event_shape 0 "$fixture_hash" "$case_digest" "$instruction_hash" "$grader_digest" "$cli_version" + mark_run_attempt_completed "$attempt_path" || die "Could not complete run-attempt state for $run_id." + rm -rf "$run_raw" + return + fi + + if [[ "$case_id" == "pivot-restart-on-stagnation-or-code-writer-blocker" ]] \ + && ! stagnation_event_shape_supported "$jsonl"; then + write_unavailable_trace "$trace_path" "$run_id" "$pair_id" "$case_id" "$trial_index" "$variant" \ + unknown_event_shape 0 "$fixture_hash" "$case_digest" "$instruction_hash" "$grader_digest" "$cli_version" + mark_run_attempt_completed "$attempt_path" || die "Could not complete run-attempt state for $run_id." + rm -rf "$run_raw" + return + fi + local semantic_extract_path="" if is_synthetic_seeded_case "$case_id"; then semantic_extract_path="$RAW_ROOT/semantic-extracts/$pair_id-$variant.json" From 634d4c3b7b42276070aa9799d76162f9365fc499 Mon Sep 17 00:00:00 2001 From: Laimis Date: Wed, 23 Sep 2026 17:26:27 +0300 Subject: [PATCH 2/6] fix: address execution-pattern eval review findings --- docs/evals/README.md | 23 +++- docs/skill-contract-design-guide.md | 2 +- .../references/skill-contract-design-guide.md | 2 +- skills/assistant-workflow/evals/cases.json | 14 +- .../p0-p4/codex-behavioral-eval-contracts.sh | 129 ++++++++++++++++-- .../p0-p4/progressive-discovery-contracts.sh | 6 + tests/p0-p4/skill-eval-contracts.sh | 88 +++++++++++- tests/p0-p4/task-packet-contracts.sh | 28 +++- tools/evals/lib/skill-eval-fixtures.sh | 42 +++++- tools/evals/lib/skill-eval-grade.sh | 4 + tools/evals/run-codex-framework-evals.sh | 57 +++++++- 11 files changed, 371 insertions(+), 24 deletions(-) diff --git a/docs/evals/README.md b/docs/evals/README.md index 644b94d..0e9d920 100644 --- a/docs/evals/README.md +++ b/docs/evals/README.md @@ -37,7 +37,14 @@ Codex adapter actually observed in its JSONL event stream. to be the first workspace command or file action; the matching command-start event may precede its successful completion. This rule and its no-web/MCP check apply only to the disposable local typo fixture, not delegated work. - A grading artifact alone cannot pass the case. + A grading artifact alone cannot pass the case. If the safe target ends in the + expected state after a later successful command but the event stream has no + `file_change`, the adapter reports unavailable: command text and final state + cannot prove which command edited the target or when. That classification is + allowed only when response grading passes, the sole workspace failure is the + missing observation, and there are no scope deviations. Observable + pre-discovery actions, external calls, and incorrect final content remain + completed failures. - `pivot-restart-on-stagnation-or-code-writer-blocker` seeds a trusted failing check, fixture-owned failure/recovery receipts, recovery action, and fresh check. Its recovery artifact retains `terminal_completed=false`: a fresh-check @@ -378,7 +385,10 @@ model-selection evidence and counts incomplete pairs. A second incomplete pair stops the batch before any later call, leaves remaining attempt records `not_started`, and withholds comparison and semantic-review artifacts. The exact limit is bound into the run plan as -`max_incomplete_pairs=1`. The runner never retries an uncertain call. +`max_incomplete_pairs=1`. Only the known pre-dispatch unavailable traces for the +`isolated-parallel-a-b-then-c-with-integration` fixture are excluded; every other +incomplete pair counts toward the limit. The runner never retries an uncertain +call. An `in_flight` record without a valid trace is quota-uncertain: resume exits before every model call, reports only the bounded run ID, and requires separate @@ -781,8 +791,9 @@ Cases may additionally define `machine_expectations.structured_json_assertions`. For these per-skill cases, the response must contain exactly one valid JSON value. The local grader applies only the fixed provider-neutral operators: `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, -`required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, and -`array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. Assertion paths are JSON arrays for safe +`required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, +`object_keys_exact`, `array_items_nonempty_fields`, and +`array_items_nonempty_array_fields`. Assertion paths are JSON arrays for safe `getpath` access. They are grader-only declarations, never executable fixture content: arbitrary jq, code, or expressions are not accepted. `array_items_nonempty_fields` requires the target array to contain at least one object, and every listed field @@ -790,6 +801,10 @@ in every object must be a non-empty string. `array_items_nonempty_array_fields` requires the target array to contain at least one object, and every listed field in every object must be a non-empty array whose every member is a nonblank string. +`object_keys_exact` requires the target to be an object whose keys exactly match +the declared `fields`; it rejects missing keys, extra keys, and non-object +values. Its path may be `[]` to check the JSON response root, or a declared +object path. It accepts at most 16 unique field names. `empty_array` requires the target path to resolve to an empty array. `array_type` requires the target path to resolve to an array and permits an empty array. `array_nonblank_strings` requires every member to be a nonblank string and a required boolean `allow_empty` declares whether an empty array is valid. In this exhaustive fixed operator list, `path_absent` passes only when its target diff --git a/docs/skill-contract-design-guide.md b/docs/skill-contract-design-guide.md index 7456267..32087dd 100644 --- a/docs/skill-contract-design-guide.md +++ b/docs/skill-contract-design-guide.md @@ -472,7 +472,7 @@ shapes native description routing and is separate from response-grade `.cases`. Local response grading is deterministic and heuristic: missing files, empty responses, fail-signal phrase hits, required substrings, and forbidden substrings. It is a provider-neutral proxy for behavior conformance and does not replace human or LLM semantic judgment. -Per-skill cases may optionally add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty. `array_nonblank_strings` requires a string array of nonblank members; `allow_empty` controls zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership in bounded scalars. `array_object_values_exact` projects every target-array object to its bounded explicit `fields` list and compares those tuples as an unordered exact multiset against `expected_objects`; this preserves each declared field correlation without making object order significant. `array_items_nonempty_fields` requires a non-empty target array and non-empty string values for every listed field in every object. `array_items_nonempty_array_fields` requires a non-empty target array and a non-empty array of nonblank string values for every listed field in every object. Paths must be non-empty JSON arrays of string object keys or non-negative integer array indexes. A selected structured case requires exactly one valid JSON response value. These declarations are grader-only data, never executable instructions: do not permit arbitrary jq, code, expressions, or fixture-provided operators outside this fixed set. +Per-skill cases may optionally add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `object_keys_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty. `array_nonblank_strings` requires a string array of nonblank members; `allow_empty` controls zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership in bounded scalars. `array_object_values_exact` projects every target-array object to its bounded explicit `fields` list and compares those tuples as an unordered exact multiset against `expected_objects`; this preserves each declared field correlation without making object order significant. `object_keys_exact` requires an object whose keys exactly match its bounded, unique `fields` list; it rejects extra keys, missing keys, and non-object values. Its path may be `[]` to check the response root or a declared nested object. `array_items_nonempty_fields` requires a non-empty target array and non-empty string values for every listed field in every object. `array_items_nonempty_array_fields` requires a non-empty target array and a non-empty array of nonblank string values for every listed field in every object. Paths are JSON arrays of string object keys or non-negative integer array indexes; only `object_keys_exact` permits the empty root path. A selected structured case requires exactly one valid JSON response value. These declarations are grader-only data, never executable instructions: do not permit arbitrary jq, code, expressions, or fixture-provided operators outside this fixed set. `one_of` accepts at most 32 declared scalar values. `array_object_values_exact` accepts at most 16 unique projected fields and 32 expected objects; it compares diff --git a/skills/assistant-skill-creator/references/skill-contract-design-guide.md b/skills/assistant-skill-creator/references/skill-contract-design-guide.md index 7456267..32087dd 100644 --- a/skills/assistant-skill-creator/references/skill-contract-design-guide.md +++ b/skills/assistant-skill-creator/references/skill-contract-design-guide.md @@ -472,7 +472,7 @@ shapes native description routing and is separate from response-grade `.cases`. Local response grading is deterministic and heuristic: missing files, empty responses, fail-signal phrase hits, required substrings, and forbidden substrings. It is a provider-neutral proxy for behavior conformance and does not replace human or LLM semantic judgment. -Per-skill cases may optionally add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty. `array_nonblank_strings` requires a string array of nonblank members; `allow_empty` controls zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership in bounded scalars. `array_object_values_exact` projects every target-array object to its bounded explicit `fields` list and compares those tuples as an unordered exact multiset against `expected_objects`; this preserves each declared field correlation without making object order significant. `array_items_nonempty_fields` requires a non-empty target array and non-empty string values for every listed field in every object. `array_items_nonempty_array_fields` requires a non-empty target array and a non-empty array of nonblank string values for every listed field in every object. Paths must be non-empty JSON arrays of string object keys or non-negative integer array indexes. A selected structured case requires exactly one valid JSON response value. These declarations are grader-only data, never executable instructions: do not permit arbitrary jq, code, expressions, or fixture-provided operators outside this fixed set. +Per-skill cases may optionally add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `object_keys_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty. `array_nonblank_strings` requires a string array of nonblank members; `allow_empty` controls zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership in bounded scalars. `array_object_values_exact` projects every target-array object to its bounded explicit `fields` list and compares those tuples as an unordered exact multiset against `expected_objects`; this preserves each declared field correlation without making object order significant. `object_keys_exact` requires an object whose keys exactly match its bounded, unique `fields` list; it rejects extra keys, missing keys, and non-object values. Its path may be `[]` to check the response root or a declared nested object. `array_items_nonempty_fields` requires a non-empty target array and non-empty string values for every listed field in every object. `array_items_nonempty_array_fields` requires a non-empty target array and a non-empty array of nonblank string values for every listed field in every object. Paths are JSON arrays of string object keys or non-negative integer array indexes; only `object_keys_exact` permits the empty root path. A selected structured case requires exactly one valid JSON response value. These declarations are grader-only data, never executable instructions: do not permit arbitrary jq, code, expressions, or fixture-provided operators outside this fixed set. `one_of` accepts at most 32 declared scalar values. `array_object_values_exact` accepts at most 16 unique projected fields and 32 expected objects; it compares diff --git a/skills/assistant-workflow/evals/cases.json b/skills/assistant-workflow/evals/cases.json index 5ac4bb1..4f56de2 100644 --- a/skills/assistant-workflow/evals/cases.json +++ b/skills/assistant-workflow/evals/cases.json @@ -366,7 +366,12 @@ {"operator":"equals","path":["execution_policy","integration_validation"],"expected":"required"}, {"operator":"equals","path":["execution_policy","integration_checks"],"expected":["cross-slice","full-scope"]}, {"operator":"equals","path":["execution_policy","fresh_review"],"expected":"required"}, - {"operator":"equals","path":["execution_policy","fresh_review_after"],"expected":"integration_validation"} + {"operator":"equals","path":["execution_policy","fresh_review_after"],"expected":"integration_validation"}, + {"operator":"object_keys_exact","path":[],"fields":["execution_policy"]}, + {"operator":"object_keys_exact","path":["execution_policy"],"fields":["source_writer_policy","read_only_analysis_policy","isolation_evidence_ref","c_start_decisions","per_slice_verification","integration_validation","integration_checks","fresh_review","fresh_review_after"]}, + {"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",0],"fields":["a_status","c_decision"]}, + {"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",1],"fields":["a_status","c_decision"]}, + {"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",2],"fields":["a_status","c_decision"]} ] } }, @@ -409,7 +414,12 @@ {"operator":"equals","path":["execution_policy","integration_validation"],"expected":"required"}, {"operator":"equals","path":["execution_policy","integration_checks"],"expected":["cross-slice","full-scope"]}, {"operator":"equals","path":["execution_policy","fresh_review"],"expected":"required"}, - {"operator":"equals","path":["execution_policy","fresh_review_after"],"expected":"integration_validation"} + {"operator":"equals","path":["execution_policy","fresh_review_after"],"expected":"integration_validation"}, + {"operator":"object_keys_exact","path":[],"fields":["execution_policy"]}, + {"operator":"object_keys_exact","path":["execution_policy"],"fields":["source_writer_policy","read_only_analysis_policy","isolation_evidence_ref","c_start_decisions","per_slice_verification","integration_validation","integration_checks","fresh_review","fresh_review_after"]}, + {"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",0],"fields":["a_status","c_decision"]}, + {"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",1],"fields":["a_status","c_decision"]}, + {"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",2],"fields":["a_status","c_decision"]} ] } }, diff --git a/tests/p0-p4/codex-behavioral-eval-contracts.sh b/tests/p0-p4/codex-behavioral-eval-contracts.sh index 2e6310f..1dd79ce 100755 --- a/tests/p0-p4/codex-behavioral-eval-contracts.sh +++ b/tests/p0-p4/codex-behavioral-eval-contracts.sh @@ -259,8 +259,10 @@ fi printf '%s' "$prompt" >"$capture_dir/call-$call_id.prompt" if [[ -f "$workspace/docs/usage.md" ]]; then printf '%s\n' 'docs-usage-present' >>"$capture_dir/call-$call_id.fixtures" - sed -i.bak 's/teh/the/g' "$workspace/docs/usage.md" - rm -f "$workspace/docs/usage.md.bak" + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" != "small-command-only" && "${FAKE_PATTERN_EVENT_MODE:-}" != "small-command-only-wrong-final" ]]; then + sed -i.bak 's/teh/the/g' "$workspace/docs/usage.md" + rm -f "$workspace/docs/usage.md.bak" + fi if [[ "${FAKE_WRONG_SMALL_EDIT:-false}" == "true" ]]; then printf '%s\n' 'This fixture contains the wrong change.' >"$workspace/docs/usage.md" fi @@ -696,7 +698,7 @@ printf '%s\n' '{"type":"turn.started"}' printf '%s\n' '{"type":"item.completed","item":{"id":"item-1","type":"agent_message","text":"phase small docs/usage.md teh"}}' if [[ -f "$workspace/docs/usage.md" ]]; then case "${FAKE_PATTERN_EVENT_MODE:-}" in - ""|small-positive|small-wrapped-positive|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed) + ""|small-positive|small-wrapped-positive|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed|small-command-only|small-command-only-wrong-final) small_discovery_command='"rg -n teh docs/usage.md"' if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-wrapped-positive" ]]; then small_discovery_command='["/bin/zsh","-lc","rg -n teh docs/usage.md"]' @@ -728,7 +730,16 @@ if [[ -f "$workspace/docs/usage.md" ]]; then elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-updated-disallowed" ]]; then printf '%s\n' '{"type":"item.updated","item":{"id":"small-updated-web","type":"web_search","query":"unrelated external lookup"}}' fi - printf '%s\n' '{"type":"item.completed","item":{"id":"small-change","type":"file_change","changes":[{"path":"docs/usage.md","kind":"update"}]}}' + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-command-only" ]]; then + (cd "$workspace/docs" && sed s/teh/the/ usage.md >usage.md.tmp && mv usage.md.tmp usage.md) + printf '%s\n' '{"type":"item.completed","item":{"id":"small-shell-edit","type":"command_execution","command":"cd docs && sed s/teh/the/ usage.md >usage.md.tmp && mv usage.md.tmp usage.md","exit_code":0,"status":"completed","aggregated_output":""}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-command-only-wrong-final" ]]; then + (cd "$workspace/docs" && sed s/teh/Wrong/ usage.md >usage.md.tmp && mv usage.md.tmp usage.md) + printf '%s\n' '{"type":"item.completed","item":{"id":"small-shell-edit","type":"command_execution","command":"cd docs && sed s/teh/Wrong/ usage.md >usage.md.tmp && mv usage.md.tmp usage.md","exit_code":0,"status":"completed","aggregated_output":""}}' + fi + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" != "small-command-only" && "${FAKE_PATTERN_EVENT_MODE:-}" != "small-command-only-wrong-final" ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"small-change","type":"file_change","changes":[{"path":"docs/usage.md","kind":"update"}]}}' + fi if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-external-symlink" ]]; then external_target="$workspace/../external-usage.md" cp "$workspace/docs/usage.md" "$external_target" @@ -6001,6 +6012,81 @@ else fail "small-fix evaluation did not require discovery-before-change or reject artifact-only and explicit external-read fixture evidence" fi +test_start "a command-only edit with correct final content is unavailable when file-change telemetry is missing" +small_command_only_output="$fixture_root/small-command-only-output" +small_command_only_wrong_output="$fixture_root/small-command-only-wrong-final-output" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-command-only "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_command_only_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'length == 2 and all(.[]; .status == "adapter_unavailable" and .error.code == "unknown_event_shape" and (has("metrics") | not))' "$small_command_only_output/traces/"*.json >/dev/null \ + && jq -e '.complete_pairs == 0 and .excluded_incomplete_pairs == 1 and .incomplete_pairs[0].case_id == "small-fix-stays-lightweight"' "$small_command_only_output/comparison.json" >/dev/null \ + && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-command-only-wrong-final "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_command_only_wrong_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'length == 2 and all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$small_command_only_wrong_output/traces/"*.json >/dev/null; then + pass +else + fail "the evaluator did not separate missing edit telemetry from an incorrect final workspace" +fi + +test_start "missing file-change telemetry does not mask plan, scope, or broad-response failures" +small_command_only_mixed_failures=() +for small_command_only_control in plan scope broad; do + small_command_only_mixed_output="$fixture_root/small-command-only-$small_command_only_control-output" + rm -f "$capture"/* + case "$small_command_only_control" in + plan) + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-command-only FAKE_SMALL_PLAN_MODE=full "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_command_only_mixed_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e ' + length == 2 + and all(.[] | select(.variant == "baseline"); .status == "adapter_unavailable" and .error.code == "unknown_event_shape" and (has("metrics") | not)) + and all(.[] | select(.variant == "candidate"); + .status == "completed" and .metrics.acceptance_passed == false + and .execution.verifier.workspace_failure_ids == ["workspace-002", "workspace-003"] + and .execution.verifier.scope_deviations == 0) + ' "$small_command_only_mixed_output/traces/"*.json >/dev/null; then + small_command_only_mixed_failures+=("plan") + fi + ;; + scope) + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-command-only FAKE_SCOPE_DEVIATION=true "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_command_only_mixed_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e ' + length == 2 and all(.[]; + .status == "completed" and .metrics.acceptance_passed == false + and .execution.verifier.workspace_failure_ids == ["workspace-003", "workspace-999"] + and .execution.verifier.scope_deviations == 1) + ' "$small_command_only_mixed_output/traces/"*.json >/dev/null; then + small_command_only_mixed_failures+=("scope") + fi + ;; + broad) + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-command-only FAKE_SMALL_BROAD=true "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_command_only_mixed_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e ' + length == 2 and all(.[]; + .status == "completed" and .metrics.acceptance_passed == false + and .execution.verifier.forbidden_hits == 1 + and .execution.verifier.workspace_failure_ids == ["workspace-003"] + and .execution.verifier.scope_deviations == 0) + ' "$small_command_only_mixed_output/traces/"*.json >/dev/null; then + small_command_only_mixed_failures+=("broad") + fi + ;; + esac +done +if [[ ${#small_command_only_mixed_failures[@]} -eq 0 ]]; then + pass +else + fail "command-only unavailable classification hid independently observed failures: ${small_command_only_mixed_failures[*]}" +fi + rm -f "$capture"/* if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-external-symlink "$runner" --execute \ --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ @@ -6244,24 +6330,47 @@ parallel_unavailable_output="$fixture_root/parallel-unavailable-output" rm -f "$capture"/* if FAKE_CODEX_CAPTURE_DIR="$capture" "$runner" --execute \ --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ - --cases isolated-parallel-a-b-then-c-with-integration --repeats 1 \ + --cases isolated-parallel-a-b-then-c-with-integration --repeats 3 \ --output "$parallel_unavailable_output" --codex-bin "$fake_codex" >/dev/null \ - && jq -s -e 'length == 2 and all(.[]; + && jq -s -e 'length == 6 and all(.[]; .case_id == "isolated-parallel-a-b-then-c-with-integration" and .status == "adapter_unavailable" and (has("metrics") | not) and .error.code == "unknown_event_shape") ' "$parallel_unavailable_output/traces/"*.json >/dev/null \ + && [[ "$(find "$parallel_unavailable_output/traces" -maxdepth 1 -type f -name '*.json' | wc -l | tr -d ' ')" -eq 6 ]] \ && jq -e ' .complete_pairs == 0 - and .excluded_incomplete_pairs == 1 - and .incomplete_pairs[0].case_id == "isolated-parallel-a-b-then-c-with-integration" + and .excluded_incomplete_pairs == 3 + and all(.incomplete_pairs[]; .case_id == "isolated-parallel-a-b-then-c-with-integration") ' "$parallel_unavailable_output/comparison.json" >/dev/null \ && [[ "$(find "$capture" -maxdepth 1 -name 'call-*.args' | wc -l | tr -d ' ')" -eq 0 ]] \ - && jq -s -e 'length == 2 and all(.[]; .state == "completed" and (.attempt_started_at | type == "array" and length == 0))' "$parallel_unavailable_output/run-attempts/"*.json >/dev/null; then + && jq -s -e 'length == 6 and all(.[]; .state == "completed" and (.attempt_started_at | type == "array" and length == 0))' "$parallel_unavailable_output/run-attempts/"*.json >/dev/null; then pass else - fail "isolated parallel evaluation accepted model narrative or promoted unavailable overlap telemetry" + fail "repeated isolated parallel unavailable pairs tripped the breaker or promoted unavailable overlap telemetry" +fi + +test_start "isolated unavailable pairs do not hide genuine incomplete-pair breaker failures" +parallel_mixed_output="$fixture_root/parallel-mixed-incomplete-output" +parallel_mixed_error="$fixture_root/parallel-mixed-incomplete-error.txt" +rm -f "$capture"/* +if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_CODEX_FAILURE_MESSAGE='network unavailable' \ + "$runner" --execute --model test-model \ + --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases isolated-parallel-a-b-then-c-with-integration,small-fix-stays-lightweight,requirements-map-through-completion,medium-final-handoff-is-reconstructable \ + --repeats 1 --output "$parallel_mixed_output" --codex-bin "$fake_codex" \ + >/dev/null 2>"$parallel_mixed_error"; then + fail "a real incomplete pair stopped tripping the breaker after deterministic pairs were excluded" +elif grep -Fq 'Stopped after 2 incomplete pairs' "$parallel_mixed_error" \ + && [[ "$(find "$capture" -maxdepth 1 -name 'call-*.args' | wc -l | tr -d ' ')" -eq 4 ]] \ + && [[ "$(jq -s '[.[] | select(.case_id == "isolated-parallel-a-b-then-c-with-integration" and .status == "adapter_unavailable" and .error.code == "unknown_event_shape" and .execution.exit_code == 0)] | length' "$parallel_mixed_output/traces/"*.json)" -eq 2 ]] \ + && [[ "$(jq -s '[.[] | select(.state == "completed")] | length' "$parallel_mixed_output/run-attempts/"*.json)" -eq 6 ]] \ + && [[ "$(jq -s '[.[] | select(.state == "not_started")] | length' "$parallel_mixed_output/run-attempts/"*.json)" -eq 2 ]] \ + && [[ ! -e "$parallel_mixed_output/comparison.json" ]]; then + pass +else + fail "the breaker did not ignore only known pre-dispatch unavailable pairs while retaining the two-real-pair stop" fi p0p4_finish_suite "${BASH_SOURCE[0]}" diff --git a/tests/p0-p4/progressive-discovery-contracts.sh b/tests/p0-p4/progressive-discovery-contracts.sh index 5c7a2ce..84deea1 100644 --- a/tests/p0-p4/progressive-discovery-contracts.sh +++ b/tests/p0-p4/progressive-discovery-contracts.sh @@ -355,6 +355,12 @@ write_workflow_eval_responses() { required_summary="$(jq -r --arg case_id "$case_id" '.cases[] | select(.id == $case_id) | .machine_expectations.required_substrings[]' "$fixture" | paste -sd ' ' -)" if jq -e --arg case_id "$case_id" '.cases[] | select(.id == $case_id) | (.machine_expectations.structured_json_assertions? // []) | length > 0' "$fixture" >/dev/null; then case "$case_id" in + native-slice-execution-uses-dependencies-not-runner-topology) + jq -n '{execution_policy:{source_writer_policy:"sequential_shared_or_unknown",read_only_analysis_policy:"parallel_permitted",isolation_evidence_ref:"not_available",c_start_decisions:[{a_status:"PENDING",c_decision:"blocked"},{a_status:"RUNNING",c_decision:"blocked"},{a_status:"VERIFIED",c_decision:"ready"}],per_slice_verification:"required",integration_validation:"required",integration_checks:["cross-slice","full-scope"],fresh_review:"required",fresh_review_after:"integration_validation"}}' >"$response_path" + ;; + isolated-independent-slices-integrate-before-review) + jq -n '{execution_policy:{source_writer_policy:"isolated_A_B_overlap_permitted",read_only_analysis_policy:"parallel_permitted",isolation_evidence_ref:"fixture-runtime-isolation-A-B-v1",c_start_decisions:[{a_status:"PENDING",c_decision:"blocked"},{a_status:"RUNNING",c_decision:"blocked"},{a_status:"VERIFIED",c_decision:"ready"}],per_slice_verification:"required",integration_validation:"required",integration_checks:["cross-slice","full-scope"],fresh_review:"required",fresh_review_after:"integration_validation"}}' >"$response_path" + ;; verification-reuse-current-scenario-matrix|verification-reuse-preserves-independent-review) write_verification_reuse_response "$case_id" "$response_path" "$required_summary" ;; diff --git a/tests/p0-p4/skill-eval-contracts.sh b/tests/p0-p4/skill-eval-contracts.sh index 3665fe8..52bd003 100644 --- a/tests/p0-p4/skill-eval-contracts.sh +++ b/tests/p0-p4/skill-eval-contracts.sh @@ -399,6 +399,12 @@ p0p4_write_skill_eval_responses() { if [[ "$skill_name" == "assistant-workflow" ]] \ && jq -e --arg id "$id" '.cases[] | select(.id == $id) | (.machine_expectations.structured_json_assertions? // []) | length > 0' "$fixture_file" >/dev/null; then case "$id" in + native-slice-execution-uses-dependencies-not-runner-topology) + jq -n '{execution_policy:{source_writer_policy:"sequential_shared_or_unknown",read_only_analysis_policy:"parallel_permitted",isolation_evidence_ref:"not_available",c_start_decisions:[{a_status:"PENDING",c_decision:"blocked"},{a_status:"RUNNING",c_decision:"blocked"},{a_status:"VERIFIED",c_decision:"ready"}],per_slice_verification:"required",integration_validation:"required",integration_checks:["cross-slice","full-scope"],fresh_review:"required",fresh_review_after:"integration_validation"}}' >"$response_path" + ;; + isolated-independent-slices-integrate-before-review) + jq -n '{execution_policy:{source_writer_policy:"isolated_A_B_overlap_permitted",read_only_analysis_policy:"parallel_permitted",isolation_evidence_ref:"fixture-runtime-isolation-A-B-v1",c_start_decisions:[{a_status:"PENDING",c_decision:"blocked"},{a_status:"RUNNING",c_decision:"blocked"},{a_status:"VERIFIED",c_decision:"ready"}],per_slice_verification:"required",integration_validation:"required",integration_checks:["cross-slice","full-scope"],fresh_review:"required",fresh_review_after:"integration_validation"}}' >"$response_path" + ;; verification-reuse-current-scenario-matrix|verification-reuse-preserves-independent-review) write_verification_reuse_response "$id" "$response_path" "$required_summary" ;; @@ -4245,6 +4251,84 @@ else fail "array_object_values_exact does not distinguish present null from missing keys: ${object_null_failures[*]}" fi +test_start "object_keys_exact rejects extra and missing root and nested keys" +object_keys_root="$(mktemp -d "${TMPDIR:-/tmp}/skill-eval-object-keys.XXXXXX")" +object_keys_skill="$object_keys_root/assistant-workflow" +object_keys_responses="$object_keys_root/responses" +object_keys_err="$object_keys_root/validation.err" +object_keys_output="$object_keys_root/grading.out" +p0p4_register_cleanup "$object_keys_root" +mkdir -p "$object_keys_skill/evals" "$object_keys_responses/assistant-workflow" +cp "$FRAMEWORK_DIR/skills/assistant-workflow/SKILL.md" "$object_keys_skill/SKILL.md" +ln -s "$FRAMEWORK_DIR/skills/assistant-workflow/contracts" "$object_keys_skill/contracts" +jq ' + .cases |= map( + if .id == "native-slice-execution-uses-dependencies-not-runner-topology" + or .id == "isolated-independent-slices-integrate-before-review" then + .machine_expectations.structured_json_assertions += [ + {operator:"object_keys_exact",path:[],fields:["execution_policy"]}, + {operator:"object_keys_exact",path:["execution_policy"],fields:["source_writer_policy","read_only_analysis_policy","isolation_evidence_ref","c_start_decisions","per_slice_verification","integration_validation","integration_checks","fresh_review","fresh_review_after"]}, + {operator:"object_keys_exact",path:["execution_policy","c_start_decisions",0],fields:["a_status","c_decision"]}, + {operator:"object_keys_exact",path:["execution_policy","c_start_decisions",1],fields:["a_status","c_decision"]}, + {operator:"object_keys_exact",path:["execution_policy","c_start_decisions",2],fields:["a_status","c_decision"]} + ] + else . end + ) +' "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json" >"$object_keys_skill/evals/cases.json" +cp "$object_keys_skill/evals/cases.json" "$object_keys_root/base-cases.json" +cat >"$object_keys_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"sequential_shared_or_unknown","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"not_available","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +cat >"$object_keys_responses/assistant-workflow/isolated-independent-slices-integrate-before-review.txt" <<'EOF' +{"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"fixture-runtime-isolation-A-B-v1","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} +EOF +object_keys_failures=() +if ! "$skill_eval_runner" --validate-fixture --skill "$object_keys_skill" >/dev/null 2>"$object_keys_err"; then + object_keys_failures+=("valid fixture: $(cat "$object_keys_err")") +else + for object_keys_case in native-slice-execution-uses-dependencies-not-runner-topology isolated-independent-slices-integrate-before-review; do + if ! "$skill_eval_runner" --responses "$object_keys_responses" --skill "$object_keys_skill" --case "$object_keys_case" >"$object_keys_output" 2>&1; then + object_keys_failures+=("valid response:$object_keys_case") + fi + done + object_keys_baseline="$object_keys_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" + cp "$object_keys_baseline" "$object_keys_root/valid-response.json" + for object_keys_mutation in \ + 'setpath(["unsafe_override"]; true)' \ + 'del(.execution_policy)' \ + 'setpath(["execution_policy","unsafe_override"]; true)' \ + 'setpath(["execution_policy","c_start_decisions",0,"unsafe_override"]; true)' \ + 'del(.execution_policy.fresh_review)' \ + 'del(.execution_policy.c_start_decisions[0].a_status)' \ + '.execution_policy = "unsafe"' \ + '.execution_policy.c_start_decisions[0] = "ready"'; do + jq "$object_keys_mutation" "$object_keys_root/valid-response.json" >"$object_keys_baseline" + if "$skill_eval_runner" --responses "$object_keys_responses" --skill "$object_keys_skill" --case native-slice-execution-uses-dependencies-not-runner-topology >"$object_keys_output" 2>&1 \ + || ! grep -Fq "structured JSON assertion failure" "$object_keys_output"; then + object_keys_failures+=("response mutation:$object_keys_mutation") + fi + cp "$object_keys_root/valid-response.json" "$object_keys_baseline" + done +fi +for object_keys_invalid in \ + '{"operator":"object_keys_exact","path":[],"fields":["unsafe_override"]}' \ + '{"operator":"object_keys_exact","path":[],"fields":["execution_policy","execution_policy"]}' \ + '{"operator":"object_keys_exact","path":["execution_policy"],"fields":["unsafe_override"]}' \ + '{"operator":"object_keys_exact","path":["execution_policy","source_writer_policy"],"fields":["value"]}' \ + '{"operator":"object_keys_exact","path":["execution_policy"],"fields":["source_writer_policy","source_writer_policy"]}'; do + jq --argjson assertion "$object_keys_invalid" '(.cases[] | select(.id == "native-slice-execution-uses-dependencies-not-runner-topology") | .machine_expectations.structured_json_assertions) += [$assertion]' "$object_keys_root/base-cases.json" >"$object_keys_root/invalid.json" + mv "$object_keys_root/invalid.json" "$object_keys_skill/evals/cases.json" + if "$skill_eval_runner" --validate-fixture --skill "$object_keys_skill" >/dev/null 2>"$object_keys_err"; then + object_keys_failures+=("invalid assertion accepted:$object_keys_invalid") + fi + cp "$object_keys_root/base-cases.json" "$object_keys_skill/evals/cases.json" +done +if [[ "${#object_keys_failures[@]}" -eq 0 ]]; then + pass +else + fail "exact object-key assertions accepted unsafe response keys or malformed schema paths: ${object_keys_failures[*]}" +fi + test_start "structured assertion declaration bounds accept exact limits and reject one-over limits" structured_bounds_root="$(mktemp -d "${TMPDIR:-/tmp}/skill-eval-structured-bounds.XXXXXX")" structured_bounds_skill="$structured_bounds_root/assistant-eval-structured-bounds" @@ -4595,6 +4679,7 @@ if grep -Fq "default eval inventory is 14 first-class \`assistant-*\` skills wit && grep -Fq '`array_type` requires the target path to resolve to an array and permits an empty array.' "$FRAMEWORK_DIR/docs/evals/README.md" \ && grep -Fq '`path_absent`' "$FRAMEWORK_DIR/docs/evals/README.md" \ && grep -Fq '`array_object_values_exact`' "$FRAMEWORK_DIR/docs/evals/README.md" \ + && grep -Fq '`object_keys_exact`' "$FRAMEWORK_DIR/docs/evals/README.md" \ && grep -Fq 'In this exhaustive fixed operator list, `path_absent` passes only when its target' "$FRAMEWORK_DIR/docs/evals/README.md" \ && grep -Fq 'path cannot resolve; a present `null` value is present and therefore fails.' "$FRAMEWORK_DIR/docs/evals/README.md" \ && grep -Fq '`absent_or_empty_array` passes when its target path is unresolved or resolves to' "$FRAMEWORK_DIR/docs/evals/README.md" \ @@ -4606,6 +4691,7 @@ if grep -Fq "default eval inventory is 14 first-class \`assistant-*\` skills wit && grep -Fq '`path_absent` accepts only absence; present `null` fails.' "$FRAMEWORK_DIR/docs/skill-contract-design-guide.md" \ && grep -Fq '`absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail.' "$FRAMEWORK_DIR/docs/skill-contract-design-guide.md" \ && grep -Fq '`array_object_values_exact` projects every target-array object' "$FRAMEWORK_DIR/docs/skill-contract-design-guide.md" \ + && grep -Fq '`object_keys_exact` requires an object whose keys exactly match' "$FRAMEWORK_DIR/docs/skill-contract-design-guide.md" \ && grep -Fq 'at most 16 unique projected fields and 32 expected objects' "$FRAMEWORK_DIR/docs/skill-contract-design-guide.md" \ && grep -Fq 'unordered multiset, preserves field correlation, and treats' "$FRAMEWORK_DIR/docs/skill-contract-design-guide.md" \ && grep -Fq 'an absent field differently from a present `null`.' "$FRAMEWORK_DIR/docs/skill-contract-design-guide.md" \ @@ -4617,7 +4703,7 @@ else fi test_start "skill eval docs enumerate the exact canonical structured operator list" -expected_structured_operator_names='equals one_of nonempty_string nonempty_array empty_array array_type array_nonblank_strings path_absent absent_or_empty_array equals_path required_when_equals array_field_values_exact array_object_values_exact array_items_nonempty_fields array_items_nonempty_array_fields' +expected_structured_operator_names='equals one_of nonempty_string nonempty_array empty_array array_type array_nonblank_strings path_absent absent_or_empty_array equals_path required_when_equals array_field_values_exact array_object_values_exact object_keys_exact array_items_nonempty_fields array_items_nonempty_array_fields' structured_operator_list_is_exact() { local document="$1" source_kind="$2" paragraph actual case "$source_kind" in diff --git a/tests/p0-p4/task-packet-contracts.sh b/tests/p0-p4/task-packet-contracts.sh index 2903717..5a3f2d2 100644 --- a/tests/p0-p4/task-packet-contracts.sh +++ b/tests/p0-p4/task-packet-contracts.sh @@ -240,6 +240,13 @@ if ! ruby -rjson -e ' assertion["fields"] == ["a_status", "c_decision"] && assertion["expected_objects"] == expected_decisions end end + exact_keys = lambda do |assertions, path, fields| + assertions.any? { |assertion| assertion["operator"] == "object_keys_exact" && assertion["path"] == path && assertion["fields"] == fields } + end + policy_fields = %w[source_writer_policy read_only_analysis_policy isolation_evidence_ref c_start_decisions per_slice_verification integration_validation integration_checks fresh_review fresh_review_after] + decision_key_assertions = lambda do |assertions| + (0..2).all? { |index| exact_keys.call(assertions, ["execution_policy", "c_start_decisions", index], %w[a_status c_decision]) } + end shared_assertions = structured.call(shared) isolated_assertions = structured.call(isolated) valid = shared_expected.include?("Sequences source-changing A and B because the workspace is shared or isolation is unknown") && @@ -256,6 +263,11 @@ if ! ruby -rjson -e ' equals.call(shared_assertions, ["execution_policy", "isolation_evidence_ref"], "not_available") && equals.call(isolated_assertions, ["execution_policy", "source_writer_policy"], "isolated_A_B_overlap_permitted") && equals.call(isolated_assertions, ["execution_policy", "isolation_evidence_ref"], "fixture-runtime-isolation-A-B-v1") && + exact_keys.call(shared_assertions, [], ["execution_policy"]) && + exact_keys.call(shared_assertions, ["execution_policy"], policy_fields) && + exact_keys.call(isolated_assertions, [], ["execution_policy"]) && + exact_keys.call(isolated_assertions, ["execution_policy"], policy_fields) && + decision_key_assertions.call(shared_assertions) && decision_key_assertions.call(isolated_assertions) && decisions.call(shared_assertions) && decisions.call(isolated_assertions) exit(valid ? 0 : 1) ' "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json"; then @@ -337,7 +349,8 @@ workflow_eval_missing_seed_responses="$workflow_eval_root/missing-seed" workflow_eval_wrong_writer_responses="$workflow_eval_root/wrong-writer" workflow_eval_wrong_integration_responses="$workflow_eval_root/wrong-integration" workflow_eval_wrong_review_responses="$workflow_eval_root/wrong-review" -mkdir -p "$workflow_eval_responses/assistant-workflow" "$workflow_eval_unsafe_responses/assistant-workflow" "$workflow_eval_missing_seed_responses/assistant-workflow" "$workflow_eval_wrong_writer_responses/assistant-workflow" "$workflow_eval_wrong_integration_responses/assistant-workflow" "$workflow_eval_wrong_review_responses/assistant-workflow" +workflow_eval_extra_responses="$workflow_eval_root/extra-keys" +mkdir -p "$workflow_eval_responses/assistant-workflow" "$workflow_eval_unsafe_responses/assistant-workflow" "$workflow_eval_missing_seed_responses/assistant-workflow" "$workflow_eval_wrong_writer_responses/assistant-workflow" "$workflow_eval_wrong_integration_responses/assistant-workflow" "$workflow_eval_wrong_review_responses/assistant-workflow" "$workflow_eval_extra_responses/assistant-workflow" cat >"$workflow_eval_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" <<'EOF' {"execution_policy":{"source_writer_policy":"sequential_shared_or_unknown","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"not_available","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"integration_validation"}} EOF @@ -362,6 +375,7 @@ EOF cat >"$workflow_eval_wrong_review_responses/assistant-workflow/isolated-independent-slices-integrate-before-review.txt" <<'EOF' {"execution_policy":{"source_writer_policy":"isolated_A_B_overlap_permitted","read_only_analysis_policy":"parallel_permitted","isolation_evidence_ref":"fixture-runtime-isolation-A-B-v1","c_start_decisions":[{"a_status":"PENDING","c_decision":"blocked"},{"a_status":"RUNNING","c_decision":"blocked"},{"a_status":"VERIFIED","c_decision":"ready"}],"per_slice_verification":"required","integration_validation":"required","integration_checks":["cross-slice","full-scope"],"fresh_review":"required","fresh_review_after":"per_slice_verification"}} EOF +cp "$workflow_eval_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" "$workflow_eval_extra_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" workflow_eval_grading_failures=() for workflow_case in native-slice-execution-uses-dependencies-not-runner-topology isolated-independent-slices-integrate-before-review; do if ! "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --responses "$workflow_eval_responses" --skill assistant-workflow --case "$workflow_case" >"$workflow_eval_root/$workflow_case-positive.out"; then @@ -372,6 +386,18 @@ for workflow_case in native-slice-execution-uses-dependencies-not-runner-topolog workflow_eval_grading_failures+=("$workflow_case:unsafe-dependent-start") fi done +for extra_key_mutation in \ + 'setpath(["unsafe_override"]; true)' \ + 'setpath(["execution_policy","unsafe_override"]; true)' \ + 'setpath(["execution_policy","c_start_decisions",0,"unsafe_override"]; true)'; do + extra_key_response="$workflow_eval_extra_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" + jq "$extra_key_mutation" "$workflow_eval_responses/assistant-workflow/native-slice-execution-uses-dependencies-not-runner-topology.txt" >"$workflow_eval_root/extra-key-mutated.json" + mv "$workflow_eval_root/extra-key-mutated.json" "$extra_key_response" + if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --responses "$workflow_eval_extra_responses" --skill assistant-workflow --case native-slice-execution-uses-dependencies-not-runner-topology >"$workflow_eval_root/extra-key.out" 2>&1 \ + || ! grep -Fq -- "structured JSON assertion failure" "$workflow_eval_root/extra-key.out"; then + workflow_eval_grading_failures+=("extra-key:$extra_key_mutation") + fi +done if "$FRAMEWORK_DIR/tools/evals/run-skill-evals.sh" --responses "$workflow_eval_missing_seed_responses" --skill assistant-workflow --case isolated-independent-slices-integrate-before-review >"$workflow_eval_root/isolated-missing-seed.out" 2>&1 \ || ! grep -Fq -- "structured JSON assertion failure" "$workflow_eval_root/isolated-missing-seed.out"; then workflow_eval_grading_failures+=("isolated-independent-slices-integrate-before-review:missing-seeded-isolation-evidence") diff --git a/tools/evals/lib/skill-eval-fixtures.sh b/tools/evals/lib/skill-eval-fixtures.sh index c07de17..47e5ee6 100644 --- a/tools/evals/lib/skill-eval-fixtures.sh +++ b/tools/evals/lib/skill-eval-fixtures.sh @@ -266,7 +266,7 @@ path_operands = lambda do |assertion| operands << ["other_path", assertion["other_path"]] if assertion.key?("other_path") operands << ["when_path", assertion["when_path"]] if assertion.key?("when_path") operands << ["field", assertion["path"] + [0, assertion["field"]]] if assertion["field"].is_a?(String) && assertion["path"].is_a?(Array) - if assertion["fields"].is_a?(Array) && assertion["path"].is_a?(Array) + if assertion["operator"] != "object_keys_exact" && assertion["fields"].is_a?(Array) && assertion["path"].is_a?(Array) assertion["fields"].each { |field| operands << ["fields", assertion["path"] + [0, field]] if field.is_a?(String) } end if assertion["expected_objects"].is_a?(Array) && assertion["path"].is_a?(Array) @@ -295,9 +295,12 @@ admitted_literal = lambda do |path, value| resolve.call(path).any? { |field| literal_valid.call(field, value) } end -fixture.fetch("cases", []).each do |test_case| + fixture.fetch("cases", []).each do |test_case| Array(test_case.dig("machine_expectations", "structured_json_assertions")).each_with_index do |assertion, index| path_operands.call(assertion).each do |operand, path| + if assertion["operator"] == "object_keys_exact" && operand == "path" && path == [] + next + end unless path.is_a?(Array) && path.first.is_a?(String) warn "case #{test_case.fetch("id")}.machine_expectations.structured_json_assertions[#{index}] invalid assertion path #{operand}: #{path.to_json}" exit 1 @@ -318,6 +321,36 @@ fixture.fetch("cases", []).each do |test_case| exit 1 end + if assertion["operator"] == "object_keys_exact" + path = assertion["path"] + fields = assertion["fields"] + if path.empty? + unknown_fields = fields.reject { |field| roots.key?(field) } + unless unknown_fields.empty? + warn "case #{test_case.fetch("id")}.machine_expectations.structured_json_assertions[#{index}] unknown assertion root object key: #{unknown_fields.first.to_json}" + exit 1 + end + else + target = resolve.call(path) + target_field = target.length == 1 ? target.first : nil + object_array_item = target_field && target_field["type"] == "object[]" && path.last.is_a?(Numeric) + unless target_field && (target_field["type"] == "object" || object_array_item) && target_field["object_fields"].is_a?(Array) + warn "case #{test_case.fetch("id")}.machine_expectations.structured_json_assertions[#{index}] object_keys_exact target must be a declared object: #{path.to_json}" + exit 1 + end + declared_fields = target_field.fetch("object_fields").map { |field| field["name"] }.sort + unless fields.sort == declared_fields + undeclared_fields = fields - declared_fields + if undeclared_fields.empty? + warn "case #{test_case.fetch("id")}.machine_expectations.structured_json_assertions[#{index}] object_keys_exact fields must exactly match the declared object schema: #{path.to_json}" + else + warn "case #{test_case.fetch("id")}.machine_expectations.structured_json_assertions[#{index}] undeclared assertion path fields: #{undeclared_fields.to_json}" + end + exit 1 + end + end + end + literal_error = case assertion["operator"] when "equals" !admitted_literal.call(assertion["path"], assertion["expected"]) @@ -904,6 +937,11 @@ validate_fixture() { if (.path? | json_path | not) or ($fields | distinct_bounded_fields | not) or (.expected_objects? | exact_expected_objects($fields) | not) then "case[\($index)].machine_expectations.structured_json_assertions[\($assertion_index)] invalid array_object_values_exact assertion" else empty end + elif .operator == "object_keys_exact" then + .fields as $fields | + if (.path? | type != "array") or ((.path | length) > 0 and (.path | json_path | not)) or ($fields | distinct_bounded_fields | not) or (($fields | unique | length) != ($fields | length)) then + "case[\($index)].machine_expectations.structured_json_assertions[\($assertion_index)] invalid object_keys_exact assertion" + else empty end else "case[\($index)].machine_expectations.structured_json_assertions[\($assertion_index)] unsupported operator: \(.operator)" end; diff --git a/tools/evals/lib/skill-eval-grade.sh b/tools/evals/lib/skill-eval-grade.sh index 878147a..d496b00 100644 --- a/tools/evals/lib/skill-eval-grade.sh +++ b/tools/evals/lib/skill-eval-grade.sh @@ -1261,6 +1261,10 @@ count_structured_json_assertion_failures() { and all(value_at($assertion.path)[]; . as $item | type == "object" and all($assertion.fields[]; . as $field | $item | has($field))) and (object_field_tuples(value_at($assertion.path); $assertion.fields) | sort) == (object_field_tuples($assertion.expected_objects; $assertion.fields) | sort) + elif $assertion.operator == "object_keys_exact" then + path_exists($assertion.path) + and (value_at($assertion.path) | type == "object") + and (value_at($assertion.path) | keys | sort) == ($assertion.fields | sort) else false end ' "$response_path" >/dev/null; then failures=$((failures + 1)) diff --git a/tools/evals/run-codex-framework-evals.sh b/tools/evals/run-codex-framework-evals.sh index bc0b26b..2746937 100755 --- a/tools/evals/run-codex-framework-evals.sh +++ b/tools/evals/run-codex-framework-evals.sh @@ -598,8 +598,21 @@ enforce_incomplete_pair_breaker() { max_incomplete_pairs="$(jq -r '.max_incomplete_pairs' "$OUTPUT_DIR/run-plan.json")" fi incomplete_pairs="$(jq -s ' + def deterministic_pre_dispatch_isolated_parallel_pair: + length == 2 + and ([.[] | .variant] | sort == ["baseline", "candidate"]) + and ([.[] | .trial_index] | unique | length == 1) + and all(.[]; + .case_id == "isolated-parallel-a-b-then-c-with-integration" + and .status == "adapter_unavailable" + and .error.code == "unknown_event_shape" + and .execution.exit_code == 0 + and .execution.raw_artifacts_retained == false + and (has("metrics") | not)); group_by(.pair_id) - | map(select(length == 2 and any(.[]; .status != "completed"))) + | map(select(length == 2 + and any(.[]; .status != "completed") + and (deterministic_pre_dispatch_isolated_parallel_pair | not))) | length ' "$OUTPUT_DIR"/traces/*.json)" [[ "$max_incomplete_pairs" =~ ^[0-9]+$ && "$incomplete_pairs" -le "$max_incomplete_pairs" ]] \ @@ -2355,10 +2368,26 @@ small_fix_event_evidence() { .type == "item.completed" and .item.type == "file_change" and changes_containing("docs/usage.md"); + def is_successful_command: + .type == "item.completed" + and .item.type == "command_execution" + and .item.exit_code == 0 + and (is_exact_command("rg -n teh docs/usage.md") | not); + def has_file_change_event: + any(.[]; + (.type == "item.started" or .type == "item.completed" or .type == "item.updated") + and .item.type == "file_change"); . as $events | [range(0; length) | select($events[.] | is_discovery)] as $discoveries | [range(0; length) | select($events[.] | is_change)] as $changes + | [range(0; length) + | . as $index + | $events[$index] + | select(is_successful_command) + | {index: $index} + ] as $successful_commands + | ($events | has_file_change_event) as $has_file_change_event | [range(0; length) | . as $index | $events[$index] @@ -2391,12 +2420,22 @@ small_fix_event_evidence() { and ($events[.].item.type == "mcp_tool_call" or $events[.].item.type == "web_search"))] as $disallowed | { source_discovery_before_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($changes | length) > 0 and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and $discoveries[0] < $changes[0]), + command_only_change_without_file_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($successful_commands | length) > 0 and ($changes | length) == 0 and ($has_file_change_event | not) and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and any($successful_commands[]; .index > $discoveries[0])), disallowed_item_count: ($disallowed | length) } JQ )" "$jsonl" } +small_fix_command_only_edit_missing_telemetry() { + local jsonl="$1" workspace="$2" event_evidence + event_evidence="$(small_fix_event_evidence "$jsonl" "$workspace")" || return 1 + # Without a file_change event, command text cannot prove that the command + # edited this file or establish write timing. Keep the result unavailable. + jq -e '.command_only_change_without_file_change == true and .disallowed_item_count == 0' \ + <<<"$event_evidence" >/dev/null +} + observed_event_shape_supported() { local jsonl="$1" allowed_item_types="$2" jq -se --argjson allowed_item_types "$allowed_item_types" ' @@ -3573,7 +3612,6 @@ execute_one_run() { rm -rf "$run_raw" return fi - if [[ "$case_id" == "pivot-restart-on-stagnation-or-code-writer-blocker" ]] \ && ! stagnation_event_shape_supported "$jsonl"; then write_unavailable_trace "$trace_path" "$run_id" "$pair_id" "$case_id" "$trial_index" "$variant" \ @@ -3608,6 +3646,21 @@ execute_one_run() { fi response_verifier="$(grade_response "$case_id" "$final_output" "$semantic_extract_path")" workspace_verifier="$(verify_workspace "$case_id" "$workspace" "$jsonl")" + if [[ "$case_id" == "small-fix-stays-lightweight" ]] \ + && small_fix_requires_unavailable_adapter_policy \ + && jq -n -e \ + --argjson response "$response_verifier" \ + --argjson workspace "$workspace_verifier" \ + '$response.status == "passed" + and $workspace.workspace_failure_ids == ["workspace-003"] + and $workspace.scope_deviations == 0' >/dev/null \ + && small_fix_command_only_edit_missing_telemetry "$jsonl" "$workspace"; then + write_unavailable_trace "$trace_path" "$run_id" "$pair_id" "$case_id" "$trial_index" "$variant" \ + unknown_event_shape 0 "$fixture_hash" "$case_digest" "$instruction_hash" "$grader_digest" "$cli_version" + mark_run_attempt_completed "$attempt_path" || die "Could not complete unavailable run-attempt state for $run_id." + rm -rf "$run_raw" + return + fi verifier="$(jq -cn \ --argjson response "$response_verifier" \ --argjson workspace "$workspace_verifier" \ From 0593434426af2aa404f2d218e349aaf2103cae25 Mon Sep 17 00:00:00 2001 From: Laimis Date: Wed, 23 Sep 2026 17:52:01 +0300 Subject: [PATCH 3/6] docs: keep assertion guide within context budget --- docs/skill-contract-design-guide.md | 2 +- .../references/skill-contract-design-guide.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/skill-contract-design-guide.md b/docs/skill-contract-design-guide.md index 32087dd..385b991 100644 --- a/docs/skill-contract-design-guide.md +++ b/docs/skill-contract-design-guide.md @@ -472,7 +472,7 @@ shapes native description routing and is separate from response-grade `.cases`. Local response grading is deterministic and heuristic: missing files, empty responses, fail-signal phrase hits, required substrings, and forbidden substrings. It is a provider-neutral proxy for behavior conformance and does not replace human or LLM semantic judgment. -Per-skill cases may optionally add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `object_keys_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty. `array_nonblank_strings` requires a string array of nonblank members; `allow_empty` controls zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership in bounded scalars. `array_object_values_exact` projects every target-array object to its bounded explicit `fields` list and compares those tuples as an unordered exact multiset against `expected_objects`; this preserves each declared field correlation without making object order significant. `object_keys_exact` requires an object whose keys exactly match its bounded, unique `fields` list; it rejects extra keys, missing keys, and non-object values. Its path may be `[]` to check the response root or a declared nested object. `array_items_nonempty_fields` requires a non-empty target array and non-empty string values for every listed field in every object. `array_items_nonempty_array_fields` requires a non-empty target array and a non-empty array of nonblank string values for every listed field in every object. Paths are JSON arrays of string object keys or non-negative integer array indexes; only `object_keys_exact` permits the empty root path. A selected structured case requires exactly one valid JSON response value. These declarations are grader-only data, never executable instructions: do not permit arbitrary jq, code, expressions, or fixture-provided operators outside this fixed set. +Per-skill cases may add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators: `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `object_keys_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty; `array_nonblank_strings` requires a string array of nonblank members, with `allow_empty` controlling zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership among bounded scalars. `array_object_values_exact` projects every target-array object to bounded explicit `fields` list and compares tuples with `expected_objects`. `object_keys_exact` requires an object whose keys exactly match its bounded, unique `fields` list; extra/missing keys or non-objects fail. `array_items_nonempty_fields` and `array_items_nonempty_array_fields` require a non-empty target array, with each listed field in each object respectively a non-empty string or non-empty array of nonblank strings. Paths are JSON arrays of string object keys or non-negative integer array indexes; only `object_keys_exact` accepts `[]` for the response root or a declared nested object. Selected structured cases require exactly one valid JSON response value. Declarations are grader-only data, not instructions: reject arbitrary jq, code, expressions, or fixture-defined operators. `one_of` accepts at most 32 declared scalar values. `array_object_values_exact` accepts at most 16 unique projected fields and 32 expected objects; it compares diff --git a/skills/assistant-skill-creator/references/skill-contract-design-guide.md b/skills/assistant-skill-creator/references/skill-contract-design-guide.md index 32087dd..385b991 100644 --- a/skills/assistant-skill-creator/references/skill-contract-design-guide.md +++ b/skills/assistant-skill-creator/references/skill-contract-design-guide.md @@ -472,7 +472,7 @@ shapes native description routing and is separate from response-grade `.cases`. Local response grading is deterministic and heuristic: missing files, empty responses, fail-signal phrase hits, required substrings, and forbidden substrings. It is a provider-neutral proxy for behavior conformance and does not replace human or LLM semantic judgment. -Per-skill cases may optionally add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `object_keys_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty. `array_nonblank_strings` requires a string array of nonblank members; `allow_empty` controls zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership in bounded scalars. `array_object_values_exact` projects every target-array object to its bounded explicit `fields` list and compares those tuples as an unordered exact multiset against `expected_objects`; this preserves each declared field correlation without making object order significant. `object_keys_exact` requires an object whose keys exactly match its bounded, unique `fields` list; it rejects extra keys, missing keys, and non-object values. Its path may be `[]` to check the response root or a declared nested object. `array_items_nonempty_fields` requires a non-empty target array and non-empty string values for every listed field in every object. `array_items_nonempty_array_fields` requires a non-empty target array and a non-empty array of nonblank string values for every listed field in every object. Paths are JSON arrays of string object keys or non-negative integer array indexes; only `object_keys_exact` permits the empty root path. A selected structured case requires exactly one valid JSON response value. These declarations are grader-only data, never executable instructions: do not permit arbitrary jq, code, expressions, or fixture-provided operators outside this fixed set. +Per-skill cases may add `machine_expectations.structured_json_assertions` when substring anchors cannot safely prove a typed response shape. Use only the fixed provider-neutral operators: `equals`, `one_of`, `nonempty_string`, `nonempty_array`, `empty_array`, `array_type`, `array_nonblank_strings`, `path_absent`, `absent_or_empty_array`, `equals_path`, `required_when_equals`, `array_field_values_exact`, `array_object_values_exact`, `object_keys_exact`, `array_items_nonempty_fields`, and `array_items_nonempty_array_fields`. `array_type` accepts arrays, including empty; `array_nonblank_strings` requires a string array of nonblank members, with `allow_empty` controlling zero members. `path_absent` accepts only absence; present `null` fails. `absent_or_empty_array` accepts only absence or `[]`; `null` and other values fail. `one_of` requires exact membership among bounded scalars. `array_object_values_exact` projects every target-array object to bounded explicit `fields` list and compares tuples with `expected_objects`. `object_keys_exact` requires an object whose keys exactly match its bounded, unique `fields` list; extra/missing keys or non-objects fail. `array_items_nonempty_fields` and `array_items_nonempty_array_fields` require a non-empty target array, with each listed field in each object respectively a non-empty string or non-empty array of nonblank strings. Paths are JSON arrays of string object keys or non-negative integer array indexes; only `object_keys_exact` accepts `[]` for the response root or a declared nested object. Selected structured cases require exactly one valid JSON response value. Declarations are grader-only data, not instructions: reject arbitrary jq, code, expressions, or fixture-defined operators. `one_of` accepts at most 32 declared scalar values. `array_object_values_exact` accepts at most 16 unique projected fields and 32 expected objects; it compares From 74e8bdfef28826443fdab56ac217fdb817135faa Mon Sep 17 00:00:00 2001 From: Laimis Date: Wed, 23 Sep 2026 18:36:36 +0300 Subject: [PATCH 4/6] fix: admit passive reasoning in observed eval streams --- .../p0-p4/codex-behavioral-eval-contracts.sh | 36 +++++++++++++++++++ tools/evals/run-codex-framework-evals.sh | 4 +-- 2 files changed, 38 insertions(+), 2 deletions(-) diff --git a/tests/p0-p4/codex-behavioral-eval-contracts.sh b/tests/p0-p4/codex-behavioral-eval-contracts.sh index 1dd79ce..40ca14b 100755 --- a/tests/p0-p4/codex-behavioral-eval-contracts.sh +++ b/tests/p0-p4/codex-behavioral-eval-contracts.sh @@ -699,6 +699,9 @@ printf '%s\n' '{"type":"item.completed","item":{"id":"item-1","type":"agent_mess if [[ -f "$workspace/docs/usage.md" ]]; then case "${FAKE_PATTERN_EVENT_MODE:-}" in ""|small-positive|small-wrapped-positive|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed|small-command-only|small-command-only-wrong-final) + if [[ "${FAKE_SMALL_PASSIVE_REASONING:-false}" == "true" ]]; then + jq -cn '{type:"item.completed",item:{id:"small-reasoning-before",type:"reasoning",text:"safe summary"}}' + fi small_discovery_command='"rg -n teh docs/usage.md"' if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-wrapped-positive" ]]; then small_discovery_command='["/bin/zsh","-lc","rg -n teh docs/usage.md"]' @@ -713,6 +716,9 @@ if [[ -f "$workspace/docs/usage.md" ]]; then printf '%s\n' '{"type":"item.started","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md"}}' fi jq -cn --argjson command "$small_discovery_command" '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:$command,exit_code:0,status:"completed",aggregated_output:"1:This fixture contains teh requested typo."}}' + if [[ "${FAKE_SMALL_PASSIVE_REASONING:-false}" == "true" ]]; then + jq -cn '{type:"item.completed",item:{id:"small-reasoning-between",type:"reasoning",text:"safe summary"}}' + fi if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-shell-edit-started-before-discovery" ]]; then printf '%s\n' '{"type":"item.completed","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md","exit_code":0,"status":"completed","aggregated_output":""}}' fi @@ -761,6 +767,9 @@ if [[ -f "$workspace/RECOVERY.md" ]]; then printf 'unsupported stagnation FAKE_PATTERN_EVENT_MODE: %s\n' "$pattern_mode" >&2 exit 2 fi + if [[ "${FAKE_STAGNATION_PASSIVE_REASONING:-false}" == "true" ]]; then + jq -cn '{type:"item.completed",item:{id:"stagnation-reasoning-before",type:"reasoning",text:"safe summary"}}' + fi if [[ "$pattern_mode" == stagnation-early-recovery-start ]]; then printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-recovery","type":"command_execution","command":"bash tests/recovery-contracts.sh"}}' elif [[ "$pattern_mode" == stagnation-early-fresh-start ]]; then @@ -768,6 +777,9 @@ if [[ -f "$workspace/RECOVERY.md" ]]; then fi if failure_output="$(bash "$workspace/tests/stagnation-contracts.sh")"; then failure_exit=0; else failure_exit=$?; fi printf '%s\n' "{\"type\":\"item.completed\",\"item\":{\"id\":\"stagnation-trusted-failure\",\"type\":\"command_execution\",\"command\":\"bash tests/stagnation-contracts.sh\",\"exit_code\":${failure_exit},\"status\":\"completed\",\"aggregated_output\":\"${failure_output}\"}}" + if [[ "${FAKE_STAGNATION_PASSIVE_REASONING:-false}" == "true" ]]; then + jq -cn '{type:"item.completed",item:{id:"stagnation-reasoning-between",type:"reasoning",text:"safe summary"}}' + fi if [[ "$pattern_mode" == stagnation-unmatched-recovery-start ]]; then printf '%s\n' '{"type":"item.started","item":{"id":"recovery-unmatched","type":"command_execution","command":"bash tests/recovery-contracts.sh"}}' fi @@ -6012,6 +6024,18 @@ else fail "small-fix evaluation did not require discovery-before-change or reject artifact-only and explicit external-read fixture evidence" fi +test_start "small-fix accepts passive reasoning before and between workspace events without counting it as an action" +small_reasoning_output="$fixture_root/small-passive-reasoning-output" +rm -f "$capture"/* +if FAKE_SMALL_PASSIVE_REASONING=true FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-positive "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_reasoning_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'length == 2 and all(.[]; .status == "completed" and .metrics.acceptance_passed == true and .metrics.tool_calls == 1 and .metrics.rework_count == 0)' "$small_reasoning_output/traces/"*.json >/dev/null; then + pass +else + fail "small-fix did not accept passive reasoning without counting it as workspace work" +fi + test_start "a command-only edit with correct final content is unavailable when file-change telemetry is missing" small_command_only_output="$fixture_root/small-command-only-output" small_command_only_wrong_output="$fixture_root/small-command-only-wrong-final-output" @@ -6196,6 +6220,18 @@ else fail "stagnation evaluation did not verify trusted failure, recovery, fresh check, and terminal incompleteness against bounded mutations" fi +test_start "stagnation accepts passive reasoning before and between workspace events without counting it as an action" +stagnation_reasoning_output="$fixture_root/stagnation-passive-reasoning-output" +rm -f "$capture"/* +if FAKE_STAGNATION_PASSIVE_REASONING=true FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=stagnation-positive "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 --output "$stagnation_reasoning_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'length == 2 and all(.[]; .status == "completed" and .metrics.acceptance_passed == true and .metrics.tool_calls == 3 and .metrics.rework_count == 0)' "$stagnation_reasoning_output/traces/"*.json >/dev/null; then + pass +else + fail "stagnation did not accept passive reasoning without counting it as workspace work" +fi + test_start "stagnation evaluation permits the declared read-only recovery probe" stagnation_recovery_read_output="$fixture_root/stagnation-recovery-read-output" rm -f "$capture"/* diff --git a/tools/evals/run-codex-framework-evals.sh b/tools/evals/run-codex-framework-evals.sh index 2746937..c26cbed 100755 --- a/tools/evals/run-codex-framework-evals.sh +++ b/tools/evals/run-codex-framework-evals.sh @@ -2449,11 +2449,11 @@ observed_event_shape_supported() { } small_fix_event_shape_supported() { - observed_event_shape_supported "$1" '["agent_message","command_execution","file_change","mcp_tool_call","web_search"]' + observed_event_shape_supported "$1" '["agent_message","reasoning","command_execution","file_change","mcp_tool_call","web_search"]' } stagnation_event_shape_supported() { - observed_event_shape_supported "$1" '["agent_message","command_execution","file_change"]' + observed_event_shape_supported "$1" '["agent_message","reasoning","command_execution","file_change"]' } small_fix_requires_unavailable_adapter_policy() { From 242141b04b977f37da0bdf5570f4b71ae606784e Mon Sep 17 00:00:00 2001 From: Laimis Date: Wed, 23 Sep 2026 20:11:26 +0300 Subject: [PATCH 5/6] fix: enforce observed scope and valid execution contracts --- README.md | 5 +- docs/evals/README.md | 19 +++++- .../contracts/phase-gates.yaml | 4 +- .../references/build-worker-protocol.md | 12 ++-- .../references/context-handoff-templates.md | 4 +- .../references/mega-and-patterns.md | 2 +- .../assistant-workflow/references/phases.md | 8 ++- .../references/phases/build.md | 8 ++- .../references/sub-task-brief-template.md | 6 +- .../p0-p4/codex-behavioral-eval-contracts.sh | 66 ++++++++++++++++++- tests/p0-p4/skill-eval-contracts.sh | 32 ++++++++- tests/p0-p4/task-packet-contracts.sh | 30 ++++++++- tools/evals/lib/skill-eval-fixtures.sh | 22 +++++-- tools/evals/run-codex-framework-evals.sh | 28 ++++++-- 14 files changed, 211 insertions(+), 35 deletions(-) diff --git a/README.md b/README.md index 9425856..a50da00 100644 --- a/README.md +++ b/README.md @@ -122,7 +122,7 @@ Core development pipeline: idea-to-action decomposition, discover, proportional |---|---|---| | Linear | One focused change with dependent steps; scale planning and review to risk. | Goal-driven checks and fresh review; a low-risk local change may use the direct lightweight path. | | Skill-driven | A recurring task has a matching installed skill and contract. | Any other pattern; the skill supplies the method. | -| Parallel | Independent work can progress together. Source-changing packets remain sequential in a shared or unknown workspace; overlap requires runtime-proven isolated workspaces. | Goal-driven integration: dependents wait for VERIFIED prerequisites, then cross-slice and full-scope validation precede fresh review. | +| Parallel | Independent work can progress together. Source-changing packets remain sequential in a shared or unknown workspace; overlap requires runtime-proven isolated workspaces. | Goal-driven integration: dependents wait for VERIFIED prerequisites, then full-scope validation precedes fresh review; cross-slice validation applies to multi-slice manifests. | | Goal-driven | Acceptance checks must remain the completion condition through repair. | Linear, skill-driven, or parallel execution; bounded repair escalates instead of claiming false completion. | For dependency-shaped uncertainty, the workflow defaults to @@ -203,8 +203,7 @@ requirement. Dependent slices wait for verified prerequisites. Read-only analysi may run in parallel for independent packets with non-overlapping ownership. Source-changing packets in a shared or unknown workspace remain sequential; parallel source-changing packets require runtime proof of isolated workspaces. -After integration, rerun cross-slice and full-scope validation and perform a -fresh review before completion. +After all slices are integrated, full-scope validation is required before fresh Review. Cross-slice validation applies only when slice_manifest contains more than one item; when it contains one item, record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code. After these integration checks, perform a fresh review before completion. For compression-safe work, the orchestrator keeps concise, root-scoped task and session state under `.codex/`. Resume reconciles that journal with the newest diff --git a/docs/evals/README.md b/docs/evals/README.md index 0e9d920..23a3e09 100644 --- a/docs/evals/README.md +++ b/docs/evals/README.md @@ -37,7 +37,11 @@ Codex adapter actually observed in its JSONL event stream. to be the first workspace command or file action; the matching command-start event may precede its successful completion. This rule and its no-web/MCP check apply only to the disposable local typo fixture, not delegated work. - A grading artifact alone cannot pass the case. If the safe target ends in the + A grading artifact alone cannot pass the case. Every observed file-change + event must name only the target or grading artifact; combined or separate + out-of-scope paths fail even when the final workspace diff is clean. Malformed + or unresolved observed paths also fail closed, and a failed completion cannot + prove a successful target edit. If the safe target ends in the expected state after a later successful command but the event stream has no `file_change`, the adapter reports unavailable: command text and final state cannot prove which command edited the target or when. That classification is @@ -804,7 +808,18 @@ array whose every member is a nonblank string. `object_keys_exact` requires the target to be an object whose keys exactly match the declared `fields`; it rejects missing keys, extra keys, and non-object values. Its path may be `[]` to check the JSON response root, or a declared -object path. It accepts at most 16 unique field names. +object path. It accepts at most 16 unique field names. Assertion paths follow +the declared shape one segment at a time: each numeric segment consumes one +array level, and string segments traverse fields only on an object. This admits +object-array element fields and primitive-array elements while rejecting +repeated or skipped indexes; schema descriptors are cloned when resolving an +array element so the declared root remains unchanged. +The assistant-workflow eval-only root registry explicitly projects the +`decision_item` and `decision_resolution` object arrays as single objects for +the `progressive-collaborative-contributor-evidence` fixture, whose prompt asks +for those projected roots. Their canonical array descriptors remain available +for indexed paths; this bounded compatibility does not permit skipped indexes +on other arrays. `empty_array` requires the target path to resolve to an empty array. `array_type` requires the target path to resolve to an array and permits an empty array. `array_nonblank_strings` requires every member to be a nonblank string and a required boolean `allow_empty` declares whether an empty array is valid. In this exhaustive fixed operator list, `path_absent` passes only when its target diff --git a/skills/assistant-workflow/contracts/phase-gates.yaml b/skills/assistant-workflow/contracts/phase-gates.yaml index d6ecfc8..82ae113 100644 --- a/skills/assistant-workflow/contracts/phase-gates.yaml +++ b/skills/assistant-workflow/contracts/phase-gates.yaml @@ -541,9 +541,9 @@ gates: on_fail: "Repair evidence according to build_execution_lane: bounded_executor records executor RED/GREEN/verification; separated_workers records Code Writer and Builder/Tester dispatch/results." - id: B13 - check: "For medium+ tasks: source-changing slices in a shared or unknown workspace are VERIFIED, including self-check result, before another source-changing slice starts. Independently executable source-changing slices may overlap only when runtime evidence proves isolated workspaces; every depends_on prerequisite has final status VERIFIED before a dependent slice starts. After all slices are integrated, cross-slice and full-scope validation are complete before Review." + check: "For medium+ tasks: source-changing slices in a shared or unknown workspace are VERIFIED, including self-check result, before another source-changing slice starts. Independently executable source-changing slices may overlap only when runtime evidence proves isolated workspaces; every depends_on prerequisite has final status VERIFIED before a dependent slice starts. Full-scope validation is required before fresh Review, after all slices are integrated. Cross-slice validation applies only when slice_manifest contains more than one item; when it contains one item, record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code." condition: "size in [medium, large, mega]" - on_fail: "This is a process violation — sequence source-changing slices in shared or unknown workspaces, require runtime-proven isolation for overlap, and do not start dependents before VERIFIED prerequisites. Record the evidence and complete integration validation before Review." + on_fail: "This is a process violation — sequence source-changing slices in shared or unknown workspaces, require runtime-proven isolation for overlap, and do not start dependents before VERIFIED prerequisites. After integration and before Review, complete full-scope validation; complete cross-slice validation only when slice_manifest contains more than one item, or record it as not_applicable with the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code." guidance_assertions: diff --git a/skills/assistant-workflow/references/build-worker-protocol.md b/skills/assistant-workflow/references/build-worker-protocol.md index 67b2c5c..5052af3 100644 --- a/skills/assistant-workflow/references/build-worker-protocol.md +++ b/skills/assistant-workflow/references/build-worker-protocol.md @@ -161,7 +161,7 @@ with reapproval, user input, or environment recovery through `pivot_restart_decision` when applicable. A changed plan version starts a new path only after required reapproval; it must not disguise a same-scope retry. -After all slices are integrated, run cross-slice and full-scope validation before entering fresh Review. Per-slice verification does not satisfy this integration barrier. +After all slices are integrated, full-scope validation is required before fresh Review. Cross-slice validation applies only when slice_manifest contains more than one item; when it contains one item, record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code. Per-slice verification does not satisfy this integration barrier. After each implementation step, apply the relevant SOLID check from `references/prompts/solid-principles.md` and fix material violations before @@ -278,9 +278,13 @@ or unknown workspace, start another source-changing slice only after the active source-changing slice is fully verified; start a dependent slice only after all its `depends_on` prerequisites are verified. -After all slices are verified and concurrent outputs are integrated, run -cross-slice and full-scope validation before entering fresh Review. Per-slice -evidence does not satisfy new integration coverage. +After all slices are verified and concurrent outputs are integrated, apply the +manifest-cardinality integration rule: full-scope validation is required before +fresh Review; cross-slice validation applies only when slice_manifest contains +more than one item, and for one item is recorded as not_applicable using the +one-item manifest and single_slice_rationale. Single-slice full-scope +validation still covers integration with existing code. Per-slice evidence +does not satisfy new integration coverage. If implementation reveals a plan problem, print `>> PLAN DEVIATION DETECTED`, record `pivot_restart_decision.reapproval_required=true` when scope/files/ behavior/risk/verification/acceptance changes, and wait for approval before diff --git a/skills/assistant-workflow/references/context-handoff-templates.md b/skills/assistant-workflow/references/context-handoff-templates.md index c541cf2..fdc64d9 100755 --- a/skills/assistant-workflow/references/context-handoff-templates.md +++ b/skills/assistant-workflow/references/context-handoff-templates.md @@ -175,10 +175,12 @@ Integrate: 1. Integrate the completed slice changes 2. Resolve conflicts 3. Confirm verified prerequisite slice outputs are present and consumed -4. Run integration checks for DI, routes, configs, data flow, and cross-slice behavior +4. Run integration checks for DI, routes, configs, data flow, and the full integrated scope 5. Run full integration test suite 6. Fix integration mismatches +After all slices are integrated, full-scope validation is required before fresh Review. Cross-slice validation applies only when slice_manifest contains more than one item; when it contains one item, record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code. + [Include relevant file paths] ``` diff --git a/skills/assistant-workflow/references/mega-and-patterns.md b/skills/assistant-workflow/references/mega-and-patterns.md index ee7c2bb..7bd58ba 100644 --- a/skills/assistant-workflow/references/mega-and-patterns.md +++ b/skills/assistant-workflow/references/mega-and-patterns.md @@ -8,7 +8,7 @@ Use the strict slice packet fields from `slice_manifest` for every executable br - Contract-only/setup-only work is valid only when it is the verified deliverable artifact slice; otherwise include enabling changes in the slice that first uses them - Each slice: Plan --> [Design] --> Build - Keep slice ownership explicit through task packets. Source-changing packets in a shared or unknown workspace are sequential. Parallel source-changing packets require the runtime explicitly proves isolated workspaces; independent read-only analysis may run in parallel. -- Verify each completed slice before dependent work starts. After all slices are integrated, run cross-slice and full-scope validation, then a fresh review. +- Verify each completed slice before dependent work starts. After all slices are integrated, full-scope validation is required before fresh Review. Cross-slice validation applies only when slice_manifest contains more than one item; when it contains one item, record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code. ## Agent Portability diff --git a/skills/assistant-workflow/references/phases.md b/skills/assistant-workflow/references/phases.md index 4b4d474..6d3135a 100644 --- a/skills/assistant-workflow/references/phases.md +++ b/skills/assistant-workflow/references/phases.md @@ -366,8 +366,12 @@ a concrete blocked/inconclusive debugging result. Do not patch until reproduction/root-cause evidence identifies a fix target or mitigation. Tests stay alongside code, not after. -After all slices are verified and concurrent outputs are integrated, run -cross-slice and full-scope validation before fresh review. +After all slices are integrated, full-scope validation is required before +fresh Review. Cross-slice validation applies only +when slice_manifest contains more than one item; when it contains one item, +record cross-slice validation as not_applicable using the one-item manifest and +single_slice_rationale. Single-slice full-scope validation still covers +integration with existing code. Apply the current-verification reuse rules in `references/build-worker-protocol.md`: reuse only full matching identity/coverage evidence and rerun every invalidated case. Per-slice evidence diff --git a/skills/assistant-workflow/references/phases/build.md b/skills/assistant-workflow/references/phases/build.md index a49815a..ca187dc 100644 --- a/skills/assistant-workflow/references/phases/build.md +++ b/skills/assistant-workflow/references/phases/build.md @@ -87,8 +87,12 @@ a concrete blocked/inconclusive debugging result. Do not patch until reproduction/root-cause evidence identifies a fix target or mitigation. Tests stay alongside code, not after. -After all slices are verified and concurrent outputs are integrated, run -cross-slice and full-scope validation before fresh review. +After all slices are integrated, full-scope validation is required before +fresh Review. Cross-slice validation applies only +when slice_manifest contains more than one item; when it contains one item, +record cross-slice validation as not_applicable using the one-item manifest and +single_slice_rationale. Single-slice full-scope validation still covers +integration with existing code. Apply the current-verification reuse rules in `references/build-worker-protocol.md`: reuse only full matching identity/coverage evidence and rerun every invalidated case. Per-slice evidence diff --git a/skills/assistant-workflow/references/sub-task-brief-template.md b/skills/assistant-workflow/references/sub-task-brief-template.md index 1ee2a4a..33974ae 100755 --- a/skills/assistant-workflow/references/sub-task-brief-template.md +++ b/skills/assistant-workflow/references/sub-task-brief-template.md @@ -168,11 +168,13 @@ keep dependent packets sequenced by `depends_on`. All slice packets are verified. Now integrate: 1. Integrate the completed slice changes and resolve conflicts 2. Confirm verified prerequisite slice outputs are present and consumed -3. Run integration checks for DI, routes, configs, data flow, and cross-slice behavior -4. Run integration tests across slice boundaries +3. Run integration checks for DI, routes, configs, data flow, and the full integrated scope +4. Run tests across slice boundaries when the manifest contains multiple items 5. Run the full relevant suite 6. Fix integration mismatches and request fresh review +After all slices are integrated, full-scope validation is required before fresh Review. Cross-slice validation applies only when slice_manifest contains more than one item; when it contains one item, record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code. + Verified slices completed: - [name]: [what was built and verified] - [name]: [what was built and verified] diff --git a/tests/p0-p4/codex-behavioral-eval-contracts.sh b/tests/p0-p4/codex-behavioral-eval-contracts.sh index 40ca14b..06cb263 100755 --- a/tests/p0-p4/codex-behavioral-eval-contracts.sh +++ b/tests/p0-p4/codex-behavioral-eval-contracts.sh @@ -698,7 +698,7 @@ printf '%s\n' '{"type":"turn.started"}' printf '%s\n' '{"type":"item.completed","item":{"id":"item-1","type":"agent_message","text":"phase small docs/usage.md teh"}}' if [[ -f "$workspace/docs/usage.md" ]]; then case "${FAKE_PATTERN_EVENT_MODE:-}" in - ""|small-positive|small-wrapped-positive|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed|small-command-only|small-command-only-wrong-final) + ""|small-positive|small-wrapped-positive|small-transient-unrelated|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed|small-command-only|small-command-only-wrong-final) if [[ "${FAKE_SMALL_PASSIVE_REASONING:-false}" == "true" ]]; then jq -cn '{type:"item.completed",item:{id:"small-reasoning-before",type:"reasoning",text:"safe summary"}}' fi @@ -716,6 +716,11 @@ if [[ -f "$workspace/docs/usage.md" ]]; then printf '%s\n' '{"type":"item.started","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md"}}' fi jq -cn --argjson command "$small_discovery_command" '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:$command,exit_code:0,status:"completed",aggregated_output:"1:This fixture contains teh requested typo."}}' + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-transient-unrelated" ]]; then + printf '%s\n' 'transient unrelated write' >"$workspace/transient-unrelated.txt" + jq -cn '{type:"item.completed",item:{id:"small-transient-unrelated",type:"file_change",path:"transient-unrelated.txt",status:"completed"}}' + rm -f "$workspace/transient-unrelated.txt" + fi if [[ "${FAKE_SMALL_PASSIVE_REASONING:-false}" == "true" ]]; then jq -cn '{type:"item.completed",item:{id:"small-reasoning-between",type:"reasoning",text:"safe summary"}}' fi @@ -3450,6 +3455,11 @@ awk ' capture { print } capture && /^}$/ { exit } ' "$runner" >>"$event_bounds_lib" +awk ' + /^path_allowed_for_case\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" awk ' /^small_fix_event_evidence\(\)/ { capture = 1 } capture { print } @@ -3562,6 +3572,38 @@ write_small_path_control() { combined) jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",changes:[{path:$path,kind:"update"},{path:".assistant-eval/workflow-decision.json",kind:"add"}]}}' >>"$jsonl" ;; + combined_unrelated) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",changes:[{path:$path,kind:"update"},{path:"README.md",kind:"update"}]}}' >>"$jsonl" + ;; + separate_before) + jq -cn '{type:"item.completed",item:{id:"small-unrelated",type:"file_change",path:"README.md"}}' >>"$jsonl" + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",path:$path}}' >>"$jsonl" + return + ;; + separate_after) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",path:$path}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"small-unrelated",type:"file_change",path:"README.md"}}' >>"$jsonl" + return + ;; + started_unrelated) + jq -cn '{type:"item.started",item:{id:"small-unrelated",type:"file_change",path:"README.md"}}' >>"$jsonl" + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",path:$path}}' >>"$jsonl" + return + ;; + failed_unrelated) + jq -cn '{type:"item.completed",item:{id:"small-unrelated",type:"file_change",path:"README.md",status:"failed"}}' >>"$jsonl" + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",path:$path}}' >>"$jsonl" + return + ;; + failed_target) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",path:$path,status:"failed"}}' >>"$jsonl" + return + ;; + malformed_extra) + jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",path:$path}}' >>"$jsonl" + jq -cn '{type:"item.completed",item:{id:"small-malformed",type:"file_change",changes:[{}]}}' >>"$jsonl" + return + ;; mixed) jq -cn --arg path "$path" '{type:"item.completed",item:{id:"small-change",type:"file_change",changes:[{path:$path,kind:"update"},{path:"/tmp/outside/usage.md",kind:"update"}]}}' >>"$jsonl" ;; @@ -3582,7 +3624,7 @@ for small_path_positive in \ small_path_jsonl="$fixture_root/small-path-positive-${small_path_mode}-${RANDOM}.jsonl" write_small_path_control "$small_path_mode" "$small_path_jsonl" "$small_path_value" if ! small_fix_event_evidence "$small_path_jsonl" "$small_path_workspace" \ - | jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0' >/dev/null; then + | jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0 and .file_change_scope_valid == true' >/dev/null; then small_path_failures+=("positive:$small_path_positive") fi done @@ -3600,6 +3642,20 @@ for small_path_negative in \ small_path_failures+=("negative:$small_path_negative") fi done +for small_path_negative in combined_unrelated separate_before separate_after started_unrelated failed_unrelated malformed_extra; do + small_path_jsonl="$fixture_root/small-path-negative-$small_path_negative.jsonl" + write_small_path_control "$small_path_negative" "$small_path_jsonl" "docs/usage.md" + if ! small_fix_event_evidence "$small_path_jsonl" "$small_path_workspace" \ + | jq -e '.source_discovery_before_change == false and .file_change_scope_valid == false' >/dev/null; then + small_path_failures+=("negative:$small_path_negative") + fi +done +small_path_jsonl="$fixture_root/small-path-failed-target.jsonl" +write_small_path_control failed_target "$small_path_jsonl" "docs/usage.md" +if ! small_fix_event_evidence "$small_path_jsonl" "$small_path_workspace" \ + | jq -e '.source_discovery_before_change == false and .file_change_scope_valid == true' >/dev/null; then + small_path_failures+=("failed target completion established successful edit evidence or invalidated allowed scope") +fi if [[ ${#small_path_failures[@]} -eq 0 ]]; then pass else @@ -5990,6 +6046,7 @@ fi test_start "small-fix evaluation requires observed low-overhead evidence rather than its grading artifact alone" small_positive_output="$fixture_root/small-observed-positive-output" small_wrapped_positive_output="$fixture_root/small-observed-wrapped-positive-output" +small_transient_unrelated_output="$fixture_root/small-observed-transient-unrelated-output" small_artifact_only_output="$fixture_root/small-observed-artifact-only-output" small_external_read_output="$fixture_root/small-observed-external-read-output" small_unsupported_output="$fixture_root/small-observed-unsupported-output" @@ -6005,6 +6062,11 @@ if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-positive "$ru --cases small-fix-stays-lightweight --repeats 1 --output "$small_wrapped_positive_output" --codex-bin "$fake_codex" >/dev/null \ && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == true)' "$small_wrapped_positive_output/traces/"*.json >/dev/null \ && rm -f "$capture"/* \ + && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-transient-unrelated "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_transient_unrelated_output" --codex-bin "$fake_codex" >/dev/null \ + && jq -s -e 'all(.[]; .status == "completed" and .metrics.acceptance_passed == false and .execution.verifier.workspace_failure_ids == ["workspace-003"] and .execution.verifier.scope_deviations == 0)' "$small_transient_unrelated_output/traces/"*.json >/dev/null \ + && rm -f "$capture"/* \ && FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-artifact-only "$runner" --execute \ --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ --cases small-fix-stays-lightweight --repeats 1 --output "$small_artifact_only_output" --codex-bin "$fake_codex" >/dev/null \ diff --git a/tests/p0-p4/skill-eval-contracts.sh b/tests/p0-p4/skill-eval-contracts.sh index 52bd003..0abbf39 100644 --- a/tests/p0-p4/skill-eval-contracts.sh +++ b/tests/p0-p4/skill-eval-contracts.sh @@ -2186,6 +2186,28 @@ else fail "fixture validation accepted unknown assertion roots: ${unknown_root_failures[*]}" fi +test_start "workflow collaborative contributor fixture resolves its bounded singleton array projections" +workflow_projection_root="$(mktemp -d "${TMPDIR:-/tmp}/skill-eval-workflow-projections.XXXXXX")" +workflow_projection_skill="$workflow_projection_root/assistant-workflow" +workflow_projection_responses="$workflow_projection_root/responses" +workflow_projection_output="$workflow_projection_root/grading.out" +p0p4_register_cleanup "$workflow_projection_root" +mkdir -p "$workflow_projection_skill/evals" "$workflow_projection_responses/assistant-workflow" +cp "$FRAMEWORK_DIR/skills/assistant-workflow/SKILL.md" "$workflow_projection_skill/SKILL.md" +ln -s "$FRAMEWORK_DIR/skills/assistant-workflow/contracts" "$workflow_projection_skill/contracts" +cp "$FRAMEWORK_DIR/skills/assistant-workflow/evals/cases.json" "$workflow_projection_skill/evals/cases.json" +cat >"$workflow_projection_responses/assistant-workflow/progressive-collaborative-contributor-evidence.txt" <<'EOF' +{"decision_item":{"interaction_mode":"collaborative"},"decision_resolution":{"contributor_evidence":[{"contributor_role":"agent","contribution":"Analyzed retention option B.","evidence_ref":"agent-analysis"},{"contributor_role":"human_or_user","contribution":"Product owner selected option B.","evidence_ref":"owner-choice"}]},"route_clear":"after joint evidence"} +EOF +if "$skill_eval_runner" --validate-fixture --skill "$workflow_projection_skill" >/dev/null 2>"$workflow_projection_output" \ + && "$skill_eval_runner" --responses "$workflow_projection_responses" --skill "$workflow_projection_skill" \ + --case progressive-collaborative-contributor-evidence >"$workflow_projection_output" 2>&1 \ + && grep -Fq "structured_json_assertion_failures=0" "$workflow_projection_output"; then + pass +else + fail "workflow collaborative contributor fixture did not validate and grade its two registered singleton projections: $(cat "$workflow_projection_output")" +fi + test_start "fixture validation rejects assertion literals outside resolved enums" impossible_literal_failures=() while IFS='|' read -r impossible_operator impossible_mutation; do @@ -4270,7 +4292,9 @@ jq ' {operator:"object_keys_exact",path:["execution_policy"],fields:["source_writer_policy","read_only_analysis_policy","isolation_evidence_ref","c_start_decisions","per_slice_verification","integration_validation","integration_checks","fresh_review","fresh_review_after"]}, {operator:"object_keys_exact",path:["execution_policy","c_start_decisions",0],fields:["a_status","c_decision"]}, {operator:"object_keys_exact",path:["execution_policy","c_start_decisions",1],fields:["a_status","c_decision"]}, - {operator:"object_keys_exact",path:["execution_policy","c_start_decisions",2],fields:["a_status","c_decision"]} + {operator:"object_keys_exact",path:["execution_policy","c_start_decisions",2],fields:["a_status","c_decision"]}, + {operator:"equals",path:["execution_policy","c_start_decisions",0,"a_status"],expected:"PENDING"}, + {operator:"equals",path:["execution_policy","integration_checks",0],expected:"cross-slice"} ] else . end ) @@ -4315,7 +4339,11 @@ for object_keys_invalid in \ '{"operator":"object_keys_exact","path":[],"fields":["execution_policy","execution_policy"]}' \ '{"operator":"object_keys_exact","path":["execution_policy"],"fields":["unsafe_override"]}' \ '{"operator":"object_keys_exact","path":["execution_policy","source_writer_policy"],"fields":["value"]}' \ - '{"operator":"object_keys_exact","path":["execution_policy"],"fields":["source_writer_policy","source_writer_policy"]}'; do + '{"operator":"object_keys_exact","path":["execution_policy"],"fields":["source_writer_policy","source_writer_policy"]}' \ + '{"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",0,0],"fields":["a_status","c_decision"]}' \ + '{"operator":"object_keys_exact","path":["execution_policy","c_start_decisions",0,0,0],"fields":["a_status","c_decision"]}' \ + '{"operator":"equals","path":["execution_policy","c_start_decisions","a_status"],"expected":"PENDING"}' \ + '{"operator":"object_keys_exact","path":["execution_policy","source_writer_policy",0],"fields":["value"]}'; do jq --argjson assertion "$object_keys_invalid" '(.cases[] | select(.id == "native-slice-execution-uses-dependencies-not-runner-topology") | .machine_expectations.structured_json_assertions) += [$assertion]' "$object_keys_root/base-cases.json" >"$object_keys_root/invalid.json" mv "$object_keys_root/invalid.json" "$object_keys_skill/evals/cases.json" if "$skill_eval_runner" --validate-fixture --skill "$object_keys_skill" >/dev/null 2>"$object_keys_err"; then diff --git a/tests/p0-p4/task-packet-contracts.sh b/tests/p0-p4/task-packet-contracts.sh index 5a3f2d2..4b866cf 100644 --- a/tests/p0-p4/task-packet-contracts.sh +++ b/tests/p0-p4/task-packet-contracts.sh @@ -72,7 +72,7 @@ for term in \ "Record verification evidence in the task journal slice verification ledger" \ "Run a small self-check/local sanity check" \ "Mark the slice \`VERIFIED\` only after all criteria pass and evidence is recorded" \ - "After all slices are integrated, run cross-slice and full-scope validation before entering fresh Review"; do + "After all slices are integrated, full-scope validation is required before fresh Review. Cross-slice validation applies only when slice_manifest contains more than one item;"; do if ! p0p4_contains_text "$build_worker_ref" "$term"; then missing_slice_phase_terms+=("$term") fi @@ -833,11 +833,14 @@ for term in \ "independently checked, passing, and recorded with command/result evidence in the task journal, validation_results, or equivalent carried-forward slice ledger" \ "record command/result evidence in the configured task journal or equivalent carried-forward state" \ "- id: B13" \ - "After all slices are integrated, cross-slice and full-scope validation are complete before Review" \ + "Full-scope validation is required before fresh Review" \ + "Cross-slice validation applies only when slice_manifest contains more than one item" \ + "record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale" \ + "Single-slice full-scope validation still covers integration with existing code" \ "shared or unknown workspace are VERIFIED, including self-check result, before another source-changing slice starts" \ "runtime evidence proves isolated workspaces" \ "every depends_on prerequisite has final status VERIFIED before a dependent slice starts" \ - "cross-slice and full-scope validation are complete before Review"; do + "Cross-slice validation applies only when slice_manifest contains more than one item"; do if ! grep -Fq -- "$term" "$FRAMEWORK_DIR/skills/assistant-workflow/contracts/phase-gates.yaml"; then missing_slice_gate_terms+=("$term") fi @@ -848,6 +851,27 @@ else fail "phase-gates.yaml missing dependency-aware slice verification gate terms: ${missing_slice_gate_terms[*]}" fi +test_start "workflow integration prompts apply cross-slice checks by manifest cardinality and keep full-scope checks" +slice_integration_rule="After all slices are integrated, full-scope validation is required before fresh Review. Cross-slice validation applies only when slice_manifest contains more than one item; when it contains one item, record cross-slice validation as not_applicable using the one-item manifest and single_slice_rationale. Single-slice full-scope validation still covers integration with existing code." +slice_integration_failures=() +for slice_integration_file in \ + "$FRAMEWORK_DIR/skills/assistant-workflow/references/build-worker-protocol.md" \ + "$FRAMEWORK_DIR/skills/assistant-workflow/references/phases.md" \ + "$FRAMEWORK_DIR/skills/assistant-workflow/references/phases/build.md" \ + "$FRAMEWORK_DIR/skills/assistant-workflow/references/mega-and-patterns.md" \ + "$FRAMEWORK_DIR/skills/assistant-workflow/references/context-handoff-templates.md" \ + "$FRAMEWORK_DIR/skills/assistant-workflow/references/sub-task-brief-template.md" \ + "$FRAMEWORK_DIR/README.md"; do + if ! p0p4_contains_text "$slice_integration_file" "$slice_integration_rule"; then + slice_integration_failures+=("$slice_integration_file") + fi +done +if [[ "${#slice_integration_failures[@]}" -eq 0 ]]; then + pass +else + fail "integration guidance is missing the single/multiple-slice applicability rule: ${slice_integration_failures[*]}" +fi + test_start "Decompose guidance permits isolated overlap while rejecting stale next-slice sequencing" decompose_source="$FRAMEWORK_DIR/skills/assistant-workflow/references/phases.md" decompose_view="$FRAMEWORK_DIR/skills/assistant-workflow/references/phases/decompose.md" diff --git a/tools/evals/lib/skill-eval-fixtures.sh b/tools/evals/lib/skill-eval-fixtures.sh index 47e5ee6..8421fe0 100644 --- a/tools/evals/lib/skill-eval-fixtures.sh +++ b/tools/evals/lib/skill-eval-fixtures.sh @@ -50,6 +50,8 @@ load_contract_roots.call(contracts_dir, skill_name == "assistant-review") # Explicitly bounded roots that eval fixtures may project outside their output # artifacts. This is deliberately not a pool of every input or handoff field. +# The collaborative progressive fixture also requests singleton projections of +# these two canonical arrays; derive each item view from the owning schema. eval_only_root_registry = { "assistant-docs" => [ { "kind" => "input_field", "name" => "architecture_decision_pack_status" }, @@ -67,7 +69,9 @@ eval_only_root_registry = { { "kind" => "input_field", "name" => "feature_preparation_scope" }, { "kind" => "handoff_field", "name" => "architecture_mapping_evidence" }, { "kind" => "handoff_field", "name" => "implementation_steps" }, - { "kind" => "output_child", "artifact" => "triage_result", "name" => "size" } + { "kind" => "output_child", "artifact" => "triage_result", "name" => "size" }, + { "kind" => "output_array_item", "artifact" => "decision_item" }, + { "kind" => "output_array_item", "artifact" => "decision_resolution" } ] } @@ -94,6 +98,14 @@ eval_only_root_registry.fetch(skill_name, []).each do |selector| output = YAML.load_file(File.join(contracts_dir, "output.yaml")) artifact = output.fetch("artifacts", []).find { |field| field["name"] == selector.fetch("artifact") } artifact ? artifact.fetch("object_fields", []).select { |field| field["name"] == selector.fetch("name") } : [] + when "output_array_item" + output = YAML.load_file(File.join(contracts_dir, "output.yaml")) + artifact = output.fetch("artifacts", []).find { |field| field["name"] == selector.fetch("artifact") } + if artifact.is_a?(Hash) && artifact["type"] == "object[]" && artifact["object_fields"].is_a?(Array) + [artifact.dup.merge("type" => "object")] + else + [] + end else [] end @@ -249,9 +261,10 @@ resolve = lambda do |path| path.drop(1).each do |segment| candidates = candidates.flat_map do |field| if segment.is_a?(Numeric) - field["type"].is_a?(String) && field["type"].end_with?("[]") ? [field] : [] + field_type = field["type"] + field_type.is_a?(String) && field_type.end_with?("[]") ? [field.dup.merge("type" => field_type.delete_suffix("[]"))] : [] elsif segment.is_a?(String) - field.fetch("object_fields", []).select { |child| child["name"] == segment } + field["type"] == "object" ? field.fetch("object_fields", []).select { |child| child["name"] == segment } : [] else [] end @@ -333,8 +346,7 @@ end else target = resolve.call(path) target_field = target.length == 1 ? target.first : nil - object_array_item = target_field && target_field["type"] == "object[]" && path.last.is_a?(Numeric) - unless target_field && (target_field["type"] == "object" || object_array_item) && target_field["object_fields"].is_a?(Array) + unless target_field && target_field["type"] == "object" && target_field["object_fields"].is_a?(Array) warn "case #{test_case.fetch("id")}.machine_expectations.structured_json_assertions[#{index}] object_keys_exact target must be a declared object: #{path.to_json}" exit 1 end diff --git a/tools/evals/run-codex-framework-evals.sh b/tools/evals/run-codex-framework-evals.sh index c26cbed..20a8b1a 100755 --- a/tools/evals/run-codex-framework-evals.sh +++ b/tools/evals/run-codex-framework-evals.sh @@ -2343,13 +2343,26 @@ PY } small_fix_event_evidence() { - local jsonl="$1" workspace="$2" path_mappings + local jsonl="$1" workspace="$2" path_mappings allowed_file_change_paths="[]" event_path path_mappings="$(workspace_event_path_mappings "$jsonl" "$workspace")" || return 1 - jq -cse --argjson path_mappings "$path_mappings" "$EVENT_EVIDENCE_NORMALIZATION_JQ$(cat <<'JQ' + while IFS= read -r event_path; do + [[ -n "$event_path" ]] || continue + if path_allowed_for_case "small-fix-stays-lightweight" "$event_path"; then + allowed_file_change_paths="$(jq -cn --argjson paths "$allowed_file_change_paths" --arg path "$event_path" '$paths + [$path] | unique')" + fi + done < <(jq -r 'to_entries[] | select(.value | type == "string") | .value' <<<"$path_mappings") + jq -cse \ + --argjson path_mappings "$path_mappings" \ + --argjson allowed_file_change_paths "$allowed_file_change_paths" \ + "$EVENT_EVIDENCE_NORMALIZATION_JQ$(cat <<'JQ' def changes_containing($expected): normalized_file_change_paths as $paths | $paths != null and ($paths | length) > 0 and all($paths[]; . != null) and any($paths[]; . == $expected); + def file_change_paths_are_allowed: + normalized_file_change_paths as $paths + | $paths != null and ($paths | length) > 0 + and all($paths[]; . as $path | $path != null and ($allowed_file_change_paths | index($path)) != null); def is_discovery: .type == "item.completed" and .item.type == "command_execution" @@ -2367,7 +2380,11 @@ small_fix_event_evidence() { def is_change: .type == "item.completed" and .item.type == "file_change" + and ((.item.status? // "completed") == "completed") and changes_containing("docs/usage.md"); + def is_observed_file_change: + (.type == "item.started" or .type == "item.completed" or .type == "item.updated") + and .item.type == "file_change"; def is_successful_command: .type == "item.completed" and .item.type == "command_execution" @@ -2381,6 +2398,8 @@ small_fix_event_evidence() { . as $events | [range(0; length) | select($events[.] | is_discovery)] as $discoveries | [range(0; length) | select($events[.] | is_change)] as $changes + | [range(0; length) | . as $index | $events[$index] | select(is_observed_file_change)] as $file_change_events + | ($file_change_events | all(.[]; file_change_paths_are_allowed)) as $file_change_scope_valid | [range(0; length) | . as $index | $events[$index] @@ -2419,9 +2438,10 @@ small_fix_event_evidence() { ($events[.].type == "item.completed" or $events[.].type == "item.started" or $events[.].type == "item.updated") and ($events[.].item.type == "mcp_tool_call" or $events[.].item.type == "web_search"))] as $disallowed | { - source_discovery_before_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($changes | length) > 0 and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and $discoveries[0] < $changes[0]), + source_discovery_before_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($changes | length) > 0 and $file_change_scope_valid and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and $discoveries[0] < $changes[0]), command_only_change_without_file_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($successful_commands | length) > 0 and ($changes | length) == 0 and ($has_file_change_event | not) and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and any($successful_commands[]; .index > $discoveries[0])), - disallowed_item_count: ($disallowed | length) + disallowed_item_count: ($disallowed | length), + file_change_scope_valid: $file_change_scope_valid } JQ )" "$jsonl" From 4d84e04dfdde99b9a7d8b08e869d2e2ba53305f3 Mon Sep 17 00:00:00 2001 From: Laimis Date: Wed, 23 Sep 2026 21:54:09 +0300 Subject: [PATCH 6/6] Close eval command scope and recovery telemetry gaps --- docs/evals/README.md | 49 +- docs/evals/framework-instruction-cases.json | 1 + .../p0-p4/codex-behavioral-eval-contracts.sh | 535 +++++++++++++++++- tools/evals/run-codex-framework-evals.sh | 118 +++- 4 files changed, 671 insertions(+), 32 deletions(-) diff --git a/docs/evals/README.md b/docs/evals/README.md index 23a3e09..fdce624 100644 --- a/docs/evals/README.md +++ b/docs/evals/README.md @@ -37,32 +37,43 @@ Codex adapter actually observed in its JSONL event stream. to be the first workspace command or file action; the matching command-start event may precede its successful completion. This rule and its no-web/MCP check apply only to the disposable local typo fixture, not delegated work. - A grading artifact alone cannot pass the case. Every observed file-change - event must name only the target or grading artifact; combined or separate - out-of-scope paths fail even when the final workspace diff is clean. Malformed - or unresolved observed paths also fail closed, and a failed completion cannot - prove a successful target edit. If the safe target ends in the - expected state after a later successful command but the event stream has no - `file_change`, the adapter reports unavailable: command text and final state - cannot prove which command edited the target or when. That classification is - allowed only when response grading passes, the sole workspace failure is the - missing observation, and there are no scope deviations. Observable - pre-discovery actions, external calls, and incorrect final content remain - completed failures. + Its admitted command vocabulary is limited to that exact read-only probe, + including bounded raw, argv, and shell-wrapper forms. Any other started or + completed command is an unsupported scope observation, regardless of whether + a later target-file event is present. The adapter reports unavailable only + when response, final-content, plan, scope, observed-path, discovery-order, and + external-tool checks reveal no independent failure. A grading artifact alone + cannot pass the case. Every observed file-change event must name only the + target or grading artifact; combined or separate out-of-scope paths fail even + when the final workspace diff is clean. Malformed raw event structure follows + the pre-grading unavailable policy, while well-formed unsafe paths and failed + target completions remain observed failures. With no `file_change`, command + text and final state still cannot prove which command edited the target or + when, so the existing unavailable route remains. Observable pre-discovery + actions, external calls, and incorrect final content remain completed + failures. - `pivot-restart-on-stagnation-or-code-writer-blocker` seeds a trusted failing check, fixture-owned failure/recovery receipts, recovery action, and fresh check. Its recovery artifact retains `terminal_completed=false`: a fresh-check pass validates the recovery protocol, not repair of the legacy bug or workflow completion. It admits only the three trusted fixture scripts and optional read-only `cat RECOVERY.md`; any other started or completed command fails the - bounded case. A present command start must have a nonempty id and one later + bounded case. Completed commands require numeric exit codes; a started command + needs no exit or output. For the three trusted marker commands, the selected + `aggregated_output`/`output` value must be a string. A numeric nonzero exit or + wrong marker in a string remains behavioral failure evidence; small-fix + output, passive message content, and optional recovery-read output are not + consumed. A present command start must have a nonempty id and one later completion with the same admitted command kind; duplicate, unmatched, or - mismatched starts, and duplicate nonempty completion ids fail. The failure - completes before recovery starts, and the recovery completes before the fresh - check starts; completion-only streams use the corresponding completion index. - Once that unique fresh check completes, any later started or completed command - or file-change event also fails. This boundary applies only to the disposable - recovery fixture. + mismatched starts, and duplicate nonempty completion ids fail. Each recovery + artifact file-change event must also occur after the recovery start (or the + completion when no start exists) and before the fresh-check start (or its + completion when no start exists). Events before the trusted failure or + recovery, after the fresh-check start, and after fresh-check completion fail; + events during a matched recovery and after recovery completion but before the + fresh check remain valid. Once that unique fresh check completes, any later + started or completed command or file-change event also fails. This boundary + applies only to the disposable recovery fixture. - `isolated-parallel-a-b-then-c-with-integration` records the required A/B/C dependency and integration policy, but current Codex CLI JSONL does not expose authoritative worker, workspace, isolation, or overlap telemetry. The adapter diff --git a/docs/evals/framework-instruction-cases.json b/docs/evals/framework-instruction-cases.json index 90b5377..97b059f 100644 --- a/docs/evals/framework-instruction-cases.json +++ b/docs/evals/framework-instruction-cases.json @@ -33,6 +33,7 @@ "The requested change is isolated to one documentation typo.", "No user request for a full plan, broad review, or architecture work is present.", "For this disposable local typo fixture, the first workspace command or file action must be target discovery with exactly rg -n teh docs/usage.md; its matching command-start event is permitted before successful completion. This command constraint is case-specific and does not generalize to other tasks.", + "The admitted command vocabulary for this disposable case is limited to that exact read-only discovery probe, including its bounded raw, argv, and shell-wrapper forms. Any other started or completed command is unsupported command-scope telemetry; the adapter reports unavailable only when response, final-content, plan, scope, observed-path, discovery-order, and external-tool checks reveal no independent failure.", "After the edit, write the disposable grading artifact .assistant-eval/workflow-decision.json with exactly schema_version, task_size, and plan_mode. Use schema_version 1.0, task_size trivial, and plan_mode none." ], "prompt": "Fix the typo 'teh' to 'the' in docs/usage.md. Keep it simple.", diff --git a/tests/p0-p4/codex-behavioral-eval-contracts.sh b/tests/p0-p4/codex-behavioral-eval-contracts.sh index 06cb263..4dcfe5a 100755 --- a/tests/p0-p4/codex-behavioral-eval-contracts.sh +++ b/tests/p0-p4/codex-behavioral-eval-contracts.sh @@ -698,7 +698,7 @@ printf '%s\n' '{"type":"turn.started"}' printf '%s\n' '{"type":"item.completed","item":{"id":"item-1","type":"agent_message","text":"phase small docs/usage.md teh"}}' if [[ -f "$workspace/docs/usage.md" ]]; then case "${FAKE_PATTERN_EVENT_MODE:-}" in - ""|small-positive|small-wrapped-positive|small-transient-unrelated|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed|small-command-only|small-command-only-wrong-final) + ""|small-positive|small-wrapped-positive|small-transient-unrelated|small-disallowed-tool|small-unsupported-shape|small-external-symlink|small-mcp-started-only|small-web-started-only|small-interleaved-shell-action|small-shell-edit-before-discovery|small-shell-edit-started-before-discovery|small-updated-unknown|small-updated-disallowed|small-command-only|small-command-only-wrong-final|small-unknown-command|small-missing-exit|small-malformed-exit|small-nonzero-exit|small-malformed-file-change) if [[ "${FAKE_SMALL_PASSIVE_REASONING:-false}" == "true" ]]; then jq -cn '{type:"item.completed",item:{id:"small-reasoning-before",type:"reasoning",text:"safe summary"}}' fi @@ -715,7 +715,39 @@ if [[ -f "$workspace/docs/usage.md" ]]; then elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-shell-edit-started-before-discovery" ]]; then printf '%s\n' '{"type":"item.started","item":{"id":"small-shell-edit","type":"command_execution","command":"sed -i.bak s/teh/the/ docs/usage.md"}}' fi - jq -cn --argjson command "$small_discovery_command" '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:$command,exit_code:0,status:"completed",aggregated_output:"1:This fixture contains teh requested typo."}}' + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-unknown-command" ]] \ + && [[ "${FAKE_SMALL_UNKNOWN_COMMAND_MODE:-}" == "pre-discovery" ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"small-unknown-before-discovery","type":"command_execution","command":"python3 transient_write_restore.py","exit_code":0,"status":"completed","aggregated_output":""}}' + fi + small_discovery_exit=0 + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-nonzero-exit" ]]; then + small_discovery_exit=2 + fi + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-missing-exit" ]]; then + jq -cn --argjson command "$small_discovery_command" '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:$command,status:"completed",aggregated_output:"1:This fixture contains teh requested typo."}}' + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-malformed-exit" ]]; then + jq -cn --argjson command "$small_discovery_command" '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:$command,exit_code:"0",status:"completed",aggregated_output:"1:This fixture contains teh requested typo."}}' + else + jq -cn --argjson command "$small_discovery_command" --argjson exit_code "$small_discovery_exit" '{type:"item.completed",item:{id:"small-discovery",type:"command_execution",command:$command,exit_code:$exit_code,status:"completed",aggregated_output:"1:This fixture contains teh requested typo."}}' + fi + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-unknown-command" ]] \ + && [[ "${FAKE_SMALL_UNKNOWN_COMMAND_MODE:-before-target}" =~ ^(before-target|started|nonzero|mixed-unrelated|failed-target)$ ]]; then + if [[ "${FAKE_SMALL_TRANSIENT_COMMAND_MUTATION:-false}" == "true" ]]; then + transient_path="$workspace/transient-command-write.txt" + printf '%s\n' 'transient command write' >"$transient_path" + transient_created_sha256="$(shasum -a 256 "$transient_path" | awk '{print $1}')" + rm -f "$workspace/transient-command-write.txt" + jq -cn --arg created_sha256 "$transient_created_sha256" \ + '{created_sha256:$created_sha256,removed:true}' \ + >"$capture_dir/call-$call_id.transient-mutation.json" + fi + if [[ "${FAKE_SMALL_UNKNOWN_COMMAND_EVENT:-completed}" == "started" ]]; then + printf '%s\n' '{"type":"item.started","item":{"id":"small-unknown-command","type":"command_execution","command":"python3 transient_write_restore.py"}}' + else + small_unknown_exit="${FAKE_SMALL_UNKNOWN_COMMAND_EXIT:-0}" + printf '%s\n' "{\"type\":\"item.completed\",\"item\":{\"id\":\"small-unknown-command\",\"type\":\"command_execution\",\"command\":\"python3 transient_write_restore.py\",\"exit_code\":${small_unknown_exit},\"status\":\"completed\",\"aggregated_output\":\"\"}}" + fi + fi if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-transient-unrelated" ]]; then printf '%s\n' 'transient unrelated write' >"$workspace/transient-unrelated.txt" jq -cn '{type:"item.completed",item:{id:"small-transient-unrelated",type:"file_change",path:"transient-unrelated.txt",status:"completed"}}' @@ -751,6 +783,23 @@ if [[ -f "$workspace/docs/usage.md" ]]; then if [[ "${FAKE_PATTERN_EVENT_MODE:-}" != "small-command-only" && "${FAKE_PATTERN_EVENT_MODE:-}" != "small-command-only-wrong-final" ]]; then printf '%s\n' '{"type":"item.completed","item":{"id":"small-change","type":"file_change","changes":[{"path":"docs/usage.md","kind":"update"}]}}' fi + if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-unknown-command" ]]; then + case "${FAKE_SMALL_UNKNOWN_COMMAND_MODE:-before-target}" in + pre-discovery|before-target|started|nonzero) ;; + after-target) + printf '%s\n' '{"type":"item.completed","item":{"id":"small-unknown-command-after-target","type":"command_execution","command":"python3 transient_write_restore.py","exit_code":0,"status":"completed","aggregated_output":""}}' + ;; + mixed-unrelated) + printf '%s\n' '{"type":"item.completed","item":{"id":"small-unrelated-change","type":"file_change","path":"README.md","status":"completed"}}' + ;; + failed-target) + printf '%s\n' '{"type":"item.completed","item":{"id":"small-failed-target","type":"file_change","path":"docs/usage.md","status":"failed"}}' + ;; + *) printf 'unsupported FAKE_SMALL_UNKNOWN_COMMAND_MODE: %s\n' "${FAKE_SMALL_UNKNOWN_COMMAND_MODE}" >&2; exit 2 ;; + esac + elif [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-malformed-file-change" ]]; then + printf '%s\n' '{"type":"item.completed","item":{"id":"small-malformed-change","type":"file_change","changes":{}}}' + fi if [[ "${FAKE_PATTERN_EVENT_MODE:-}" == "small-external-symlink" ]]; then external_target="$workspace/../external-usage.md" cp "$workspace/docs/usage.md" "$external_target" @@ -768,7 +817,7 @@ if [[ -f "$workspace/docs/usage.md" ]]; then fi if [[ -f "$workspace/RECOVERY.md" ]]; then pattern_mode="${FAKE_PATTERN_EVENT_MODE:-stagnation-positive}" - if [[ "$pattern_mode" != stagnation-positive && "$pattern_mode" != stagnation-paired-start-completion-positive && "$pattern_mode" != stagnation-missing-recovery && "$pattern_mode" != stagnation-retry-after-bound && "$pattern_mode" != stagnation-repeated-trusted-check && "$pattern_mode" != stagnation-duplicate-recovery && "$pattern_mode" != stagnation-duplicate-fresh && "$pattern_mode" != stagnation-failed-recovery-then-retry && "$pattern_mode" != stagnation-failed-fresh-then-retry && "$pattern_mode" != stagnation-transient-source-change && "$pattern_mode" != stagnation-false-completion && "$pattern_mode" != stagnation-unknown-completed-action && "$pattern_mode" != stagnation-post-fresh-shell-mutate-revert && "$pattern_mode" != stagnation-post-fresh-command-started-only && "$pattern_mode" != stagnation-post-fresh-file-change-started-only && "$pattern_mode" != stagnation-updated-unknown && "$pattern_mode" != stagnation-updated-disallowed && "$pattern_mode" != stagnation-pre-fresh-shell-mutate-revert && "$pattern_mode" != stagnation-pre-fresh-command-started-only && "$pattern_mode" != stagnation-pre-fresh-file-change-started-only && "$pattern_mode" != stagnation-recovery-read-positive && "$pattern_mode" != stagnation-early-recovery-start && "$pattern_mode" != stagnation-early-fresh-start && "$pattern_mode" != stagnation-unmatched-recovery-start && "$pattern_mode" != stagnation-duplicate-recovery-start && "$pattern_mode" != stagnation-matched-recovery-id-wrong-command && "$pattern_mode" != stagnation-duplicate-recovery-completed-id ]]; then + if [[ "$pattern_mode" != stagnation-positive && "$pattern_mode" != stagnation-paired-start-completion-positive && "$pattern_mode" != stagnation-missing-recovery && "$pattern_mode" != stagnation-retry-after-bound && "$pattern_mode" != stagnation-repeated-trusted-check && "$pattern_mode" != stagnation-duplicate-recovery && "$pattern_mode" != stagnation-duplicate-fresh && "$pattern_mode" != stagnation-failed-recovery-then-retry && "$pattern_mode" != stagnation-failed-fresh-then-retry && "$pattern_mode" != stagnation-transient-source-change && "$pattern_mode" != stagnation-false-completion && "$pattern_mode" != stagnation-unknown-completed-action && "$pattern_mode" != stagnation-post-fresh-shell-mutate-revert && "$pattern_mode" != stagnation-post-fresh-command-started-only && "$pattern_mode" != stagnation-post-fresh-file-change-started-only && "$pattern_mode" != stagnation-updated-unknown && "$pattern_mode" != stagnation-updated-disallowed && "$pattern_mode" != stagnation-pre-fresh-shell-mutate-revert && "$pattern_mode" != stagnation-pre-fresh-command-started-only && "$pattern_mode" != stagnation-pre-fresh-file-change-started-only && "$pattern_mode" != stagnation-recovery-read-positive && "$pattern_mode" != stagnation-early-recovery-start && "$pattern_mode" != stagnation-early-fresh-start && "$pattern_mode" != stagnation-unmatched-recovery-start && "$pattern_mode" != stagnation-duplicate-recovery-start && "$pattern_mode" != stagnation-matched-recovery-id-wrong-command && "$pattern_mode" != stagnation-duplicate-recovery-completed-id && "$pattern_mode" != stagnation-premature-recovery-artifact && "$pattern_mode" != stagnation-malformed-output && "$pattern_mode" != stagnation-wrong-marker && "$pattern_mode" != stagnation-missing-exit ]]; then printf 'unsupported stagnation FAKE_PATTERN_EVENT_MODE: %s\n' "$pattern_mode" >&2 exit 2 fi @@ -780,8 +829,21 @@ if [[ -f "$workspace/RECOVERY.md" ]]; then elif [[ "$pattern_mode" == stagnation-early-fresh-start ]]; then printf '%s\n' '{"type":"item.started","item":{"id":"stagnation-fresh-check","type":"command_execution","command":"bash tests/stagnation-contracts.sh --after-recovery"}}' fi + if [[ "$pattern_mode" == stagnation-premature-recovery-artifact ]]; then + mkdir -p "$workspace/.assistant-eval" + printf '%s\n' '{"premature":true}' >"$workspace/.assistant-eval/stagnation-recovery.json" + jq -cn '{type:"item.completed",item:{id:"stagnation-premature-artifact",type:"file_change",path:".assistant-eval/stagnation-recovery.json",status:"completed"}}' + fi if failure_output="$(bash "$workspace/tests/stagnation-contracts.sh")"; then failure_exit=0; else failure_exit=$?; fi - printf '%s\n' "{\"type\":\"item.completed\",\"item\":{\"id\":\"stagnation-trusted-failure\",\"type\":\"command_execution\",\"command\":\"bash tests/stagnation-contracts.sh\",\"exit_code\":${failure_exit},\"status\":\"completed\",\"aggregated_output\":\"${failure_output}\"}}" + if [[ "$pattern_mode" == stagnation-malformed-output ]]; then + jq -cn --arg output "$failure_output" --argjson exit_code "$failure_exit" '{type:"item.completed",item:{id:"stagnation-trusted-failure",type:"command_execution",command:"bash tests/stagnation-contracts.sh",exit_code:$exit_code,status:"completed",aggregated_output:{text:$output}}}' + elif [[ "$pattern_mode" == stagnation-wrong-marker ]]; then + jq -cn --argjson exit_code "$failure_exit" '{type:"item.completed",item:{id:"stagnation-trusted-failure",type:"command_execution",command:"bash tests/stagnation-contracts.sh",exit_code:$exit_code,status:"completed",aggregated_output:"ordinary command output"}}' + elif [[ "$pattern_mode" == stagnation-missing-exit ]]; then + jq -cn --arg output "$failure_output" '{type:"item.completed",item:{id:"stagnation-trusted-failure",type:"command_execution",command:"bash tests/stagnation-contracts.sh",status:"completed",aggregated_output:$output}}' + else + printf '%s\n' "{\"type\":\"item.completed\",\"item\":{\"id\":\"stagnation-trusted-failure\",\"type\":\"command_execution\",\"command\":\"bash tests/stagnation-contracts.sh\",\"exit_code\":${failure_exit},\"status\":\"completed\",\"aggregated_output\":\"${failure_output}\"}}" + fi if [[ "${FAKE_STAGNATION_PASSIVE_REASONING:-false}" == "true" ]]; then jq -cn '{type:"item.completed",item:{id:"stagnation-reasoning-between",type:"reasoning",text:"safe summary"}}' fi @@ -3460,11 +3522,36 @@ awk ' capture { print } capture && /^}$/ { exit } ' "$runner" >>"$event_bounds_lib" +awk ' + /^observed_event_shape_supported\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" +awk ' + /^case_event_payload_shape_supported\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" +awk ' + /^small_fix_event_shape_supported\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" +awk ' + /^stagnation_event_shape_supported\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" awk ' /^small_fix_event_evidence\(\)/ { capture = 1 } capture { print } capture && /^}$/ { exit } ' "$runner" >>"$event_bounds_lib" +awk ' + /^small_fix_command_scope_missing_telemetry\(\)/ { capture = 1 } + capture { print } + capture && /^}$/ { exit } +' "$runner" >>"$event_bounds_lib" awk ' /^stagnation_recovery_event_evidence\(\)/ { capture = 1 } capture { print } @@ -3750,6 +3837,207 @@ else fail "small-fix discovery lifecycle accepted malformed or rejected valid evidence: ${small_lifecycle_failures[*]}" fi +test_start "small-fix unknown command scope is independent of allowed target file events" +small_command_scope_workspace="$fixture_root/small-command-scope-workspace" +mkdir -p "$small_command_scope_workspace/docs" +write_small_command_scope_control() { + local mode="$1" jsonl="$2" + + : >"$jsonl" + if [[ "$mode" == pre-discovery ]]; then + jq -cn '{type:"item.completed",item:{id:"unknown",type:"command_execution",command:"python3 transient_write_restore.py",exit_code:0}}' >>"$jsonl" + fi + jq -cn '{type:"item.completed",item:{id:"discovery",type:"command_execution",command:"rg -n teh docs/usage.md",exit_code:0}}' >>"$jsonl" + if [[ "$mode" == unknown-started ]]; then + jq -cn '{type:"item.started",item:{id:"unknown",type:"command_execution",command:"python3 transient_write_restore.py"}}' >>"$jsonl" + elif [[ "$mode" == unknown-nonzero || "$mode" == unknown-before || "$mode" == unknown-unrelated || "$mode" == unknown-failed-target ]]; then + jq -cn '{type:"item.completed",item:{id:"unknown",type:"command_execution",command:"python3 transient_write_restore.py",exit_code:7}}' >>"$jsonl" + fi + jq -cn '{type:"item.completed",item:{id:"target",type:"file_change",path:"docs/usage.md"}}' >>"$jsonl" + case "$mode" in + unknown-after) + jq -cn '{type:"item.completed",item:{id:"unknown-after",type:"command_execution",command:"python3 transient_write_restore.py",exit_code:0}}' >>"$jsonl" + ;; + unknown-unrelated) + jq -cn '{type:"item.completed",item:{id:"unrelated",type:"file_change",path:"README.md"}}' >>"$jsonl" + ;; + unknown-failed-target) + jq -cn '{type:"item.completed",item:{id:"failed-target",type:"file_change",path:"docs/usage.md",status:"failed"}}' >>"$jsonl" + ;; + known-passive) + jq -cn '{type:"item.completed",item:{id:"reasoning",type:"reasoning",text:{summary:"passive"}}}' >>"$jsonl" + ;; + esac +} +small_command_scope_failures=() +for small_command_scope_mode in unknown-before unknown-after unknown-started unknown-nonzero; do + small_command_scope_jsonl="$fixture_root/small-command-scope-$small_command_scope_mode.jsonl" + write_small_command_scope_control "$small_command_scope_mode" "$small_command_scope_jsonl" + if ! small_fix_event_evidence "$small_command_scope_jsonl" "$small_command_scope_workspace" \ + | jq -e '.unknown_command_count == 1 and .command_scope_supported == false' >/dev/null \ + || ! small_fix_command_scope_missing_telemetry "$small_command_scope_jsonl" "$small_command_scope_workspace"; then + small_command_scope_failures+=("unavailable:$small_command_scope_mode") + fi +done +for small_command_scope_mode in pre-discovery unknown-unrelated unknown-failed-target; do + small_command_scope_jsonl="$fixture_root/small-command-scope-$small_command_scope_mode.jsonl" + write_small_command_scope_control "$small_command_scope_mode" "$small_command_scope_jsonl" + if small_fix_command_scope_missing_telemetry "$small_command_scope_jsonl" "$small_command_scope_workspace"; then + small_command_scope_failures+=("masked-violation:$small_command_scope_mode") + fi +done +small_command_scope_jsonl="$fixture_root/small-command-scope-known-passive.jsonl" +write_small_command_scope_control known-passive "$small_command_scope_jsonl" +if ! small_fix_event_evidence "$small_command_scope_jsonl" "$small_command_scope_workspace" \ + | jq -e '.unknown_command_count == 0 and .command_scope_supported == true' >/dev/null \ + || small_fix_command_scope_missing_telemetry "$small_command_scope_jsonl" "$small_command_scope_workspace"; then + small_command_scope_failures+=("known-probe-or-passive-event") +fi +if [[ ${#small_command_scope_failures[@]} -eq 0 ]]; then + pass +else + fail "small-fix command telemetry gap was not kept separate from observed failures: ${small_command_scope_failures[*]}" +fi + +test_start "case payload shape checks validate consumed fields without rejecting passive or unconsumed values" +payload_shape_failures=() +payload_event_shape_supported() { + local jsonl="$1" case_id="$2" + case "$case_id" in + small-fix-stays-lightweight) small_fix_event_shape_supported "$jsonl" ;; + pivot-restart-on-stagnation-or-code-writer-blocker) stagnation_event_shape_supported "$jsonl" ;; + *) return 1 ;; + esac +} +write_command_payload() { + local path="$1" case_id="$2" mode="$3" + case "$mode" in + started) + jq -cn --arg case_id "$case_id" '{type:"item.started",item:{id:"command",type:"command_execution",command:(if $case_id == "small-fix-stays-lightweight" then "rg -n teh docs/usage.md" else "bash tests/stagnation-contracts.sh" end)}}' >"$path" + ;; + missing-exit) + jq -cn --arg case_id "$case_id" '{type:"item.completed",item:{id:"command",type:"command_execution",command:(if $case_id == "small-fix-stays-lightweight" then "rg -n teh docs/usage.md" else "bash tests/stagnation-contracts.sh" end),aggregated_output:"STAGNATION_TRUSTED_FAILURE"}}' >"$path" + ;; + null-exit|string-exit|object-exit|boolean-exit|nonzero) + jq -cn --arg case_id "$case_id" --arg mode "$mode" '{type:"item.completed",item:{id:"command",type:"command_execution",command:(if $case_id == "small-fix-stays-lightweight" then "rg -n teh docs/usage.md" else "bash tests/stagnation-contracts.sh" end),exit_code:(if $mode == "null-exit" then null elif $mode == "string-exit" then "0" elif $mode == "object-exit" then {} elif $mode == "boolean-exit" then false else 2 end),aggregated_output:"STAGNATION_TRUSTED_FAILURE"}}' >"$path" + ;; + output-null|output-number|output-object|output-array|aggregated-object-with-output-string|wrong-marker) + jq -cn --arg mode "$mode" '{type:"item.completed",item:{id:"command",type:"command_execution",command:"bash tests/stagnation-contracts.sh",exit_code:1,aggregated_output:(if $mode == "output-null" then null elif $mode == "output-number" then 3 elif $mode == "output-object" or $mode == "aggregated-object-with-output-string" then {text:"STAGNATION_TRUSTED_FAILURE"} elif $mode == "output-array" then ["STAGNATION_TRUSTED_FAILURE"] elif $mode == "wrong-marker" then "ordinary output" else null end),output:(if $mode == "aggregated-object-with-output-string" then "STAGNATION_TRUSTED_FAILURE" else "STAGNATION_TRUSTED_FAILURE" end)}}' >"$path" + ;; + recovery-read-object) + jq -cn '{type:"item.completed",item:{id:"read",type:"command_execution",command:"cat RECOVERY.md",exit_code:0,aggregated_output:{text:"unconsumed"}}}' >"$path" + ;; + esac +} +write_file_payload() { + local path="$1" case_id="$2" mode="$3" + jq -cn --arg case_id "$case_id" --arg mode "$mode" ' + {type:"item.completed",item:{id:"file",type:"file_change"}} + | if $mode == "missing" then . + elif $mode == "nonarray" then .item.changes={bad:true} + elif $mode == "empty" then .item.changes=[] + elif $mode == "nonobject" then .item.changes=["docs/usage.md"] + elif $mode == "nonstring-path" then .item.changes=[{path:17}] + elif $mode == "null-item-path" then .item.path=null + elif $mode == "outside" then .item.path="/tmp/outside.txt" + elif $mode == "status-absent" then .item.path="docs/usage.md" + elif $mode == "status-failed" then .item.path="docs/usage.md" | .item.status="failed" + elif $mode == "status-null" then .item.path="docs/usage.md" | .item.status=null + elif $mode == "status-boolean" then .item.path="docs/usage.md" | .item.status=false + elif $mode == "status-unknown" then .item.path="docs/usage.md" | .item.status="pending" + else .item.changes=[{path:"docs/usage.md",kind:"update"}] end + ' >"$path" +} +for payload_case in small-fix-stays-lightweight pivot-restart-on-stagnation-or-code-writer-blocker; do + for payload_mode in missing-exit null-exit string-exit object-exit boolean-exit; do + payload_path="$fixture_root/payload-$payload_case-$payload_mode.jsonl" + write_command_payload "$payload_path" "$payload_case" "$payload_mode" + if payload_event_shape_supported "$payload_path" "$payload_case"; then + payload_shape_failures+=("accepted-$payload_case-$payload_mode") + fi + done + payload_path="$fixture_root/payload-$payload_case-nonzero.jsonl" + write_command_payload "$payload_path" "$payload_case" nonzero + if ! payload_event_shape_supported "$payload_path" "$payload_case"; then + payload_shape_failures+=("rejected-numeric-nonzero-$payload_case") + fi + payload_path="$fixture_root/payload-$payload_case-started.jsonl" + write_command_payload "$payload_path" "$payload_case" started + if ! payload_event_shape_supported "$payload_path" "$payload_case"; then + payload_shape_failures+=("rejected-started-without-exit-output-$payload_case") + fi +done +for payload_mode in missing nonarray empty nonobject nonstring-path null-item-path; do + for payload_case in small-fix-stays-lightweight pivot-restart-on-stagnation-or-code-writer-blocker; do + payload_path="$fixture_root/payload-file-$payload_case-$payload_mode.jsonl" + write_file_payload "$payload_path" "$payload_case" "$payload_mode" + if payload_event_shape_supported "$payload_path" "$payload_case"; then + payload_shape_failures+=("accepted-malformed-file-$payload_case-$payload_mode") + fi + done +done +for payload_mode in status-absent status-failed; do + payload_path="$fixture_root/payload-file-small-$payload_mode.jsonl" + write_file_payload "$payload_path" small-fix-stays-lightweight "$payload_mode" + if ! payload_event_shape_supported "$payload_path" small-fix-stays-lightweight; then + payload_shape_failures+=("rejected-supported-file-status-$payload_mode") + fi +done +for payload_mode in status-null status-boolean status-unknown; do + payload_path="$fixture_root/payload-file-small-$payload_mode.jsonl" + write_file_payload "$payload_path" small-fix-stays-lightweight "$payload_mode" + if payload_event_shape_supported "$payload_path" small-fix-stays-lightweight; then + payload_shape_failures+=("accepted-unsupported-file-status-$payload_mode") + fi +done +payload_path="$fixture_root/payload-file-small-outside.jsonl" +write_file_payload "$payload_path" small-fix-stays-lightweight outside +if ! small_fix_event_shape_supported "$payload_path" \ + || ! small_fix_event_evidence "$payload_path" "$small_command_scope_workspace" \ + | jq -e '.file_change_scope_valid == false' >/dev/null; then + payload_shape_failures+=("outside-path-was-treated-as-malformed-shape") +fi +payload_path="$fixture_root/payload-file-stagnation-unconsumed-status.jsonl" +write_file_payload "$payload_path" pivot-restart-on-stagnation-or-code-writer-blocker status-boolean +if ! stagnation_event_shape_supported "$payload_path"; then + payload_shape_failures+=("constrained-unconsumed-stagnation-file-status") +fi +payload_path="$fixture_root/payload-small-output-object.jsonl" +write_command_payload "$payload_path" small-fix-stays-lightweight output-object +if ! small_fix_event_shape_supported "$payload_path"; then + payload_shape_failures+=("small-fix-constrained-unconsumed-command-output") +fi +payload_path="$fixture_root/payload-stagnation-output.jsonl" +write_command_payload "$payload_path" pivot-restart-on-stagnation-or-code-writer-blocker output-object +if payload_event_shape_supported "$payload_path" pivot-restart-on-stagnation-or-code-writer-blocker; then + payload_shape_failures+=("accepted-object-trusted-command-output") +fi +for payload_mode in output-number output-array aggregated-object-with-output-string; do + payload_path="$fixture_root/payload-stagnation-$payload_mode.jsonl" + write_command_payload "$payload_path" pivot-restart-on-stagnation-or-code-writer-blocker "$payload_mode" + if payload_event_shape_supported "$payload_path" pivot-restart-on-stagnation-or-code-writer-blocker; then + payload_shape_failures+=("accepted-malformed-selected-output-$payload_mode") + fi +done +for payload_mode in output-null wrong-marker recovery-read-object; do + payload_path="$fixture_root/payload-stagnation-$payload_mode.jsonl" + write_command_payload "$payload_path" pivot-restart-on-stagnation-or-code-writer-blocker "$payload_mode" + if ! stagnation_event_shape_supported "$payload_path"; then + payload_shape_failures+=("rejected-string-or-unconsumed-output-$payload_mode") + fi +done +payload_path="$fixture_root/payload-passive-object.jsonl" +jq -cn '{type:"item.completed",item:{id:"passive",type:"reasoning",text:{summary:"unconsumed"}}}' >"$payload_path" +if ! small_fix_event_shape_supported "$payload_path" \ + || ! stagnation_event_shape_supported "$payload_path"; then + payload_shape_failures+=("constrained-passive-reasoning-payload") +fi +if [[ ${#payload_shape_failures[@]} -eq 0 ]]; then + pass +else + fail "case consumed-payload validation admitted malformed telemetry or rejected compatible values: ${payload_shape_failures[*]}" +fi + test_start "stagnation event paths accept only normalized recovery-artifact changes" stagnation_path_workspace="$fixture_root/stagnation-path-controls-workspace" mkdir -p "$stagnation_path_workspace/.assistant-eval" @@ -3816,6 +4104,105 @@ else fail "stagnation event path normalization accepted or rejected an incorrect workspace boundary: ${stagnation_path_failures[*]}" fi +test_start "stagnation recovery-artifact events stay inside the recovery-to-fresh-check interval" +recovery_window_workspace="$fixture_root/recovery-window-workspace" +mkdir -p "$recovery_window_workspace/.assistant-eval" +emit_window_event() { + local jsonl="$1" type="$2" id="$3" command="$4" exit_code="$5" output="$6" + if [[ "$type" == started ]]; then + jq -cn --arg id "$id" --arg command "$command" '{type:"item.started",item:{id:$id,type:"command_execution",command:$command}}' >>"$jsonl" + else + jq -cn --arg id "$id" --arg command "$command" --argjson exit_code "$exit_code" --arg output "$output" '{type:"item.completed",item:{id:$id,type:"command_execution",command:$command,exit_code:$exit_code,aggregated_output:$output}}' >>"$jsonl" + fi +} +emit_window_file_change() { + local jsonl="$1" event_type="$2" + jq -cn --arg type "$event_type" '{type:("item." + $type),item:{id:"recovery-artifact",type:"file_change",path:".assistant-eval/stagnation-recovery.json"}}' >>"$jsonl" +} +write_recovery_window_control() { + local position="$1" event_type="$2" jsonl="$3" + : >"$jsonl" + case "$position" in + before-failure) + emit_window_file_change "$jsonl" "$event_type" + emit_window_event "$jsonl" completed failure 'bash tests/stagnation-contracts.sh' 1 STAGNATION_TRUSTED_FAILURE + emit_window_event "$jsonl" completed recovery 'bash tests/recovery-contracts.sh' 0 RECOVERY_APPLIED + emit_window_event "$jsonl" completed fresh 'bash tests/stagnation-contracts.sh --after-recovery' 0 STAGNATION_FRESH_CHECK_PASS + ;; + after-failure|before-recovery) + emit_window_event "$jsonl" completed failure 'bash tests/stagnation-contracts.sh' 1 STAGNATION_TRUSTED_FAILURE + emit_window_file_change "$jsonl" "$event_type" + emit_window_event "$jsonl" started recovery 'bash tests/recovery-contracts.sh' '' '' + emit_window_event "$jsonl" completed recovery 'bash tests/recovery-contracts.sh' 0 RECOVERY_APPLIED + emit_window_event "$jsonl" started fresh 'bash tests/stagnation-contracts.sh --after-recovery' '' '' + emit_window_event "$jsonl" completed fresh 'bash tests/stagnation-contracts.sh --after-recovery' 0 STAGNATION_FRESH_CHECK_PASS + ;; + during-recovery) + emit_window_event "$jsonl" completed failure 'bash tests/stagnation-contracts.sh' 1 STAGNATION_TRUSTED_FAILURE + emit_window_event "$jsonl" started recovery 'bash tests/recovery-contracts.sh' '' '' + emit_window_file_change "$jsonl" "$event_type" + emit_window_event "$jsonl" completed recovery 'bash tests/recovery-contracts.sh' 0 RECOVERY_APPLIED + emit_window_event "$jsonl" started fresh 'bash tests/stagnation-contracts.sh --after-recovery' '' '' + emit_window_event "$jsonl" completed fresh 'bash tests/stagnation-contracts.sh --after-recovery' 0 STAGNATION_FRESH_CHECK_PASS + ;; + after-recovery) + emit_window_event "$jsonl" completed failure 'bash tests/stagnation-contracts.sh' 1 STAGNATION_TRUSTED_FAILURE + emit_window_event "$jsonl" completed recovery 'bash tests/recovery-contracts.sh' 0 RECOVERY_APPLIED + emit_window_file_change "$jsonl" "$event_type" + emit_window_event "$jsonl" completed fresh 'bash tests/stagnation-contracts.sh --after-recovery' 0 STAGNATION_FRESH_CHECK_PASS + ;; + after-fresh-start) + emit_window_event "$jsonl" completed failure 'bash tests/stagnation-contracts.sh' 1 STAGNATION_TRUSTED_FAILURE + emit_window_event "$jsonl" completed recovery 'bash tests/recovery-contracts.sh' 0 RECOVERY_APPLIED + emit_window_event "$jsonl" started fresh 'bash tests/stagnation-contracts.sh --after-recovery' '' '' + emit_window_file_change "$jsonl" "$event_type" + emit_window_event "$jsonl" completed fresh 'bash tests/stagnation-contracts.sh --after-recovery' 0 STAGNATION_FRESH_CHECK_PASS + ;; + after-fresh-completion) + emit_window_event "$jsonl" completed failure 'bash tests/stagnation-contracts.sh' 1 STAGNATION_TRUSTED_FAILURE + emit_window_event "$jsonl" completed recovery 'bash tests/recovery-contracts.sh' 0 RECOVERY_APPLIED + emit_window_event "$jsonl" completed fresh 'bash tests/stagnation-contracts.sh --after-recovery' 0 STAGNATION_FRESH_CHECK_PASS + emit_window_file_change "$jsonl" "$event_type" + ;; + no-change) + emit_window_event "$jsonl" completed failure 'bash tests/stagnation-contracts.sh' 1 STAGNATION_TRUSTED_FAILURE + emit_window_event "$jsonl" completed recovery 'bash tests/recovery-contracts.sh' 0 RECOVERY_APPLIED + emit_window_event "$jsonl" completed fresh 'bash tests/stagnation-contracts.sh --after-recovery' 0 STAGNATION_FRESH_CHECK_PASS + ;; + esac +} +recovery_window_failures=() +for recovery_window_event_type in started completed; do + for recovery_window_position in before-failure after-failure before-recovery after-fresh-start after-fresh-completion; do + recovery_window_jsonl="$fixture_root/recovery-window-$recovery_window_event_type-$recovery_window_position.jsonl" + write_recovery_window_control "$recovery_window_position" "$recovery_window_event_type" "$recovery_window_jsonl" + recovery_window_evidence="$(stagnation_recovery_event_evidence "$recovery_window_jsonl" "$recovery_window_workspace")" + if jq -e '.reject_patch_or_retry_after_bound == true' <<<"$recovery_window_evidence" >/dev/null; then + recovery_window_failures+=("accepted-$recovery_window_event_type-$recovery_window_position") + fi + done + for recovery_window_position in during-recovery after-recovery; do + recovery_window_jsonl="$fixture_root/recovery-window-$recovery_window_event_type-$recovery_window_position.jsonl" + write_recovery_window_control "$recovery_window_position" "$recovery_window_event_type" "$recovery_window_jsonl" + recovery_window_evidence="$(stagnation_recovery_event_evidence "$recovery_window_jsonl" "$recovery_window_workspace")" + if ! stagnation_event_shape_supported "$recovery_window_jsonl" \ + || ! jq -e '.trusted_failure_before_recovery == true and .recovery_before_fresh_check == true and .reject_patch_or_retry_after_bound == true' <<<"$recovery_window_evidence" >/dev/null; then + recovery_window_failures+=("rejected-$recovery_window_event_type-$recovery_window_position") + fi + done +done +recovery_window_jsonl="$fixture_root/recovery-window-no-change.jsonl" +write_recovery_window_control no-change completed "$recovery_window_jsonl" +recovery_window_evidence="$(stagnation_recovery_event_evidence "$recovery_window_jsonl" "$recovery_window_workspace")" +if ! jq -e '.trusted_failure_before_recovery == true and .recovery_before_fresh_check == true and .reject_patch_or_retry_after_bound == true' <<<"$recovery_window_evidence" >/dev/null; then + recovery_window_failures+=("rejected-no-file-event") +fi +if [[ ${#recovery_window_failures[@]} -eq 0 ]]; then + pass +else + fail "stagnation recovery artifact event crossed a phase boundary: ${recovery_window_failures[*]}" +fi + test_start "VIEWING evidence mutations are isolated to the candidate verifier" viewing_mutation_failures=() for viewing_mutation in omitted_ref stale_hashes bad_path bad_symbol bad_assertion bad_event_ref missing_event extra_top_key extra_evidence_key extra_item_key extra_design_key extra_implementation_key extra_trace_key extra_behavioral_test_key extra_result_key prepended_document appended_document; do @@ -6173,6 +6560,146 @@ else fail "command-only unavailable classification hid independently observed failures: ${small_command_only_mixed_failures[*]}" fi +test_start "small-fix unknown command scope excludes clean pairs and preserves observed failures" +small_unknown_command_failures=() +run_small_unknown_command_case() { + local output="$1" mode="$2" + shift 2 + rm -f "$capture"/* + env FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-unknown-command "$@" \ + "$runner" --execute --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$output" --codex-bin "$fake_codex" >/dev/null +} +for small_unknown_mode in before-target after-target started nonzero; do + small_unknown_output="$fixture_root/small-unknown-$small_unknown_mode-output" + case "$small_unknown_mode" in + before-target) + if ! run_small_unknown_command_case "$small_unknown_output" "$small_unknown_mode" \ + FAKE_SMALL_UNKNOWN_COMMAND_MODE=before-target FAKE_SMALL_TRANSIENT_COMMAND_MUTATION=true; then + small_unknown_command_failures+=("runner:$small_unknown_mode") + elif [[ "$(find "$capture" -maxdepth 1 -name 'call-*.transient-mutation.json' | wc -l | tr -d ' ')" -ne 2 ]] \ + || ! jq -s -e 'length == 2 and all(.[]; .created_sha256 | type == "string" and length == 64) and all(.[]; .removed == true)' "$capture"/call-*.transient-mutation.json >/dev/null; then + small_unknown_command_failures+=("mutation-not-observed:$small_unknown_mode") + fi + ;; + started) + if ! run_small_unknown_command_case "$small_unknown_output" "$small_unknown_mode" \ + FAKE_SMALL_UNKNOWN_COMMAND_MODE=started FAKE_SMALL_UNKNOWN_COMMAND_EVENT=started; then + small_unknown_command_failures+=("runner:$small_unknown_mode") + fi + ;; + nonzero) + if ! run_small_unknown_command_case "$small_unknown_output" "$small_unknown_mode" \ + FAKE_SMALL_UNKNOWN_COMMAND_MODE=nonzero FAKE_SMALL_UNKNOWN_COMMAND_EXIT=7; then + small_unknown_command_failures+=("runner:$small_unknown_mode") + fi + ;; + after-target) + if ! run_small_unknown_command_case "$small_unknown_output" "$small_unknown_mode" \ + FAKE_SMALL_UNKNOWN_COMMAND_MODE=after-target; then + small_unknown_command_failures+=("runner:$small_unknown_mode") + fi + ;; + esac + if [[ -d "$small_unknown_output/traces" ]] \ + && jq -s -e 'length == 2 and all(.[]; .status == "adapter_unavailable" and (has("metrics") | not) and .error.code == "unknown_event_shape")' "$small_unknown_output/traces/"*.json >/dev/null \ + && jq -e '.complete_pairs == 0 and .excluded_incomplete_pairs == 1 and .incomplete_pairs[0].case_id == "small-fix-stays-lightweight"' "$small_unknown_output/comparison.json" >/dev/null; then + : + else + small_unknown_command_failures+=("not-excluded:$small_unknown_mode") + fi +done +for small_unknown_mode in pre-discovery mixed-unrelated failed-target; do + small_unknown_output="$fixture_root/small-unknown-$small_unknown_mode-output" + if ! run_small_unknown_command_case "$small_unknown_output" "$small_unknown_mode" \ + FAKE_SMALL_UNKNOWN_COMMAND_MODE="$small_unknown_mode"; then + small_unknown_command_failures+=("runner:$small_unknown_mode") + elif ! jq -s -e 'length == 2 and all(.[]; .status == "completed" and .metrics.acceptance_passed == false and (.execution.verifier.workspace_failure_ids | index("workspace-003")) != null)' "$small_unknown_output/traces/"*.json >/dev/null; then + small_unknown_command_failures+=("masked-observed-failure:$small_unknown_mode") + fi +done +for small_unknown_guard in plan scope response final; do + small_unknown_output="$fixture_root/small-unknown-guard-$small_unknown_guard-output" + case "$small_unknown_guard" in + plan) small_unknown_env=(FAKE_SMALL_PLAN_MODE=full) ;; + scope) small_unknown_env=(FAKE_SCOPE_DEVIATION=true) ;; + response) small_unknown_env=(FAKE_SMALL_BROAD=true) ;; + final) small_unknown_env=(FAKE_WRONG_SMALL_EDIT=true) ;; + esac + if ! run_small_unknown_command_case "$small_unknown_output" before-target "${small_unknown_env[@]}"; then + small_unknown_command_failures+=("runner:guard-$small_unknown_guard") + elif [[ "$small_unknown_guard" == plan ]] \ + && ! jq -s -e 'length == 2 and all(.[] | select(.variant == "candidate"); .status == "completed" and .metrics.acceptance_passed == false and .execution.verifier.workspace_failure_ids == ["workspace-002", "workspace-003"])' "$small_unknown_output/traces/"*.json >/dev/null; then + small_unknown_command_failures+=("masked-guard-$small_unknown_guard") + elif [[ "$small_unknown_guard" != plan ]] \ + && ! jq -s -e 'length == 2 and all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$small_unknown_output/traces/"*.json >/dev/null; then + small_unknown_command_failures+=("masked-guard-$small_unknown_guard") + fi +done +if [[ ${#small_unknown_command_failures[@]} -eq 0 ]]; then + pass +else + fail "unknown command scope was promoted or concealed an observed small-fix failure: ${small_unknown_command_failures[*]}" +fi + +test_start "small-fix malformed consumed event shapes are unavailable while numeric failures remain behavioral" +small_payload_runtime_failures=() +for small_payload_mode in small-missing-exit small-malformed-exit small-malformed-file-change; do + small_payload_output="$fixture_root/$small_payload_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$small_payload_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_payload_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e 'length == 2 and all(.[]; .status == "adapter_unavailable" and (has("metrics") | not) and .error.code == "unknown_event_shape")' "$small_payload_output/traces/"*.json >/dev/null \ + || ! jq -e '.complete_pairs == 0 and .excluded_incomplete_pairs == 1' "$small_payload_output/comparison.json" >/dev/null; then + small_payload_runtime_failures+=("unsupported-shape:$small_payload_mode") + fi +done +small_nonzero_output="$fixture_root/small-nonzero-exit-output" +rm -f "$capture"/* +if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-nonzero-exit "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases small-fix-stays-lightweight --repeats 1 --output "$small_nonzero_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e 'length == 2 and all(.[]; .status == "completed" and .metrics.acceptance_passed == false)' "$small_nonzero_output/traces/"*.json >/dev/null; then + small_payload_runtime_failures+=("numeric-nonzero-was-unavailable-or-passed") +fi +if [[ ${#small_payload_runtime_failures[@]} -eq 0 ]]; then + pass +else + fail "small-fix malformed payload or valid nonzero exit classification regressed: ${small_payload_runtime_failures[*]}" +fi + +test_start "stagnation consumed command payloads fail closed while temporal violations remain graded" +stagnation_payload_runtime_failures=() +for stagnation_payload_mode in stagnation-malformed-output stagnation-missing-exit; do + stagnation_payload_output="$fixture_root/$stagnation_payload_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$stagnation_payload_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 \ + --output "$stagnation_payload_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e 'length == 2 and all(.[]; .status == "adapter_unavailable" and (has("metrics") | not) and .error.code == "unknown_event_shape")' "$stagnation_payload_output/traces/"*.json >/dev/null \ + || ! jq -e '.complete_pairs == 0 and .excluded_incomplete_pairs == 1' "$stagnation_payload_output/comparison.json" >/dev/null; then + stagnation_payload_runtime_failures+=("unsupported-shape:$stagnation_payload_mode") + fi +done +for stagnation_payload_mode in stagnation-wrong-marker stagnation-premature-recovery-artifact; do + stagnation_payload_output="$fixture_root/$stagnation_payload_mode-output" + rm -f "$capture"/* + if ! FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE="$stagnation_payload_mode" "$runner" --execute \ + --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ + --cases pivot-restart-on-stagnation-or-code-writer-blocker --repeats 1 \ + --output "$stagnation_payload_output" --codex-bin "$fake_codex" >/dev/null \ + || ! jq -s -e 'length == 2 and all(.[]; .status == "completed" and .metrics.acceptance_passed == false and (.execution.verifier.workspace_failure_ids | index("workspace-003")) != null)' "$stagnation_payload_output/traces/"*.json >/dev/null; then + stagnation_payload_runtime_failures+=("behavioral-failure:$stagnation_payload_mode") + fi +done +if [[ ${#stagnation_payload_runtime_failures[@]} -eq 0 ]]; then + pass +else + fail "stagnation consumed payload or recovery-event window was misclassified: ${stagnation_payload_runtime_failures[*]}" +fi + rm -f "$capture"/* if FAKE_CODEX_CAPTURE_DIR="$capture" FAKE_PATTERN_EVENT_MODE=small-external-symlink "$runner" --execute \ --model test-model --baseline-variant "$baseline" --candidate-variant "$candidate" \ diff --git a/tools/evals/run-codex-framework-evals.sh b/tools/evals/run-codex-framework-evals.sh index 20a8b1a..899240a 100755 --- a/tools/evals/run-codex-framework-evals.sh +++ b/tools/evals/run-codex-framework-evals.sh @@ -2380,7 +2380,7 @@ small_fix_event_evidence() { def is_change: .type == "item.completed" and .item.type == "file_change" - and ((.item.status? // "completed") == "completed") + and ((.item | has("status") | not) or .item.status == "completed") and changes_containing("docs/usage.md"); def is_observed_file_change: (.type == "item.started" or .type == "item.completed" or .type == "item.updated") @@ -2394,6 +2394,12 @@ small_fix_event_evidence() { any(.[]; (.type == "item.started" or .type == "item.completed" or .type == "item.updated") and .item.type == "file_change"); + def is_completed_file_change: + .type == "item.completed" and .item.type == "file_change"; + def is_unknown_command: + (.type == "item.started" or .type == "item.completed") + and .item.type == "command_execution" + and (is_exact_command("rg -n teh docs/usage.md") | not); . as $events | [range(0; length) | select($events[.] | is_discovery)] as $discoveries @@ -2407,6 +2413,10 @@ small_fix_event_evidence() { | {index: $index} ] as $successful_commands | ($events | has_file_change_event) as $has_file_change_event + | ([range(0; length) | select($events[.] | is_unknown_command)] | length) as $unknown_command_count + | ([range(0; length) | $events[.] | select(is_completed_file_change and (.item | has("status")))]) as $status_events + | ($status_events | all(.[]; .item.status == "completed")) as $file_change_status_valid + | ([range(0; length) | select($events[.] | is_completed_file_change and .item.status == "failed")] | length) as $failed_file_change_count | [range(0; length) | . as $index | $events[$index] @@ -2434,14 +2444,28 @@ small_fix_event_evidence() { )) as $discovery_lifecycles_valid | [range(0; length) | select($events[.] | is_workspace_action)] as $workspace_actions | [range(0; length) | select($events[.] | is_discovery_action)] as $discovery_actions + | (($discoveries | length) > 0 + and $discovery_lifecycles_valid + and ($workspace_actions | length) > 0 + and ($discovery_actions | length) > 0 + and $workspace_actions[0] == $discovery_actions[0] + and all($workspace_actions[]; . as $action_index + | $action_index >= $discoveries[0] + or ($discovery_actions | index($action_index)) != null) + and (($changes | length) == 0 or $discoveries[0] < $changes[0])) as $discovery_sequence_valid | [range(0; length) | select( ($events[.].type == "item.completed" or $events[.].type == "item.started" or $events[.].type == "item.updated") and ($events[.].item.type == "mcp_tool_call" or $events[.].item.type == "web_search"))] as $disallowed | { - source_discovery_before_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($changes | length) > 0 and $file_change_scope_valid and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and $discoveries[0] < $changes[0]), + source_discovery_before_change: ($discovery_sequence_valid and ($changes | length) > 0 and $file_change_scope_valid and $file_change_status_valid), command_only_change_without_file_change: (($discoveries | length) > 0 and $discovery_lifecycles_valid and ($workspace_actions | length) > 0 and ($discovery_actions | length) > 0 and ($successful_commands | length) > 0 and ($changes | length) == 0 and ($has_file_change_event | not) and $workspace_actions[0] == $discovery_actions[0] and all($workspace_actions[]; . as $action_index | $action_index >= $discoveries[0] or ($discovery_actions | index($action_index)) != null) and any($successful_commands[]; .index > $discoveries[0])), + unknown_command_count: $unknown_command_count, + command_scope_supported: ($unknown_command_count == 0), disallowed_item_count: ($disallowed | length), - file_change_scope_valid: $file_change_scope_valid + file_change_scope_valid: $file_change_scope_valid, + file_change_status_valid: $file_change_status_valid, + failed_file_change_count: $failed_file_change_count, + discovery_sequence_valid: $discovery_sequence_valid } JQ )" "$jsonl" @@ -2456,6 +2480,18 @@ small_fix_command_only_edit_missing_telemetry() { <<<"$event_evidence" >/dev/null } +small_fix_command_scope_missing_telemetry() { + local jsonl="$1" workspace="$2" event_evidence + event_evidence="$(small_fix_event_evidence "$jsonl" "$workspace")" || return 1 + jq -e ' + .discovery_sequence_valid == true + and .file_change_scope_valid == true + and .file_change_status_valid == true + and .disallowed_item_count == 0 + and ((.unknown_command_count > 0) or .command_only_change_without_file_change == true) + ' <<<"$event_evidence" >/dev/null +} + observed_event_shape_supported() { local jsonl="$1" allowed_item_types="$2" jq -se --argjson allowed_item_types "$allowed_item_types" ' @@ -2468,12 +2504,59 @@ observed_event_shape_supported() { ' "$jsonl" >/dev/null } +case_event_payload_shape_supported() { + local jsonl="$1" case_id="$2" allowed_item_types + case "$case_id" in + small-fix-stays-lightweight) + allowed_item_types='["agent_message","reasoning","command_execution","file_change","mcp_tool_call","web_search"]' + ;; + pivot-restart-on-stagnation-or-code-writer-blocker) + allowed_item_types='["agent_message","reasoning","command_execution","file_change"]' + ;; + *) return 1 ;; + esac + observed_event_shape_supported "$jsonl" "$allowed_item_types" || return 1 + jq -cse --arg case_id "$case_id" "$EVENT_EVIDENCE_NORMALIZATION_JQ$(cat <<'JQ' + all(.[]; + if (.type == "item.started" or .type == "item.completed") + and .item.type == "command_execution" then + (.item.command // .item.command_line // null) as $command + | (($command | type) == "string" + or (($command | type) == "array" + and ($command | length) > 0 + and all($command[]; type == "string"))) + and (.type == "item.started" or (.item.exit_code | type) == "number") + and (if $case_id == "pivot-restart-on-stagnation-or-code-writer-blocker" + and .type == "item.completed" + and (is_exact_command("bash tests/stagnation-contracts.sh") + or is_exact_command("bash tests/recovery-contracts.sh") + or is_exact_command("bash tests/stagnation-contracts.sh --after-recovery")) then + ((.item.aggregated_output // .item.output // null) | type) == "string" + else true end) + elif (.type == "item.started" or .type == "item.completed") + and .item.type == "file_change" then + ((.item | has("path") | not) or (.item.path | type) == "string") + and ((.item | has("changes") | not) + or ((.item.changes | type) == "array" + and (.item.changes | length) > 0 + and all(.item.changes[]; type == "object" and has("path") and (.path | type) == "string"))) + and ($case_id != "small-fix-stays-lightweight" + or .type != "item.completed" + or (.item | has("status") | not) + or ((.item.status | type) == "string" + and (.item.status == "completed" or .item.status == "failed"))) + and ((.item | has("path")) or (.item | has("changes"))) + else true end) +JQ + )" "$jsonl" >/dev/null 2>&1 +} + small_fix_event_shape_supported() { - observed_event_shape_supported "$1" '["agent_message","reasoning","command_execution","file_change","mcp_tool_call","web_search"]' + case_event_payload_shape_supported "$1" "small-fix-stays-lightweight" } stagnation_event_shape_supported() { - observed_event_shape_supported "$1" '["agent_message","reasoning","command_execution","file_change"]' + case_event_payload_shape_supported "$1" "pivot-restart-on-stagnation-or-code-writer-blocker" } small_fix_requires_unavailable_adapter_policy() { @@ -2491,7 +2574,7 @@ stagnation_recovery_event_evidence() { def changes_only($expected): normalized_file_change_paths as $paths | $paths != null and ($paths | length) > 0 and all($paths[]; . == $expected); - def command_output: (.item.aggregated_output // .item.output // "" | tostring); + def command_output: (.item.aggregated_output // .item.output // null); def has_numeric_exit_code: (.item.exit_code | type) == "number"; def is_initial_stagnation_invocation: @@ -2500,6 +2583,7 @@ stagnation_recovery_event_evidence() { def is_failed_stagnation: is_initial_stagnation_invocation and has_numeric_exit_code and .item.exit_code != 0 + and (command_output | type) == "string" and (command_output | contains("STAGNATION_TRUSTED_FAILURE")); def is_recovery_invocation: .type == "item.completed" and .item.type == "command_execution" @@ -2507,6 +2591,7 @@ stagnation_recovery_event_evidence() { def is_recovery: is_recovery_invocation and has_numeric_exit_code and .item.exit_code == 0 + and (command_output | type) == "string" and (command_output | contains("RECOVERY_APPLIED")); def is_fresh_check_invocation: .type == "item.completed" and .item.type == "command_execution" @@ -2514,6 +2599,7 @@ stagnation_recovery_event_evidence() { def is_fresh_check: is_fresh_check_invocation and has_numeric_exit_code and .item.exit_code == 0 + and (command_output | type) == "string" and (command_output | contains("STAGNATION_FRESH_CHECK_PASS")); def recovery_command_kind: if is_exact_command("bash tests/stagnation-contracts.sh") then "trusted_failure" @@ -2536,6 +2622,10 @@ stagnation_recovery_event_evidence() { (.type == "item.started" or .type == "item.completed") and .item.type == "file_change" and (changes_only(".assistant-eval/stagnation-recovery.json") | not); + def is_recovery_artifact_change: + (.type == "item.started" or .type == "item.completed") + and .item.type == "file_change" + and changes_only(".assistant-eval/stagnation-recovery.json"); def is_workspace_action: (.type == "item.started" or .type == "item.completed") and (.item.type == "command_execution" or .item.type == "file_change"); @@ -2571,6 +2661,16 @@ stagnation_recovery_event_evidence() { | [range(0; length) | select($events[.] | is_forbidden_recovery_command)] as $forbidden_commands | [range(0; length) | select($events[.] | is_forbidden_retry_or_patch)] as $forbidden | [range(0; length) | select($events[.] | is_forbidden_source_change)] as $forbidden_changes + | [range(0; length) + | . as $index + | select(($events[$index] | is_recovery_artifact_change) + and (($recovery_start_or_completion == null) + or ($fresh_check_start_or_completion == null) + or ($failures | length) != 1 + or $index <= $failures[0] + or $index <= $recovery_start_or_completion + or $index >= $fresh_check_start_or_completion)) + ] as $out_of_window_changes | [range(0; length) | select( . as $index | ($events[$index] | is_workspace_action) @@ -2580,7 +2680,7 @@ stagnation_recovery_event_evidence() { | { trusted_failure_before_recovery: (($initial_invocations | length) == 1 and ($failures | length) == 1 and ($recovery_invocations | length) == 1 and ($recoveries | length) == 1 and $command_lifecycles_valid and $failures[0] < $recovery_start_or_completion), recovery_before_fresh_check: (($recovery_invocations | length) == 1 and ($fresh_check_invocations | length) == 1 and ($fresh_checks | length) == 1 and $command_lifecycles_valid and $recoveries[0] < $fresh_check_start_or_completion), - reject_patch_or_retry_after_bound: (($forbidden_commands | length) == 0 and ($forbidden | length) == 0 and ($forbidden_changes | length) == 0 and $command_lifecycles_valid and ($fresh_checks | length) == 1 and ($post_bound_workspace_actions | length) == 0), + reject_patch_or_retry_after_bound: (($forbidden_commands | length) == 0 and ($forbidden | length) == 0 and ($forbidden_changes | length) == 0 and ($out_of_window_changes | length) == 0 and $command_lifecycles_valid and ($fresh_checks | length) == 1 and ($post_bound_workspace_actions | length) == 0), terminal_completed: false } JQ @@ -2906,7 +3006,7 @@ verify_workspace() { and .plan_mode == "none"' event_evidence="$(small_fix_event_evidence "$jsonl" "$workspace_identity")" || event_evidence='{}' if small_fix_requires_unavailable_adapter_policy \ - && jq -e '.source_discovery_before_change == true and .disallowed_item_count == 0' <<<"$event_evidence" >/dev/null; then + && jq -e '.source_discovery_before_change == true and .command_scope_supported == true and .disallowed_item_count == 0' <<<"$event_evidence" >/dev/null; then workspace_record_check "workspace-003" true else workspace_record_check "workspace-003" false @@ -3674,7 +3774,7 @@ execute_one_run() { '$response.status == "passed" and $workspace.workspace_failure_ids == ["workspace-003"] and $workspace.scope_deviations == 0' >/dev/null \ - && small_fix_command_only_edit_missing_telemetry "$jsonl" "$workspace"; then + && small_fix_command_scope_missing_telemetry "$jsonl" "$workspace"; then write_unavailable_trace "$trace_path" "$run_id" "$pair_id" "$case_id" "$trial_index" "$variant" \ unknown_event_shape 0 "$fixture_hash" "$case_digest" "$instruction_hash" "$grader_digest" "$cli_version" mark_run_attempt_completed "$attempt_path" || die "Could not complete unavailable run-attempt state for $run_id."