diff --git a/apps/presentation/dashboard/src/data/status.ts b/apps/presentation/dashboard/src/data/status.ts index 61d91e2d41..a835dd9388 100644 --- a/apps/presentation/dashboard/src/data/status.ts +++ b/apps/presentation/dashboard/src/data/status.ts @@ -363,8 +363,8 @@ export const projectAssetTodoProjectionGapSchema = z.object({ export const nativeChildActivitySchema = z.object({ schema_version: z.literal("native_subagent_activity_v0"), - observation: z.enum(["unknown", "coordinator_reported"]), - host_attested: z.literal(false), + observation: z.enum(["unknown", "coordinator_reported", "host_observed", "mixed"]), + host_attested: z.boolean(), configured_limit: z.number().int().nonnegative(), launched_count: z.number().int().nonnegative(), skipped_count: z.number().int().nonnegative(), diff --git a/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx b/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx index 6415d75441..42c79930cd 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx @@ -801,10 +801,12 @@ export function ContextDrawer({ agents, attentionHistory = [], onSelectAttention : null} - {selection.item.nativeChildActivity?.observation === "coordinator_reported" ? ( + {selection.item.nativeChildActivity && selection.item.nativeChildActivity.observation !== "unknown" ? (

{t("drawer.subagentReportTitle")}

-

{t("drawer.subagentReportedActivity", { +

{t(selection.item.nativeChildActivity.observation === "host_observed" + ? "drawer.subagentHostActivity" : selection.item.nativeChildActivity.observation === "mixed" + ? "drawer.subagentMixedActivity" : "drawer.subagentReportedActivity", { started: selection.item.nativeChildActivity.launched_count, skipped: selection.item.nativeChildActivity.skipped_count, rejected: selection.item.nativeChildActivity.capacity_rejected_count, diff --git a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx index bb592e78a2..1f0ecc3a84 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx @@ -296,6 +296,8 @@ const en = { "drawer.subagentCurrentBoundary": "Current task-domain restriction", "drawer.subagentDescription": "Allows the runtime to create temporary child agents for independent tasks only after Todo, quota, capability, and write-scope gates pass. It does not force parallel work or grant durable authority.", "drawer.subagentReportTitle": "Child activity", + "drawer.subagentHostActivity": "Host-observed activity: {started} starts, {rejected} capacity rejections, {failed} host failures, {accepted} parent-accepted results.", + "drawer.subagentMixedActivity": "Host observations and coordinator reports: {started} starts, {skipped} skips, {rejected} capacity rejections, {failed} host failures, {accepted} parent-accepted results. Some decisions are unverified.", "drawer.subagentReportedActivity": "Latest coordinator report: {started} starts, {skipped} skips, {rejected} capacity rejections, {failed} host failures, {accepted} parent-accepted results. Host verification is unavailable.", "drawer.subagentDisable": "Preview turning off sub-agent execution", "drawer.subagentDisableSummary": "New child-agent execution will be disabled for this Goal. Existing Todo ownership and execution records stay unchanged.", @@ -1574,6 +1576,8 @@ const zhCN: Record = { "drawer.subagentCurrentBoundary": "当前任务领域限制", "drawer.subagentDescription": "仅在 Todo、配额、能力和写入范围门禁全部通过后,允许运行时为相互独立的任务临时创建子代理;不会强制并行,也不会授予持久权限。", "drawer.subagentReportTitle": "子代理活动", + "drawer.subagentHostActivity": "宿主已观察:启动 {started} 次、容量拒绝 {rejected} 次、宿主失败 {failed} 次、主 Agent 验收 {accepted} 项。", + "drawer.subagentMixedActivity": "宿主观察与主 Agent 回报:启动 {started} 次、跳过 {skipped} 次、容量拒绝 {rejected} 次、宿主失败 {failed} 次、主 Agent 验收 {accepted} 项;部分决策未经宿主核验。", "drawer.subagentReportedActivity": "最近一轮主 Agent 回报:启动 {started} 次、跳过 {skipped} 次、容量拒绝 {rejected} 次、宿主失败 {failed} 次、主 Agent 验收 {accepted} 项;目前没有宿主核验。", "drawer.subagentDisable": "预览关闭子代理执行", "drawer.subagentDisableSummary": "这个 Goal 将不再创建新的子代理;现有 Todo 归属和执行记录不受影响。", diff --git a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-model.ts b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-model.ts index def13b066e..03f703fd50 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-model.ts +++ b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-model.ts @@ -159,8 +159,8 @@ export type WorkspaceGoal = { subagentExecution?: WorkspaceGoalSubagentConfiguration; nativeChildActivity?: { turn_instance_id: string; - observation: "unknown" | "coordinator_reported"; - host_attested: false; + observation: "unknown" | "coordinator_reported" | "host_observed" | "mixed"; + host_attested: boolean; launched_count: number; skipped_count: number; capacity_rejected_count: number; diff --git a/apps/presentation/dashboard/src/views/dashboard-page.tsx b/apps/presentation/dashboard/src/views/dashboard-page.tsx index f39a930c7a..2e8fd7679d 100644 --- a/apps/presentation/dashboard/src/views/dashboard-page.tsx +++ b/apps/presentation/dashboard/src/views/dashboard-page.tsx @@ -468,8 +468,8 @@ type PersonalGoalItem = { hasRunObservation: boolean; nativeChildActivity?: { turn_instance_id: string; - observation: "unknown" | "coordinator_reported"; - host_attested: false; + observation: "unknown" | "coordinator_reported" | "host_observed" | "mixed"; + host_attested: boolean; launched_count: number; skipped_count: number; capacity_rejected_count: number; diff --git a/docs/architecture/rfcs/STATUS.md b/docs/architecture/rfcs/STATUS.md index c16dd68b3a..6a3df593d2 100644 --- a/docs/architecture/rfcs/STATUS.md +++ b/docs/architecture/rfcs/STATUS.md @@ -61,7 +61,7 @@ appendix may keep dated history, but no dated log heading may precede it. | [RFC: Research Exploration Control Plane v0](research-exploration-control-plane-v0.md) | Accepted | none | — | | [RFC: Semantic Vocabulary Convergence and Commit-Time Drift Checks (v0)](semantic-vocabulary-convergence-v0.md) | Accepted | none | [5 entries](ledger/semantic-vocabulary-convergence-v0/) | | [RFC: Shared Goal Alignment and Governed Amendment Protocol (v0)](shared-goal-alignment-and-governed-amendment-v0.md) | Accepted | none | [2 entries](ledger/shared-goal-alignment-and-governed-amendment-v0/) | -| [RFC: LoopX Shared Control-Plane Authority and Pluggable State Providers (v0)](shared-goal-authority-state-provider-v0.md) | Accepted | none | [23 entries](ledger/shared-goal-authority-state-provider-v0/) | +| [RFC: LoopX Shared Control-Plane Authority and Pluggable State Providers (v0)](shared-goal-authority-state-provider-v0.md) | Accepted | none | [24 entries](ledger/shared-goal-authority-state-provider-v0/) | | [RFC: Single-Owner Local Daemon (v0)](single-owner-local-daemon-v0.md) | Accepted | none | — | | [RFC: TypeScript Control-Plane Migration Direction v0](typescript-control-plane-migration-v0.md) | Accepted | none | [14 entries](ledger/typescript-control-plane-migration-v0/) | diff --git a/docs/architecture/rfcs/STATUS.zh-CN.md b/docs/architecture/rfcs/STATUS.zh-CN.md index fe9189d5b8..33c158d11e 100644 --- a/docs/architecture/rfcs/STATUS.zh-CN.md +++ b/docs/architecture/rfcs/STATUS.zh-CN.md @@ -58,7 +58,7 @@ | [RFC:研究型探索控制面 v0](research-exploration-control-plane-v0.zh-CN.md) | 已接受 | 无 | — | | [RFC:语义词表收敛与提交期漂移检查(v0)](semantic-vocabulary-convergence-v0.zh-CN.md) | 已接受 | 无 | [5 条](ledger/semantic-vocabulary-convergence-v0/) | | [RFC:共享 Goal 对齐与受治理 Amendment 协议(v0)](shared-goal-alignment-and-governed-amendment-v0.zh-CN.md) | 已接受 | 无 | [2 条](ledger/shared-goal-alignment-and-governed-amendment-v0/) | -| [RFC:LoopX 共享控制面权威与可插拔状态 Provider(v0)](shared-goal-authority-state-provider-v0.zh-CN.md) | 已接受 | 无 | [23 条](ledger/shared-goal-authority-state-provider-v0/) | +| [RFC:LoopX 共享控制面权威与可插拔状态 Provider(v0)](shared-goal-authority-state-provider-v0.zh-CN.md) | 已接受 | 无 | [24 条](ledger/shared-goal-authority-state-provider-v0/) | | [RFC: Single-Owner Local Daemon (v0)](single-owner-local-daemon-v0.md) | 已接受 | none | — | | [RFC:LoopX 控制面 TypeScript 渐进迁移方向 v0](typescript-control-plane-migration-v0.zh-CN.md) | 已接受 | 无 | [14 条](ledger/typescript-control-plane-migration-v0/) | diff --git a/docs/architecture/rfcs/capable-manager-semantic-handoff-v0.md b/docs/architecture/rfcs/capable-manager-semantic-handoff-v0.md index 5b02283204..59c0e130ac 100644 --- a/docs/architecture/rfcs/capable-manager-semantic-handoff-v0.md +++ b/docs/architecture/rfcs/capable-manager-semantic-handoff-v0.md @@ -219,17 +219,23 @@ grant for linked Core details. The shipped managed-Goal context grant below is a bounded step; full planning/effect inheritance still needs its typed grant chain, installed receiver adoption and original-route acceptance. -Use managed Goal scope for an authenticated owner's context-delegation grant: -all current and future registered Agents within those Goals inherit it. Avoid -requiring separate enrollment every time a worker joins. Retain exact-recipient -grants for restricted sources, and explicit recipient revocations that override -the Goal grant. New Goals, evidence-read scope and execution permissions are -separate authority; registration or a quoted request cannot expand them. -The shared `collaboration/source_grants.ts` owner resolves the same current -policy for the catalog, direct handoff and peer forwarding. The existing local -operator command can configure a Goal target by omitting `--agent-id`, with -preview, locked apply and readback. This bounded configuration slice does not -qualify settings-UI editing, native receiver adoption or the full M1–M3 journey. +**Local context-delivery default.** An operator-configured source with a verified authorized +sender defaults to all active registered recipients on its selected local registry, +across Goals and later registrations. `local_delivery_scope=selected` deliberately +retains an enrollment boundary; an old enrollment list alone no longer restricts +the default. Explicit Agent and Goal exclusions survive broad restoration and +apply at direct delivery, replay and each parent-forwarding hop. Missing or +malformed source provenance cannot activate the default. Context delivery grants +no evidence-read expansion, remote delivery, execution, claim/lease or protected +operation. Shared TypeScript `collaboration/source_grants.ts` owns the decision; +Python observes registration/provenance and persists operator changes. Qualify +cross-Goal delivery, future registration, revoked replay and original-route return +through the existing App/Lark conversation, without claiming worker adoption from +catalog access. The existing local operator commands preview, apply and read back exceptions. +Restoring a Goal retains its individually revoked Agents; selected scope can +enroll a whole Goal by omitting `--agent-id`. Editing source policy in packaged +settings remains an unqualified configuration journey. Native receiver adoption +and full M1–M3 execution are separate acceptance gates. LoopX state mutations always use the existing typed command boundary, even if initiated through shell. The manager does not edit registry/authority files behind the control plane. Repository modifications use the project's normal worktree/review practice. Scoped merge/deploy authorization may be reused; unrelated payment or trading authority cannot be inferred from it. diff --git a/docs/architecture/rfcs/external-evidence-research-capability-v0.md b/docs/architecture/rfcs/external-evidence-research-capability-v0.md index c0046e5783..e1b5412c8f 100644 --- a/docs/architecture/rfcs/external-evidence-research-capability-v0.md +++ b/docs/architecture/rfcs/external-evidence-research-capability-v0.md @@ -78,11 +78,11 @@ inventory-only row for the same provider id. ## Product surfaces -- CLI: `external-evidence discover|plan|receipt|admit|retire`. -- Managed Turn: the same five effect-runtime methods. -- Frontend/Lark: not changed in this Core slice. A companion slice should render - the same typed plan/admission projection and readback; it must not invent a - second registry or lifecycle. +- CLI: `external-evidence discover|plan|execute|receipt|admit|readback|retire`. +- Managed Turn: the same five typed effect-runtime methods; explicit provider + execution and ledger projection use the capability's CLI owner. +- Frontend/Lark: existing conversation answer/report and Markdown transports + render the shared validated readback. No independent registry or lifecycle. ## Acceptance @@ -102,6 +102,29 @@ inventory-only row for the same provider id. - retirement waits for downstream coverage of every admitted source; - CLI and effect-runtime TypeScript tests pass from the source checkout. +## Delivery checkpoint (2026-10-02) + +The public GitHub method now completes a bounded real journey: anonymous pinned +file reads, exact-plan receipt validation, a separate parent decision, projection +into the existing deepresearch source ledger, actual lineage readback and retirement. +Optional source refs and literal search terms are bound into the request/plan digest; +legacy requests retain their existing identity. The provider is bundled in extensions +under `method:public-github`; capability and ledger owners remain unchanged. + +Passed: real public-provider/source CLI journey; negative cases for private or stale +readiness, malformed/unpinned sources, plan/admission mutation, partial/empty/failed +reads, independent admission and coverage, wrong-question projection, budget failure +and idempotent replay; packaged desktop/mobile conversation readback and reload; +existing Lark Markdown presentation. Source bodies are not persisted. The shared +Markdown readback uses existing answer/report and Lark transports; no frontend +configuration or parallel evidence authority is needed. + +Commands are in the [versioned capability guide](../../../loopx/capabilities/external_research/README.md#public-github-method--公开-github-方法). +Live Lark delivery, authenticated connector execution and broader semantic research +quality remain untested; this checkpoint does not promote those providers or close +S6/S8. Failed or partial results preserve original-source fallback, and neither a +successful read nor a parent admission certifies evidence completeness. + ## Non-goals - a universal browser/search engine; diff --git a/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md b/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md index 898efaee2f..d939db48c7 100644 --- a/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md +++ b/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md @@ -60,10 +60,11 @@ Connector registry 继续只拥有库存与遥测。`supported` 绝不映射为 ## 产品入口 -- CLI:`external-evidence discover|plan|receipt|admit|retire`; -- Managed Turn:复用同五个 effect-runtime 方法; -- Frontend/Lark:本 Core 切片不修改。后续 companion slice 只渲染同源 plan/admission - 投影与读回,不建立第二个 registry 或生命周期。 +- CLI:`external-evidence discover|plan|execute|receipt|admit|readback|retire`; +- Managed Turn:复用五个 typed effect-runtime 方法;显式 provider 执行与账本投影 + 使用能力的 CLI owner; +- Frontend/Lark:现有会话答复/报告和 Markdown 运输渲染同源校验回读, + 不建立独立 registry 或生命周期。 ## 验收 @@ -80,6 +81,24 @@ Connector registry 继续只拥有库存与遥测。`supported` 绝不映射为 - 全部被采纳来源完成下游覆盖前不得退休; - CLI 与 effect-runtime TypeScript 测试在源码 checkout 中通过。 +## 交付检查点(2026-10-02) + +公开 GitHub method 已完成有界真实链路:匿名读取固定提交文件、精确 plan 回执校验、 +独立父 Agent 决定、投影到现有 deepresearch 来源账本、实际 lineage 回读与退休。 +可选 source refs 和字面检索词进入 request/plan digest;旧请求身份保持兼容。 +provider 以 `method:public-github` 内置在 extensions,能力和账本 owner 不变。 + +通过:真实公开 provider/源码 CLI 链路;私有或过期 readiness、无效/未固定来源、 +plan/admission 篡改、部分/空/失败读取、独立采纳与覆盖、问题不匹配、预算耗尽及 +幂等重放等负向用例;打包桌面/移动会话回读与重载;现有 Lark Markdown 展示。 +不持久化来源正文。同源 Markdown 沿用现有答复/报告和 Lark 运输路径, +无需新增前端配置或并行证据权威。 + +命令参见[版本化能力指南](../../../loopx/capabilities/external_research/README.md#public-github-method--公开-github-方法)。 +真实 Lark 送达、带凭据 connector 执行和更广泛语义研究质量尚未验证; +该检查点不晋升这些 provider,也不关闭 S6/S8。失败或部分结果保留原始来源退路; +读取成功和父 Agent 采纳均不证明证据完整性。 + ## 非目标 - 通用浏览器或搜索引擎; diff --git a/docs/architecture/rfcs/ledger/shared-goal-authority-state-provider-v0/2026-10-03-delegation-stop-lease-fence.md b/docs/architecture/rfcs/ledger/shared-goal-authority-state-provider-v0/2026-10-03-delegation-stop-lease-fence.md new file mode 100644 index 0000000000..132a8af72e --- /dev/null +++ b/docs/architecture/rfcs/ledger/shared-goal-authority-state-provider-v0/2026-10-03-delegation-stop-lease-fence.md @@ -0,0 +1,188 @@ +# Proposed delegated operation stop: canonical lease revocation + +- Baseline: `e55489c77`, measured October 3, 2026. +- Outcome: overall roadmap S4 ("restart/cancel/drain/stop retain work and + fence old executors") and the R2 bounded single-operation stop; the + revocation half of delivery 2 in the + [September 27 host-supervision plan](2026-09-27-host-supervision.md) + ("cancellation on expiry/reclaim/revocation"). No provider, capability, + configuration surface or lease vocabulary is introduced. +- Status: proposed alternative, pending an explicit maintainer decision. + This entry neither replaces the stop contract under review in #5308 nor + claims that the stop surface has shipped. Its implementation qualification + applies only if this alternative is selected. +- [中文](2026-10-03-delegation-stop-lease-fence.zh-CN.md). + +## What was measured + +[#5308](https://github.com/loopx-project/loopx/pull/5308) is open again. Its +[October 3 review](https://github.com/loopx-project/loopx/pull/5308#pullrequestreview-5401677786) +requires canonical lease-obligation readback (R1), complete execution drain +observation in the existing Host boundary (R2), and a typed separation between +next actions and final receipts (R3). That review explicitly removes historical +failure-by-failure attribution as a merge prerequisite. Earlier failures remain +historical evidence, not proof that its current design cannot be repaired. + +The two proposals address the same caller outcome with different guarantees: + +| Boundary | #5308 under review | This proposed alternative | +| --- | --- | --- | +| Ordering | Prove the original execution drained before releasing its lease | Revoke the lease first; observe drain separately | +| Completion feedback | `settled` requires ACK, released holders, Host drain and resolved lease obligation | `revoked` proves loss of commit authority; only `drained` reports execution exit | +| Authority modes | Includes existing unleased routes with the dispatch fence | Requires canonical `hard_lease`; refuses unleased routes | + +These are alternative public contracts, not interchangeable phase names. Do +not implement both under the same `delegation stop` / `stop_delegation` entry +points. The current #5308 repair follows R1–R3. Selecting this alternative would +require an explicit supersession decision, the CLI/MCP/readback/docs companions, +and the implementation evidence below; merging a design note alone does not +change the runtime contract. Neither approach establishes whole-team or Goal +completion, and neither makes revocation proof of physical drain. + +At the measured baseline, `main` already provides reusable lease fencing and +supervision: + +- `Delegations._complete_delegated_todo` refuses to commit without the + execution's acquired lease and completes through the canonical lease CAS + ([#5466](https://github.com/loopx-project/loopx/pull/5466)). +- `runLeasedHostProcess` re-proves the original owner/key/epoch at + `min(30 s, remaining/2)` and requests cancellation when a renewal is + rejected, current proof is lost or the last proven `expires_at` passes; + the forced group termination follows a six-second + grace ([#5436](https://github.com/loopx-project/loopx/pull/5436)). Each + lease command may run for 60 seconds and a lost reply is retried once with + the same intent, while the proven expiry stays armed throughout: + `tests/control_plane/test_leased_host_process.py::test_real_renewal_faults_keep_original_deadline_and_identity` + shows on real File and SQLite authority that a hung renewal does not + disarm expiry-driven cancellation in the tested supervisor topology. +- `tests/test_delegation_lease_lifetime.py::test_real_revocation_or_new_execution_stops_nested_host_without_acceptance` + proves on real File and SQLite authority, with functioning nested + supervision, that releasing that lease stops the nested Host and its + descendants before the worker returns, leaves the Todo + open, and that retrying the operation neither reacquires the old execution + nor launches the Host again. + +A replay of the retired acquire receipt is the only way the old execution key +can reach the authority again. The acquire receipt identity is deterministic +in `(goal_id, todo_id, owner, idempotency_key)`, and replay requires current +proof, so a released lease retires its key permanently without a new status. + +## Proposed contract + +1. **Scope.** One authorized, bound delegated operation on the local host + authority. Not a team or Goal stop, not coordinator pause, not cross-host + signalling, not a frontend control beyond a recorded-state label. +2. **Intent.** `stop` is the explicit intent of the binding's requester for + one operation. It is persisted beside the operation record before any + fence write, carries the requester identity and one stable `stop_id`, and + cannot be inferred from a signal, a timeout or progress. Repeated stops + return the same receipt. +3. **Fence.** The execution's own canonical hard lease + (`owner`, `idempotency_key`, `lease_epoch`) is the only fence. The + Delegations host releases it through the existing canonical lifecycle with + a CAS on the current version, retrying only a version-mismatch race. This + is the same trust the host already exercises when it claims, renews and + completes on the member's behalf. After release the authority rejects + every renewal, completion CAS and acquire replay from that execution; late + Todo completion and result acceptance are impossible by construction. No + file lock, worker acknowledgement, lane probe or process-group record is + part of this canonical-write guarantee. It does not undo shell commands, + network requests or other external effects already launched by the Host. +4. **Receipt.** The typed TypeScript owner derives one phase from current + facts on every read; no phase is persisted. + - `requested`: intent persisted, the execution has not yet exposed a lease + to release. Read again; the worker observes the intent before it + launches a Host. + - `revoked`: the release committed, or the execution is already fenced by + another epoch or by expiry. The old execution cannot commit effects + guarded by canonical authority. This alone does not qualify overlapping + external work or a resource handoff. + - `drained`: additionally, the existing Host owner proves that the original + execution and every attributed process group have exited, or proves + that no Host was launched and no launch remains possible. A returned + leased supervisor, a `stopped` operation or elapsed grace is insufficient. + - `noop`: the operation was `accepted` or `rejected` before the fence took + effect. Its prior conclusion stands and nothing is written. + Drain is an observation, never a settlement condition. A dead worker + leaves `revoked` with `host_supervision: unobserved`; only complete Host + evidence tied to the original execution can establish drain. An unavailable + or interrupted inner supervisor leaves drain unproven even after outer return. +5. **Worker observation.** The worker checks the intent before acquiring a + lease and again before launching a Host, releases its own lease on either + checkpoint, and records `stopped` after any supervised execution returns + while the intent exists. That observation says the worker handled stop; it + does not prove complete drain. Those checkpoints avoid wasted work; the + canonical-write guarantee comes from the lease fence. +6. **Authority mode.** Stop requires the Goal's canonical `hard_lease` mode. + On `legacy` or `soft_claim` authority there is no execution lease and + therefore no fence; `stop --execute` is refused before any write with a + reason naming the mode. Promoting the Goal is the enabling step. +7. **Lifecycle.** A stopped operation refuses `resume`; continuing requires a + new operation, which acquires a new lease epoch. Stop never completes the + Todo, settles the Goal or changes an accepted result. +8. **Drain latency.** Revocation and resource exit are separate facts. The + fence holds once release commits. The existing leased supervisor requests + cancellation on rejected renewal, failed current proof or its last proven + expiry; that expiry remains armed while authority replies are in flight. + These are cancellation triggers, not an unconditional deadline for every + nested process to exit. Outer supervisor return and expiry plus six-second + grace do not prove inner drain if a nested supervisor is interrupted or its + cleanup cannot be observed. The roughly thirty-six-second healthy path is + nominal only, requiring promptly answered authority requests, no in-flight + renewal at release and functioning supervision. Slow/lost replies or failed + cleanup must remain visible as `revoked` with unproven drain. Report actual + Host evidence before `drained`; this proposal adds no hard drain deadline, + new cleanup service or second process-lifecycle owner. +9. **Surfaces.** CLI `delegation stop --execute` and MCP `stop_delegation` + share `Delegations.stop`; `read`, `wait` and the inventory expose the + receipt and the `stopped` observation. The dashboard shows a recorded + stop, not a claim that execution resources were released. + +## Decisions proposed here + +- **D1, hard-lease only.** This reduces the stop guarantee to the canonical + lease fence, at the cost of refusing existing unleased routes. #5308 instead + retains their dispatch-fence path; that tradeoff needs a maintainer decision. +- **D2, release instead of a new `revoked` lease status.** A new status would + extend a vocabulary consumed by lifecycle, proof, retirement, migration and + recovery owners; release already retires the key, as measured. +- **D3, drain reported, not required for revocation.** This makes authority + loss observable while processes may still be running. It does not satisfy + #5308's `settled` promise or qualify immediate resource handoff. + +## Qualification the implementation owes + +The implementation PR shows each of these on real processes against File and +SQLite authority, records the observed release-to-drain durations, and never +asserts the nominal thirty-six seconds: + +- **Healthy revocation, the positive control.** + `test_real_revocation_or_new_execution_stops_nested_host_without_acceptance` + keeps passing: releasing the lease stops the nested Host and its + descendants before the worker returns, and the Todo stays open. +- **Release while a renewal is in flight.** With a long TTL (for example 180 + seconds), the stop's release commits after a renewal has started and while + that renewal's authority reply is delayed. The receipt is `revoked` once + the release commits and does not report `drained` while the nested Host is + still running. No renewal, Todo completion or acceptance from the old + execution commits. Observe the existing cancellation trigger separately + from complete Host exit; retain `revoked` whenever drain cannot be proved. +- **Lost authority replies.** When the renewal command fails twice or never + answers within its timeout, the same receipt and fence properties hold, + and the last proven expiry remains armed for cancellation. A cancellation + observation is not complete nested-process drain. +- **Interrupted nested supervisor.** Pause the inner supervisor after the + actual Host starts, release the original canonical lease, and wait for the + outer call to return. If an independently observed descendant still runs, + including after expiry plus grace, the receipt must remain `revoked` with + unproven drain. Only subsequent complete, original-execution Host evidence + may report `drained`. Retain the healthy-supervisor control and ensure the + fixture cleans its own groups even when the assertion fails. + +## What this entry does not establish + +The stop surface is not implemented by this entry. Windows native drain, +PostgreSQL re-qualification, cross-host stop, Lark controls, whole-team stop +and installed-product acceptance are outside the slice. Lease records are +opaque JSON to the File, SQLite and PostgreSQL providers, so no provider +change is expected, but that is a reviewed claim of the implementation PR. diff --git a/docs/architecture/rfcs/ledger/shared-goal-authority-state-provider-v0/2026-10-03-delegation-stop-lease-fence.zh-CN.md b/docs/architecture/rfcs/ledger/shared-goal-authority-state-provider-v0/2026-10-03-delegation-stop-lease-fence.zh-CN.md new file mode 100644 index 0000000000..0a4424e120 --- /dev/null +++ b/docs/architecture/rfcs/ledger/shared-goal-authority-state-provider-v0/2026-10-03-delegation-stop-lease-fence.zh-CN.md @@ -0,0 +1,147 @@ +# 委派停止替代提案:canonical lease 撤销 + +- 基线:`e55489c77`,2026 年 10 月 3 日测量。 +- 结果:总体 roadmap S4("restart/cancel/drain/stop 保留工作并 fence 旧执行者") + 与 R2 的有界单操作停止;对应 + [9 月 27 日 host-supervision 计划](2026-09-27-host-supervision.zh-CN.md) + 中交付 2 的撤销部分("到期/回收/撤销时取消")。不引入 provider、 + capability、配置面或 lease 词表。 +- 状态:替代提案,待维护者明确决定。本条目不替换 #5308 正在评审的停止契约, + 也不声称停止能力已交付。只有选定本替代方案后,下述实现验收才适用。 +- [English](2026-10-03-delegation-stop-lease-fence.md)。 + +## 测量到什么 + +[#5308](https://github.com/loopx-project/loopx/pull/5308) 已重新打开。 +[10 月 3 日最新评审](https://github.com/loopx-project/loopx/pull/5308#pullrequestreview-5401677786) +要求从 canonical authority 读回租约义务(R1)、由既有 Host 边界提供完整执行的 +退出观察(R2),以及在类型化 owner 中区分下一步动作和最终回执(R3)。该评审 +已明确取消逐次追查历史失败原因这一合入前置条件。旧失败仍是历史证据,不能据此 +断言当前设计无法修复。 + +两份方案服务于同一调用者结果,但保证不同: + +| 边界 | #5308 正在评审的实现 | 本替代提案 | +| --- | --- | --- | +| 顺序 | 先证明原执行退出,再释放其租约 | 先撤销租约,单独观察进程退出 | +| 完成反馈 | `settled` 要求 ACK、holder 释放、Host 退出和租约义务已解析 | `revoked` 证明提交权限失效;只有 `drained` 报告执行退出 | +| authority 模式 | 通过 dispatch fence 保留既有无租约路径 | 要求 canonical `hard_lease`,拒绝无租约路径 | + +这是两份替代的公共契约,不是可以互换的 phase 名称,不能同时实现在同一个 +`delegation stop` / `stop_delegation` 入口下。当前 #5308 的修复遵循 R1–R3。 +选择本替代方案需要明确的替代决定、CLI/MCP/读回/文档的配套修改,以及下述实现 +验收;仅合入设计文档不会改变运行时契约。两者均不代表团队或 Goal 已完成, +也不能用撤销权限证明物理进程已退出。 + +在测量基线上,`main` 已提供可复用的租约 fence 和 supervision: + +- `Delegations._complete_delegated_todo` 没有本次执行已取得的 lease 就拒绝提交, + 并通过 canonical lease CAS 完成 + ([#5466](https://github.com/loopx-project/loopx/pull/5466))。 +- `runLeasedHostProcess` 以 `min(30 s, remaining/2)` 的节奏重新证明原 + owner/key/epoch,在续期被拒绝、当前证明丢失或最后已证明的 `expires_at` 到达时 + 请求取消;强制进程组终止前有六秒 grace + ([#5436](https://github.com/loopx-project/loopx/pull/5436))。每个 lease 命令 + 最长运行 60 秒,回复丢失时以同一意图重试一次,而已证明的到期计时始终有效: + `tests/control_plane/test_leased_host_process.py::test_real_renewal_faults_keep_original_deadline_and_identity` + 在真实 File 与 SQLite authority 上证明,在该测试的监督结构内,续期挂起不会 + 撤销由到期时刻触发的取消。 +- `tests/test_delegation_lease_lifetime.py::test_real_revocation_or_new_execution_stops_nested_host_without_acceptance` + 在真实 File 与 SQLite authority 上证明:嵌套监督正常运行时,释放该 lease 后, + 嵌套 Host 及其子进程 + 在 worker 返回前停止,Todo 保持未完成,重试该操作既不会重新取得旧执行也不会 + 再次启动 Host。 + +旧执行 key 再次到达 authority 的唯一途径是重放已退役的 acquire 回执。acquire +回执身份由 `(goal_id, todo_id, owner, idempotency_key)` 确定性生成,而重放要求 +当前证明,因此 lease 一旦 released,其 key 就永久退役,不需要新状态。 + +## 提议的契约 + +1. **范围。** 本机 authority 上一个经授权、有 binding 的委派操作。不是团队或 + Goal 停止,不是协调者暂停,不是跨宿主信号,前端只有一个记录状态标签。 +2. **意图。** `stop` 是 binding 的 requester 对一个操作的明确意图。它在任何 + fence 写入之前持久化在操作记录旁,携带 requester 身份和一个稳定的 + `stop_id`,不能由信号、超时或进度推导。重复 stop 返回同一回执。 +3. **Fence。** 该执行自己的 canonical hard lease(`owner`、`idempotency_key`、 + `lease_epoch`)是唯一的 fence。Delegations host 通过既有 canonical lifecycle + 以当前 version 的 CAS 释放它,只重试 version 不匹配这一种竞争。这与 host + 代表成员 claim、renew、complete 时行使的是同一份信任。释放之后,authority + 拒绝该执行的一切续期、完成 CAS 与 acquire 重放;迟到的 Todo 完成与结果 + 验收在构造上不可能。文件锁、worker ACK、lane 探测、进程组记录都不是保证的 + 一部分。该保证只约束经过 canonical authority 校验的写入,不能撤回 Host 已经 + 发出的 shell 命令、网络请求或其他外部副作用。 +4. **回执。** 类型化的 TypeScript owner 在每次读取时由当前事实推导一个 + phase;不持久化 phase。 + - `requested`:意图已持久化,执行尚未暴露可释放的 lease。再次读取;worker + 会在启动 Host 前观察到该意图。 + - `revoked`:释放已提交,或该执行已被另一个 epoch 或到期 fence。旧执行不能 + 提交受 canonical authority 校验的效果;这本身不证明可以重叠执行外部工作 + 或交接资源。 + - `drained`:在此之上,既有 Host owner 证明原执行及全部归属进程组已经退出, + 或证明 Host 从未启动且已不存在继续启动的可能。leased supervisor 返回、 + 操作记录为 `stopped` 或 grace 已经过期,都不足以证明这一点。 + - `noop`:操作在 fence 生效前已经 `accepted` 或 `rejected`。原结论保留, + 不写任何内容。 + Drain 是观察,绝不是结算条件。worker 已死时停留在 `revoked` 且 + `host_supervision: unobserved`;只有绑定原执行的完整 Host 证据才能建立 drain。 + 内层 supervisor 不可用或被中断时,即使外层已经返回,drain 仍未获证明。 +5. **Worker 观察。** worker 在取得 lease 前和启动 Host 前各检查一次意图,任一 + 检查点命中时释放自己的 lease;在意图存在时任何被监督执行返回后记录 + `stopped`。这只说明 worker 处理过停止,不能证明完整 drain。检查点用于避免 + 浪费工作;canonical 写入保证来自 lease fence。 +6. **Authority 模式。** stop 要求 Goal 的 canonical `hard_lease` 模式。在 + `legacy` 或 `soft_claim` authority 上没有执行 lease,因此没有 fence; + `stop --execute` 在任何写入前被拒绝,原因中写明模式。提升 Goal 是启用步骤。 +7. **生命周期。** 已停止的操作拒绝 `resume`;继续需要新操作,它会取得新的 + lease epoch。stop 永不完成 Todo、不结算 Goal、不改变已验收结果。 +8. **Drain 延迟。** 撤销与资源退出是两个事实。release 提交后 fence 生效;既有 + leased supervisor 在续期被拒绝、当前证明失败或最后已证明的到期时刻请求取消, + authority 回复在途时也保留这个到期计时器。这些是取消触发条件,不是所有嵌套 + 进程退出的无条件期限。内层 supervisor 被中断或无法观察其清理时,外层返回和 + 到期加六秒 grace 都不能证明内层 drain。约三十六秒的健康路径只是名义值,要求 + authority 及时响应、release 时没有在途续期且监督正常运行。慢响应、回复丢失或 + 清理失败时,保留 `revoked` 与 drain 未证明的事实。只有真实完整的 Host 证据才 + 能得到 `drained`;本提案不新增 drain 硬期限、清理服务或第二个进程生命周期 owner。 +9. **入口。** CLI `delegation stop --execute` 与 MCP `stop_delegation` 共用 + `Delegations.stop`;`read`、`wait` 与 inventory 暴露回执与 `stopped` 观察。 + dashboard 展示"停止已登记",不声称执行资源已释放。 + +## 本条目提议的决定 + +- **D1,仅限 hard lease。** 把停止保证限定在 canonical lease fence,代价是 + 拒绝既有无租约路径。#5308 保留这些路径的 dispatch fence,取舍需由维护者决定。 +- **D2,用 release 而非新增 `revoked` lease 状态。** 新状态会扩展被 + lifecycle、proof、retirement、migration 与 recovery 多个 owner 消费的词表; + 如测量所示,release 已经使 key 退役。 +- **D3,drain 单独报告,不作为撤销的条件。** 允许在进程仍运行时报告提交权限 + 已失效;这不满足 #5308 的 `settled` 承诺,也不证明可以立即交接执行资源。 + +## 实现 PR 必须给出的验收 + +实现 PR 需在 File 与 SQLite authority 上以真实进程逐项证明以下各点,记录观测到的 +release 到 drain 耗时,且从不断言名义上的三十六秒: + +- **健康撤销,作为正向对照。** + `test_real_revocation_or_new_execution_stops_nested_host_without_acceptance` + 继续通过:释放 lease 后,嵌套 Host 及其子进程在 worker 返回前停止,Todo 保持 + 未完成。 +- **续期进行中时 release。** 使用长 TTL(例如 180 秒),在一次续期已开始、且其 + authority 回复被延迟时提交 stop 的 release。release 提交后回执即为 `revoked`, + 嵌套 Host 仍在运行时不报告 `drained`。旧执行的续期、Todo 完成与验收都不能提交; + 分别观察既有取消触发和完整 Host 退出;不能证明 drain 时保留 `revoked`。 +- **authority 回复丢失。** 续期命令两次失败或在超时内始终无回复时,回执与 fence + 的性质不变,最后已证明的到期计时器仍负责触发取消。取消观察不等于完整嵌套 + 进程 drain。 +- **内层 supervisor 中断。** 实际 Host 启动后暂停内层 supervisor,释放原 canonical + lease,等待外层调用返回。若独立观察到后代仍在运行,包括到期加 grace 之后, + 回执必须保持 `revoked` 且 drain 未证明。只有后续绑定原执行的完整 Host 证据才能 + 报告 `drained`。保留 supervisor 正常运行的正控,并保证断言失败时 fixture 仍清理 + 自己的进程组。 + +## 本条目不建立什么 + +停止能力不由本条目实现。Windows 原生 drain、PostgreSQL 重新资格化、跨宿主 +停止、Lark 控件、整团队停止与安装态验收都在切片之外。lease 记录对 File、 +SQLite、PostgreSQL provider 是不透明 JSON,预计不需要 provider 改动,但这是 +实现 PR 需要评审的声明。 diff --git a/docs/architecture/rfcs/loopx-overall-roadmap-v0.md b/docs/architecture/rfcs/loopx-overall-roadmap-v0.md index 08022d753f..732c26ad7a 100644 --- a/docs/architecture/rfcs/loopx-overall-roadmap-v0.md +++ b/docs/architecture/rfcs/loopx-overall-roadmap-v0.md @@ -429,6 +429,10 @@ Use scoped discovery and permitted recovery before requesting manual IDs. Private-owner discovery is broad by default across registered local resources and authorized connected sources; a delivery allowlist, missing live binding or bounded first page must not hide an otherwise visible responsible Agent. +Sender-bound local context delivery now defaults to active registered recipients +across Goals, with current/future registration and explicit revocations resolved +by the shared TS source owner. Selected-audience enrollment remains explicit; +this is neither a remote grant nor proof of worker adoption. Keep discovery, audience evidence access, delegation and execution readiness separate, as specified by [manager §5.5](capable-manager-semantic-handoff-v0.md#55-responsibility-discovery-and-receiver-owned-planning). diff --git a/docs/integrations/host-native-child-receipts.md b/docs/integrations/host-native-child-receipts.md index b30eeaf878..597555b7a5 100644 --- a/docs/integrations/host-native-child-receipts.md +++ b/docs/integrations/host-native-child-receipts.md @@ -16,20 +16,32 @@ Goal 事件流,以 Turn ID、稳定操作 ID 和不限定宿主的 `entrypoint 记录动作不会启动子代理、调度新 Turn、授予写入权限或消耗配额。 `max_children` 只是配置上限,不代表当前可用槽位,也不是必须启动的数量。 -`native-child record` currently accepts the coordinator's typed report. Its -`observation` is `coordinator_reported` and `host_attested` is always `false`. -LoopX cannot intercept an arbitrary external host's native tool call. A future -host adapter must observe that call at its own boundary and extend this event -and read-model contract with a separately verified provenance variant; this -v0 recorder cannot claim host attestation. Missing records remain -`unknown`; neither a missing record nor `max_children > 0` proves that a child -was created or deliberately skipped. - -目前 `native-child record` 接受主 Agent 的类型化上报,因此 `observation` 为 -`coordinator_reported`,`host_attested` 始终为 `false`。LoopX 无法拦截任意 -外部宿主的原生工具调用。后续宿主适配器须在自己的边界观察调用,给事件与 -读模型扩展单独核验的来源类型;当前 v0 上报器不能声称宿主核验。缺少回执 -就是 `unknown`;没有回执或配置上限大于零,都不能证明已启动或主动跳过。 +`native-child record` accepts a coordinator report and cannot select host +provenance. Managed Codex CLI and operation-equipped app-server Turns also +observe native collaboration items directly on their owned connection. A +successful spawn, failed call and observed child completion become durable +`host_observed` records before Turn settlement. A parent review is a separate +explicit record; a host completion does not adopt the evidence. + +The shared projection distinguishes `host_observed`, `coordinator_reported`, +`mixed` and `unknown`. `host_attested` is true only when every decision in the +Turn came from the host. Configured capacity remains an upper bound. Failed +Codex collaboration items do not carry a typed capacity error, so the adapter +records `host_failed` and forbids report-based same-Turn retry rather than +classifying provider prose. Missing native events stay unknown. Persisted Codex +history is not used to reconstruct native activity because supported host +versions may omit collaboration items from that history. + +`native-child record` 仍是主 Agent 上报入口,不能指定宿主来源。托管 Codex CLI +与启用操作工具的 app-server Turn 会在自身连接上直接观察原生协作事件, +把实际启动、宿主失败和观察到的结果写入同一事件流。结果完成之后仍须由主 +Agent 明确记录验收;宿主完成不代表证据已被采纳。 + +共享投影区分宿主观察、主 Agent 上报、混合来源和未知;仅全部决策均来自 +宿主时 `host_attested` 为真。Codex 的失败协作项没有容量错误码,适配器保留 +通用 `host_failed` 和禁止同 Turn 上报重试的规则,不从错误文字推断容量。 +缺少原生事件仍是未知。部分宿主版本的持久化历史会遗漏协作项,因此不用于 +重建活动。第三方宿主上报仍不会自动获得宿主核验标记。 ## Lifecycle / 生命周期 @@ -106,3 +118,65 @@ for the parent validation of the underlying work. 已绑定的 LoopX delegation 仍以自身操作回执为权威。原生子代理上报不能代替 delegation 回执,也不能代替主 Agent 对工作结果的实际核验。 + +## Codex host qualification / Codex 宿主验证 + +The adapter belongs to the built-in Codex Turn host and the existing +`multi_subagent` receipt owner. It installs no scheduler and makes no additional +provider request. Feature-off Turns create no observer and retain their host +request/result contract. Ordinary CLI and Lark status use the same shared +projection as the dashboard; Lark has no native child configuration to change. + +A resumed CLI invocation can receive only a new `wait` completion. The adapter +resolves its hashed child reference against the latest started host-observed +spawn or followup in the +same admitted Goal instance, coordinator and LoopX Turn, including receipts +outside the bounded status window. A spawn/followup item's terminal snapshot +belongs to its own stable decision and receivers; replaying an old spawn never +completes a later followup. A terminal wait first resolves the immutable binding +of its session, invocation, native item and child identity. Only its first +observation uses the latest child association. Each distinct wait retains a +hashed scalar reference on the existing result event; exact replay preserves +that original operation across restart, and changed outcomes or reassignment +are rejected. Multiple waits observing one result do not add operations, +launches, parent acceptance or quota. No raw host identifiers or content enter +the public activity projection. +It does not adopt coordinator reports or scan external host history. Spawn IDs +retain their existing child binding. +CLI followup and failed-call IDs use the Turn journal's durable `host_attempt` +plus the owned parent session and native item ID; app-server calls use their +native Turn ID. A real retry advances the journal attempt before launch; +replaying the same binding and item remains idempotent. Direct enabled CLI +adapter calls require that attempt; feature-off calls retain their original +request. Only compact hashed child references are retained for correlation; +they do not enter public activity rows or replace independent parent review. +No prompt or raw result is added to a receipt. + +恢复 CLI 时可能只收到新的 `wait` 完成事件。适配器从同一已准入 Goal 实例、 +主 Agent、LoopX Turn 的最近一次已启动宿主 spawn 或 followup 回执恢复关联, +覆盖状态窗口外的操作。spawn/followup 的完成快照只归属自身稳定决策和接收者; +重放旧启动事件不会完成后来的跟进任务。wait 首先恢复其会话、调用、原生项和 +接收子 Agent 身份对应的首次结果关联;只有首次观察才使用最近操作。哈希关联 +作为标量留在现有结果事件中,重启重放仍归原操作,结果冲突或重新归属会被拒绝。 +多个 wait 观察同一结果不会增加操作、启动、主 Agent 验收或配额;公开活动投影 +不暴露原始宿主标识或内容。 +不采纳主 Agent 上报或扫描外部宿主历史。启动保留既有子代理绑定;CLI 跟进和 +失败调用复用 Turn 日志持久化的 `host_attempt`、父会话和原生工具 ID, +app-server 使用其原生 Turn ID。真实重试在启动前递增尝试次数;同一绑定与 +事件的重放仍幂等。直接调用已启用的 CLI 适配器须提供该尝试次数,关闭能力 +时请求不变。关联只保留紧凑的子代理哈希引用,不进入公开活动行, +也不替代独立父任务验收。回执不增加原始提示或结果内容。 + +A live isolated Codex 0.142.5 test observed one successful spawn, a second failed +spawn at `agents.max_threads=1`, child completion and independent parent +acceptance. Durable readback preserved one launch and one accepted result. The +native failure subtype remains unqualified: Codex emitted only `failed`, not +`agent_thread_limit_reached`. Synthetic typed capacity-report tests cover the +existing no-same-Turn retry rule without pretending that this host supplies that +error code. No live production Goal or raw child output is part of this evidence. + +适配器复用内置 Codex Turn 宿主与 `multi_subagent` 回执所有者,不新增调度器 +或模型调用。关闭能力时不创建观察器。CLI、Lark 状态和仪表板共用同一投影。 +隔离的真实 Codex 0.142.5 验证观察到了一个成功启动、上限为 1 时第二次启动 +失败、首个子任务完成以及独立的父任务验收。持久回读保留一个启动和一个 +采纳结果。失败子类型仍有宿主协议缺口,不能声称已核验容量错误码。 diff --git a/docs/product/use-cases/steward/golden-queries.md b/docs/product/use-cases/steward/golden-queries.md index 632438ae75..ba66db022c 100644 --- a/docs/product/use-cases/steward/golden-queries.md +++ b/docs/product/use-cases/steward/golden-queries.md @@ -76,6 +76,16 @@ an ordinary public issue and a qualified receiver already doing unrelated review Delivery and read must not pass the test. Require an independent assessment, reuse an explicitly linked existing task when needed, inspect current facts, post the authorized checked reply and return its link to the original request. +A configured, sender-bound source delivers to all registered active local +Agents by default, across Goals and later registrations, without recipient +enrollment. Exercise direct delivery and a receiver's subsequent peer handoff +through the same current source rule. Revoke one Agent and one Goal: the next +attempt must be denied, and restoring the Goal must preserve the Agent exception. +Exercise both revocation orders, including an Agent revoked while its Goal is disabled. +An explicit selected scope retains enrollment; an unknown sender, stopped Goal +or remote binding never gains authority from the local default. Delivery remains +separate from evidence access, task acceptance and execution. + A short answer requires no invented Todo. A deferral names its actual condition and continuation; a monitor responsibility or worker activity is not a task result. diff --git a/examples/personal-workspace-browser-smoke.mjs b/examples/personal-workspace-browser-smoke.mjs index 3f6291729b..1f6df697af 100644 --- a/examples/personal-workspace-browser-smoke.mjs +++ b/examples/personal-workspace-browser-smoke.mjs @@ -1,5 +1,7 @@ #!/usr/bin/env node +import {nativeChildActivityScenario} from "./personal-workspace-browser/native-child-activity.mjs"; import {conversationImageRequestScenario} from "./personal-workspace-browser/conversation-image-request.mjs"; +import {externalEvidenceReadbackScenario} from "./personal-workspace-browser/external-evidence-readback.mjs"; // Isolated browser acceptance scenarios for the personal Agent workspace. import { mkdir, writeFile } from "node:fs/promises"; @@ -72,6 +74,8 @@ scenarioCatalog.push(turnStepsScenario); scenarioCatalog.push(goalWorkMapScenario); scenarioCatalog.push(performanceDiagnosisScenario); scenarioCatalog.push(blockedNoticeSettingsScenario); +scenarioCatalog.push(nativeChildActivityScenario); +scenarioCatalog.push(externalEvidenceReadbackScenario); const requestedScenario = process.env.LOOPX_PERSONAL_WORKSPACE_SCENARIO; const scenarios = requestedScenario ? scenarioCatalog.filter((scenario) => scenario.id === requestedScenario) diff --git a/examples/personal-workspace-browser/external-evidence-readback.mjs b/examples/personal-workspace-browser/external-evidence-readback.mjs new file mode 100644 index 0000000000..ed2ec67cdf --- /dev/null +++ b/examples/personal-workspace-browser/external-evidence-readback.mjs @@ -0,0 +1,57 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { resolve } from "node:path"; +import { resolveTestPython } from "../../scripts/test-python.mjs"; +import { outputDir, repoRoot } from "./fixture.mjs"; +import { openWorkspacePage } from "./scenario-context.mjs"; + +export const externalEvidenceReadbackScenario = { + id: "external-evidence-readback", + async run({browser, collectCoverage, url}) { + // Exercise the product's typed plan/admission and actual downstream ledger, + // then hand the resulting shared readback to the existing answer surface. + const result = spawnSync(resolveTestPython(), ["-X", "utf8", "-c", ` +import tempfile +from pathlib import Path +from loopx.control_plane.effect_runtime import effect_runtime_result as effect +from loopx.capabilities.deep_research.runtime import start_research +from loopx.capabilities.external_research.projection import readback, render_readback +provider = {"provider_id":"method:public-github", "provider_kind":"method", "protocol":"external_evidence_research_v0", "declared":True, "installed":True, "enabled":True, "ready":True, "unavailable_reason":None} +ref = "https://github.com/example/public/blob/" + "a"*40 + "/README.md" +plan = effect("external_evidence.plan", {"request":{"objective":"Inspect public fixture", "user_activity":"Choose a source", "decision":"Whether to use the fixture", "evidence_kinds":["literal_match"], "source_refs":[ref]}, "providers":[provider]}) +receipt = {"schema_version":"loopx_external_evidence_receipt_v0", "plan_id":plan["plan_id"], "request_id":plan["request"]["request_id"], "provider_id":provider["provider_id"], "provider_kind":"method", "status":"succeeded", "summary":"Read pinned public fixture", "completed_at":"2026-10-02T00:00:00Z", "sources":[{"source_ref":ref, "source_family":"github_repository_file", "basis":"observed", "finding":"Literal fixture marker observed at line 1", "content_digest":"sha256:"+"b"*64, "accessed_at":"2026-10-02T00:00:00Z", "limitation":"Retrieval only; completeness unverified"}]} +admission = effect("external_evidence.admit", {"plan":plan, "receipt":receipt, "decision":{"disposition":"admit", "reason":"Synthetic parent checked the direct source", "admitted_source_refs":[ref]}}) +with tempfile.TemporaryDirectory(prefix="lxe-ui-") as folder: + project=Path(folder) + start_research(project, question=plan["request"]["objective"], max_sources=8, max_subquestions=4) + print(render_readback(readback(plan, receipt, admission, project=project, execute=True))) +`], {cwd:repoRoot, encoding:"utf8", env:{...process.env, PYTHONPATH:repoRoot}, timeout:45000}); + assert.equal(result.status, 0, result.stderr); + const markdown = result.stdout; + const context = await openWorkspacePage(browser, url, {collectCoverage}); + const {api, page} = context; + try { + api.answerForMessage = () => markdown; + await page.getByRole("navigation", {name:"管家视图"}).getByRole("button", {name:/^(Chat|对话)$/}).click(); + await page.getByLabel("向 LoopX 发送消息").fill("Show the external evidence readback"); + await page.getByRole("button", {name:"发送",exact:true}).click(); + const answer = page.locator(".personal-channel-timeline .personal-message.is-assistant", {hasText:"External evidence"}); + await answer.waitFor(); + assert.match(await answer.innerText(), /retire_ready/u); + assert.match(await answer.innerText(), /Downstream coverage.*1 admitted/u); + assert.match(await answer.innerText(), /Evidence completeness is unverified/u); + await answer.scrollIntoViewIfNeeded(); + await page.screenshot({path:resolve(outputDir,"external-evidence-desktop.png"),fullPage:false,animations:"disabled"}); + await page.setViewportSize({width:390,height:844}); + await answer.scrollIntoViewIfNeeded(); + assert.equal(await answer.evaluate(el => el.scrollWidth <= el.clientWidth + 1), true); + await page.screenshot({path:resolve(outputDir,"external-evidence-mobile.png"),fullPage:false,animations:"disabled"}); + await page.reload({waitUntil:"networkidle"}); + await page.getByRole("navigation", {name:"管家视图"}).getByRole("button", {name:/^(Chat|对话)$/}).click(); + await answer.waitFor(); + assert.match(await answer.innerText(), /retire_ready/u); + assert.equal(context.errors.length, 0); + return {coverageEntries:context.coverageEntries, note:"Shared typed evidence readback survives conversation reload and fits desktop/mobile."}; + } finally { await context.close(); } + }, +}; diff --git a/examples/personal-workspace-browser/fixture.mjs b/examples/personal-workspace-browser/fixture.mjs index 829db0433f..7326fa376d 100644 --- a/examples/personal-workspace-browser/fixture.mjs +++ b/examples/personal-workspace-browser/fixture.mjs @@ -558,6 +558,10 @@ export async function installApi(page, { goalSubagentConfigurationEnabled = true }); } const first = fixture.attention_queue?.items?.[0]; + if (first && state.nativeChildActivity) { + first.project_asset ??= {owner: "codex", gate: "ready", next_action: first.recommended_action ?? "Review the fixture", stop_condition: "Fixture accepted"}; + first.project_asset.native_child_activity = state.nativeChildActivity; + } if (first) { first.waiting_on = "user_or_controller"; const gateDecided = state.decidedGateTodoIds.has("todo-browser-user-gate"); diff --git a/examples/personal-workspace-browser/native-child-activity.mjs b/examples/personal-workspace-browser/native-child-activity.mjs new file mode 100644 index 0000000000..3c13d0ac8e --- /dev/null +++ b/examples/personal-workspace-browser/native-child-activity.mjs @@ -0,0 +1,55 @@ +import assert from "node:assert/strict"; +import {resolve} from "node:path"; +import {outputDir} from "./fixture.mjs"; +import {openWorkspacePage} from "./scenario-context.mjs"; + +export const nativeChildActivityScenario = { + id: "native-child-activity", + async run({browser, collectCoverage, url}) { + const coverageEntries = []; + for (const observation of ["host_observed", "coordinator_reported", "mixed", "unknown"]) { + const context = await openWorkspacePage(browser, url, {collectCoverage, + beforeGoto(api, page) { + api.goalSubagentConfigurationEnabled = true; + page.__loopxRuntime.goalSubagentConfigurations.set("loopx-meta", + {mode: "multi_subagent", spawn_allowed: true, max_children: 3, allowed_domains: []}); + api.nativeChildActivity = {schema_version: "native_subagent_activity_v0", + observation, host_attested: observation === "host_observed", configured_limit: 3, + launched_count: 1, skipped_count: 0, capacity_rejected_count: 0, + host_failed_count: 1, parent_accepted_count: 1, turn_instance_id: "turn-browser-native"}; + }, + }); + try { + const {page} = context; + await page.locator(".personal-goal-link").filter({hasText: "LoopX meta"}).click(); + await page.getByRole("navigation", {name: "Goal 视图"}).getByRole("button", {name: "概览", exact: true}).click(); + await page.getByRole("button", {name: "Goal 信息", exact: true}).click(); + const drawer = page.locator('.personal-context-drawer[data-context-kind="goal"]'); + await drawer.waitFor(); + const activity = drawer.locator(".personal-native-child-activity"); + if (observation === "unknown") { + assert.equal(await activity.count(), 0); + } else { + await activity.waitFor(); + const text = await activity.innerText(); + assert.match(text, /启动 1 次/); + assert.match(text, /主 Agent 验收 1 项/); + assert.match(text, observation === "host_observed" ? /宿主已观察/ + : observation === "mixed" ? /部分决策未经宿主核验/ : /目前没有宿主核验/); + if (observation === "host_observed") { + await activity.scrollIntoViewIfNeeded(); + await page.screenshot({path: resolve(outputDir, "native-child-desktop.png"), animations: "disabled"}); + await page.setViewportSize({width: 390, height: 844}); + await activity.scrollIntoViewIfNeeded(); + assert(await activity.evaluate(el => el.scrollWidth <= el.clientWidth)); + await page.screenshot({path: resolve(outputDir, "native-child-mobile.png"), animations: "disabled"}); + } + } + assert.deepEqual(context.errors, []); + } finally { + coverageEntries.push(...await context.close()); + } + } + return {coverageEntries, note: "Host, coordinator, mixed and unknown native child activity preserve provenance in the packaged Goal drawer."}; + }, +}; diff --git a/examples/public-github-evidence-live-smoke.py b/examples/public-github-evidence-live-smoke.py new file mode 100644 index 0000000000..f3587892e5 --- /dev/null +++ b/examples/public-github-evidence-live-smoke.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Opt-in public GitHub exact-plan journey through the shipped source CLI. + +Only anonymous public GETs and a disposable synthetic research ledger are used. +The explicit synthetic parent decision is separate from provider execution. +""" +from __future__ import annotations + +import argparse +import json +import subprocess +import sys +import tempfile +from pathlib import Path + + +def qualify(source: str, root: Path) -> dict: + def run(*args, json_output=True): + result = subprocess.run([sys.executable, "-X", "utf8", "-m", "loopx.entrypoint", + "--runtime-root", str(root / "runtime"), "--registry", str(root / "registry.json"), + *args, "--format", "json" if json_output else "markdown"], capture_output=True, + text=True, encoding="utf-8", timeout=90, check=True) + return json.loads(result.stdout) if json_output else result.stdout + + def save(name, value): + path = root / name + path.write_text(json.dumps(value), encoding="utf-8") + return str(path) + + objective = "Inspect the public pinned README for the LoopX literal" + plan = run("external-evidence", "plan", "--objective", objective, + "--user-activity", "Choose a public source", "--decision", "Whether the source contains LoopX", + "--evidence-kind", "literal_match", "--public-github", "--source", source, "--search-term", "LoopX") + assert plan["status"] == "ready" + plan_path = save("plan.json", plan) + execution = run("external-evidence", "execute", "--plan-json", plan_path, "--execute") + receipt = execution["receipt"] + assert receipt["status"] == "succeeded" and len(receipt["sources"]) == 1 + assert "observed at lines" in receipt["sources"][0]["finding"] + receipt_path = save("receipt.json", execution) + before = run("external-evidence", "readback", "--plan-json", plan_path, "--receipt-json", receipt_path) + assert before["parent_admission"] is None and before["retirement"] is None + admitted = run("external-evidence", "admit", "--plan-json", plan_path, "--receipt-json", receipt_path, + "--decision", "admit", "--reason", "Synthetic parent checked the pinned file and literal-match finding", + "--admit-source", source) + admission_path = save("admission.json", admitted) + project = root / "research" + run("deepresearch", "start", "--project", str(project), "--question", objective) + args = ("external-evidence", "readback", "--plan-json", plan_path, "--receipt-json", receipt_path, + "--admission-json", admission_path, "--project", str(project)) + assert run(*args)["retirement"]["status"] == "retained" + result = run(*args, "--execute") + assert result["downstream_source_refs"] == [source] + assert result["retirement"]["status"] == "retire_ready" + assert run(*args, "--execute") == result + markdown = run(*args, json_output=False) + assert source in markdown and "retire_ready" in markdown and "Evidence completeness is unverified" in markdown + return {"ok": True, "provider": "method:public-github", "sources_observed": 1, + "explicit_parent_admission": True, "actual_ledger_readback": True, + "retained_before_projection": True, "idempotent_projection": True, + "raw_content_persisted": False, "source_ref": source} + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--execute-public-provider", action="store_true") + parser.add_argument("--source", required=True) + args = parser.parse_args() + if not args.execute_public_provider: + parser.error("--execute-public-provider is required for anonymous public GETs") + with tempfile.TemporaryDirectory(prefix="lxe-") as folder: + print(json.dumps(qualify(args.source, Path(folder)), sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/loopx/capabilities/deep_research/runtime.py b/loopx/capabilities/deep_research/runtime.py index bbabbaed81..0abf998ea2 100644 --- a/loopx/capabilities/deep_research/runtime.py +++ b/loopx/capabilities/deep_research/runtime.py @@ -13,6 +13,7 @@ from pathlib import Path from typing import Any +from ...control_plane.content_digest import ENVELOPED_SHA256_PATTERN from ...file_lock import exclusive_file_lock COMMAND = "/loopx-deepresearch" @@ -238,6 +239,8 @@ def add_source( tool: str, title: str | None, claims: list[dict[str, Any]], + external_evidence: dict[str, str] | None = None, + expected_question: str | None = None, ) -> dict[str, Any]: # Load, validate the whole batch, allocate ids, mutate, and save all under # one project-level lock: concurrent deepresearch commands are a normal @@ -245,13 +248,31 @@ def add_source( # when the atomic rename itself succeeds. with exclusive_file_lock(state_path(project), operation="deepresearch_add_source"): state = _require_active_state(project) + if expected_question is not None and state["question"] != expected_question: + raise ValueError("external evidence objective does not match the active research question") + if external_evidence is not None: + fields = {"plan_id", "admission_id", "receipt_digest", "content_digest"} + if set(external_evidence) != fields or any( + not isinstance(value, str) or not ENVELOPED_SHA256_PATTERN.fullmatch(value) + for value in external_evidence.values() + ): + raise ValueError("external evidence lineage requires exact content-addressed identities") url_or_path = url_or_path.strip() tool = tool.strip() or "unspecified" if not url_or_path: raise ValueError("--url-or-path must be non-empty") normalized = _normalize_source_ref(url_or_path) for source in state["sources"]: - if _normalize_source_ref(str(source["url_or_path"])) == normalized: + same_source = ( + str(source["url_or_path"]).strip().rstrip("/") == url_or_path.rstrip("/") + if external_evidence is not None else + _normalize_source_ref(str(source["url_or_path"])) == normalized + ) + if same_source: + if (external_evidence is not None and source.get("external_evidence") == external_evidence + and [claim["text"] for claim in state["claims"] if claim["id"] in source["claims"]] + == [str(claim.get("text", "")).strip() for claim in claims]): + return {"source_id": source["id"], "claim_ids": source["claims"], "state": state} raise ValueError( f"source already recorded as {source['id']} " f"({source['url_or_path']}); reuse its claims instead of re-reading" @@ -325,6 +346,7 @@ def add_source( "title": (title or "").strip() or None, "accessed_at": _now_iso(), "claims": claim_ids, + **({"external_evidence": dict(external_evidence)} if external_evidence is not None else {}), } ) _save_state(project, state) diff --git a/loopx/capabilities/external_research/README.md b/loopx/capabilities/external_research/README.md index ef3874156e..3535d58a41 100644 --- a/loopx/capabilities/external_research/README.md +++ b/loopx/capabilities/external_research/README.md @@ -106,10 +106,73 @@ credentials, and private notes remain provider-private. `external_evidence.discover`, `external_evidence.plan`, `external_evidence.receipt`, `external_evidence.admit`, and `external_evidence.retire`. -- Frontend and Lark are companion slices. They should render the same plan and - admission projection; neither gets an independent provider registry or - evidence state machine. +- CLI `readback` renders the same validated plan, source, parent decision, + actual deepresearch-ledger coverage and retirement as Markdown. Existing + conversation answer/report and Lark Markdown transports consume that output; + no new UI configuration or evidence state machine is introduced. TypeScript 是 discovery 真值边界、请求身份、provider 准入、provenance 校验、父 Agent 采纳、紧凑投影与 退休条件的唯一语义 owner。Python 仅适配 CLI 与 effect-runtime transport。Managed -Turn 复用同一方法;frontend/Lark 后续只渲染同源投影,不新建 registry 或状态机。 +Turn 复用同一方法;CLI `readback` 的同源 Markdown 可由现有会话答复/报告和 Lark Markdown 运输路径展示,不新建 registry 或状态机。 + +## Public GitHub method / 公开 GitHub 方法 + +The bundled `method:public-github` provider performs anonymous, bounded HTTPS +GETs only. It reads UTF-8 files from explicitly selected full commit SHA URLs. +No token, cookie, private repository, branch-head URL, redirect, proxy credential, +raw page persistence or automatic admission is used. Repository public visibility +is checked at planning and again for each execution read. A saved ready row alone +cannot authorize or prove a successful read. + +该内置 method 只做匿名、有界 HTTPS GET,只接受显式选择的完整 commit SHA 文件 URL。 +规划和执行均检查仓库当前公开性;不使用 token、cookie、私有仓库、分支 head、重定向、 +代理凭据、原文持久化或自动采纳。保留的 ready 行本身不证明执行成功。 + +```bash +loopx external-evidence plan --public-github \ + --objective "Inspect public source" --user-activity "Choose a source" \ + --decision "Whether a literal is present" --evidence-kind literal_match \ + --source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md \ + --search-term LoopX --format json > plan.json +loopx external-evidence execute --plan-json plan.json --execute --format json > execution.json +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json +# The parent separately inspects findings and chooses admit/reject. +loopx external-evidence admit --plan-json plan.json --receipt-json execution.json \ + --decision admit --reason "Direct source answers this bounded decision" \ + --admit-source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md \ + --format json > admission.json +loopx deepresearch start --project research --question "Inspect public source" +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json \ + --admission-json admission.json --project research --execute --format json +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json \ + --admission-json admission.json --project research +``` + +`execute --execute` authorizes source reads only. `readback --execute` authorizes +writing already explicitly admitted compact sources to the **existing** research +ledger. It requires the active research question to match the plan objective. +Retries for the same admission are idempotent. Wrong questions, unrelated existing +sources and exhausted source budgets remain blockers; partial projection stays +retained until every admitted source is actually read back with matching lineage. +Omit `--execute` for read-only projection; omit `--public-github` to avoid the +provider readiness probe. There is no persistent provider enablement to uninstall. +Original sources remain available on partial, empty and failed results. Literal +matches prove neither semantic conclusions nor evidence completeness. + +`execute --execute` 只授权读取来源;`readback --execute` 只把已明确采纳的紧凑证据 +写入现有研究账本,且要求研究问题与 plan objective 相同。同一 admission 可幂等重试; +问题不匹配、已有不相关来源或预算耗尽仍为 blocker。部分下游投影保持 retained, +直至全部已采纳来源以匹配 lineage 实际回读。省略 `--execute` 可只读回读;省略 +`--public-github` 不探测该 provider。没有持久开关需要卸载。部分、空或失败证据 +保留原始来源退路;字面匹配不证明语义结论或证据完整性。 + +Real qualification (anonymous network reads, disposable synthetic ledger): +`uv run --extra test python examples/public-github-evidence-live-smoke.py --execute-public-provider --source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md`. +Packaged conversation readback: set `LOOPX_PERSONAL_WORKSPACE_PACKAGED=1` and +`LOOPX_PERSONAL_WORKSPACE_SCENARIO=external-evidence-readback`, then run +`node examples/personal-workspace-browser-smoke.mjs`. +Live authenticated connector and Lark delivery qualification remain separate; +this method grants no connector credentials or outbound-message authority. + +真实验收会进行匿名网络读取并使用一次性合成账本;打包会话验收沿用上述环境变量和命令。 +真实带凭据 connector 与 Lark 送达资格仍是独立边界;本方法不授予 connector 凭据或外发消息权限。 diff --git a/loopx/capabilities/external_research/catalog_entry.py b/loopx/capabilities/external_research/catalog_entry.py index 047db8bc36..97d874b3ea 100644 --- a/loopx/capabilities/external_research/catalog_entry.py +++ b/loopx/capabilities/external_research/catalog_entry.py @@ -21,12 +21,12 @@ ), "user_value": ( "Discover method and connector inventory, select only a currently ready provider, " - "bind a caller-presented provider receipt to its exact plan, and admit or reject compact " + "execute explicitly selected public GitHub sources, bind a provider receipt to its exact plan, and admit or reject compact " "provenance without copying raw provider content." ), "next_real_step": ( - "run `loopx external-evidence discover --connector-registry`, then provide a current " - "provider inventory to plan and admit the returned provider receipt" + "run `loopx external-evidence plan --public-github --help`, then inspect the returned " + "receipt before explicit admission and downstream ledger readback" ), "entry_command": "loopx external-evidence discover --help", "commands": [ @@ -40,6 +40,16 @@ "purpose": "Bind object, user activity, decision, and evidence kinds to one ready provider.", "write_boundary": "read-only", }, + { + "command": "loopx external-evidence execute --plan-json plan.json --execute", + "purpose": "Read explicit pinned public GitHub sources; parent admission remains separate.", + "write_boundary": "anonymous public HTTPS reads; no raw persistence", + }, + { + "command": "loopx external-evidence readback --plan-json plan.json --receipt-json execution.json ...", + "purpose": "Show the same source lineage, parent decision and actual downstream coverage.", + "write_boundary": "read-only by default; --execute projects admitted sources to the existing research ledger", + }, { "command": "loopx external-evidence receipt --plan-json plan.json --receipt-json receipt.json", "purpose": ( diff --git a/loopx/capabilities/external_research/cli.py b/loopx/capabilities/external_research/cli.py index fd3c4da9db..697741e8bf 100644 --- a/loopx/capabilities/external_research/cli.py +++ b/loopx/capabilities/external_research/cli.py @@ -8,6 +8,8 @@ from ...control_plane.effect_runtime import effect_runtime_result from ..connector_registry.core import load_connector_registry +from ...extensions.public_github_research import execute_public_github, inspect_provider +from .projection import readback, render_readback, validate_receipt PrintPayload = Callable[ @@ -32,6 +34,14 @@ def _load_object(path_text: str, *, label: str) -> dict[str, Any]: return value +def _load_receipt(path_text: str) -> dict: + value = _load_object(path_text, label="external evidence receipt") + receipt = value.get("receipt", value) + if not isinstance(receipt, dict): + raise ValueError("external evidence receipt must be an object") + return receipt + + def _provider_inventory(value: Mapping[str, Any]) -> list[dict[str, object]]: providers = value.get("providers") if not isinstance(providers, list): @@ -80,6 +90,8 @@ def _merge_providers( def _render(payload: dict[str, object]) -> str: + if payload.get("schema_version") == "loopx_external_evidence_readback_v0": + return render_readback(payload) lines = ["# LoopX External Evidence", ""] for field in ( "status", @@ -137,11 +149,27 @@ def register_external_evidence_commands( plan.add_argument("--decision", required=True) plan.add_argument("--evidence-kind", action="append", required=True) plan.add_argument("--constraint", action="append", default=[]) - plan.add_argument("--provider-inventory-json", required=True) + plan.add_argument("--provider-inventory-json") + plan.add_argument("--public-github", action="store_true", help="Opt in to a fresh anonymous public GitHub readiness probe.") + plan.add_argument("--source", action="append", default=[]) + plan.add_argument("--search-term", action="append", default=[]) plan.add_argument("--connector-registry", nargs="?", const="") plan.add_argument("--preferred-provider-id") add_subcommand_format(plan) + execute = actions.add_parser("execute", help="Execute the exact public GitHub plan without admitting evidence.") + execute.add_argument("--plan-json", required=True) + execute.add_argument("--execute", action="store_true", help="Authorize bounded anonymous public-source reads.") + add_subcommand_format(execute) + + project = actions.add_parser("readback", help="Show receipt, parent decision and actual research-ledger coverage.") + project.add_argument("--plan-json", required=True) + project.add_argument("--receipt-json", required=True) + project.add_argument("--admission-json") + project.add_argument("--project", help="Existing deepresearch project; no implicit run is created.") + project.add_argument("--execute", action="store_true", help="Write explicitly admitted sources to the existing ledger.") + add_subcommand_format(project) + receipt = actions.add_parser( "receipt", help="Validate and bind a caller-presented provider receipt to its exact plan.", @@ -194,10 +222,14 @@ def handle_external_evidence_command( {"providers": providers}, ) elif args.external_evidence_action == "plan": - inventory = _load_object( - args.provider_inventory_json, - label="external evidence provider inventory", - ) + if not args.provider_inventory_json and not getattr(args, "public_github", False): + raise ValueError("plan requires provider inventory or explicit --public-github") + inventory = _load_object(args.provider_inventory_json, + label="external evidence provider inventory") if args.provider_inventory_json else {"providers": []} + if getattr(args, "public_github", False): + if len(getattr(args, "source", [])) > 8: + raise ValueError("public GitHub plan allows at most eight sources") + inventory["providers"] = _merge_providers(_provider_inventory(inventory), [inspect_provider(getattr(args, "source", []))]) providers = _merge_providers( _connector_inventory(args.connector_registry), _provider_inventory(inventory), @@ -211,11 +243,31 @@ def handle_external_evidence_command( "decision": args.decision, "evidence_kinds": args.evidence_kind, "constraints": args.constraint, + **({"source_refs": getattr(args, "source", [])} if getattr(args, "source", []) else {}), + **({"search_terms": getattr(args, "search_term", [])} if getattr(args, "search_term", []) else {}), }, "providers": providers, "preferred_provider_id": args.preferred_provider_id, }, ) + elif args.external_evidence_action == "execute": + if not args.execute: + raise ValueError("--execute is required for real public-source reads") + plan = _load_object(args.plan_json, label="external evidence plan") + # Canonical identity must pass the typed owner before any HTTP call. + probe_receipt = {"schema_version": "loopx_external_evidence_receipt_v0", + "plan_id": plan.get("plan_id"), "request_id": plan.get("request", {}).get("request_id"), + "provider_id": plan.get("selected_provider", {}).get("provider_id"), + "provider_kind": plan.get("selected_provider", {}).get("provider_kind"), + "status": "failed", "sources": [], "summary": "Validation only", "completed_at": "not-executed"} + validate_receipt(plan, probe_receipt) + payload = execute_public_github(plan) + payload["observation"] = validate_receipt(plan, payload["receipt"]) + elif args.external_evidence_action == "readback": + payload = readback(_load_object(args.plan_json, label="external evidence plan"), + _load_receipt(args.receipt_json), + _load_object(args.admission_json, label="external evidence admission") if args.admission_json else None, + project=Path(args.project).expanduser() if args.project else None, execute=args.execute) elif args.external_evidence_action == "receipt": payload = effect_runtime_result( "external_evidence.receipt", @@ -223,9 +275,7 @@ def handle_external_evidence_command( "plan": _load_object( args.plan_json, label="external evidence plan" ), - "receipt": _load_object( - args.receipt_json, label="external evidence receipt" - ), + "receipt": _load_receipt(args.receipt_json), }, ) elif args.external_evidence_action == "admit": @@ -235,9 +285,7 @@ def handle_external_evidence_command( "plan": _load_object( args.plan_json, label="external evidence plan" ), - "receipt": _load_object( - args.receipt_json, label="external evidence receipt" - ), + "receipt": _load_receipt(args.receipt_json), "decision": { "disposition": args.decision, "reason": args.reason, diff --git a/loopx/capabilities/external_research/projection.py b/loopx/capabilities/external_research/projection.py new file mode 100644 index 0000000000..3d01582468 --- /dev/null +++ b/loopx/capabilities/external_research/projection.py @@ -0,0 +1,111 @@ +"""Shared external evidence readback for CLI and existing conversation surfaces. + +Typed effect-runtime reductions remain the sole admission/retirement authority. +Deep-research owns its existing durable source ledger; no parallel evidence store. +""" +from __future__ import annotations + +import re +from pathlib import Path + +from ...control_plane.effect_runtime import effect_runtime_result +from ..deep_research.runtime import add_source, load_state + + +def validate_receipt(plan: dict, receipt: dict) -> dict: + return effect_runtime_result("external_evidence.receipt", {"plan": plan, "receipt": receipt}) + + +def _validate_admission(plan: dict, receipt: dict, admission: dict) -> dict: + normalized = effect_runtime_result("external_evidence.admit", {"plan": plan, "receipt": receipt, + "decision": {"disposition": admission.get("disposition"), "reason": admission.get("reason"), + "admitted_source_refs": admission.get("admitted_source_refs", [])}}) + if normalized != admission: + raise ValueError("admission does not match its exact plan and receipt") + return normalized + + +def _lineage(admission: dict, source: dict) -> dict: + return {"plan_id": admission["plan_id"], "admission_id": admission["admission_id"], + "receipt_digest": admission["receipt_digest"], "content_digest": source["content_digest"]} + + +def readback(plan: dict, receipt: dict, admission: dict | None = None, + *, project: Path | None = None, execute: bool = False) -> dict: + observation = validate_receipt(plan, receipt) + receipt = observation["receipt"] + if execute and (admission is None or project is None): + raise ValueError("downstream write requires an explicit parent admission and research project") + normalized = _validate_admission(plan, receipt, admission) if admission is not None else None + write_blockers = [] + if execute and normalized["disposition"] == "admit": + for source in normalized["downstream_projection"]["sources"]: + try: + add_source(project, url_or_path=source["source_ref"], tool="external-evidence", + title="Public evidence", claims=[{"text": source["finding"], "stance": "neutral"}], + external_evidence=_lineage(normalized, source), expected_question=plan["request"]["objective"]) + except ValueError as error: + # A bounded partial projection is retained. Retrying is idempotent + # for the same identity; failed sources never count as covered. + write_blockers.append(str(error)) + break + covered = [] + if project is not None and normalized is not None: + state = load_state(project) + if state is not None and state["question"] == plan["request"]["objective"]: + for source in normalized["downstream_projection"]["sources"]: + for row in state["sources"]: + if (row["url_or_path"] == source["source_ref"] and + row.get("external_evidence") == _lineage(normalized, source) and + any(claim["id"] in row["claims"] and claim["text"] == source["finding"] + for claim in state["claims"])): + covered.append(source["source_ref"]) + break + retirement = effect_runtime_result("external_evidence.retire", {"admission": normalized, + "downstream_source_refs": covered}) if normalized is not None else None + return {"schema_version": "loopx_external_evidence_readback_v0", "plan_id": plan["plan_id"], + "request": plan["request"], "provider_id": receipt["provider_id"], "receipt_status": receipt["status"], + "sources": receipt["sources"], "summary": receipt["summary"], "limitations": receipt["limitations"], + "parent_admission": None if normalized is None else {"disposition": normalized["disposition"], + "reason": normalized["reason"], "admission_id": normalized["admission_id"], + "admitted_source_refs": normalized["admitted_source_refs"]}, + "downstream_source_refs": covered, "retirement": retirement, "write_blockers": write_blockers, + "original_source_fallback_allowed": True, "automatic_admission": False, + "evidence_coverage_observed": False} + + +def _text(value: object) -> str: + return re.sub(r"([\\`*_{}\[\]()<>#!|])", r"\\\1", str(value)).replace("\n", " ") + + +def render_readback(payload: dict) -> str: + admission = payload["parent_admission"] + retirement = payload["retirement"] + lines = ["## External evidence / 外部证据", "", _text(payload["request"]["objective"]), "", + "- Provider / 来源:" + _text(payload["provider_id"]), + "- Receipt / 读取结果:" + payload["receipt_status"], + "- Parent decision / 父 Agent 决定:" + ("pending / 待采纳" if admission is None else _text(admission["disposition"])), + f"- Downstream coverage / 下游覆盖:{len(payload['downstream_source_refs'])} admitted sources / 已采纳来源", + "- Retirement / 退休:" + ("pending / 待决定" if retirement is None else retirement["status"]), + "- Original-source fallback remains available / 可继续使用原始来源。", + "- Evidence completeness is unverified / 证据完整性尚未验证。", ""] + if admission is not None: + lines.extend(["Parent reason / 采纳理由:" + _text(admission["reason"]), ""]) + for ref in payload["request"].get("source_refs", []): + lines.extend(["Original source / 原始来源:" + _text(ref), ""]) + admitted = set(admission["admitted_source_refs"]) if admission is not None else set() + covered = set(payload["downstream_source_refs"]) + for index, source in enumerate(payload["sources"], 1): + ref = source["source_ref"] + lines.extend([f"### Source {index} / 来源 {index}", "", _text(ref), "", + _text(source["finding"]), "", + "- Evidence basis / 依据:" + _text(source["basis"]), + "- Accessed / 读取时间:" + _text(source["accessed_at"]), + "- Digest / 摘要:" + _text(source["content_digest"]), + "- Admission / 采纳:" + ("admitted" if ref in admitted else "not admitted"), + "- Downstream / 下游:" + ("observed" if ref in covered else "not observed"), + "- Limitation / 限制:" + _text(source.get("limitation") or "unspecified"), ""]) + for limitation in [*payload["limitations"], *payload["write_blockers"]]: + lines.append("- " + _text(limitation)) + lines.extend(["", "Plan identity / 计划身份:" + _text(payload["plan_id"])]) + return "\n".join(lines) + "\n" diff --git a/loopx/capabilities/manager_context/README.md b/loopx/capabilities/manager_context/README.md index 70389bbb0c..39d7a46f61 100644 --- a/loopx/capabilities/manager_context/README.md +++ b/loopx/capabilities/manager_context/README.md @@ -2,55 +2,56 @@ Built-in capability for original intent delivery and receiver-owned replanning. The owner's local manager channel uses registered workers automatically. -External channels need an owner-configured grant in +An external conversation requires a trusted operator's sender-bound policy in `/.local/manager-context/policy.json`: ```json {"schema_version":"loopx_manager_context_policy_v1","sources":{ - "manager.external.example":{ - "sender_ids":["exact-provider-sender"], - "targets":[{"goal_id":"research"}] + "manager.external.0123456789abcdef01234567":{ + "sender_ids":["exact-provider-sender"] } }} ``` -Use the actual connection channel and provider sender identity. Keep this file -private (0600); do not commit it. Missing grants disable external delivery. -For an existing channel with an authorized sender, use the local operator CLI -to preview, grant, or revoke a managed Goal without editing the -policy file by hand: +**Default behavior change:** configured senders can now deliver context to every +active registered Agent in this local registry, across Goals and including future +registrations. Existing enrollment lists do not narrow this default. To retain a +selected audience's previous enrollment boundary, explicitly set +`"local_delivery_scope":"selected"` alongside its `targets` list before upgrading. +Missing policy, missing source, a wrong sender or malformed scope grants nothing. +Keep the policy private (0600); do not commit it. A read grant without a sender +grant is insufficient. This does not grant remote delivery, evidence reads, +worker launch, Todo/lease changes or protected operations. + +The existing local operator commands preview, apply and verify exceptions: ```sh -loopx manager-inbox grant-delivery-target --channel-id manager.external.0123456789abcdef01234567 --goal-id research -loopx manager-inbox grant-delivery-target --channel-id manager.external.0123456789abcdef01234567 --goal-id research --execute +loopx manager-inbox revoke-delivery-target --channel-id manager.external.0123456789abcdef01234567 --goal-id research --agent-id worker loopx manager-inbox revoke-delivery-target --channel-id manager.external.0123456789abcdef01234567 --goal-id research --agent-id worker --execute loopx manager-inbox revoke-delivery-target --channel-id manager.external.0123456789abcdef01234567 --goal-id research --execute +loopx manager-inbox grant-delivery-target --channel-id manager.external.0123456789abcdef01234567 --goal-id research --execute +loopx manager-inbox grant-delivery-target --channel-id manager.external.0123456789abcdef01234567 --goal-id research --agent-id worker --execute ``` -Pass the same `--registry` and `--runtime-root` used by the manager connection. -Without `--execute`, these commands only preview the target and count change. -Omitting `--agent-id` covers all current and future registered Agents in that -Goal. Use this for the owner's managed scope; a newly registered Agent then needs -no separate enrollment. Supplying `--agent-id` retains one-recipient enrollment -or revocation. Individual revocation is stored in `blocked_targets`, overrides -the Goal grant, and survives reapplying that Goal grant. Explicitly grant the -Agent to restore it. Revoking a Goal removes both its broad and individual grants. -Existing exact-recipient policies retain their scope until a trusted operator -promotes them; evidence read scope alone never becomes delegation authority. - -Grant requires an active registered Goal (and a registered Agent when specified), an existing sender-bound -channel, and membership in any explicit audience Goal read scope. The command -does not create a sender grant, launch the Agent, or grant protected-operation -authority. Revocation also works when the former Agent is no longer registered. -Remove a source/target grant to revoke future delivery, including replay attempts. -The shared TypeScript source-recipient owner resolves registration and exceptions -for discovery, direct Chat handoff and later peer consultation. Provider adapters -verify ingress and perform locked file IO. Missing, malformed or revoked grants -do not record a request; stopped or unreadable Goals are excluded. This config -slice has a CLI preview/apply/readback; the existing Chat uses its resulting -catalog. Editing source grants in the packaged settings UI remains unqualified. -Provider ingress receipts bind the current message digest, channel and sender; -a model cannot create that provenance through its response. +Pass the manager connection's `--registry` and `--runtime-root`. Without +`--execute`, no policy bytes change. `blocked_targets` retains exact Agent and +whole-Goal revocations, including future members. Restoring a Goal preserves +individual revocations, including those recorded while the Goal is disabled; +restore the Goal first, then an individual Agent when needed. Revocation still +works after unregistration. In `selected` mode, granting +a Goal enrolls its current/future Agents; exact grants enroll only that Agent. +An explicit audience read scope still bounds enrollment in selected mode. + +The shared TypeScript source-recipient owner resolves the same current rule for +discovery, direct Chat delivery, replay and later peer consultation. Python verifies +provider ingress and performs locked file IO. Stopped or unreadable Goal activation +is excluded before admission. The existing App and Lark conversation paths consume +this catalog; registration and delivery never prove receiver adoption or execution. +The operator grant editor in packaged settings remains unqualified; this repair +adds no separate configuration page. Remove the sender/source grant to disable +external delivery, including future replay. Existing inbox records are retained. +Provider ingress receipts bind message digest, channel and sender; a model cannot +create or widen that provenance through its response. The existing worker turn-start hook exposes only a bounded pending count and required read command, without copying private content into status projections. diff --git a/loopx/capabilities/multi_subagent/native_child_receipts.py b/loopx/capabilities/multi_subagent/native_child_receipts.py index 1044b845b5..e823a2859d 100644 --- a/loopx/capabilities/multi_subagent/native_child_receipts.py +++ b/loopx/capabilities/multi_subagent/native_child_receipts.py @@ -1,10 +1,4 @@ -"""Turn-bound reports of host-native child-tool decisions. - -The native tool belongs to the host. LoopX can durably reconcile the -coordinator's typed report of its result, but cannot attest that a host call -occurred unless the host itself supplies an integration. This distinction is -part of the projection, not an implicit promise of configured capacity. -""" +"""Turn-bound native child decisions, with explicit report/host provenance.""" from __future__ import annotations @@ -119,13 +113,17 @@ def native_child_activity( attempted = sum(row.get("operation") in {"spawn", "followup"} for row in ordered) rejected = sum(row.get("outcome") == "capacity_rejected" for row in ordered) host_failed = sum(row.get("outcome") == "host_failed" for row in ordered) + sources = {row.get("observation_source") for row in ordered} + observation = ("unknown" if not sources else "host_observed" + if sources == {"host_observed"} else "mixed" + if "host_observed" in sources else "coordinator_reported") return { "schema_version": NATIVE_SUBAGENT_ACTIVITY_SCHEMA_VERSION, "goal_id": goal_id, "agent_id": agent_id, "turn_instance_id": turn_instance_id, "entrypoint_scope": "host_native_child_tools", - "observation": "coordinator_reported" if ordered else "unknown", - "host_attested": False, + "observation": observation, + "host_attested": observation == "host_observed", "configured_limit_kind": "upper_bound", "configured_limit": configured_limit, "observed_capacity": "capacity_rejection_reported" if rejected else @@ -145,12 +143,12 @@ def native_child_activity( } -def load_native_child_activity( +def _load_native_child_events( runtime_root: Path, *, goal_id: str, agent_id: str, - turn_instance_id: str, configured_limit: int, + turn_instance_id: str, goal_ref: Mapping[str, Any] | None = None, registry_path: Path | None = None, -) -> dict[str, Any]: +) -> list[dict[str, Any]]: with quota_accounting_admission( runtime_root=runtime_root, registry_path=registry_path, @@ -183,6 +181,7 @@ def load_native_child_activity( event for event in source if event.get("event_kind") in EVENT_KINDS.values() + and event.get("goal_id") == goal_id and event.get("agent_id") == agent_id and event.get("run_id") == turn_instance_id and ( @@ -191,14 +190,25 @@ def load_native_child_activity( else "goal_ref" not in event ) ] - return native_child_activity( - events, - goal_id=goal_id, - agent_id=agent_id, - turn_instance_id=turn_instance_id, - configured_limit=configured_limit, - goal_ref=goal_ref, - ) + return events + + +def load_native_child_activity( + runtime_root: Path, *, goal_id: str, agent_id: str, + turn_instance_id: str, configured_limit: int, + goal_ref: Mapping[str, Any] | None = None, + registry_path: Path | None = None, +) -> dict[str, Any]: + events = _load_native_child_events( + runtime_root, goal_id=goal_id, agent_id=agent_id, + turn_instance_id=turn_instance_id, goal_ref=goal_ref, + registry_path=registry_path, + ) + return native_child_activity( + events, goal_id=goal_id, agent_id=agent_id, + turn_instance_id=turn_instance_id, configured_limit=configured_limit, + goal_ref=goal_ref, + ) def latest_native_child_activity( @@ -284,6 +294,9 @@ def _record_native_child( registry_path: Path | None = None, goal_ref: Mapping[str, Any] | None = None, source_admission: Mapping[str, Any] | None = None, + _host_observed: bool = False, + _host_child_refs: Sequence[str] | None = None, + _host_wait_ref: str | None = None, ) -> dict[str, Any]: """Preview or append a typed report; never launch a child or spend quota.""" goal_id = _id(goal_id, field="goal_id") @@ -292,10 +305,26 @@ def _record_native_child( operation_id = _id(operation_id, field="operation_id") if isinstance(configured_limit, bool) or not isinstance(configured_limit, int) or configured_limit < 1: raise ValueError("enabled multi_subagent configured_limit must be positive") - fields = _normalized_fields( + fields: dict[str, Any] = _normalized_fields( stage=stage, operation=operation, outcome=outcome, entrypoint_id=entrypoint_id, reason_code=reason_code, evidence_ref=evidence_ref, validation_ref=validation_ref, ) + if _host_observed: + if stage == "review" or operation == "skip": + raise ValueError("host observation cannot attest a parent review or skip") + fields["observation_source"] = "host_observed" + if _host_child_refs is not None: + if (not _host_observed or stage != "decision" or outcome != "started" + or isinstance(_host_child_refs, (str, bytes)) or not _host_child_refs): + raise ValueError("child correlation requires a started host-observed decision") + # Rollout details are scalar; each opaque binding remains exact and + # cannot collide with the decision fields or be text-truncated. + fields.update({"host_child_ref_" + _id(ref, field="host_child_ref"): True + for ref in sorted(set(_host_child_refs))}) + if _host_wait_ref is not None: + if not _host_observed or stage != "result": + raise ValueError("wait correlation requires a host-observed result") + fields["host_wait_ref"] = _id(_host_wait_ref, field="host_wait_ref") log_path = rollout_event_log_path(runtime_root, goal_id) events = load_rollout_events(log_path) prior = _events_for_turn( @@ -307,10 +336,35 @@ def _record_native_child( ) existing = next((event for event in prior if event.get("case_id") == operation_id - and event.get("event_kind") == EVENT_KINDS[stage]), None) + and event.get("event_kind") == EVENT_KINDS[stage] + and (stage != "result" or _host_wait_ref is None + or _details(event).get("host_wait_ref") == _host_wait_ref)), None) + if stage == "result" and _host_wait_ref is None and existing is not None: + # A later decision snapshot can confirm the same typed result without + # replacing the causal wait binding already stored on that result. + wait_ref = _details(existing).get("host_wait_ref") + if wait_ref is not None: + fields["host_wait_ref"] = wait_ref if existing is not None and _details(existing) != fields: raise ValueError("operation identity already has a conflicting native child report") + def validate_result_identity(current: Sequence[Mapping[str, Any]]) -> None: + if stage != "result": + return + for result in current: + if result.get("event_kind") != EVENT_KINDS["result"]: + continue + details = _details(result) + if (_host_wait_ref is not None and details.get("host_wait_ref") == _host_wait_ref + and result.get("case_id") != operation_id): + raise ValueError("wait identity already has a conflicting native child binding") + if (result.get("case_id") == operation_id + and any(details.get(key) != fields.get(key) + for key in ("outcome", "observation_source"))): + raise ValueError("operation identity already has a conflicting native child report") + + validate_result_identity(prior) + def report_admission() -> Mapping[str, Any]: readback = read_heartbeat_settlement( runtime_root, goal_id=goal_id, agent_id=agent_id, todo_id=None, @@ -337,12 +391,15 @@ def validate_transition(observed: Sequence[Mapping[str, Any]]) -> None: current = _events_for_turn(observed, goal_id=goal_id, agent_id=agent_id, turn_instance_id=turn_instance_id, goal_ref=goal_ref) + validate_result_identity(current) decisions = {str(item.get("case_id")): item for item in current if item.get("event_kind") == EVENT_KINDS["decision"]} if stage == "decision": if admission["report_permission"] != "new_operation": raise ValueError("native child decision requires an open, work-admitted Turn guard") - if fields["operation"] in {"spawn", "followup"} and any( + # Host observations record calls that already happened, including + # violations; recording one cannot authorize another host call. + if not _host_observed and fields["operation"] in {"spawn", "followup"} and any( _details(item).get("outcome") in {"capacity_rejected", "host_failed"} for item in decisions.values() ): @@ -381,6 +438,7 @@ def validate_transition(observed: Sequence[Mapping[str, Any]]) -> None: "run_id", "case_id", *(("goal_ref",) if goal_ref is not None else ()), + *(("details",) if _host_wait_ref is not None else ()), ), precondition=lambda: validate_transition(load_rollout_events(log_path)), ) @@ -414,6 +472,9 @@ def record_native_child( evidence_ref: str | None = None, validation_ref: str | None = None, execute: bool = False, registry_path: Path | None = None, goal_ref: Mapping[str, Any] | None = None, + _host_observed: bool = False, + _host_child_refs: Sequence[str] | None = None, + _host_wait_ref: str | None = None, ) -> dict[str, Any]: """Preview or append one report under its exact quota owner.""" @@ -443,4 +504,7 @@ def record_native_child( registry_path=registry_path, goal_ref=goal_ref, source_admission=source_admission, + _host_observed=_host_observed, + _host_child_refs=_host_child_refs, + _host_wait_ref=_host_wait_ref, ) diff --git a/loopx/capabilities/pr_review_queue/README.md b/loopx/capabilities/pr_review_queue/README.md index 8ffd8eb3ea..3ef057c28a 100644 --- a/loopx/capabilities/pr_review_queue/README.md +++ b/loopx/capabilities/pr_review_queue/README.md @@ -1102,7 +1102,31 @@ the immutable base and exact head, or an independently evidenced external outage, is not a reason to request code changes on an unrelated PR when its changed invariant has separate passing coverage. Record the red check and its owner; approval does not make a blocked merge ready. A new, worsened or -unattributed failure remains a review blocker. Disabling CI waiting +unattributed current failure remains a review blocker. Policy revision 18 scopes +the matrix to the reviewed head. Keep relevant older failures in existing +result/evidence text with their source and explain why current independent +evidence covers the exposed invariant and conditions. An unknown historical +cause alone is not a veto, nor does current passing evidence prove that cause +was fixed. A selected green rerun cannot dismiss material intermittency or a +missing negative case: record the current gap as failed/unverified and name +the smallest discriminating check. An explicit accepted contract may still +require causal attribution. No new result schema, review-dismissal authority or +merge exception is introduced. + +CI completion is a separate merge decision. With `wait_for_ci=true`, observe +available CI and retain the configured merge gate, but do not require every +remote job to finish or succeed before approving independently verified code. +`repository_required_checks` records decisive repository validation for the +review; each matrix row's `required` flag means review evidence, not GitHub +branch protection. Record merely pending remote jobs as diagnostic rows with +`required=false` when current independent coverage establishes their relevant +invariants. If a queued job is the only decisive coverage, leave that invariant +required and unverified. Pending CI alone never justifies `REQUEST_CHANGES`; +missing relevant evidence, current regressions and material instability still +do. An earned approval may coexist with `ready=false`, and changing the CI +waiting configuration requires the existing owner's authorization. + +Disabling CI waiting also removes CI requests and waiting instructions; legacy supplied summaries are diagnostic only. It grants no publication, merge, or admin-bypass authority. diff --git a/loopx/capabilities/pr_review_queue/review_contract.py b/loopx/capabilities/pr_review_queue/review_contract.py index 0ebf6c9313..77f2b853cc 100644 --- a/loopx/capabilities/pr_review_queue/review_contract.py +++ b/loopx/capabilities/pr_review_queue/review_contract.py @@ -8,7 +8,7 @@ from .approval_closeout import approval_closeout_contract # Increment when review requirements change without changing the packet shape. -REVIEW_POLICY_REVISION = 16 +REVIEW_POLICY_REVISION = 18 # Reuse the existing evidence fields for publication, rather than inventing a # second problem assessment or treating a jargon denylist as comprehension. @@ -49,7 +49,7 @@ ], "external_fields": ["independent_evidence", "retry_or_recovery_owner"], "rule": ( - "Classify every required failed or skipped validation before choosing a review verdict. " + "Classify every currently required failed or skipped validation before choosing a review verdict. " "A pre-existing failure is non-blocking for review only when the same check on an " "immutable base and exact head has the same normalized failing identity and detail, " "the PR does not alter that failure's causal path, and the changed invariant has " @@ -60,6 +60,23 @@ "their recovery separately from the PR verdict: APPROVE may be correct while merge " "readiness remains on hold. Never relax a hard limit or required check to make it green." ), + "evidence_scope": ( + "The validation matrix assesses the exact reviewed head, not the union of all historical " + "test failures. Retain relevant earlier runs, revisions, commands and failure signatures " + "in the existing result/evidence text; explain which current evidence supersedes them " + "and why it covers the originally exposed invariant and conditions. An unexplained " + "historical cause alone does not require REQUEST_CHANGES when independent current " + "evidence is sufficient. Do not invent a causal explanation or claim the old failure " + "was fixed. A later green run alone does not resolve intermittency: consider the whole " + "bounded run set, concurrency, environment and coverage; do not select only successes, " + "remove assertions or loosen limits. Mark a still-material instability or coverage gap " + "failed or unverified, even if the latest command passed. Name the affected invariant, " + "present evidence gap and smallest discriminating check in the existing finding fields. " + "Full historical root-cause attribution is required only where it is necessary to " + "resolve that current risk or an explicit accepted contract requires it. Preserve " + "unresolved risks and separate merge gates; history is neither an automatic veto nor " + "permission to dismiss an existing review." + ), } OUTCOME_IMPACT_ASSESSMENT = { @@ -889,9 +906,17 @@ def build_review_execution_contract(*, wait_for_ci: bool = True) -> dict[str, An "ci_policy": "required" if wait_for_ci else "not_consulted", "wait_for_ci": wait_for_ci, "validation_source": ( - "Repository-native local validation and final CI observation are required. " - "Attribute failed checks before judging the PR; an unrelated red check " - "may hold merging without requiring code changes on this PR." + "Repository-native validation must establish the changed invariants at the " + "reviewed head. Observe currently available CI without making completion " + "or success of every remote job a prerequisite for APPROVE. " + "repository_required_checks records the validation required for the code " + "judgment; required means review evidence, not GitHub branch protection. " + "Record pending remote jobs separately as diagnostic rows with required=false " + "when independent current evidence already covers their relevant invariants. " + "If a pending job is the only decisive coverage, keep that invariant's row " + "required and unverified. Attribute current failures before judging the PR. " + "Pending CI alone does not justify REQUEST_CHANGES; merge readiness still " + "enforces its configured CI policy." if wait_for_ci else "Repository-native local validation at the reviewed head. " "Do not fetch, poll, or wait for GitHub CI. Missing, pending, " @@ -1242,7 +1267,9 @@ def build_review_execution_contract(*, wait_for_ci: bool = True) -> dict[str, An "pre-existing failure or external infrastructure, and the PR's changed " "invariant is covered. Record the separate merge-readiness hold; do not ask " "this PR to repair unrelated code or budgets. Unattributed, introduced, or " - "worsened failures still block approval." + "worsened current failures still block approval. Apply validation_matrix's " + "evidence_scope to earlier observations; historical root-cause completeness " + "is not an independent approval gate." ), "open_pr_unjustified_delivery": ( "REQUEST_CHANGES when problem_context is off_goal, fragmented or " @@ -1472,7 +1499,7 @@ def build_agent_response_contract(*, wait_for_ci: bool = True) -> dict[str, Any] "Before evidence commands, obey pull_requests[].review_action_kind. A null action stays in pull_requests inventory but is excluded from review_sequence, carries no execution artifacts, and remains readback-only; generic re-review wording selects the PR but does not force duplicate evidence for an already concluded or merged no-action row.", "Execute each non-null pull_requests[].review_plan against the shared review_execution_contract before drafting prose.", "Do not infer verified evidence from title, labels, changed-file counts, metadata_risk_hint, or green CI alone.", - ("Observe final CI in addition to repository-native local validation, then attribute red checks before judging this PR; review approval and merge readiness are separate." if wait_for_ci else "Do not fetch, poll, or wait for CI for review or merge readiness. repository_required_checks means repository-native local validation; attribute base-equivalent failures and keep missing affected-invariant evidence blocking."), + ("Observe available CI alongside repository-native validation. APPROVE does not require every CI job to finish or succeed when independent current evidence covers the changed invariants; pending CI is a separate merge-readiness hold. Keep missing decisive coverage and material current failures blocking, and apply validation_matrix's evidence_scope to history." if wait_for_ci else "Do not fetch, poll, or wait for CI for review or merge readiness. repository_required_checks means repository-native local validation; attribute base-equivalent failures and keep missing affected-invariant evidence blocking."), "Recheck the exact remote head before verdict and publication.", "After publishing and reading back APPROVE, execute review_execution_contract.approval_closeout; approval alone does not clear another reviewer's effective blocking review.", "Render the verified result through a non-null pull_requests[].review_template; host skills must not maintain a competing depth checklist.", diff --git a/loopx/chat_agent.py b/loopx/chat_agent.py index 97720aff59..917af87fb0 100644 --- a/loopx/chat_agent.py +++ b/loopx/chat_agent.py @@ -933,6 +933,7 @@ def send( attachments: list[dict[str, Any]] | None = None, on_event: Callable[[str, dict[str, Any]], None] | None = None, output_schema: dict[str, Any] | None = None, + on_native_item: Callable[[dict[str, Any]], None] | None = None, ) -> dict[str, Any]: text = " ".join(str(user_message or "").split()) if not text: @@ -1030,6 +1031,10 @@ def send( self.current_turn_id = turn_id method = str(message.get("method") or "") params = message.get("params") + if method == "item/completed" and isinstance(params, dict) and on_native_item: + native_item = params.get("item") + if isinstance(native_item, dict) and native_item.get("type") == "collabAgentToolCall": + on_native_item(native_item) if on_event: phase = { "turn/started": "Agent 已开始处理", diff --git a/loopx/chat_configuration_api.py b/loopx/chat_configuration_api.py index d7f16df9b4..8d09835d63 100644 --- a/loopx/chat_configuration_api.py +++ b/loopx/chat_configuration_api.py @@ -2,7 +2,7 @@ from collections.abc import Callable -from . import chat_goal_ownership_api as ownership_api +from .presentation import goal_ownership_api as ownership_api from . import chat_usage_statistics_api as usage_api from . import chat_goal_configuration_api as goal_api from . import chat_machine_configuration_api as machine_api diff --git a/loopx/cli_commands/agent_context.py b/loopx/cli_commands/agent_context.py index 1465b050d7..4d96a51027 100644 --- a/loopx/cli_commands/agent_context.py +++ b/loopx/cli_commands/agent_context.py @@ -203,7 +203,8 @@ def handle_agent_context(args, registry_path, runtime_root, print_payload, outpu ) if native_activity is not None: payload["native_child_activity"] = native_activity - payload["host_receipts_scope"] = "turn_bound_coordinator_report" + payload["host_receipts_observed"] = native_activity["host_attested"] + payload["host_receipts_scope"] = "turn_bound_native_child_receipts" print_payload(payload, output_format(args), render_agent_context) return 0 diff --git a/loopx/control_plane/capabilities/external_evidence.ts b/loopx/control_plane/capabilities/external_evidence.ts index 4dcf8ed30b..260f0e8246 100644 --- a/loopx/control_plane/capabilities/external_evidence.ts +++ b/loopx/control_plane/capabilities/external_evidence.ts @@ -92,6 +92,19 @@ function normalizeRequest(value: unknown): JsonObject { (normalized.evidence_kinds as string[]).length > 0, "request.evidence_kinds must not be empty", ); + // Optional source selection is part of the exact request identity. Preserve + // legacy request digests when no source-bound provider is requested. + if (request.source_refs !== undefined) { + const refs = boundedStrings(request.source_refs, "request.source_refs", 8, 2048); + requireThat(refs.length > 0 && new Set(refs).size === refs.length, + "request.source_refs must be nonempty and unique"); + requireThat(refs.every((ref) => SOURCE_REF_RE.test(ref) && !ref.startsWith("file://")), + "request.source_refs must be non-file provenance URIs"); + normalized.source_refs = refs; + } + if (request.search_terms !== undefined) { + normalized.search_terms = boundedStrings(request.search_terms, "request.search_terms", 8, 256); + } normalized.request_id = digest(normalized); return normalized; } diff --git a/loopx/control_plane/collaboration/README.md b/loopx/control_plane/collaboration/README.md index 0e276712c3..1ab94b7adc 100644 --- a/loopx/control_plane/collaboration/README.md +++ b/loopx/control_plane/collaboration/README.md @@ -44,9 +44,12 @@ ingress and policy reader lives in `source_grant_observation.py`; the Chat capability retains its public `authority` API and supplies its own instruction. This removes a dependency from shared coordination to a product adapter without introducing a second policy writer or migrating existing records. The typed -`source_grants.ts` owner resolves exact recipients and managed Goal targets; -Goal grants include future registered Agents while explicit recipient exclusions -remain effective. The existing operator configuration path delegates changes to +`source_grants.ts` owner resolves exact recipients and managed Goal targets. +Authorized sender-bound sources default to all currently registered local +recipients, across Goals and future registrations. Explicit `selected` scope +retains enrollment; exact and whole-Goal exclusions override the default, including +parent forwarding and replay. Provider observations contain only this host's +active registrations; neither scope admits remote execution. The existing operator configuration path delegates changes to that same owner. Read grants and executor permissions do not imply delegation. `inbox.py` adapts the existing private file stores; typed request validation stays diff --git a/loopx/control_plane/collaboration/source_grants.ts b/loopx/control_plane/collaboration/source_grants.ts index edb3406f1c..f9b1c601ec 100644 --- a/loopx/control_plane/collaboration/source_grants.ts +++ b/loopx/control_plane/collaboration/source_grants.ts @@ -3,6 +3,7 @@ import { EffectRuntimeRequestError } from "../effect_runtime_errors.ts"; import { requireBoolean, requireJsonObject, requireNonEmptyString, requireStringArray } from "../runtime_decode.ts"; type Recipient = { goal_id: string; agent_id?: string }; +type LocalDeliveryScope = "all_registered" | "selected"; function recipients(value: unknown, label: string, exact: boolean): Recipient[] { if (!Array.isArray(value)) throw new EffectRuntimeRequestError(`${label} must be an array`); @@ -18,21 +19,32 @@ function recipients(value: unknown, label: string, exact: boolean): Recipient[] function grants(source: JsonObject) { return { targets: recipients(Object.hasOwn(source, "targets") ? source.targets : [], "context delivery targets", false), - blocked: recipients(Object.hasOwn(source, "blocked_targets") ? source.blocked_targets : [], "blocked context recipients", true), + blocked: recipients(Object.hasOwn(source, "blocked_targets") ? source.blocked_targets : [], "blocked context recipients", false), + scope: localScope(source), }; } +function localScope(source: JsonObject): LocalDeliveryScope { + const scope = Object.hasOwn(source, "local_delivery_scope") ? source.local_delivery_scope : "all_registered"; + if (scope !== "all_registered" && scope !== "selected") { + throw new EffectRuntimeRequestError("local delivery scope must be all_registered or selected"); + } + return scope; +} + function matches(grant: Recipient, target: Recipient): boolean { return grant.goal_id === target.goal_id && (grant.agent_id === undefined || grant.agent_id === target.agent_id); } -function granted(target: Recipient, targets: Recipient[], blocked: Recipient[]): boolean { - return targets.some(row => matches(row, target)) && !blocked.some(row => matches(row, target)); +function granted(target: Recipient, targets: Recipient[], blocked: Recipient[], scope: LocalDeliveryScope): boolean { + return (scope === "all_registered" || targets.some(row => matches(row, target))) && + !blocked.some(row => matches(row, target)); } /** Source provenance is verified by the provider adapter; registration is observed - * afresh. A Goal target covers its current/future Agents, never other Goals, - * evidence reads, protected operations or executor readiness. + * afresh on this host. Authorized sources default to all registered local + * recipients; selected scope opts into enrollment. Neither scope grants remote + * delivery, evidence reads, protected operations or executor readiness. */ export function resolveSourceRecipients(params: JsonObject): JsonObject { const source = requireJsonObject(params.source, "source policy"); @@ -40,9 +52,9 @@ export function resolveSourceRecipients(params: JsonObject): JsonObject { if (!requireStringArray(source.sender_ids, "source senders").includes(sender)) { throw new EffectRuntimeRequestError("source sender is not authorized"); } - const { targets, blocked } = grants(source); + const { targets, blocked, scope } = grants(source); const available = recipients(params.available, "registered recipients", true); - const selected = available.filter(target => granted(target, targets, blocked)); + const selected = available.filter(target => granted(target, targets, blocked, scope)); const unique = new Map(selected.map(row => [JSON.stringify([row.goal_id, row.agent_id]), row])); return { targets: [...unique.values()].sort((a, b) => a.goal_id.localeCompare(b.goal_id) || a.agent_id!.localeCompare(b.agent_id!)) }; @@ -53,7 +65,7 @@ export function resolveSourceRecipients(params: JsonObject): JsonObject { */ export function configureSourceRecipient(params: JsonObject): JsonObject { const source = requireJsonObject(params.source, "source policy"); - const { targets, blocked } = grants(source); + const { targets, blocked, scope } = grants(source); const goal_id = requireNonEmptyString(params.goal_id, "delivery target Goal"); const agent_id = params.agent_id === null || params.agent_id === undefined ? undefined : requireNonEmptyString(params.agent_id, "delivery target Agent"); @@ -70,7 +82,7 @@ export function configureSourceRecipient(params: JsonObject): JsonObject { !available.some(row => matches(target, row)))) { throw new EffectRuntimeRequestError("delivery target must be a registered Agent or all Agents in an active Goal"); } - if (Object.hasOwn(source, "evidence_goal_ids")) { + if (scope === "selected" && Object.hasOwn(source, "evidence_goal_ids")) { const readGoals = requireStringArray(source.evidence_goal_ids, "source read Goals"); if (readGoals.some(id => !/^[A-Za-z0-9][A-Za-z0-9._-]{0,159}$/.test(id)) || !readGoals.includes(goal_id)) { throw new EffectRuntimeRequestError("target Goal is outside the channel read scope"); @@ -78,33 +90,42 @@ export function configureSourceRecipient(params: JsonObject): JsonObject { } } const before = agent_id === undefined - ? targets.some(row => row.goal_id === goal_id && row.agent_id === undefined) - : granted(target, targets, blocked); + ? (scope === "all_registered" || targets.some(row => row.goal_id === goal_id && row.agent_id === undefined)) && + !blocked.some(row => row.goal_id === goal_id && row.agent_id === undefined) + : granted(target, targets, blocked, scope); let updatedTargets = targets; let updatedBlocked = blocked; if (grant) { - if (!targets.some(row => matches(row, target))) updatedTargets = [...targets, target]; - if (agent_id !== undefined) updatedBlocked = blocked.filter(row => !matches(target, row)); + if (agent_id !== undefined && blocked.some(row => row.goal_id === goal_id && row.agent_id === undefined)) { + throw new EffectRuntimeRequestError("restore the Goal delivery grant before restoring an individual Agent"); + } + if (scope === "selected" && !targets.some(row => matches(row, target))) updatedTargets = [...targets, target]; + // Regrant only this scope; a Goal regrant preserves individual revocations. + updatedBlocked = blocked.filter(row => !(row.goal_id === goal_id && row.agent_id === agent_id)); } else if (agent_id === undefined) { - // Revoking a Goal also revokes individually enrolled members of that Goal. updatedTargets = targets.filter(row => row.goal_id !== goal_id); - updatedBlocked = blocked.filter(row => row.goal_id !== goal_id); + if (scope === "all_registered" && !blocked.some(row => row.goal_id === goal_id && row.agent_id === undefined)) { + updatedBlocked = [...blocked, target]; + } + // Preserve individual exceptions when the Goal is restored later. } else { updatedTargets = targets.filter(row => !matches(target, row)); - if (updatedTargets.some(row => matches(row, target)) && !blocked.some(row => matches(row, target))) { + // Record the exact revocation even if a Goal block or missing enrollment + // already disables delivery. Restoring that broader scope must not erase it. + if (!blocked.some(row => row.goal_id === goal_id && row.agent_id === agent_id)) { updatedBlocked = [...blocked, target]; } } const changed = JSON.stringify(updatedTargets) !== JSON.stringify(targets) || JSON.stringify(updatedBlocked) !== JSON.stringify(blocked); - // Preserve provider metadata when the semantic recipient set is unchanged. + // Preserve provider metadata when the declared grants and exceptions are unchanged. const updatedSource: JsonObject = { ...source }; if (changed) { updatedSource.targets = updatedTargets; if (updatedBlocked.length) updatedSource.blocked_targets = updatedBlocked; else delete updatedSource.blocked_targets; } - return { source: updatedSource, target, would_change: changed, + return { source: updatedSource, target, local_delivery_scope: scope, would_change: changed, granted_before: before, granted_after: grant, existing_target_count: targets.length, resulting_target_count: updatedTargets.length, includes_future_agents: agent_id === undefined && grant }; diff --git a/loopx/control_plane/subagent_context.ts b/loopx/control_plane/subagent_context.ts index 72f4d23aae..7ca3df64be 100644 --- a/loopx/control_plane/subagent_context.ts +++ b/loopx/control_plane/subagent_context.ts @@ -109,14 +109,14 @@ function boundedNativeChildActivity(value: unknown): JsonObject | null { if (!source || source.schema_version !== "native_subagent_activity_v0" || source.entrypoint_scope !== "host_native_child_tools") return null; const observation = String(source.observation ?? ""); - if (!["unknown", "coordinator_reported"].includes(observation)) return null; + if (!["unknown", "coordinator_reported", "host_observed", "mixed"].includes(observation)) return null; const count = (key: string) => Number.isInteger(source[key]) && Number(source[key]) >= 0 ? Math.min(Number(source[key]), 10_000) : 0; const result: JsonObject = { schema_version: "native_subagent_activity_v0", entrypoint_scope: "host_native_child_tools", observation, - host_attested: false, + host_attested: observation === "host_observed", configured_limit_kind: "upper_bound", configured_limit: count("configured_limit"), attempted_count: count("attempted_count"), diff --git a/loopx/control_plane/turn_driver/codex_cli.py b/loopx/control_plane/turn_driver/codex_cli.py index d2e5ab370c..93224f246c 100644 --- a/loopx/control_plane/turn_driver/codex_cli.py +++ b/loopx/control_plane/turn_driver/codex_cli.py @@ -15,7 +15,7 @@ from .subagent_execution_topology import ( child_execution_receipts_json_schema, ) -from .driver import SUPPORTED_ITERATION_CONTEXT_POLICIES +from ...extensions.codex_native_child import native_child_observer from .codex_sessions import ( CODEX_CLI_SESSION_SCHEMA_VERSION as CODEX_CLI_SESSION_SCHEMA_VERSION, _discard_codex_cli_session, @@ -28,6 +28,7 @@ codex_session_profile_digest, require_codex_session_profile, ) +from .driver import SUPPORTED_ITERATION_CONTEXT_POLICIES from .executor import ( HOST_AGENT_VISION_JSON_MAX_CHARS, HOST_REWARD_MEMORY_REFLECTION_JSON_MAX_CHARS, @@ -721,6 +722,15 @@ def commit() -> None: else: goal_admission.accept_result(commit) + child_observer = native_child_observer(request, runtime_root=runtime_root, lineage=lineage, + registry_path=goal_admission.registry_path if goal_admission is not None else None) + invocation_id = "" + if child_observer is not None: + attempt = request.get("host_attempt") + if isinstance(attempt, bool) or not isinstance(attempt, int) or attempt < 1: + raise ValueError("native child CLI observation requires the durable host attempt") + invocation_id = f"exec:{request['turn_key']}:{attempt}" + with tempfile.TemporaryDirectory(prefix="loopx-turn-codex-") as directory: temporary = Path(directory) schema_path = temporary / "result-schema.json" @@ -757,6 +767,16 @@ def observe_event(line: str) -> None: candidate = codex_cli_event_session_id(event) if candidate and candidate not in observed_session: observed_session.append(candidate) + item = event.get("item") + if (child_observer is not None and observed_session + and event.get("type") == "item.completed" and isinstance(item, Mapping)): + def record_child() -> None: + child_observer.observe(item, session_id=observed_session[0], invocation_id=invocation_id) + + if goal_admission is None: + record_child() + else: + goal_admission.accept_result(record_child) structured, diagnostic = _event_failure_categories(event) if structured: structured_failure_categories.add(structured) diff --git a/loopx/control_plane/turn_driver/codex_operation_host.py b/loopx/control_plane/turn_driver/codex_operation_host.py index 5135d93d5d..1df534c970 100644 --- a/loopx/control_plane/turn_driver/codex_operation_host.py +++ b/loopx/control_plane/turn_driver/codex_operation_host.py @@ -37,6 +37,7 @@ _store_codex_cli_session, load_codex_cli_session, ) +from ...extensions.codex_native_child import native_child_observer from .executor import LOOPX_TURN_HOST_REQUEST_SCHEMA_VERSION from .host_failure import BuiltInHostError @@ -430,7 +431,22 @@ def on_event(kind: str, event: dict[str, Any]) -> None: recovery_kind="resume_session", ) from exc - return session.send( + child_observer = native_child_observer(request, runtime_root=runtime_root, lineage=lineage, + registry_path=goal_admission.registry_path if goal_admission is not None else None) + + def observe_child(item: Mapping[str, Any]) -> None: + if child_observer is None: + return + def record_child() -> None: + child_observer.observe(item, session_id=session.thread_id, + invocation_id=session.current_turn_id) + + if goal_admission is None: + record_child() + else: + goal_admission.accept_result(record_child) + + result = session.send( _prompt(request) + "\nUse loopx_operation for context/pending/prepare/inspect/consume/report. " "Source conversations are not executor identity. context/pending/inspect do not require consumption. " @@ -446,7 +462,9 @@ def on_event(kind: str, event: dict[str, Any]) -> None: if continuations else ""), output_schema=codex_cli_result_schema(request), on_event=on_event, + **({"on_native_item": observe_child} if child_observer is not None else {}), ) + return result except CodexChatAgentError as exc: raise BuiltInHostError( "codex_operation_host_" + exc.error_code, diff --git a/loopx/control_plane/turn_driver/executor.py b/loopx/control_plane/turn_driver/executor.py index dc83e8eee1..4c2903f14a 100644 --- a/loopx/control_plane/turn_driver/executor.py +++ b/loopx/control_plane/turn_driver/executor.py @@ -161,6 +161,10 @@ def build_loopx_turn_host_request(plan: Mapping[str, Any]) -> dict[str, Any]: if isinstance(reward_memory_recall, Mapping): request["reward_memory_recall"] = dict(reward_memory_recall) request.update(subagent.subagent_host_request_projection(plan)) + from ...extensions.codex_native_child import configured_native_child_limit + + if configured_native_child_limit(request) is not None: + request["turn_instance_id"] = transaction.get("turn_instance_id") or turn_key return request @@ -797,6 +801,11 @@ def _host_result_stage( if "typed_result" not in completed_phases: journal["host_attempt_count"] = int(journal.get("host_attempt_count") or 0) + 1 persist_journal(journal) + from ...extensions.codex_native_child import configured_native_child_limit + if configured_native_child_limit(request) is not None: + # Reuse the journal's durable attempt identity; replay never creates + # another identity, and a real host retry always advances it. + request = {**request, "host_attempt": journal["host_attempt_count"]} # The attempt is durable now, so a later restart must not resume this # reservation. Confirmation failure stops before the host starts. if confirm_start is not None: diff --git a/loopx/extensions/codex_native_child.py b/loopx/extensions/codex_native_child.py new file mode 100644 index 0000000000..1dff414114 --- /dev/null +++ b/loopx/extensions/codex_native_child.py @@ -0,0 +1,170 @@ +"""Transient Codex host events adapted to native-child receipts. + +Only opaque identities and typed outcomes reach the existing multi_subagent +log. Prompts, child messages and raw tool output are never retained. +""" +from __future__ import annotations + +import hashlib +import json +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +from ..capabilities.multi_subagent.native_child_receipts import ( + _load_native_child_events, record_native_child, +) + + +def configured_native_child_limit(request: Mapping[str, Any]) -> int | None: + envelope = request.get("turn_envelope") + context = envelope.get("agent_context") if isinstance(envelope, Mapping) else None + contributions = context.get("contributions") if isinstance(context, Mapping) else None + if not isinstance(contributions, list): + return None + for contribution in contributions: + if not isinstance(contribution, Mapping) or contribution.get("capability_id") != "multi_subagent": + continue + facts = contribution.get("facts") + count = facts.get("max_children") if isinstance(facts, Mapping) else None + if isinstance(count, int) and not isinstance(count, bool) and count > 0: + return count + return None + + +class CodexNativeChildObserver: + """Bound to one owned host invocation and one admitted LoopX Turn. + + The public recorder cannot select host provenance. Codex exec and app-server + use different field casing; both are normalized here at the provider seam. + Failed collab items lack a typed capacity error, so they remain host_failed. + A host retry is recorded as observed fact, never authorized by this adapter. + """ + + def __init__(self, *, runtime_root: Path, lineage: Mapping[str, str], + turn_instance_id: str, configured_limit: int, + goal_ref: Mapping[str, Any] | None = None, registry_path: Path | None = None): + self.runtime_root = runtime_root + self.lineage = lineage + self.turn_instance_id = turn_instance_id + self.configured_limit = configured_limit + self.goal_ref = goal_ref + self.registry_path = registry_path + + def _record(self, *, stage: str, **record: Any) -> None: + record_native_child( + runtime_root=self.runtime_root, goal_id=self.lineage["goal_id"], + agent_id=self.lineage["agent_id"], turn_instance_id=self.turn_instance_id, + configured_limit=self.configured_limit, stage=stage, + entrypoint_id="codex_native_tools" if stage == "decision" else None, + execute=True, _host_observed=True, goal_ref=self.goal_ref, + registry_path=self.registry_path, **record, + ) + + def _restore_operation(self, child: str, *, wait_ref: str) -> str | None: + # Restore the latest successful host decision for this opaque child, + # including followups and rows outside the presentation window. The + # existing log supplies order and exact Goal/agent/Turn ownership. + child_ref = "codex-child-" + hashlib.sha256(child.encode()).hexdigest()[:32] + legacy_spawn = "codex-" + hashlib.sha256(child.encode()).hexdigest()[:32] + events = _load_native_child_events( + self.runtime_root, goal_id=self.lineage["goal_id"], agent_id=self.lineage["agent_id"], + turn_instance_id=self.turn_instance_id, goal_ref=self.goal_ref, + registry_path=self.registry_path, + ) + # The first terminal observation owns this wait identity permanently. + # Replayed waits must never reinterpret a later child association. + for event in events: + details = event.get("details") + if (event.get("event_kind") == "native_child_result" + and isinstance(details, Mapping) + and details.get("observation_source") == "host_observed" + and details.get("host_wait_ref") == wait_ref): + return str(event["case_id"]) + for event in reversed(events): + details = event.get("details") + if not isinstance(details, Mapping): + continue + if (event.get("event_kind") == "native_child_decision" + and details.get("operation") in {"spawn", "followup"} + and details.get("outcome") == "started" + and details.get("observation_source") == "host_observed" + and details.get("entrypoint_id") == "codex_native_tools"): + if (details.get("host_child_ref_" + child_ref) is True + or (details.get("operation") == "spawn" + and event.get("case_id") == legacy_spawn)): + return str(event["case_id"]) + return None + + def observe(self, item: Mapping[str, Any], *, session_id: str, invocation_id: str) -> None: + item_type = item.get("type") + if item_type not in {"collab_tool_call", "collabAgentToolCall"}: + return + snake = item_type == "collab_tool_call" + sender = item.get("sender_thread_id" if snake else "senderThreadId") + status = item.get("status") + native_id = item.get("id") + if sender != session_id or not isinstance(native_id, str) or not native_id or status not in {"completed", "failed"}: + return + tool = {"spawn_agent": "spawn", "spawnAgent": "spawn", "send_input": "followup", + "sendInput": "followup", "resumeAgent": "followup", "wait": "wait"}.get(item.get("tool")) + if tool is None: + return + receivers = item.get("receiver_thread_ids" if snake else "receiverThreadIds") + decision_id: str | None = None + if tool != "wait": + started = status == "completed" and isinstance(receivers, list) and bool(receivers) + if started and any(not isinstance(child, str) or not child for child in receivers): + return + # Successful spawn identity survives host event replay/restart; host + # exec display-item counters alone are not globally unique. + if not invocation_id: + raise ValueError("native child decision requires its owned host invocation") + identity = (receivers[0] if tool == "spawn" and started else + json.dumps([session_id, invocation_id, native_id], separators=(",", ":"))) + operation_id = "codex-" + hashlib.sha256(identity.encode()).hexdigest()[:32] + self._record(stage="decision", operation_id=operation_id, operation=tool, + outcome="started" if started else "host_failed", + **({"reason_code": "host_failed"} if not started else { + "_host_child_refs": sorted({"codex-child-" + hashlib.sha256(child.encode()).hexdigest()[:32] + for child in receivers})})) + if started: + decision_id = operation_id + states = item.get("agents_states" if snake else "agentsStates") + if not isinstance(states, Mapping): + return + for child, state in states.items(): + if not isinstance(child, str) or not child or not isinstance(state, Mapping): + continue + # A spawn/followup snapshot belongs to that stable decision, even + # when replayed after later work. A wait first restores its own + # consumed binding, then falls back to the latest child operation. + if tool != "wait" and (not isinstance(receivers, list) or child not in receivers): + continue + outcome = {"completed": "completed", "errored": "failed", "shutdown": "cancelled"}.get(state.get("status")) + if outcome is None: + continue + wait_ref = None + if tool == "wait": + if not invocation_id: + raise ValueError("native child wait requires its owned host invocation") + identity = json.dumps([session_id, invocation_id, native_id, child], separators=(",", ":")) + wait_ref = "codex-wait-" + hashlib.sha256(identity.encode()).hexdigest()[:32] + operation_id = self._restore_operation(child, wait_ref=wait_ref) if wait_ref else decision_id + if operation_id is not None: + self._record(stage="result", operation_id=operation_id, outcome=outcome, + _host_wait_ref=wait_ref) + + +def native_child_observer(request: Mapping[str, Any], *, runtime_root: Path, + lineage: Mapping[str, str], + registry_path: Path | None = None) -> CodexNativeChildObserver | None: + limit = configured_native_child_limit(request) + turn = request.get("turn_instance_id") + if limit is None or not isinstance(turn, str) or not turn: + return None + goal_ref = request.get("goal_ref") + return CodexNativeChildObserver(runtime_root=runtime_root, lineage=lineage, + turn_instance_id=turn, configured_limit=limit, + goal_ref=goal_ref if isinstance(goal_ref, Mapping) else None, + registry_path=registry_path) diff --git a/loopx/extensions/public_github_research.py b/loopx/extensions/public_github_research.py new file mode 100644 index 0000000000..6903c9d32e --- /dev/null +++ b/loopx/extensions/public_github_research.py @@ -0,0 +1,131 @@ +"""Explicit public GitHub source-inspection method; no credentials or raw persistence. + +This bundled provider owns HTTP access. External evidence's TypeScript contract +owns exact-plan validation, parent admission and retirement. +""" +from __future__ import annotations + +import hashlib +import json +import re +from datetime import datetime, timezone +from urllib.error import HTTPError, URLError +from urllib.parse import quote, unquote, urlsplit +from urllib.request import HTTPRedirectHandler, ProxyHandler, Request, build_opener + +PROVIDER_ID = "method:public-github" +MAX_SOURCE_BYTES = 1_000_000 +SOURCE = re.compile(r"/([A-Za-z0-9_.-]+)/([A-Za-z0-9_.-]+)/blob/([0-9a-f]{40})/(.+)") + + +class _NoRedirect(HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +def _read(url: str) -> bytes: + # No provider credentials, cookies, user-configured proxy credentials or + # caller-selected origins enter the request. Redirects fail closed. + request = Request(url, headers={"User-Agent": "LoopX-public-evidence", + "Accept": "application/vnd.github+json" if url.startswith("https://api.github.com/") else "text/plain"}) + with build_opener(ProxyHandler({}), _NoRedirect()).open(request, timeout=15) as response: + result = response.read(MAX_SOURCE_BYTES + 1) + if len(result) > MAX_SOURCE_BYTES: + raise ValueError("public source exceeds the bounded read limit") + return result + + +def _source(ref: str) -> tuple[str, str, str, str]: + parsed = urlsplit(ref) + match = SOURCE.fullmatch(parsed.path) + if (parsed.scheme != "https" or parsed.netloc != "github.com" or parsed.query + or parsed.fragment or match is None): + raise ValueError("public GitHub sources require https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/PATH") + owner, repo, revision, path = match.groups() + decoded = unquote(path) + if (any(part in {"", ".", ".."} for part in decoded.split("/")) or "\\" in decoded + or any(ord(char) < 32 for char in decoded) or "%" in decoded + or owner in {".", ".."} or repo in {".", ".."}): + raise ValueError("public GitHub source path is invalid") + return owner, repo, revision, decoded + + +def _public_repository(owner: str, repo: str) -> None: + value = json.loads(_read(f"https://api.github.com/repos/{owner}/{repo}")) + if not isinstance(value, dict) or value.get("private") is not False: + raise ValueError("provider only reads currently public GitHub repositories") + + +def inspect_provider(source_refs: list[str]) -> dict: + """Current opt-in readiness, not registry inventory or execution proof.""" + sources = [_source(ref) for ref in source_refs] + if not sources: + raise ValueError("public GitHub inspection requires at least one pinned source") + reason = None + try: + for owner, repo in sorted({(source[0], source[1]) for source in sources}): + _public_repository(owner, repo) + except (OSError, ValueError, URLError): + reason = "public_github_readiness_unavailable" + return {"provider_id": PROVIDER_ID, "provider_kind": "method", + "protocol": "external_evidence_research_v0", "declared": True, + "installed": True, "enabled": True, "ready": reason is None, + "unavailable_reason": reason} + + +def execute_public_github(plan: dict) -> dict: + """Read only exact-plan sources, with fresh public-visibility checks. + + Caller validates the canonical plan through the typed owner before entry. + Findings are retrieval and literal-match facts, not autonomous conclusions. + """ + selected = plan["selected_provider"] + if selected["provider_id"] != PROVIDER_ID or selected["provider_kind"] != "method": + raise ValueError("this executor requires the selected public GitHub method") + request = plan["request"] + refs = request.get("source_refs", []) + sources = [_source(ref) for ref in refs] + if not sources or len(sources) > 8: + raise ValueError("public GitHub execution requires one to eight pinned sources") + terms = request.get("search_terms", []) + records, failures = [], [] + now = datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") + for index, (ref, (owner, repo, revision, path)) in enumerate(zip(refs, sources), 1): + try: + _public_repository(owner, repo) + raw = _read(f"https://raw.githubusercontent.com/{owner}/{repo}/{revision}/{quote(path, safe='/')}") + text = raw.decode("utf-8") + if "\x00" in text: + raise ValueError("source is not UTF-8 text") + if not text.strip(): + failures.append(f"Source {index}: empty source") + continue + matches = [] + for term in terms: + lines = [str(index) for index, line in enumerate(text.splitlines(), 1) if term in line] + matches.append(f"Literal term {json.dumps(term)}: " + + ("observed at lines " + ", ".join(lines[:16]) if lines else "not observed") + + (" (additional matches omitted)" if len(lines) > 16 else "")) + finding = f"Read pinned file {path}; {len(raw)} UTF-8 bytes. " + " ".join(matches) + if len(finding) > 4096: + finding = finding[:4040] + " (additional match metadata omitted)" + records.append({"source_ref": ref, "source_family": "github_repository_file", + "basis": "observed", "finding": finding, + "limitation": "Retrieval/literal matches only; no semantic conclusion, execution test or completeness claim.", + "publication_date": None, "accessed_at": now, + "content_digest": "sha256:" + hashlib.sha256(raw).hexdigest()}) + except HTTPError as error: + failures.append(f"Source {index}: HTTP {error.code}") + except (OSError, ValueError, UnicodeError, URLError): + failures.append(f"Source {index}: source read unavailable") + status = "succeeded" if records else "no_evidence" if all("empty source" in item for item in failures) else "failed" + receipt = {"schema_version": "loopx_external_evidence_receipt_v0", + "plan_id": plan["plan_id"], "request_id": request["request_id"], + "provider_id": PROVIDER_ID, "provider_kind": "method", "status": status, + "sources": records, "summary": f"Read {len(records)} of {len(refs)} requested public sources.", + "limitations": ["Original-source fallback remains available; parent admission is required.", *failures], + "completed_at": now} + return {"receipt": receipt, "execution": {"provider_id": PROVIDER_ID, + "source_reads_observed": len(records), "requested_source_count": len(refs), + "raw_content_persisted": False, "credentials_used": False, + "automatic_admission": False, "evidence_coverage_observed": False}} diff --git a/loopx/chat_goal_ownership_api.py b/loopx/presentation/goal_ownership_api.py similarity index 99% rename from loopx/chat_goal_ownership_api.py rename to loopx/presentation/goal_ownership_api.py index 83a1e7e387..ae1849056a 100644 --- a/loopx/chat_goal_ownership_api.py +++ b/loopx/presentation/goal_ownership_api.py @@ -11,7 +11,7 @@ from urllib.parse import parse_qs, urlparse from uuid import uuid4 -from .control_plane.todos.provider_handoff_mode import ( +from ..control_plane.todos.provider_handoff_mode import ( migrate_registered_handoff_mode, read_canonical_handoff_mode, ) diff --git a/loopx/presentation/renderers/status_markdown.py b/loopx/presentation/renderers/status_markdown.py index 7ab021b9a5..768a5588c0 100644 --- a/loopx/presentation/renderers/status_markdown.py +++ b/loopx/presentation/renderers/status_markdown.py @@ -1275,11 +1275,12 @@ def _append_project_asset_runtime_policy_markdown( if isinstance(project_asset.get("native_child_activity"), dict) else {} ) - if native_child_activity.get("observation") == "coordinator_reported": + if native_child_activity.get("observation") in {"coordinator_reported", "host_observed", "mixed"}: lines.append( " - native_child_activity: " f"turn={markdown_scalar(native_child_activity.get('turn_instance_id'))} " - "source=coordinator_reported host_attested=false " + f"source={native_child_activity.get('observation')} " + f"host_attested={str(native_child_activity.get('host_attested')).lower()} " f"configured_max={native_child_activity.get('configured_limit')} " f"starts={native_child_activity.get('launched_count')} " f"skips={native_child_activity.get('skipped_count')} " diff --git a/skills/loopx-pr-review/SKILL.md b/skills/loopx-pr-review/SKILL.md index 7b693670b0..395b887574 100644 --- a/skills/loopx-pr-review/SKILL.md +++ b/skills/loopx-pr-review/SKILL.md @@ -90,7 +90,7 @@ When `review_action_kind` is null, the row stays in `pull_requests` inventory bu Each PR needs independent evidence and a standalone card; a queue table is a preface only. -For managed review, pass `--goal-id GOAL` and follow the packet’s resolved `wait_for_ci`: false means never fetch, poll, or wait for CI; true retains CI observation. Apply the packet's `validation_matrix.failure_attribution` before treating a red required check as a PR blocker. An independently verified unchanged baseline failure or external outage can hold merge readiness without forcing `REQUEST_CHANGES` on an unrelated PR; missing attribution or missing affected-invariant coverage still blocks approval. Configure one Goal with `configure-goal --goal-id GOAL --no-pr-review-wait-for-ci --execute`; clear with `--clear-pr-review-configuration --execute`. +For managed review, pass `--goal-id GOAL` and follow the packet’s resolved `wait_for_ci`: false means never fetch, poll, or wait for CI; true retains available CI observation and the merge gate's CI policy. It does not require every CI job to finish or succeed before `APPROVE` when independent current evidence covers the changed invariants. Follow `validation_matrix.validation_source`: `required` marks decisive review evidence, not branch protection; record merely pending remote jobs as separate diagnostic rows, and keep an invariant unverified when CI is its only decisive coverage. Pending CI alone is never a `REQUEST_CHANGES` reason. Apply `validation_matrix.failure_attribution`, including its `evidence_scope`, to current required checks. An independently attributed unchanged baseline failure or external outage can hold merge readiness without forcing `REQUEST_CHANGES` on an unrelated PR. Preserve earlier failures and the current evidence that supersedes them in existing evidence fields; historical root-cause completeness alone is not an approval gate. Current unattributed failures, material instability and missing affected-invariant coverage still block; selecting one successful rerun does not resolve them. Configure one Goal with `configure-goal --goal-id GOAL --no-pr-review-wait-for-ci --execute`; clear with `--clear-pr-review-configuration --execute`. ## Publish And Read Back For an open PR, publish validated actionable findings by default unless the user diff --git a/tests/architecture/test_content_digest_single_owner.py b/tests/architecture/test_content_digest_single_owner.py index d3c40bf38f..c454d42233 100644 --- a/tests/architecture/test_content_digest_single_owner.py +++ b/tests/architecture/test_content_digest_single_owner.py @@ -161,6 +161,7 @@ "loopx.capabilities.benchmark_toolkit.runtime_continuity", "loopx.capabilities.benchmark_toolkit.study_projection", "loopx.capabilities.content_ops.item_lifecycle", + "loopx.capabilities.deep_research.runtime", "loopx.capabilities.issue_fix.outcome_projection", "loopx.capabilities.issue_fix.reviewer_notification", "loopx.capabilities.machine_configuration.store", diff --git a/tests/capabilities/test_capability_extension_registry.py b/tests/capabilities/test_capability_extension_registry.py index 29aea2a1e1..67e30928e1 100644 --- a/tests/capabilities/test_capability_extension_registry.py +++ b/tests/capabilities/test_capability_extension_registry.py @@ -46,6 +46,7 @@ "performance-diagnosis", "reliability-diagnostics", "progress-review-sentinel", + "goal-capability-organization", ] diff --git a/tests/capabilities/test_codex_native_child_receipts.py b/tests/capabilities/test_codex_native_child_receipts.py new file mode 100644 index 0000000000..178f866434 --- /dev/null +++ b/tests/capabilities/test_codex_native_child_receipts.py @@ -0,0 +1,431 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from loopx.extensions.codex_native_child import ( + configured_native_child_limit, CodexNativeChildObserver, native_child_observer, +) +from loopx.capabilities.multi_subagent.native_child_receipts import load_native_child_activity, record_native_child +from loopx.rollout_event_log import load_rollout_events, rollout_event_log_path +from tests.capabilities.test_native_child_receipts import _admit, GOAL, AGENT, TURN + + +def _turn(): + return {"id": "host-turn-1", "itemsView": "full", "status": "completed", "items": [ + {"type": "collabAgentToolCall", "id": "call-1", "senderThreadId": "parent-1", + "tool": "spawnAgent", "status": "completed", "receiverThreadIds": ["child-1"], + "prompt": "private child instructions", "agentsStates": {"child-1": {"status": "running"}}}, + {"type": "collabAgentToolCall", "id": "wait-1", "senderThreadId": "parent-1", + "tool": "wait", "status": "completed", "agentsStates": { + "child-1": {"status": "completed", "message": "private child result"}}}, + ]} + + +def _observe(root, items, session_id="parent-1"): + observer = CodexNativeChildObserver(runtime_root=root, + lineage={"goal_id": GOAL, "agent_id": AGENT}, + turn_instance_id=TURN, configured_limit=3) + for item in items: + observer.observe(item, session_id=session_id, invocation_id="host-turn-1") + + +def test_host_spawn_result_parent_review_and_restart(tmp_path: Path): + _admit(tmp_path) + _observe(tmp_path, _turn()["items"]) + _observe(tmp_path, _turn()["items"]) + activity = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert activity["observation"] == "host_observed" + assert activity["host_attested"] is True + assert activity["launched_count"] == 1 + assert activity["parent_accepted_count"] == 0 + [operation] = activity["operations"] + assert operation["result"] == "completed" + record_native_child(runtime_root=tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3, operation_id=operation["operation_id"], + stage="review", outcome="accepted", evidence_ref="evidence-1", + validation_ref="validation-1", execute=True) + readback = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert readback["parent_accepted_count"] == 1 + assert readback["host_attested"] is True + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + assert len(events) == 4 + assert "private child" not in json.dumps(events) + assert all(event.get("event_kind") != "quota_spend" for event in events) + + +@pytest.mark.parametrize("items", [[], [{"type": "agentMessage", "text": "I spawned three children"}], + [{**_turn()["items"][0], "status": "inProgress"}], + [{**_turn()["items"][0], "senderThreadId": "historical-parent"}]]) +def test_missing_or_unrelated_host_events_stay_unknown(tmp_path: Path, items): + _admit(tmp_path) + _observe(tmp_path, items) + activity = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert activity["observation"] == "unknown" + assert activity["launched_count"] == 0 + + +def test_exec_casing_and_display_counter_replay_do_not_duplicate_spawn(tmp_path: Path): + _admit(tmp_path) + snake = {"type": "collab_tool_call", "id": "item_1", "sender_thread_id": "parent-1", + "tool": "spawn_agent", "status": "completed", "receiver_thread_ids": ["child-1"], + "agents_states": {"child-1": {"status": "completed", "message": "private content"}}} + _observe(tmp_path, [snake, {**snake, "id": "item_7"}]) + activity = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert activity["host_attested"] is True + assert activity["launched_count"] == 1 + assert activity["operations"][0]["result"] == "completed" + assert len(load_rollout_events(rollout_event_log_path(tmp_path, GOAL))) == 3 + + +def test_host_failure_does_not_infer_capacity_from_prose(tmp_path: Path): + _admit(tmp_path) + failed = {**_turn()["items"][0], "status": "failed", "receiverThreadIds": [], + "agentsStates": {"child-1": {"status": "errored", "message": "agent_thread_limit_reached"}}} + _observe(tmp_path, [failed]) + activity = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert activity["host_attested"] is True + assert activity["host_failed_count"] == 1 + assert activity["capacity_rejected_count"] == 0 + assert activity["retry_same_turn"] is False + + +def test_feature_off_has_no_observer_and_reports_cannot_attest_reviews(tmp_path: Path): + assert native_child_observer({"turn_envelope": {}}, runtime_root=tmp_path, + lineage={"goal_id": GOAL, "agent_id": AGENT}) is None + assert configured_native_child_limit({"turn_envelope": {}}) is None + assert configured_native_child_limit({"turn_envelope": {"agent_context": { + "contributions": [{"capability_id": "multi_subagent", "facts": {"max_children": 0}}]}}}) is None + with pytest.raises(ValueError, match="cannot attest"): + record_native_child(runtime_root=tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3, operation_id="review-1", stage="review", + outcome="accepted", evidence_ref="evidence-1", validation_ref="validation-1", + execute=True, _host_observed=True) + + +def test_cli_host_collects_native_items_before_returning_parent_result(tmp_path: Path, monkeypatch): + import sys + from loopx.control_plane.turn_driver import codex_cli + from tests.test_loopx_turn_codex_cli import _request + + _admit(tmp_path) + request = _request() + request["turn_instance_id"] = TURN + request["host_attempt"] = 1 + request["turn_envelope"].update(goal_id=GOAL, agent_id=AGENT, agent_context={ + "contributions": [{"capability_id": "multi_subagent", "facts": {"max_children": 3}}]}) + request["turn_envelope"]["action"]["selected_todo"]["todo_id"] = "todo_native_1" + + def host(command, **kwargs): + kwargs["on_stdout"](json.dumps({"type": "thread.started", "thread_id": "parent-1"}) + "\n") + for item in _turn()["items"]: + kwargs["on_stdout"](json.dumps({"type": "item.completed", "item": item}) + "\n") + Path(command[command.index("--output-last-message") + 1]).write_text(json.dumps({"parent_work": "preserved"})) + return {"returncode": 0, "outcome": "exited", "output_complete": True} + + monkeypatch.setattr(codex_cli, "run_host_process", host) + assert codex_cli.run_codex_cli_host(request, runtime_root=tmp_path, project=tmp_path, + codex_bin=sys.executable) == {"parent_work": "preserved"} + activity = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert activity["host_attested"] is True + assert activity["launched_count"] == 1 + assert activity["operations"][0]["result"] == "completed" + + +def _real_cli_calls(tmp_path, monkeypatch, batches, *, host_attempts=None): + import sys + from loopx.control_plane.turn_driver import codex_cli + from tests.test_loopx_turn_codex_cli import _request + + script = tmp_path / "host.py" + script.write_text(""" +import json, sys +from pathlib import Path +sys.stdin.read() +print(json.dumps({"type": "thread.started", "thread_id": "parent-1"}), flush=True) +for item in json.loads(Path(sys.argv[1]).read_text(encoding="utf-8")): + print(json.dumps({"type": "item.completed", "item": item}), flush=True) +Path(sys.argv[2]).write_text(json.dumps({"parent_work": "preserved"}), encoding="utf-8") +""", encoding="utf-8") + event_file = tmp_path / "events.json" + sessions = [] + + def command(**kwargs): + sessions.append(kwargs["session_id"]) + return [sys.executable, str(script), str(event_file), str(kwargs["output_path"])] + + # Replace only executable selection; run the real process, stream parser, + # session store, receipt admission and durable readback. + monkeypatch.setattr(codex_cli, "_codex_command", command) + for attempt, items in enumerate(batches, 1): + request = _request(session_action="start_new" if attempt == 1 else "resume") + request["turn_instance_id"] = TURN + request["host_attempt"] = host_attempts[attempt - 1] if host_attempts else attempt + request["turn_envelope"].update(goal_id=GOAL, agent_id=AGENT, agent_context={ + "contributions": [{"capability_id": "multi_subagent", "facts": {"max_children": 3}}]}) + request["turn_envelope"]["action"]["selected_todo"]["todo_id"] = "todo_native_1" + event_file.write_text(json.dumps(items), encoding="utf-8") + assert codex_cli.run_codex_cli_host(request, runtime_root=tmp_path, project=tmp_path, + codex_bin=sys.executable) == {"parent_work": "preserved"} + assert sessions == [None] + ["parent-1"] * (len(batches) - 1) + return load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + + +def test_real_cli_resume_wait_only_completes_original_spawn(tmp_path, monkeypatch): + _admit(tmp_path) + spawn, wait = _turn()["items"] + activity = _real_cli_calls(tmp_path, monkeypatch, [[spawn], [wait, wait]]) + assert activity["launched_count"] == 1 + assert activity["operation_count"] == 1 + assert activity["operations"][0]["result"] == "completed" + assert activity["quota_spend_slots"] == 0 + assert len(load_rollout_events(rollout_event_log_path(tmp_path, GOAL))) == 3 + + +def test_real_cli_resume_counter_reuse_and_exact_replay(tmp_path, monkeypatch): + _admit(tmp_path) + followup = {"type": "collab_tool_call", "id": "item_0", "sender_thread_id": "parent-1", + "tool": "send_input", "status": "completed", "receiver_thread_ids": ["child-1"]} + other_child = {**followup, "receiver_thread_ids": ["child-2"]} + failed = {**followup, "tool": "spawn_agent", "status": "failed", "receiver_thread_ids": []} + activity = _real_cli_calls(tmp_path, monkeypatch, [ + [followup, followup], [other_child, other_child], [followup, followup], + [failed, failed], [failed, failed]]) + assert activity["operation_count"] == 5 + assert activity["attempted_count"] == 5 + assert activity["host_failed_count"] == 2 + assert activity["launched_count"] == 0 + assert activity["quota_spend_slots"] == 0 + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + assert len(events) == 6 + assert "private child" not in json.dumps(events) + + +def test_resume_restores_spawn_outside_the_visible_window(tmp_path): + _admit(tmp_path) + spawn, wait = _turn()["items"] + items = [{**spawn, "receiverThreadIds": [f"child-{i}"], "agentsStates": {}} + for i in range(1, 11)] + _observe(tmp_path, items) + _observe(tmp_path, [wait]) + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + assert sum(event["event_kind"] == "native_child_decision" for event in events) == 10 + assert sum(event["event_kind"] == "native_child_result" for event in events) == 1 + + +def test_resume_does_not_attest_coordinator_reported_or_unknown_children(tmp_path): + import hashlib + _admit(tmp_path) + operation = "codex-" + hashlib.sha256(b"child-1").hexdigest()[:32] + record_native_child(runtime_root=tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3, operation_id=operation, stage="decision", + operation="spawn", outcome="started", entrypoint_id="codex_native_tools", execute=True) + _observe(tmp_path, [_turn()["items"][1]]) + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + assert not any(event["event_kind"] == "native_child_result" for event in events) + + +def test_restart_replay_uses_the_original_invocation_binding(tmp_path): + _admit(tmp_path) + followup = {**_turn()["items"][0], "tool": "sendInput", "agentsStates": {}} + _observe(tmp_path, [followup]) + _observe(tmp_path, [followup]) + assert len(load_rollout_events(rollout_event_log_path(tmp_path, GOAL))) == 2 + + +@pytest.mark.parametrize('with_spawn', [False, True]) +@pytest.mark.parametrize('terminal_status,expected', [('completed', 'completed'), ('errored', 'failed')]) +def test_real_cli_resumed_followup_owns_its_result(tmp_path, monkeypatch, with_spawn, terminal_status, expected): + _admit(tmp_path) + spawn, wait = _turn()['items'] + followup = {**spawn, 'id': 'item_0', 'tool': 'sendInput', 'agentsStates': {}} + terminal = {**wait, 'agentsStates': {'child-1': {'status': terminal_status}}} + batches = ([[spawn, wait]] if with_spawn else []) + [[followup], [terminal, terminal], [terminal]] + activity = _real_cli_calls(tmp_path, monkeypatch, batches, + host_attempts=[*range(1, len(batches)), len(batches) - 1]) + followups = [row for row in activity['operations'] if row['operation'] == 'followup'] + assert len(followups) == 1 and followups[0]['result'] == expected + assert activity['operation_count'] == 1 + int(with_spawn) + assert activity['launched_count'] == int(with_spawn) + assert activity['quota_spend_slots'] == 0 + if with_spawn: + original = next(row for row in activity['operations'] if row['operation'] == 'spawn') + assert original['result'] == 'completed' + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + assert sum(event['event_kind'] == 'native_child_result' for event in events) == 1 + int(with_spawn) + assert '"child-1"' not in json.dumps(events) and 'private child' not in json.dumps(events) + + +def test_real_cli_consecutive_followups_restore_the_latest_owned_operation(tmp_path, monkeypatch): + _admit(tmp_path) + spawn, wait = _turn()['items'] + followup = {**spawn, 'id': 'item_0', 'tool': 'sendInput', 'agentsStates': {}} + failed = {**wait, 'agentsStates': {'child-1': {'status': 'errored'}}} + activity = _real_cli_calls(tmp_path, monkeypatch, + [[spawn, wait], [followup], [wait], [followup], [failed, failed], [failed]]) + followups = [row for row in activity['operations'] if row['operation'] == 'followup'] + assert len(followups) == 2 + assert {row['result'] for row in followups} == {'completed', 'failed'} + assert activity['operation_count'] == 3 and activity['launched_count'] == 1 + assert activity['quota_spend_slots'] == 0 + + +@pytest.mark.parametrize('host_observed,stage,outcome', [(False, 'decision', 'started'), (True, 'result', 'completed')]) +def test_child_correlation_cannot_attest_a_report_or_result(tmp_path, host_observed, stage, outcome): + _admit(tmp_path) + with pytest.raises(ValueError, match='child correlation requires'): + record_native_child(runtime_root=tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3, operation_id='correlation-1', + stage=stage, outcome=outcome, operation='followup' if stage == 'decision' else None, + entrypoint_id='codex_native_tools' if stage == 'decision' else None, + execute=True, _host_observed=host_observed, _host_child_refs=['codex-child-opaque']) + assert not any(event['event_kind'] == 'native_child_decision' + for event in load_rollout_events(rollout_event_log_path(tmp_path, GOAL))) + + + +@pytest.mark.parametrize("snake", [False, True]) +@pytest.mark.parametrize("new_wait", [False, True]) +def test_real_cli_old_spawn_snapshot_cannot_complete_a_later_followup( + tmp_path, monkeypatch, snake, new_wait, +): + _admit(tmp_path) + spawn, wait = _turn()["items"] + spawn = {**spawn, "agentsStates": {"child-1": {"status": "completed"}}} + followup = {**spawn, "id": "item_0", "tool": "sendInput", + "agentsStates": {"child-1": {"status": "running"}}} + if snake: + def exec_item(item): + return {"type": "collab_tool_call", "id": item["id"], + "sender_thread_id": item["senderThreadId"], + "tool": {"spawnAgent": "spawn_agent", "sendInput": "send_input", "wait": "wait"}[item["tool"]], + "status": item["status"], "receiver_thread_ids": item.get("receiverThreadIds", []), + "agents_states": item["agentsStates"]} + spawn, followup, wait = map(exec_item, (spawn, followup, wait)) + # The third invocation replays only the old spawn's terminal snapshot. It + # contains no new wait or followup result and cannot prove future work done. + batches = [[spawn], [followup], [spawn]] + ([[wait, wait]] if new_wait else []) + activity = _real_cli_calls(tmp_path, monkeypatch, batches) + assert activity["operation_count"] == 2 and activity["launched_count"] == 1 + original = next(row for row in activity["operations"] if row["operation"] == "spawn") + later = next(row for row in activity["operations"] if row["operation"] == "followup") + assert original["result"] == "completed" + assert later.get("result") == ("completed" if new_wait else None) + assert activity["parent_accepted_count"] == 0 and activity["quota_spend_slots"] == 0 + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + assert sum(row["event_kind"] == "native_child_result" for row in events) == 1 + int(new_wait) + + +@pytest.mark.parametrize("tool", ["spawnAgent", "sendInput"]) +def test_nonwait_terminal_snapshot_ignores_unrelated_receivers(tmp_path, tool): + _admit(tmp_path) + spawn = _turn()["items"][0] + other = {**spawn, "receiverThreadIds": ["child-2"], "agentsStates": {}} + unrelated_snapshot = {**spawn, "tool": tool, + "agentsStates": {"child-2": {"status": "completed"}}} + _observe(tmp_path, [other, unrelated_snapshot]) + activity = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert activity["operation_count"] == 2 + assert all(row.get("result") is None for row in activity["operations"]) + + +@pytest.mark.parametrize("snake", [False, True]) +@pytest.mark.parametrize("restart", [False, True]) +def test_real_cli_consumed_wait_replay_cannot_complete_a_later_followup( + tmp_path, monkeypatch, snake, restart, +): + _admit(tmp_path) + spawn, wait = _turn()["items"] + followup = {**spawn, "id": "followup-1", "tool": "sendInput", + "agentsStates": {"child-1": {"status": "running"}}} + if snake: + def exec_item(item): + return {"type": "collab_tool_call", "id": item["id"], + "sender_thread_id": item["senderThreadId"], + "tool": {"spawnAgent": "spawn_agent", "sendInput": "send_input", "wait": "wait"}[item["tool"]], + "status": item["status"], "receiver_thread_ids": item.get("receiverThreadIds", []), + "agents_states": item["agentsStates"]} + spawn, followup, wait = map(exec_item, (spawn, followup, wait)) + # Distinct native IDs within one invocation exclude counter reuse. Restart + # must preserve the original attempt for an exact replay of the old wait. + batches = [[spawn, wait], [followup], [wait]] if restart else [[spawn, wait, followup, wait]] + activity = _real_cli_calls(tmp_path, monkeypatch, batches, + host_attempts=[1, 2, 1] if restart else [1]) + original = next(row for row in activity["operations"] if row["operation"] == "spawn") + later = next(row for row in activity["operations"] if row["operation"] == "followup") + assert original["result"] == "completed" + assert later.get("result") is None + assert activity["launched_count"] == 1 and activity["operation_count"] == 2 + assert activity["parent_accepted_count"] == 0 and activity["quota_spend_slots"] == 0 + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + assert sum(event["event_kind"] == "native_child_result" for event in events) == 1 + assert '"child-1"' not in json.dumps(events) and "private child" not in json.dumps(events) + + +@pytest.mark.parametrize("spawn_completed", [False, True]) +def test_distinct_waits_keep_their_first_binding_and_reject_changed_outcomes(tmp_path, spawn_completed): + _admit(tmp_path) + spawn, wait = _turn()["items"] + if spawn_completed: + spawn = {**spawn, "agentsStates": {"child-1": {"status": "completed"}}} + followup = {**spawn, "id": "followup-1", "tool": "sendInput", "agentsStates": {}} + fresh_wait = {**wait, "id": "wait-2"} + snapshot = {**spawn, "agentsStates": {"child-1": {"status": "completed"}}} + _observe(tmp_path, [spawn, wait, followup, snapshot, wait]) + pending = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + later = next(row for row in pending["operations"] if row["operation"] == "followup") + assert later.get("result") is None + _observe(tmp_path, [fresh_wait, fresh_wait]) + completed = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert all(row["result"] == "completed" for row in completed["operations"]) + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + with pytest.raises(ValueError, match="conflicting"): + _observe(tmp_path, [{**wait, "agentsStates": {"child-1": {"status": "errored"}}}]) + assert load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) == events + assert completed["parent_accepted_count"] == 0 and completed["quota_spend_slots"] == 0 + + +def test_wait_binding_is_per_child_and_cannot_be_reassigned(tmp_path): + _admit(tmp_path) + spawn, wait = _turn()["items"] + other = {**spawn, "id": "spawn-2", "receiverThreadIds": ["child-2"], "agentsStates": {}} + both = {**wait, "agentsStates": {"child-1": {"status": "completed"}, + "child-2": {"status": "shutdown"}}} + followup = {**spawn, "id": "followup-1", "tool": "sendInput", "agentsStates": {}} + _observe(tmp_path, [spawn, other, both, followup, both]) + activity = load_native_child_activity(tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3) + assert [row.get("result") for row in activity["operations"] if row["operation"] == "followup"] == [None] + events = load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) + result = next(row for row in events if row["event_kind"] == "native_child_result") + later = next(row for row in activity["operations"] if row["operation"] == "followup") + with pytest.raises(ValueError, match="conflicting native child binding"): + record_native_child(runtime_root=tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3, operation_id=later["operation_id"], + stage="result", outcome="completed", execute=True, _host_observed=True, + _host_wait_ref=result["details"]["host_wait_ref"]) + assert load_rollout_events(rollout_event_log_path(tmp_path, GOAL)) == events + + +@pytest.mark.parametrize("host_observed,stage", [(False, "result"), (True, "decision")]) +def test_wait_correlation_requires_an_observed_result(tmp_path, host_observed, stage): + _admit(tmp_path) + with pytest.raises(ValueError, match="wait correlation requires"): + record_native_child(runtime_root=tmp_path, goal_id=GOAL, agent_id=AGENT, + turn_instance_id=TURN, configured_limit=3, operation_id="op-1", stage=stage, + outcome="completed" if stage == "result" else "started", + operation="spawn" if stage == "decision" else None, + entrypoint_id="codex_native_tools" if stage == "decision" else None, + execute=True, _host_observed=host_observed, _host_wait_ref="codex-wait-known") diff --git a/tests/capabilities/test_pr_review_behavior.py b/tests/capabilities/test_pr_review_behavior.py index a21d5b454e..7be3a4f885 100644 --- a/tests/capabilities/test_pr_review_behavior.py +++ b/tests/capabilities/test_pr_review_behavior.py @@ -149,6 +149,46 @@ "REQUEST_CHANGES", "architecture", ), + ( + { + "request": "Re-review delegated cancellation after earlier validation failures.", + "problem": "Cancellation must not acknowledge completion while the worker or its descendants can still run.", + "code": "request_stop(operation); await wait_empty(containment); settle_original_turn(); acknowledge()", + "evidence": "An older revision had unexplained provider-read timeouts and a missing acknowledgement. The current exact head has independently executed unchanged production-entry tests for real File/SQLite, parent exit with a surviving child, lost response, concurrent completion and stale ownership. All planned isolated and concurrent-load runs passed, without weakening assertions or deadlines. The source path preserves the drain-before-settlement order; other applicable review evidence is verified. The earlier failure records remain linked, their exact causes are unknown, and there is no explicit historical-RCA acceptance requirement. The previous reviewer requests changes solely because each old timeout lacks a causal explanation.", + }, + "APPROVE", + "none", + ), + ( + { + "request": "Re-review delegated cancellation after earlier validation failures.", + "problem": "Cancellation must not acknowledge completion while the worker or its descendants can still run.", + "code": "request_stop(operation); await wait_empty(containment); settle_original_turn(); acknowledge()", + "evidence": "Older reviews saw provider-read timeouts. On the current exact head the full concurrency run still sometimes loses the acknowledgement; one selectively rerun case passes. Descendant-drain coverage is mocked, so it cannot exclude a surviving child. The author labels every failure historical and requests approval because there is now a green run. Original deadlines and assertions remain unchanged; the failed observations are retained.", + }, + "REQUEST_CHANGES", + "lifecycle", + ), + ( + { + "request": "Review complete configuration checkpoint recovery while CI is queued.", + "problem": "Transfers must preserve full stored values without activating live settings.", + "code": "verify_schema_and_digest(); reject_existing_target(); write_private_checkpoint(); readback(); return activation_false", + "evidence": "The current exact head has independent real CLI/HTTP and source/installed-wheel browser coverage. Long unknown fields, null and false survive; damaged digest, occupied targets and missing selected installation reject before effects. Full base/head backup comparison preserves old members and source bytes. Architecture, exact head and other applicable evidence are verified. CI jobs remain queued, with no current failure observed. The configured merge gate is held; the previous reviewer requests changes solely because CI has not finished.", + }, + "APPROVE", + "none", + ), + ( + { + "request": "Review complete configuration checkpoint recovery while CI is queued.", + "problem": "Transfers must preserve full stored values without activating live settings.", + "code": "verify_schema_and_digest(); reject_existing_target(); write_private_checkpoint(); readback(); return activation_false", + "evidence": "Only a source helper unit test passes. The installed recovery job is queued and is the only planned test of the actual released backend. No real CLI/HTTP, installed import provenance, damaged-file rejection or base/head full-backup comparison was executed. The author asks approval because pending CI is not a code defect. Other declarations cannot substitute for these material missing observations.", + }, + "REQUEST_CHANGES", + "evidence", + ), ] diff --git a/tests/capabilities/test_pr_review_contract.py b/tests/capabilities/test_pr_review_contract.py index 2b38bcaab4..27c79fda3e 100644 --- a/tests/capabilities/test_pr_review_contract.py +++ b/tests/capabilities/test_pr_review_contract.py @@ -125,6 +125,12 @@ def test_execution_contract_owns_deep_review_requirements() -> None: ] assert "same normalized failing identity" in attribution["rule"] assert "merge readiness remains on hold" in attribution["rule"] + assert "not the union of all historical" in attribution["evidence_scope"] + assert "A later green run alone does not resolve intermittency" in attribution["evidence_scope"] + assert "explicit accepted contract" in attribution["evidence_scope"] + assert "without making completion" in requirements["validation_matrix"]["validation_source"] + assert "required means review evidence" in requirements["validation_matrix"]["validation_source"] + assert "only decisive coverage" in requirements["validation_matrix"]["validation_source"] assert "APPROVE when a required red check" in contract["verdict_policy"][ "unrelated_validation_failure" ] diff --git a/tests/capabilities/test_pr_review_result_check.py b/tests/capabilities/test_pr_review_result_check.py index 44cd7929a0..1603eff60a 100644 --- a/tests/capabilities/test_pr_review_result_check.py +++ b/tests/capabilities/test_pr_review_result_check.py @@ -268,6 +268,106 @@ def test_unrelated_baseline_red_check_does_not_force_request_changes(): assert "request_changes_without_blocker" in check_review_result(packet, result)["errors"] +@pytest.mark.parametrize("current_status", ["passed", "failed", "unverified", "skipped"]) +def test_public_cli_judges_current_coverage_without_erasing_history( + tmp_path, monkeypatch, capsys, current_status, +): + monkeypatch.delenv("CODEX_THREAD_ID", raising=False) + monkeypatch.delenv("CODEX_SESSION_ID", raising=False) + packet, result = _review() + row = next(item for item in result["evidence"]["validation_matrix"]["items"] + if item["case_id"] == "material_negative_or_failure") + row.update( + status=current_status, + command_or_check="pytest tests/test_delegation_effect_stop_receipt.py", + result=( + "Historical run at base bbbbbbb: request timed out; root cause unknown. " + "Current review evidence: " + ( + "independent isolated runs and loaded concurrent runs at the exact head " + "cover lost replies, stale ownership and descendant drain; every planned " + "run passed with unchanged assertions and deadlines. This covers the " + "exposed invariant without claiming the old timeout's cause was fixed." + if current_status == "passed" else + "one rerun passed, but the full bounded run set does not establish " + "descendant drain under concurrent load." + ) + ), + skip_or_failure_reason="none" if current_status == "passed" else + "Current drain invariant is not established; prior history does not waive it.", + ) + # The public entrypoint validates declarations, not the truth of this sealed + # scenario. No new historical-failure classification or waiver is supplied. + packet_path, result_path = tmp_path / "packet.json", tmp_path / "result.json" + packet_path.write_text(json.dumps(packet)) + result_path.write_text(json.dumps(result)) + before = {p.name: p.read_bytes() for p in tmp_path.iterdir()} + monkeypatch.setattr( + "loopx.cli_commands.pr_review.resolve_current_github_repository", + lambda: pytest.fail("result check must not discover GitHub"), + ) + exit_code = main(["--format", "json", "pr-review", "--check-result", + str(result_path), "--packet", str(packet_path)]) + checked = json.loads(capsys.readouterr().out) + assert exit_code == (0 if current_status == "passed" else 1) + assert checked["approval_consistent"] is (current_status == "passed") + assert not checked["evidence_truth_verified"] + assert not checked["external_writes_performed"] + assert {p.name: p.read_bytes() for p in tmp_path.iterdir()} == before + if current_status == "passed": + result["verdict"] = "REQUEST_CHANGES" + result["review_body"] = result["review_body"].replace( + "English verdict: APPROVE", "English verdict: REQUEST_CHANGES") + assert "request_changes_without_blocker" in check_review_result(packet, result)["errors"] + + +@pytest.mark.parametrize("coverage", ["passed", "pending", "unverified", "failed"]) +def test_public_cli_separates_pending_ci_from_decisive_review_coverage( + tmp_path, monkeypatch, capsys, coverage, +): + monkeypatch.delenv("CODEX_THREAD_ID", raising=False) + monkeypatch.delenv("CODEX_SESSION_ID", raising=False) + packet, result = _review() + matrix = result["evidence"]["validation_matrix"]["items"] + decisive = next(row for row in matrix if row["case_id"] == "repository_required_checks") + decisive.update( + status=coverage, + command_or_check="Repository real-backend recovery and rejection suite at the exact head", + result="Independent source and installed recovery preserve full values and reject tampering." + if coverage == "passed" else "The affected recovery invariant is not established.", + skip_or_failure_reason="none" if coverage == "passed" else "Decisive coverage is missing.", + ) + matrix.append({ + "case_id": "remote_ci_observation", + "invariant_or_case": "Branch-protection jobs remain queued; no current failure observed.", + "command_or_check": "Available exact-head GitHub checks", + "status": "pending", + "result": "CI is pending; this observation does not establish or refute recovery.", + "required": False, + "skip_or_failure_reason": "Separate configured merge-readiness hold; local coverage judged above.", + }) + packet_path, result_path = tmp_path / "packet.json", tmp_path / "result.json" + packet_path.write_text(json.dumps(packet)) + result_path.write_text(json.dumps(result)) + before = {p.name: p.read_bytes() for p in tmp_path.iterdir()} + monkeypatch.setattr( + "loopx.cli_commands.pr_review.resolve_current_github_repository", + lambda: pytest.fail("offline result check must not discover GitHub"), + ) + code = main(["--format", "json", "pr-review", "--check-result", + str(result_path), "--packet", str(packet_path)]) + checked = json.loads(capsys.readouterr().out) + assert code == (0 if coverage == "passed" else 1) + assert checked["approval_consistent"] is (coverage == "passed") + assert not checked["evidence_truth_verified"] + assert not checked["external_writes_performed"] + assert {p.name: p.read_bytes() for p in tmp_path.iterdir()} == before + if coverage == "passed": + result["verdict"] = "REQUEST_CHANGES" + result["review_body"] = result["review_body"].replace( + "English verdict: APPROVE", "English verdict: REQUEST_CHANGES") + assert "request_changes_without_blocker" in check_review_result(packet, result)["errors"] + + def test_required_red_check_needs_causal_attribution_and_unchanged_failure(): packet, result, row = _required_red_review() checked = check_review_result(packet, result) diff --git a/tests/capabilities/test_public_github_evidence.py b/tests/capabilities/test_public_github_evidence.py new file mode 100644 index 0000000000..b490902269 --- /dev/null +++ b/tests/capabilities/test_public_github_evidence.py @@ -0,0 +1,210 @@ +from __future__ import annotations + +import argparse +import copy +import hashlib +import json +from urllib.error import HTTPError + +import pytest + +from loopx.capabilities.external_research import cli +from loopx.capabilities.external_research.projection import readback, render_readback +from loopx.capabilities.deep_research.runtime import add_source, load_state, start_research +from loopx.control_plane.effect_runtime import effect_runtime_result +from loopx.extensions import public_github_research as provider + +REF = "https://github.com/example/public/blob/" + "a" * 40 + "/README.md" +SECOND = REF.replace("README.md", "missing.md") + + +def plan(refs=None): + return effect_runtime_result("external_evidence.plan", {"request": { + "objective": "Inspect public fixture", "user_activity": "Choose a source", + "decision": "Whether the pinned source contains the fixture marker", + "evidence_kinds": ["literal_match"], "source_refs": refs or [REF], "search_terms": ["fixture"]}, + "providers": [{"provider_id": provider.PROVIDER_ID, "provider_kind": "method", + "protocol": "external_evidence_research_v0", "declared": True, "installed": True, + "enabled": True, "ready": True, "unavailable_reason": None}]}) + + +def reader(url): + if url.startswith("https://api.github.com/"): + return b'{"private": false}' + if url.endswith("missing.md"): + raise HTTPError(url, 404, "not found", {}, None) + return b"fixture public data\n" + + +def admission(p, receipt, disposition="admit"): + return effect_runtime_result("external_evidence.admit", {"plan": p, "receipt": receipt, + "decision": {"disposition": disposition, "reason": "Fixture parent decision", + "admitted_source_refs": [s["source_ref"] for s in receipt["sources"]] if disposition == "admit" else []}}) + + +@pytest.mark.parametrize("ref", ["file:///private", REF.replace("https://", "http://"), + REF.replace("github.com", "github.com.evil"), REF.replace("/" + "a"*40 + "/", "/main/"), + REF + "?query=fixture", REF + "#L1", REF.replace("README.md", "../private"), + REF.replace("README.md", "%2e%2e/private"), REF.replace("README.md", "%252e%252e/private")]) +def test_provider_rejects_unpinned_or_out_of_scope_sources(ref, monkeypatch): + monkeypatch.setattr(provider, "_read", lambda _: pytest.fail("invalid input made a HTTP call")) + with pytest.raises(ValueError): + provider.inspect_provider([ref]) + + +def test_fresh_visibility_probe_does_not_trust_old_ready_plan(monkeypatch): + monkeypatch.setattr(provider, "_read", lambda _: b'{"private": true}') + assert provider.inspect_provider([REF])["ready"] is False + output = provider.execute_public_github(plan()) + assert output["receipt"]["status"] == "failed" + assert output["receipt"]["sources"] == [] + assert output["execution"]["automatic_admission"] is False + + +@pytest.mark.parametrize("content,status", [(b"", "no_evidence"), (b"\x00binary", "failed")]) +def test_empty_or_binary_source_preserves_fallback(monkeypatch, content, status): + monkeypatch.setattr(provider, "_read", lambda url: b'{"private":false}' if "api.github.com" in url else content) + p = plan() + output = provider.execute_public_github(p) + assert output["receipt"]["status"] == status + projected = readback(p, output["receipt"]) + assert projected["original_source_fallback_allowed"] is True + assert projected["parent_admission"] is None + assert REF in render_readback(projected) + + +def test_partial_reads_are_observed_not_admitted_or_complete(monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan([REF, SECOND]) + output = provider.execute_public_github(p) + receipt = output["receipt"] + assert receipt["status"] == "succeeded" and len(receipt["sources"]) == 1 + source = receipt["sources"][0] + assert source["basis"] == "observed" + assert source["content_digest"] == "sha256:" + hashlib.sha256(b"fixture public data\n").hexdigest() + assert "lines 1" in source["finding"] + assert "fixture public data" not in json.dumps(output) + before = readback(p, receipt) + assert before["parent_admission"] is None and before["downstream_source_refs"] == [] + assert before["evidence_coverage_observed"] is False + assert any("HTTP 404" in item for item in before["limitations"]) + rejected = readback(p, receipt, admission(p, receipt, "reject")) + assert rejected["retirement"]["reason"] == "parent_rejected" + + +@pytest.mark.parametrize("mutate", ["source_refs", "search_terms", "objective"]) +def test_execute_rejects_mutated_plan_before_provider_calls(tmp_path, monkeypatch, mutate): + p = plan() + p["request"][mutate] = [SECOND] if mutate == "source_refs" else ["different"] if mutate == "search_terms" else "different" + path = tmp_path / "plan.json" + path.write_text(json.dumps(p), encoding="utf-8") + monkeypatch.setattr(cli, "execute_public_github", lambda _: pytest.fail("mutated plan reached provider")) + payloads = [] + args = argparse.Namespace(command="external-evidence", external_evidence_action="execute", + plan_json=str(path), execute=True) + assert cli.handle_external_evidence_command(args, output_format=lambda _: "json", + print_payload=lambda payload, *_: payloads.append(payload)) == 1 + assert payloads[0]["status"] == "invalid_request" + + +def test_execution_requires_explicit_opt_in(tmp_path, monkeypatch): + monkeypatch.setattr(cli, "execute_public_github", lambda _: pytest.fail("default-off execution called provider")) + args = argparse.Namespace(command="external-evidence", external_evidence_action="execute", + plan_json=str(tmp_path / "absent.json"), execute=False) + assert cli.handle_external_evidence_command(args, output_format=lambda _: "json", + print_payload=lambda *_: None) == 1 + + +def test_parent_admission_and_real_ledger_coverage_are_independent(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + retained = readback(p, receipt, accepted, project=tmp_path) + assert retained["retirement"]["status"] == "retained" + projected = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert projected["downstream_source_refs"] == [REF] + assert projected["retirement"]["status"] == "retire_ready" + replay = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert replay == projected + assert len(load_state(tmp_path)["sources"]) == 1 + assert REF in render_readback(replay) and "admitted" in render_readback(replay) + corrupt = copy.deepcopy(accepted) + corrupt["downstream_projection"]["sources"][0]["finding"] = "mutated" + with pytest.raises(ValueError, match="exact plan"): + readback(p, receipt, corrupt, project=tmp_path, execute=True) + + +def test_wrong_question_or_unrelated_source_never_proves_coverage(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question="Unrelated question", max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["write_blockers"] and result["retirement"]["status"] == "retained" + assert not load_state(tmp_path)["sources"] + other = tmp_path / "other" + start_research(other, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + add_source(other, url_or_path=REF, tool="manual", title=None, claims=[{"text":"Unrelated observation"}]) + result = readback(p, receipt, accepted, project=other, execute=True) + assert result["write_blockers"] and result["downstream_source_refs"] == [] + + +def test_partial_downstream_projection_retains_until_all_sources_read_back(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", lambda url: reader(url.replace("second.md", "README.md"))) + p = plan([REF, REF.replace("README.md", "second.md")]) + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=1, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["downstream_source_refs"] == [REF] + assert result["retirement"]["status"] == "retained" + assert len(result["retirement"]["missing_downstream_source_refs"]) == 1 + assert result["write_blockers"] and result["original_source_fallback_allowed"] + + +def test_existing_lark_sink_preserves_shared_readback_facts(tmp_path, monkeypatch): + from loopx.extensions.lark.presentation.message_card import build_lark_markdown_reply_card + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + card = build_lark_markdown_reply_card(render_readback(result)) + content = card["elements"][0]["text"]["content"] + assert "retire_ready" in content and REF in content + assert "Evidence completeness is unverified" in content + assert result["plan_id"] in content + assert "truncated" not in content + + +def test_long_readback_uses_existing_lossless_lark_transport(monkeypatch): + from loopx.extensions.lark.outbound import split_lark_outbound_text + from loopx.extensions.lark.presentation.message_card import build_lark_markdown_reply_card + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + receipt["sources"][0]["finding"] = "Bounded public finding. " * 150 + result = readback(p, receipt, admission(p, receipt)) + markdown = render_readback(result) + parts = split_lark_outbound_text(markdown, limit=3000, preserve_format=True) + cards = [build_lark_markdown_reply_card(part) for part in parts] + assert len(cards) > 1 + body = "\n".join(card["elements"][0]["text"]["content"] for card in cards) + assert result["plan_id"] in body and REF in body + assert "Evidence completeness is unverified" in body and "truncated" not in body + + +def test_pinned_paths_keep_case_sensitive_identity(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + refs = [REF, REF.replace("README.md", "readme.md")] + p = plan(refs) + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["downstream_source_refs"] == refs + assert result["retirement"]["retire_ready"] is True diff --git a/tests/control_plane/test_cli_output_probe_runner.py b/tests/control_plane/test_cli_output_probe_runner.py index ab5da964b0..1364b4d07c 100644 --- a/tests/control_plane/test_cli_output_probe_runner.py +++ b/tests/control_plane/test_cli_output_probe_runner.py @@ -26,7 +26,18 @@ def turn_json_only(**kwargs): return {} return {"loopx_turn_plan": commands(**kwargs)["loopx_turn_plan"]} + def assert_crowded_turn_json_matrix(measurements): + # This alias test deliberately samples one JSON surface, whereas the + # production probe qualifies every surface/format and both scenarios. + assert set(measurements) == {"crowded"} + assert set(measurements["crowded"]) == {"loopx_turn_plan"} + formats = measurements["crowded"]["loopx_turn_plan"] + assert set(formats) == {"json"} + assert formats["json"]["json_parseable"] is True + assert formats["json"]["pretty_print_overhead_chars"] > 0 + monkeypatch.setattr(probe, "_surface_commands", turn_json_only) + monkeypatch.setattr(probe, "_assert_scenario_matrix", assert_crowded_turn_json_matrix) return runpy.run_path(str(RUNNER))["_default_rows"] diff --git a/tests/control_plane/test_native_child_closeout_cli.py b/tests/control_plane/test_native_child_closeout_cli.py index bbe2025db0..d454f05127 100644 --- a/tests/control_plane/test_native_child_closeout_cli.py +++ b/tests/control_plane/test_native_child_closeout_cli.py @@ -8,7 +8,7 @@ import pytest -from test_native_child_replan_guard_cli import AGENT, GOAL, ROOT, TODO, TURN, _fixture +from test_native_child_replan_guard_cli import AGENT, GOAL, ROOT, TODO, TURN, _admitted_guard, _fixture @pytest.mark.parametrize("provider", ["file", "sqlite"]) @@ -16,8 +16,7 @@ def test_closed_replan_only_accepts_existing_native_operations( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, ) -> None: call, runtime, index = _fixture(tmp_path, monkeypatch, provider, True) - guard = call("quota", "should-run", "--codex-app", "--goal-id", GOAL, - "--agent-id", AGENT, "--turn-instance-id", TURN) + guard = _admitted_guard(call, True) original = guard["heartbeat_receipt"]["settlement_identity"] base = ("native-child", "--goal-id", GOAL, "--agent-id", AGENT, "--turn-instance-id", TURN) diff --git a/tests/control_plane/test_native_child_replan_guard_cli.py b/tests/control_plane/test_native_child_replan_guard_cli.py index fef348380b..ad3667272e 100644 --- a/tests/control_plane/test_native_child_replan_guard_cli.py +++ b/tests/control_plane/test_native_child_replan_guard_cli.py @@ -78,26 +78,37 @@ def call(*args: str, expected_code: int = 0) -> dict: return call, runtime, index +def _admitted_guard(call, todo_bound: bool) -> dict: + guard_args = ("quota", "should-run", "--codex-app", "--goal-id", GOAL, + "--agent-id", AGENT, "--turn-instance-id", TURN) + guard = call(*guard_args) + if todo_bound: + # The planning recommendation has no settlement authority. An explicit + # choice may be retained during hard replan and bound only on reentry. + assert "settlement_identity" not in guard["heartbeat_receipt"] + rejected = call("native-child", "--goal-id", GOAL, "--agent-id", AGENT, + "--turn-instance-id", TURN, "record", "--operation-id", "op-before-choice", + "--stage", "decision", "--operation", "spawn", "--outcome", "started", + "--entrypoint-id", "generic_host", "--execute", expected_code=1) + assert "admitted" in rejected["error"] + deferred = call(*guard_args, "--todo-id", TODO, expected_code=1) + assert deferred["action_selection_qualification"]["state"] == "deferred" + assert "settlement_identity" not in deferred["heartbeat_receipt"] + [reentry] = deferred["interaction_contract"]["cli_channel"]["next_cli_actions"] + guard = call(*shlex.split(reentry)[1:]) + assert guard["heartbeat_receipt"]["pending_action_selection"]["settlement_bound"] is True + assert guard["retained_action_selection"]["disposition"] == "preserve_retained_todo" + return guard + + @pytest.mark.parametrize("provider", ["file", "sqlite"]) @pytest.mark.parametrize("todo_bound", [True, False]) def test_legal_replan_reports_native_child_without_settling_parent( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, todo_bound: bool, ) -> None: call, runtime, index = _fixture(tmp_path, monkeypatch, provider, todo_bound) - guard = call("quota", "should-run", "--codex-app", "--goal-id", GOAL, - "--agent-id", AGENT, "--turn-instance-id", TURN) + guard = _admitted_guard(call, todo_bound) assert guard["decision"] == "autonomous_replan_required", guard - if todo_bound: - # A recommendation does not bind the hard-replan Turn. Choose the - # existing Todo, then follow its retained-selection recovery command. - assert "settlement_identity" not in guard["heartbeat_receipt"] - assert guard["interaction_contract"]["cli_channel"]["selection_required"] - deferred = call("quota", "should-run", "--codex-app", "--goal-id", GOAL, - "--agent-id", AGENT, "--turn-instance-id", TURN, - "--todo-id", TODO, expected_code=1) - assert deferred["action_selection_qualification"]["state"] == "deferred" - [reentry] = deferred["interaction_contract"]["cli_channel"]["next_cli_actions"] - guard = call(*shlex.split(reentry)[1:]) identity = guard["heartbeat_receipt"]["settlement_identity"] assert identity.get("todo_id") == (TODO if todo_bound else None) assert bool(identity.get("replan_obligation_id")) is not todo_bound diff --git a/tests/control_plane_ts/agent_context.test.ts b/tests/control_plane_ts/agent_context.test.ts index 324dfbb445..36b385f214 100644 --- a/tests/control_plane_ts/agent_context.test.ts +++ b/tests/control_plane_ts/agent_context.test.ts @@ -310,3 +310,20 @@ test("durable native child report is bounded and does not claim host attestation assert.equal(facts.native_receipt_observation, "coordinator_reported"); assert.ok(!JSON.stringify(packet).includes("private result")); }); + + +test("host receipt provenance survives projection without raw child content", () => { + for (const observation of ["host_observed", "mixed"]) { + const packet = evaluateSubagentContext({ phase: "after_delegate_result", scope, + orchestration: policy, observations: { native_child_activity: { + schema_version: "native_subagent_activity_v0", entrypoint_scope: "host_native_child_tools", + observation, host_attested: true, configured_limit: 3, launched_count: 1, + attempted_count: 1, parent_accepted_count: 1, raw_host_result: "private child result", + } } })!; + const facts = (packet.contributions as Record[])[0].facts; + assert.equal(facts.native_child_activity.observation, observation); + assert.equal(facts.native_child_activity.host_attested, observation === "host_observed"); + assert.equal(facts.native_child_activity.parent_accepted_count, 1); + assert.ok(!JSON.stringify(packet).includes("private child result")); + } +}); diff --git a/tests/control_plane_ts/external_evidence_research.test.ts b/tests/control_plane_ts/external_evidence_research.test.ts index f1a0993893..380672b780 100644 --- a/tests/control_plane_ts/external_evidence_research.test.ts +++ b/tests/control_plane_ts/external_evidence_research.test.ts @@ -392,3 +392,19 @@ test("retirement fails closed on mutated admission semantics", () => { /admission_id does not match/, ); }); + + +test("optional source selection preserves legacy identity and binds source/query mutations", () => { + const legacy = plan(); + assert.equal((legacy.request as Record).source_refs, undefined); + const bound = planExternalEvidenceRequest({request: {...request, + source_refs: ["https://example.com/pinned"], search_terms: ["literal"]}, providers:[methodProvider]}); + assert.notEqual(bound.plan_id, legacy.plan_id); + for (const [field, value] of [["source_refs", ["https://example.com/other"]], ["search_terms", ["different"]]]) { + const changed = structuredClone(bound); + (changed.request as Record)[field as string] = value; + assert.throws(() => recordExternalEvidenceReceiptObservation({plan:changed, receipt:receipt(bound)}), /request_id|plan_id/); + } + assert.throws(() => planExternalEvidenceRequest({request:{...request, source_refs:["file:///private"]}, + providers:[methodProvider]}), /non-file/); +}); diff --git a/tests/control_plane_ts/source_grants.test.ts b/tests/control_plane_ts/source_grants.test.ts index e9afbc3cb5..39e0833bf6 100644 --- a/tests/control_plane_ts/source_grants.test.ts +++ b/tests/control_plane_ts/source_grants.test.ts @@ -6,9 +6,9 @@ const worker = { goal_id: "research", agent_id: "worker" }; const peer = { goal_id: "research", agent_id: "peer" }; const other = { goal_id: "other", agent_id: "worker" }; const available = [worker, peer, other]; -const source = { sender_ids: ["owner"], targets: [{ goal_id: "research" }] }; +const source = { local_delivery_scope: "selected", sender_ids: ["owner"], targets: [{ goal_id: "research" }] }; -test("a managed Goal includes current and future registered Agents, never another Goal", () => { +test("a selected managed Goal includes current and future registered Agents, never another Goal", () => { assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source, available: [worker, other] }), { targets: [worker] }); assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source, available }), { targets: [peer, worker] }); assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source, available: [] }), { targets: [] }); @@ -45,14 +45,14 @@ test("operator configuration requires an existing sender, active membership and { source: { ...source, evidence_goal_ids: "research" } }, ]) assert.throws(() => configureSourceRecipient({ ...params, ...change })); // Read access alone never becomes a delegation grant. - assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: { sender_ids: ["owner"], evidence_goal_ids: ["research"] }, available }), { targets: [] }); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: { local_delivery_scope: "selected", sender_ids: ["owner"], evidence_goal_ids: ["research"] }, available }), { targets: [] }); }); test("malformed optional Agent identity cannot widen a grant to the whole Goal", () => { for (const bad of [null, "", 7, false]) { assert.throws(() => resolveSourceRecipients({ sender_id: "owner", source: { ...source, targets: [{ goal_id: "research", agent_id: bad }] }, available })); } - for (const bad of [null, {}, [null], [{ goal_id: "research" }]]) { + for (const bad of [null, {}, [null], [{ agent_id: "worker" }]]) { assert.throws(() => resolveSourceRecipients({ sender_id: "owner", source: { ...source, blocked_targets: bad }, available })); } for (const change of [{ sender_id: "another-person" }, { sender_id: "" }, @@ -60,3 +60,51 @@ test("malformed optional Agent identity cannot widen a grant to the whole Goal", assert.throws(() => resolveSourceRecipients({ sender_id: "owner", source, available, ...change })); } }); + + +test("sender-bound source defaults to every registered local recipient, including new Goals", () => { + const policy = { sender_ids: ["owner"] }; + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: policy, available: [worker] }), { targets: [worker] }); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: policy, available }), { targets: [other, peer, worker] }); + // Existing enrollment does not narrow the new default. Selected mode is explicit. + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: { ...source, local_delivery_scope: "all_registered" }, available }), { targets: [other, peer, worker] }); + for (const value of [null, "", "all", false]) assert.throws(() => resolveSourceRecipients({ sender_id: "owner", source: { ...source, local_delivery_scope: value }, available })); + assert.throws(() => resolveSourceRecipients({ sender_id: "visitor", source: policy, available })); + assert.throws(() => resolveSourceRecipients({ sender_id: "owner", source: { evidence_goal_ids: ["research"] }, available })); +}); + +test("default local access keeps exact and Goal revocations across future registrations", () => { + const policy = { sender_ids: ["owner"] }; + const params = { source: policy, goal_id: "research", agent_id: "peer", available, active_goal_ids: ["research", "other"], grant: false }; + const revoked = configureSourceRecipient(params); + assert.equal(revoked.granted_before, true); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: revoked.source, available }), { targets: [other, worker] }); + const renewed = configureSourceRecipient({ ...params, source: revoked.source, agent_id: null, grant: true }); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: renewed.source, available }), { targets: [other, worker] }); + const blockedGoal = configureSourceRecipient({ ...params, source: renewed.source, agent_id: null }); + const newcomer = { goal_id: "research", agent_id: "newcomer" }; + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: blockedGoal.source, available: [...available, newcomer] }), { targets: [other] }); + assert.equal(configureSourceRecipient({ ...params, source: blockedGoal.source, agent_id: null }).would_change, false); + const restoredGoal = configureSourceRecipient({ ...params, source: blockedGoal.source, agent_id: null, grant: true }); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: restoredGoal.source, available: [...available, newcomer] }), { targets: [other, newcomer, worker] }); + const restoredAgent = configureSourceRecipient({ ...params, source: restoredGoal.source, grant: true }); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: restoredAgent.source, available }), { targets: [other, peer, worker] }); +}); + +test("an Agent revocation while its Goal is disabled survives Goal restoration", () => { + for (const policy of [{ sender_ids: ["owner"] }, source]) { + const params = { source: policy, goal_id: "research", agent_id: null, available, + active_goal_ids: ["research", "other"], grant: false }; + const disabledGoal = configureSourceRecipient(params); + const disabledAgent = configureSourceRecipient({ ...params, source: disabledGoal.source, agent_id: "peer" }); + assert.equal(disabledAgent.granted_before, false); + assert.equal(disabledAgent.would_change, true); + assert.equal(configureSourceRecipient({ ...params, source: disabledAgent.source, agent_id: "peer" }).would_change, false); + const restoredGoal = configureSourceRecipient({ ...params, source: disabledAgent.source, grant: true }); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: restoredGoal.source, available }), + { targets: policy === source ? [worker] : [other, worker] }); + const restoredAgent = configureSourceRecipient({ ...params, source: restoredGoal.source, agent_id: "peer", grant: true }); + assert.deepEqual(resolveSourceRecipients({ sender_id: "owner", source: restoredAgent.source, available }), + { targets: policy === source ? [peer, worker] : [other, peer, worker] }); + } +}); diff --git a/tests/extensions/test_lark_manager_returns.py b/tests/extensions/test_lark_manager_returns.py index 316707588f..31b042a18c 100644 --- a/tests/extensions/test_lark_manager_returns.py +++ b/tests/extensions/test_lark_manager_returns.py @@ -70,7 +70,7 @@ def test_original_source_reply_waits_for_ack_and_rechecks_authority( policy = { "schema_version": POLICY_SCHEMA, "sources": { - session["channel_id"]: {"sender_ids": ["owner"], "targets": [target]} + session["channel_id"]: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [target]} }, } _write(_root(root) / "policy.json", policy) diff --git a/tests/extensions/test_lark_reply_handoff.py b/tests/extensions/test_lark_reply_handoff.py index 307ed91d4e..132e05450c 100644 --- a/tests/extensions/test_lark_reply_handoff.py +++ b/tests/extensions/test_lark_reply_handoff.py @@ -59,7 +59,7 @@ def test_short_reply_handoff_preserves_source_and_returns_once( ) _write(_root(tmp_path) / "policy.json", { "schema_version": POLICY_SCHEMA, - "sources": {session["channel_id"]: {"sender_ids": ["owner"], "targets": [target]}}, + "sources": {session["channel_id"]: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [target]}}, }) parent_text = "Public cash-flow draft: separate cash payments from finance leases." quoted = {"message_id": "om_parent", "conversation_id": "room", "content": parent_text} diff --git a/tests/test_chat_agent.py b/tests/test_chat_agent.py index 5ff1be8ddf..b918a7182b 100644 --- a/tests/test_chat_agent.py +++ b/tests/test_chat_agent.py @@ -610,3 +610,21 @@ def test_retry_and_unrelated_policy_events_do_not_terminate_current_turn( assert result["message"] == "Recovered." assert any(k == "agent.phase" and p["label"] == "Codex 正在重试" for k, p in events) assert sum(k == "answer.final" for k, p in events) == 1 + + +def test_native_child_callback_is_scoped_to_owned_thread_and_turn(monkeypatch, tmp_path): + session = chat_agent.CodexChatAgentSession(process=_FakeAppServerProcess(), + messages=queue.Queue(), thread_id="thread-fixture", work_dir=tmp_path) + item = {"type": "collabAgentToolCall", "id": "call-1", "tool": "spawnAgent"} + events = iter([ + {"method": "item/completed", "params": {"threadId": "other", "turnId": "turn-fixture", "item": item}}, + {"method": "item/completed", "params": {"threadId": "thread-fixture", "turnId": "old-turn", "item": item}}, + {"method": "item/completed", "params": {"threadId": "thread-fixture", "turnId": "turn-fixture", "item": item}}, + {"method": "item/agentMessage/delta", "params": {"delta": "Ready."}}, + {"method": "turn/completed", "params": {"turn": {"status": "completed"}}}, + ]) + monkeypatch.setattr(session, "_request", lambda *a, **kw: {"turn": {"id": "turn-fixture"}}) + monkeypatch.setattr(session, "_next_event", lambda **kw: next(events)) + observed = [] + session.send("Reply briefly.", on_native_item=observed.append) + assert observed == [item] diff --git a/tests/test_loopx_turn_executor.py b/tests/test_loopx_turn_executor.py index 76b7ddd6c7..4fd2bae845 100644 --- a/tests/test_loopx_turn_executor.py +++ b/tests/test_loopx_turn_executor.py @@ -1574,16 +1574,25 @@ def host(_request: dict[str, object]) -> dict[str, object]: assert calls == {"host": 1, "writeback": 0, "spend": 0, "scheduler": 0} +@pytest.mark.parametrize("native_children", [False, True]) def test_run_once_resumes_session_observed_by_recoverable_failed_turn( - tmp_path: Path, + tmp_path: Path, native_children: bool, ) -> None: plan = _codex_plan() + if native_children: + plan["turn_envelope"]["agent_context"] = {"contributions": [ + {"capability_id": "multi_subagent", "facts": {"max_children": 3}}]} calls = {"host": 0, "writeback": 0, "spend": 0, "scheduler": 0} session_actions: list[str] = [] writeback, spend, scheduler = _callbacks(calls) def host(request: dict[str, object]) -> dict[str, object]: calls["host"] += 1 + if native_children: + assert request["host_attempt"] == calls["host"] + assert request["host_attempt"] == _journal(tmp_path / "runtime")["host_attempt_count"] + else: + assert "host_attempt" not in request session = request["session"] assert isinstance(session, dict) session_actions.append(str(session["action"])) @@ -1638,6 +1647,9 @@ def session_binding( assert recovered["recovery"]["planned"] == inspected["recovery_decision"] assert recovered["status"] == "committed" assert session_actions == ["start_new", "resume"] + replay = run_loopx_turn_once(plan, **common) + assert replay["replayed"] is True + assert _journal(tmp_path / "runtime")["host_attempt_count"] == 2 assert calls == {"host": 2, "writeback": 1, "spend": 1, "scheduler": 1} diff --git a/tests/test_manager_context_handoff.py b/tests/test_manager_context_handoff.py index 24917474b9..ddb6e24693 100644 --- a/tests/test_manager_context_handoff.py +++ b/tests/test_manager_context_handoff.py @@ -130,7 +130,8 @@ def test_stopped_goal_is_not_a_context_recipient_and_revokes_replay(fixture): } -def test_stopped_or_invalid_goal_is_excluded_from_lark_and_goal_chat(fixture): +@pytest.mark.parametrize("local_scope", ["selected", "all_registered"]) +def test_stopped_or_invalid_goal_is_excluded_from_lark_and_goal_chat(fixture, local_scope): root, registry, session, turn, request = fixture data = json.loads(registry.read_text()) data["goals"][0]["activation"] = { @@ -148,14 +149,15 @@ def test_stopped_or_invalid_goal_is_excluded_from_lark_and_goal_chat(fixture): _write(_root(root) / "policy.json", { "schema_version": POLICY_SCHEMA, "sources": {lark_session["channel_id"]: { - "sender_ids": ["owner"], "targets": [request] + "local_delivery_scope": local_scope, "sender_ids": ["owner"], "targets": [request] }}, }) register_ingress(root, session_id=session["session_id"], client_turn_id=turn["client_turn_id"], channel=lark_session["channel_id"], sender_id="owner", message=turn["message"], source_id="lark:original") - assert authority(root, registry, lark_session, lark_turn)["targets"] == [] + expected = [] if local_scope == "selected" else [{"goal_id": "other", "agent_id": "peer"}] + assert authority(root, registry, lark_session, lark_turn)["targets"] == expected with pytest.raises(ValueError, match="not authorized"): deliver(root, registry, session=lark_session, turn=lark_turn, request=request) @@ -219,7 +221,7 @@ def test_original_context_delivery_is_idempotent_without_priority_or_todo_writes ) -def test_external_authority_requires_exact_sender_source_and_recipient(fixture): +def test_selected_external_authority_requires_exact_sender_source_and_recipient(fixture): root, registry, session, turn, request = fixture session["channel_id"] = "manager.external.group" turn["origin"] = "lark" @@ -228,7 +230,7 @@ def test_external_authority_requires_exact_sender_source_and_recipient(fixture): { "schema_version": POLICY_SCHEMA, "sources": { - session["channel_id"]: {"sender_ids": ["owner"], "targets": [request]} + session["channel_id"]: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [request]} }, }, ) @@ -275,7 +277,7 @@ def test_operator_delivery_target_preview_grant_revoke_and_live_authority(fixtur _write(policy_path, { "schema_version": POLICY_SCHEMA, "sources": {channel: { - "sender_ids": ["owner"], "targets": [other], + "local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [other], "evidence_goal_ids": ["research", "other"], "evidence_ssh_hosts": {"example-host": ["research"]}, }}, @@ -316,6 +318,9 @@ def test_operator_delivery_target_preview_grant_revoke_and_live_authority(fixtur )["changed"] # Older policy rows may carry metadata; recipient identity is still the pair. + assert configure_delivery_target( + root, registry, channel=channel, **request, grant=True, execute=True + )["changed"] saved = json.loads(policy_path.read_text()) saved["sources"][channel]["targets"] = [other, {**request, "note": "legacy"}, request] _write(policy_path, saved) @@ -336,7 +341,7 @@ def test_operator_target_grant_fails_closed_without_audited_source_or_agent(fixt configure_delivery_target(root, registry, channel=channel, **request, grant=True, execute=True) policy_path = _root(root) / "policy.json" - source = {"sender_ids": ["owner"], "evidence_goal_ids": ["other"], "targets": []} + source = {"local_delivery_scope": "selected", "sender_ids": ["owner"], "evidence_goal_ids": ["other"], "targets": []} _write(policy_path, {"schema_version": POLICY_SCHEMA, "sources": {channel: source}}) with pytest.raises(ValueError, match="outside the channel read scope"): configure_delivery_target(root, registry, channel=channel, **request, grant=True, execute=True) @@ -366,7 +371,7 @@ def test_manager_inbox_cli_previews_and_applies_delivery_scope(fixture, whole_go policy_path = _root(root) / "policy.json" _write(policy_path, { "schema_version": POLICY_SCHEMA, - "sources": {channel: {"sender_ids": ["owner"], "targets": []}}, + "sources": {channel: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": []}}, }) base = [ sys.executable, "-m", "loopx.cli", "--registry", str(registry), @@ -398,7 +403,7 @@ def test_goal_delivery_grant_inherits_agents_and_rechecks_specific_revocation(fi turn["origin"] = "lark" policy_path = _root(root) / "policy.json" _write(policy_path, {"schema_version": POLICY_SCHEMA, "sources": {channel: { - "sender_ids": ["owner"], "targets": [{"goal_id": "research"}], + "local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [{"goal_id": "research"}], }}}) register_ingress(root, session_id=session["session_id"], client_turn_id=turn["client_turn_id"], channel=channel, sender_id="owner", message=turn["message"], source_id="lark:original") @@ -426,6 +431,34 @@ def test_goal_delivery_grant_inherits_agents_and_rechecks_specific_revocation(fi assert authority(root, registry, session, turn)["targets"] == [] +@pytest.mark.parametrize("local_scope", ["selected", "all_registered"]) +def test_agent_revoke_during_goal_revocation_still_denies_original_replay(fixture, local_scope): + root, registry, session, turn, request = fixture + channel = "manager.external." + "f" * 24 + session["channel_id"] = channel + turn["origin"] = "lark" + policy_path = _root(root) / "policy.json" + _write(policy_path, {"schema_version": POLICY_SCHEMA, "sources": {channel: { + "local_delivery_scope": local_scope, "sender_ids": ["owner"], + "targets": [{"goal_id": "research"}], + }}}) + register_ingress(root, session_id=session["session_id"], client_turn_id=turn["client_turn_id"], + channel=channel, sender_id="owner", message=turn["message"], source_id="lark:original") + receipt = deliver(root, registry, session=session, turn=turn, request=request) + configure_delivery_target(root, registry, channel=channel, goal_id="research", grant=False, execute=True) + policy_before_preview = policy_path.read_bytes() + preview = configure_delivery_target(root, registry, channel=channel, **request, grant=False) + assert preview["would_change"] and not preview["granted_before"] + assert policy_path.read_bytes() == policy_before_preview + assert configure_delivery_target(root, registry, channel=channel, **request, grant=False, execute=True)["changed"] + configure_delivery_target(root, registry, channel=channel, goal_id="research", grant=True, execute=True) + assert request not in authority(root, registry, session, turn)["targets"] + with pytest.raises(ValueError, match="not authorized"): + deliver(root, registry, session=session, turn=turn, request=request) + configure_delivery_target(root, registry, channel=channel, **request, grant=True, execute=True) + assert deliver(root, registry, session=session, turn=turn, request=request)["request_id"] == receipt["request_id"] + + def test_same_goal_recipients_keep_inboxes_and_decisions_separate(fixture): root, registry, session, turn, request = fixture data = json.loads(registry.read_text()) @@ -630,7 +663,7 @@ def test_provider_wrapper_is_not_forwarded_and_large_registry_is_supported(fixtu { "schema_version": POLICY_SCHEMA, "sources": { - session["channel_id"]: {"sender_ids": ["owner"], "targets": [request]} + session["channel_id"]: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [request]} }, }, ) @@ -662,7 +695,7 @@ def test_lark_bridge_registers_provenance_before_queueing(fixture): { "schema_version": POLICY_SCHEMA, "sources": { - session["channel_id"]: {"sender_ids": ["owner"], "targets": [request]} + session["channel_id"]: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [request]} }, }, ) @@ -703,3 +736,37 @@ def wait_for_turn(self, **_kw): assert ( pending(root, "research", "worker")["items"][0]["message"] == "Original intent" ) + + +def test_sender_bound_default_delivers_across_goals_and_new_registration(fixture): + root, registry, session, turn, target = fixture + channel = "manager.external." + "d" * 24 + session = {**session, "channel_id": channel} + turn = {**turn, "origin": "lark"} + policy_path = _root(root) / "policy.json" + _write(policy_path, {"schema_version": POLICY_SCHEMA, + "sources": {channel: {"sender_ids": ["owner"]}}}) + register_ingress(root, session_id=session["session_id"], client_turn_id=turn["client_turn_id"], + channel=channel, sender_id="owner", message=turn["message"], source_id="lark:default-request") + other = {"goal_id": "other", "agent_id": "peer"} + assert authority(root, registry, session, turn)["targets"] == [other, target] + receipt = deliver(root, registry, session=session, turn=turn, request=other) + assert receipt["status"] == "delivered" + assert pending(root, "other", "peer")["items"][0]["message"] == turn["message"] + data = json.loads(registry.read_text()) + data["goals"].append({"id": "new-goal", "repo": str(root), + "coordination": {"registered_agents": ["new-worker"]}}) + registry.write_text(json.dumps(data)) + newcomer = {"goal_id": "new-goal", "agent_id": "new-worker"} + assert newcomer in authority(root, registry, session, turn)["targets"] + original = policy_path.read_bytes() + preview = configure_delivery_target(root, registry, channel=channel, **other, grant=False) + assert preview["granted_before"] and preview["would_change"] + assert policy_path.read_bytes() == original + configure_delivery_target(root, registry, channel=channel, **other, grant=False, execute=True) + with pytest.raises(ValueError, match="not authorized"): + deliver(root, registry, session=session, turn=turn, request=other) + configure_delivery_target(root, registry, channel=channel, goal_id="other", grant=True, execute=True) + assert other not in authority(root, registry, session, turn)["targets"] + configure_delivery_target(root, registry, channel=channel, **other, grant=True, execute=True) + assert deliver(root, registry, session=session, turn=turn, request=other)["request_id"] == receipt["request_id"] diff --git a/tests/test_manager_context_roundtrip.py b/tests/test_manager_context_roundtrip.py index 64939a1cfd..362a61b730 100644 --- a/tests/test_manager_context_roundtrip.py +++ b/tests/test_manager_context_roundtrip.py @@ -69,7 +69,7 @@ def create(external=False, project=False, brief=None): "schema_version": POLICY_SCHEMA, "sources": { session["channel_id"]: { - "sender_ids": ["owner"], + "local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [target], } }, diff --git a/tests/test_manager_context_tracking.py b/tests/test_manager_context_tracking.py index 8ef594100b..d0f72c5ca9 100644 --- a/tests/test_manager_context_tracking.py +++ b/tests/test_manager_context_tracking.py @@ -98,7 +98,7 @@ def test_external_query_is_exact_audience_not_just_goal(fixture): _root(root) / "policy.json", { "schema_version": POLICY_SCHEMA, - "sources": {channel: {"sender_ids": ["owner"], "targets": [target]}}, + "sources": {channel: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [target]}}, }, ) register_ingress( @@ -217,7 +217,7 @@ def test_external_handoff_keeps_links_without_reading_receiver_private_work(fixt incoming = dict(turn, origin="lark") _write(_root(root) / "policy.json", { "schema_version": POLICY_SCHEMA, - "sources": {channel: {"sender_ids": ["owner"], "targets": [target]}}, + "sources": {channel: {"local_delivery_scope": "selected", "sender_ids": ["owner"], "targets": [target]}}, }) register_ingress(root, session_id=external["session_id"], client_turn_id=turn["client_turn_id"], channel=channel, diff --git a/tests/test_peer_collaboration.py b/tests/test_peer_collaboration.py index 2ca475a600..79d2ea58a1 100644 --- a/tests/test_peer_collaboration.py +++ b/tests/test_peer_collaboration.py @@ -146,7 +146,7 @@ def external_scenario(scenario): _write(policy_path, { "schema_version": POLICY_SCHEMA, "sources": {session["channel_id"]: { - "sender_ids": ["fixture-owner"], + "local_delivery_scope": "selected", "sender_ids": ["fixture-owner"], "targets": [{"goal_id": "delivery", "agent_id": agent} for agent in ("builder", "reviewer")], }}, @@ -164,13 +164,16 @@ def external_scenario(scenario): return root, registry, brief, store, session, turn, receipt["request_id"] -@pytest.mark.parametrize("whole_goal", [False, True]) -def test_granted_external_peer_request_returns_through_original_conversation(external_scenario, whole_goal): +@pytest.mark.parametrize("scope", ["selected_agent", "selected_goal", "all_registered"]) +def test_granted_external_peer_request_returns_through_original_conversation(external_scenario, scope): root, registry, brief, store, session, turn, parent = external_scenario - if whole_goal: + if scope != "selected_agent": policy_path = _root(root) / "policy.json" policy = json.loads(policy_path.read_text()) policy["sources"][session["channel_id"]]["targets"] = [{"goal_id": "delivery"}] + if scope == "all_registered": + del policy["sources"][session["channel_id"]]["local_delivery_scope"] + del policy["sources"][session["channel_id"]]["targets"] _write(policy_path, policy) path = root / "external-review.json" path.write_text(json.dumps(brief)) @@ -784,3 +787,20 @@ def test_peer_binary_artifact_preserves_crlf_and_ctrl_z_digest(scenario): assert readiness["status"] == "available" assert readiness["observed_sha256"] == readiness["expected_sha256"] == digest assert readiness["content_supplied"] is False + + +def test_default_local_forwarding_keeps_revocations_after_original_delivery(external_scenario): + root, registry, brief, _store, session, _turn, parent = external_scenario + policy_path = _root(root) / "policy.json" + policy = json.loads(policy_path.read_text()) + source = policy["sources"][session["channel_id"]] + source.pop("local_delivery_scope") + source.pop("targets") + _write(policy_path, policy) + first = request(root, registry, "delivery", "builder", "analyst", "default-hop", brief, parent) + assert first["request_id"] + # The source's current exceptions also protect later hops, not only Chat. + policy["sources"][session["channel_id"]]["blocked_targets"] = [{"goal_id": "delivery", "agent_id": "reviewer"}] + _write(policy_path, policy) + with pytest.raises(ValueError, match="reviewer is not authorized"): + request(root, registry, "delivery", "analyst", "reviewer", "blocked-hop", brief, first["request_id"])