From 7511aeaf45a9c7ad02f1aaa6c8d1dce91ed9e3c1 Mon Sep 17 00:00:00 2001 From: valentimarco Date: Tue, 25 Aug 2026 12:28:36 +0200 Subject: [PATCH 01/57] feat(plugin): add opencode v2 port with event-driven architecture - Add src/v2/index.ts: full port using Plugin.define + ctx.event.subscribe() - Add docs/v2/migration.md: migration notes and API mapping - Update README.md: v2 callout banner and install instructions Ports from v1 hooks-object API to v2 promise-plugin API: - Event pump via AsyncIterable instead of event hook callback - Flattened SDK calls (session.prompt({sessionID, text})) - Assistant text accumulated from streaming events (no message history access) - New guards: permission-aware pausing, subagent-aware waiting, failure-recovery arming - All detection/recovery features preserved: stall watchdog, failure recovery, tool-call-as-text, ready-to-continue nudges, hallucination loop guard, etc. Tested against opencode2 v0.0.0-beta-18050, @opencode-ai/plugin 0.0.0-next-17403. --- README.md | 26 ++ docs/v2/migration.md | 155 +++++++ src/v2/index.ts | 931 +++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 1112 insertions(+) create mode 100644 docs/v2/migration.md create mode 100644 src/v2/index.ts diff --git a/README.md b/README.md index 56213b3..abb7ec7 100644 --- a/README.md +++ b/README.md @@ -2,6 +2,10 @@ **Plugin for [OpenCode](https://github.com/anomalyco/opencode) that automatically detects and recovers from LLM session failures — stalls, broken tool calls, hallucination loops, stuck subagent parents, and more. Fully silent, zero UI pollution.** +> **OpenCode v2 is here.** This repo now ships both the v1 plugin (`src/index.ts`) +> and a complete v2 port (`src/v2/index.ts`). See [docs/v2/migration.md](docs/v2/migration.md) +> for the full migration notes, or jump to the [Installation](#installation) section. + ## What it does LLM sessions fail in predictable ways. This plugin monitors all sessions and automatically recovers without user intervention. Each recovery path below references the upstream OpenCode issues that motivated it — these are problems not yet resolved in the official project. @@ -303,6 +307,28 @@ With options: } ``` +### OpenCode v2 + +The v2 plugin uses the new `Plugin.define` API with `ctx.event.subscribe()` (AsyncIterable) instead of the v1 hooks-object pattern. Add to your `opencode.json`: + +```jsonc +{ + "plugins": [ + { + "package": "./plugins/auto-resume-v2.ts", + "options": { + "chunkTimeoutMs": 45000, + "maxRetries": 3 + } + } + ] +} +``` + +Or place `src/v2/index.ts` directly in `~/.config/opencode/plugins/` for auto-discovery (no config entry needed). + +Disable via `"-auto-resume.v2"` in `plugins`. Full v2 migration notes: [docs/v2/migration.md](docs/v2/migration.md). + ### Configurable options | Option | Default | Description | diff --git a/docs/v2/migration.md b/docs/v2/migration.md new file mode 100644 index 0000000..e3c52f2 --- /dev/null +++ b/docs/v2/migration.md @@ -0,0 +1,155 @@ +# opencode-auto-resume — OpenCode v2 port: what changed + +> Saved from ~/.config/opencode/plugins/README.md on 2026-08-25. +> Drop this into the PR description (or keep as docs/v2-migration.md upstream). +> Target: https://github.com/Mte90/opencode-auto-resume +> Runtime tested against: opencode2 v0.0.0-beta-18050, @opencode-ai/plugin 0.0.0-next-17403 + +## Summary + +Ports the plugin from the v1 hooks API (`Plugin` factory returning a hooks +object) to the v2 promise-plugin API (`Plugin.define({ id, setup })` + +`ctx.event.subscribe()`). All detection/recovery features are preserved. +Strict-mode typechecked against the real installed `@opencode-ai/plugin` +types; validated by a mocked-context runtime suite covering every recovery +path (12/12 checks). + +## 1. Config key renamed: `plugin` → `plugins` + +```jsonc +// v1 (removed) +{ "plugin": ["some-package", ["./local.ts", { "opt": 1 }]] } + +// v2 +{ + "plugins": [ + "some-package", + { "package": "./local.ts", "options": { "opt": 1 } } + ] +} +``` + +The tuple form `[path, options]` is gone — use the object form with +`package` + `options`. Local paths must start with `./` or `../` and resolve +relative to the config file. Disable directives (`"-plugin-id"`, `"*"`) +still work and are matched against the plugin's exported `id`. + +## 2. Module shape: hooks-object → `{ id, setup }` + +```ts +// v1: export a factory that returns a Hooks object +export const MyPlugin: Plugin = async ({ client, $, directory }) => ({ + event: async ({ event }) => { /* ... */ }, + "tool.execute.before": async (input, output) => { /* ... */ }, +}) + +// v2: default-export Plugin.define with a unique id and a setup fn. +// setup registers long-lived behavior and MAY return a cleanup function, +// which OpenCode awaits on disable/reload/shutdown. +import { Plugin } from "@opencode-ai/plugin" + +export default Plugin.define({ + id: "acme.thing", // unique — used by disable directives + setup: async (ctx) => { + /* register timers/subscriptions here */ + return () => { /* cleanup */ } + }, +}) +``` + +Legacy single-function modules are still loaded through a compatibility +shim, but new code should target v2 directly. + +## 3. Context capabilities replace most hooks + +The v2 context is essentially a typed server client plus registration APIs: + +| v2 capability | Replaces (v1) | +|---|---| +| `ctx.event.subscribe()` → **AsyncIterable** of events | `event` hook | +| `ctx.session.prompt/interrupt/create/get/command/synthetic/generate` + `ctx.session.hook("context")` | direct SDK calls / chat message hooks | +| `ctx.tool.transform` / `ctx.tool.hook` | `tool` map, `tool.execute.before/after` | +| `ctx.agent.transform` | agent config mutation | +| `ctx.options` | second `options` argument of the v1 factory | +| `ctx.app` | `{ name, version, channel }` only — **no `app.log()`** | + +Notes: +- **Subscribe, don't block**: start the event pump in the background inside + `setup`; do not `await` an infinite loop there. +- **Flattened SDK calls**: nested `{ path, query, body }` envelopes are gone. + Example: `session.prompt({ path: { id }, body: { parts } })` became + `session.prompt({ sessionID, text })`. +- **No history access**: the v2 plugin context does not expose + `message.list()`. Plugins that need assistant text must accumulate it from + streaming events (`session.text.delta`, `session.reasoning.delta`). +- **Logging**: write to the console with a prefix instead of `app.log()`; + opencode captures stdout/stderr into its logs. + +## 4. Event payload shapes changed + +v1 events were `{ type, properties }`; v2 events are flat +`{ type, created, data }`. Useful v2 session events for watchdog-style +plugins: `session.execution.started/succeeded/failed/interrupted`, +`session.step.started/ended/failed`, `session.text.delta`, +`session.reasoning.delta`, `session.tool.called/progress/success/failed`, +`session.retry.scheduled`, `session.idle`, `permission.asked/replied`. + +## 5. Concrete changes in this port + +- v1 hooks → v2 `Plugin.define({ id: "auto-resume.v2", setup })` + + background `ctx.event.subscribe()` pump; cleanup returned from `setup` + clears the watchdog interval. +- `client.session.prompt({path, body})` → `ctx.session.prompt({ sessionID, + text })` (a nested-shape fallback is kept while the beta API settles). +- `session.messages()` forensics → live accumulation of assistant text from + `session.text.delta` / `session.reasoning.delta` events (message history + is not reachable from the v2 plugin context). +- `ctx.client.app.log(...)` → prefixed console logging. +- Status polling demoted: session state comes primarily from lifecycle + events; a low-frequency interval cross-checks active sessions for silence + only. +- New guards made possible/necessary by v2 semantics: + - `permission.asked` / `permission.replied` hold off recovery while a + permission dialog is open. + - Subagent-aware waiting: a parent silent right after dispatching a task + tool with other sessions running is left alone. + - Failure-triggered recoveries arm a flag so the failure's own idle + transition doesn't cancel the scheduled retry prompt, while a healthy + completion still cancels stale ones. + +## Feature parity retained + +Stall watchdog · failure recovery with exponential backoff → interrupt+resume +escalation · tool-call-as-text detection · "ready to continue"/action-intent +nudges · done-claim verification · hallucination-loop guard (N continues in +window → abort+resume) · tool-loop pattern detection · user-interrupt +respected. + +## Options + +Unchanged names/defaults from v1: `chunkTimeoutMs` (45000), `gracePeriodMs` +(3000), `checkIntervalMs` (5000), `maxRetries` (3), `baseBackoffMs` (1000), +`maxBackoffMs` (8000), `loopMaxContinues` (3), `loopWindowMs` (600000), +plus `maxRecoveryRetries`, `continuePrompt`, `debug`. + +```jsonc +{ + "plugins": [ + { + "package": "./plugins/auto-resume-v2.ts", + "options": { "chunkTimeoutMs": 45000, "maxRetries": 3 } + } + ] +} +``` + +Disable by id without touching other plugins: add `"-auto-resume.v2"`. + +## Testing + +- `tsc --strict --noEmit` clean against `@opencode-ai/plugin@0.0.0-next-17403`. +- Mocked-context runtime suite (bun): export shape · stall watchdog → prompt · + ready-to-continue nudge · tool-call-as-text recovery · loop guard → interrupt · + permission-hold blocks recovery · execution-failure recovery · stale recovery + does not disturb healthy idle sessions · user interrupt honored · cleanup + quiesces timers — **12/12 passing**. diff --git a/src/v2/index.ts b/src/v2/index.ts new file mode 100644 index 0000000..57ee468 --- /dev/null +++ b/src/v2/index.ts @@ -0,0 +1,931 @@ +/** + * opencode-auto-resume — adapted for OpenCode v2 plugin API. + * + * Port of https://github.com/Mte90/opencode-auto-resume (v1 hooks API) to the + * v2 promise-plugin API (`Plugin.define` + `ctx.event.subscribe()`). + * + * What changed vs. v1: + * - Default export is `{ id, setup }` via `Plugin.define`; setup returns a cleanup fn. + * - Events come from `ctx.event.subscribe()` (AsyncIterable) with flat payloads + * (`{ type, data }`) instead of the v1 `event` hook (`{ type, properties }`). + * - SDK calls are flattened: `ctx.session.prompt({ sessionID, text })`, + * `ctx.session.interrupt({ sessionID })`, `ctx.session.active()`. + * - There is no message-history access from the v2 plugin context, so assistant + * text is reconstructed from `session.text.delta` events instead of + * `session.messages()` polling. + * - No `ctx.app.log` in v2 — logs go to the console (captured by opencode logs). + * + * Detection/recovery features ported: + * - Stalled stream watchdog (busy session with no events for chunkTimeoutMs) + * - Execution/step failure recovery (`session.execution.failed`, `session.step.failed`) + * - Provider retry awareness (`session.retry.scheduled`) + * - Tool-call-printed-as-text detection + targeted recovery prompt + * - "Ready to continue" / stalled-intent ("...:") nudges + * - "Done" claim without work verification + * - Hallucination loop guard (too many auto-continues → interrupt + resume) + * - Subagent-awareness: does not recover a parent blocked on a running subagent + * - Permission-awareness: never recovers while a permission dialog is open + * + * Install: drop this file in ~/.config/opencode/plugins/ (auto-loaded), or add: + * { "plugins": [{ "package": "./plugins/auto-resume-v2.ts", "options": { ... } }] } + */ + +import { Plugin } from "@opencode-ai/plugin" + +// --------------------------------------------------------------------------- +// Types +// --------------------------------------------------------------------------- + +interface ToolCallRecord { + toolName: string + at: number +} + +/** Minimal structural view of a V2Event (avoids depending on client internals). */ +interface V2Event { + type: string + created: number + data?: Record +} + +interface SessionWatch { + createdAt: number + lastActivityAt: number + status: "busy" | "idle" | "unknown" + userCancelled: boolean + resumeAttempts: number + lastRetryAt: number + gaveUp: boolean + aborting: boolean + recovering: boolean + + // Assistant text accumulation (v2 replacement for session.messages()) + textParts: Map // assistantMessageID -> accumulated text + lastAssistantText: string + lastAssistantMessageID: string | null + + // Recovery budgets + toolTextAttempts: number + continueTimestamps: number[] + doneClaimAttempts: number + intentNudgeAttempts: number + /** Set when a failure-triggered recovery is pending, so the delayed prompt isn't cancelled by the failure's own idle transition. */ + pendingRecoveryArmed: boolean + + // Guards + permissionPending: boolean + waitingOnSubagent: boolean + lastWasTaskTool: boolean + idleSince: number | null + + // Tool-loop tracking + recentToolCalls: ToolCallRecord[] + + // Agent/model from last step (informational logging only; sessions are stateful in v2) + agent?: string + model?: string +} + +export interface AutoResumeOptions { + chunkTimeoutMs?: number + gracePeriodMs?: number + checkIntervalMs?: number + maxRetries?: number + baseBackoffMs?: number + maxBackoffMs?: number + loopMaxContinues?: number + loopWindowMs?: number + maxRecoveryRetries?: number + continuePrompt?: string + toolTextRecoveryPrompt?: string + doneWithoutWorkPrompt?: string + actionIntentPrompt?: string + debug?: boolean +} + +// --------------------------------------------------------------------------- +// Constants & defaults +// --------------------------------------------------------------------------- + +const DEFAULT_CHUNK_TIMEOUT_MS = 45_000 +const DEFAULT_CHECK_INTERVAL_MS = 5_000 +const DEFAULT_GRACE_PERIOD_MS = 3_000 +const DEFAULT_MAX_RETRIES = 3 +const DEFAULT_BASE_BACKOFF_MS = 1_000 +const DEFAULT_MAX_BACKOFF_MS = 8_000 +const DEFAULT_LOOP_MAX_CONTINUES = 3 +const DEFAULT_LOOP_WINDOW_MS = 10 * 60_000 +const DEFAULT_MAX_RECOVERY_RETRIES = 2 +const DEFAULT_DEBUG = false + +const MAX_IDLE_SESSIONS = 50 +const IDLE_CLEANUP_MS = 10 * 60_000 +const TEXT_BUFFER_TRIM_LEN = 20_000 + +const CONTINUE_PROMPT = "continue" + +const TOOL_TEXT_RECOVERY_PROMPT = + "Your last message contained a raw tool call printed as text instead of being executed. " + + "Please use the proper tool calling mechanism to execute it." + +const DONE_WITHOUT_WORK_PROMPT = + "I need you to verify more carefully that you have actually completed all the required tasks. " + + "Your response indicated you're done, but no work was detected. Please check your todo list " + + "and complete any remaining work." + +const TOOL_LOOP_RECOVERY_PROMPT = + "I notice you've been calling the same tool multiple times in a row without making progress. " + + "Please step back and reassess your approach. Consider: " + + "1) Are you stuck in a loop? 2) Do you need different information first? " + + "3) Should you try a different tool or break the task into smaller steps? " + + "Take a moment to think about what's blocking you and propose a different strategy." + +const TASK_TOOL_HINTS = ["task", "agent", "subagent", "dispatch"] + +// --------------------------------------------------------------------------- +// Pattern lists (ported verbatim from upstream where pure) +// --------------------------------------------------------------------------- + +const TOOL_TEXT_PATTERNS = [ + //i, + /<\/function>/i, + //i, + /<\/parameter>/i, + /]/i, + /<\/tool_call>/i, + /]*)?\s*(?:\/>|>)/i, + /{"type":\s*"function"/i, + /{"name":\s*"[a-zA-Z_]/i, + /\{\s*"type"\s*:?$/im, + /\{\s*"name"\s*:?$/im, +] + +const TRUNCATED_XML_PATTERNS = [ + { open: /]*>/i, close: /<\/function>/i }, + { open: /]*>/i, close: /<\/parameter>/i }, + { open: /]*>/i, close: /<\/tool_call>/i }, + { open: /\{\s*"type"\s*:/i, close: /}/ }, + { open: /\{\s*"name"\s*:/i, close: /}/ }, +] + +const READY_TO_CONTINUE_PATTERNS = [ + /ready to continue with task/i, + /continuing with task/i, + /continue with task/i, + /proceeding with task/i, + /ready to proceed with task/i, + /will continue with task/i, + /moving on to task/i, +] + +const DONE_CLAIM_PATTERNS = [ + /^task\s+done[.!]*$/im, + /^done[.!]*$/im, + /^all\s+done[.!]*$/im, + /^finished[.!]*$/im, + /^complete[.!]*$/im, + /^task\s+complete[.!]*$/im, + /^task\s+completed[.!]*$/im, + /^all\s+tasks?\s+complete[.!]*$/im, + /^all\s+tasks?\s+completed[.!]*$/im, + /^(?:i['’]?m\s+)?done\s+with\s+task/im, + /\bdone\s+with\s+(?:the\s+)?(?:task|work|implementation)/im, + /\bfinished\s+(?:the\s+)?(?:task|work|implementation)/im, + /\b(?:all|everything)\s+(?:is\s+)?(?:complete|done|finished)/im, + /\bnothing\s+(?:else\s+)?(?:left|remaining|to do)/im, +] + +// Error signatures that indicate a transient streaming/provider failure worth retrying +const STREAMING_FAILURE_MESSAGE_PATTERNS = [ + "stream.*fail", + "stream.*timeout", + "connection.*reset", + "connection.*closed", + "connection.*error", + "socket.*hang", + "econnreset", + "etimedout", + "rate.?limit", + "overloaded", + "server error", + "internal error", +] + +// --------------------------------------------------------------------------- +// Pure helpers +// --------------------------------------------------------------------------- + +function stripCodeBlocks(text: string): string { + return text.replace(/```[\s\S]*?```/g, "").replace(/`[^`\n]+`/g, "") +} + +function containsToolCallAsText(text: string): boolean { + if (text.length <= 10) return false + const stripped = stripCodeBlocks(text) + if (TOOL_TEXT_PATTERNS.some((pat) => pat.test(stripped))) return true + for (const { open, close } of TRUNCATED_XML_PATTERNS) { + if (open.test(stripped) && !close.test(stripped)) return true + } + return false +} + +function containsReadyToContinuePattern(text: string): boolean { + const lines = text.split("\n") + const lastLines = lines.slice(-3).join("\n") + return READY_TO_CONTINUE_PATTERNS.some((pat) => pat.test(lastLines)) +} + +function containsDoneClaimPattern(text: string): boolean { + const lines = text.split("\n") + const lastLines = lines.slice(-5).join("\n") + return DONE_CLAIM_PATTERNS.some((pat) => pat.test(lastLines)) +} + +/** Model ends with ":" announcing intent without executing. */ +function containsActionIntent(text: string): boolean { + if (text.length <= 15) return false + const cleaned = text.replace(/<[a-zA-Z/?][^>]*>/g, "").trim() + const lines = cleaned.split("\n") + let lastLine = "" + for (let i = lines.length - 1; i >= 0; i--) { + if (lines[i].trim().length > 0) { + lastLine = lines[i].trim() + break + } + } + return lastLine.endsWith(":") && lastLine.length > 5 && lastLine.length < 500 +} + +function backoffMs(attempt: number, base: number, max: number): number { + return Math.min(base * Math.pow(2, attempt - 1), max) +} + +function isStreamingFailure(message: string): boolean { + const lower = message.toLowerCase() + if (!lower) return false + return STREAMING_FAILURE_MESSAGE_PATTERNS.some((pattern) => { + try { + return new RegExp(pattern, "i").test(lower) + } catch { + return lower.includes(pattern.toLowerCase()) + } + }) +} + +/** Repeating tool-call patterns: A-A-A or A-B-A-B-A-B etc. */ +function detectPatternLoop(recentTools: string[]): boolean { + if (recentTools.length < 6) return false + for (const patternLen of [1, 2, 3]) { + if (recentTools.length < patternLen * 3) continue + const pattern = recentTools.slice(-patternLen) + let matches = 0 + for (let i = recentTools.length - patternLen * 2; i >= 0; i -= patternLen) { + const slice = recentTools.slice(i, i + patternLen) + if (slice.length !== patternLen) break + if (!slice.every((t, idx) => t === pattern[idx])) break + matches++ + } + if (matches >= 2) return true + } + return false +} + +function trackToolCall(w: SessionWatch, toolName: string): boolean { + const now = Date.now() + w.recentToolCalls = w.recentToolCalls.filter((c) => now - c.at < 120_000) + w.recentToolCalls.push({ toolName, at: now }) + const recentTools = w.recentToolCalls.slice(-12).map((c) => c.toolName) + if (recentTools.length < 6) return false + const lastTool = recentTools[recentTools.length - 1] + const consecutiveSame = recentTools.slice(-4).filter((t) => t === lastTool).length + if (consecutiveSame >= 4) return true + return detectPatternLoop(recentTools) +} + +function short(sid: string): string { + return sid.length > 12 ? `…${sid.slice(-8)}` : sid +} + +function sidOf(ev: V2Event): string | undefined { + const sid = ev.data?.sessionID + return typeof sid === "string" ? sid : undefined +} + +function isTaskToolCall(ev: V2Event): boolean { + const tool = ev.data?.tool ?? ev.data?.toolName + if (typeof tool === "string") { + const lower = tool.toLowerCase() + if (TASK_TOOL_HINTS.some((h) => lower.includes(h))) return true + } + // Also inspect input for agent-ish payloads + const desc = ev.data?.input?.description ?? ev.data?.input?.subagent_type ?? ev.data?.input?.agent + return typeof desc === "string" +} + +// --------------------------------------------------------------------------- +// Plugin +// --------------------------------------------------------------------------- + +export default Plugin.define({ + id: "auto-resume.v2", + + setup: async (ctx) => { + const opts = (ctx.options ?? {}) as AutoResumeOptions + + const chunkTimeoutMs = opts.chunkTimeoutMs ?? DEFAULT_CHUNK_TIMEOUT_MS + const checkIntervalMs = opts.checkIntervalMs ?? DEFAULT_CHECK_INTERVAL_MS + const gracePeriodMs = opts.gracePeriodMs ?? DEFAULT_GRACE_PERIOD_MS + const maxRetries = opts.maxRetries ?? DEFAULT_MAX_RETRIES + const baseBackoff = opts.baseBackoffMs ?? DEFAULT_BASE_BACKOFF_MS + const maxBackoff = opts.maxBackoffMs ?? DEFAULT_MAX_BACKOFF_MS + const loopMaxContinues = opts.loopMaxContinues ?? DEFAULT_LOOP_MAX_CONTINUES + const loopWindowMs = opts.loopWindowMs ?? DEFAULT_LOOP_WINDOW_MS + const maxRecoveryRetries = opts.maxRecoveryRetries ?? DEFAULT_MAX_RECOVERY_RETRIES + const debug = opts.debug ?? DEFAULT_DEBUG + + const dbg = (...args: unknown[]) => { + if (debug) console.log("[auto-resume:debug]", ...args) + } + + function log(level: "info" | "warn" | "error", msg: string) { + const line = `[auto-resume] ${msg}` + if (level === "error") console.error(line) + else if (level === "warn") console.warn(line) + else console.log(line) + } + + // --------------------------------------------------------------------- + // State + // --------------------------------------------------------------------- + + const sessions = new Map() + + function ensureWatch(sid: string): SessionWatch { + let w = sessions.get(sid) + if (!w) { + w = { + createdAt: Date.now(), + lastActivityAt: Date.now(), + status: "unknown", + userCancelled: false, + resumeAttempts: 0, + lastRetryAt: 0, + gaveUp: false, + aborting: false, + recovering: false, + textParts: new Map(), + lastAssistantText: "", + lastAssistantMessageID: null, + toolTextAttempts: 0, + continueTimestamps: [], + doneClaimAttempts: 0, + intentNudgeAttempts: 0, + pendingRecoveryArmed: false, + permissionPending: false, + waitingOnSubagent: false, + lastWasTaskTool: false, + idleSince: null, + recentToolCalls: [], + } + sessions.set(sid, w) + } + return w + } + + function touch(sid: string) { + const w = ensureWatch(sid) + w.lastActivityAt = Date.now() + } + + function markBusy(sid: string) { + const w = ensureWatch(sid) + if (w.status !== "busy") { + dbg(`${short(sid)} idle/unknown -> busy`) + // Fresh busy cycle: reset per-turn budgets. + // NOTE: continueTimestamps is intentionally preserved — the + // hallucination-loop detector counts across busy cycles by design. + w.resumeAttempts = 0 + w.toolTextAttempts = 0 + w.doneClaimAttempts = 0 + w.intentNudgeAttempts = 0 + w.gaveUp = false + w.recentToolCalls = [] + w.textParts.clear() + w.lastAssistantText = "" + w.waitingOnSubagent = false + } + w.status = "busy" + w.idleSince = null + w.lastActivityAt = Date.now() + } + + function markIdle(sid: string) { + const w = ensureWatch(sid) + if (w.status !== "idle") { + dbg(`${short(sid)} ${w.status} -> idle`) + w.status = "idle" + w.idleSince = Date.now() + } + w.permissionPending = false + } + + function recordContinue(sid: string) { + const w = sessions.get(sid) + if (!w) return + const now = Date.now() + w.continueTimestamps.push(now) + const cutoff = now - loopWindowMs + while (w.continueTimestamps.length > 0 && w.continueTimestamps[0] < cutoff) { + w.continueTimestamps.shift() + } + } + + function continuesInWindow(w: SessionWatch): number { + const cutoff = Date.now() - loopWindowMs + while (w.continueTimestamps.length > 0 && w.continueTimestamps[0] < cutoff) { + w.continueTimestamps.shift() + } + return w.continueTimestamps.length + } + + function cleanupIdleSessions() { + const now = Date.now() + const busy = new Set() + for (const [sid, w] of sessions) { + if (w.status === "busy") busy.add(sid) + } + let idleCount = 0 + const toDelete: string[] = [] + for (const [sid, w] of sessions) { + if (w.status !== "busy") { + idleCount++ + if (w.idleSince && now - w.idleSince > IDLE_CLEANUP_MS) toDelete.push(sid) + } + } + if (idleCount > MAX_IDLE_SESSIONS) { + const entries: Array<{ sid: string; since: number }> = [] + for (const [sid, w] of sessions) { + if (w.status !== "busy" && w.idleSince) entries.push({ sid, since: w.idleSince }) + } + entries.sort((a, b) => a.since - b.since) + const excess = idleCount - MAX_IDLE_SESSIONS + for (let i = 0; i < excess && i < entries.length; i++) { + if (!toDelete.includes(entries[i].sid)) toDelete.push(entries[i].sid) + } + } + for (const sid of toDelete) sessions.delete(sid) + if (toDelete.length > 0) dbg(`cleaned ${toDelete.length} idle sessions, total=${sessions.size}`) + } + + /** + * Accumulate assistant text from deltas. This replaces v1's + * `session.messages()` polling, which no longer exists on the v2 ctx. + */ + function appendText(sid: string, messageID: string | undefined, delta: string) { + const w = ensureWatch(sid) + const mid = messageID ?? "_anon" + w.textParts.set(mid, (w.textParts.get(mid) ?? "") + delta) + w.lastAssistantMessageID = mid + w.lastAssistantText = w.textParts.get(mid) ?? "" + // Keep memory bounded: keep only the two most recent messages' text + if (w.textParts.size > 2) { + const oldest = w.textParts.keys().next().value + if (oldest !== undefined && oldest !== mid) w.textParts.delete(oldest) + } + if (w.lastAssistantText.length > TEXT_BUFFER_TRIM_LEN) { + w.textParts.set(mid, w.lastAssistantText.slice(-TEXT_BUFFER_TRIM_LEN)) + w.lastAssistantText = w.textParts.get(mid) ?? "" + } + } + + // --------------------------------------------------------------------- + // Recovery actions + // --------------------------------------------------------------------- + + async function sendPrompt(sid: string, text: string): Promise { + try { + await ctx.session.prompt({ sessionID: sid, text }) + return true + } catch (err) { + const msg = err instanceof Error ? err.message : String(err) + log("warn", `${short(sid)} prompt failed: ${msg}`) + // One flattened-shape fallback for beta drift safety + try { + await (ctx.session as any).prompt({ path: { id: sid }, body: { parts: [{ type: "text", text }] } }) + return true + } catch { + log("error", `${short(sid)} prompt failed twice: ${msg}`) + return false + } + } + } + + async function tryAbortAndResume(sid: string, w: SessionWatch): Promise { + if (w.aborting) return false + w.aborting = true + log("warn", `${short(sid)} escalating: interrupt + fresh continue`) + try { + await ctx.session.interrupt({ sessionID: sid }) + } catch (err) { + const msg = err instanceof Error ? err.message : String(err) + log("warn", `${short(sid)} interrupt failed: ${msg}`) + } + // Give the runtime a beat to settle the interrupted turn + await new Promise((r) => setTimeout(r, 2_000)) + w.aborting = false + w.resumeAttempts = 0 + const ok = await sendPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT) + if (ok) { + recordContinue(sid) + w.lastRetryAt = Date.now() + log("info", `${short(sid)} resumed after abort`) + } + return ok + } + + /** + * Core recovery ladder for a stuck/failed session. + * plain continue with backoff -> more attempts -> abort+resume escalation. + */ + async function recover(sid: string, reason: string) { + const w = ensureWatch(sid) + if (w.recovering || w.aborting || w.gaveUp || w.userCancelled || w.permissionPending) return + // Record intent first, then evaluate the loop guard, so the Nth + // continue within the window is the one that escalates. + recordContinue(sid) + if (continuesInWindow(w) >= loopMaxContinues) { + log("warn", `${short(sid)} hallucination loop (${loopMaxContinues} continues/${loopWindowMs / 1000}s) — abort+resume`) + await tryAbortAndResume(sid, w) + w.continueTimestamps = [] + return + } + if (w.resumeAttempts >= maxRetries) { + log("warn", `${short(sid)} giving up after ${maxRetries} attempts (${reason})`) + w.gaveUp = true + return + } + w.recovering = true + w.resumeAttempts++ + const delay = backoffMs(w.resumeAttempts, baseBackoff, maxBackoff) + const attempt = w.resumeAttempts + log( + "info", + `${short(sid)} stall detected (${reason}) — resume attempt ${attempt}/${maxRetries} in ${delay}ms`, + ) + setTimeout(async () => { + try { + // Skip only if the session genuinely turned healthy again + // (a normal completion clears pendingRecoveryArmed) or the + // user took over. + if (w.userCancelled || w.gaveUp) return + if (w.status === "idle" && !w.pendingRecoveryArmed) return // recovered by itself meanwhile + const ok = await sendPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT) + w.pendingRecoveryArmed = false + if (ok) { + w.lastRetryAt = Date.now() + touch(sid) + markBusy(sid) + } else if (attempt >= maxRetries) { + await tryAbortAndResume(sid, w) + } + } finally { + w.recovering = false + } + }, delay) + } + + /** Targeted recovery prompts (tool-as-text, done-claims, intent nudges). */ + async function targetedRecovery(sid: string, kind: string, prompt: string, budgetKey: "toolTextAttempts" | "doneClaimAttempts" | "intentNudgeAttempts") { + const w = ensureWatch(sid) + if (w.recovering || w.userCancelled || w.permissionPending) return + recordContinue(sid) + if (continuesInWindow(w) >= loopMaxContinues) { + log("warn", `${short(sid)} loop guard before ${kind} nudge — abort+resume`) + await tryAbortAndResume(sid, w) + w.continueTimestamps = [] + return + } + if (w[budgetKey] >= maxRetries) { + dbg(`${short(sid)} ${kind} budget exhausted`) + return + } + w[budgetKey]++ + log("info", `${short(sid)} ${kind} detected — sending targeted prompt (${w[budgetKey]}/${maxRetries})`) + w.recovering = true + const ok = await sendPrompt(sid, prompt) + w.recovering = false + if (ok) { + recordContinue(sid) + touch(sid) + markBusy(sid) + } + } + + // --------------------------------------------------------------------- + // Idle-time forensics (runs once when a turn finishes) + // --------------------------------------------------------------------- + + async function inspectOnIdle(sid: string) { + const w = ensureWatch(sid) + const text = w.lastAssistantText + if (!text) return + + if (containsToolCallAsText(text)) { + await targetedRecovery(sid, "tool-call-as-text", opts.toolTextRecoveryPrompt ?? TOOL_TEXT_RECOVERY_PROMPT, "toolTextAttempts") + return + } + if (containsReadyToContinuePattern(text)) { + await targetedRecovery(sid, "ready-to-continue", opts.continuePrompt ?? CONTINUE_PROMPT, "intentNudgeAttempts") + return + } + if (containsActionIntent(text)) { + await targetedRecovery(sid, "action-intent", opts.actionIntentPrompt ?? opts.continuePrompt ?? CONTINUE_PROMPT, "intentNudgeAttempts") + return + } + if (containsDoneClaimPattern(text) && w.doneClaimAttempts < 1) { + // Single verification nudge for suspiciously terse completions + const trimmed = text.trim() + if (trimmed.length < 400) { + await targetedRecovery(sid, "done-claim-no-details", opts.doneWithoutWorkPrompt ?? DONE_WITHOUT_WORK_PROMPT, "doneClaimAttempts") + } + } + } + + // --------------------------------------------------------------------- + // Watchdog timer + // --------------------------------------------------------------------- + + /** + * `session.active()` exists on the v2 client but is not part of the + * plugin SessionDomain pick — access defensively; empty map if missing. + */ + async function getActiveSessions(): Promise { + try { + const fn = (ctx.session as any).active + if (typeof fn !== "function") return [] + const active = await fn.call(ctx.session) + return Object.keys(active ?? {}).filter((k) => typeof k === "string") + } catch { + return [] + } + } + + async function checkActiveSessions() { + const now = Date.now() + + // Cross-check our busy set against server truth (also catches sessions + // we never saw an execution.started for, e.g. after plugin reload). + const activeIDs = await getActiveSessions() + for (const sid of activeIDs) { + const w = ensureWatch(sid) + if (w.status !== "busy") markBusy(sid) + } + + for (const [sid, w] of sessions) { + if (w.status !== "busy" || w.userCancelled) continue + const silence = now - w.lastActivityAt + if (silence < chunkTimeoutMs + gracePeriodMs) continue + if (w.permissionPending) { + dbg(`${short(sid)} silent but permission pending — skipping`) + continue + } + if (w.waitingOnSubagent) { + dbg(`${short(sid)} silent but waiting on subagent — skipping`) + continue + } + // If another session is actively running and this one went silent + // right after dispatching a task tool, treat it as a parent wait. + if (w.lastWasTaskTool) { + const others = (await getActiveSessions()).filter((s) => s !== sid) + if (others.length > 0) { + dbg(`${short(sid)} silent after task tool with ${others.length} active child(ren) — waiting`) + continue + } + } + await recover(sid, `no activity for ${Math.ceil(silence / 1000)}s`) + } + cleanupIdleSessions() + } + + const watchdog = setInterval(() => { + checkActiveSessions().catch((e) => + log("error", `watchdog failed: ${e instanceof Error ? e.message : String(e)}`), + ) + }, checkIntervalMs) + + // --------------------------------------------------------------------- + // Event stream + // --------------------------------------------------------------------- + + function handleEvent(ev: V2Event) { + switch (ev.type) { + // --- lifecycle ------------------------------------------------- + case "session.execution.started": { + const sid = sidOf(ev) + if (!sid) return + markBusy(sid) + return + } + case "session.execution.succeeded": + case "session.idle": { + const sid = sidOf(ev) + if (!sid) return + markIdle(sid) + ensureWatch(sid).pendingRecoveryArmed = false + void inspectOnIdle(sid) + return + } + case "session.execution.interrupted": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + // Only user interrupts should suppress us; ours reset flags themselves. + w.userCancelled = !w.aborting + w.recovering = false + markIdle(sid) + return + } + case "session.deleted": + case "session.reverted": { + const sid = sidOf(ev) + if (!sid) return + sessions.delete(sid) + return + } + + // --- activity -------------------------------------------------- + case "session.step.started": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + w.agent = typeof ev.data?.agent === "string" ? ev.data.agent : w.agent + w.model = + ev.data?.model && typeof ev.data.model === "object" + ? `${ev.data.model.providerID ?? ev.data.model.provider ?? "?"}/${ev.data.model.modelID ?? ev.data.model.id ?? "?"}` + : w.model + w.lastWasTaskTool = false + markBusy(sid) + return + } + case "session.step.ended": { + const sid = sidOf(ev) + if (!sid) return + touch(sid) + return + } + case "session.text.delta": + case "session.reasoning.delta": { + const sid = sidOf(ev) + if (!sid) return + appendText(sid, ev.data?.assistantMessageID, typeof ev.data?.delta === "string" ? ev.data.delta : "") + touch(sid) + return + } + case "session.text.ended": + case "session.reasoning.ended": { + const sid = sidOf(ev) + if (!sid) return + touch(sid) + return + } + + // --- tools ------------------------------------------------------ + case "session.tool.called": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + const name = typeof ev.data?.tool === "string" ? ev.data.tool : (ev.data?.id as string | undefined) ?? "tool" + w.lastWasTaskTool = isTaskToolCall(ev) + if (trackToolCall(w, name)) { + void targetedRecovery(sid, "tool-loop", TOOL_LOOP_RECOVERY_PROMPT, "intentNudgeAttempts").catch(() => {}) + } + touch(sid) + return + } + case "session.tool.progress": + case "session.shell.started": + case "session.shell.ended": { + const sid = sidOf(ev) + if (!sid) return + touch(sid) + return + } + case "session.tool.success": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + w.lastWasTaskTool = false + touch(sid) + return + } + case "session.tool.failed": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + w.lastWasTaskTool = false + touch(sid) + return + } + + // --- permissions ------------------------------------------------- + case "permission.asked": { + const sid = sidOf(ev) + if (!sid) return + ensureWatch(sid).permissionPending = true + return + } + case "permission.replied": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + w.permissionPending = false + touch(sid) + return + } + + // --- failures ---------------------------------------------------- + case "session.retry.scheduled": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + touch(sid) // provider-level retry is progress of its own kind + const errType = String(ev.data?.error?.type ?? "") + const errMsg = String(ev.data?.error?.message ?? "") + log("info", `${short(sid)} provider retry #${ev.data?.attempt ?? "?"}: ${errType || errMsg}`) + if (!isStreamingFailure(errMsg) && errType !== "retryable") return + // Let the provider retries play out first; only intervene if it stays quiet + if (nowSilenceTooLong(w)) { + w.pendingRecoveryArmed = true + void recover(sid, "streaming failure with scheduled retry") + } + return + } + case "session.step.failed": + case "session.execution.failed": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + const errMsg = String(ev.data?.error?.message ?? "") + const errType = String(ev.data?.error?.type ?? "") + markIdle(sid) + w.pendingRecoveryArmed = true // our delayed recovery must survive this idle transition + if (errMsg.includes("interrupted by user") || errType.includes("cancel")) { + dbg(`${short(sid)} failure was user-initiated — not recovering`) + w.pendingRecoveryArmed = false + return + } + log("warn", `${short(sid)} ${ev.type}: ${errType || "error"} ${errMsg.slice(0, 160)}`) + void recover(sid, `${ev.type}${isStreamingFailure(errMsg) ? " (streaming)" : ""}`) + return + } + + default: + return + } + } + + function nowSilenceTooLong(w: SessionWatch): boolean { + return Date.now() - w.lastActivityAt > chunkTimeoutMs + gracePeriodMs + } + + // Subscribe and pump events in the background (setup must not block). + let running = true + const pump = async () => { + try { + const stream = ctx.event.subscribe() + for await (const raw of stream) { + if (!running) return + const ev = raw as unknown as V2Event + if (!ev || typeof ev.type !== "string") continue + try { + handleEvent(ev) + } catch (e) { + dbg("handler error:", e instanceof Error ? e.message : String(e)) + } + } + } catch (e) { + if (running) log("error", `event stream ended: ${e instanceof Error ? e.message : String(e)}`) + } + } + void pump() + + log( + "info", + `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s`, + ) + + // Cleanup: stop timers and the pump; OpenCode awaits this on disable/reload/shutdown. + return () => { + running = false + clearInterval(watchdog) + sessions.clear() + log("info", "stopped") + } + }, +}) From 5e04aded99bfa92f7f8466e2b60be635c9344774 Mon Sep 17 00:00:00 2001 From: valentimarco Date: Wed, 26 Aug 2026 11:25:34 +0200 Subject: [PATCH 02/57] feat(plugin): show visible notification when recovering from stall/failure MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace session.prompt() with session.synthetic() which injects a visible message into the session timeline (shown in TUI) so the user knows the plugin intervened. The synthetic message also resumes the session when resume=true, replacing the separate prompt call. Notification text per recovery path: - "auto-resume: stalled — retrying" (watchdog stall) - "auto-resume: recovering: " (tool-text, ready-to-continue, etc.) - "auto-resume: abort+resume escalation" (loop guard / max retries) Falls back to session.prompt() if synthetic is not available (beta compat). 12/12 typecheck + smoke tests pass. --- src/v2/index.ts | 43 +++++++++++++++++++++++++++++++++---------- 1 file changed, 33 insertions(+), 10 deletions(-) diff --git a/src/v2/index.ts b/src/v2/index.ts index 57ee468..6f417e5 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -508,20 +508,43 @@ export default Plugin.define({ // Recovery actions // --------------------------------------------------------------------- - async function sendPrompt(sid: string, text: string): Promise { + /** + * Show a visible notification in the session timeline and optionally + * resume it. Uses `session.synthetic()` which is exposed on the v2 + * promise-plugin context — the synthetic message appears in the TUI + * so the user knows the plugin intervened. + * + * When `resume` is true the synthetic also acts as a user turn that + * kicks the session back to life, replacing the separate `prompt()`. + */ + async function notifyAndPrompt(sid: string, text: string, notification: string, resume = true): Promise { try { - await ctx.session.prompt({ sessionID: sid, text }) + await ctx.session.synthetic({ + sessionID: sid, + text, + description: `auto-resume: ${notification}`, + resume, + }) return true } catch (err) { const msg = err instanceof Error ? err.message : String(err) - log("warn", `${short(sid)} prompt failed: ${msg}`) - // One flattened-shape fallback for beta drift safety + log("warn", `${short(sid)} synthetic failed: ${msg}`) + // Flat-shape fallback for beta drift try { - await (ctx.session as any).prompt({ path: { id: sid }, body: { parts: [{ type: "text", text }] } }) + await (ctx.session as any).synthetic({ + path: { id: sid }, + body: { text, description: `auto-resume: ${notification}`, resume }, + }) return true } catch { - log("error", `${short(sid)} prompt failed twice: ${msg}`) - return false + // Last resort: bare prompt + try { + await ctx.session.prompt({ sessionID: sid, text }) + return true + } catch { + log("error", `${short(sid)} all recovery attempts failed: ${msg}`) + return false + } } } } @@ -540,7 +563,7 @@ export default Plugin.define({ await new Promise((r) => setTimeout(r, 2_000)) w.aborting = false w.resumeAttempts = 0 - const ok = await sendPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT) + const ok = await notifyAndPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "abort+resume escalation") if (ok) { recordContinue(sid) w.lastRetryAt = Date.now() @@ -585,7 +608,7 @@ export default Plugin.define({ // user took over. if (w.userCancelled || w.gaveUp) return if (w.status === "idle" && !w.pendingRecoveryArmed) return // recovered by itself meanwhile - const ok = await sendPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT) + const ok = await notifyAndPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "stalled — retrying") w.pendingRecoveryArmed = false if (ok) { w.lastRetryAt = Date.now() @@ -618,7 +641,7 @@ export default Plugin.define({ w[budgetKey]++ log("info", `${short(sid)} ${kind} detected — sending targeted prompt (${w[budgetKey]}/${maxRetries})`) w.recovering = true - const ok = await sendPrompt(sid, prompt) + const ok = await notifyAndPrompt(sid, prompt, "recovering: " + kind) w.recovering = false if (ok) { recordContinue(sid) From ad92b3984767e7bbd1da722304af8788c298b7b3 Mon Sep 17 00:00:00 2001 From: valentimarco Date: Thu, 17 Sep 2026 15:03:54 +0200 Subject: [PATCH 03/57] feat(v2): target stable @opencode/plugin 2.0.5 API - import from @opencode/plugin (stable package; @opencode-ai/plugin was beta-only) - read authoritative assistant text from ctx.session.context() at idle time, keeping the delta accumulation for liveness - pass an AbortSignal to ctx.event.subscribe() and abort it on cleanup - honour session.execution.interrupted data.reason (only "user" disables recovery) - drop the beta-era nested synthetic({ path, body }) fallback - document session.reverted as v1 legacy (v2 emits session.revert.*) --- src/v2/index.ts | 86 ++++++++++++++++++++++++++++++++++++------------- 1 file changed, 63 insertions(+), 23 deletions(-) diff --git a/src/v2/index.ts b/src/v2/index.ts index 6f417e5..8f4fe0e 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -10,10 +10,12 @@ * (`{ type, data }`) instead of the v1 `event` hook (`{ type, properties }`). * - SDK calls are flattened: `ctx.session.prompt({ sessionID, text })`, * `ctx.session.interrupt({ sessionID })`, `ctx.session.active()`. - * - There is no message-history access from the v2 plugin context, so assistant - * text is reconstructed from `session.text.delta` events instead of - * `session.messages()` polling. + * - Assistant text is accumulated from `session.text.delta` events for liveness; + * `ctx.session.context()` (stable v2) supplies the authoritative final + * assistant text at idle time — replacing v1's `session.messages()` polling. * - No `ctx.app.log` in v2 — logs go to the console (captured by opencode logs). + * - Targets the stable v2 API (`@opencode/plugin`). Event names are unchanged + * from the beta port; `session.execution.interrupted` now carries a `reason`. * * Detection/recovery features ported: * - Stalled stream watchdog (busy session with no events for chunkTimeoutMs) @@ -30,7 +32,7 @@ * { "plugins": [{ "package": "./plugins/auto-resume-v2.ts", "options": { ... } }] } */ -import { Plugin } from "@opencode-ai/plugin" +import { Plugin } from "@opencode/plugin" // --------------------------------------------------------------------------- // Types @@ -529,22 +531,14 @@ export default Plugin.define({ } catch (err) { const msg = err instanceof Error ? err.message : String(err) log("warn", `${short(sid)} synthetic failed: ${msg}`) - // Flat-shape fallback for beta drift + // Last resort: a plain prompt still resumes the session (just + // without the visible synthetic notification in the TUI). try { - await (ctx.session as any).synthetic({ - path: { id: sid }, - body: { text, description: `auto-resume: ${notification}`, resume }, - }) + await ctx.session.prompt({ sessionID: sid, text }) return true } catch { - // Last resort: bare prompt - try { - await ctx.session.prompt({ sessionID: sid, text }) - return true - } catch { - log("error", `${short(sid)} all recovery attempts failed: ${msg}`) - return false - } + log("error", `${short(sid)} all recovery attempts failed: ${msg}`) + return false } } } @@ -654,9 +648,40 @@ export default Plugin.define({ // Idle-time forensics (runs once when a turn finishes) // --------------------------------------------------------------------- + /** + * Read the last assistant message's text from the session message history + * (`ctx.session.context()`, stable v2 API). Returns "" when unavailable. + * Guarded so a failure in this forensic path never breaks the watchdog. + */ + async function lastAssistantTextFromContext(sid: string): Promise { + try { + const messages = await ctx.session.context({ sessionID: sid }) + if (!Array.isArray(messages)) return "" + for (let i = messages.length - 1; i >= 0; i--) { + const msg = messages[i] as { + type?: string + content?: Array<{ type?: string; text?: string }> + } + if (!msg || msg.type !== "assistant" || !Array.isArray(msg.content)) continue + const text = msg.content + .filter((part) => part?.type === "text" && typeof part.text === "string") + .map((part) => part.text as string) + .join("") + if (text) return text + } + return "" + } catch (e) { + dbg("session.context() fallback failed:", e instanceof Error ? e.message : String(e)) + return "" + } + } + async function inspectOnIdle(sid: string) { const w = ensureWatch(sid) - const text = w.lastAssistantText + // Prefer the live delta buffer; fall back to the authoritative message + // history when it is empty (e.g. the plugin loaded mid-turn) or stale. + let text = w.lastAssistantText + if (!text) text = await lastAssistantTextFromContext(sid) if (!text) return if (containsToolCallAsText(text)) { @@ -768,13 +793,24 @@ export default Plugin.define({ const sid = sidOf(ev) if (!sid) return const w = ensureWatch(sid) - // Only user interrupts should suppress us; ours reset flags themselves. - w.userCancelled = !w.aborting + // Stable v2 carries the interrupt `reason`. Only a genuine user + // interrupt should suppress recovery; shutdown/superseded/inactivity + // (or a missing reason on older runtimes) must not. + const reason = typeof ev.data?.reason === "string" ? ev.data.reason : undefined + w.userCancelled = !w.aborting && (reason === undefined || reason === "user") w.recovering = false markIdle(sid) return } - case "session.deleted": + case "session.deleted": { + const sid = sidOf(ev) + if (!sid) return + sessions.delete(sid) + return + } + // v1 legacy: not emitted in v2 (reverts now surface as + // `session.revert.cleared` / `session.revert.committed`). Kept + // defensively so older runtimes still drop their watch state. case "session.reverted": { const sid = sidOf(ev) if (!sid) return @@ -918,10 +954,13 @@ export default Plugin.define({ } // Subscribe and pump events in the background (setup must not block). + // Stable v2 recommends passing an AbortSignal so the stream is torn down + // promptly on plugin unload instead of staying suspended in `for await`. let running = true + const eventAbort = new AbortController() const pump = async () => { try { - const stream = ctx.event.subscribe() + const stream = ctx.event.subscribe({ signal: eventAbort.signal }) for await (const raw of stream) { if (!running) return const ev = raw as unknown as V2Event @@ -943,9 +982,10 @@ export default Plugin.define({ `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s`, ) - // Cleanup: stop timers and the pump; OpenCode awaits this on disable/reload/shutdown. + // Cleanup: stop timers and the event pump; OpenCode awaits this on disable/reload/shutdown. return () => { running = false + eventAbort.abort() clearInterval(watchdog) sessions.clear() log("info", "stopped") From aa89252b3e0d3b99f7f40c4a5d25871d5239a170 Mon Sep 17 00:00:00 2001 From: valentimarco Date: Thu, 17 Sep 2026 15:03:54 +0200 Subject: [PATCH 04/57] docs(v2): add install guide and stable migration notes - docs/v2/installing.md: step-by-step v2 install (drop-in + plugins entry), options reference, verification and troubleshooting - docs/v2/migration.md: verified stable-vs-beta delta (package rename, event surface, new session methods, subscribe signal) and updated test notes - README: point at the new install guide and stable API --- README.md | 17 ++++-- docs/v2/installing.md | 136 ++++++++++++++++++++++++++++++++++++++++++ docs/v2/migration.md | 102 ++++++++++++++++++++++++------- 3 files changed, 227 insertions(+), 28 deletions(-) create mode 100644 docs/v2/installing.md diff --git a/README.md b/README.md index abb7ec7..dfe777e 100644 --- a/README.md +++ b/README.md @@ -2,9 +2,11 @@ **Plugin for [OpenCode](https://github.com/anomalyco/opencode) that automatically detects and recovers from LLM session failures — stalls, broken tool calls, hallucination loops, stuck subagent parents, and more. Fully silent, zero UI pollution.** -> **OpenCode v2 is here.** This repo now ships both the v1 plugin (`src/index.ts`) -> and a complete v2 port (`src/v2/index.ts`). See [docs/v2/migration.md](docs/v2/migration.md) -> for the full migration notes, or jump to the [Installation](#installation) section. +> **OpenCode v2 is here (stable).** This repo ships both the v1 plugin +> (`src/index.ts`) and a complete v2 port (`src/v2/index.ts`) targeting the +> stable `@opencode/plugin` 2.0.5 API. Install it with the +> [v2 install guide](docs/v2/installing.md); see +> [docs/v2/migration.md](docs/v2/migration.md) for the full migration notes. ## What it does @@ -307,9 +309,9 @@ With options: } ``` -### OpenCode v2 +### OpenCode v2 (stable) -The v2 plugin uses the new `Plugin.define` API with `ctx.event.subscribe()` (AsyncIterable) instead of the v1 hooks-object pattern. Add to your `opencode.json`: +The v2 plugin uses the `Plugin.define` API with `ctx.event.subscribe()` (AsyncIterable) instead of the v1 hooks-object pattern, and targets the stable **`@opencode/plugin` 2.0.5** API (opencode v2.0.5+). Add to your `opencode.json`: ```jsonc { @@ -327,7 +329,10 @@ The v2 plugin uses the new `Plugin.define` API with `ctx.event.subscribe()` (Asy Or place `src/v2/index.ts` directly in `~/.config/opencode/plugins/` for auto-discovery (no config entry needed). -Disable via `"-auto-resume.v2"` in `plugins`. Full v2 migration notes: [docs/v2/migration.md](docs/v2/migration.md). +Disable via `"-auto-resume.v2"` in `plugins`. + +- **Step-by-step install guide:** [docs/v2/installing.md](docs/v2/installing.md) +- **Migration notes (v1 → v2, stable validation):** [docs/v2/migration.md](docs/v2/migration.md) ### Configurable options diff --git a/docs/v2/installing.md b/docs/v2/installing.md new file mode 100644 index 0000000..00af2d8 --- /dev/null +++ b/docs/v2/installing.md @@ -0,0 +1,136 @@ +# Installing opencode-auto-resume on OpenCode v2 + +This guide installs the **v2 port** of the plugin. It targets the stable v2 +plugin API (`@opencode/plugin` 2.0.5, shipped with `opencode` v2.0.5). + +> Coming from OpenCode v1? Read [`migration.md`](./migration.md) first — the v1 +> plugin file does **not** run on v2 and must be replaced by the v2 port. + +## Requirements + +- **opencode v2.0.5 or newer** (`opencode --version`). +- Node.js/Bun is only needed if you run the test/typecheck tooling; the plugin + itself is loaded by opencode. +- The plugin source: [`src/v2/index.ts`](../../src/v2/index.ts) from this repo. + +## Install + +### Option A — drop-in file (no config) + +OpenCode auto-loads every plugin found in these directories: + +- Global: `~/.config/opencode/plugins/` +- Per project: `/.opencode/plugins/` + +Copy the v2 port there: + +```sh +mkdir -p ~/.config/opencode/plugins +cp src/v2/index.ts ~/.config/opencode/plugins/auto-resume-v2.ts +``` + +That's it — no `opencode.json` change is required. Restart opencode (or reload +plugins) and the plugin is active. + +### Option B — `opencode.json(c)` entry + +Use this when you want to load the file from another location, pass options, or +pin a published package. Add an entry to the `plugins` array: + +```jsonc title="~/.config/opencode/opencode.jsonc" +{ + "$schema": "https://opencode.ai/config.json", + "plugins": [ + // 1. plain path (relative to the config file) or absolute path + "./plugins/auto-resume-v2.ts", + + // 2. with options + { + "package": "./plugins/auto-resume-v2.ts", + "options": { + "chunkTimeoutMs": 45000, + "maxRetries": 3 + } + } + ] +} +``` + +Both `.opencode/plugin/` (v1 directory name) and `.opencode/plugins/` are +discovered; use `.opencode/plugins/` for v2 files. + +## Options + +All options are optional and are read from `ctx.options` in `setup`. + +| Option | Default | Description | +|---|---|---| +| `chunkTimeoutMs` | `45000` | Silence on a **busy** session before recovery is considered. | +| `gracePeriodMs` | `3000` | Extra grace added to the timeout before acting. | +| `checkIntervalMs` | `5000` | Watchdog polling interval. | +| `maxRetries` | `3` | Resume attempts per stall before escalating. | +| `baseBackoffMs` | `1000` | Base delay for exponential backoff between attempts. | +| `maxBackoffMs` | `8000` | Ceiling for the backoff delay. | +| `loopMaxContinues` | `3` | Continues allowed inside `loopWindowMs` before forcing interrupt + resume. | +| `loopWindowMs` | `600000` | Window (ms) for the hallucination-loop guard (default 10 min). | +| `maxRecoveryRetries` | `2` | Cap for targeted recovery prompts (tool-as-text / intent nudges). | +| `continuePrompt` | `"continue"` | Prompt used for a plain resume / ready-to-continue nudge. | +| `toolTextRecoveryPrompt` | _(built-in)_ | Prompt used when a tool call is printed as text instead of executed. | +| `doneWithoutWorkPrompt` | _(built-in)_ | Prompt used to verify a suspicious terse "done" claim. | +| `actionIntentPrompt` | _(falls back to `continuePrompt`)_ | Prompt used when the model ends with an unexecuted intent. | +| `debug` | `false` | Verbose `[auto-resume:debug]` logging. | + +Example: + +```jsonc +{ + "plugins": [ + { + "package": "./plugins/auto-resume-v2.ts", + "options": { "chunkTimeoutMs": 60000, "debug": true } + } + ] +} +``` + +## Verify it loaded + +Start opencode; the plugin logs its banner at load: + +``` +[auto-resume] ready (opencode v2). timeout=45000ms interval=5000ms retries=3 loop=3/600s +``` + +To see an intervention in action, let a session go quiet past +`chunkTimeoutMs`; the plugin injects a visible **synthetic** message in the +session timeline (`auto-resume: …`) and resumes the turn. Every recovery is +also appended to the opencode log with an `[auto-resume]` prefix. + +## Disable / uninstall + +- **Disable by id** without touching other plugins: add `"-auto-resume.v2"` to + the `plugins` array. +- **Temporarily off**: delete/move the file out of the `plugins/` directory. +- **Uninstall**: remove the file and any `plugins` entry you added. + +## Troubleshooting + +| Symptom | Check | +|---|---| +| No `[auto-resume] ready …` banner | File is in a `plugins/` dir opencode scans, or listed in `plugins`; restart opencode. | +| Nothing happens on a stall | Increase verbosity with `"debug": true`; confirm `chunkTimeoutMs` isn't larger than your real stall. | +| Recovers but you don't see a notice | Your model/provider may reject `session.synthetic()`; the plugin falls back to `session.prompt()` (no TUI banner). | +| Never recovers a parent waiting on a subagent | Intentional: parent sessions blocked on a running subagent are left alone. | +| Never recovers while a permission dialog is open | Intentional: recovery is held until the permission is answered. | + +## Development + +```sh +# typecheck the v2 port against the stable types +bun add -d @opencode/plugin@2.0.5 typescript +bunx tsc --noEmit --strict --target ESNext --module ESNext \ + --moduleResolution bundler --skipLibCheck src/v2/index.ts +``` + +See [`migration.md`](./migration.md) for the full v1→v2 mapping and the +stable-vs-beta validation notes. diff --git a/docs/v2/migration.md b/docs/v2/migration.md index e3c52f2..c1ebd82 100644 --- a/docs/v2/migration.md +++ b/docs/v2/migration.md @@ -1,17 +1,18 @@ # opencode-auto-resume — OpenCode v2 port: what changed -> Saved from ~/.config/opencode/plugins/README.md on 2026-08-25. > Drop this into the PR description (or keep as docs/v2-migration.md upstream). > Target: https://github.com/Mte90/opencode-auto-resume -> Runtime tested against: opencode2 v0.0.0-beta-18050, @opencode-ai/plugin 0.0.0-next-17403 +> Runtime: **opencode v2.0.5 (stable)** with **@opencode/plugin 2.0.5**. +> Originally ported against opencode2 v0.0.0-beta-18050 / @opencode-ai/plugin +> 0.0.0-next-17403; re-validated against the stable release (see §7). ## Summary Ports the plugin from the v1 hooks API (`Plugin` factory returning a hooks object) to the v2 promise-plugin API (`Plugin.define({ id, setup })` + `ctx.event.subscribe()`). All detection/recovery features are preserved. -Strict-mode typechecked against the real installed `@opencode-ai/plugin` -types; validated by a mocked-context runtime suite covering every recovery +Strict-mode typechecked against the real `@opencode/plugin@2.0.5` types; +validated by a mocked-context runtime suite covering every recovery path (12/12 checks). ## 1. Config key renamed: `plugin` → `plugins` @@ -46,7 +47,7 @@ export const MyPlugin: Plugin = async ({ client, $, directory }) => ({ // v2: default-export Plugin.define with a unique id and a setup fn. // setup registers long-lived behavior and MAY return a cleanup function, // which OpenCode awaits on disable/reload/shutdown. -import { Plugin } from "@opencode-ai/plugin" +import { Plugin } from "@opencode/plugin" export default Plugin.define({ id: "acme.thing", // unique — used by disable directives @@ -67,7 +68,7 @@ The v2 context is essentially a typed server client plus registration APIs: | v2 capability | Replaces (v1) | |---|---| | `ctx.event.subscribe()` → **AsyncIterable** of events | `event` hook | -| `ctx.session.prompt/interrupt/create/get/command/synthetic/generate` + `ctx.session.hook("context")` | direct SDK calls / chat message hooks | +| `ctx.session.prompt/interrupt/create/get/command/synthetic/generate` + `context/wait/switchAgent/switchModel/rename/update/move` + `ctx.session.hook(...)` | direct SDK calls / chat message hooks | | `ctx.tool.transform` / `ctx.tool.hook` | `tool` map, `tool.execute.before/after` | | `ctx.agent.transform` | agent config mutation | | `ctx.options` | second `options` argument of the v1 factory | @@ -79,9 +80,12 @@ Notes: - **Flattened SDK calls**: nested `{ path, query, body }` envelopes are gone. Example: `session.prompt({ path: { id }, body: { parts } })` became `session.prompt({ sessionID, text })`. -- **No history access**: the v2 plugin context does not expose - `message.list()`. Plugins that need assistant text must accumulate it from - streaming events (`session.text.delta`, `session.reasoning.delta`). +- **History access**: beta exposed no message-history API, so plugins that + needed assistant text had to accumulate it from streaming events + (`session.text.delta`, `session.reasoning.delta`). **Stable adds + `ctx.session.context({ sessionID })` → `SessionMessageInfo[]`**; the port + keeps the deltas for liveness and uses `context()` as the authoritative + final text at idle time. - **Logging**: write to the console with a prefix instead of `app.log()`; opencode captures stdout/stderr into its logs. @@ -90,9 +94,20 @@ Notes: v1 events were `{ type, properties }`; v2 events are flat `{ type, created, data }`. Useful v2 session events for watchdog-style plugins: `session.execution.started/succeeded/failed/interrupted`, -`session.step.started/ended/failed`, `session.text.delta`, -`session.reasoning.delta`, `session.tool.called/progress/success/failed`, -`session.retry.scheduled`, `session.idle`, `permission.asked/replied`. +`session.step.started/streamed/ended/failed`, `session.text.started/delta/ended`, +`session.reasoning.started/delta/ended`, +`session.tool.input.started/delta/ended`, +`session.tool.called/progress/success/failed`, `session.shell.started/ended`, +`session.retry.scheduled`, `session.compaction.*`, `session.status`, +`session.idle`, `session.deleted`, `permission.asked/replied`. + +Two shape details the port relies on: + +- `session.execution.interrupted` carries `data.reason` + (`"user" | "shutdown" | "superseded" | "inactivity"`) — only `"user"` should + suppress recovery. +- `session.idle` is a separate ephemeral event (`data.sessionID`); + `session.status` carries `data.status.type` (`"idle" | "retry" | "busy"`). ## 5. Concrete changes in this port @@ -100,10 +115,14 @@ plugins: `session.execution.started/succeeded/failed/interrupted`, background `ctx.event.subscribe()` pump; cleanup returned from `setup` clears the watchdog interval. - `client.session.prompt({path, body})` → `ctx.session.prompt({ sessionID, - text })` (a nested-shape fallback is kept while the beta API settles). + text })` (flat input; the beta-era nested-shape fallback was removed once + stable confirmed the flat contract). - `session.messages()` forensics → live accumulation of assistant text from - `session.text.delta` / `session.reasoning.delta` events (message history - is not reachable from the v2 plugin context). + `session.text.delta` / `session.reasoning.delta`, with + `ctx.session.context({ sessionID })` as the authoritative idle-time read. +- Recovery notifications use `ctx.session.synthetic({ sessionID, text, + description, resume })` so the intervention is visible in the TUI, falling + back to `session.prompt({ sessionID, text })`. - `ctx.client.app.log(...)` → prefixed console logging. - Status polling demoted: session state comes primarily from lifecycle events; a low-frequency interval cross-checks active sessions for silence @@ -116,6 +135,10 @@ plugins: `session.execution.started/succeeded/failed/interrupted`, - Failure-triggered recoveries arm a flag so the failure's own idle transition doesn't cancel the scheduled retry prompt, while a healthy completion still cancels stale ones. + - `ctx.event.subscribe({ signal })` + `AbortController` so the event stream + is torn down on unload instead of staying suspended in `for await`. + - User-interrupt detection uses `session.execution.interrupted`'s `reason` + (stable) instead of assuming every non-plugin interrupt was the user. ## Feature parity retained @@ -130,7 +153,9 @@ respected. Unchanged names/defaults from v1: `chunkTimeoutMs` (45000), `gracePeriodMs` (3000), `checkIntervalMs` (5000), `maxRetries` (3), `baseBackoffMs` (1000), `maxBackoffMs` (8000), `loopMaxContinues` (3), `loopWindowMs` (600000), -plus `maxRecoveryRetries`, `continuePrompt`, `debug`. +`maxRecoveryRetries` (2), `debug` (false), plus the prompt overrides +`continuePrompt`, `toolTextRecoveryPrompt`, `doneWithoutWorkPrompt`, +`actionIntentPrompt`. ```jsonc { @@ -147,9 +172,42 @@ Disable by id without touching other plugins: add `"-auto-resume.v2"`. ## Testing -- `tsc --strict --noEmit` clean against `@opencode-ai/plugin@0.0.0-next-17403`. -- Mocked-context runtime suite (bun): export shape · stall watchdog → prompt · - ready-to-continue nudge · tool-call-as-text recovery · loop guard → interrupt · - permission-hold blocks recovery · execution-failure recovery · stale recovery - does not disturb healthy idle sessions · user interrupt honored · cleanup - quiesces timers — **12/12 passing**. +- `tsc --strict --noEmit` clean against **`@opencode/plugin@2.0.5`** (stable). +- Mocked-context runtime suite (bun, 12 tests) against the stable package: + definition shape · `subscribe({ signal })` · abort-on-cleanup · stall watchdog + → `synthetic` with visible `description` + `resume` · idle forensics via the + `session.context()` fallback · healthy idle turn is a no-op · permission-hold + blocks recovery · `interrupted` `reason="user"` disables recovery · + `reason="inactivity"` does **not** · execution-failure recovery · + `synthetic`→`prompt` fallback · `session.deleted` drops state — + **12/12 passing**. + +## 7. Re-validated against stable v2 (2.0.5) + +The port was originally written against the beta (`@opencode-ai/plugin` +`0.0.0-next-17403`). Re-checked against the stable release: + +- **Package renamed**: the dependency is now **`@opencode/plugin`** (stable + `2.0.5`), published from the opencode repo. The beta `@opencode-ai/plugin` + package is not the stable API. Imports must use `@opencode/plugin`. + Config/discovery is unchanged (`plugins` key, object form, and plugins under + `.opencode/plugin/` **or** `.opencode/plugins/` are auto-loaded). +- **Every event the plugin matches still exists** in 2.0.5, with the same names + (`session.execution.succeeded` is still `succeeded`, not `completed`). New, + unused-by-us events: `session.status`, `session.text.started`, + `session.step.streamed`, `session.tool.input.*`, `session.compaction.*`, + `session.message.content.updated`. +- **`session.execution.interrupted` gained `data.reason`**; the port now treats + only `"user"` as a user-cancel. +- **`session.reverted` is gone** in v2 (reverts surface as + `session.revert.cleared` / `session.revert.committed`); the legacy matcher is + kept defensively and `session.deleted` handles cleanup. +- **New session methods** on the plugin context: `context()` (message history), + `wait()`, `switchAgent()`, `switchModel()`, `rename()`. The port adopts + `context()` for idle-time forensics; `wait()` is intentionally unused (the + event-driven watchdog is retained). +- **`ctx.session.synthetic()` still accepts `description` and `resume`** in its + input (both optional), so the visible-notification + resume path is unchanged. + `ctx.session.interrupt()` input is `{ sessionID, resume? }`. +- **Docs recommend `ctx.event.subscribe({ signal })`** and aborting the stream + during cleanup; the port now does this with an `AbortController`. From b5fd8aa5bf7d59550c401da4b64a382db09fe28f Mon Sep 17 00:00:00 2001 From: valentimarco Date: Thu, 17 Sep 2026 15:07:40 +0200 Subject: [PATCH 05/57] docs(v2): document @opencode/plugin resolution requirement A local .ts plugin is imported by opencode with normal ESM resolution and the plugin dependency graph is resolved at server startup, so @opencode/plugin must be installed in the config dir (~/.config/opencode) and opencode restarted. Without it the log shows "failed to load plugin ... Cannot find package '@opencode/plugin'". --- docs/v2/installing.md | 37 +++++++++++++++++++++++++++++++++---- docs/v2/migration.md | 6 ++++++ 2 files changed, 39 insertions(+), 4 deletions(-) diff --git a/docs/v2/installing.md b/docs/v2/installing.md index 00af2d8..f82263a 100644 --- a/docs/v2/installing.md +++ b/docs/v2/installing.md @@ -9,13 +9,35 @@ plugin API (`@opencode/plugin` 2.0.5, shipped with `opencode` v2.0.5). ## Requirements - **opencode v2.0.5 or newer** (`opencode --version`). -- Node.js/Bun is only needed if you run the test/typecheck tooling; the plugin - itself is loaded by opencode. +- **`@opencode/plugin` must be resolvable by opencode.** A local plugin is loaded + with a normal ESM import, so the API package has to be installed next to it + (see [Step 1](#step-1--install-the-plugin-api-package)). Without it opencode + logs `failed to load plugin … Cannot find package '@opencode/plugin'`. +- Node.js/Bun is only needed to install the API package and run the + test/typecheck tooling; opencode itself loads the plugin. - The plugin source: [`src/v2/index.ts`](../../src/v2/index.ts) from this repo. ## Install -### Option A — drop-in file (no config) +### Step 1 — install the plugin API package + +Global plugins resolve imports from `~/.config/opencode`, so install the stable +API package there once: + +```sh +cd ~/.config/opencode +bun add @opencode/plugin@2.0.5 # or: npm install @opencode/plugin@2.0.5 +``` + +For a per-project plugin, install it in the project instead: + +```sh +npm install @opencode/plugin@2.0.5 +``` + +### Step 2 — add the plugin + +#### Option A — drop-in file (no config) OpenCode auto-loads every plugin found in these directories: @@ -32,7 +54,7 @@ cp src/v2/index.ts ~/.config/opencode/plugins/auto-resume-v2.ts That's it — no `opencode.json` change is required. Restart opencode (or reload plugins) and the plugin is active. -### Option B — `opencode.json(c)` entry +#### Option B — `opencode.json(c)` entry Use this when you want to load the file from another location, pass options, or pin a published package. Add an entry to the `plugins` array: @@ -59,6 +81,12 @@ pin a published package. Add an entry to the `plugins` array: Both `.opencode/plugin/` (v1 directory name) and `.opencode/plugins/` are discovered; use `.opencode/plugins/` for v2 files. +### Step 3 — restart opencode + +Restart opencode after installing. The plugin loader resolves plugin +dependencies when the server starts, so a hot file reload is **not** enough to +pick up a newly installed `@opencode/plugin` package. + ## Options All options are optional and are read from `ctx.options` in `setup`. @@ -118,6 +146,7 @@ also appended to the opencode log with an `[auto-resume]` prefix. | Symptom | Check | |---|---| | No `[auto-resume] ready …` banner | File is in a `plugins/` dir opencode scans, or listed in `plugins`; restart opencode. | +| Logs say `failed to load plugin … Cannot find package '@opencode/plugin'` | Install the API package next to the plugin (Step 1) **and restart** opencode. | | Nothing happens on a stall | Increase verbosity with `"debug": true`; confirm `chunkTimeoutMs` isn't larger than your real stall. | | Recovers but you don't see a notice | Your model/provider may reject `session.synthetic()`; the plugin falls back to `session.prompt()` (no TUI banner). | | Never recovers a parent waiting on a subagent | Intentional: parent sessions blocked on a running subagent are left alone. | diff --git a/docs/v2/migration.md b/docs/v2/migration.md index c1ebd82..907bfee 100644 --- a/docs/v2/migration.md +++ b/docs/v2/migration.md @@ -211,3 +211,9 @@ The port was originally written against the beta (`@opencode-ai/plugin` `ctx.session.interrupt()` input is `{ sessionID, resume? }`. - **Docs recommend `ctx.event.subscribe({ signal })`** and aborting the stream during cleanup; the port now does this with an `AbortController`. +- **Local-file installs need the API package resolvable.** opencode loads a + local `.ts` plugin with a normal ESM import and snapshots plugin + dependencies at server startup, so `@opencode/plugin` must be installed in + the config dir (`~/.config/opencode`) and opencode restarted. A *published* + plugin should instead declare `@opencode/plugin` in its `dependencies` + (replacing `@opencode-ai/plugin`). From 0a0a9618cc0d252015cbb960b5bb1a9cc5fa366b Mon Sep 17 00:00:00 2001 From: famewolf Date: Sat, 26 Sep 2026 19:53:23 -0400 Subject: [PATCH 06/57] OOC lock (parity with v1 loop-fix) Mirrors the v1 src/index.ts fix onto the v2 port: - oocLocked / oocLockReason / selfRecovery on SessionWatch - OOC_ERROR_RE (exceeds the available context size / too large to compact / too many tokens / prompt is too long) - maybeLockOoc latches on retry.scheduled + failed events carrying an OOC error; recover() and tryAbortAndResume refuse while locked - execution.started from a genuine (non-self-recovery) run clears the lock; self-recovery busy edges don't reset the retry counter v1 full suite on this branch: 578 pass / 0 fail (v1 source untouched). v2 build: clean. --- src/v2/index.ts | 44 +++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 43 insertions(+), 1 deletion(-) diff --git a/src/v2/index.ts b/src/v2/index.ts index 8f4fe0e..bbb8110 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -60,6 +60,11 @@ interface SessionWatch { gaveUp: boolean aborting: boolean recovering: boolean + /** Latched when a failure carries an OOC error that `continue` can never clear; recovery (continue + abort+resume) is refused until a genuine user/agent turn. */ + oocLocked: boolean + oocLockReason: string | null + /** True while one of our own recovery continues is in flight — distinguishes a self-caused execution.started from a genuine turn. */ + selfRecovery: boolean // Assistant text accumulation (v2 replacement for session.messages()) textParts: Map // assistantMessageID -> accumulated text @@ -120,6 +125,9 @@ const DEFAULT_LOOP_WINDOW_MS = 10 * 60_000 const DEFAULT_MAX_RECOVERY_RETRIES = 2 const DEFAULT_DEBUG = false +/** OOC (out-of-context) errors that `continue` can never clear — recovery is locked out on these. */ +const OOC_ERROR_RE = /exceeds the available context size|context size \(\d+\)|too large to compact|too many tokens|prompt is too long/i + const MAX_IDLE_SESSIONS = 50 const IDLE_CLEANUP_MS = 10 * 60_000 const TEXT_BUFFER_TRIM_LEN = 20_000 @@ -381,6 +389,9 @@ export default Plugin.define({ gaveUp: false, aborting: false, recovering: false, + oocLocked: false, + oocLockReason: null, + selfRecovery: false, textParts: new Map(), lastAssistantText: "", lastAssistantMessageID: null, @@ -520,6 +531,7 @@ export default Plugin.define({ * kicks the session back to life, replacing the separate `prompt()`. */ async function notifyAndPrompt(sid: string, text: string, notification: string, resume = true): Promise { + ensureWatch(sid).selfRecovery = true try { await ctx.session.synthetic({ sessionID: sid, @@ -545,6 +557,10 @@ export default Plugin.define({ async function tryAbortAndResume(sid: string, w: SessionWatch): Promise { if (w.aborting) return false + if (w.oocLocked) { + dbg(`${short(sid)} oocLocked — refusing abort+resume escalation`) + return false + } w.aborting = true log("warn", `${short(sid)} escalating: interrupt + fresh continue`) try { @@ -566,6 +582,16 @@ export default Plugin.define({ return ok } + /** Latch the OOC lock when a failure carries an out-of-context error (which `continue` can never clear). */ + function maybeLockOoc(sid: string, errMsg: string) { + const w = ensureWatch(sid) + if (w.oocLocked) return + if (!OOC_ERROR_RE.test(errMsg)) return + w.oocLocked = true + w.oocLockReason = errMsg.slice(0, 200) + log("warn", `${short(sid)} OOC error latched — recovery locked out until a genuine turn: ${errMsg.slice(0, 120)}`) + } + /** * Core recovery ladder for a stuck/failed session. * plain continue with backoff -> more attempts -> abort+resume escalation. @@ -573,6 +599,10 @@ export default Plugin.define({ async function recover(sid: string, reason: string) { const w = ensureWatch(sid) if (w.recovering || w.aborting || w.gaveUp || w.userCancelled || w.permissionPending) return + if (w.oocLocked) { + dbg(`${short(sid)} oocLocked — refusing recovery (reason: ${w.oocLockReason?.slice(0, 80)})`) + return + } // Record intent first, then evaluate the loop guard, so the Nth // continue within the window is the one that escalates. recordContinue(sid) @@ -777,6 +807,14 @@ export default Plugin.define({ case "session.execution.started": { const sid = sidOf(ev) if (!sid) return + const w = ensureWatch(sid) + if (w.selfRecovery) { + w.selfRecovery = false + } else if (w.oocLocked) { + w.oocLocked = false + w.oocLockReason = null + log("info", `${short(sid)} genuine execution — clearing OOC lock`) + } markBusy(sid) return } @@ -785,7 +823,9 @@ export default Plugin.define({ const sid = sidOf(ev) if (!sid) return markIdle(sid) - ensureWatch(sid).pendingRecoveryArmed = false + const w = ensureWatch(sid) + w.pendingRecoveryArmed = false + w.selfRecovery = false // stale — no in-flight recovery prompt by the time we idle void inspectOnIdle(sid) return } @@ -916,6 +956,7 @@ export default Plugin.define({ touch(sid) // provider-level retry is progress of its own kind const errType = String(ev.data?.error?.type ?? "") const errMsg = String(ev.data?.error?.message ?? "") + maybeLockOoc(sid, errMsg) log("info", `${short(sid)} provider retry #${ev.data?.attempt ?? "?"}: ${errType || errMsg}`) if (!isStreamingFailure(errMsg) && errType !== "retryable") return // Let the provider retries play out first; only intervene if it stays quiet @@ -932,6 +973,7 @@ export default Plugin.define({ const w = ensureWatch(sid) const errMsg = String(ev.data?.error?.message ?? "") const errType = String(ev.data?.error?.type ?? "") + maybeLockOoc(sid, errMsg) markIdle(sid) w.pendingRecoveryArmed = true // our delayed recovery must survive this idle transition if (errMsg.includes("interrupted by user") || errType.includes("cancel")) { From 9a446996e571fc35cc6b8ac847452ac2c4581413 Mon Sep 17 00:00:00 2001 From: famewolf Date: Sun, 27 Sep 2026 15:49:58 -0400 Subject: [PATCH 07/57] v2: never interrupt compaction or live busy sessions Root cause: recovery injects a synthetic 'continue' turn to a session even while it is BUSY; in v2 a new turn interrupts the in-flight step ('Step interrupted'). The 45s stall watchdog falsely fires during long 27B thinking and native compaction (little/no text.delta), so in-flight compaction was interrupted 3x back-to-back. FIX A (compaction guard): - SessionWatch.compacting / compactionStartedAt set on session.compaction.started, cleared on ended/failed and on execution.interrupted (feature-tolerant: no-op if the runtime emits no such events), stale-cleared after 30 min - refuse recovery in recover(), tryAbortAndResume(), targetedRecovery(); watchdog skips stall detection while compacting FIX B (live-busy inject guard): - at inject time, skip the synthetic turn if the session is still busy AND live within the chunk window: a working session is not stalled; the watchdog re-arms if it truly stalls --- src/v2/index.ts | 76 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 76 insertions(+) diff --git a/src/v2/index.ts b/src/v2/index.ts index bbb8110..4425c53 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -84,6 +84,10 @@ interface SessionWatch { waitingOnSubagent: boolean lastWasTaskTool: boolean idleSince: number | null + /** Set while the session is mid native compaction — recovery must never interrupt it. */ + compacting: boolean + /** Timestamp of the latest `session.compaction.started` (stale-guard for the flag). */ + compactionStartedAt: number | null // Tool-loop tracking recentToolCalls: ToolCallRecord[] @@ -130,6 +134,8 @@ const OOC_ERROR_RE = /exceeds the available context size|context size \(\d+\)|to const MAX_IDLE_SESSIONS = 50 const IDLE_CLEANUP_MS = 10 * 60_000 +/** A compaction silent for this long without an ended/failed event is wedged — stale-clear the flag. */ +const COMPACTION_STALE_TTL_MS = 30 * 60_000 const TEXT_BUFFER_TRIM_LEN = 20_000 const CONTINUE_PROMPT = "continue" @@ -404,6 +410,8 @@ export default Plugin.define({ waitingOnSubagent: false, lastWasTaskTool: false, idleSince: null, + compacting: false, + compactionStartedAt: null, recentToolCalls: [], } sessions.set(sid, w) @@ -561,6 +569,10 @@ export default Plugin.define({ dbg(`${short(sid)} oocLocked — refusing abort+resume escalation`) return false } + if (w.compacting) { + dbg(`${short(sid)} mid-compaction — refusing abort+resume escalation`) + return false + } w.aborting = true log("warn", `${short(sid)} escalating: interrupt + fresh continue`) try { @@ -603,6 +615,10 @@ export default Plugin.define({ dbg(`${short(sid)} oocLocked — refusing recovery (reason: ${w.oocLockReason?.slice(0, 80)})`) return } + if (w.compacting) { + dbg(`${short(sid)} mid-compaction — refusing recovery (reason: ${reason})`) + return + } // Record intent first, then evaluate the loop guard, so the Nth // continue within the window is the one that escalates. recordContinue(sid) @@ -632,6 +648,19 @@ export default Plugin.define({ // user took over. if (w.userCancelled || w.gaveUp) return if (w.status === "idle" && !w.pendingRecoveryArmed) return // recovered by itself meanwhile + // A turn sent to a busy session interrupts its in-flight step + // ("Step interrupted"). Never interrupt mid-compaction, and + // never interrupt a busy session that is still live (events + // within the chunk window): it is working, not stalled. The + // watchdog re-arms if it truly stalls. + if (w.compacting) { + dbg(`${short(sid)} mid-compaction at inject time — not interrupting`) + return + } + if (w.status === "busy" && Date.now() - w.lastActivityAt < chunkTimeoutMs) { + dbg(`${short(sid)} still busy and live at inject time (${Math.round((Date.now() - w.lastActivityAt) / 1000)}s since last event) — not interrupting`) + return + } const ok = await notifyAndPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "stalled — retrying") w.pendingRecoveryArmed = false if (ok) { @@ -651,6 +680,10 @@ export default Plugin.define({ async function targetedRecovery(sid: string, kind: string, prompt: string, budgetKey: "toolTextAttempts" | "doneClaimAttempts" | "intentNudgeAttempts") { const w = ensureWatch(sid) if (w.recovering || w.userCancelled || w.permissionPending) return + if (w.compacting) { + dbg(`${short(sid)} mid-compaction — skipping ${kind} nudge`) + return + } recordContinue(sid) if (continuesInWindow(w) >= loopMaxContinues) { log("warn", `${short(sid)} loop guard before ${kind} nudge — abort+resume`) @@ -767,6 +800,19 @@ export default Plugin.define({ for (const [sid, w] of sessions) { if (w.status !== "busy" || w.userCancelled) continue + // Stale compaction flag: a compaction that never reports + // ended/failed within the TTL is wedged — clear the guard so + // recovery can eventually intervene. + if (w.compacting && w.compactionStartedAt && now - w.compactionStartedAt > COMPACTION_STALE_TTL_MS) { + w.compacting = false + w.compactionStartedAt = null + log("warn", `${short(sid)} compaction flag stale (>${COMPACTION_STALE_TTL_MS / 60000}m) — clearing guard`) + } + // NEVER interrupt a session that is mid-compaction. + if (w.compacting) { + dbg(`${short(sid)} mid-compaction — skipping stall detection`) + continue + } const silence = now - w.lastActivityAt if (silence < chunkTimeoutMs + gracePeriodMs) continue if (w.permissionPending) { @@ -838,6 +884,9 @@ export default Plugin.define({ // (or a missing reason on older runtimes) must not. const reason = typeof ev.data?.reason === "string" ? ev.data.reason : undefined w.userCancelled = !w.aborting && (reason === undefined || reason === "user") + // A user interrupt also ends any in-flight compaction. + w.compacting = false + w.compactionStartedAt = null w.recovering = false markIdle(sid) return @@ -858,6 +907,33 @@ export default Plugin.define({ return } + // --- compaction (NEVER interrupt a session that is compacting) --- + case "session.compaction.started": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + if (!w.compacting) { + w.compacting = true + w.compactionStartedAt = Date.now() + log("info", `${short(sid)} compaction started — auto-resume recovery suspended until it ends`) + } + touch(sid) + return + } + case "session.compaction.ended": + case "session.compaction.failed": { + const sid = sidOf(ev) + if (!sid) return + const w = ensureWatch(sid) + if (w.compacting) { + w.compacting = false + w.compactionStartedAt = null + log("info", `${short(sid)} compaction ${ev.type.endsWith("failed") ? "failed" : "ended"} — guard cleared`) + } + touch(sid) + return + } + // --- activity -------------------------------------------------- case "session.step.started": { const sid = sidOf(ev) From 0dd040342a88d01436b4feb0921930cc0f06b446 Mon Sep 17 00:00:00 2001 From: famewolf Date: Sun, 27 Sep 2026 17:38:41 -0400 Subject: [PATCH 08/57] fix(v2): never nudge a turn that hands off to the user A turn that ends with a question or an explicit prompt is awaiting the user's reply. Nudging it (a synthetic 'continue') starts an in-flight step that the user's slow reply then interrupts -> 'Step interrupted'. Add isUserHandoff() and gate inspectOnIdle() on it. Purely additive; legit nudges (tool-as-text / ready-to-continue / non-hand-off done-claim) are unchanged. --- src/v2/index.ts | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/src/v2/index.ts b/src/v2/index.ts index 4425c53..24d6424 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -278,6 +278,30 @@ function containsActionIntent(text: string): boolean { return lastLine.endsWith(":") && lastLine.length > 5 && lastLine.length < 500 } +/** + * True when a turn cleanly ends by handing control back to the user — a + * question, or an explicit prompt for their input. Nudging such a turn + * injects a synthetic `continue` that starts an in-flight step, and when the + * user's real reply then arrives it interrupts that step → "Step interrupted" + * (the execution.interrupted event). A hand-off turn is, by design, awaiting + * the user, so it must never be nudged. + */ +function isUserHandoff(text: string): boolean { + const lines = text.split("\n") + let last = "" + for (let i = lines.length - 1; i >= 0; i--) { + const t = lines[i].trim() + if (t.length > 0) { + last = t + break + } + } + if (!last) return false + if (last.endsWith("?") || last.endsWith("?")) return true + const patterns = [/let\s+me\s+know/i, /your\s+call/i, /up\s+to\s+you/i, /which\s+(would|do\s+you|option)/i, /should\s+I/i, /want\s+me\s+to/i, /would\s+you\s+like/i] + return patterns.some((p) => p.test(last)) +} + function backoffMs(attempt: number, base: number, max: number): number { return Math.min(base * Math.pow(2, attempt - 1), max) } @@ -747,6 +771,15 @@ export default Plugin.define({ if (!text) text = await lastAssistantTextFromContext(sid) if (!text) return + // A turn that ends by handing control back to the user (a question or an + // explicit prompt) is awaiting their reply; a synthetic nudge here starts + // an in-flight step that their real reply then interrupts ("Step + // interrupted"). Never nudge a hand-off turn. + if (isUserHandoff(text)) { + dbg(`${short(sid)} idle turn ends with a user hand-off — skipping targeted recovery`) + return + } + if (containsToolCallAsText(text)) { await targetedRecovery(sid, "tool-call-as-text", opts.toolTextRecoveryPrompt ?? TOOL_TEXT_RECOVERY_PROMPT, "toolTextAttempts") return From 3c1d88cf16be4df3302cf0a51bb40b43884fe37e Mon Sep 17 00:00:00 2001 From: famewolf Date: Sun, 27 Sep 2026 20:43:22 -0400 Subject: [PATCH 09/57] v2 back-port: shouldStandDownForUser guard (v1 parity) - stand down on a pending tool_use/question or recent user activity; + activeUserWindowMs option (default 15min) --- bun.lock | 46 ++++++++++++++++++++++++++++++++++++++++++++++ src/v2/index.ts | 36 +++++++++++++++++++++++++++++++++++- 2 files changed, 81 insertions(+), 1 deletion(-) create mode 100644 bun.lock diff --git a/bun.lock b/bun.lock new file mode 100644 index 0000000..6515c4f --- /dev/null +++ b/bun.lock @@ -0,0 +1,46 @@ +{ + "lockfileVersion": 2, + "configVersion": 0, + "workspaces": { + "": { + "name": "opencode-auto-resume", + "dependencies": { + "@opencode-ai/plugin": "latest", + }, + "devDependencies": { + "@opencode-ai/sdk": "latest", + "@types/bun": "latest", + "typescript": "latest", + }, + }, + }, + "packages": { + "@opencode-ai/plugin": ["@opencode-ai/plugin@1.4.0", "", { "dependencies": { "@opencode-ai/sdk": "1.4.0", "zod": "4.1.8" }, "peerDependencies": { "@opentui/core": ">=0.1.97", "@opentui/solid": ">=0.1.97" }, "optionalPeers": ["@opentui/core", "@opentui/solid"] }, "sha512-VFIff6LHp/RVaJdrK3EQ1ijx0K1tV5i1DY5YJ+pRqwC6trunPHbvqSN0GHSTZX39RdnSc+XuzCTZQCy1W2qNOg=="], + + "@opencode-ai/sdk": ["@opencode-ai/sdk@1.4.0", "", { "dependencies": { "cross-spawn": "7.0.6" } }, "sha512-mfa3MzhqNM+Az4bgPDDXL3NdG+aYOHClXmT6/4qLxf2ulyfPpMNHqb9Dfmo4D8UfmrDsPuJHmbune73/nUQnuw=="], + + "@types/bun": ["@types/bun@1.3.11", "", { "dependencies": { "bun-types": "1.3.11" } }, "sha512-5vPne5QvtpjGpsGYXiFyycfpDF2ECyPcTSsFBMa0fraoxiQyMJ3SmuQIGhzPg2WJuWxVBoxWJ2kClYTcw/4fAg=="], + + "@types/node": ["@types/node@25.5.2", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-tO4ZIRKNC+MDWV4qKVZe3Ql/woTnmHDr5JD8UI5hn2pwBrHEwOEMZK7WlNb5RKB6EoJ02gwmQS9OrjuFnZYdpg=="], + + "bun-types": ["bun-types@1.3.11", "", { "dependencies": { "@types/node": "*" } }, "sha512-1KGPpoxQWl9f6wcZh57LvrPIInQMn2TQ7jsgxqpRzg+l0QPOFvJVH7HmvHo/AiPgwXy+/Thf6Ov3EdVn1vOabg=="], + + "cross-spawn": ["cross-spawn@7.0.6", "", { "dependencies": { "path-key": "^3.1.0", "shebang-command": "^2.0.0", "which": "^2.0.1" } }, "sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA=="], + + "isexe": ["isexe@2.0.0", "", {}, "sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw=="], + + "path-key": ["path-key@3.1.1", "", {}, "sha512-ojmeN0qd+y0jszEtoY48r0Peq5dwMEkIlCOu6Q5f41lfkswXuKtYrhgoTpLnyIcHm24Uhqx+5Tqm2InSwLhE6Q=="], + + "shebang-command": ["shebang-command@2.0.0", "", { "dependencies": { "shebang-regex": "^3.0.0" } }, "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA=="], + + "shebang-regex": ["shebang-regex@3.0.0", "", {}, "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A=="], + + "typescript": ["typescript@6.0.2", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-bGdAIrZ0wiGDo5l8c++HWtbaNCWTS4UTv7RaTH/ThVIgjkveJt83m74bBHMJkuCbslY8ixgLBVZJIOiQlQTjfQ=="], + + "undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="], + + "which": ["which@2.0.2", "", { "dependencies": { "isexe": "^2.0.0" }, "bin": { "node-which": "bin/node-which" } }, "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA=="], + + "zod": ["zod@4.1.8", "", {}, "sha512-5R1P+WwQqmmMIEACyzSvo4JXHY5WiAFHRMg+zBZKgKS+Q1viRa0C1hmUKtHltoIFKtIdki3pRxkmpP74jnNYHQ=="], + } +} diff --git a/src/v2/index.ts b/src/v2/index.ts index 24d6424..cd9db64 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -112,6 +112,7 @@ export interface AutoResumeOptions { doneWithoutWorkPrompt?: string actionIntentPrompt?: string debug?: boolean + activeUserWindowMs?: number } // --------------------------------------------------------------------------- @@ -128,6 +129,7 @@ const DEFAULT_LOOP_MAX_CONTINUES = 3 const DEFAULT_LOOP_WINDOW_MS = 10 * 60_000 const DEFAULT_MAX_RECOVERY_RETRIES = 2 const DEFAULT_DEBUG = false +const DEFAULT_ACTIVE_USER_WINDOW_MS = 15 * 60_000 /** OOC (out-of-context) errors that `continue` can never clear — recovery is locked out on these. */ const OOC_ERROR_RE = /exceeds the available context size|context size \(\d+\)|too large to compact|too many tokens|prompt is too long/i @@ -388,6 +390,7 @@ export default Plugin.define({ const loopWindowMs = opts.loopWindowMs ?? DEFAULT_LOOP_WINDOW_MS const maxRecoveryRetries = opts.maxRecoveryRetries ?? DEFAULT_MAX_RECOVERY_RETRIES const debug = opts.debug ?? DEFAULT_DEBUG + const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const dbg = (...args: unknown[]) => { if (debug) console.log("[auto-resume:debug]", ...args) @@ -763,7 +766,33 @@ export default Plugin.define({ } } - async function inspectOnIdle(sid: string) { + async function shouldStandDownForUser(sid: string, activeUserWindowMs: number): Promise { + try { + const result = await ctx.session.context({ sessionID: sid }) + const messages: unknown[] = Array.isArray(result) ? result : ((result as { messages?: unknown[] })?.messages ?? []) + if (messages.length === 0) return false + const newest = messages[messages.length - 1] as { type?: string; content?: { type?: string; state?: { status?: string } }[]; time?: { created?: number }; info?: { time?: { created?: number } } } + // (a) A pending tool call (e.g. the `question` tool) means the model is waiting + // on the user's answer — a synthetic nudge here would be interrupted by their + // real reply ("Step interrupted"). + if (newest?.type === "assistant") { + for (const part of newest.content ?? []) { + const t = part?.type ?? "" + if ((t === "tool_use" || t === "tool" || t === "tool_call" || t.startsWith("tool")) && part?.state?.status === "pending") return true + } + } + // (b) The user was recently active — they are mid-conversation, not stuck. + if (newest?.type === "user") { + const ts = newest.time?.created ?? newest.info?.time?.created + if (typeof ts === "number" && Date.now() - ts < activeUserWindowMs) return true + } + return false + } catch { + return false + } +} + +async function inspectOnIdle(sid: string) { const w = ensureWatch(sid) // Prefer the live delta buffer; fall back to the authoritative message // history when it is empty (e.g. the plugin loaded mid-turn) or stale. @@ -780,6 +809,11 @@ export default Plugin.define({ return } + if (await shouldStandDownForUser(sid, activeUserWindowMs)) { + dbg(`${short(sid)} user has pending input or was recently active — standing down`) + return + } + if (containsToolCallAsText(text)) { await targetedRecovery(sid, "tool-call-as-text", opts.toolTextRecoveryPrompt ?? TOOL_TEXT_RECOVERY_PROMPT, "toolTextAttempts") return From 875d7e470e40896e23540e2bad9e78fb4433a8bd Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 28 Sep 2026 12:29:55 -0400 Subject: [PATCH 10/57] fix(v2): stop self-inflicted "Step interrupted" and synthetic-continue bursts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The plugin could interrupt a session and then recover from its own interrupt, injecting a synthetic `continue` that superseded the step it had just killed. The UI surfaces that as "Opencode failed to send message with error: Step interrupted". Root causes fixed here: * Interrupt-shaped failures were never recognised. The filter only matched "interrupted by user" / a type containing "cancel", but v2 reports an interrupt as {type:"aborted", message:"Step interrupted"}, so the plugin recovered from its own abort forever. Added isAbortError() and made the plugin refuse to recover from any abort-shaped failure, its own or the user's. * "Is this abort mine?" was tracked with a transient boolean that the runtime resets before it delivers session.execution.interrupted asynchronously. Replaced with a selfAbortAt latch + TTL, and the idle handler no longer runs its heuristics on top of our own abort. * There was no single choke point for recovery injection: the stall watchdog, the intent/tool-loop nudges and the abort+resume escalation each called notifyAndPrompt directly, with no rate limit. Added injectOnce(), which all three paths now use, refusing injections inside our own abort window, into a busy-but-live session (a turn sent to a busy session supersedes its in-flight step), and more often than injectIntervalMs (15s). The abort+resume escalation takes its own lane, since swallowing its continue would leave a session interrupted with nothing to restart it. * The loop guard counted the plugin's own continues, so it escalated on its own output; and targetedRecovery recorded the same continue twice. The guard is now evaluated before recording. * getActiveSessions() returned [] because v2's plugin SessionDomain does not expose session.active, which made the subagent-wait guard permanently dead — a parent that went quiet while its child ran was declared stalled and interrupted. Falls back to busySessions(), derived from our own busy flags. * Interrupts are now hard-capped per window, since an interrupt persists an errored assistant message that the UI shows as a send failure. Also: * Added the missing liveness events (session.step.streamed, *.started, session.synthetic, session.tool.input.*) so a step that starts and then produces no tokens, or a long tool-input generation, is not mistaken for a stall. * Removed dead code: waitingOnSubagent (never set true, so its guard could never fire) and maxRecoveryRetries (read from options, never used). * Made this file buildable: it imported a runtime Plugin.define from "@opencode/plugin", which is not a dependency of this repo. Use a local identity helper instead; `tsc --noEmit` now exits 0 where it previously failed with 2 errors. * log() now prefers ctx.app.log, since plugin console output is not captured anywhere retrievable. Verified with tsc (exit 0), the existing test suite (578 passing, 0 failing) and a behavioral harness that asserts zero injections during healthy streaming, zero after a user stop, zero in response to the plugin's own abort events, and a working stall recovery path. --- .gitignore | 3 + src/v2/index.ts | 294 +++++++++++++++++++++++++++++++++++++++++++----- 2 files changed, 266 insertions(+), 31 deletions(-) diff --git a/.gitignore b/.gitignore index 9451024..7a54f87 100644 --- a/.gitignore +++ b/.gitignore @@ -2,3 +2,6 @@ node_modules/ dist/ .DS_Store *.log + +# v2 build output (bun build src/v2/index.ts --outdir dist-v2 --target bun) +dist-v2/ diff --git a/src/v2/index.ts b/src/v2/index.ts index cd9db64..8622ffb 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -32,7 +32,39 @@ * { "plugins": [{ "package": "./plugins/auto-resume-v2.ts", "options": { ... } }] } */ -import { Plugin } from "@opencode/plugin" +/** + * v2 entrypoint shape. + * + * The package depends on `@opencode-ai/plugin`, which exports `Plugin` only as a + * *type* — `(input, options?) => Promise`. There is no runtime + * `Plugin.define` in it, so the previous `import { Plugin } from + * "@opencode/plugin"` did not resolve and the file could not be built or + * type-checked at all (see the build script, which only ever built v1's + * `src/index.ts`). The v2 contract is simply `{ id, setup }`, so define the + * identity helper locally rather than importing a symbol that does not exist. + */ +type AutoResumePlugin = { + id: string + setup: (ctx: AutoResumePluginInput) => unknown +} + +const define = (plugin: T): T => plugin + +/** + * The subset of the v2 plugin input this plugin actually consumes. Typed + * structurally so the file type-checks standalone, without pinning a + * @opencode-ai/plugin version whose `Plugin` type describes the v1 contract. + */ +interface AutoResumePluginInput { + /** Per-plugin options from config. */ + options?: Record + /** Event stream (AsyncIterable of `{ type, data }`). */ + event: { subscribe: (opts?: { signal?: AbortSignal }) => AsyncIterable } + /** Server-side session/messaging operations. */ + session: Record any> + /** Application logger, when the host provides one. */ + app?: { log?: (level: string, message: string) => unknown } +} // --------------------------------------------------------------------------- // Types @@ -59,6 +91,14 @@ interface SessionWatch { lastRetryAt: number gaveUp: boolean aborting: boolean + /** Timestamp of the last plugin-initiated `session.interrupt()`. Any abort-shaped + * event inside `SELF_ABORT_TTL_MS` of it is ours, not the user's. */ + selfAbortAt: number + /** Count of plugin-initiated interrupts in the current `INTERRUPT_WINDOW_MS`. */ + interruptsThisWindow: number + interruptWindowStart: number + /** Timestamp of the last recovery injection, for `injectIntervalMs` debouncing. */ + lastInjectAt: number recovering: boolean /** Latched when a failure carries an OOC error that `continue` can never clear; recovery (continue + abort+resume) is refused until a genuine user/agent turn. */ oocLocked: boolean @@ -81,7 +121,6 @@ interface SessionWatch { // Guards permissionPending: boolean - waitingOnSubagent: boolean lastWasTaskTool: boolean idleSince: number | null /** Set while the session is mid native compaction — recovery must never interrupt it. */ @@ -106,7 +145,8 @@ export interface AutoResumeOptions { maxBackoffMs?: number loopMaxContinues?: number loopWindowMs?: number - maxRecoveryRetries?: number + /** Minimum gap between recovery injections for one session. */ + injectIntervalMs?: number continuePrompt?: string toolTextRecoveryPrompt?: string doneWithoutWorkPrompt?: string @@ -127,13 +167,39 @@ const DEFAULT_BASE_BACKOFF_MS = 1_000 const DEFAULT_MAX_BACKOFF_MS = 8_000 const DEFAULT_LOOP_MAX_CONTINUES = 3 const DEFAULT_LOOP_WINDOW_MS = 10 * 60_000 -const DEFAULT_MAX_RECOVERY_RETRIES = 2 const DEFAULT_DEBUG = false const DEFAULT_ACTIVE_USER_WINDOW_MS = 15 * 60_000 /** OOC (out-of-context) errors that `continue` can never clear — recovery is locked out on these. */ const OOC_ERROR_RE = /exceeds the available context size|context size \(\d+\)|too large to compact|too many tokens|prompt is too long/i +/** + * Window after `session.interrupt()` during which any abort-shaped event is + * attributed to us rather than to the user. The runtime delivers + * `session.execution.interrupted` asynchronously, so a boolean `aborting` flag + * cleared on a fixed timer is not a reliable "this abort was mine" marker. + */ +const SELF_ABORT_TTL_MS = 30_000 +/** Interrupt-shaped failure signatures. v2 reports an interrupt as + * `{type:"aborted", message:"Step interrupted"}`, which the older + * `("interrupted by user" | type contains "cancel")` filter never matched. */ +const ABORT_ERROR_TYPE_RE = /^(abort|aborted|cancel|cancelled|canceled|interrupted)$/i +const ABORT_ERROR_MSG_RE = /step interrupted|interrupted by user|request cancelled|request canceled/i +/** Hard cap on plugin-initiated interrupts. An interrupt persists an errored + * assistant message that the UI surfaces as a send failure, so it is a last + * resort, not a routine recovery step. */ +const MAX_INTERRUPTS_PER_WINDOW = 2 +const INTERRUPT_WINDOW_MS = 10 * 60_000 +/** + * Minimum gap between two recovery injections for the same session. + * Synthetic turns are not free: sending one to a session that is already + * running supersedes its in-flight step, which the runtime reports as + * "Step interrupted". Debouncing collapses the several independent recovery + * paths (stall watchdog, intent nudge, tool-loop nudge) into at most one + * injection per interval. + */ +const DEFAULT_INJECT_INTERVAL_MS = 15_000 + const MAX_IDLE_SESSIONS = 50 const IDLE_CLEANUP_MS = 10 * 60_000 /** A compaction silent for this long without an ended/failed event is wedged — stale-clear the flag. */ @@ -374,10 +440,10 @@ function isTaskToolCall(ev: V2Event): boolean { // Plugin // --------------------------------------------------------------------------- -export default Plugin.define({ +export default define({ id: "auto-resume.v2", - setup: async (ctx) => { + setup: async (ctx: AutoResumePluginInput) => { const opts = (ctx.options ?? {}) as AutoResumeOptions const chunkTimeoutMs = opts.chunkTimeoutMs ?? DEFAULT_CHUNK_TIMEOUT_MS @@ -388,9 +454,9 @@ export default Plugin.define({ const maxBackoff = opts.maxBackoffMs ?? DEFAULT_MAX_BACKOFF_MS const loopMaxContinues = opts.loopMaxContinues ?? DEFAULT_LOOP_MAX_CONTINUES const loopWindowMs = opts.loopWindowMs ?? DEFAULT_LOOP_WINDOW_MS - const maxRecoveryRetries = opts.maxRecoveryRetries ?? DEFAULT_MAX_RECOVERY_RETRIES const debug = opts.debug ?? DEFAULT_DEBUG - const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS + const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS + const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS const dbg = (...args: unknown[]) => { if (debug) console.log("[auto-resume:debug]", ...args) @@ -398,6 +464,16 @@ export default Plugin.define({ function log(level: "info" | "warn" | "error", msg: string) { const line = `[auto-resume] ${msg}` + // Prefer the server log sink so plugin output is actually retrievable + // (console output from a hosted plugin is not captured anywhere useful). + try { + const appLog = (ctx as any)?.app?.log + if (typeof appLog === "function") { + void appLog.call((ctx as any).app, level === "info" ? "info" : level, line) + } + } catch { + // fall through to console + } if (level === "error") console.error(line) else if (level === "warn") console.warn(line) else console.log(line) @@ -421,6 +497,10 @@ export default Plugin.define({ lastRetryAt: 0, gaveUp: false, aborting: false, + selfAbortAt: 0, + interruptsThisWindow: 0, + interruptWindowStart: 0, + lastInjectAt: 0, recovering: false, oocLocked: false, oocLockReason: null, @@ -434,7 +514,6 @@ export default Plugin.define({ intentNudgeAttempts: 0, pendingRecoveryArmed: false, permissionPending: false, - waitingOnSubagent: false, lastWasTaskTool: false, idleSince: null, compacting: false, @@ -466,7 +545,6 @@ export default Plugin.define({ w.recentToolCalls = [] w.textParts.clear() w.lastAssistantText = "" - w.waitingOnSubagent = false } w.status = "busy" w.idleSince = null @@ -502,6 +580,84 @@ export default Plugin.define({ return w.continueTimestamps.length } + /** + * True while a plugin-initiated abort is still in flight. The runtime delivers + * `session.execution.interrupted` asynchronously after `interrupt()` returns, so + * the transient `w.aborting` boolean is not a reliable discriminator. + */ + function selfAbortActive(w: SessionWatch): boolean { + return w.selfAbortAt > 0 && Date.now() - w.selfAbortAt < SELF_ABORT_TTL_MS + } + + /** Does this failure signature mean "the turn was interrupted"? */ + function isAbortError(errType: string, errMsg: string): boolean { + return ABORT_ERROR_TYPE_RE.test(errType.trim()) || ABORT_ERROR_MSG_RE.test(errMsg) + } + + /** Remaining plugin-initiated interrupts allowed in the current window. */ + function interruptBudget(w: SessionWatch): number { + if (w.interruptWindowStart === 0 || Date.now() - w.interruptWindowStart > INTERRUPT_WINDOW_MS) { + w.interruptWindowStart = Date.now() + w.interruptsThisWindow = 0 + } + return MAX_INTERRUPTS_PER_WINDOW - w.interruptsThisWindow + } + + /** + * Fallback source of truth for "which sessions are running". v2's plugin + * SessionDomain does not expose `session.active()` (see Mte90/opencode-auto-resume#33), + * so when it is missing we derive the set from our own event-derived busy flags. + * Without this the subagent-wait guard below can never fire. + */ + function busySessions(): string[] { + const out: string[] = [] + for (const [sid, w] of sessions) { + if (w.status === "busy" && !w.userCancelled) out.push(sid) + } + return out + } + + /** + * Single choke point for every recovery injection. Guarantees at most one + * synthetic per `injectIntervalMs` and refuses to talk to a live session, + * because a turn sent to a busy session supersedes its in-flight step + * ("Step interrupted"). + * + * `allowDuringSelfAbort` is for the abort+resume escalation, which by + * construction runs inside the plugin's own abort window: it is the one + * injection that is *supposed* to follow `interrupt()`. That path skips + * both the abort-window check and the busy guard, because: + * - we just killed the step ourselves, so "this session is working" is + * false, and the runtime may not have delivered the idle transition + * yet; and + * - swallowing the continue there leaves a session interrupted with + * nothing to restart it, which is the failure this whole fix targets. + * It stays rate-limited by `w.aborting` and `MAX_INTERRUPTS_PER_WINDOW`. + */ + async function injectOnce( + sid: string, + text: string, + notification: string, + allowDuringSelfAbort = false, + ): Promise { + const w = ensureWatch(sid) + if (!allowDuringSelfAbort && selfAbortActive(w)) { + dbg(`${short(sid)} injection refused — inside our own abort window`) + return false + } + if (!allowDuringSelfAbort && w.status === "busy" && Date.now() - w.lastActivityAt < chunkTimeoutMs) { + dbg(`${short(sid)} injection refused — session busy and live (${Math.round((Date.now() - w.lastActivityAt) / 1000)}s since last event)`) + return false + } + const since = Date.now() - w.lastInjectAt + if (!allowDuringSelfAbort && w.lastInjectAt > 0 && since < injectIntervalMs) { + dbg(`${short(sid)} injection debounced — ${Math.round(since / 1000)}s since last (min ${injectIntervalMs / 1000}s)`) + return false + } + w.lastInjectAt = Date.now() + return notifyAndPrompt(sid, text, notification) + } + function cleanupIdleSessions() { const now = Date.now() const busy = new Set() @@ -591,7 +747,7 @@ export default Plugin.define({ } async function tryAbortAndResume(sid: string, w: SessionWatch): Promise { - if (w.aborting) return false + if (w.aborting || selfAbortActive(w)) return false if (w.oocLocked) { dbg(`${short(sid)} oocLocked — refusing abort+resume escalation`) return false @@ -600,8 +756,15 @@ export default Plugin.define({ dbg(`${short(sid)} mid-compaction — refusing abort+resume escalation`) return false } + if (interruptBudget(w) <= 0) { + w.continueTimestamps = [] + log("warn", `${short(sid)} interrupt budget exhausted (${MAX_INTERRUPTS_PER_WINDOW}/${INTERRUPT_WINDOW_MS / 60000}m) — skipping interrupt, loop guard reset`) + return false + } w.aborting = true - log("warn", `${short(sid)} escalating: interrupt + fresh continue`) + w.selfAbortAt = Date.now() + w.interruptsThisWindow++ + log("warn", `${short(sid)} escalating: interrupt + fresh continue (${w.interruptsThisWindow}/${MAX_INTERRUPTS_PER_WINDOW} per ${INTERRUPT_WINDOW_MS / 60000}m)`) try { await ctx.session.interrupt({ sessionID: sid }) } catch (err) { @@ -612,9 +775,15 @@ export default Plugin.define({ await new Promise((r) => setTimeout(r, 2_000)) w.aborting = false w.resumeAttempts = 0 - const ok = await notifyAndPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "abort+resume escalation") + // Deliberately does not recordContinue(): our own escalation must not feed + // the loop guard that triggered it, or the guard trips on its own output. + // Through the same choke point as every other path. The runtime + // delivers session.execution.interrupted asynchronously, so an + // escalation that lands on top of a concurrent recovery injection is + // exactly the case that produced the simultaneous bursts of synthetic + // continues (and the "Step interrupted" they caused). + const ok = await injectOnce(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "abort+resume escalation", true) if (ok) { - recordContinue(sid) w.lastRetryAt = Date.now() log("info", `${short(sid)} resumed after abort`) } @@ -646,15 +815,19 @@ export default Plugin.define({ dbg(`${short(sid)} mid-compaction — refusing recovery (reason: ${reason})`) return } - // Record intent first, then evaluate the loop guard, so the Nth - // continue within the window is the one that escalates. - recordContinue(sid) + if (selfAbortActive(w)) { + dbg(`${short(sid)} self-abort in flight — refusing recovery (reason: ${reason})`) + return + } + // Evaluate the loop guard BEFORE recording this continue, so the plugin + // never escalates on continues it injected itself. if (continuesInWindow(w) >= loopMaxContinues) { log("warn", `${short(sid)} hallucination loop (${loopMaxContinues} continues/${loopWindowMs / 1000}s) — abort+resume`) await tryAbortAndResume(sid, w) w.continueTimestamps = [] return } + recordContinue(sid) if (w.resumeAttempts >= maxRetries) { log("warn", `${short(sid)} giving up after ${maxRetries} attempts (${reason})`) w.gaveUp = true @@ -688,7 +861,7 @@ export default Plugin.define({ dbg(`${short(sid)} still busy and live at inject time (${Math.round((Date.now() - w.lastActivityAt) / 1000)}s since last event) — not interrupting`) return } - const ok = await notifyAndPrompt(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "stalled — retrying") + const ok = await injectOnce(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "stalled — retrying") w.pendingRecoveryArmed = false if (ok) { w.lastRetryAt = Date.now() @@ -711,13 +884,17 @@ export default Plugin.define({ dbg(`${short(sid)} mid-compaction — skipping ${kind} nudge`) return } - recordContinue(sid) + if (selfAbortActive(w)) { + dbg(`${short(sid)} self-abort in flight — skipping ${kind} nudge`) + return + } if (continuesInWindow(w) >= loopMaxContinues) { log("warn", `${short(sid)} loop guard before ${kind} nudge — abort+resume`) await tryAbortAndResume(sid, w) w.continueTimestamps = [] return } + recordContinue(sid) if (w[budgetKey] >= maxRetries) { dbg(`${short(sid)} ${kind} budget exhausted`) return @@ -725,10 +902,11 @@ export default Plugin.define({ w[budgetKey]++ log("info", `${short(sid)} ${kind} detected — sending targeted prompt (${w[budgetKey]}/${maxRetries})`) w.recovering = true - const ok = await notifyAndPrompt(sid, prompt, "recovering: " + kind) + // injectOnce debounces: this nudge counts once, not alongside a + // concurrent stall-watchdog injection. + const ok = await injectOnce(sid, prompt, "recovering: " + kind) w.recovering = false if (ok) { - recordContinue(sid) touch(sid) markBusy(sid) } @@ -843,14 +1021,22 @@ async function inspectOnIdle(sid: string) { * `session.active()` exists on the v2 client but is not part of the * plugin SessionDomain pick — access defensively; empty map if missing. */ + /** + * `session.active()` exists on the v2 client but is not part of the plugin + * SessionDomain pick, so it is normally absent — see + * Mte90/opencode-auto-resume#33. Fall back to our own event-derived busy + * flags rather than returning an empty set: returning `[]` made the + * subagent-wait guard below unreachable, so a parent session that went + * quiet while its child ran was declared stalled and interrupted. + */ async function getActiveSessions(): Promise { try { const fn = (ctx.session as any).active - if (typeof fn !== "function") return [] + if (typeof fn !== "function") return busySessions() const active = await fn.call(ctx.session) return Object.keys(active ?? {}).filter((k) => typeof k === "string") } catch { - return [] + return busySessions() } } @@ -886,8 +1072,8 @@ async function inspectOnIdle(sid: string) { dbg(`${short(sid)} silent but permission pending — skipping`) continue } - if (w.waitingOnSubagent) { - dbg(`${short(sid)} silent but waiting on subagent — skipping`) + if (selfAbortActive(w)) { + dbg(`${short(sid)} silent but inside our own abort window — skipping`) continue } // If another session is actively running and this one went silent @@ -939,6 +1125,13 @@ async function inspectOnIdle(sid: string) { const w = ensureWatch(sid) w.pendingRecoveryArmed = false w.selfRecovery = false // stale — no in-flight recovery prompt by the time we idle + // Never run the idle heuristics on top of our own abort: the idle + // transition caused by interrupt() would otherwise trigger a second, + // independent injection. + if (selfAbortActive(w)) { + dbg(`${short(sid)} idle after plugin-initiated abort — skipping targeted recovery`) + return + } void inspectOnIdle(sid) return } @@ -950,7 +1143,13 @@ async function inspectOnIdle(sid: string) { // interrupt should suppress recovery; shutdown/superseded/inactivity // (or a missing reason on older runtimes) must not. const reason = typeof ev.data?.reason === "string" ? ev.data.reason : undefined - w.userCancelled = !w.aborting && (reason === undefined || reason === "user") + // Use the latch, not the transient `aborting` flag: the runtime + // delivers this event asynchronously, after `aborting` may have + // already been cleared. Only a genuine user stop latches + // userCancelled; `shutdown`/`superseded`/`inactivity` must not + // permanently disable recovery for the session. + const mine = w.aborting || selfAbortActive(w) + if (!mine) w.userCancelled = reason === undefined || reason === "user" // A user interrupt also ends any in-flight compaction. w.compacting = false w.compactionStartedAt = null @@ -1021,6 +1220,28 @@ async function inspectOnIdle(sid: string) { touch(sid) return } + // Liveness between step start and the first token. Without these a + // step that starts and then produces nothing for `chunkTimeoutMs` + // looks identical to a stall. + case "session.step.streamed": + case "session.text.started": + case "session.reasoning.started": + case "session.synthetic": { + const sid = sidOf(ev) + if (!sid) return + touch(sid) + return + } + // A long tool-input generation (e.g. a big heredoc write) emits no + // text deltas; treating that silence as a stall resumes a session + // that is in fact working. + case "session.tool.input.started": + case "session.tool.input.delta": { + const sid = sidOf(ev) + if (!sid) return + touch(sid) + return + } case "session.text.delta": case "session.reasoning.delta": { const sid = sidOf(ev) @@ -1116,14 +1337,25 @@ async function inspectOnIdle(sid: string) { const w = ensureWatch(sid) const errMsg = String(ev.data?.error?.message ?? "") const errType = String(ev.data?.error?.type ?? "") - maybeLockOoc(sid, errMsg) - markIdle(sid) - w.pendingRecoveryArmed = true // our delayed recovery must survive this idle transition - if (errMsg.includes("interrupted by user") || errType.includes("cancel")) { - dbg(`${short(sid)} failure was user-initiated — not recovering`) + // Our own abort: do not recover from it, and do not arm the delayed + // recovery (arming here is what kept the watchdog injecting into an + // already-interrupted session). + if (w.aborting || selfAbortActive(w)) { + dbg(`${short(sid)} ${ev.type} is this plugin's own abort (${errType || "error"}) — not recovering`) + w.recovering = false w.pendingRecoveryArmed = false return } + // Any interrupt-shaped failure (ours or the user's) is not recoverable. + // v2 reports these as {type:"aborted", message:"Step interrupted"}. + if (isAbortError(errType, errMsg)) { + dbg(`${short(sid)} failure was an interrupt (${errType || "error"}) — not recovering`) + w.pendingRecoveryArmed = false + return + } + maybeLockOoc(sid, errMsg) + markIdle(sid) + w.pendingRecoveryArmed = true // our delayed recovery must survive this idle transition log("warn", `${short(sid)} ${ev.type}: ${errType || "error"} ${errMsg.slice(0, 160)}`) void recover(sid, `${ev.type}${isStreamingFailure(errMsg) ? " (streaming)" : ""}`) return From 67c34ff0fe13aa185a8042f5675b9af8aa53cbb0 Mon Sep 17 00:00:00 2001 From: famewolf Date: Tue, 29 Sep 2026 12:00:09 -0400 Subject: [PATCH 11/57] fix(v2): stand down when a question or permission is awaiting the user MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The v2 stand-down guard could never fire. It tested `part.state.status === "pending"`, but the v2 message projection never writes "pending" — an in-flight tool part reads "running" and settles to "completed" or "error" (`pending` was the v1 spelling). With that branch unreachable, auto-resume injected a synthetic `continue` over an unanswered `question`, which surfaced as the tool call being interrupted / "Step interrupted". Reported in #33. A second, independent defect: markIdle() cleared `permissionPending` immediately before inspectOnIdle() consulted it. The `session.idle` case calls markIdle() and then inspectOnIdle(), so the flag was always false by the time the guard that honours it ran — dead on exactly the transition it exists for. `permission.replied` is now the only thing that clears it, plus a TTL stale-clear mirroring the compaction one, so a `permission.asked` whose reply never arrives cannot strand a session forever. Interactive tools (`question`, `permission`, `ask`, `confirm`) stand down unless they reached "completed" — a `question` that came back "error" was interrupted, not answered. Verified against the real v2 message shape: ctx.session.context() returns a plain array, and the tool-part states actually observed in a live session are completed / error / running, never pending. Tests: src/v2/index.stand-down.test.ts drives the real v2 seam (event.subscribe -> async iterable, session.synthetic). It includes a mandatory control asserting recovery DOES fire when nothing is pending, plus over-blocking controls (answered question, non-interactive tool erroring, permission asked-then-replied) so the fix cannot silently disable auto-resume. 7 of its 13 assertions fail against the previous code. Full suite: 591 pass / 0 fail. tsc --noEmit clean. --- src/v2/index.stand-down.test.ts | 244 ++++++++++++++++++++++++++++++++ src/v2/index.ts | 78 +++++++++- 2 files changed, 315 insertions(+), 7 deletions(-) create mode 100644 src/v2/index.stand-down.test.ts diff --git a/src/v2/index.stand-down.test.ts b/src/v2/index.stand-down.test.ts new file mode 100644 index 0000000..16371b7 --- /dev/null +++ b/src/v2/index.stand-down.test.ts @@ -0,0 +1,244 @@ +import { describe, test, expect } from "bun:test" +import { readFileSync } from "node:fs" +import { join } from "node:path" +import plugin from "./index" + +const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_standdown" +/** Older than the 15m default activeUserWindowMs, so guard (b) cannot mask a failure. */ +const OLD_USER_MS = 30 * 60_000 + +/** Text that trips `ready-to-continue` but NOT `isUserHandoff` (no "?", no "should I"). */ +const READY_TEXT = "Ready to continue with task" + +// --------------------------------------------------------------------------- +// Mock host: drives the real v2 seam (`event.subscribe` async iterable) and +// records the real injection primitive (`session.synthetic`). +// --------------------------------------------------------------------------- + +type Injected = { kind: string; text?: string; description?: string } + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const textPart = (t: string) => ({ type: "text", text: t }) + +function toolPart(name: string, status: string) { + return { + type: "tool", + id: "prt_" + name, + name, + callID: "call_" + name, + executed: status === "completed", + state: { status, input: { questions: [{ header: "Pick", question: "A or B?" }] } }, + time: { created: Date.now() }, + } +} + +const oldUserTurn = () => ({ + type: "user", + id: "msg_u0", + time: { created: Date.now() - OLD_USER_MS }, + content: [textPart("go ahead")], +}) + +/** Assistant turn carrying recovery-triggering text and an optional tool part. */ +function assistantTurn(text: string, tool?: ReturnType) { + const content: any[] = [textPart(text)] + if (tool) content.push(tool) + return { type: "assistant", id: "msg_a1", time: { created: Date.now() - 60_000 }, content } +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +/** Events for one assistant turn that streams `text`, then goes idle. */ +function turnEvents(text: string, toolName?: string) { + const seq: any[] = [ + ev("session.execution.started"), + ev("session.text.delta", { messageID: "msg_a1", delta: text }), + ev("session.text.ended", { messageID: "msg_a1" }), + ] + if (toolName) seq.push(ev("session.tool.called", { tool: toolName })) + seq.push(ev("session.idle")) + return seq +} + +/** Replay a turn and report whether the plugin injected a recovery. */ +async function replay(opts: { messages: any[]; events: any[] }): Promise { + const injected: Injected[] = [] + const stream = makeEventStream() + const ctx: any = { + event: stream, + app: { log: () => {} }, + session: { + // v2 returns a plain ARRAY here, not { messages }. Asserted separately below. + context: async () => opts.messages, + active: async () => ({}), + synthetic: async (a: any) => { + injected.push({ kind: "synthetic", ...a }) + return {} + }, + prompt: async (a: any) => { + injected.push({ kind: "prompt", ...a }) + return {} + }, + }, + } + const cleanup = await (plugin as any).setup(ctx) + for (const e of opts.events) { + stream.push(e) + await wait(10) + } + await wait(500) // handleEvent is sync; its work is async + ;(cleanup as (() => void) | undefined)?.() + return injected +} + +// ============================================================================ +// BEHAVIOURAL — drive the real plugin and assert on real injections. +// ============================================================================ + +describe("v2: stand down while the user owes us an answer", () => { + test("CONTROL: nothing pending -> recovery fires (otherwise the rest proves nothing)", async () => { + const injected = await replay({ + messages: [oldUserTurn(), assistantTurn(READY_TEXT)], + events: turnEvents(READY_TEXT), + }) + expect(injected.length).toBeGreaterThan(0) + expect(injected[0].kind).toBe("synthetic") + }) + + // The v2 message projection never writes "pending" — it writes "running". + // The old guard tested for "pending" literally, so it could never fire and + // every nudge went out over an unanswered `question` (#33 / valentimarco). + for (const status of ["running", "error", "pending"]) { + test(`question tool unanswered, state.status=${status} -> no injection`, async () => { + const injected = await replay({ + messages: [oldUserTurn(), assistantTurn(READY_TEXT, toolPart("question", status))], + events: turnEvents(READY_TEXT, "question"), + }) + expect(injected).toEqual([]) + }) + } + + test("question tool ANSWERED (status=completed) -> recovery may fire", async () => { + const injected = await replay({ + messages: [oldUserTurn(), assistantTurn(READY_TEXT, toolPart("question", "completed"))], + events: turnEvents(READY_TEXT, "question"), + }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("non-interactive tool that errored (status=error) -> recovery may fire", async () => { + const injected = await replay({ + messages: [oldUserTurn(), assistantTurn(READY_TEXT, toolPart("shell", "error"))], + events: turnEvents(READY_TEXT, "shell"), + }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("permission.asked then session.idle -> no injection", async () => { + const injected = await replay({ + messages: [oldUserTurn(), assistantTurn(READY_TEXT)], + events: [ + ev("session.execution.started"), + ev("session.text.delta", { messageID: "msg_a1", delta: READY_TEXT }), + ev("session.text.ended", { messageID: "msg_a1" }), + ev("permission.asked"), + ev("session.idle"), + ], + }) + expect(injected).toEqual([]) + }) + + test("permission asked then REPLIED, then session.idle -> recovery may fire", async () => { + const injected = await replay({ + messages: [oldUserTurn(), assistantTurn(READY_TEXT)], + events: [ + ev("session.execution.started"), + ev("session.text.delta", { messageID: "msg_a1", delta: READY_TEXT }), + ev("session.text.ended", { messageID: "msg_a1" }), + ev("permission.asked"), + ev("permission.replied"), + ev("session.idle"), + ], + }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("unanswered text question -> no injection (hand-off guard still works)", async () => { + const injected = await replay({ + messages: [oldUserTurn(), assistantTurn("Which option do you want me to take?")], + events: turnEvents("Which option do you want me to take?"), + }) + expect(injected).toEqual([]) + }) +}) + +// ============================================================================ +// CONTRACT — fail deterministically if someone reverts the fix. +// ============================================================================ + +describe("v2: contract assertions on source", () => { + test("FIX: the tool-state test is not a literal 'pending'-only comparison", () => { + // The old single condition could never be true on v2. + expect(SOURCE).not.toMatch(/t\.startsWith\("tool"\)\)\s*&&\s*part\?\.state\?\.status === "pending"/) + expect(SOURCE).toMatch(/TOOL_STATE_RUNNING/) + expect(SOURCE).toMatch(/TOOL_STATE_COMPLETED/) + }) + + test("FIX: markIdle no longer clears permissionPending", () => { + const start = SOURCE.indexOf("function markIdle") + expect(start).toBeGreaterThan(-1) + // Scope to markIdle's own body: the stale-clear helper that follows it is + // allowed to reset the flag, markIdle is not. + const end = SOURCE.indexOf("function clearStalePermissionFlag", start) + expect(end).toBeGreaterThan(start) + const body = SOURCE.slice(start, end) + expect(body).not.toMatch(/w\.permissionPending = false/) + // ...but a stale unanswered prompt still cannot strand the session forever. + expect(body).toMatch(/clearStalePermissionFlag/) + }) + + test("FIX: permission.replied is paired with its timestamp", () => { + expect(SOURCE).toMatch(/case "permission\.asked"[\s\S]{0,400}permissionPendingAt = Date\.now\(\)/) + expect(SOURCE).toMatch(/case "permission\.replied"[\s\S]{0,400}permissionPendingAt = null/) + }) + + test("FIX: a stale permission flag is cleared by TTL, not by the idle transition", () => { + expect(SOURCE).toMatch(/PERMISSION_STALE_TTL_MS/) + expect(SOURCE).toMatch(/function clearStalePermissionFlag/) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index 8622ffb..639d582 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -121,6 +121,8 @@ interface SessionWatch { // Guards permissionPending: boolean + /** Timestamp of the latest `permission.asked` (stale-guard for the flag). */ + permissionPendingAt: number | null lastWasTaskTool: boolean idleSince: number | null /** Set while the session is mid native compaction — recovery must never interrupt it. */ @@ -204,8 +206,29 @@ const MAX_IDLE_SESSIONS = 50 const IDLE_CLEANUP_MS = 10 * 60_000 /** A compaction silent for this long without an ended/failed event is wedged — stale-clear the flag. */ const COMPACTION_STALE_TTL_MS = 30 * 60_000 +/** A `permission.asked` unanswered this long is treated as abandoned — stale-clear the flag. */ +const PERMISSION_STALE_TTL_MS = 30 * 60_000 const TEXT_BUFFER_TRIM_LEN = 20_000 +/** + * Tool-part states as reported by the v2 message projection. + * + * v2 never writes "pending": an in-flight tool part reads "running", and it + * settles to "completed" or "error". "pending" was the v1 spelling, so a guard + * that tested for it literally could never fire on v2. + */ +const TOOL_STATE_RUNNING = "running" +/** Kept for v1-shaped payloads; harmless on v2. */ +const TOOL_STATE_PENDING = "pending" +const TOOL_STATE_COMPLETED = "completed" + +/** + * Tools whose only real completion is the user answering. A `question` that + * comes back "error" was interrupted, not answered, so it must still hold the + * turn open. + */ +const AWAITING_USER_TOOLS = new Set(["question", "permission", "ask", "confirm"]) + const CONTINUE_PROMPT = "continue" const TOOL_TEXT_RECOVERY_PROMPT = @@ -514,6 +537,7 @@ export default define({ intentNudgeAttempts: 0, pendingRecoveryArmed: false, permissionPending: false, + permissionPendingAt: null, lastWasTaskTool: false, idleSince: null, compacting: false, @@ -558,7 +582,26 @@ export default define({ w.status = "idle" w.idleSince = Date.now() } - w.permissionPending = false + // NOTE: this used to clear `permissionPending`. The `session.idle` handler + // calls markIdle() and then inspectOnIdle(), so the flag was always false by + // the time the guard that honours it ran — the permission guard was dead on + // exactly the transition it exists for. `permission.replied` is now the only + // thing that clears it, plus a stale-clear for a `permission.asked` whose + // reply never arrives. + clearStalePermissionFlag(sid, w) + } + + /** + * A `permission.asked` with no matching `permission.replied` would otherwise + * stand the session down forever. Mirrors the compaction stale-guard. + */ + function clearStalePermissionFlag(sid: string, w: SessionWatch) { + if (!w.permissionPending) return + if (w.permissionPendingAt && Date.now() - w.permissionPendingAt > PERMISSION_STALE_TTL_MS) { + dbg(`${short(sid)} permission prompt silent >${PERMISSION_STALE_TTL_MS / 60000}m — clearing flag`) + w.permissionPending = false + w.permissionPendingAt = null + } } function recordContinue(sid: string) { @@ -949,14 +992,29 @@ export default define({ const result = await ctx.session.context({ sessionID: sid }) const messages: unknown[] = Array.isArray(result) ? result : ((result as { messages?: unknown[] })?.messages ?? []) if (messages.length === 0) return false - const newest = messages[messages.length - 1] as { type?: string; content?: { type?: string; state?: { status?: string } }[]; time?: { created?: number }; info?: { time?: { created?: number } } } - // (a) A pending tool call (e.g. the `question` tool) means the model is waiting - // on the user's answer — a synthetic nudge here would be interrupted by their - // real reply ("Step interrupted"). + const newest = messages[messages.length - 1] as { + type?: string + content?: { type?: string; name?: string; state?: { status?: string } }[] + time?: { created?: number } + info?: { time?: { created?: number } } + } + // (a) A tool call still awaiting its result means the model is waiting on the + // user — a synthetic nudge here would be interrupted by their real reply + // ("Step interrupted"). + // + // This used to test `state.status === "pending"`, which v2 never emits, so + // the branch was unreachable and the nudge fired with an unanswered + // `question` on screen. See TOOL_STATE_* above. if (newest?.type === "assistant") { for (const part of newest.content ?? []) { const t = part?.type ?? "" - if ((t === "tool_use" || t === "tool" || t === "tool_call" || t.startsWith("tool")) && part?.state?.status === "pending") return true + if (!(t === "tool_use" || t === "tool" || t === "tool_call" || t.startsWith("tool"))) continue + const status = part?.state?.status + if (status === TOOL_STATE_RUNNING || status === TOOL_STATE_PENDING) return true + // An interactive tool only reaches a terminal state once the user has + // actually answered. "error" here means the question was interrupted, + // not that it was answered, so keep standing down. + if (status !== TOOL_STATE_COMPLETED && AWAITING_USER_TOOLS.has(part?.name ?? "")) return true } } // (b) The user was recently active — they are mid-conversation, not stuck. @@ -1068,6 +1126,9 @@ async function inspectOnIdle(sid: string) { } const silence = now - w.lastActivityAt if (silence < chunkTimeoutMs + gracePeriodMs) continue + // Same stale-guard as the compaction flag: an unanswered `permission.asked` + // must not stand this session down forever. + clearStalePermissionFlag(sid, w) if (w.permissionPending) { dbg(`${short(sid)} silent but permission pending — skipping`) continue @@ -1300,7 +1361,9 @@ async function inspectOnIdle(sid: string) { case "permission.asked": { const sid = sidOf(ev) if (!sid) return - ensureWatch(sid).permissionPending = true + const w = ensureWatch(sid) + w.permissionPending = true + w.permissionPendingAt = Date.now() return } case "permission.replied": { @@ -1308,6 +1371,7 @@ async function inspectOnIdle(sid: string) { if (!sid) return const w = ensureWatch(sid) w.permissionPending = false + w.permissionPendingAt = null touch(sid) return } From 4855d1854133c642816d3f3c171de5260f368e47 Mon Sep 17 00:00:00 2001 From: famewolf Date: Tue, 29 Sep 2026 21:14:48 -0400 Subject: [PATCH 12/57] fix(v2): handle the session.revert.* family so a rewind drops watch state MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The port documented that v2 reverts surface as `session.revert.cleared` / `session.revert.committed` and kept a defensive legacy `session.reverted` case, but it never handled the v2 events themselves. On a real v2 runtime no case matched, so a revert left the session's watch state — including any armed recovery — intact against turns that had just been rewound. Verified against the running opencode server: all three names are declared event schemas with live store handlers. - `session.revert.staged` is intermediate (the server sets a `revert` field there), so it only counts as activity and keeps the state. - `session.revert.cleared` and `session.revert.committed` are terminal (the server clears that field on both) and drop the state, matching the existing `session.deleted` behaviour. The legacy `session.reverted` case stays for v1-shaped runtimes. --- src/v2/index.ts | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/src/v2/index.ts b/src/v2/index.ts index 639d582..6b127b4 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -1224,6 +1224,25 @@ async function inspectOnIdle(sid: string) { sessions.delete(sid) return } + // --- reverts --- + // v2 emits a three-stage revert family. `staged` is intermediate + // (the user can still clear it), while `cleared` and `committed` + // are terminal; drop our watch state on the terminal stages so a + // rewind cannot leave a stale recovery armed against old turns. + case "session.revert.staged": { + const sid = sidOf(ev) + if (!sid) return + // Staging is user activity, but not terminal: keep the state. + touch(sid) + return + } + case "session.revert.cleared": + case "session.revert.committed": { + const sid = sidOf(ev) + if (!sid) return + sessions.delete(sid) + return + } // v1 legacy: not emitted in v2 (reverts now surface as // `session.revert.cleared` / `session.revert.committed`). Kept // defensively so older runtimes still drop their watch state. From 7135c468a74e8d2fd626d34c48541625f308b071 Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 14:15:23 -0400 Subject: [PATCH 13/57] feat(v2): read every v1 option, warn on unknown keys, and complete busyStallStrategy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Additive to the existing v2 port on this branch — no rebase, no changes to src/index.ts. The v1 side of the plugin is untouched here. The v2 port read 15 of v1's 31 options and silently ignored the other 16, so a v1 config ported to v2 quietly ran on defaults. Three changes close that: - All 31 v1 option names are now read. `maxRecoveryRetries` is accepted as a fallback for v2's `maxRetries`, so a v1 config needs no edits. - 8 of them are accepted but not applied, because the v2 feature they tune is not ported (no token read, no reachable todo state, no discovery sweep). They are listed as accepted-but-inert in the startup line and explained in docs/known-issues-v2.md rather than left to look like live behaviour. - Any option key this build does not know produces one warn line at startup. v1 dropped unknown keys silently, which is how the gap went unnoticed. busyStallStrategy is now complete rather than half-wired: "off" skips the stall and "abort" interrupts the wedged step before continuing, mirroring the v1 branch. Previously only "off" was handled, so "abort" silently behaved as "continue" while appearing in the recognised set. src/v2/index.options.test.ts covers all of the above behaviourally, with a control per group, plus a source contract that fails if the declared options type and the recognised set drift apart. 630 pass, 0 fail. Also mirrors two upstream changes the v2 build had not picked up: - e1b8374: targetedRecovery now counts an attempt only after the prompt is delivered, so a rejected send no longer burns a retry. - DEFAULT_ACTIVE_USER_WINDOW_MS 15 min -> 5 min, matching v1. build:v2 was missing from package.json, so dist/v2/ could not be rebuilt from this branch at all. Added, along with dist/v2/index.js in files. --- README.md | 20 +- docs/known-issues-v2.md | 57 ++++ package.json | 5 +- src/v2/index.open-shell.test.ts | 238 ++++++++++++++++ src/v2/index.options.test.ts | 413 ++++++++++++++++++++++++++++ src/v2/index.ts | 468 ++++++++++++++++++++++++++++++-- 6 files changed, 1176 insertions(+), 25 deletions(-) create mode 100644 docs/known-issues-v2.md create mode 100644 src/v2/index.open-shell.test.ts create mode 100644 src/v2/index.options.test.ts diff --git a/README.md b/README.md index cc130aa..a86127b 100644 --- a/README.md +++ b/README.md @@ -418,6 +418,18 @@ Disable via `"-auto-resume.v2"` in `plugins`. - **Step-by-step install guide:** [docs/v2/installing.md](docs/v2/installing.md) - **Migration notes (v1 → v2, stable validation):** [docs/v2/migration.md](docs/v2/migration.md) +- **What the v2 port does and does not yet do:** [docs/known-issues-v2.md](docs/known-issues-v2.md) + +The v2 build reads every option in the table below. A handful of them are +accepted without being applied yet, because the v2 feature they tune is not +ported — they are listed in the startup line as `accepted-but-inert=…` and +explained in [docs/known-issues-v2.md](docs/known-issues-v2.md). An option name +this build does not know at all produces a single warning at startup, so a typo +or an unported key is never silent. + +### Configurable options + +Defaults are the same on v1 and v2 unless a row says otherwise. ### Configurable options @@ -453,6 +465,12 @@ Disable via `"-auto-resume.v2"` in `plugins`. | `contextSaturationThreshold` | `0.85` | Ratio of used/usable context that routes a saturated parent to magic-context `ctx-wrapup` (only when magic-context is installed) | | `activeUserWindowMs` | `900000` | Inbound-user-message recency window (15 min) during which idle nudges stand down (user likely composing) | | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | +| `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | + +Accepted but **not applied** on v2: `contextSaturationThreshold`, +`subagentNativeCompactionEnabled`, `silentDeadStreamMinTokens`, `subagentWaitMs`, +`discoveryDelayMs`, `toolTextCheckDelayMs`, `thinkingToolRecoveryPrompt`, +`doneWithoutWorkPrompt`. See [docs/known-issues-v2.md](docs/known-issues-v2.md) for why. Message patterns are matched case-insensitively. Error names use exact match. @@ -463,7 +481,7 @@ Message patterns are matched case-insensitively. Error names use exact match. | `ABORT_CONTINUE_DELAY_MS` | `2000` | Delay between abort and continue | | `MAX_IDLE_SESSIONS` | `50` | Idle session map cap before cleanup | | `IDLE_CLEANUP_MS` | `600000` | Idle session age before cleanup (10 min) | -| `SESSION_DISCOVERY_INTERVAL_MS` | `60000` | `session.list()` poll interval (60s) | +| `SESSION_DISCOVERY_INTERVAL_MS` | `60000` | `session.list()` poll interval (60s) — v1 only; the v2 build has no discovery sweep | ## Verification diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md new file mode 100644 index 0000000..9f4bfe1 --- /dev/null +++ b/docs/known-issues-v2.md @@ -0,0 +1,57 @@ +# Known issues — v2 port + +Status of the v1 → v2 port as of the current branch. This file is the +authoritative list; the code refers back to it by name from the startup path, so +a config key that this build ignores is never silent. + +## Options that are accepted but not applied + +These keys are read without error, and are listed in the plugin's startup line as +`accepted-but-inert=…`. They stay in `AutoResumeOptions` so an existing v1 +config keeps loading unchanged. + +| Option | Why it is inert on v2 | +| --- | --- | +| `contextSaturationThreshold` | v2 has no token or context-limit read on the plugin context, so there is nothing to compare the threshold against. | +| `subagentNativeCompactionEnabled` | Same: without a context-limit read, saturation cannot be detected, so there is nothing to gate a compaction on. | +| `silentDeadStreamMinTokens` | Same. The heuristic it tunes compares generated token count against a floor. | +| `subagentWaitMs` | The v1 orphan-watch timer that this delays has no v2 counterpart; the v2 port decides parent-vs-stalled from its own event-derived busy set. | +| `discoveryDelayMs` | v2 has no discovery sweep. The port starts watching from the events it receives rather than by enumerating existing sessions on a delay. | +| `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | +| `thinkingToolRecoveryPrompt` | The thinking-contains-a-tool-call detector is part of the v1 idle-nudge pass and is not ported. | +| `doneWithoutWorkPrompt` | Both v1 use sites for this prompt are gated on tracked todo state. v2 has no todo state, so the prompt has no trigger. | + +`doneWithoutDetailsPrompt` **is** applied on v2, and is the replacement for +`doneWithoutWorkPrompt`: it fires on a terse done-claim regardless of todos. + +## Unrecognised options warn once + +Any key in the plugin's config that is not in the recognised set produces a +single `warn` line at startup: + +``` +[auto-resume] ignoring unrecognised option(s): foo, bar — this v2 build does not read them. +``` + +This is a v2 addition. v1 silently ignored unknown keys, which is how a typo or +a dropped port went unnoticed. + +## Counter semantics changed from v1 + +`targetedRecovery` now counts an attempt only after the prompt is actually +delivered. v1 incremented first and lost the budget when a send was rejected; +the fix is mte090's `e1b8374`, applied to the v2 code path. The stall path +(`recover`) keeps v1's ordering — it escalates on a stall, not on a send. + +## Renamed and aliased + +- `maxRetries` is the v2 name. `maxRecoveryRetries` is v1's name for the same + knob and is still accepted as a fallback, so a v1 config ports unchanged. +- `activeUserWindowMs` defaults to `300000` (5 min) on both v1 and v2, matching + upstream `e1b8374`. The v2 port shipped with `900000`, which stood down for + three times longer than v1 after any user message. + +## v2-only options + +- `injectIntervalMs` — minimum gap between recovery injections for one session. + No v1 equivalent. diff --git a/package.json b/package.json index 12e30af..7856ea7 100644 --- a/package.json +++ b/package.json @@ -22,12 +22,15 @@ "type": "module", "main": "dist/index.js", "scripts": { - "build": "bun build src/index.ts --outdir dist --target bun", + "build": "bun run build:v1 && bun run build:v2", + "build:v1": "bun build src/index.ts --outdir dist --target bun", + "build:v2": "bun build src/v2/index.ts --outfile dist/v2/index.js --target bun", "dev": "bun build src/index.ts --outdir dist --target bun --watch", "test": "bun test", "prepublishOnly": "bun run build" }, "files": [ "dist/index.js", + "dist/v2/index.js", "README.md", "LICENSE" ], diff --git a/src/v2/index.open-shell.test.ts b/src/v2/index.open-shell.test.ts new file mode 100644 index 0000000..5318abc --- /dev/null +++ b/src/v2/index.open-shell.test.ts @@ -0,0 +1,238 @@ +import { describe, test, expect } from "bun:test" +import { readFileSync } from "node:fs" +import { join } from "node:path" +import plugin from "./index" + +const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_openshell" +const OTHER_SID = "ses_someone_else" +/** Older than the 15m default activeUserWindowMs, so guard (b) cannot mask a failure. */ +const OLD_USER_MS = 30 * 60_000 +/** Trips `ready-to-continue` but NOT `isUserHandoff` (no "?", no "should I"). */ +const READY_TEXT = "Ready to continue with task" + +type Injected = { kind: string; text?: string; description?: string } + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const textPart = (t: string) => ({ type: "text", text: t }) + +const oldUserTurn = () => ({ + type: "user", + id: "msg_u0", + time: { created: Date.now() - OLD_USER_MS }, + content: [textPart("go ahead")], +}) + +function assistantTurn(text: string) { + return { type: "assistant", id: "msg_a1", time: { created: Date.now() - 60_000 }, content: [textPart(text)] } +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +// --- shell process-registry events ----------------------------------------- +// These deliberately mirror the real payload shape observed on the running +// server: `shell.created` carries NO top-level sessionID — it is nested at +// `data.info.metadata.sessionID` — and the exit events carry only the shell id. +// `shell.deleted` is given a different id family, as observed. + +const shellCreated = (id: string, sessionID: string = SID) => ({ + type: "shell.created", + data: { info: { id, command: "sleep 12", status: "running", metadata: { sessionID } } }, +}) +/** Built explicitly: a default parameter would silently substitute a valid id. */ +const shellCreatedRaw = (info: Record) => ({ type: "shell.created", data: { info } }) +const shellExited = (id: string) => ({ type: "shell.exited", data: { id, status: "exited", exit: 0 } }) +const shellDeleted = (id: string) => ({ type: "shell.deleted", data: { id } }) + +/** One assistant turn that streams `text`, then goes idle. */ +function turnEvents(text: string) { + return [ + ev("session.execution.started"), + ev("session.text.delta", { messageID: "msg_a1", delta: text }), + ev("session.text.ended", { messageID: "msg_a1" }), + ev("session.idle"), + ] +} + +/** Replay events and report whether the plugin injected a recovery. */ +async function replay(events: any[], messages?: any[]): Promise { + const injected: Injected[] = [] + const stream = makeEventStream() + const ctx: any = { + event: stream, + app: { log: () => {} }, + session: { + context: async () => messages ?? [oldUserTurn(), assistantTurn(READY_TEXT)], + // Empty: no other session is active, so the `lastWasTaskTool` branch + // (which needs `others.length > 0`) cannot mask any failure below. + active: async () => ({}), + synthetic: async (a: any) => { + injected.push({ kind: "synthetic", ...a }) + return {} + }, + prompt: async (a: any) => { + injected.push({ kind: "prompt", ...a }) + return {} + }, + }, + } + const cleanup = await (plugin as any).setup(ctx) + for (const e of events) { + stream.push(e) + await wait(10) + } + await wait(500) // handleEvent is sync; its work is async + ;(cleanup as (() => void) | undefined)?.() + return injected +} + +// ============================================================================ +// BEHAVIOURAL +// ============================================================================ + +describe("v2: a session running a shell is working, not idle", () => { + test("CONTROL: no shell involved -> recovery fires (otherwise the rest proves nothing)", async () => { + const injected = await replay(turnEvents(READY_TEXT)) + expect(injected.length).toBeGreaterThan(0) + }) + + test("shell open -> no injection (parent parked on a background job)", async () => { + const injected = await replay([shellCreated("sh_aaa"), ...turnEvents(READY_TEXT)]) + expect(injected).toEqual([]) + }) + + test("shell open, then exited -> recovery may fire", async () => { + const injected = await replay([shellCreated("sh_aaa"), shellExited("sh_aaa"), ...turnEvents(READY_TEXT)]) + expect(injected.length).toBeGreaterThan(0) + }) + + test("two shells open, one exits -> still suppressed", async () => { + const injected = await replay([ + shellCreated("sh_aaa"), + shellCreated("sh_bbb"), + shellExited("sh_aaa"), + ...turnEvents(READY_TEXT), + ]) + expect(injected).toEqual([]) + }) + + test("another session's open shell does not suppress us", async () => { + const injected = await replay([shellCreated("sh_zzz", OTHER_SID), ...turnEvents(READY_TEXT)]) + expect(injected.length).toBeGreaterThan(0) + }) + + // A phantom entry from any of these would suppress recovery forever. + for (const [label, info] of [ + ["metadata present but sessionID missing", { id: "sh_bad", metadata: {} }], + ["metadata absent entirely", { id: "sh_bad" }], + ["sessionID is not a string", { id: "sh_bad", metadata: { sessionID: 42 } }], + ["no id", { metadata: { sessionID: SID } }], + ["info is a bare string", "not-an-object"], + ] as Array<[string, unknown]>) { + test(`malformed shell.created (${label}) is ignored, not recorded`, async () => { + const injected = await replay([shellCreatedRaw(info as Record), ...turnEvents(READY_TEXT)]) + expect(injected.length).toBeGreaterThan(0) + }) + } + + test("shell.exited for an unknown id does not throw or suppress", async () => { + const injected = await replay([shellExited("sh_never_seen"), ...turnEvents(READY_TEXT)]) + expect(injected.length).toBeGreaterThan(0) + }) + + test("shell.deleted (different id family) cannot close an open shell", async () => { + // Documents the reason `shell.deleted` is not wired up: its id never + // matches the one recorded at create, so it is not a usable exit signal. + const injected = await replay([shellCreated("sh_aaa"), shellDeleted("sh_bbb"), ...turnEvents(READY_TEXT)]) + expect(injected).toEqual([]) + }) +}) + +// ============================================================================ +// CONTRACT — fail deterministically if someone reverts the fix. +// ============================================================================ + +describe("v2: contract assertions on source", () => { + test("FIX: the registry family is handled, not just the session-scoped one", () => { + expect(SOURCE).toMatch(/case "shell\.created"/) + expect(SOURCE).toMatch(/case "shell\.exited"/) + // The session-scoped family is also correct and must be kept. + expect(SOURCE).toMatch(/case "session\.shell\.started"/) + expect(SOURCE).toMatch(/case "session\.shell\.ended"/) + }) + + test("FIX: the session is read from the nested path the runtime actually uses", () => { + expect(SOURCE).toMatch(/case "shell\.created"[\s\S]{0,600}metadata\?\.sessionID/) + }) + + test("FIX: both injection paths are gated, not just the stall watchdog", () => { + // The proactive nudge fires on session.idle, long before the watchdog + // would look, so gating only the watchdog would not prevent the nag. + const gate = /openShellCount\(sid\)[^\n]*\n\s*if \(busyShells > 0\)/ + const all = SOURCE.match(new RegExp(gate, "g")) ?? [] + expect(all.length).toBe(2) + // ...one in the targeted-nudge guard chain, one in the stall watchdog. + const targeted = SOURCE.indexOf("async function targetedRecovery") + const watchdog = SOURCE.indexOf("async function checkActiveSessions") + expect(targeted).toBeGreaterThan(-1) + expect(watchdog).toBeGreaterThan(targeted) + const targetedBody = SOURCE.slice(targeted, watchdog) + expect(targetedBody).toMatch(/openShellCount\(sid\)/) + }) + + test("FIX: a dropped exit event cannot suppress recovery forever", () => { + expect(SOURCE).toMatch(/SHELL_OPEN_MAX_MS/) + expect(SOURCE).toMatch(/now - s\.startedAt > SHELL_OPEN_MAX_MS/) + }) + + test("FIX: cleanup cannot delete the watch state of a session running a shell", () => { + const start = SOURCE.indexOf("function cleanupIdleSessions") + expect(start).toBeGreaterThan(-1) + const end = SOURCE.indexOf("function ", start + 10) + const body = SOURCE.slice(start, end > start ? end : start + 2000) + expect(body).toMatch(/openShellCount\(sid\) > 0\) continue/) + }) + + test("FIX: every sessions.delete drops that session's shell entries", () => { + const deletes = [...SOURCE.matchAll(/^\s*sessions\.delete\(sid\)$/gm)] + expect(deletes.length).toBeGreaterThan(0) + for (const d of deletes) { + const before = SOURCE.slice(Math.max(0, d.index! - 120), d.index!) + expect(before).toMatch(/forgetShells\(sid\)/) + } + }) +}) diff --git a/src/v2/index.options.test.ts b/src/v2/index.options.test.ts new file mode 100644 index 0000000..280f7fd --- /dev/null +++ b/src/v2/index.options.test.ts @@ -0,0 +1,413 @@ +import { describe, test, expect } from "bun:test" +import { readFileSync } from "node:fs" +import { join } from "node:path" +import plugin from "./index" + +const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_opts" + +const textPart = (t: string) => ({ type: "text", text: t }) +/** Older than every activeUserWindowMs we pass, so guard (b) cannot mask a failure. */ +const oldUserTurn = () => ({ + type: "user", + id: "msg_u0", + time: { created: Date.now() - 60 * 60_000 }, + content: [textPart("go ahead")], +}) +const assistantTurn = (text: string) => ({ + type: "assistant", + id: "msg_a1", + time: { created: Date.now() - 60_000 }, + content: [textPart(text)], +}) + +/** Trips `ready-to-continue` but not `isUserHandoff` (no "?", no "should I"). */ +const READY_TEXT = "Ready to continue with task" +/** + * Ends in ":" so `containsActionIntent` is true. Must not trip the + * ready-to-continue patterns and must not read as a user hand-off. + */ +const INTENT_TEXT = "I will run the migration now:" + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +type Harness = { + injected: Array<{ kind: string; text?: string }> + interrupts: string[] + logs: string[] +} + +/** + * Replay `events` and report what the plugin did. + * `sendFails` makes every delivery attempt throw, which is how the + * "a rejected send must not burn the retry budget" case is exercised. + */ +async function replay( + events: any[], + opts: Record = {}, + harnessOpts: { sendFails?: boolean; text?: string } = {}, +): Promise { + const injected: Harness["injected"] = [] + const interrupts: string[] = [] + const logs: string[] = [] + const stream = makeEventStream() + const text = harnessOpts.text ?? READY_TEXT + + const ctx: any = { + event: stream, + // The plugin reads its config from `ctx.options`; the second argument to + // `setup()` is ignored. Passing options the other way makes every + // negative assertion below pass for the wrong reason. + options: opts, + app: { + log: (_level: string, line: string) => { + logs.push(String(line)) + }, + }, + session: { + context: async () => [oldUserTurn(), assistantTurn(text)], + // Empty: no other session is active, so the `lastWasTaskTool` branch + // (which needs `others.length > 0`) cannot mask a failure below. + active: async () => ({}), + interrupt: async (a: any) => { + interrupts.push(a?.sessionID) + return {} + }, + synthetic: async (a: any) => { + if (harnessOpts.sendFails) throw new Error("synthetic rejected") + injected.push({ kind: "synthetic", text: a?.text }) + return {} + }, + prompt: async (a: any) => { + if (harnessOpts.sendFails) throw new Error("prompt rejected") + injected.push({ kind: "prompt", text: a?.text }) + return {} + }, + }, + } + + const cleanup = await (plugin as any).setup(ctx) + for (const e of events) { + stream.push(e) + await wait(10) + } + await wait(600) // handleEvent is sync; its work is async + ;(cleanup as (() => void) | undefined)?.() + return { injected, interrupts, logs } +} + +/** A turn that streams `text` and then goes idle. */ +function turnEvents(text = READY_TEXT) { + return [ + ev("session.execution.started"), + ev("session.text.delta", { messageID: "msg_a1", delta: text }), + ev("session.text.ended", { messageID: "msg_a1" }), + ev("session.idle"), + ] +} + +/** A turn that starts, streams nothing, and never idles: a stall candidate. */ +const busyStallEvents = [ev("session.execution.started")] + +/** Timings small enough that the watchdog fires inside the harness wait. */ +const FAST = { chunkTimeoutMs: 50, gracePeriodMs: 0, checkIntervalMs: 20, warmupMs: 0, baseBackoffMs: 1, maxBackoffMs: 2 } + +// ============================================================================ +// busyStallStrategy +// ============================================================================ + +describe("v2: busyStallStrategy", () => { + test("CONTROL: default 'continue' -> a stalled busy session gets a continue", async () => { + const { injected } = await replay(busyStallEvents, { ...FAST, loopMaxContinues: 99, injectIntervalMs: 0 }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("'off' -> the stall is ignored and nothing is injected", async () => { + const { injected, interrupts } = await replay(busyStallEvents, { + ...FAST, + busyStallStrategy: "off", + loopMaxContinues: 99, + injectIntervalMs: 0, + }) + expect(injected).toEqual([]) + expect(interrupts).toEqual([]) + }) + + test("'abort' -> the step is interrupted before the continue is sent", async () => { + const { interrupts } = await replay(busyStallEvents, { + ...FAST, + busyStallStrategy: "abort", + loopMaxContinues: 99, + injectIntervalMs: 0, + }) + expect(interrupts).toEqual([SID]) + }) + + test("'continue' (explicit) does not interrupt", async () => { + const { interrupts } = await replay(busyStallEvents, { + ...FAST, + busyStallStrategy: "continue", + loopMaxContinues: 99, + injectIntervalMs: 0, + }) + expect(interrupts).toEqual([]) + }) +}) + +// ============================================================================ +// warmupMs +// ============================================================================ + +describe("v2: warmupMs", () => { + test("CONTROL: warmup 0 -> the stall is acted on", async () => { + const { injected } = await replay(busyStallEvents, { ...FAST, loopMaxContinues: 99, injectIntervalMs: 0 }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("a session still inside its warmup window is not treated as stalled", async () => { + const { injected } = await replay(busyStallEvents, { + ...FAST, + warmupMs: 10 * 60_000, + loopMaxContinues: 99, + injectIntervalMs: 0, + }) + expect(injected).toEqual([]) + }) +}) + +// ============================================================================ +// maxRecoveryRetries alias +// ============================================================================ + +describe("v2: maxRecoveryRetries is v1's name for maxRetries", () => { + test("CONTROL: maxRetries 3 -> recovery fires", async () => { + const { injected } = await replay(busyStallEvents, { ...FAST, maxRetries: 3, loopMaxContinues: 99, injectIntervalMs: 0 }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("maxRetries 0 -> the session gives up instead of recovering", async () => { + const { injected } = await replay(busyStallEvents, { ...FAST, maxRetries: 0, loopMaxContinues: 99, injectIntervalMs: 0 }) + expect(injected).toEqual([]) + }) + + // If the alias were ignored, maxRetries would fall back to 3 and this would inject. + test("maxRecoveryRetries 0 is honoured as the same knob", async () => { + const { injected } = await replay(busyStallEvents, { + ...FAST, + maxRecoveryRetries: 0, + loopMaxContinues: 99, + injectIntervalMs: 0, + }) + expect(injected).toEqual([]) + }) +}) + +// ============================================================================ +// resumeOnActionIntent +// ============================================================================ + +describe("v2: resumeOnActionIntent", () => { + test("CONTROL: default true -> an action-intent turn is nudged", async () => { + const { injected } = await replay(turnEvents(INTENT_TEXT), { injectIntervalMs: 0 }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("false -> an action-intent turn is left alone", async () => { + const { injected } = await replay(turnEvents(INTENT_TEXT), { + resumeOnActionIntent: false, + injectIntervalMs: 0, + }) + expect(injected).toEqual([]) + }) +}) + +// ============================================================================ +// Unrecognised / accepted-but-inert option reporting +// ============================================================================ + +describe("v2: option reporting at startup", () => { + test("an unknown option key produces exactly one warning naming it", async () => { + const { logs } = await replay([], { chunkTimeoutMs: 5000, definitelyNotAnOption: true }) + const warnings = logs.filter((l) => l.includes("unrecognised")) + expect(warnings).toHaveLength(1) + expect(warnings[0]).toContain("definitelyNotAnOption") + }) + + test("known v1 options do not warn, even the ones v2 does not act on", async () => { + const { logs } = await replay([], { + chunkTimeoutMs: 5000, + maxRecoveryRetries: 2, + subagentWaitMs: 15_000, + discoveryDelayMs: 5_000, + contextSaturationThreshold: 0.85, + silentDeadStreamMinTokens: 200, + subagentNativeCompactionEnabled: false, + toolTextCheckDelayMs: 3_000, + thinkingToolRecoveryPrompt: "x", + doneWithoutWorkPrompt: "y", + }) + expect(logs.filter((l) => l.includes("unrecognised"))).toEqual([]) + }) + + test("the startup line names the accepted-but-inert options in use", async () => { + const { logs } = await replay([], { chunkTimeoutMs: 5000, subagentWaitMs: 15_000, discoveryDelayMs: 5_000 }) + const ready = logs.filter((l) => l.includes("ready (opencode v2)")) + expect(ready).toHaveLength(1) + expect(ready[0]).toContain("accepted-but-inert=") + expect(ready[0]).toContain("subagentWaitMs") + expect(ready[0]).toContain("discoveryDelayMs") + }) + + test("the startup line carries no accepted-but-inert list when none are set", async () => { + const { logs } = await replay([], { chunkTimeoutMs: 5000 }) + const ready = logs.filter((l) => l.includes("ready (opencode v2)")) + expect(ready).toHaveLength(1) + expect(ready[0]).not.toContain("accepted-but-inert=") + }) +}) + +// ============================================================================ +// e1b8374: a rejected send must not burn the retry budget +// ============================================================================ + +describe("v2: a nudge that was never delivered does not consume a retry", () => { + // Two idles inside one turn: no `execution.started` between them, so + // markBusy does not reset the per-turn budget and the second attempt is + // genuinely attempt #2 if the failed send was counted. + const twoIdles = [ + ev("session.execution.started"), + ev("session.text.delta", { messageID: "msg_a1", delta: READY_TEXT }), + ev("session.text.ended", { messageID: "msg_a1" }), + ev("session.idle"), + ev("session.idle"), + ] + + test("CONTROL: both sends succeed -> each nudge is 1/3, because markBusy resets the per-turn budget", async () => { + const { logs } = await replay(twoIdles, { maxRetries: 3, loopMaxContinues: 99, injectIntervalMs: 0 }) + const nudges = logs.filter((l) => l.includes("ready-to-continue detected")) + expect(nudges.length).toBeGreaterThanOrEqual(2) + // This is what makes the test below meaningful: on the success path the + // counter is reset by markBusy, so the number the user sees does not + // discriminate. On the failure path it does. + for (const n of nudges) expect(n).toContain("(1/3)") + }) + + test("both sends are rejected -> the budget is untouched, so it is still 1/3", async () => { + const { logs, injected } = await replay( + twoIdles, + { maxRetries: 3, loopMaxContinues: 99, injectIntervalMs: 0 }, + { sendFails: true }, + ) + expect(injected).toEqual([]) + const nudges = logs.filter((l) => l.includes("ready-to-continue detected")) + expect(nudges.length).toBeGreaterThanOrEqual(2) + // Nothing was delivered, so nothing was spent. The pre-e1b8374 ordering + // incremented before the send and reported 1/3 then 2/3 here. + for (const n of nudges) expect(n).toContain("(1/3)") + expect(nudges.some((l) => l.includes("(2/3)"))).toBe(false) + }) + + test("the budget write is inside the success branch, not before the send", () => { + // Guards the exact reordering e1b8374 makes, so a well-meaning "tidy up" + // that moves the increment back above injectOnce fails here. + expect(SOURCE).toMatch(/const attemptNum = w\[budgetKey\] \+ 1[\s\S]{0,400}w\[budgetKey\] = attemptNum/) + const body = SOURCE.match(/async function targetedRecovery[\s\S]*?\n\t\t\}/)?.[0] ?? "" + const incr = body.indexOf("w[budgetKey]") + const send = body.indexOf("injectOnce(") + expect(incr).toBeGreaterThan(-1) + expect(send).toBeGreaterThan(-1) + expect(incr).toBeLessThan(send) + expect(body.slice(send)).toContain("w[budgetKey] = attemptNum") + }) +}) + +// ============================================================================ +// CONTRACT — fail deterministically if someone reverts a ported option. +// ============================================================================ + +describe("v2: contract assertions on source", () => { + test("every v1 option name is in the recognised set", () => { + const block = SOURCE.match(/const RECOGNISED_OPTIONS = new Set\(\[([\s\S]*?)\]\)/) + expect(block).not.toBeNull() + const recognised = new Set([...(block![1].match(/"([a-zA-Z][a-zA-Z0-9]*)"/g) ?? [])].map((s) => s.replace(/"/g, ""))) + // Sampled across every category: prompts, timing, pattern lists, the + // v1-only alias, and the three that v2 recognises but does not act on. + for (const key of [ + "continuePrompt", + "actionIntentPrompt", + "toolTextRecoveryPrompt", + "doneWithoutDetailsPrompt", + "doneWithoutWorkPrompt", + "thinkingToolRecoveryPrompt", + "chunkTimeoutMs", + "warmupMs", + "minActivityGapMs", + "toolTextCheckDelayMs", + "subagentWaitMs", + "discoveryDelayMs", + "maxRecoveryRetries", + "busyStallStrategy", + "resumeOnActionIntent", + "doneClaimPatterns", + "readyToContinuePatterns", + "streamingFailureErrorNames", + "streamingFailureMessagePatterns", + "contextSaturationThreshold", + "subagentNativeCompactionEnabled", + "silentDeadStreamMinTokens", + "injectIntervalMs", + ]) { + expect(recognised.has(key)).toBe(true) + } + }) + + test("the declared options type and the recognised set cannot drift apart", () => { + const declared = new Set( + [...(SOURCE.match(/export interface AutoResumeOptions \{([\s\S]*?)\n\}/)?.[1] ?? "").matchAll( + /^\t([a-zA-Z][a-zA-Z0-9]*)\?:/gm, + )].map((m) => m[1]), + ) + const block = SOURCE.match(/const RECOGNISED_OPTIONS = new Set\(\[([\s\S]*?)\]\)/) + const recognised = new Set([...(block![1].match(/"([a-zA-Z][a-zA-Z0-9]*)"/g) ?? [])].map((s) => s.replace(/"/g, ""))) + expect([...declared].sort()).toEqual([...recognised].sort()) + }) + + test("the v2 active-user window matches v1 (5 min), not the 15 min the port shipped with", () => { + expect(SOURCE).toMatch(/DEFAULT_ACTIVE_USER_WINDOW_MS = 5 \* 60_000/) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index 6b127b4..0e4d9ef 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -62,6 +62,14 @@ interface AutoResumePluginInput { event: { subscribe: (opts?: { signal?: AbortSignal }) => AsyncIterable } /** Server-side session/messaging operations. */ session: Record any> + /** + * Opencode API client, when the host provides one. Declared on the real + * `@opencode-ai/plugin` `PluginInput` (`client: ReturnType`); typed here structurally so the file still + * type-checks standalone. Optional: a host may not supply it, and every + * use must degrade gracefully when it is absent. + */ + client?: { session: { get: (opts: { path: { id: string } }) => Promise<{ data?: { parentID?: string } }> } } /** Application logger, when the host provides one. */ app?: { log?: (level: string, message: string) => unknown } } @@ -124,6 +132,10 @@ interface SessionWatch { /** Timestamp of the latest `permission.asked` (stale-guard for the flag). */ permissionPendingAt: number | null lastWasTaskTool: boolean + /** Cached verdict from `isSubAgentSession()`: true when the server reports a + * `parentID` for this session, i.e. it is a child and must never be injected + * into or interrupted. `undefined` = not yet resolved. */ + isSubAgent?: boolean idleSince: number | null /** Set while the session is mid native compaction — recovery must never interrupt it. */ compacting: boolean @@ -155,13 +167,55 @@ export interface AutoResumeOptions { actionIntentPrompt?: string debug?: boolean activeUserWindowMs?: number + + // ---- Ported from v1 (Mte90/opencode-auto-resume#33) ---- + /** v1 name for `maxRetries`. Still accepted so v1 configs port unchanged. */ + maxRecoveryRetries?: number + /** How a stall is recovered: "continue" (default), "abort", or "off". */ + busyStallStrategy?: "continue" | "abort" | "off" + /** Fraction of the usable context window that counts as saturated. */ + contextSaturationThreshold?: number + /** Trigger native compaction on saturation for subagent sessions. */ + subagentNativeCompactionEnabled?: boolean + /** Resume on action-intent detection (default true). */ + resumeOnActionIntent?: boolean + /** Delay before the first session discovery sweep. */ + discoveryDelayMs?: number + /** Quiet period after startup before stall recovery arms. */ + warmupMs?: number + /** Minimum gap between recorded activity timestamps. */ + minActivityGapMs?: number + /** Grace before a raw-tool-call-as-text check fires. */ + toolTextCheckDelayMs?: number + /** Wait for a subagent to report before treating it as stalled. */ + subagentWaitMs?: number + /** Token floor for the silent-dead-stream heuristic. */ + silentDeadStreamMinTokens?: number + /** Error names that mark a streaming failure. */ + streamingFailureErrorNames?: string[] + /** Regex sources that mark a streaming failure message. */ + streamingFailureMessagePatterns?: string[] + /** Regex sources that mark a "task is done" claim. */ + doneClaimPatterns?: string[] + /** Regex sources that mark the model ready to continue. */ + readyToContinuePatterns?: string[] + /** Prompt sent when a done-claim arrives with no work to show. */ + doneWithoutDetailsPrompt?: string + /** Prompt sent when a tool call was emitted inside reasoning. */ + thinkingToolRecoveryPrompt?: string } // --------------------------------------------------------------------------- // Constants & defaults // --------------------------------------------------------------------------- -const DEFAULT_CHUNK_TIMEOUT_MS = 45_000 +// 45s -> 180s (2026-09-29): silence is the ONLY stall signal here, and 45s + +// 3s grace misfires on a single shared GPU where one subagent turn legitimately +// emits no events for over a minute. Widening the window is a mitigation, not +// the principled fix — that is positive hang detection (a tool call stuck in +// running/pending is a real hang; model generation is not), which changes +// recovery semantics and is still an open decision. +const DEFAULT_CHUNK_TIMEOUT_MS = 180_000 const DEFAULT_CHECK_INTERVAL_MS = 5_000 const DEFAULT_GRACE_PERIOD_MS = 3_000 const DEFAULT_MAX_RETRIES = 3 @@ -170,7 +224,84 @@ const DEFAULT_MAX_BACKOFF_MS = 8_000 const DEFAULT_LOOP_MAX_CONTINUES = 3 const DEFAULT_LOOP_WINDOW_MS = 10 * 60_000 const DEFAULT_DEBUG = false -const DEFAULT_ACTIVE_USER_WINDOW_MS = 15 * 60_000 +// Mirrors v1 (Mte90 e1b8374): an inbound user message this recent means the +// user is engaged (likely composing), so idle nudges stand down. Was 15min here. +const DEFAULT_ACTIVE_USER_WINDOW_MS = 5 * 60_000 + +// ---- Defaults ported from v1 (Mte90/opencode-auto-resume#33) ---- +const DEFAULT_DISCOVERY_DELAY_MS = 5_000 +const DEFAULT_WARMUP_MS = 15_000 +const DEFAULT_MIN_ACTIVITY_GAP_MS = 1_000 +const DEFAULT_TOOL_TEXT_CHECK_DELAY_MS = 3_000 +const DEFAULT_SUBAGENT_WAIT_MS = 15_000 +const DEFAULT_SILENT_DEAD_STREAM_MIN_TOKENS = 200 +const DEFAULT_CONTEXT_SATURATION_THRESHOLD = 0.85 +const DEFAULT_MAX_RECOVERY_RETRIES = 2 +// Referenced from FEATURE_GATED_OPTIONS docs; kept so the intended v1 default is +// recorded next to the gate that explains why it is not applied yet. + +const DEFAULT_STREAMING_FAILURE_ERROR_NAMES = [ + "ProviderError", + "APIError", + "StreamError", + "ConnectionError", + "TimeoutError", +] + +const DEFAULT_STREAMING_FAILURE_MESSAGE_PATTERNS = [ + "streaming response failed", + "stream.*fail", + "connection.*reset", + "connection.*closed", + "aborted due to timeout", +] + +const DEFAULT_DONE_CLAIM_PATTERNS = [ + "task\\s+done[.!]*", + "done[.!]*", + "all\\s+done[.!]*", + "finished[.!]*", + "complete[.!]*", + "task\\s+complete[.!]*", + "task\\s+completed[.!]*", + "all\\s+tasks?\\s+complete[.!]*", + "all\\s+tasks?\\s+completed[.!]*", + "(?:i['’]?m\\s+)?done\\s+with\\s+task", + "done\\s+with\\s+(?:the\\s+)?(?:task|work|implementation)", + "finished\\s+(?:the\\s+)?(?:task|work|implementation)", + "(?:all|everything)\\s+(?:is\\s+)?(?:complete|done|finished)", + "nothing\\s+(?:else\\s+)?(?:left|remaining|to do)", +] + +const DEFAULT_READY_TO_CONTINUE_PATTERNS = [ + "ready to continue with task", + "continuing with task", + "continue with task", + "proceeding with task", + "ready to proceed with task", + "will continue with task", + "moving on to task", +] + +const THINKING_TOOL_RECOVERY_PROMPT = + "I noticed you have a tool call generated in your thinking/reasoning. " + + "Please execute it using the proper tool calling mechanism instead of keeping it in reasoning." + +const DONE_WITHOUT_DETAILS_PROMPT = + "Your last response claimed the task is complete but contained no work description. This is not acceptable. " + + "You MUST respond now with a full, detailed report of everything you did: " + + "for each file you modified, state the full path and the exact changes; " + + "list every command you ran to verify and its result; state the final outcome. " + + "Do NOT reply with 'done', 'task completed', or any short acknowledgment — " + + "your ONLY acceptable response right now is this detailed report. Write it now." + +/** + * Upper bound on how long a `shell.created` → `shell.exited` pair is trusted + * to mean "this session is busy". A shell that never reports an exit (session + * torn down, event dropped) would otherwise suppress recovery forever, so + * entries are pruned past this. Set well above any realistic long build. + */ +const SHELL_OPEN_MAX_MS = 30 * 60_000 /** OOC (out-of-context) errors that `continue` can never clear — recovery is locked out on these. */ const OOC_ERROR_RE = /exceeds the available context size|context size \(\d+\)|too large to compact|too many tokens|prompt is too long/i @@ -342,16 +473,16 @@ function containsToolCallAsText(text: string): boolean { return false } -function containsReadyToContinuePattern(text: string): boolean { +function containsReadyToContinuePattern(text: string, patterns: RegExp[] = READY_TO_CONTINUE_PATTERNS): boolean { const lines = text.split("\n") const lastLines = lines.slice(-3).join("\n") - return READY_TO_CONTINUE_PATTERNS.some((pat) => pat.test(lastLines)) + return patterns.some((pat) => pat.test(lastLines)) } -function containsDoneClaimPattern(text: string): boolean { +function containsDoneClaimPattern(text: string, patterns: RegExp[] = DONE_CLAIM_PATTERNS): boolean { const lines = text.split("\n") const lastLines = lines.slice(-5).join("\n") - return DONE_CLAIM_PATTERNS.some((pat) => pat.test(lastLines)) + return patterns.some((pat) => pat.test(lastLines)) } /** Model ends with ":" announcing intent without executing. */ @@ -397,10 +528,10 @@ function backoffMs(attempt: number, base: number, max: number): number { return Math.min(base * Math.pow(2, attempt - 1), max) } -function isStreamingFailure(message: string): boolean { +function isStreamingFailure(message: string, patterns: string[] = STREAMING_FAILURE_MESSAGE_PATTERNS): boolean { const lower = message.toLowerCase() if (!lower) return false - return STREAMING_FAILURE_MESSAGE_PATTERNS.some((pattern) => { + return patterns.some((pattern) => { try { return new RegExp(pattern, "i").test(lower) } catch { @@ -409,6 +540,11 @@ function isStreamingFailure(message: string): boolean { }) } +/** True when an error `name` is one of the configured streaming-failure names. */ +function isStreamingFailureName(name: string, names: string[]): boolean { + return names.some((candidate) => candidate.toLowerCase() === name.toLowerCase()) +} + /** Repeating tool-call patterns: A-A-A or A-B-A-B-A-B etc. */ function detectPatternLoop(recentTools: string[]): boolean { if (recentTools.length < 6) return false @@ -458,7 +594,6 @@ function isTaskToolCall(ev: V2Event): boolean { const desc = ev.data?.input?.description ?? ev.data?.input?.subagent_type ?? ev.data?.input?.agent return typeof desc === "string" } - // --------------------------------------------------------------------------- // Plugin // --------------------------------------------------------------------------- @@ -472,7 +607,9 @@ export default define({ const chunkTimeoutMs = opts.chunkTimeoutMs ?? DEFAULT_CHUNK_TIMEOUT_MS const checkIntervalMs = opts.checkIntervalMs ?? DEFAULT_CHECK_INTERVAL_MS const gracePeriodMs = opts.gracePeriodMs ?? DEFAULT_GRACE_PERIOD_MS - const maxRetries = opts.maxRetries ?? DEFAULT_MAX_RETRIES + // `maxRecoveryRetries` is v1's name for the same knob. Prefer the v2 name, + // fall back to the v1 alias so an existing v1 config ports unchanged. + const maxRetries = opts.maxRetries ?? opts.maxRecoveryRetries ?? DEFAULT_MAX_RETRIES const baseBackoff = opts.baseBackoffMs ?? DEFAULT_BASE_BACKOFF_MS const maxBackoff = opts.maxBackoffMs ?? DEFAULT_MAX_BACKOFF_MS const loopMaxContinues = opts.loopMaxContinues ?? DEFAULT_LOOP_MAX_CONTINUES @@ -481,6 +618,119 @@ export default define({ const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS + // ---- Ported from v1 (Mte90/opencode-auto-resume#33) ---- + const rawBusyStallStrategy = opts.busyStallStrategy ?? "continue" + const busyStallStrategy: "continue" | "abort" | "off" = + rawBusyStallStrategy === "abort" || rawBusyStallStrategy === "off" ? rawBusyStallStrategy : "continue" + // Feature-gated on v2 (see FEATURE_GATED_OPTIONS): accepted for config + // compatibility, not yet acted on. Not bound to locals so they cannot be + // mistaken for live behaviour. + const resumeOnActionIntent = opts.resumeOnActionIntent !== false + const warmupMs = opts.warmupMs ?? DEFAULT_WARMUP_MS + const minActivityGapMs = opts.minActivityGapMs ?? DEFAULT_MIN_ACTIVITY_GAP_MS + const streamingFailureErrorNames = opts.streamingFailureErrorNames ?? DEFAULT_STREAMING_FAILURE_ERROR_NAMES + const streamingFailureMessagePatterns = opts.streamingFailureMessagePatterns ?? DEFAULT_STREAMING_FAILURE_MESSAGE_PATTERNS + const doneWithoutDetailsPrompt = opts.doneWithoutDetailsPrompt ?? DONE_WITHOUT_DETAILS_PROMPT + const thinkingToolRecoveryPrompt = opts.thinkingToolRecoveryPrompt ?? THINKING_TOOL_RECOVERY_PROMPT + + /** Compile a user-supplied regex-source list, skipping anything invalid. */ + function compilePatterns( + raw: string[] | undefined, + fallback: string[], + flags: string, + ): RegExp[] { + const sources = Array.isArray(raw) && raw.length > 0 ? raw : fallback + const out: RegExp[] = [] + for (const source of sources) { + try { + out.push(new RegExp(source, flags)) + } catch { + // An invalid pattern is dropped rather than failing startup. + } + } + return out + } + + const doneClaimPatterns = compilePatterns(opts.doneClaimPatterns, DEFAULT_DONE_CLAIM_PATTERNS, "im") + const readyToContinuePatterns = compilePatterns( + opts.readyToContinuePatterns, + DEFAULT_READY_TO_CONTINUE_PATTERNS, + "i", + ) + const streamingFailureMessageRegexes = compilePatterns( + streamingFailureMessagePatterns, + DEFAULT_STREAMING_FAILURE_MESSAGE_PATTERNS, + "i", + ) + + // Options accepted but not yet acted on, because the v2 feature they tune + // is not ported. Listed (not warned) so an existing v1 config stays valid + // and the gap is documented rather than surprising. See + // docs/known-issues-v2.md. + const FEATURE_GATED_OPTIONS = [ + "contextSaturationThreshold", + "subagentNativeCompactionEnabled", + "silentDeadStreamMinTokens", + "subagentWaitMs", + "discoveryDelayMs", + "toolTextCheckDelayMs", + "thinkingToolRecoveryPrompt", + "doneWithoutWorkPrompt", + ] as const + + // Options this build understands. Anything else in the user's config is + // reported once at startup so a silent fallback is visible rather than + // looking like a bug. (Mte90/opencode-auto-resume#33) + const RECOGNISED_OPTIONS = new Set([ + "chunkTimeoutMs", + "checkIntervalMs", + "gracePeriodMs", + "maxRetries", + "maxRecoveryRetries", + "baseBackoffMs", + "maxBackoffMs", + "loopMaxContinues", + "loopWindowMs", + "debug", + "activeUserWindowMs", + "injectIntervalMs", + "continuePrompt", + "toolTextRecoveryPrompt", + "actionIntentPrompt", + "doneWithoutWorkPrompt", + "busyStallStrategy", + "contextSaturationThreshold", + "subagentNativeCompactionEnabled", + "resumeOnActionIntent", + "discoveryDelayMs", + "warmupMs", + "minActivityGapMs", + "toolTextCheckDelayMs", + "subagentWaitMs", + "silentDeadStreamMinTokens", + "streamingFailureErrorNames", + "streamingFailureMessagePatterns", + "doneClaimPatterns", + "readyToContinuePatterns", + "doneWithoutDetailsPrompt", + "thinkingToolRecoveryPrompt", + "doneWithoutWorkPrompt", + ]) + const unknownOptions = Object.keys(opts).filter((key) => !RECOGNISED_OPTIONS.has(key)) + if (unknownOptions.length > 0) { + log( + "warn", + `ignoring unrecognised option(s): ${unknownOptions.join(", ")} — ` + + `this v2 build does not read them. See docs/known-issues-v2.md.`, + ) + } + // Feature-gated options are valid but inert on v2. Reported in the startup + // line rather than as a warning — the user cannot act on it, so a per-start + // warn would be pure noise. See docs/known-issues-v2.md. + const gatedInUse = Object.keys(opts).filter((key) => + (FEATURE_GATED_OPTIONS as readonly string[]).includes(key), + ) + const dbg = (...args: unknown[]) => { if (debug) console.log("[auto-resume:debug]", ...args) } @@ -508,6 +758,38 @@ export default define({ const sessions = new Map() + /** + * Open shells from the process-registry event family, keyed by shell id + * → `{ sessionID, startedAt }`. Populated by `shell.created`, cleared by + * `shell.exited`. A session with an entry here is running a command and + * must not be treated as a stalled parent. + * + * `shell.created` is the only shell event that carries its session, and it + * nests it at `data.info.metadata.sessionID`; the exit events carry only + * the shell id, so the reverse lookup has to be recorded here. + */ + const openShells = new Map() + + /** Number of shells currently open for `sid`, pruning stale entries. */ + function openShellCount(sid: string, now = Date.now()): number { + let n = 0 + for (const [id, s] of openShells) { + if (now - s.startedAt > SHELL_OPEN_MAX_MS) { + openShells.delete(id) + continue + } + if (s.sessionID === sid) n++ + } + return n + } + + /** Drop every shell entry owned by a session that no longer exists. */ + function forgetShells(sid: string): void { + for (const [id, s] of openShells) { + if (s.sessionID === sid) openShells.delete(id) + } + } + function ensureWatch(sid: string): SessionWatch { let w = sessions.get(sid) if (!w) { @@ -551,7 +833,12 @@ export default define({ function touch(sid: string) { const w = ensureWatch(sid) - w.lastActivityAt = Date.now() + // `minActivityGapMs` debounces activity: without it, a burst of events + // inside one tool call keeps resetting the stall clock and a genuinely + // wedged step never trips the watchdog. + const now = Date.now() + if (now - w.lastActivityAt < minActivityGapMs) return + w.lastActivityAt = now } function markBusy(sid: string) { @@ -660,6 +947,40 @@ export default define({ return out } + /** + * Is this session itself a subagent? Definitive test: ask the server for + * our own record and look at `parentID`. + * + * This plugin is parent-scoped — it exists to recover sessions a human is + * waiting on. A child is not that: it never reads the parent's AGENTS.md, + * so a prompt-level "ignore injected continues" rule in the child agent + * loses to an injection that arrives as a real task turn. Observed in the + * wild: a worker answered an invisible prompt mid-task and never returned. + * + * Deliberately independent of the `lastWasTaskTool` heuristic used for + * parents in `checkActiveSessions` — that one asks "did this session just + * dispatch a child?", this one asks "is this session a child?". A silent + * child gets no protection from the parent-side heuristic, which is + * exactly the gap this closes. + * + * Cached on the watch record. `ctx.client` is optional and any failure + * degrades to `false` — i.e. precisely the pre-guard behavior — so a host + * without a client can neither break recovery nor fail to boot. + */ + async function isSubAgentSession(sid: string): Promise { + const w = ensureWatch(sid) + if (typeof w.isSubAgent === "boolean") return w.isSubAgent + let sub = false + try { + const res = await ctx.client?.session.get({ path: { id: sid } }) + sub = !!res?.data?.parentID + } catch { + sub = false + } + w.isSubAgent = sub + return sub + } + /** * Single choke point for every recovery injection. Guarantees at most one * synthetic per `injectIntervalMs` and refuses to talk to a live session, @@ -684,6 +1005,12 @@ export default define({ allowDuringSelfAbort = false, ): Promise { const w = ensureWatch(sid) + // A subagent is not ours to recover. Checked first, before every other + // guard, so no code path below can reach a child. + if (await isSubAgentSession(sid)) { + dbg(`${short(sid)} subagent session — injection refused (parent owns recovery)`) + return false + } if (!allowDuringSelfAbort && selfAbortActive(w)) { dbg(`${short(sid)} injection refused — inside our own abort window`) return false @@ -726,8 +1053,17 @@ export default define({ if (!toDelete.includes(entries[i].sid)) toDelete.push(entries[i].sid) } } - for (const sid of toDelete) sessions.delete(sid) - if (toDelete.length > 0) dbg(`cleaned ${toDelete.length} idle sessions, total=${sessions.size}`) + let cleaned = 0 + for (const sid of toDelete) { + // A session waiting on a long command emits no activity events and + // therefore looks idle. Don't drop its watch state out from under + // a shell that is still running. + if (openShellCount(sid) > 0) continue + forgetShells(sid) + sessions.delete(sid) + cleaned++ + } + if (cleaned > 0) dbg(`cleaned ${cleaned} idle sessions, total=${sessions.size}`) } /** @@ -790,6 +1126,13 @@ export default define({ } async function tryAbortAndResume(sid: string, w: SessionWatch): Promise { + // Guarded separately because this path calls `interrupt()` *before* + // delegating to injectOnce — guarding only the injection would still + // let us interrupt a running child. + if (await isSubAgentSession(sid)) { + dbg(`${short(sid)} subagent session — abort+resume refused`) + return false + } if (w.aborting || selfAbortActive(w)) return false if (w.oocLocked) { dbg(`${short(sid)} oocLocked — refusing abort+resume escalation`) @@ -927,6 +1270,15 @@ export default define({ dbg(`${short(sid)} mid-compaction — skipping ${kind} nudge`) return } + // A session with a shell still running is working, not idle. A parent + // parked on a background job goes idle the moment the tool call returns + // — long before the process finishes — so without this it collects a + // nudge on the spot and the stall watchdog never even gets a look. + const busyShells = openShellCount(sid) + if (busyShells > 0) { + dbg(`${short(sid)} ${busyShells} shell(s) still running — skipping ${kind} nudge`) + return + } if (selfAbortActive(w)) { dbg(`${short(sid)} self-abort in flight — skipping ${kind} nudge`) return @@ -942,16 +1294,23 @@ export default define({ dbg(`${short(sid)} ${kind} budget exhausted`) return } - w[budgetKey]++ - log("info", `${short(sid)} ${kind} detected — sending targeted prompt (${w[budgetKey]}/${maxRetries})`) + // Port of v1 e1b8374 ("Fix todoNudgeAttempts burning retries on failed + // sends"): only count an attempt once the prompt actually landed. A + // rejected send costs nothing, so a transient failure no longer eats a + // retry and silences the nudge for the rest of the session. + const attemptNum = w[budgetKey] + 1 + log("info", `${short(sid)} ${kind} detected — sending targeted prompt (${attemptNum}/${maxRetries})`) w.recovering = true // injectOnce debounces: this nudge counts once, not alongside a // concurrent stall-watchdog injection. const ok = await injectOnce(sid, prompt, "recovering: " + kind) w.recovering = false if (ok) { + w[budgetKey] = attemptNum touch(sid) markBusy(sid) + } else { + dbg(`${short(sid)} ${kind} nudge not delivered — attempt ${attemptNum} not counted`) } } @@ -1054,19 +1413,19 @@ async function inspectOnIdle(sid: string) { await targetedRecovery(sid, "tool-call-as-text", opts.toolTextRecoveryPrompt ?? TOOL_TEXT_RECOVERY_PROMPT, "toolTextAttempts") return } - if (containsReadyToContinuePattern(text)) { + if (containsReadyToContinuePattern(text, readyToContinuePatterns)) { await targetedRecovery(sid, "ready-to-continue", opts.continuePrompt ?? CONTINUE_PROMPT, "intentNudgeAttempts") return } - if (containsActionIntent(text)) { + if (resumeOnActionIntent && containsActionIntent(text)) { await targetedRecovery(sid, "action-intent", opts.actionIntentPrompt ?? opts.continuePrompt ?? CONTINUE_PROMPT, "intentNudgeAttempts") return } - if (containsDoneClaimPattern(text) && w.doneClaimAttempts < 1) { + if (containsDoneClaimPattern(text, doneClaimPatterns) && w.doneClaimAttempts < 1) { // Single verification nudge for suspiciously terse completions const trimmed = text.trim() if (trimmed.length < 400) { - await targetedRecovery(sid, "done-claim-no-details", opts.doneWithoutWorkPrompt ?? DONE_WITHOUT_WORK_PROMPT, "doneClaimAttempts") + await targetedRecovery(sid, "done-claim-no-details", doneWithoutDetailsPrompt, "doneClaimAttempts") } } } @@ -1111,6 +1470,10 @@ async function inspectOnIdle(sid: string) { for (const [sid, w] of sessions) { if (w.status !== "busy" || w.userCancelled) continue + // Warmup: a session that only just went busy has not had a chance to + // emit anything. Without this a freshly-started turn can be declared + // stalled while it is still queueing its first model call. + if (now - w.createdAt < warmupMs) continue // Stale compaction flag: a compaction that never reports // ended/failed within the TTL is wedged — clear the guard so // recovery can eventually intervene. @@ -1126,6 +1489,10 @@ async function inspectOnIdle(sid: string) { } const silence = now - w.lastActivityAt if (silence < chunkTimeoutMs + gracePeriodMs) continue + if (busyStallStrategy === "off") { + dbg(`stall on ${short(sid)} ignored (busyStallStrategy=off)`) + continue + } // Same stale-guard as the compaction flag: an unanswered `permission.asked` // must not stand this session down forever. clearStalePermissionFlag(sid, w) @@ -1137,6 +1504,16 @@ async function inspectOnIdle(sid: string) { dbg(`${short(sid)} silent but inside our own abort window — skipping`) continue } + // A session with a shell still running is working, not stalled. This + // covers the parked-parent case the task-tool heuristic below cannot + // see: a backgrounded `shell` spawns no child session, so + // `lastWasTaskTool` stays false and `others.length` never gets a + // chance to say "this parent is waiting on something". + const busyShells = openShellCount(sid) + if (busyShells > 0) { + dbg(`${short(sid)} silent with ${busyShells} shell(s) still running — waiting`) + continue + } // If another session is actively running and this one went silent // right after dispatching a task tool, treat it as a parent wait. if (w.lastWasTaskTool) { @@ -1146,7 +1523,16 @@ async function inspectOnIdle(sid: string) { continue } } - await recover(sid, `no activity for ${Math.ceil(silence / 1000)}s`) + // busyStallStrategy picks how a stall is answered. "abort" interrupts + // the wedged step before continuing, because a stream that went + // silent mid-generation will not pick the prompt up on its own. + // Mirrors the v1 branch at src/index.ts. + if (busyStallStrategy === "abort" && w.resumeAttempts < maxRetries) { + log("info", `${short(sid)} stall (busyStallStrategy=abort): aborting before continue`) + await tryAbortAndResume(sid, w) + } else { + await recover(sid, `no activity for ${Math.ceil(silence / 1000)}s`) + } } cleanupIdleSessions() } @@ -1221,6 +1607,7 @@ async function inspectOnIdle(sid: string) { case "session.deleted": { const sid = sidOf(ev) if (!sid) return + forgetShells(sid) sessions.delete(sid) return } @@ -1240,6 +1627,7 @@ async function inspectOnIdle(sid: string) { case "session.revert.committed": { const sid = sidOf(ev) if (!sid) return + forgetShells(sid) sessions.delete(sid) return } @@ -1249,6 +1637,7 @@ async function inspectOnIdle(sid: string) { case "session.reverted": { const sid = sidOf(ev) if (!sid) return + forgetShells(sid) sessions.delete(sid) return } @@ -1359,6 +1748,38 @@ async function inspectOnIdle(sid: string) { touch(sid) return } + // --- shell process registry --- + // `session.shell.*` above is the session-scoped family. The runtime + // also emits a process-registry family, and which one fires depends + // on the tool path (the bash/Code Mode shell emits only these). We + // need the registry pair because it is the one that reports a + // backgrounded job's real completion: `session.tool.success` fires + // as soon as the *tool call* returns, which for `background: true` + // is a few hundred ms — long before the process finishes. + // + // `shell.deleted` is deliberately not used: it carries a different + // id family than `created`/`exited` (verified against the running + // server), so it cannot close an entry. `shell.exited` can. + case "shell.created": { + const info = ev.data?.info + if (!info || typeof info !== "object") return + const rec = info as { id?: unknown; metadata?: { sessionID?: unknown } } + const id = rec.id + const sid = rec.metadata?.sessionID + if (typeof id !== "string" || typeof sid !== "string") return + openShells.set(id, { sessionID: sid, startedAt: Date.now() }) + touch(sid) + return + } + case "shell.exited": { + const id = ev.data?.id + if (typeof id !== "string") return + const sid = openShells.get(id)?.sessionID + if (sid === undefined) return + openShells.delete(id) + touch(sid) + return + } case "session.tool.success": { const sid = sidOf(ev) if (!sid) return @@ -1405,7 +1826,7 @@ async function inspectOnIdle(sid: string) { const errMsg = String(ev.data?.error?.message ?? "") maybeLockOoc(sid, errMsg) log("info", `${short(sid)} provider retry #${ev.data?.attempt ?? "?"}: ${errType || errMsg}`) - if (!isStreamingFailure(errMsg) && errType !== "retryable") return + if (!isStreamingFailure(errMsg, streamingFailureMessagePatterns) && !isStreamingFailureName(errType, streamingFailureErrorNames) && errType !== "retryable") return // Let the provider retries play out first; only intervene if it stays quiet if (nowSilenceTooLong(w)) { w.pendingRecoveryArmed = true @@ -1440,7 +1861,7 @@ async function inspectOnIdle(sid: string) { markIdle(sid) w.pendingRecoveryArmed = true // our delayed recovery must survive this idle transition log("warn", `${short(sid)} ${ev.type}: ${errType || "error"} ${errMsg.slice(0, 160)}`) - void recover(sid, `${ev.type}${isStreamingFailure(errMsg) ? " (streaming)" : ""}`) + void recover(sid, `${ev.type}${isStreamingFailure(errMsg, streamingFailureMessagePatterns) ? " (streaming)" : ""}`) return } @@ -1479,7 +1900,8 @@ async function inspectOnIdle(sid: string) { log( "info", - `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s`, + `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s warmup=${warmupMs}ms stall=${busyStallStrategy}` + + (gatedInUse.length > 0 ? ` accepted-but-inert=${gatedInUse.join(",")}` : ""), ) // Cleanup: stop timers and the event pump; OpenCode awaits this on disable/reload/shutdown. From c871c2956f262d12064d32d1cd455655c368f83d Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 14:44:36 -0400 Subject: [PATCH 14/57] feat(v2): sweep sessions on startup, and give the plugin a log file at all MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two features ported from v1, and one defect fixed that made the plugin unobservable. Session discovery. v1 swept session.list() on an interval and trusted a `status` field on each row. v2 has something better — session.active() is the server's own record of what is running — so the sweep seeds a watch for every existing session and adopts the running ones as busy. Without it, a turn that was already mid-flight when the plugin attached produced no `session.execution.started` for us to see, so the stall watchdog had nothing to inspect and a wedged session stayed wedged. Both calls are read defensively: the v2 plugin `session` domain is a narrowed Pick that omits `list` and `active`, so each falls back to the raw client, and a host that supplies neither degrades to the event-derived busy set rather than failing. `discoveryDelayMs` is live again as a result, so it leaves the accepted-but-inert list (7 remain, down from 8). Logging. v1 logged through `ctx.client.app.log({ body: {...} })`, a real endpoint that reached the opencode log file. v2 removed it — `ctx.app` is `{name, version, channel}`, there is no `app` group in the protocol, and a hosted plugin's console.log is not captured by OpenChamber. The v2 build was therefore completely silent: the deployed one had zero log lines, and its only startup line predated the current work by days. A running watchdog and a dead one were indistinguishable, which is exactly what made a stall look like "nothing happened". This build appends to `~/.local/state/opencode-v2/auto-resume.log` (override with the new `logFile` option or `AUTO_RESUME_LOG_FILE`), capped at 2 MB, every write best-effort so an unwritable directory can never break the watchdog. Console output is kept alongside it. src/v2/index.discovery.test.ts covers both API shapes with a control per group, the `{data}` envelope, malformed rows, a throwing list(), and teardown leaking no timers. The option tests now read the log file rather than a stubbed `ctx.app.log`, so they exercise the real sink. 639 pass, 0 fail. Docs follow the code: known-issues-v2.md drops the stale discoveryDelayMs row and documents both features, and README's activeUserWindowMs row now says plainly that v1 on this branch is still 15 min while v2 is 5 min. --- README.md | 7 +- docs/known-issues-v2.md | 54 +++++++- src/v2/index.discovery.test.ts | 245 +++++++++++++++++++++++++++++++++ src/v2/index.options.test.ts | 40 ++++-- src/v2/index.ts | 202 +++++++++++++++++++++++++-- 5 files changed, 520 insertions(+), 28 deletions(-) create mode 100644 src/v2/index.discovery.test.ts diff --git a/README.md b/README.md index a86127b..a82a74a 100644 --- a/README.md +++ b/README.md @@ -463,13 +463,14 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `silentDeadStreamMinTokens` | `200` | Min output tokens to treat a textless `finish:"unknown"` message as a dead stream | | `busyStallStrategy` | `"continue"` | Busy-stall response: `"continue"`, `"abort"` (abort-first), or `"off"` (disabled) | | `contextSaturationThreshold` | `0.85` | Ratio of used/usable context that routes a saturated parent to magic-context `ctx-wrapup` (only when magic-context is installed) | -| `activeUserWindowMs` | `900000` | Inbound-user-message recency window (15 min) during which idle nudges stand down (user likely composing) | +| `activeUserWindowMs` | `300000` on v2, `900000` on v1 | Inbound-user-message recency window during which idle nudges stand down (user likely composing). v1 is stale at 15 min on this branch; upstream `e1b8374` already moved it to 5 min and the v2 build matches that | | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | +| `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | Accepted but **not applied** on v2: `contextSaturationThreshold`, `subagentNativeCompactionEnabled`, `silentDeadStreamMinTokens`, `subagentWaitMs`, -`discoveryDelayMs`, `toolTextCheckDelayMs`, `thinkingToolRecoveryPrompt`, +`toolTextCheckDelayMs`, `thinkingToolRecoveryPrompt`, `doneWithoutWorkPrompt`. See [docs/known-issues-v2.md](docs/known-issues-v2.md) for why. Message patterns are matched case-insensitively. Error names use exact match. @@ -481,7 +482,7 @@ Message patterns are matched case-insensitively. Error names use exact match. | `ABORT_CONTINUE_DELAY_MS` | `2000` | Delay between abort and continue | | `MAX_IDLE_SESSIONS` | `50` | Idle session map cap before cleanup | | `IDLE_CLEANUP_MS` | `600000` | Idle session age before cleanup (10 min) | -| `SESSION_DISCOVERY_INTERVAL_MS` | `60000` | `session.list()` poll interval (60s) — v1 only; the v2 build has no discovery sweep | +| `SESSION_DISCOVERY_INTERVAL_MS` | `60000` | Session discovery sweep interval (60s). Same on both builds; v2 reads the busy set from `session.active()` rather than a `status` field per row | ## Verification diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 9f4bfe1..6236d08 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -16,7 +16,6 @@ config keeps loading unchanged. | `subagentNativeCompactionEnabled` | Same: without a context-limit read, saturation cannot be detected, so there is nothing to gate a compaction on. | | `silentDeadStreamMinTokens` | Same. The heuristic it tunes compares generated token count against a floor. | | `subagentWaitMs` | The v1 orphan-watch timer that this delays has no v2 counterpart; the v2 port decides parent-vs-stalled from its own event-derived busy set. | -| `discoveryDelayMs` | v2 has no discovery sweep. The port starts watching from the events it receives rather than by enumerating existing sessions on a delay. | | `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | | `thinkingToolRecoveryPrompt` | The thinking-contains-a-tool-call detector is part of the v1 idle-nudge pass and is not ported. | | `doneWithoutWorkPrompt` | Both v1 use sites for this prompt are gated on tracked todo state. v2 has no todo state, so the prompt has no trigger. | @@ -47,11 +46,60 @@ the fix is mte090's `e1b8374`, applied to the v2 code path. The stall path - `maxRetries` is the v2 name. `maxRecoveryRetries` is v1's name for the same knob and is still accepted as a fallback, so a v1 config ports unchanged. -- `activeUserWindowMs` defaults to `300000` (5 min) on both v1 and v2, matching +- `activeUserWindowMs` defaults to `300000` (5 min) in the v2 build, matching upstream `e1b8374`. The v2 port shipped with `900000`, which stood down for - three times longer than v1 after any user message. + three times longer than intended after any user message. The v1 file on this + branch still reads `900000` because it predates `e1b8374`; the PR does not + touch v1, and master already carries the fix. ## v2-only options - `injectIntervalMs` — minimum gap between recovery injections for one session. No v1 equivalent. +- `logFile` — where this build writes its log. See "Logging" below. + +## Logging: there is no server log sink in v2 + +v1 logged through `ctx.client.app.log({ body: { service, level, message } })`, +which landed in the opencode log file. v2 removed that endpoint: `ctx.app` is +`{ name, version, channel }` (`packages/plugin/src/app.ts`), there is no `app` +group under `packages/protocol/src/groups/`, and a hosted plugin's +`console.log` is not captured by the OpenChamber process. + +A build that only writes to the console is therefore **silent** — a running +watchdog and a dead one look identical, and there is no way to tell from +outside whether a stall was seen, skipped, or recovered. This v2 build appends +to a file instead: + +``` +~/.local/state/opencode-v2/auto-resume.log # default +``` + +Override with the `logFile` option or the `AUTO_RESUME_LOG_FILE` environment +variable; the option wins. The file is truncated once it passes 2 MB, and every +write is best-effort — an unwritable log directory never breaks the watchdog. + +Each line is `ISO-8601 LEVEL [auto-resume] message`. The startup line lists the +effective timings, so a build that loaded at all is provable from the file: + +``` +2026-10-01T18:02:28.412Z INFO [auto-resume] ready (opencode v2). timeout=180000ms interval=5000ms ... +``` + +## Session discovery + +v1 swept `session.list()` on an interval and trusted a `status` field per row. +v2 has something better: `session.active()` is the server's own record of what +is running, so "busy" is read rather than guessed. The sweep runs once at +`discoveryDelayMs` (default 5s) and then every 60s, and it: + +- seeds a watch for any session that exists, so cleanup and revert handling know + about sessions that predate the plugin load, and +- marks any session the server reports as running as busy, so a turn that was + already mid-flight when the plugin attached starts its stall clock now rather + than never. + +Both calls are read defensively — off the plugin `session` domain first, then +off the raw client — because v2 hands plugins a narrowed `Pick` that omits both. +A host that supplies neither degrades to the event-derived busy set rather than +failing. diff --git a/src/v2/index.discovery.test.ts b/src/v2/index.discovery.test.ts new file mode 100644 index 0000000..b129d75 --- /dev/null +++ b/src/v2/index.discovery.test.ts @@ -0,0 +1,245 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) + +/** Scratch log path per case, so parallel tests cannot read each other's lines. */ +let counter = 0 +function scratchLog(): string { + const p = join(tmpdir(), `auto-resume-discovery-${process.pid}-${counter++}.log`) + rmSync(p, { force: true }) + return p +} +function readLogs(p: string): string[] { + const logs = existsSync(p) ? readFileSync(p, "utf8").split("\n") : [] + rmSync(p, { force: true }) + return logs +} + +/** + * Discovery sweep tests. + * + * The sweep is what makes a session that was ALREADY running when the plugin + * loaded visible to the watchdog. Without it, a session that started before + * attach never emits `session.execution.started` to us, so nothing is marked + * busy and the stall watchdog has nothing to inspect. + * + * Two shapes are exercised, because v2's plugin `session` domain is a narrowed + * Pick that may or may not carry `list`/`active`: + * - onDomain: the methods hang off ctx.session (what this build prefers) + * - onClient: they are absent there and only reachable via ctx.client + * + * Every group carries a CONTROL that fails if discovery silently no-ops, which + * is the failure mode this whole feature has. + */ + +type ApiShape = "onDomain" | "onClient" | "neither" + +type Harness = { + logs: string[] + calls: string[] + listRows: Array> + activeIDs: string[] +} + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +/** A session already mid-turn at attach time: `list` knows it, `active` says busy. */ +const RUNNING = "ses_running_before_attach" + +async function run( + shape: ApiShape, + opts: Record = {}, + overrides: { listThrows?: boolean } = {}, +): Promise { + const calls: string[] = [] + const stream = makeEventStream() + const logFile = scratchLog() + + const listRows = [{ id: RUNNING, title: "already running" }] + const activeIDs = [RUNNING] + + const list = async () => { + calls.push("list") + if (overrides.listThrows) throw new Error("list unavailable") + return listRows + } + const active = async () => { + calls.push("active") + return Object.fromEntries(activeIDs.map((id) => [id, { type: "running" }])) + } + + const session: Record = { + context: async () => [], + interrupt: async () => ({}), + synthetic: async () => ({}), + prompt: async () => ({}), + } + if (shape === "onDomain") { + session.list = list + session.active = active + } + + const client: Record | undefined = + shape === "onClient" ? { session: { list, active } } : shape === "onDomain" ? undefined : {} + + const ctx: any = { + event: stream, + options: { warmupMs: 0, discoveryDelayMs: 20, logFile, ...opts }, + session, + } + if (client) ctx.client = client + + const cleanup = await (plugin as any).setup(ctx) + await wait(220) // initial sweep is discoveryDelayMs out + ;(cleanup as (() => void) | undefined)?.() + return { logs: readLogs(logFile), calls, listRows, activeIDs } +} + +const discoveryLine = (logs: string[]) => logs.find((l) => l.includes("discovery:")) + +describe("v2: session discovery sweep", () => { + test("CONTROL (list on the session domain): an already-running session is adopted as busy", async () => { + const { logs, calls } = await run("onDomain") + expect(calls).toContain("list") + expect(calls).toContain("active") + const line = discoveryLine(logs) + expect(line).toBeDefined() + expect(line).toContain("adopted 1 already running") + }) + + test("list on the session domain seeds a session the plugin has never seen", async () => { + const { logs } = await run("onDomain") + // Seeded and adopted are separate counts; a session that exists but is + // not running must still be seeded so revert/cleanup know about it. + expect(discoveryLine(logs)).toContain("seeded 1 session(s)") + }) + + test("CONTROL (list only on ctx.client): the sweep still works via the client fallback", async () => { + const { logs, calls } = await run("onClient") + expect(calls).toContain("list") + expect(discoveryLine(logs)).toContain("adopted 1 already running") + }) + + test("a host exposing neither degrades quietly instead of throwing", async () => { + // No `list`, no `client`. The sweep must not raise; the plugin falls back + // to the event-derived busy set, which is what it did before this feature. + const { logs, calls } = await run("neither") + expect(calls).toEqual([]) + expect(logs.filter((l) => l.includes("discovery failed"))).toEqual([]) + }) + + test("a throwing list() is contained and reported at debug level, not as a crash", async () => { + const { logs } = await run("onDomain", {}, { listThrows: true }) + expect(logs.filter((l) => l.includes("session discovery failed"))).toEqual([]) + // The sweep still completed — it simply had nothing to adopt. + expect(logs.filter((l) => l.includes("watchdog failed"))).toEqual([]) + }) + + test("rows without a usable id are skipped, not treated as sessions", async () => { + const logFile = scratchLog() + const stream = makeEventStream() + const rows = [{ id: "not-a-session" }, { id: 42 }, {}, null] + const ctx: any = { + event: stream, + options: { warmupMs: 0, discoveryDelayMs: 20, logFile }, + session: { + context: async () => [], + active: async () => ({}), + list: async () => rows, + }, + } + const cleanup = await (plugin as any).setup(ctx) + await wait(220) + ;(cleanup as (() => void) | undefined)?.() + const logs = readLogs(logFile) + expect(discoveryLine(logs)).not.toBeDefined() + expect(logs.filter((l) => l.includes("discovery failed"))).toEqual([]) + }) + + test("a { data } envelope is unwrapped, matching the client's response shape", async () => { + const logFile = scratchLog() + const stream = makeEventStream() + const ctx: any = { + event: stream, + options: { warmupMs: 0, discoveryDelayMs: 20, logFile }, + session: { + context: async () => [], + active: async () => ({ [RUNNING]: { type: "running" } }), + list: async () => ({ data: [{ id: RUNNING }] }), + }, + } + const cleanup = await (plugin as any).setup(ctx) + await wait(220) + ;(cleanup as (() => void) | undefined)?.() + expect(discoveryLine(readLogs(logFile))).toContain("adopted 1 already running") + }) + + test("cleanup stops every timer — no API calls after teardown", async () => { + const calls: string[] = [] + const stream = makeEventStream() + const logFile = scratchLog() + const ctx: any = { + event: stream, + // Tight watchdog so a leaked interval is visible inside the wait. + options: { warmupMs: 0, discoveryDelayMs: 20, checkIntervalMs: 20, chunkTimeoutMs: 10_000, logFile }, + session: { + context: async () => [], + active: async () => { + calls.push("active") + return {} + }, + list: async () => { + calls.push("list") + return [{ id: RUNNING }] + }, + }, + } + const cleanup = (await (plugin as any).setup(ctx)) as () => void + await wait(120) + // Precondition: the watchdog really is running, so a zero afterwards is + // the result of teardown rather than of a plugin that never started. + expect(calls.filter((c) => c === "active").length).toBeGreaterThan(0) + + cleanup() + const atTeardown = calls.length + await wait(200) + expect(calls.length).toBe(atTeardown) + rmSync(logFile, { force: true }) + }) +}) \ No newline at end of file diff --git a/src/v2/index.options.test.ts b/src/v2/index.options.test.ts index 280f7fd..e1d7343 100644 --- a/src/v2/index.options.test.ts +++ b/src/v2/index.options.test.ts @@ -1,5 +1,6 @@ import { describe, test, expect } from "bun:test" -import { readFileSync } from "node:fs" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" import { join } from "node:path" import plugin from "./index" @@ -7,6 +8,9 @@ const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) const SID = "ses_opts" +/** Scratch log path per replay, so parallel cases cannot read each other's lines. */ +let counter = 0 + const textPart = (t: string) => ({ type: "text", text: t }) /** Older than every activeUserWindowMs we pass, so guard (b) cannot mask a failure. */ const oldUserTurn = () => ({ @@ -84,21 +88,21 @@ async function replay( ): Promise { const injected: Harness["injected"] = [] const interrupts: string[] = [] - const logs: string[] = [] const stream = makeEventStream() const text = harnessOpts.text ?? READY_TEXT + // v2 removed v1's server log endpoint, so the plugin writes to a file. Point + // it at a scratch path and read the lines back after teardown — this is also + // what proves the file sink works, rather than assuming it does. + const logFile = join(tmpdir(), `auto-resume-test-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + const ctx: any = { event: stream, // The plugin reads its config from `ctx.options`; the second argument to // `setup()` is ignored. Passing options the other way makes every // negative assertion below pass for the wrong reason. - options: opts, - app: { - log: (_level: string, line: string) => { - logs.push(String(line)) - }, - }, + options: { ...opts, logFile }, session: { context: async () => [oldUserTurn(), assistantTurn(text)], // Empty: no other session is active, so the `lastWasTaskTool` branch @@ -128,6 +132,8 @@ async function replay( } await wait(600) // handleEvent is sync; its work is async ;(cleanup as (() => void) | undefined)?.() + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) return { injected, interrupts, logs } } @@ -285,12 +291,26 @@ describe("v2: option reporting at startup", () => { }) test("the startup line names the accepted-but-inert options in use", async () => { - const { logs } = await replay([], { chunkTimeoutMs: 5000, subagentWaitMs: 15_000, discoveryDelayMs: 5_000 }) + const { logs } = await replay([], { chunkTimeoutMs: 5000, subagentWaitMs: 15_000 }) + const ready = logs.filter((l) => l.includes("ready (opencode v2)")) + expect(ready).toHaveLength(1) + expect(ready[0]).toContain("accepted-but-inert=") + expect(ready[0]).toContain("subagentWaitMs") + }) + + test("discoveryDelayMs is live on v2, so it is no longer listed as inert", async () => { + // Control: a genuinely inert option, set the same way, IS still listed. + // Without this the assertion below would pass for the wrong reason. + const { logs } = await replay([], { + chunkTimeoutMs: 5000, + discoveryDelayMs: 5_000, + subagentWaitMs: 15_000, + }) const ready = logs.filter((l) => l.includes("ready (opencode v2)")) expect(ready).toHaveLength(1) expect(ready[0]).toContain("accepted-but-inert=") expect(ready[0]).toContain("subagentWaitMs") - expect(ready[0]).toContain("discoveryDelayMs") + expect(ready[0]).not.toContain("discoveryDelayMs") }) test("the startup line carries no accepted-but-inert list when none are set", async () => { diff --git a/src/v2/index.ts b/src/v2/index.ts index 0e4d9ef..8eec638 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -13,7 +13,11 @@ * - Assistant text is accumulated from `session.text.delta` events for liveness; * `ctx.session.context()` (stable v2) supplies the authoritative final * assistant text at idle time — replacing v1's `session.messages()` polling. - * - No `ctx.app.log` in v2 — logs go to the console (captured by opencode logs). + * - No `ctx.app.log` in v2: `ctx.app` is only `{name, version, channel}` (see + * `packages/plugin/src/app.ts`), and v2 has no `app.log` endpoint. Console + * output from a hosted plugin is not captured anywhere retrievable, so v1's + * `ctx.client.app.log(...)` has no equivalent. This build therefore appends + * to its own log file — see LOG_FILE below. * - Targets the stable v2 API (`@opencode/plugin`). Event names are unchanged * from the beta port; `session.execution.interrupted` now carries a `reason`. * @@ -50,6 +54,10 @@ type AutoResumePlugin = { const define = (plugin: T): T => plugin +import { appendFileSync, existsSync, mkdirSync, statSync, writeFileSync } from "node:fs" +import { homedir } from "node:os" +import { dirname, join } from "node:path" + /** * The subset of the v2 plugin input this plugin actually consumes. Typed * structurally so the file type-checks standalone, without pinning a @@ -203,6 +211,61 @@ export interface AutoResumeOptions { doneWithoutDetailsPrompt?: string /** Prompt sent when a tool call was emitted inside reasoning. */ thinkingToolRecoveryPrompt?: string + /** + * Append this build's log here instead of the default path + * (`~/.local/state/opencode-v2/auto-resume.log`). + * + * v2 removed v1's `app.log` endpoint, so this is the only retrievable + * record of what the plugin did. The `AUTO_RESUME_LOG_FILE` environment + * variable sets the same thing and wins over neither. + */ + logFile?: string +} + +// --------------------------------------------------------------------------- +// Logging +// --------------------------------------------------------------------------- + +/** + * Where this build's log lines go, and why not somewhere official. + * + * v1 wrote through `ctx.client.app.log({ body: { service, level, message } })`, + * a real endpoint that landed in the opencode log file. v2 removed it: + * `ctx.app` is `{ name, version, channel }` (`packages/plugin/src/app.ts`), + * there is no `app` group in `packages/protocol/src/groups/`, and a hosted + * plugin's `console.log` is not captured by the OpenChamber process. Without + * this the plugin is completely silent — you cannot tell a working watchdog + * from a dead one, which is exactly the failure that made a stall + * indistinguishable from "nothing happened". + * + * Override with AUTO_RESUME_LOG_FILE. Size-capped so an unattended run cannot + * fill the disk. + */ +const DEFAULT_LOG_FILE = join(homedir(), ".local", "state", "opencode-v2", "auto-resume.log") +const LOG_FILE_MAX_BYTES = 2 * 1024 * 1024 + +/** + * Append one line to `target`, best effort. + * + * Every failure here is swallowed on purpose: logging must never be the reason + * the watchdog stops running. The size check races harmlessly between + * processes — losing a line beats throwing. + */ +function appendLogFile(target: string, level: string, line: string): void { + try { + mkdirSync(dirname(target), { recursive: true }) + try { + if (existsSync(target) && statSync(target).size > LOG_FILE_MAX_BYTES) { + writeFileSync(target, "") + } + } catch { + // Size check is an optimisation, not a requirement. + } + appendFileSync(target, `${new Date().toISOString()} ${level.toUpperCase().padEnd(5)} ${line}\n`) + } catch { + // Unwritable log dir, read-only fs, quota — none of these should + // propagate into the recovery path. + } } // --------------------------------------------------------------------------- @@ -229,7 +292,10 @@ const DEFAULT_DEBUG = false const DEFAULT_ACTIVE_USER_WINDOW_MS = 5 * 60_000 // ---- Defaults ported from v1 (Mte90/opencode-auto-resume#33) ---- +// v1 keeps SESSION_DISCOVERY_INTERVAL_MS as an internal constant rather than an +// option, so it is not configurable here either. 60s matches v1. const DEFAULT_DISCOVERY_DELAY_MS = 5_000 +const DEFAULT_SESSION_DISCOVERY_INTERVAL_MS = 60_000 const DEFAULT_WARMUP_MS = 15_000 const DEFAULT_MIN_ACTIVITY_GAP_MS = 1_000 const DEFAULT_TOOL_TEXT_CHECK_DELAY_MS = 3_000 @@ -617,6 +683,7 @@ export default define({ const debug = opts.debug ?? DEFAULT_DEBUG const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS + const logFile = opts.logFile ?? process.env.AUTO_RESUME_LOG_FILE ?? DEFAULT_LOG_FILE // ---- Ported from v1 (Mte90/opencode-auto-resume#33) ---- const rawBusyStallStrategy = opts.busyStallStrategy ?? "continue" @@ -627,6 +694,8 @@ export default define({ // mistaken for live behaviour. const resumeOnActionIntent = opts.resumeOnActionIntent !== false const warmupMs = opts.warmupMs ?? DEFAULT_WARMUP_MS + // Live since the v2 discovery sweep landed (see discoverSessions). + const discoveryDelayMs = opts.discoveryDelayMs ?? DEFAULT_DISCOVERY_DELAY_MS const minActivityGapMs = opts.minActivityGapMs ?? DEFAULT_MIN_ACTIVITY_GAP_MS const streamingFailureErrorNames = opts.streamingFailureErrorNames ?? DEFAULT_STREAMING_FAILURE_ERROR_NAMES const streamingFailureMessagePatterns = opts.streamingFailureMessagePatterns ?? DEFAULT_STREAMING_FAILURE_MESSAGE_PATTERNS @@ -672,7 +741,6 @@ export default define({ "subagentNativeCompactionEnabled", "silentDeadStreamMinTokens", "subagentWaitMs", - "discoveryDelayMs", "toolTextCheckDelayMs", "thinkingToolRecoveryPrompt", "doneWithoutWorkPrompt", @@ -715,6 +783,7 @@ export default define({ "doneWithoutDetailsPrompt", "thinkingToolRecoveryPrompt", "doneWithoutWorkPrompt", + "logFile", ]) const unknownOptions = Object.keys(opts).filter((key) => !RECOGNISED_OPTIONS.has(key)) if (unknownOptions.length > 0) { @@ -737,16 +806,11 @@ export default define({ function log(level: "info" | "warn" | "error", msg: string) { const line = `[auto-resume] ${msg}` - // Prefer the server log sink so plugin output is actually retrievable - // (console output from a hosted plugin is not captured anywhere useful). - try { - const appLog = (ctx as any)?.app?.log - if (typeof appLog === "function") { - void appLog.call((ctx as any).app, level === "info" ? "info" : level, line) - } - } catch { - // fall through to console - } + // v2 exposes no server log sink to plugins (ctx.app is + // {name,version,channel}), so the file is the only retrievable record. + appendLogFile(logFile, level, line) + // Console too: harmless when nobody captures it, and the reason the + // startup line is visible at all when running under `opencode serve`. if (level === "error") console.error(line) else if (level === "warn") console.warn(line) else console.log(line) @@ -1543,6 +1607,118 @@ async function inspectOnIdle(sid: string) { ) }, checkIntervalMs) + // Discovery sweep. Separate timer from the watchdog because the two have + // different jobs and different costs: the watchdog runs every few seconds + // and must stay cheap, discovery lists every session and runs every 60s. + // The initial sweep waits out discoveryDelayMs so plugin load does not + // compete with the turn that triggered it. + const discoveryTimer = setInterval(() => { + discoverSessions().catch((e) => + log("error", `discovery failed: ${e instanceof Error ? e.message : String(e)}`), + ) + }, DEFAULT_SESSION_DISCOVERY_INTERVAL_MS) + const initialDiscovery = setTimeout(() => { + discoverSessions().catch(() => {}) + }, discoveryDelayMs) + + // --------------------------------------------------------------------- + // Session discovery + // --------------------------------------------------------------------- + + /** + * v2 hands plugins a narrowed `ctx.session` domain (see + * `packages/plugin/src/promise/session.ts`) that omits `list` and + * `active`, so both are read defensively: off the session domain first, + * then off the raw client. A host that supplies neither degrades to the + * event-derived busy set rather than failing. + */ + async function callSessionApi(name: string): Promise { + const fromDomain = (ctx.session as any)[name] + if (typeof fromDomain === "function") { + try { + return (await fromDomain.call(ctx.session)) as T + } catch (e) { + dbg(`ctx.session.${name}() failed:`, e instanceof Error ? e.message : String(e)) + } + } + const fromClient = ctx.client?.session?.[name as "get"] + if (typeof fromClient === "function") { + try { + return (await (fromClient as any).call(ctx.client!.session)) as T + } catch (e) { + dbg(`ctx.client.session.${name}() failed:`, e instanceof Error ? e.message : String(e)) + } + } + return undefined + } + + /** Unwrap the `{ data }` envelope the client uses, or pass an array through. */ + function unwrapList(response: unknown): Array> { + if (Array.isArray(response)) return response as Array> + if (response && typeof response === "object") { + const data = (response as { data?: unknown }).data + if (Array.isArray(data)) return data as Array> + } + return [] + } + + /** + * Seed watch state for sessions that already exist, and mark the ones + * currently running as busy. + * + * v1 did this by polling `session.list()` and trusting a `status` field + * on each row. v2 has something better: `session.active()` is the + * server's own record of what is running right now, so "busy" is read + * rather than guessed. A session that was already mid-turn when the + * plugin loaded would otherwise be invisible — no `execution.started` + * ever reaches us for it, so the stall watchdog would have nothing to + * watch. + */ + async function discoverSessions() { + try { + const listed = unwrapList(await callSessionApi("list")) + let seeded = 0 + for (const row of listed) { + const sid = row?.id + if (typeof sid !== "string" || !sid.startsWith("ses_")) continue + if (!sessions.has(sid)) { + ensureWatch(sid) + seeded++ + } + } + + // Authoritative busy set. Anything listed as running but not + // already tracked busy starts its stall clock now, not from + // whenever the plugin happened to attach. + const active = await callSessionApi>("active") + let adopted = 0 + if (active && typeof active === "object") { + for (const [sid, val] of Object.entries(active)) { + if (!sid.startsWith("ses_")) continue + const w = ensureWatch(sid) + if (w.status !== "busy") { + markBusy(sid) + adopted++ + } else { + // Already busy from events; refresh nothing, but make + // sure the session is not left `createdAt`-stale so + // the warmup window does not swallow its first stall. + void val + } + } + } + + if (seeded > 0 || adopted > 0) { + log("info", `discovery: seeded ${seeded} session(s), adopted ${adopted} already running`) + } else { + dbg("discovery: nothing new") + } + } catch (err) { + const msg = err instanceof Error ? err.message : String(err) + dbg(`session discovery failed: ${msg}`) + } + } + // --------------------------------------------------------------------- // Event stream // --------------------------------------------------------------------- @@ -1909,6 +2085,8 @@ async function inspectOnIdle(sid: string) { running = false eventAbort.abort() clearInterval(watchdog) + clearInterval(discoveryTimer) + clearTimeout(initialDiscovery) sessions.clear() log("info", "stopped") } From e61dfec7e22b8ca49d189d5e14ac662f71853396 Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 14:58:48 -0400 Subject: [PATCH 15/57] feat(v2): catch the premature stop, and make activeUserWindowMs real MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The stall watchdog only inspects busy sessions, so a turn that ends cleanly with the work unfinished is invisible to it by construction — no stall, no error, no streaming failure, just idle. Two v1 detectors for that case are ported onto the idle path. A done-claim with no work report in it. v2 gated the details prompt on a 400-character threshold, which cannot tell "Task done." from a real two-line summary: it over-nudged terse reports and let short real ones pass. It now uses v1's containsWorkDescription — a backticked or bare path with a dotted extension, or a report header (Changed / Verification / Tests run / Results / Commands run). Asking again after a genuine report loops forever (#26). A trailing 🎉. The model's own "finished" signal, so completion latches instead of being nudged over. Per-turn: a new turn re-opens the question, and a later idle with no new text does not re-derive it forever. That exposed a second defect in activeUserWindowMs. The default was lowered to 5 minutes to match e1b8374, but the implementation asked "is the newest message a user message?" instead of v1's "was any inbound user message inside the window?". At idle time the newest message is the assistant turn that just finished, so the option never applied and the plugin nudged straight over a user who was mid-conversation. It now walks back for the most recent user message. dbg() also goes to the log file now, for the same reason the rest of the logging does: debug output is exactly when you are working out what the plugin did, and console-only output in v2 reaches nobody. src/v2/index.premature-stop.test.ts, 17 tests: a control per group, plus the two halves of the length heuristic, one-nudge-per-turn, budget re-arm, maxRetries, latch survival, latch reset, the handoff and active-user guards, and the session.context() fallback. 656 pass, 0 fail. --- docs/known-issues-v2.md | 30 +++ src/v2/index.premature-stop.test.ts | 295 ++++++++++++++++++++++++++++ src/v2/index.ts | 105 +++++++++- 3 files changed, 422 insertions(+), 8 deletions(-) create mode 100644 src/v2/index.premature-stop.test.ts diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 6236d08..bb4b42c 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -52,6 +52,13 @@ the fix is mte090's `e1b8374`, applied to the v2 code path. The stall path branch still reads `900000` because it predates `e1b8374`; the PR does not touch v1, and master already carries the fix. + Lowering the default was not enough on its own: the v2 implementation asked + "is the newest message a user message?" rather than v1's "was any inbound user + message inside the window?". At idle time the newest message is the assistant + turn that just finished, so the window never applied and the nudge fired + straight over a user who was mid-conversation. The check now walks back for + the most recent user message, which is what the option describes. + ## v2-only options - `injectIntervalMs` — minimum gap between recovery injections for one session. @@ -103,3 +110,26 @@ Both calls are read defensively — off the plugin `session` domain first, then off the raw client — because v2 hands plugins a narrowed `Pick` that omits both. A host that supplies neither degrades to the event-derived busy set rather than failing. + +## Premature stop — the case the stall watchdog cannot see + +A session can end its turn cleanly while the work is not done: no stall, no +error, no streaming failure, just a short "Task done." and then idle. The stall +watchdog only looks at *busy* sessions, so this is invisible to it by +construction. Two detectors run on the idle path instead, both ported from v1: + +- **A done-claim with no work report in it.** v2 used to gate the details prompt + on a 400-character threshold, which cannot tell "Task done." from a real + two-line summary — it both over-nudged terse reports and let short real ones + pass. It now uses v1's `containsWorkDescription`: a backticked or bare path + with a dotted extension, or a report header (`Changed`, `Verification`, + `Tests run`, `Results`, `Commands run`). Asking again after a genuine report + loops forever, which is issue #26. +- **A trailing 🎉.** That is the model's own "I finished" signal, so the plugin + latches completion and stops nudging rather than talking over a deliberate + stop. The latch is per-turn: a new turn re-opens the question, and a later + idle with no new text does not re-derive it forever. + +v1 cross-checks both against tracked todo state before latching. v2 has no todo +state yet, so the emoji latches on its own — see `doneWithoutWorkPrompt` in the +inert table above for what that costs. diff --git a/src/v2/index.premature-stop.test.ts b/src/v2/index.premature-stop.test.ts new file mode 100644 index 0000000..ccaa307 --- /dev/null +++ b/src/v2/index.premature-stop.test.ts @@ -0,0 +1,295 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_premature" + +/** + * Premature-stop detection. + * + * This is the failure mode where the model ends its turn cleanly — no stall, no + * error, no streaming failure — while the work is not actually finished. The + * stall watchdog cannot see it: the session is idle, not wedged, so nothing is + * ever "too silent". What catches it is reading the last assistant text and + * asking whether it actually reports work. + * + * Two detectors: + * - endsWithCelebration: the model's own "finished" signal, which must latch + * rather than be nudged over. + * - containsWorkDescription: a done-claim with no work report in it, which gets + * exactly one request for details. + * + * Every group carries a control. The failure mode that matters here is a test + * that passes because the nudge never fired at all. + */ + +type Harness = { + injected: Array<{ kind: string; text?: string }> + logs: string[] +} + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +/** Old enough that activeUserWindowMs cannot mask a failure. */ +const userTurn = (ageMs = 60 * 60_000) => ({ + type: "user", + id: "msg_u0", + time: { created: Date.now() - ageMs }, + content: [{ type: "text", text: "do the thing" }], +}) +const assistantTurn = (text: string) => ({ + type: "assistant", + id: "msg_a1", + time: { created: Date.now() - 60_000 }, + content: [{ type: "text", text }], +}) + +/** One turn that streams `text` and then goes idle. */ +function turnEvents(text: string) { + return [ + ev("session.execution.started"), + ev("session.text.delta", { messageID: "msg_a1", delta: text }), + ev("session.text.ended", { messageID: "msg_a1" }), + ev("session.idle"), + ] +} + +/** Idle again with no new turn — what a latched completion must survive. */ +const idleAgain = [ev("session.idle")] + +const OPTIONS = { + // Long enough that the stall watchdog never fires: these tests are about the + // idle path only, and a stall injection would mask which path produced a nudge. + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + // Without this the second nudge in a two-turn test is debounced away as + // "too soon after the last injection", and the re-arm test fails for the + // wrong reason. + injectIntervalMs: 0, +} + +async function replay( + events: any[], + opts: Record = {}, + harnessOpts: { userAgeMs?: number } = {}, +): Promise { + const injected: Harness["injected"] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-premature-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile, ...opts }, + session: { + context: async () => [userTurn(harnessOpts.userAgeMs), assistantTurn(TEXT_FOR_HARNESS.value)], + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => { + injected.push({ kind: "synthetic", text: a?.text }) + return {} + }, + prompt: async (a: any) => { + injected.push({ kind: "prompt", text: a?.text }) + return {} + }, + }, + } + + const cleanup = await (plugin as any).setup(ctx) + for (const e of events) { + stream.push(e) + await wait(10) + } + await wait(600) + ;(cleanup as (() => void) | undefined)?.() + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { injected, logs } +} + +/** + * The idle handler prefers the streamed delta buffer and only falls back to + * `session.context()` when it is empty. Tests that stream text exercise the + * buffer; this lets a case that must exercise the fallback set it explicitly. + */ +const TEXT_FOR_HARNESS = { value: "unrelated" } + +/** Terse done-claim with no work report — the premature stop. */ +const BARE_DONE = "Task done." + +/** Long, but no actual work report. Length alone must not save it. */ +const LONG_BUT_EMPTY = `Task done.\n\n${"I considered the situation carefully. ".repeat(30)}` + +/** Short, but a real report. Length alone must not condemn it. */ +const SHORT_BUT_REAL = "Done. Changed src/v2/index.ts" + +describe("v2: premature stop — a done-claim with no work in it", () => { + test("CONTROL: a bare 'Task done.' with nothing after it IS nudged", async () => { + const { injected } = await replay(turnEvents(BARE_DONE)) + expect(injected.length).toBeGreaterThan(0) + expect(injected[0].text).toContain("verify") + }) + + test("a long done-claim with no work report is still nudged", async () => { + // The bug this replaces: a 400-character gate let this through. + expect(LONG_BUT_EMPTY.length).toBeGreaterThan(400) + const { injected } = await replay(turnEvents(LONG_BUT_EMPTY)) + expect(injected.length).toBeGreaterThan(0) + }) + + test("a short done-claim that names a changed file is NOT nudged", async () => { + // The other half of the same bug: SHORT_BUT_REAL is under 400 characters. + expect(SHORT_BUT_REAL.length).toBeLessThan(400) + const { injected } = await replay(turnEvents(SHORT_BUT_REAL)) + expect(injected).toEqual([]) + }) + + test("a done-claim reporting tests run is NOT nudged", async () => { + const { injected } = await replay( + turnEvents("Done.\n\nTests run: 639 pass, 0 fail\nResults: all green"), + ) + expect(injected).toEqual([]) + }) + + test("a done-claim reporting verification is NOT nudged", async () => { + const { injected } = await replay(turnEvents("All done.\n\nVerification: tsc clean")) + expect(injected).toEqual([]) + }) + + test("the nudge is sent once per turn, not once per idle event", async () => { + const { injected } = await replay([...turnEvents(BARE_DONE), ...idleAgain, ...idleAgain]) + expect(injected.length).toBe(1) + }) + + test("a new turn re-arms the budget", async () => { + const { injected } = await replay([ + ...turnEvents(BARE_DONE), + ...idleAgain, + ...turnEvents(BARE_DONE), + ]) + expect(injected.length).toBe(2) + }) + + test("maxRetries caps how many details prompts one busy cycle can spend", async () => { + const { injected } = await replay(turnEvents(BARE_DONE), { maxRetries: 1 }) + expect(injected.length).toBe(1) + }) +}) + +describe("v2: celebration latches completion", () => { + test("CONTROL: a bare done-claim without 🎉 is still nudged", async () => { + const { injected } = await replay(turnEvents(BARE_DONE)) + expect(injected.length).toBeGreaterThan(0) + }) + + test("a trailing 🎉 latches completion instead of nudging", async () => { + const { injected, logs } = await replay(turnEvents("All set.\n\nChanged the parser. 🎉")) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("latching completion"))).toBe(true) + }) + + test("the latch survives a later idle with no new text", async () => { + // debug:true so the second (silent) visit is observable at all — before + // the log-sink fix these lines only reached a console nobody captures. + const { injected, logs } = await replay( + [...turnEvents("All set. 🎉"), ...idleAgain, ...idleAgain], + { debug: true }, + ) + expect(injected).toEqual([]) + expect(logs.filter((l) => l.includes("latching completion"))).toHaveLength(1) + expect(logs.some((l) => l.includes("completion already latched"))).toBe(true) + }) + + test("a new turn clears the latch, so a later done-claim is judged on its own", async () => { + const { injected, logs } = await replay([ + ...turnEvents("All set. 🎉"), + ...idleAgain, + // A fresh turn: budgets reset, latch resets, and this one is bare. + ...turnEvents(BARE_DONE), + ]) + expect(injected.length).toBeGreaterThan(0) + expect(logs.filter((l) => l.includes("latching completion"))).toHaveLength(1) + }) + + test("punctuation after the emoji still counts as celebration", async () => { + // v1 normalises trailing [.!?] before the check. + const { injected } = await replay(turnEvents("Wrapped it up. 🎉.")) + expect(injected).toEqual([]) + }) +}) + +describe("v2: premature-stop detection does not fight the other idle guards", () => { + test("a turn that hands off to the user is never nudged", async () => { + // A question is a legitimate stop, however terse. + const { injected } = await replay(turnEvents("Task done.\n\nWhich file should I edit?")) + expect(injected).toEqual([]) + }) + + test("a recent user message stands the nudge down", async () => { + // The user turn must be INSIDE the window for this to mean anything; the + // default harness turn is an hour old and would not stand down at all. + const { injected } = await replay(turnEvents(BARE_DONE), { activeUserWindowMs: 600_000 }, { userAgeMs: 30_000 }) + expect(injected).toEqual([]) + }) + + test("an old user message does NOT stand the nudge down", async () => { + // The control for the test above: same window, older turn, nudge fires. + const { injected } = await replay(turnEvents(BARE_DONE), { activeUserWindowMs: 600_000 }, { userAgeMs: 3_600_000 }) + expect(injected.length).toBeGreaterThan(0) + }) + + test("the fallback to session.context() judges the same way", async () => { + // No streamed text at all: the handler must read the authoritative history. + TEXT_FOR_HARNESS.value = BARE_DONE + try { + const { injected } = await replay([ev("session.execution.started"), ev("session.idle")]) + expect(injected.length).toBeGreaterThan(0) + } finally { + TEXT_FOR_HARNESS.value = "unrelated" + } + }) +}) \ No newline at end of file diff --git a/src/v2/index.ts b/src/v2/index.ts index 8eec638..c21ab9e 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -132,6 +132,9 @@ interface SessionWatch { continueTimestamps: number[] doneClaimAttempts: number intentNudgeAttempts: number + /** Set when the model's own completion signal was seen (a trailing 🎉). + * Latches so a finished session stops being nudged. */ + completionSignaled: boolean /** Set when a failure-triggered recovery is pending, so the delayed prompt isn't cancelled by the failure's own idle transition. */ pendingRecoveryArmed: boolean @@ -551,6 +554,48 @@ function containsDoneClaimPattern(text: string, patterns: RegExp[] = DONE_CLAIM_ return patterns.some((pat) => pat.test(lastLines)) } +/** + * True when a done-claim already carries a concrete work report. + * + * v1 asks for details once per done-claim and then trusts the answer. A length + * check cannot tell the difference between "Task done." (nothing was done) and + * a two-line summary, so v2 replaces the 400-char threshold with the same + * structural test v1 uses. Prompting again after a real report would loop + * forever (#26). + * + * Three signals, in order of specificity: + * - a backticked span naming a dotted file: `src/index.ts` + * - a bare path with a slash and a dotted extension: /a/b.py + * - a report header: changed / verification / results / commands run + */ +function containsWorkDescription(text: string): boolean { + // Backticked span mentioning a dotted filename. + if (/`[^`\n]*\.[a-zA-Z0-9]{1,8}[^`\n]*`/.test(text)) return true + // Bare path with a slash and a dotted extension. + if (/[\w\-~.][\w\-.~\/]*\/[\w\-.~]*\.[a-zA-Z]{1,8}\b/.test(text)) return true + // Report section headers. + if ( + /^(changed|modified|deleted|created|updated|renamed|moved|files?\s+changed|verification|verified|tests?(?:\s+run|\s+passing|\s+pass)?|results?|outcome|commands?\s+(?:run|executed))/im.test( + text, + ) + ) + return true + return false +} + +/** + * True when the last assistant turn closes with a celebration emoji. + * + * A 🎉 is the model's own "I finished" signal. v1 uses it to latch completion + * rather than keep nudging — but only when no work is left, otherwise a + * premature 🎉 would be treated as a real finish. Ported without the todo + * cross-check, so on v2 it latches on the emoji alone. + */ +function endsWithCelebration(text: string): boolean { + const normalized = text.trim().replace(/[.!?]+$/, "") + return normalized.endsWith("🎉") +} + /** Model ends with ":" announcing intent without executing. */ function containsActionIntent(text: string): boolean { if (text.length <= 15) return false @@ -800,8 +845,13 @@ export default define({ (FEATURE_GATED_OPTIONS as readonly string[]).includes(key), ) + // Debug output goes to the same file as everything else. Console-only debug + // is invisible in v2 for the same reason the rest of the logging was, and + // debug is exactly when you are trying to work out what the plugin did. const dbg = (...args: unknown[]) => { - if (debug) console.log("[auto-resume:debug]", ...args) + if (!debug) return + appendLogFile(logFile, "debug", `[auto-resume:debug] ${args.map((a) => String(a)).join(" ")}`) + console.log("[auto-resume:debug]", ...args) } function log(level: "info" | "warn" | "error", msg: string) { @@ -881,6 +931,7 @@ export default define({ continueTimestamps: [], doneClaimAttempts: 0, intentNudgeAttempts: 0, + completionSignaled: false, pendingRecoveryArmed: false, permissionPending: false, permissionPendingAt: null, @@ -916,6 +967,8 @@ export default define({ w.toolTextAttempts = 0 w.doneClaimAttempts = 0 w.intentNudgeAttempts = 0 + // A new turn re-opens the question of whether the work is finished. + w.completionSignaled = false w.gaveUp = false w.recentToolCalls = [] w.textParts.clear() @@ -1441,10 +1494,27 @@ export default define({ } } // (b) The user was recently active — they are mid-conversation, not stuck. - if (newest?.type === "user") { - const ts = newest.time?.created ?? newest.info?.time?.created - if (typeof ts === "number" && Date.now() - ts < activeUserWindowMs) return true + // + // v1 stamps `lastUserMessageAt` on ANY inbound user message and asks + // "was there one inside activeUserWindowMs?". This used to ask "is the + // newest message a user message?", which at idle time is essentially never + // true — the newest message is the assistant turn that just finished — so + // the whole window was dead. Walk back for the most recent user message. + let lastUserAt: number | undefined + for (let i = messages.length - 1; i >= 0; i--) { + const m = messages[i] as { + type?: string + role?: string + time?: { created?: number } + info?: { time?: { created?: number }; role?: string } + } + const isUser = m?.type === "user" || m?.role === "user" || m?.info?.role === "user" + if (!isUser) continue + const ts = m.time?.created ?? m.info?.time?.created + if (typeof ts === "number") lastUserAt = ts + break } + if (lastUserAt !== undefined && Date.now() - lastUserAt < activeUserWindowMs) return true return false } catch { return false @@ -1473,6 +1543,20 @@ async function inspectOnIdle(sid: string) { return } + // The model's own completion signal. A trailing 🎉 means it considers the + // work finished, so nudging here would talk over a deliberate stop. + // Latched rather than re-derived, because the next idle with no new text + // would otherwise re-check the same turn forever. + if (endsWithCelebration(text)) { + if (!w.completionSignaled) { + w.completionSignaled = true + log("info", `${short(sid)} turn ends with a celebration — latching completion, not nudging`) + } else { + dbg(`${short(sid)} completion already latched — skipping`) + } + return + } + if (containsToolCallAsText(text)) { await targetedRecovery(sid, "tool-call-as-text", opts.toolTextRecoveryPrompt ?? TOOL_TEXT_RECOVERY_PROMPT, "toolTextAttempts") return @@ -1485,11 +1569,16 @@ async function inspectOnIdle(sid: string) { await targetedRecovery(sid, "action-intent", opts.actionIntentPrompt ?? opts.continuePrompt ?? CONTINUE_PROMPT, "intentNudgeAttempts") return } - if (containsDoneClaimPattern(text, doneClaimPatterns) && w.doneClaimAttempts < 1) { - // Single verification nudge for suspiciously terse completions - const trimmed = text.trim() - if (trimmed.length < 400) { + if (containsDoneClaimPattern(text, doneClaimPatterns) && w.doneClaimAttempts < maxRetries) { + // Ask once for the work report. v2 used to gate this on a 400-char + // length, which cannot tell "Task done." from a real summary and so + // both over-nudged terse reports and let short-but-real ones pass. + // containsWorkDescription is the same structural test v1 uses, and + // prompting again after a real report loops forever (#26). + if (!containsWorkDescription(text)) { await targetedRecovery(sid, "done-claim-no-details", doneWithoutDetailsPrompt, "doneClaimAttempts") + } else { + dbg(`${short(sid)} done-claim carries a work description — skipping details prompt`) } } } From f93dfa127fc89e029d7ffc2d5ac5f728e57589fd Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 15:17:14 -0400 Subject: [PATCH 16/57] feat(v2): read the token window and route saturated sessions like v1 does MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A session can fill its context window without ever looking stalled — it just keeps working until it chokes. v1 caught that on idle by comparing the token count against the model's usable window. The v2 port declared the option inert because there was nothing to compare against; there was, it just needed reading. Tokens: session.usage.updated carries the session's current usage. tokenTotalOf sums input + output + reasoning + cache.read + cache.write, which is exactly TokenUsage.total in the schema and exactly what v1 accumulated. The window: ctx.model.get(providerID, modelID) hands over Model.Info.limit directly, where v1 had to walk the raw provider list. Usable is limit.context - Math.min(20_000, limit.output) — v1's arithmetic, unchanged, so the same threshold means the same thing on both builds. Cached per model. Routing is v1's. A subagent, with subagentNativeCompactionEnabled, requests native compaction: v1 called session.summarize(), v2 spells it session.compact, and it is not on the plugin session Pick, so it goes through the same defensive lookup that carries discovery. A parent goes to magic-context's ctx-wrapup command, but only when magic-context is installed — its setup disables native compaction, so compacting here would double-compress. v1 detected that with config.get().plugin; v2 removed the config domain, so it now reads ctx.plugin.list() and matches both the plugin id and its source spec, which is what makes it work for a local path and a package target alike. Both paths are one-shot per turn and fail-safe: no limit, no token count, a user cancellation, or a signalled completion means no intervention. contextSaturationThreshold and subagentNativeCompactionEnabled are live, so they leave the accepted-but-inert list (5 remain, down from 8). 16 new tests with a control per group, covering both routings, both compaction transports, the package-spec and local-path magic-context shapes, the one-shot budget, the user interrupt guard, the reserve arithmetic, cache and reasoning in the total, the threshold as a fraction, and the missing-model and missing-inventory fallbacks. 672 pass, 0 fail. --- README.md | 5 +- docs/known-issues-v2.md | 35 +++- src/v2/index.saturation.test.ts | 350 ++++++++++++++++++++++++++++++++ src/v2/index.ts | 273 +++++++++++++++++++++++-- 4 files changed, 644 insertions(+), 19 deletions(-) create mode 100644 src/v2/index.saturation.test.ts diff --git a/README.md b/README.md index a82a74a..349b812 100644 --- a/README.md +++ b/README.md @@ -462,14 +462,13 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `readyToContinuePatterns` | `READY_TO_CONTINUE_PATTERNS` | Array of regex strings overriding the default ready-to-continue detection patterns (case-insensitive). Invalid regexes are skipped. Empty array falls back to defaults. | | `silentDeadStreamMinTokens` | `200` | Min output tokens to treat a textless `finish:"unknown"` message as a dead stream | | `busyStallStrategy` | `"continue"` | Busy-stall response: `"continue"`, `"abort"` (abort-first), or `"off"` (disabled) | -| `contextSaturationThreshold` | `0.85` | Ratio of used/usable context that routes a saturated parent to magic-context `ctx-wrapup` (only when magic-context is installed) | +| `contextSaturationThreshold` | `0.85` | Ratio of used/usable context that routes a saturated session to reclamation: a parent to magic-context `ctx-wrapup` (only when magic-context is installed), a subagent to native compaction (only when `subagentNativeCompactionEnabled`) | | `activeUserWindowMs` | `300000` on v2, `900000` on v1 | Inbound-user-message recency window during which idle nudges stand down (user likely composing). v1 is stale at 15 min on this branch; upstream `e1b8374` already moved it to 5 min and the v2 build matches that | | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | -Accepted but **not applied** on v2: `contextSaturationThreshold`, -`subagentNativeCompactionEnabled`, `silentDeadStreamMinTokens`, `subagentWaitMs`, +Accepted but **not applied** on v2: `silentDeadStreamMinTokens`, `subagentWaitMs`, `toolTextCheckDelayMs`, `thinkingToolRecoveryPrompt`, `doneWithoutWorkPrompt`. See [docs/known-issues-v2.md](docs/known-issues-v2.md) for why. diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index bb4b42c..5ff0ce5 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -12,9 +12,7 @@ config keeps loading unchanged. | Option | Why it is inert on v2 | | --- | --- | -| `contextSaturationThreshold` | v2 has no token or context-limit read on the plugin context, so there is nothing to compare the threshold against. | -| `subagentNativeCompactionEnabled` | Same: without a context-limit read, saturation cannot be detected, so there is nothing to gate a compaction on. | -| `silentDeadStreamMinTokens` | Same. The heuristic it tunes compares generated token count against a floor. | +| `silentDeadStreamMinTokens` | The heuristic it tunes compares generated token count against a floor. That count is a single assistant message's output, not the session's cumulative usage, and v2 exposes no per-message stream for it. | | `subagentWaitMs` | The v1 orphan-watch timer that this delays has no v2 counterpart; the v2 port decides parent-vs-stalled from its own event-derived busy set. | | `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | | `thinkingToolRecoveryPrompt` | The thinking-contains-a-tool-call detector is part of the v1 idle-nudge pass and is not ported. | @@ -133,3 +131,34 @@ construction. Two detectors run on the idle path instead, both ported from v1: v1 cross-checks both against tracked todo state before latching. v2 has no todo state yet, so the emoji latches on its own — see `doneWithoutWorkPrompt` in the inert table above for what that costs. + +## Context saturation + +A session can fill its context window without ever looking stalled — it just keeps +working until it chokes. On idle, when used/usable crosses +`contextSaturationThreshold` (default 0.85), routing depends on session kind, +exactly as in v1: + +- **Subagent** — opt-in only. With `subagentNativeCompactionEnabled: true` the + plugin requests native compaction. v1 called `session.summarize()`; v2 spells it + `session.compact`. +- **Parent** — only when magic-context is installed, because its setup disables + native compaction and compacting here would double-compress. The plugin sends + the `ctx-wrapup` command through `session.command`, not as prompt text, since + prompt text is not expanded into a command. + +Both are one-shot per turn, and both are fail-safe: a missing limit, a missing +token count, a user cancellation, or a signalled completion means no intervention +at all. + +Three v2 API shapes are worth recording, because each replaced something v1 had: + +| Need | v1 | v2 | +|---|---|---| +| Tokens in the window | accumulated from `message.updated` | `session.usage.updated`, summing `input + output + reasoning + cache.read + cache.write` — the same five fields v1 added up, matching `TokenUsage.total` | +| The model's window | walk the raw provider list for `limit` | `ctx.model.get(providerID, modelID)` → `Model.Info.limit` | +| Is magic-context installed? | `config.get().plugin` | `ctx.plugin.list()` → `Plugin.Info[]`; v2 removed the `config` domain | + +The usable window is `limit.context - Math.min(20_000, limit.output)` — v1's +arithmetic, kept identical so the same threshold means the same thing on both +builds. diff --git a/src/v2/index.saturation.test.ts b/src/v2/index.saturation.test.ts new file mode 100644 index 0000000..5df393f --- /dev/null +++ b/src/v2/index.saturation.test.ts @@ -0,0 +1,350 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_saturation" +const PROVIDER = "llama-server" +const MODEL = "coder" + +/** + * Context saturation. + * + * A session can fill its context window without ever looking stalled — it just + * keeps working until it chokes. That failure mode is invisible to the stall + * watchdog by construction, so it gets its own check on the idle path. + * + * Routing, as in v1: + * subagent — opt-in (`subagentNativeCompactionEnabled`) then native compaction. + * magic-context is not involved. + * parent — only when magic-context is installed, because its setup disables + * native compaction and compacting here would double-compress. + * + * The v2 API shape is what these tests are really pinning: the token count comes + * from the `session.usage.updated` event, the window from `ctx.model.get()`, the + * magic-context verdict from `ctx.plugin.list()` (v2 removed the `config` domain + * v1 used), and compaction from `session.compact` — which is not on the plugin's + * narrowed `session` Pick, so it goes through the client fallback. + * + * Every group carries a control. A test that passes because the saturation branch + * never ran is worse than no test here. + */ + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const userTurn = () => ({ + type: "user", + id: "msg_u0", + // Old enough that activeUserWindowMs cannot stand the nudge down. + time: { created: Date.now() - 60 * 60_000 }, + content: [{ type: "text", text: "keep going" }], +}) + +/** `TokenUsage.Info` for the given totals; `cache` is always present, as in the schema. */ +const usage = (input: number, output = 0, reasoning = 0, read = 0, write = 0) => ({ + input, + output, + reasoning, + cache: { read, write }, +}) + +type HarnessOpts = { + /** Tokens reported by `session.usage.updated`. Omit for "no usage event". */ + tokens?: Record + /** `Model.Info.limit`. */ + limit?: { context: number; output: number } + parentID?: string + /** Installed plugins as `ctx.plugin.list()` returns them. */ + plugins?: unknown[] + /** Host exposes `session.compact` on the plugin domain. */ + domainCompact?: boolean + /** Host exposes `session.compact` only via the client. */ + clientCompact?: boolean + /** A genuine user interrupt lands before the idle. */ + interruptFirst?: boolean + opts?: Record +} + +async function replay(h: HarnessOpts) { + const commands: Array<{ sessionID: string; name: string }> = [] + const compacted: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-saturation-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const limit = h.limit ?? { context: 200_000, output: 32_000 } + const model = { id: MODEL, providerID: PROVIDER, limit, name: MODEL } + + const session: Record = { + context: async () => [userTurn(), { type: "assistant", id: "msg_a1", model: { providerID: PROVIDER, id: MODEL } }], + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async () => ({}), + prompt: async () => ({}), + command: async (a: any) => { + commands.push({ sessionID: a?.sessionID, name: a?.name }) + return {} + }, + } + if (h.domainCompact) session.compact = async (a: any) => (compacted.push(a?.sessionID), {}) + if (!h.domainCompact) session.compact = undefined + + const client: Record = { + session: { + get: async () => ({ data: h.parentID ? { parentID: h.parentID } : {} }), + }, + } + if (h.clientCompact) (client.session as any).compact = async (a: any) => compacted.push(a?.sessionID) + + const ctx: any = { + event: stream, + options: { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + injectIntervalMs: 0, + logFile, + ...(h.opts ?? {}), + }, + session, + client, + model: { get: (providerID: string, modelID: string) => (providerID === PROVIDER && modelID === MODEL ? model : undefined) }, + plugin: { list: async () => h.plugins ?? [] }, + } + + const cleanup = await (plugin as any).setup(ctx) + + const events: any[] = [ + ev("session.execution.started", { agent: "build", model: { providerID: PROVIDER, id: MODEL } }), + ev("session.text.delta", { messageID: "msg_a1", delta: "Working through the list." }), + ev("session.text.ended", { messageID: "msg_a1" }), + ] + if (h.tokens) events.splice(1, 0, ev("session.usage.updated", { tokens: usage(h.tokens.input, h.tokens.output, h.tokens.reasoning, h.tokens.read, h.tokens.write) })) + if (h.interruptFirst) events.push(ev("session.execution.interrupted", { reason: "user" })) + events.push(ev("session.idle")) + + for (const e of events) { + stream.push(e) + await wait(10) + } + await wait(600) + ;(cleanup as (() => void) | undefined)?.() + + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { commands, compacted, logs } +} + +/** A parent with magic-context installed and a window well over the threshold. */ +const SATURATED_PARENT: HarnessOpts = { + tokens: { input: 160_000 }, + plugins: [{ id: "magic-context", source: { type: "local", path: "/x/magic-context" } }], +} +/** Same, but only a small slice of the window used. */ +const HEALTHY_PARENT: HarnessOpts = { ...SATURATED_PARENT, tokens: { input: 10_000 } } + +describe("v2: context saturation — parent sessions", () => { + test("CONTROL: a saturated parent with magic-context installed gets the wrapup command", async () => { + const { commands, logs } = await replay(SATURATED_PARENT) + expect(commands).toHaveLength(1) + // Sent as a command, not as prompt text: prompt text is not expanded into + // a command, which was v1's documented reason for using the endpoint. + expect(commands[0]).toEqual({ sessionID: SID, name: "ctx-wrapup" }) + expect(logs.some((l) => l.includes("context saturation:"))).toBe(true) + }) + + test("a parent below the threshold is left alone", async () => { + const { commands, compacted, logs } = await replay(HEALTHY_PARENT) + expect(commands).toEqual([]) + expect(compacted).toEqual([]) + expect(logs.filter((l) => l.includes("context saturation"))).toEqual([]) + }) + + test("no usage event means no token count, so no intervention", async () => { + const { commands } = await replay({ plugins: SATURATED_PARENT.plugins }) + expect(commands).toEqual([]) + }) + + test("a saturated parent with no magic-context is NOT compacted — that would double-compress", async () => { + const { commands, compacted } = await replay({ tokens: { input: 160_000 }, plugins: [] }) + expect(commands).toEqual([]) + expect(compacted).toEqual([]) + }) + + test("magic-context is detected from a package spec, not just an id", async () => { + // v1 matched `config.get().plugin` entries, which are package specs. v2 has + // no config domain, so the source has to be matched too. + const { commands } = await replay({ + tokens: { input: 160_000 }, + plugins: [{ id: "pkg-1", source: { type: "package", target: "@someone/opencode-magic-context@1.2.3" } }], + }) + expect(commands).toHaveLength(1) + }) + + test("a host with no plugin inventory treats magic-context as absent", async () => { + const logs: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-saturation-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + const ctx: any = { + event: stream, + options: { warmupMs: 0, checkIntervalMs: 20, chunkTimeoutMs: 600_000, gracePeriodMs: 0, logFile }, + session: { context: async () => [], active: async () => ({}), interrupt: async () => ({}), synthetic: async () => ({}) }, + model: { get: () => ({ limit: { context: 200_000, output: 32_000 } }) }, + } + const cleanup = await (plugin as any).setup(ctx) + stream.push(ev("session.step.started", { model: { providerID: PROVIDER, id: MODEL } })) + stream.push(ev("session.usage.updated", { tokens: usage(160_000) })) + stream.push(ev("session.idle")) + await wait(400) + ;(cleanup as (() => void) | undefined)?.() + if (existsSync(logFile)) logs.push(...readFileSync(logFile, "utf8").split("\n")) + rmSync(logFile, { force: true }) + // Must not raise, and must not invent a verdict. + expect(logs.filter((l) => l.includes("ERROR"))).toEqual([]) + }) + + test("the intervention is one-shot: a second idle does not repeat it", async () => { + const { commands } = await replay(SATURATED_PARENT) + expect(commands).toHaveLength(1) + }) + + test("a genuine user interrupt suppresses the intervention entirely", async () => { + // The guard reads the `userCancelled` latch, which only a non-self abort sets. + // Assert the negative branch: the same saturated parent, interrupted first. + const { commands, compacted, logs } = await replay({ ...SATURATED_PARENT, interruptFirst: true }) + expect(commands).toEqual([]) + expect(compacted).toEqual([]) + expect(logs.filter((l) => l.includes("context saturation"))).toEqual([]) + }) +}) + +describe("v2: context saturation — subagent sessions", () => { + const SATURATED_SUBAGENT: HarnessOpts = { ...SATURATED_PARENT, parentID: "ses_parent" } + + test("CONTROL: a saturated subagent with the opt-in enabled triggers native compaction", async () => { + const { compacted, commands, logs } = await replay({ + ...SATURATED_SUBAGENT, + domainCompact: true, + opts: { subagentNativeCompactionEnabled: true }, + }) + expect(compacted).toEqual([SID]) + // magic-context must not be involved on this path even when installed. + expect(commands).toEqual([]) + expect(logs.some((l) => l.includes("context saturation (subagent)"))).toBe(true) + }) + + test("a saturated subagent without the opt-in is left alone", async () => { + const { compacted, commands } = await replay({ ...SATURATED_SUBAGENT, domainCompact: true }) + expect(compacted).toEqual([]) + expect(commands).toEqual([]) + }) + + test("compaction reaches the host through the client fallback when the domain omits it", async () => { + // `session.compact` is NOT in v2's plugin `session` Pick. The same defensive + // lookup that carries discovery must carry this. + const { compacted } = await replay({ + ...SATURATED_SUBAGENT, + clientCompact: true, + opts: { subagentNativeCompactionEnabled: true }, + }) + expect(compacted).toEqual([SID]) + }) + + test("a host exposing neither compaction path warns instead of throwing", async () => { + const { logs } = await replay({ ...SATURATED_SUBAGENT, opts: { subagentNativeCompactionEnabled: true } }) + expect(logs.some((l) => l.includes("native compaction is not exposed on this host"))).toBe(true) + expect(logs.filter((l) => l.includes("ERROR"))).toEqual([]) + }) +}) + +describe("v2: the saturation arithmetic matches v1", () => { + test("the output reserve is min(20000, model output limit)", async () => { + // 200k context, 32k output -> usable 180k. 170k used is below 0.85*180k=153k? + // No: 170k > 153k, so it fires. Drop to 150k and it must not. + const fired = await replay({ ...SATURATED_PARENT, tokens: { input: 170_000 } }) + expect(fired.commands).toHaveLength(1) + + const quiet = await replay({ ...SATURATED_PARENT, tokens: { input: 150_000 } }) + expect(quiet.commands).toEqual([]) + }) + + test("cache read and reasoning count toward the total", async () => { + // 140k input alone is below the threshold; the cache read pushes it over. + // `TokenUsage.total` includes both, and so does this. + const without = await replay({ ...SATURATED_PARENT, tokens: { input: 140_000 } }) + expect(without.commands).toEqual([]) + + const withCache = await replay({ ...SATURATED_PARENT, tokens: { input: 140_000, read: 60_000 } }) + expect(withCache.commands).toHaveLength(1) + }) + + test("contextSaturationThreshold is honoured as a fraction, not a percentage", async () => { + const at99 = await replay({ ...SATURATED_PARENT, tokens: { input: 178_000 }, opts: { contextSaturationThreshold: 0.99 } }) + expect(at99.commands).toEqual([]) + + const at90 = await replay({ ...SATURATED_PARENT, tokens: { input: 178_000 }, opts: { contextSaturationThreshold: 0.9 } }) + expect(at90.commands).toHaveLength(1) + }) + + test("an unknown model yields no limit, so no intervention", async () => { + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-saturation-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + const ctx: any = { + event: stream, + options: { warmupMs: 0, checkIntervalMs: 20, chunkTimeoutMs: 600_000, gracePeriodMs: 0, logFile }, + session: { context: async () => [], active: async () => ({}), interrupt: async () => ({}), synthetic: async () => ({}) }, + model: { get: () => undefined }, + plugin: { list: async () => [{ id: "magic-context" }] }, + } + const cleanup = await (plugin as any).setup(ctx) + stream.push(ev("session.step.started", { model: { providerID: PROVIDER, id: "not-installed" } })) + stream.push(ev("session.usage.updated", { tokens: usage(999_000) })) + stream.push(ev("session.idle")) + await wait(400) + ;(cleanup as (() => void) | undefined)?.() + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8") : "" + rmSync(logFile, { force: true }) + expect(logs).not.toContain("context saturation") + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index c21ab9e..5858fed 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -77,7 +77,20 @@ interface AutoResumePluginInput { * type-checks standalone. Optional: a host may not supply it, and every * use must degrade gracefully when it is absent. */ - client?: { session: { get: (opts: { path: { id: string } }) => Promise<{ data?: { parentID?: string } }> } } + client?: { session: Record any> } + /** + * Model catalogue. v2's `ModelDomain.get(providerID, modelID)` returns + * `Model.Info`, whose `limit` is `{ context, input?, output }` — the usable + * context window for context-saturation checks. Optional: a host may not + * supply it, and the saturation check degrades to "no intervention". + */ + model?: { get: (providerID: string, modelID: string) => unknown } + /** + * Installed-plugin inventory (`ctx.plugin.list()`). v2 removed the `config` + * domain, so this is how a plugin detects that magic-context is present before + * routing a saturated parent to it. + */ + plugin?: { list: () => Promise } /** Application logger, when the host provides one. */ app?: { log?: (level: string, message: string) => unknown } } @@ -135,6 +148,10 @@ interface SessionWatch { /** Set when the model's own completion signal was seen (a trailing 🎉). * Latches so a finished session stops being nudged. */ completionSignaled: boolean + /** Tokens currently in the context window, from `session.usage.updated`. */ + lastTokenTotal: number + /** Saturation intervention is one-shot per turn, like v1's `contextWrapupAttempts`. */ + contextWrapupAttempts: number /** Set when a failure-triggered recovery is pending, so the delayed prompt isn't cancelled by the failure's own idle transition. */ pendingRecoveryArmed: boolean @@ -159,6 +176,8 @@ interface SessionWatch { // Agent/model from last step (informational logging only; sessions are stateful in v2) agent?: string model?: string + /** `Model.Ref` from the last step, for the usable-context lookup. */ + modelRef?: { providerID: string; modelID: string } } export interface AutoResumeOptions { @@ -271,6 +290,13 @@ function appendLogFile(target: string, level: string, line: string): void { } } +/** + * The command magic-context registers to reclaim context on a saturated parent. + * Sent through `session.command` rather than as prompt text, because prompt text is + * not expanded into a command. + */ +const CTX_WRAPUP_TRIGGER = "ctx-wrapup" + // --------------------------------------------------------------------------- // Constants & defaults // --------------------------------------------------------------------------- @@ -742,6 +768,9 @@ export default define({ // Live since the v2 discovery sweep landed (see discoverSessions). const discoveryDelayMs = opts.discoveryDelayMs ?? DEFAULT_DISCOVERY_DELAY_MS const minActivityGapMs = opts.minActivityGapMs ?? DEFAULT_MIN_ACTIVITY_GAP_MS + // Live since the v2 token read landed (see tokenTotalOf / checkContextSaturation). + const contextSaturationThreshold = opts.contextSaturationThreshold ?? DEFAULT_CONTEXT_SATURATION_THRESHOLD + const subagentNativeCompactionEnabled = opts.subagentNativeCompactionEnabled ?? false const streamingFailureErrorNames = opts.streamingFailureErrorNames ?? DEFAULT_STREAMING_FAILURE_ERROR_NAMES const streamingFailureMessagePatterns = opts.streamingFailureMessagePatterns ?? DEFAULT_STREAMING_FAILURE_MESSAGE_PATTERNS const doneWithoutDetailsPrompt = opts.doneWithoutDetailsPrompt ?? DONE_WITHOUT_DETAILS_PROMPT @@ -782,8 +811,6 @@ export default define({ // and the gap is documented rather than surprising. See // docs/known-issues-v2.md. const FEATURE_GATED_OPTIONS = [ - "contextSaturationThreshold", - "subagentNativeCompactionEnabled", "silentDeadStreamMinTokens", "subagentWaitMs", "toolTextCheckDelayMs", @@ -932,6 +959,8 @@ export default define({ doneClaimAttempts: 0, intentNudgeAttempts: 0, completionSignaled: false, + lastTokenTotal: 0, + contextWrapupAttempts: 0, pendingRecoveryArmed: false, permissionPending: false, permissionPendingAt: null, @@ -1581,6 +1610,13 @@ async function inspectOnIdle(sid: string) { dbg(`${short(sid)} done-claim carries a work description — skipping details prompt`) } } + + // A full context window is a separate failure mode from a stalled one: the + // session keeps working happily until it chokes. Judged last, because it is + // the only check here that acts on the session rather than the text, and a + // saturated session that also produced a terse done-claim wants the + // reclamation, not the details prompt. + await checkContextSaturation(sid) } // --------------------------------------------------------------------- @@ -1711,21 +1747,100 @@ async function inspectOnIdle(sid: string) { }, discoveryDelayMs) // --------------------------------------------------------------------- - // Session discovery + // Context saturation // --------------------------------------------------------------------- + /** + * v2 emits `session.usage.updated` with the session's current usage, so the + * token count is read rather than reconstructed. `TokenUsage.total` in the + * schema sums input + output + reasoning + cache read + cache write; v1 + * accumulated the same way. + */ + function tokenTotalOf(tokens: unknown): number { + if (!tokens || typeof tokens !== "object") return 0 + const t = tokens as { + input?: number + output?: number + reasoning?: number + cache?: { read?: number; write?: number } + } + const cache = t.cache ?? {} + return posNum(t.input) + posNum(t.output) + posNum(t.reasoning) + posNum(cache.read) + posNum(cache.write) + } + function posNum(v: unknown): number { + return typeof v === "number" && Number.isFinite(v) && v > 0 ? v : 0 + } + + /** + * The usable context window for a session's model: the full window minus the + * smaller of a 20k output reserve and the model's own output limit. This is + * v1's arithmetic (`limit.context - Math.min(20_000, limit.output)`), kept + * identical so the same threshold means the same thing on both builds. + * + * v2 hands the limit over directly on `ctx.model.get()` — `Model.Info.limit` + * is `{ context, input?, output }` — where v1 had to walk the raw provider + * list. Cached per model, because the answer never changes at runtime. + */ + const usableLimitCache = new Map() + async function getUsableContextLimit(sid: string): Promise { + try { + let ref = sessions.get(sid)?.modelRef + if (!ref) { + // No step observed yet (plugin loaded mid-session): take the ref off + // the newest assistant message instead. + const msgs = await ctx.session.context({ sessionID: sid }) + if (Array.isArray(msgs)) { + for (let i = msgs.length - 1; i >= 0; i--) { + const m = msgs[i] as { + type?: string + model?: { providerID?: string; id?: string; modelID?: string } + } + if (m?.type !== "assistant" || !m.model) continue + const providerID = m.model.providerID + const modelID = m.model.modelID ?? m.model.id + if (typeof providerID === "string" && typeof modelID === "string") { + ref = { providerID, modelID } + const w = sessions.get(sid) + if (w) w.modelRef = ref + } + break + } + } + } + if (!ref) return null + const key = `${ref.providerID}/${ref.modelID}` + const cached = usableLimitCache.get(key) + if (cached !== undefined) return cached + + const info = (ctx.model as any)?.get?.(ref.providerID, ref.modelID) + if (!info) return null + const limit = (info as { limit?: { context?: number; output?: number } }).limit + if (!limit || typeof limit.context !== "number" || limit.context === 0) return null + const usable = limit.context - Math.min(20_000, limit.output ?? 0) + if (!Number.isFinite(usable) || usable <= 0) return null + usableLimitCache.set(key, usable) + return usable + } catch (e) { + dbg(`usable-context-limit lookup failed for ${short(sid)}:`, e instanceof Error ? e.message : String(e)) + return null + } + } + /** * v2 hands plugins a narrowed `ctx.session` domain (see - * `packages/plugin/src/promise/session.ts`) that omits `list` and - * `active`, so both are read defensively: off the session domain first, + * `packages/plugin/src/promise/session.ts`) that omits `list`, `active` and + * `compact`, so those are read defensively: off the session domain first, * then off the raw client. A host that supplies neither degrades to the * event-derived busy set rather than failing. + * + * `args` matters for the client's methods, which take a request object + * (`{ sessionID }`), unlike the plugin-domain wrappers. */ - async function callSessionApi(name: string): Promise { + async function callSessionApi(name: string, args?: Record): Promise { const fromDomain = (ctx.session as any)[name] if (typeof fromDomain === "function") { try { - return (await fromDomain.call(ctx.session)) as T + return (await fromDomain.call(ctx.session, args)) as T } catch (e) { dbg(`ctx.session.${name}() failed:`, e instanceof Error ? e.message : String(e)) } @@ -1733,7 +1848,7 @@ async function inspectOnIdle(sid: string) { const fromClient = ctx.client?.session?.[name as "get"] if (typeof fromClient === "function") { try { - return (await (fromClient as any).call(ctx.client!.session)) as T + return (await (fromClient as any).call(ctx.client!.session, args)) as T } catch (e) { dbg(`ctx.client.session.${name}() failed:`, e instanceof Error ? e.message : String(e)) } @@ -1741,6 +1856,42 @@ async function inspectOnIdle(sid: string) { return undefined } + /** + * Is magic-context installed? + * + * v1 read `config.get().plugin`. v2 has no `config` domain on the plugin + * context, but `ctx.plugin.list()` carries the same information in the + * server's own words: `Plugin.Info[]` with an `id` plus a `source` that is a + * package spec, a local path, or an SDK module. Matching id *and* source is + * what makes this work regardless of how the plugin was installed. + * + * Cached, negative verdict included — an idle check must not re-list plugins. + */ + let magicContextDetected: boolean | null = null + async function isMagicContextInstalled(): Promise { + if (magicContextDetected !== null) return magicContextDetected + try { + const list = (ctx.plugin as any)?.list + if (typeof list !== "function") { + dbg("magic-context detection: ctx.plugin.list unavailable, treating as not installed") + return false + } + const plugins = unwrapList(await list.call(ctx.plugin)) as Array<{ id?: string; source?: unknown }> + magicContextDetected = plugins.some((p) => { + const id = typeof p?.id === "string" ? p.id : "" + if (id.toLowerCase().includes("magic-context")) return true + const src = p?.source as { target?: string; path?: string } | undefined + const spec = src?.target ?? src?.path ?? "" + return typeof spec === "string" && spec.toLowerCase().includes("magic-context") + }) + dbg(`magic-context detection: ${magicContextDetected ? "installed" : "not installed"}`) + return magicContextDetected + } catch (e) { + dbg("magic-context detection failed, treating as not installed:", e instanceof Error ? e.message : String(e)) + return false + } + } + /** Unwrap the `{ data }` envelope the client uses, or pass an array through. */ function unwrapList(response: unknown): Array> { if (Array.isArray(response)) return response as Array> @@ -1751,6 +1902,78 @@ async function inspectOnIdle(sid: string) { return [] } + /** + * v1's saturation routing, ported to the v2 API. + * + * A session can fill its window without stalling — it just keeps working + * until it chokes. On idle, when used/usable crosses the threshold, routing + * depends on session kind: + * + * subagent — opt-in only (`subagentNativeCompactionEnabled`). v1 called + * `session.summarize()`; v2 spells it `session.compact`, which + * is NOT in the plugin `session` Pick, so it goes through the same + * defensive lookup as `list`/`active`. + * parent — only when magic-context is installed, because its setup + * disables native compaction and compacting here would + * double-compress. v1 sent the `ctx-wrapup` command through the + * client; `session.command` IS on the plugin domain in v2. + * + * Fail-safe, as in v1: a missing limit, a missing token count, a user + * cancellation, or a signalled completion means no intervention at all. + * The intervention is one-shot per turn. + */ + async function checkContextSaturation(sid: string): Promise { + const w = sessions.get(sid) + if (!w) return + if (w.lastTokenTotal <= 0) return + if (w.contextWrapupAttempts >= 1) return + if (w.userCancelled || w.completionSignaled || w.aborting) return + + const usable = await getUsableContextLimit(sid) + if (!usable) return + if (w.lastTokenTotal / usable < contextSaturationThreshold) return + + const pct = Math.round((w.lastTokenTotal / usable) * 100) + if (await isSubAgentSession(sid)) { + if (!subagentNativeCompactionEnabled) { + dbg(`${short(sid)} subagent at ${pct}% of usable context; native compaction is opt-in`) + return + } + w.contextWrapupAttempts++ + log( + "warn", + `${short(sid)} context saturation (subagent): ${w.lastTokenTotal}/${usable} tokens (${pct}% of usable); triggering native compaction`, + ) + // Re-check the latches: the awaits above left a window for the user + // to cancel in. + if (w.userCancelled || w.completionSignaled) return + const compacted = await callSessionApi("compact", { sessionID: sid }) + if (compacted === undefined) { + log("warn", `${short(sid)} native compaction is not exposed on this host — skipping`) + } + return + } + + if (!(await isMagicContextInstalled())) { + dbg(`${short(sid)} at ${pct}% of usable context; magic-context not installed, not intervening`) + return + } + w.contextWrapupAttempts++ + log( + "warn", + `${short(sid)} context saturation: ${w.lastTokenTotal}/${usable} tokens (${pct}% of usable); sending magic-context wrapup command`, + ) + if (w.userCancelled || w.completionSignaled) return + try { + await (ctx.session as any).command({ sessionID: sid, name: CTX_WRAPUP_TRIGGER }) + } catch (e) { + log( + "warn", + `${short(sid)} magic-context wrapup command failed: ${e instanceof Error ? e.message : String(e)}`, + ) + } + } + /** * Seed watch state for sessions that already exist, and mark the ones * currently running as busy. @@ -1829,6 +2052,21 @@ async function inspectOnIdle(sid: string) { markBusy(sid) return } + case "session.usage.updated": { + // The session's current context usage. `tokenTotalOf` mirrors + // `TokenUsage.total` in the schema (input + output + reasoning + + // cache read + cache write) so the ratio means the same thing here + // as in v1, which accumulated the same five fields. + const sid = sidOf(ev) + if (!sid) return + const total = tokenTotalOf(ev.data?.tokens) + if (total <= 0) return + const w = ensureWatch(sid) + w.lastTokenTotal = total + dbg(`${short(sid)} usage: ${total} tokens`) + return + } + case "session.execution.succeeded": case "session.idle": { const sid = sidOf(ev) @@ -1940,10 +2178,19 @@ async function inspectOnIdle(sid: string) { if (!sid) return const w = ensureWatch(sid) w.agent = typeof ev.data?.agent === "string" ? ev.data.agent : w.agent - w.model = - ev.data?.model && typeof ev.data.model === "object" - ? `${ev.data.model.providerID ?? ev.data.model.provider ?? "?"}/${ev.data.model.modelID ?? ev.data.model.id ?? "?"}` - : w.model + { + const m = ev.data?.model as { providerID?: string; provider?: string; modelID?: string; id?: string } | undefined + const providerID = m?.providerID ?? m?.provider + const modelID = m?.modelID ?? m?.id + if (typeof providerID === "string" && typeof modelID === "string") { + w.modelRef = { providerID, modelID } + w.model = `${providerID}/${modelID}` + } else if (m) { + w.model = `${providerID ?? "?"}/${modelID ?? "?"}` + } else { + w.model = w.model + } + } w.lastWasTaskTool = false markBusy(sid) return From d301bd2c09b88617ac0c39ef836bcea68b320638 Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 15:27:45 -0400 Subject: [PATCH 17/57] feat(v2): catch the stream that finished without saying anything MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A turn can end having produced nothing the user can see — reasoning only, or a finish reason the provider did not describe. v2 records the message as completed and the session goes idle, so no stall timer expires, no streaming failure fires, and the stall watchdog never sees it, because it only looks at busy sessions. v1 had a detector for exactly this; the v2 port declared the option inert, on the mistaken grounds that the token count was unavailable. It is available. The judgment is on the message, not the event stream: walk back to the newest assistant message that HAS a finish reason. If it carried text, the session answered and there is nothing to recover. If it did not, and it generated at least silentDeadStreamMinTokens output tokens, the stream died mid-response. The walk skips messages with no finish reason on purpose — an intermediate tool-call step has none, and stopping there would report a dead stream for every session that used a tool. Two v2 specifics worth the record. The judge needs session.context(), every message since the last compaction, so the idle inspection now fetches it once and shares it across four checks — this port was fetching it three times. And before injecting the plugin asks the server whether the session is running again, not only its own event-derived flag: a provider quietly retrying looks identical from the event stream, and the recovery event may not have arrived by the time the turn ends. That is the last of the token-based detectors, so accepted-but-inert is down to four: subagentWaitMs, toolTextCheckDelayMs, thinkingToolRecoveryPrompt, doneWithoutWorkPrompt. 11 new tests with a control per group, including a control for the recovering-provider guard. 683 pass, 0 fail. --- README.md | 6 +- docs/known-issues-v2.md | 31 +++- src/v2/index.dead-stream.test.ts | 269 +++++++++++++++++++++++++++++++ src/v2/index.ts | 151 +++++++++++++---- 4 files changed, 426 insertions(+), 31 deletions(-) create mode 100644 src/v2/index.dead-stream.test.ts diff --git a/README.md b/README.md index 349b812..d069549 100644 --- a/README.md +++ b/README.md @@ -468,9 +468,9 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | -Accepted but **not applied** on v2: `silentDeadStreamMinTokens`, `subagentWaitMs`, -`toolTextCheckDelayMs`, `thinkingToolRecoveryPrompt`, -`doneWithoutWorkPrompt`. See [docs/known-issues-v2.md](docs/known-issues-v2.md) for why. +Accepted but **not applied** on v2: `subagentWaitMs`, `toolTextCheckDelayMs`, +`thinkingToolRecoveryPrompt`, `doneWithoutWorkPrompt`. See +[docs/known-issues-v2.md](docs/known-issues-v2.md) for why. Message patterns are matched case-insensitively. Error names use exact match. diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 5ff0ce5..e7162e6 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -12,7 +12,6 @@ config keeps loading unchanged. | Option | Why it is inert on v2 | | --- | --- | -| `silentDeadStreamMinTokens` | The heuristic it tunes compares generated token count against a floor. That count is a single assistant message's output, not the session's cumulative usage, and v2 exposes no per-message stream for it. | | `subagentWaitMs` | The v1 orphan-watch timer that this delays has no v2 counterpart; the v2 port decides parent-vs-stalled from its own event-derived busy set. | | `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | | `thinkingToolRecoveryPrompt` | The thinking-contains-a-tool-call detector is part of the v1 idle-nudge pass and is not ported. | @@ -162,3 +161,33 @@ Three v2 API shapes are worth recording, because each replaced something v1 had: The usable window is `limit.context - Math.min(20_000, limit.output)` — v1's arithmetic, kept identical so the same threshold means the same thing on both builds. + +## Silent dead stream + +A turn can end having produced nothing the user can see: reasoning only, or a +finish reason the provider did not describe. OpenCode records the message as +completed and the session goes idle, so no stall timer expires, no streaming +failure fires, and the stall watchdog — which only looks at *busy* sessions — +never sees it. v1's rule is unchanged on v2: walk back to the newest assistant +message that **has** a finish reason; if it carried no text and generated at least +`silentDeadStreamMinTokens` output tokens, the stream died mid-response. + +The walk skips messages with no finish reason on purpose. An intermediate +tool-call step has none, and stopping at it would report a dead stream for every +session that used a tool. + +Two things are worth recording about the v2 API: + +- The judge is the message, not the event stream, so this needs + `session.context()` — every message since the last compaction. The idle + inspection now fetches it **once** and shares it across four checks (dead + stream, text fallback, pending tool call, active user), where v1 fetched once + for the same reason and this port initially fetched three times. +- Before injecting, the plugin asks the server whether the session is running + again (`session.active()`), not only its own event-derived flag. A provider + quietly retrying looks identical from the event stream, and the recovery event + may not have arrived by the time the turn ends. + +`thinkingToolRecoveryPrompt` is the sibling detector that is still inert: it +covers a *thinking* block containing a raw tool call, which v1 judged from the +message parts on the same pass. diff --git a/src/v2/index.dead-stream.test.ts b/src/v2/index.dead-stream.test.ts new file mode 100644 index 0000000..b3218e8 --- /dev/null +++ b/src/v2/index.dead-stream.test.ts @@ -0,0 +1,269 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_deadstream" + +/** + * Silent dead stream. + * + * The model can end a turn having produced nothing the user can see: reasoning + * only, or a finish reason the provider did not describe. The turn looks + * complete, so no stall timer expires and no streaming failure fires — the + * session just goes quiet with the work unfinished. Nothing in v1's event + * stream catches it either; it takes reading the finished message itself. + * + * The rule, from v1: the newest assistant message that HAS a finish reason + * decides. If it carried text, the session answered and there is nothing to + * recover. If it did not, and it generated at least `silentDeadStreamMinTokens` + * output tokens, the stream died mid-response. + * + * The walk deliberately skips messages with no finish reason. An intermediate + * tool-call step has none, and stopping at it would report a dead stream for + * every session that used a tool — which is why "walk back past a delivered + * answer to an intermediate step" matters. + * + * Every group carries a control. + */ + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const userTurn = () => ({ + type: "user", + id: "msg_u0", + time: { created: Date.now() - 60 * 60_000 }, + content: [{ type: "text", text: "do the thing" }], +}) + +/** An assistant message that finished with a `finish` reason and `output` tokens. */ +const finished = (opts: { finish?: string; text?: string; output?: number; reasoning?: string }) => ({ + type: "assistant", + id: "msg_a1", + time: { created: Date.now() - 60_000 }, + ...(opts.text !== undefined + ? { content: [{ type: "text", text: opts.text }] } + : { content: [{ type: "reasoning", text: opts.reasoning ?? "thinking about the answer" }] }), + finish: opts.finish ?? "stop", + tokens: { input: 100, output: opts.output ?? 400, reasoning: 0, cache: { read: 0, write: 0 } }, +}) + +/** An intermediate tool-call step: no finish reason, so the walk skips it. */ +const toolStep = () => ({ + type: "assistant", + id: "msg_a2", + time: { created: Date.now() - 30_000 }, + content: [{ type: "tool", id: "call_1", name: "bash", state: { status: "completed" } }], + tokens: { input: 200, output: 30, reasoning: 0, cache: { read: 0, write: 0 } }, +}) + +const OPTIONS = { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + // Several assertions here are about a silent skip, which is exactly what the + // debug log is for — and in v2 that log goes to the file, not the console. + debug: true, +} + +type Harness = { injected: Array<{ text?: string }>; logs: string[] } + +async function replay( + messages: unknown[], + opts: Record = {}, + extra: { serverRunning?: boolean } = {}, +): Promise { + const injected: Harness["injected"] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-deadstream-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile, ...opts }, + session: { + context: async () => [userTurn(), ...messages], + // The server's own record of what is running, which is what the + // recovering-provider guard consults. + active: async () => (extra.serverRunning ? { [SID]: { sessionID: SID } } : {}), + interrupt: async () => ({}), + synthetic: async (a: any) => { + injected.push({ text: a?.text }) + return {} + }, + prompt: async (a: any) => { + injected.push({ text: a?.text }) + return {} + }, + }, + client: { session: { get: async () => ({ data: {} }) } }, + } + + const cleanup = await (plugin as any).setup(ctx) + + for (const e of [ + ev("session.execution.started"), + ev("session.step.started"), + ev("session.step.ended"), + ev("session.idle"), + ]) { + stream.push(e) + await wait(10) + } + await wait(700) + ;(cleanup as (() => void) | undefined)?.() + + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { injected, logs } +} + +/** Reasoning-only finish with plenty of output: the real dead stream. */ +const DEAD = [finished({ output: 400 })] + +describe("v2: silent dead stream", () => { + test("CONTROL: a finished message with no text and enough output tokens resumes", async () => { + const { injected, logs } = await replay(DEAD) + expect(injected).toHaveLength(1) + expect(logs.some((l) => l.includes("silent dead stream: finish=stop, 400 output tokens"))).toBe(true) + expect(logs.some((l) => l.includes("Silent dead stream (stop)"))).toBe(true) + }) + + test("a finished message that delivered text is a normal completion", async () => { + const { injected, logs } = await replay([finished({ text: "The change is in src/index.ts and tests pass.", output: 400 })]) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("silent dead stream"))).toBe(false) + }) + + test("below the token floor it is a short answer, not a dead stream", async () => { + // 40 output tokens with no text: the model simply had nothing to say. The + // floor exists so a one-word turn is not nudged forever. + const { injected, logs } = await replay([finished({ output: 40 })]) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("only 40 output tokens (floor 200)"))).toBe(true) + }) + + test("the floor is configurable", async () => { + const quiet = await replay([finished({ output: 40 })], { silentDeadStreamMinTokens: 200 }) + expect(quiet.injected).toEqual([]) + const loud = await replay([finished({ output: 40 })], { silentDeadStreamMinTokens: 10 }) + expect(loud.injected).toHaveLength(1) + }) + + test("the walk skips a tool-call step and judges the answer behind it", async () => { + // Newest is an intermediate step with no finish reason. The finished + // message behind it has text, so this session just used a tool. + const { injected } = await replay([finished({ text: "Ran the tests; 42 pass.", output: 400 }), toolStep()]) + expect(injected).toEqual([]) + }) + + test("a tool-call step in front of a dead stream is still found", async () => { + // The other direction: skipping the unfinished step must not hide the + // dead finish behind it. + const { injected } = await replay([finished({ output: 400 }), toolStep()]) + expect(injected).toHaveLength(1) + }) + + test("a session the server still reports as running is left alone", async () => { + // A provider that is quietly retrying looks identical from the event + // stream. Injecting into that would turn a recovering session into a + // stalled one, which is the harm the guard prevents — and the event may + // not have arrived yet, so the guard asks the server. + const { injected, logs } = await replay(DEAD, {}, { serverRunning: true }) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("running again"))).toBe(true) + }) + + test("CONTROL for that guard: the same dead stream on an idle session does resume", async () => { + // Otherwise the guard could pass because the detector stopped working. + const { injected } = await replay(DEAD, {}, { serverRunning: false }) + expect(injected).toHaveLength(1) + }) + + test("no finished message at all is not a dead stream", async () => { + // Mid-first-turn: nothing has finished yet, so there is nothing to judge. + const { injected } = await replay([toolStep()]) + expect(injected).toEqual([]) + }) + + test("a user-cancelled session is never resumed", async () => { + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-deadstream-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + const injected: unknown[] = [] + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async () => [userTurn(), finished({ output: 400 })], + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async () => ({}), + prompt: async () => (injected.push(1), {}), + }, + client: { session: { get: async () => ({ data: {} }) } }, + } + const cleanup = await (plugin as any).setup(ctx) + for (const e of [ + ev("session.execution.started"), + ev("session.execution.interrupted", { reason: "user" }), + ev("session.idle"), + ]) { + stream.push(e) + await wait(10) + } + await wait(600) + ;(cleanup as (() => void) | undefined)?.() + rmSync(logFile, { force: true }) + expect(injected).toEqual([]) + }) + + test("the recovery budget is finite", async () => { + // Two dead turns in a row must not queue unbounded injections. + const { injected } = await replay([finished({ output: 400 })], { maxRetries: 2 }) + expect(injected.length).toBeGreaterThan(0) + expect(injected.length).toBeLessThanOrEqual(2) + }) +}) \ No newline at end of file diff --git a/src/v2/index.ts b/src/v2/index.ts index 5858fed..ddadd67 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -771,6 +771,8 @@ export default define({ // Live since the v2 token read landed (see tokenTotalOf / checkContextSaturation). const contextSaturationThreshold = opts.contextSaturationThreshold ?? DEFAULT_CONTEXT_SATURATION_THRESHOLD const subagentNativeCompactionEnabled = opts.subagentNativeCompactionEnabled ?? false + // Live since the silent-dead-stream detector landed (lastSilentDeadStream). + const silentDeadStreamMinTokens = opts.silentDeadStreamMinTokens ?? DEFAULT_SILENT_DEAD_STREAM_MIN_TOKENS const streamingFailureErrorNames = opts.streamingFailureErrorNames ?? DEFAULT_STREAMING_FAILURE_ERROR_NAMES const streamingFailureMessagePatterns = opts.streamingFailureMessagePatterns ?? DEFAULT_STREAMING_FAILURE_MESSAGE_PATTERNS const doneWithoutDetailsPrompt = opts.doneWithoutDetailsPrompt ?? DONE_WITHOUT_DETAILS_PROMPT @@ -811,7 +813,6 @@ export default define({ // and the gap is documented rather than surprising. See // docs/known-issues-v2.md. const FEATURE_GATED_OPTIONS = [ - "silentDeadStreamMinTokens", "subagentWaitMs", "toolTextCheckDelayMs", "thinkingToolRecoveryPrompt", @@ -1465,37 +1466,126 @@ export default define({ // --------------------------------------------------------------------- /** - * Read the last assistant message's text from the session message history - * (`ctx.session.context()`, stable v2 API). Returns "" when unavailable. - * Guarded so a failure in this forensic path never breaks the watchdog. + * Fetch a session's message history once per idle inspection. + * + * Four separate checks below want the same array — the dead-stream + * detector, the text fallback, the pending-tool and active-user lookups — + * and `ctx.session.context()` is the expensive call in all of them (it is + * every message since the last compaction). v1 fetched once for the same + * reason. The cache lives only for the duration of one inspection, so a + * later idle always sees the freshest history. + * + * Returns `[]` on any failure: this is forensic, and every caller already + * treats "no history" as "cannot conclude". */ - async function lastAssistantTextFromContext(sid: string): Promise { + async function loadMessages(sid: string): Promise { try { - const messages = await ctx.session.context({ sessionID: sid }) - if (!Array.isArray(messages)) return "" - for (let i = messages.length - 1; i >= 0; i--) { - const msg = messages[i] as { - type?: string - content?: Array<{ type?: string; text?: string }> - } - if (!msg || msg.type !== "assistant" || !Array.isArray(msg.content)) continue - const text = msg.content - .filter((part) => part?.type === "text" && typeof part.text === "string") - .map((part) => part.text as string) - .join("") - if (text) return text - } - return "" + const res = await ctx.session.context({ sessionID: sid }) + return Array.isArray(res) ? res : ((res as { messages?: unknown[] })?.messages ?? []) } catch (e) { - dbg("session.context() fallback failed:", e instanceof Error ? e.message : String(e)) - return "" + dbg(`${short(sid)} session.context() failed:`, e instanceof Error ? e.message : String(e)) + return [] } } - async function shouldStandDownForUser(sid: string, activeUserWindowMs: number): Promise { + /** The newest assistant message that delivered text, joined. "" when none. */ + function lastAssistantTextFrom(messages: unknown[]): string { + for (let i = messages.length - 1; i >= 0; i--) { + const msg = messages[i] as { + type?: string + content?: Array<{ type?: string; text?: string }> + } + if (!msg || msg.type !== "assistant" || !Array.isArray(msg.content)) continue + const text = msg.content + .filter((part) => part?.type === "text" && typeof part.text === "string") + .map((part) => part.text as string) + .join("") + if (text) return text + } + return "" + } + + /** + * v1's `getLastSilentDeadStream`, on the v2 message shape. + * + * The model can finish a turn having produced no text at all — reasoning + * only, or a `finish=unknown` that the provider did not describe. The + * turn looks complete, so nothing raises, and the session simply stops. + * A `finish` with real text behind it means the session answered normally + * and there is nothing to recover. + * + * Only the newest assistant message that *has* a finish is judged, and the + * walk skips messages without one — an intermediate tool-call step has no + * finish, and walking back past a delivered answer to one of those would + * recover a session that just used a tool. + * + * Returns `null` when the newest finished message carried text, or when + * there is no finished message at all. + */ + function lastSilentDeadStream(messages: unknown[]): { finish: string; outputTokens: number } | null { + for (let i = messages.length - 1; i >= 0; i--) { + const msg = messages[i] as { + type?: string + content?: Array<{ type?: string; text?: string }> + finish?: string + tokens?: { output?: number } + } + if (msg?.type !== "assistant") continue + const finish = typeof msg.finish === "string" ? msg.finish : undefined + if (!finish) continue + const hasText = Array.isArray(msg.content) && + msg.content.some((p) => p?.type === "text" && typeof p.text === "string" && p.text.length > 0) + if (hasText) return null + return { finish, outputTokens: posNum(msg.tokens?.output) } + } + return null + } + + /** + * Recover a stream that finished without ever delivering text. + * + * Ported from v1 including its guard: re-check the server's own status + * first, because a provider that is quietly retrying looks identical from + * the event stream, and interrupting that would turn a recovering session + * into a stalled one. Then arm the pending-recovery latch — without it + * `recover()`'s inject-time check reads an idle session as "recovered by + * itself" and drops the injection. + * + * Returns true when it took action, so the caller stops looking. + */ + async function recoverSilentDeadStream(sid: string, messages: unknown[]): Promise { + const dead = lastSilentDeadStream(messages) + if (!dead) return false + if (dead.outputTokens < silentDeadStreamMinTokens) { + dbg( + `${short(sid)} last finished message has no text, but only ${dead.outputTokens} output tokens (floor ${silentDeadStreamMinTokens}) — not treating it as a dead stream`, + ) + return false + } + const w = ensureWatch(sid) + // Ask the server before injecting, not just our own flag. A provider + // that is quietly retrying looks exactly like a dead stream from the + // event stream, and the event may not have arrived yet when the turn + // ends — which is v1's own reason for re-checking live status here. + if (w.status === "busy" || (await getActiveSessions()).includes(sid)) { + dbg(`${short(sid)} silent dead stream, but the session is running again — likely a provider retry`) + return true + } + w.pendingRecoveryArmed = true + log( + "info", + `${short(sid)} silent dead stream: finish=${dead.finish}, ${dead.outputTokens} output tokens, no text parts; resuming`, + ) + await recover(sid, `Silent dead stream (${dead.finish})`) + return true + } + + async function shouldStandDownForUser( + sid: string, + messages: unknown[], + activeUserWindowMs: number, + ): Promise { try { - const result = await ctx.session.context({ sessionID: sid }) - const messages: unknown[] = Array.isArray(result) ? result : ((result as { messages?: unknown[] })?.messages ?? []) if (messages.length === 0) return false const newest = messages[messages.length - 1] as { type?: string @@ -1552,10 +1642,17 @@ export default define({ async function inspectOnIdle(sid: string) { const w = ensureWatch(sid) + const messages = await loadMessages(sid) + + // Judged first, and before anything that needs text: a stream that died + // before delivering any text is precisely the case where every + // text-based check below would find nothing to look at. + if (await recoverSilentDeadStream(sid, messages)) return + // Prefer the live delta buffer; fall back to the authoritative message // history when it is empty (e.g. the plugin loaded mid-turn) or stale. let text = w.lastAssistantText - if (!text) text = await lastAssistantTextFromContext(sid) + if (!text) text = lastAssistantTextFrom(messages) if (!text) return // A turn that ends by handing control back to the user (a question or an @@ -1567,7 +1664,7 @@ async function inspectOnIdle(sid: string) { return } - if (await shouldStandDownForUser(sid, activeUserWindowMs)) { + if (await shouldStandDownForUser(sid, messages, activeUserWindowMs)) { dbg(`${short(sid)} user has pending input or was recently active — standing down`) return } From 887ae947a9904554a7985a9b256333f05fa88fee Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 15:36:10 -0400 Subject: [PATCH 18/57] feat(v2): ask for the tool call when the model writes one in its reasoning MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A model that writes raw tool-call markup inside its thinking produces no tool call: nothing is tagged as one, so no session-side code runs it. The turn completes, the prose reads fine, and the work silently does not happen. v1 already caught this on the same pass as the text variant, and answered with a different prompt — the fix is different, because the model is not forgetting the tool mechanism, it is writing in the wrong channel. v2 had joined reasoning and text into one string, so the two were indistinguishable and only the text one was judged. AssistantContent tags reasoning and text distinctly, which makes this two reads instead of one filter. Three decisions worth naming: Reasoning is judged first. A message can carry both, and the reasoning one is the one that silently does nothing; reporting the text one would send the model off to fix the wrong thing. The budget is shared with the text variant. One phenomenon, two symptoms — two budgets would let the plugin spend twice the retries on a model that keeps doing it. The same code-block stripping applies. A fenced example of the call format, or an inline path in backticks, is not a real call. thinkingToolRecoveryPrompt is live, so accepted-but-inert is down to three: subagentWaitMs, toolTextCheckDelayMs, doneWithoutWorkPrompt. 7 new tests, with a control per behaviour including the two-channel precedence. 690 pass, 0 fail. One thing the tests surfaced and the fixtures had to respect: a message that finished with no text part at all is a silent dead stream, and that detector fires first. Reasoning-only fixtures need a normal text part or they exercise the wrong branch. --- README.md | 2 +- docs/known-issues-v2.md | 24 +++- src/v2/index.thinking-tool.test.ts | 213 +++++++++++++++++++++++++++++ src/v2/index.ts | 38 ++++- 4 files changed, 271 insertions(+), 6 deletions(-) create mode 100644 src/v2/index.thinking-tool.test.ts diff --git a/README.md b/README.md index d069549..334eb0f 100644 --- a/README.md +++ b/README.md @@ -469,7 +469,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | Accepted but **not applied** on v2: `subagentWaitMs`, `toolTextCheckDelayMs`, -`thinkingToolRecoveryPrompt`, `doneWithoutWorkPrompt`. See +`doneWithoutWorkPrompt`. See [docs/known-issues-v2.md](docs/known-issues-v2.md) for why. Message patterns are matched case-insensitively. Error names use exact match. diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index e7162e6..910c120 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -14,7 +14,6 @@ config keeps loading unchanged. | --- | --- | | `subagentWaitMs` | The v1 orphan-watch timer that this delays has no v2 counterpart; the v2 port decides parent-vs-stalled from its own event-derived busy set. | | `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | -| `thinkingToolRecoveryPrompt` | The thinking-contains-a-tool-call detector is part of the v1 idle-nudge pass and is not ported. | | `doneWithoutWorkPrompt` | Both v1 use sites for this prompt are gated on tracked todo state. v2 has no todo state, so the prompt has no trigger. | `doneWithoutDetailsPrompt` **is** applied on v2, and is the replacement for @@ -188,6 +187,23 @@ Two things are worth recording about the v2 API: quietly retrying looks identical from the event stream, and the recovery event may not have arrived by the time the turn ends. -`thinkingToolRecoveryPrompt` is the sibling detector that is still inert: it -covers a *thinking* block containing a raw tool call, which v1 judged from the -message parts on the same pass. +## A tool call written into the reasoning block + +When a model writes raw tool-call markup inside its thinking instead of calling +the tool, nothing executes and nothing raises. No part is tagged as a tool call, +so no session-side code runs it; the turn completes normally, the prose may read +fine, and the work silently does not happen. + +v1 caught this on the same pass as the text variant and answered with a different +prompt, because the fix is different — the model is not forgetting the tool +mechanism, it is writing in the wrong channel. v2 separates the two reads instead +of filtering them into one joined string, since `AssistantContent` tags reasoning +and text distinctly: + +- reasoning parts are judged first, because a message can carry both and the + reasoning one is the one that silently does nothing; +- both variants share the `toolTextAttempts` budget, because they are one + phenomenon with two symptoms — two budgets would let the plugin spend twice the + retries on a model that keeps doing it; +- both use the same code-block stripping, so a fenced example of the format or an + inline path in backticks is not mistaken for a real call. diff --git a/src/v2/index.thinking-tool.test.ts b/src/v2/index.thinking-tool.test.ts new file mode 100644 index 0000000..0c57de8 --- /dev/null +++ b/src/v2/index.thinking-tool.test.ts @@ -0,0 +1,213 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_thinking_tool" + +/** + * A tool call written into the reasoning block. + * + * When a model emits raw tool-call markup inside its thinking instead of calling + * the tool, nothing executes and nothing raises: the turn completes, the text the + * user sees may be fine, and the work silently does not happen. The text variant + * of this failure has its own detector and its own prompt; the reasoning variant + * needs a different one, because the model is not forgetting the mechanism, it is + * writing in the wrong channel. + * + * v2 makes this a separate read rather than a filter inside a joined string: + * `AssistantContent` tags reasoning and text distinctly, so the newest message's + * reasoning parts can be judged on their own. + * + * Every group carries a control. + */ + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const userTurn = () => ({ + type: "user", + id: "msg_u0", + time: { created: Date.now() - 60 * 60_000 }, + content: [{ type: "text", text: "fix the failing test" }], +}) + +const assistantTurn = (parts: Array>) => ({ + type: "assistant", + id: "msg_a1", + time: { created: Date.now() - 60_000 }, + content: parts, + finish: "stop", + tokens: { input: 100, output: 200, reasoning: 0, cache: { read: 0, write: 0 } }, +}) + +/** + * Every fixture carries a normal text part. + * + * Without one the message is a *silent dead stream* — finished, no text, plenty + * of output tokens — and that detector fires first with its own "continue". That + * is correct behaviour and it has its own tests; here it would just mask which + * branch is under test. + */ + +/** A raw tool call sitting in the reasoning, with normal prose alongside. */ +const REASONING_WITH_TOOL_CALL = [ + { type: "text", text: "Let me start by reading the failing test." }, + { + type: "reasoning", + text: "I should read the test file first.\n\nsrc/a.test.ts\n", + }, +] +/** Ordinary reasoning, no markup. */ +const REASONING_ONLY = [ + { type: "text", text: "Checking which assertion is wrong." }, + { type: "reasoning", text: "Let me think about which assertion is wrong here." }, +] +/** The same markup, but in the text part where the text detector already sees it. */ +const TEXT_WITH_TOOL_CALL = [{ type: "text", text: "src/a.test.ts" }] + +const OPTIONS = { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + debug: true, +} + +async function replay( + parts: Array>, + opts: Record = {}, +): Promise<{ injected: Array<{ text?: string }>; logs: string[] }> { + const injected: Array<{ text?: string }> = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-thinktool-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile, ...opts }, + session: { + context: async () => [userTurn(), assistantTurn(parts)], + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ text: a?.text }), {}), + }, + client: { session: { get: async () => ({ data: {} }) } }, + } + + const cleanup = await (plugin as any).setup(ctx) + for (const e of [ + ev("session.execution.started"), + ev("session.step.started"), + ev("session.step.ended"), + ev("session.idle"), + ]) { + stream.push(e) + await wait(10) + } + await wait(700) + ;(cleanup as (() => void) | undefined)?.() + + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { injected, logs } +} + +describe("v2: a tool call written into the reasoning block", () => { + test("CONTROL: markup in the reasoning asks the model to use the tool mechanism", async () => { + const { injected } = await replay(REASONING_WITH_TOOL_CALL) + expect(injected).toHaveLength(1) + // The reasoning prompt is specific to this failure: the tool call is + // well-formed, it was just written in the wrong channel. + expect(injected[0].text).toContain("thinking/reasoning") + }) + + test("CONTROL: the same markup in a text part asks for the tool mechanism generally", async () => { + const { injected } = await replay(TEXT_WITH_TOOL_CALL) + expect(injected).toHaveLength(1) + expect(injected[0].text).not.toContain("thinking/reasoning") + }) + + test("reasoning with no tool call in it is left alone", async () => { + const { injected, logs } = await replay(REASONING_ONLY) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("tool-call-in-reasoning"))).toBe(false) + }) + + test("the reasoning prompt is configurable", async () => { + const { injected } = await replay(REASONING_WITH_TOOL_CALL, { + thinkingToolRecoveryPrompt: "Use the real tool call, not reasoning.", + }) + expect(injected).toHaveLength(1) + expect(injected[0].text).toBe("Use the real tool call, not reasoning.") + }) + + test("code blocks in reasoning are not mistaken for a tool call", async () => { + // Same code-stripping the text detector uses: a fenced example of the + // shape, or an inline path in backticks, must not trigger a nudge. + const { injected } = await replay([ + { type: "text", text: "Checking how the tool-call format is documented." }, + { + type: "reasoning", + text: "The format is documented as:\n```\n\nx\n\n```\n", + }, + ]) + expect(injected).toEqual([]) + }) + + test("both channels in one message: the reasoning one is judged first", async () => { + // The reasoning variant silently does nothing, so reporting the text one + // would send the model off to fix the wrong thing. + const { injected } = await replay([...TEXT_WITH_TOOL_CALL, ...REASONING_WITH_TOOL_CALL]) + expect(injected).toHaveLength(1) + expect(injected[0].text).toContain("thinking/reasoning") + }) + + test("the budget is shared with the text variant, not doubled", async () => { + // One phenomenon with two symptoms. Two independent budgets would let the + // plugin spend twice the retries on a model that keeps doing this. + const { injected } = await replay(REASONING_WITH_TOOL_CALL, { maxRetries: 1 }) + expect(injected.length).toBeLessThanOrEqual(1) + }) +}) \ No newline at end of file diff --git a/src/v2/index.ts b/src/v2/index.ts index ddadd67..6da4012 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -815,7 +815,6 @@ export default define({ const FEATURE_GATED_OPTIONS = [ "subagentWaitMs", "toolTextCheckDelayMs", - "thinkingToolRecoveryPrompt", "doneWithoutWorkPrompt", ] as const @@ -1505,6 +1504,34 @@ export default define({ return "" } + /** + * The newest assistant message's reasoning text, joined. "" when none. + * + * A reasoning block is the one place a model writes a tool call that never + * becomes one: the raw markup lands in the reasoning and nothing executes. + * v1 caught that on the same pass as the text variant and answered with a + * different prompt, because the fix is different — the model is not + * forgetting the mechanism, it is writing in the wrong channel. + * + * v2 separates the two cleanly, since `AssistantContent` tags reasoning and + * text distinctly where v1 filtered them together into one string. + */ + function lastAssistantReasoning(messages: unknown[]): string { + for (let i = messages.length - 1; i >= 0; i--) { + const msg = messages[i] as { + type?: string + content?: Array<{ type?: string; text?: string }> + } + if (!msg || msg.type !== "assistant" || !Array.isArray(msg.content)) continue + const reasoning = msg.content + .filter((part) => part?.type === "reasoning" && typeof part.text === "string") + .map((part) => part.text as string) + .join("") + if (reasoning) return reasoning + } + return "" + } + /** * v1's `getLastSilentDeadStream`, on the v2 message shape. * @@ -1683,6 +1710,15 @@ async function inspectOnIdle(sid: string) { return } + // A raw tool call written into the reasoning block never executes: no + // part is tagged as a tool call, so nothing on the session side ever + // runs it. Judged before the text variant because a message can have + // both, and the reasoning one is the one that silently does nothing. + const reasoning = lastAssistantReasoning(messages) + if (reasoning && containsToolCallAsText(reasoning)) { + await targetedRecovery(sid, "tool-call-in-reasoning", thinkingToolRecoveryPrompt, "toolTextAttempts") + return + } if (containsToolCallAsText(text)) { await targetedRecovery(sid, "tool-call-as-text", opts.toolTextRecoveryPrompt ?? TOOL_TEXT_RECOVERY_PROMPT, "toolTextAttempts") return From 16fdc92b2d59a8f0ae1d8fa1ad83d67e0a7b301b Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 16:19:56 -0400 Subject: [PATCH 19/57] feat(v2): read the todo list, and stop trusting the emoji on its own MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v1 tracks a session's todos from a todo.updated event, falling back to a server API. v2 has neither: the todo table is created in the v2 database (migration 20260127222353_familiar_lady_ursula.ts) but no route reaches it and nothing emits an event for it, so there is no way to observe the list changing. The port declared doneWithoutWorkPrompt inert on that basis. There is another way in. v2 has a storage domain, and the installed todo tool already writes the list there under a stable per-session key — storage.set("todos/", { todos, updatedAt }). auto-resume reads that key instead of owning a list of its own, so it is a consumer rather than a second owner: it works with whichever todo tool is installed, and there is no copy to drift. It only ever calls get — asserted by a test that hands the harness a set and a remove and checks neither is called. Three things consume it. A trailing 🎉 is no longer trusted on its own: a model that finishes early celebrates early, and latching on that turns a false positive into permanent silence, so with items still open the celebration is treated as a false positive and the reminder names what is unfinished. That branch deliberately does not latch, so the next turn is free to judge again. doneWithoutWorkPrompt now fires for a done-claim with open todos, on a budget separate from doneClaimAttempts — two different problems, so spending one must not silence the other. And with no storage domain, or a record that will not parse, it falls back to "no list", which means the older behaviour and never a nudge on the strength of a list it failed to read. One fix came along with it. The done-claim budgets no longer reset on every busy cycle — only on a genuinely new inbound user message. v1 moved both out of the busy reset for #26: a model that re-announces completion each turn was otherwise handed a fresh budget each time it announced, so the nudge never stopped. v2 still had that. A new user message is recognised by message id rather than by clock time, because comparing timestamps would read any drift between two reads of the history as a new request and reintroduce the same loop. accepted-but-inert is down to two: subagentWaitMs, toolTextCheckDelayMs. 14 new tests with a control per behaviour. 704 pass, 0 fail. --- README.md | 15 +- docs/known-issues-v2.md | 52 +++++- src/v2/index.todo.test.ts | 333 ++++++++++++++++++++++++++++++++++++++ src/v2/index.ts | 242 ++++++++++++++++++++++++--- 4 files changed, 614 insertions(+), 28 deletions(-) create mode 100644 src/v2/index.todo.test.ts diff --git a/README.md b/README.md index 334eb0f..9912ff7 100644 --- a/README.md +++ b/README.md @@ -231,6 +231,8 @@ Repeat calls with no new user message in between are guarded: the first acknowle An assistant message ending with 🎉 resets the tool-text timer and prevents a trigger — the emoji signals the agent considers the task complete. +The emoji alone is not trusted: a model that finishes early celebrates early, and latching on that turns a false positive into silence. Both builds cross-check the session's todo list first, and with items still open the 🎉 is treated as a false positive — the reminder names what is unfinished instead. See [The todo list](docs/known-issues-v2.md#the-todo-list) for how v2 reads that list. + --- ### Ready-to-continue auto-resume @@ -316,6 +318,14 @@ Paths: orphan parent, subagent stuck (parent side), hallucination loop, streamin ## Architecture +The diagram below is v1's: it is drawn against the v1 SSE event stream, the v1 +`todo.updated` event and the v1 `session.todo()` API. The v2 build reads the same +information through different doors — `session.usage.updated` for tokens, +`ctx.model.get()` for the window, `ctx.plugin.list()` to detect magic-context, +`ctx.session.context()` for message history, and `ctx.storage` for the todo list +(the v2 `todo` table has no route and emits no event). See +[docs/known-issues-v2.md](docs/known-issues-v2.md) for each substitution. + ``` Any SSE Event ├─ has sessionID? → touchSession(sid) — reset only that session's timer @@ -456,7 +466,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `actionIntentPrompt` | same as `continuePrompt` | Prompt sent on action-intent detection | | `toolTextRecoveryPrompt` | `TOOL_TEXT_RECOVERY_PROMPT` | Override the tool-call-as-text recovery prompt | | `thinkingToolRecoveryPrompt` | `THINKING_TOOL_RECOVERY_PROMPT` | Override the thinking-tool recovery prompt | -| `doneWithoutWorkPrompt` | `DONE_WITHOUT_WORK_PROMPT` | Override the done-claim-with-open-todos prompt | +| `doneWithoutWorkPrompt` | `DONE_WITHOUT_WORK_PROMPT` | Override the done-claim-with-open-todos prompt. Needs a todo tool writing `todos/` into storage; without one it never fires | | `doneWithoutDetailsPrompt` | `DONE_WITHOUT_DETAILS_PROMPT` | Override the done-claim-with-no-todos report prompt | | `doneClaimPatterns` | `DONE_CLAIM_PATTERNS` | Array of regex strings overriding the default done-claim detection patterns (case-insensitive, multiline). Invalid regexes are skipped. Empty array falls back to defaults. | | `readyToContinuePatterns` | `READY_TO_CONTINUE_PATTERNS` | Array of regex strings overriding the default ready-to-continue detection patterns (case-insensitive). Invalid regexes are skipped. Empty array falls back to defaults. | @@ -468,8 +478,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | -Accepted but **not applied** on v2: `subagentWaitMs`, `toolTextCheckDelayMs`, -`doneWithoutWorkPrompt`. See +Accepted but **not applied** on v2: `subagentWaitMs`, `toolTextCheckDelayMs`. See [docs/known-issues-v2.md](docs/known-issues-v2.md) for why. Message patterns are matched case-insensitively. Error names use exact match. diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 910c120..e0cff3b 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -14,10 +14,11 @@ config keeps loading unchanged. | --- | --- | | `subagentWaitMs` | The v1 orphan-watch timer that this delays has no v2 counterpart; the v2 port decides parent-vs-stalled from its own event-derived busy set. | | `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | -| `doneWithoutWorkPrompt` | Both v1 use sites for this prompt are gated on tracked todo state. v2 has no todo state, so the prompt has no trigger. | -`doneWithoutDetailsPrompt` **is** applied on v2, and is the replacement for -`doneWithoutWorkPrompt`: it fires on a terse done-claim regardless of todos. +`doneWithoutDetailsPrompt` and `doneWithoutWorkPrompt` **are** both applied on +v2, and they are the two halves of v1's done-claim handling: the first asks for a +work report when there is no list to check, the second when the list still has +items open. ## Unrecognised options warn once @@ -126,9 +127,48 @@ construction. Two detectors run on the idle path instead, both ported from v1: stop. The latch is per-turn: a new turn re-opens the question, and a later idle with no new text does not re-derive it forever. -v1 cross-checks both against tracked todo state before latching. v2 has no todo -state yet, so the emoji latches on its own — see `doneWithoutWorkPrompt` in the -inert table above for what that costs. +v1 cross-checks both against tracked todo state before latching. v2 now does too — +see "The todo list" below. + +## The todo list + +v1 tracked a session's todos from a `todo.updated` event, falling back to a server +API. v2 has neither: the `todo` table is created in the v2 database (migration +`20260127222353_familiar_lady_ursula.ts`) but no route reaches it and nothing +emits an event for it, so there is no way to observe the list changing. + +What v2 does have is the storage domain, and the installed todo tool already +writes the list there under a stable per-session key — +`ctx.storage.set("todos/", { todos, updatedAt })`. auto-resume reads +that key instead of owning a list of its own. + +The consequence worth stating: auto-resume is a **consumer**, not a second owner. +It works with whichever todo tool is installed rather than requiring its own, and +there is no copy to drift. It only ever calls `get`. With no `ctx.storage` at all, +or with a record it cannot parse, it falls back to "no list" — which means the +older behaviour (latch on the emoji, ask for details on a bare done-claim), never +a nudge on the strength of a list it failed to read. + +Three places consume it: + +- **The 🎉 cross-check.** The emoji alone is not trustworthy: a model that + finishes early celebrates early, and latching on that turns a false positive + into permanent silence. With items still open, the celebration is a false + positive and the reminder names what is unfinished. This branch deliberately + does **not** latch, so the next turn is free to judge again. +- **`doneWithoutWorkPrompt`.** A done-claim with open todos. Distinct from + `doneWithoutDetailsPrompt`, and on a separate budget — two different problems, + so spending one must not silence the other. +- **`todoCheckAttempts`.** The last v1 use of the list that is not here: a turn + that says "ready to continue" while every todo is already closed gets two + chances before a plain `continue`. + +One fix came with the list. The done-claim budgets no longer reset on every busy +cycle — only on a genuinely new inbound user message. v1 moved both out of the +busy reset for #26: a model that re-announces completion each turn was otherwise +handed a fresh budget each time it announced, so the nudge never stopped. v2 +still had that, and now does not. The open-todos nudge is the opposite case and +does reset per cycle, because an open list is new information each turn. ## Context saturation diff --git a/src/v2/index.todo.test.ts b/src/v2/index.todo.test.ts new file mode 100644 index 0000000..fa95416 --- /dev/null +++ b/src/v2/index.todo.test.ts @@ -0,0 +1,333 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_todo" + +/** + * The todo list as the plugin under test reads it. + * + * auto-resume does not own a todo list — it reads the one the installed todo tool + * writes, under the key `todos/`. So these tests supply that key and + * nothing else, which is the point: they would pass unchanged against a real + * `todowrite` tool, and they assert the plugin never writes there itself. + * + * Three things are under test: + * + * The celebration cross-check. A trailing 🎉 is the model's own "finished" + * signal, and latching on it is what stops the nudge. But a model that finishes + * early celebrates early, so the emoji alone is not trustworthy: with work still + * listed, the celebration is a false positive and the right answer is to name + * what is unfinished. + * + * Two done-claim prompts, two budgets. A done-claim with open todos and a + * done-claim with no detail report are different problems; spending one budget + * must not silence the other. + * + * When a budget re-arms. #26 was an unbounded done-claim nudge. The fix is that + * it re-arms only on a genuinely new work cycle — an inbound user message — + * not on every turn the model announces completion again. + * + * Every group carries a control. + */ + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const userMessage = (text: string, at: number) => ({ + type: "user", + id: `msg_u_${at}`, + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantMessage = (text: string, at: number) => ({ + type: "assistant", + id: `msg_a_${at}`, + time: { created: at }, + content: [{ type: "text", text }], + finish: "stop", + tokens: { input: 100, output: 200, reasoning: 0, cache: { read: 0, write: 0 } }, +}) + +const OPEN = [ + { content: "Write the migration guide", status: "pending", priority: "high" }, + { content: "Delete the temp fixtures", status: "in_progress", priority: "low" }, +] +const CLOSED = [ + { content: "Write the migration guide", status: "completed", priority: "high" }, + { content: "Delete the temp fixtures", status: "cancelled", priority: "low" }, +] + +/** An hour ago: outside the 5-minute active-user window, so idle nudges do not stand down. */ +const OLD = Date.now() - 60 * 60_000 +/** Ten minutes ago: still outside that window, but newer than OLD. */ +const RECENT_BUT_STALE = Date.now() - 10 * 60_000 + +const OPTIONS = { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + debug: true, +} + +type Harness = { + injected: Array<{ text?: string }> + logs: string[] + storageWrites: string[] + storageReads: string[] +} + +async function replay( + turns: Array<{ text: string; userMessages?: unknown[] }>, + todos: unknown[] | undefined, + opts: Record = {}, + storageShape: { omitStorage?: boolean } = {}, +): Promise { + const injected: Harness["injected"] = [] + const storageWrites: string[] = [] + const storageReads: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-todo-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + // The history grows turn by turn, exactly as a real session's does. + let history: unknown[] = [] + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile, ...opts }, + session: { + context: async () => history, + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ text: a?.text }), {}), + }, + client: { session: { get: async () => ({ data: {} }) } }, + } + if (!storageShape.omitStorage) { + ctx.storage = { + get: async (key: string) => { + storageReads.push(key) + return { todos: todos ?? [], updatedAt: Date.now() } + }, + // Present so that "the plugin never writes here" is an assertion the + // harness can actually make, rather than an assumption. + set: async (key: string) => storageWrites.push(key), + remove: async (key: string) => storageWrites.push(key), + } + } + + const cleanup = await (plugin as any).setup(ctx) + for (const turn of turns) { + for (const m of turn.userMessages ?? []) history.push(m) + history.push(assistantMessage(turn.text, Date.now() - 30_000)) + for (const e of [ + ev("session.execution.started"), + ev("session.step.started"), + ev("session.step.ended"), + ev("session.idle"), + ]) { + stream.push(e) + await wait(10) + } + } + await wait(700) + ;(cleanup as (() => void) | undefined)?.() + + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { injected, logs, storageWrites, storageReads } +} + +/** A done-claim with nothing in it — the premature stop. */ +const BARE_DONE = "Task done." +/** The same, but closed out with the model's own finished signal. */ +const CELEBRATED = "Everything is in place. 🎉" + +describe("v2: the todo list, read from the tool that owns it", () => { + test("CONTROL: a celebration with open todos is a false positive, and is named", async () => { + const { injected, logs } = await replay([{ text: CELEBRATED }], OPEN) + expect(injected).toHaveLength(1) + expect(injected[0].text).toContain("Write the migration guide") + expect(injected[0].text).toContain("Delete the temp fixtures") + expect(logs.some((l) => l.includes("open-todos-celebration-false-positive"))).toBe(true) + }) + + test("CONTROL: a celebration with everything closed latches and never nudges", async () => { + const { injected, logs } = await replay([{ text: CELEBRATED }, { text: CELEBRATED }], CLOSED) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("no open todos — latching completion"))).toBe(true) + }) + + test("CONTROL: a bare done-claim with open todos gets the work prompt, not the details prompt", async () => { + const { injected } = await replay([{ text: BARE_DONE }], OPEN) + expect(injected).toHaveLength(1) + // v1's wording for this branch names the todo list. + expect(injected[0].text).toContain("todo list") + }) + + test("CONTROL: the same bare done-claim with no todos still gets the details prompt", async () => { + const { injected } = await replay([{ text: BARE_DONE }], []) + expect(injected).toHaveLength(1) + expect(injected[0].text).toContain("no work description") + expect(injected[0].text).not.toContain("todo list") + }) + + test("both done-claim prompts stay live independently", async () => { + // Two different problems. If the todo branch spent the same budget as the + // details branch, exhausting one would silently disable the other. + const withTodos = await replay([{ text: BARE_DONE }, { text: BARE_DONE }], OPEN, { maxRetries: 1 }) + const withoutTodos = await replay([{ text: BARE_DONE }, { text: BARE_DONE }], [], { maxRetries: 1 }) + expect(withTodos.injected).toHaveLength(1) + expect(withoutTodos.injected).toHaveLength(1) + expect(withTodos.injected[0].text).not.toBe(withoutTodos.injected[0].text) + }) + + test("the done-claim budget does not re-arm just because the model repeated itself", async () => { + // This is #26. A model that re-announces completion every turn used to be + // handed a fresh budget every time it announced, so the nudge never stopped. + const { injected } = await replay([{ text: BARE_DONE }, { text: BARE_DONE }, { text: BARE_DONE }], [], { + maxRetries: 1, + }) + expect(injected).toHaveLength(1) + }) + + test("a genuinely new user message does re-arm it", async () => { + // The other half of the fix: the budget must be recoverable, or one bad + // stretch would silence the plugin for the rest of the session. + const { injected } = await replay( + [ + { text: BARE_DONE, userMessages: [userMessage("first ask", OLD)] }, + { text: BARE_DONE }, + { text: BARE_DONE, userMessages: [userMessage("actually, also this", RECENT_BUT_STALE)] }, + ], + [], + { maxRetries: 1 }, + ) + expect(injected).toHaveLength(2) + }) + + test("a celebration with open todos keeps being caught on later turns", async () => { + // Proof that the false-positive branch does not latch: a latched turn + // would go silent here. + const { injected } = await replay([{ text: CELEBRATED }, { text: CELEBRATED }], OPEN, { maxRetries: 1 }) + expect(injected).toHaveLength(2) + }) + + test("the list is read under the todo tool's own key", async () => { + const { storageReads } = await replay([{ text: CELEBRATED }], OPEN) + expect(storageReads).toContain(`todos/${SID}`) + }) + + test("the plugin never writes to that key", async () => { + // It is a consumer, not a second owner. A write here would mean auto-resume + // was maintaining a copy that could drift from the real one. + const { storageWrites } = await replay([{ text: CELEBRATED }], OPEN) + expect(storageWrites).toEqual([]) + }) + + test("the list is read once per turn, not once per check", async () => { + // The storage call is cheap but the cache is the difference between one read + // and two on every idle where both a celebration and a done-claim are judged. + const { storageReads } = await replay([{ text: CELEBRATED }], OPEN) + expect(storageReads.filter((k) => k === `todos/${SID}`).length).toBeLessThanOrEqual(2) + }) + + test("no storage domain at all is not a crash", async () => { + // A host without ctx.storage must degrade to "cannot conclude", which means + // the old behaviour — latch on the emoji — rather than a thrown error. + const { injected, logs } = await replay([{ text: CELEBRATED }], OPEN, {}, { omitStorage: true }) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("latching completion"))).toBe(true) + }) + + test("a malformed record is treated as no list, not as no work", async () => { + // The dangerous direction is "list unreadable" → "nothing is open" → nudge. + // Both malformed shapes must fall back to the old behaviour instead. + for (const bad of [{ todos: "nope" }, { todos: [null, 7] }, null]) { + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-todo-bad-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + const injected: unknown[] = [] + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async () => [userMessage("ask", OLD), assistantMessage(CELEBRATED, Date.now())], + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push(a?.text), {}), + prompt: async (a: any) => (injected.push(a?.text), {}), + }, + client: { session: { get: async () => ({ data: {} }) } }, + storage: { get: async () => bad, set: async () => {}, remove: async () => {} }, + } + const cleanup = await (plugin as any).setup(ctx) + for (const e of [ + ev("session.execution.started"), + ev("session.step.started"), + ev("session.step.ended"), + ev("session.idle"), + ]) { + stream.push(e) + await wait(10) + } + await wait(600) + ;(cleanup as (() => void) | undefined)?.() + rmSync(logFile, { force: true }) + expect(injected).toEqual([]) + } + }) + + test("the work prompt is configurable", async () => { + const { injected } = await replay([{ text: BARE_DONE }], OPEN, { + doneWithoutWorkPrompt: "Your todo list still has open items.", + }) + expect(injected).toHaveLength(1) + expect(injected[0].text).toBe("Your todo list still has open items.") + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index 6da4012..a99556e 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -91,6 +91,12 @@ interface AutoResumePluginInput { * routing a saturated parent to it. */ plugin?: { list: () => Promise } + /** + * Key/value store scoped to the plugin (`ctx.storage.get/set/remove/scan`). + * Read-only here: auto-resume does not own a todo list, it reads the one the + * installed todo tool writes, so this only ever calls `get`. + */ + storage?: { get: (key: string) => Promise } /** Application logger, when the host provides one. */ app?: { log?: (level: string, message: string) => unknown } } @@ -144,10 +150,25 @@ interface SessionWatch { toolTextAttempts: number continueTimestamps: number[] doneClaimAttempts: number + /** Done-claim with todos still open. Separate from `doneClaimAttempts`: the two + * prompts ask for different things, so spending one budget must not silence the other. */ + doneClaimOpenTodosAttempts: number + /** Open-todos reminders (the celebration false positive). Persists across turns — + * an open list does not become finished by the model saying it is. */ + todoNudgeAttempts: number intentNudgeAttempts: number /** Set when the model's own completion signal was seen (a trailing 🎉). * Latches so a finished session stops being nudged. */ completionSignaled: boolean + /** The session's todo list, read from the todo tool's storage key. */ + todos: Todo[] + todosFetchedAt: number + /** Id of the newest inbound user message already acted on, so a replayed or + * re-delivered message does not re-arm the done-claim budgets. Identity, not + * clock time — see noteInboundUserMessage. */ + lastUserMessageID?: string + /** Timestamp fallback for the same check, for messages carrying no id. */ + lastUserMessageSeenAt: number /** Tokens currently in the context window, from `session.usage.updated`. */ lastTokenTotal: number /** Saturation intervention is one-shot per turn, like v1's `contextWrapupAttempts`. */ @@ -331,6 +352,8 @@ const DEFAULT_TOOL_TEXT_CHECK_DELAY_MS = 3_000 const DEFAULT_SUBAGENT_WAIT_MS = 15_000 const DEFAULT_SILENT_DEAD_STREAM_MIN_TOKENS = 200 const DEFAULT_CONTEXT_SATURATION_THRESHOLD = 0.85 +/** How long a fetched todo list is trusted before being re-read. */ +const TODO_CACHE_TTL_MS = 3_000 const DEFAULT_MAX_RECOVERY_RETRIES = 2 // Referenced from FEATURE_GATED_OPTIONS docs; kept so the intended v1 default is // recorded next to the gate that explains why it is not applied yet. @@ -612,16 +635,60 @@ function containsWorkDescription(text: string): boolean { /** * True when the last assistant turn closes with a celebration emoji. * - * A 🎉 is the model's own "I finished" signal. v1 uses it to latch completion - * rather than keep nudging — but only when no work is left, otherwise a - * premature 🎉 would be treated as a real finish. Ported without the todo - * cross-check, so on v2 it latches on the emoji alone. + * A 🎉 is the model's own "I finished" signal, used to latch completion rather + * than keep nudging. The emoji alone is not trustworthy: a model that finishes + * early celebrates early, and latching on that turns a false positive into + * silence. The todo cross-check that catches it lives at the call site, where + * the fetched list is in hand. */ function endsWithCelebration(text: string): boolean { const normalized = text.trim().replace(/[.!?]+$/, "") return normalized.endsWith("🎉") } +/** + * One entry of a session's todo list. + * + * Read from the todo tool that is already installed, not from a list this + * plugin keeps. The record shape is the one `todowrite` writes — + * `storage.set("todos/", {todos, updatedAt})` — so whatever todo + * tool the session uses is what this sees, and auto-resume never has to own a + * second copy that can drift. + */ +interface Todo { + content: string + status: "pending" | "in_progress" | "completed" | "cancelled" + priority: "high" | "medium" | "low" +} + +function isOpenTodo(t: Todo): boolean { + return t.status === "pending" || t.status === "in_progress" +} + +function getOpenTodos(todos: Todo[]): Todo[] { + if (!Array.isArray(todos)) return [] + return todos.filter(isOpenTodo) +} + +/** + * The prompt for a turn that claims to be finished with work still listed. + * + * Names the open items rather than saying "continue", because the model has + * already decided it is done — a bare continuation prompt gets answered with + * another done-claim. Falls back to a plain "continue" when the list is + * unusable, which is strictly better than sending an empty reminder. + */ +function buildOpenTodosReminder(todos: Todo[]): string { + if (!Array.isArray(todos)) return "continue" + const open = todos.filter(isOpenTodo) + if (open.length === 0) return "continue" + const list = open.map((t, i) => `${i + 1}. [${t.status}] ${t.content}`).join("\n") + const plural = open.length > 1 ? "s" : "" + const taskWord = open.length > 1 ? "tasks" : "task" + const thisWord = open.length > 1 ? "these" : "this" + return `You have ${open.length} unfinished task${plural}:\n${list}\n\nPlease continue working on ${thisWord} ${taskWord}.` +} + /** Model ends with ":" announcing intent without executing. */ function containsActionIntent(text: string): boolean { if (text.length <= 15) return false @@ -776,6 +843,9 @@ export default define({ const streamingFailureErrorNames = opts.streamingFailureErrorNames ?? DEFAULT_STREAMING_FAILURE_ERROR_NAMES const streamingFailureMessagePatterns = opts.streamingFailureMessagePatterns ?? DEFAULT_STREAMING_FAILURE_MESSAGE_PATTERNS const doneWithoutDetailsPrompt = opts.doneWithoutDetailsPrompt ?? DONE_WITHOUT_DETAILS_PROMPT + // Live since the todo read landed (see readTodos): a done-claim with items still + // listed gets this prompt instead of the details prompt. + const doneWithoutWorkPrompt = opts.doneWithoutWorkPrompt ?? DONE_WITHOUT_WORK_PROMPT const thinkingToolRecoveryPrompt = opts.thinkingToolRecoveryPrompt ?? THINKING_TOOL_RECOVERY_PROMPT /** Compile a user-supplied regex-source list, skipping anything invalid. */ @@ -815,7 +885,6 @@ export default define({ const FEATURE_GATED_OPTIONS = [ "subagentWaitMs", "toolTextCheckDelayMs", - "doneWithoutWorkPrompt", ] as const // Options this build understands. Anything else in the user's config is @@ -957,8 +1026,13 @@ export default define({ toolTextAttempts: 0, continueTimestamps: [], doneClaimAttempts: 0, + doneClaimOpenTodosAttempts: 0, + todoNudgeAttempts: 0, intentNudgeAttempts: 0, completionSignaled: false, + todos: [], + todosFetchedAt: 0, + lastUserMessageSeenAt: 0, lastTokenTotal: 0, contextWrapupAttempts: 0, pendingRecoveryArmed: false, @@ -994,7 +1068,16 @@ export default define({ // hallucination-loop detector counts across busy cycles by design. w.resumeAttempts = 0 w.toolTextAttempts = 0 - w.doneClaimAttempts = 0 + // Deliberately NOT resetting doneClaimAttempts / doneClaimOpenTodosAttempts. + // v1 moved both out of the busy-cycle reset for a reason (#26): a model + // that keeps re-announcing completion would get a fresh budget on every + // turn it announced, so the nudge never stopped. They re-arm only on a + // genuine new work cycle — an inbound user message, via + // noteInboundUserMessage(). + // The open-todos nudge is the opposite case and does reset here: an open + // list is new information each turn, not a model that keeps saying the + // same thing. + w.todoNudgeAttempts = 0 w.intentNudgeAttempts = 0 // A new turn re-opens the question of whether the work is finished. w.completionSignaled = false @@ -1409,7 +1492,53 @@ export default define({ } /** Targeted recovery prompts (tool-as-text, done-claims, intent nudges). */ - async function targetedRecovery(sid: string, kind: string, prompt: string, budgetKey: "toolTextAttempts" | "doneClaimAttempts" | "intentNudgeAttempts") { + /** + * Read the session's todo list. + * + * v1 tracked todos from a `todo.updated` event and a server API. v2 has + * neither: the todo table exists in the v2 database but no route reaches + * it and nothing emits an event for it. What v2 does have is the storage + * domain, and the installed todo tool already writes the list there under + * a stable, per-session key. + * + * So this reads the key rather than owning a list. That is the difference + * between a second copy that can drift and the real one — and it means + * auto-resume works with whichever todo tool is installed instead of + * requiring its own. + * + * Cached for a short TTL: this is read on every idle inspection, and the + * list can only change while the model is working, which is not when we + * ask. Returns `[]` on any failure — every caller treats "no list" as + * "cannot conclude", never as "nothing is open". + */ + async function readTodos(sid: string): Promise { + const w = ensureWatch(sid) + const now = Date.now() + if (w.todosFetchedAt && now - w.todosFetchedAt < TODO_CACHE_TTL_MS) return w.todos + if (!ctx.storage) return w.todos + try { + const raw = await ctx.storage.get(`todos/${sid}`) + const record = raw as { todos?: unknown; updatedAt?: unknown } | null | undefined + const list = Array.isArray(record?.todos) ? record.todos : [] + w.todos = list.filter( + (t): t is Todo => + !!t && typeof t === "object" && typeof (t as Todo).content === "string" && + typeof (t as Todo).status === "string", + ) + w.todosFetchedAt = now + return w.todos + } catch (e) { + dbg(`${short(sid)} todo read failed:`, e instanceof Error ? e.message : String(e)) + return w.todos + } + } + + async function targetedRecovery( + sid: string, + kind: string, + prompt: string, + budgetKey: "toolTextAttempts" | "doneClaimAttempts" | "doneClaimOpenTodosAttempts" | "todoNudgeAttempts" | "intentNudgeAttempts", + ) { const w = ensureWatch(sid) if (w.recovering || w.userCancelled || w.permissionPending) return if (w.compacting) { @@ -1667,9 +1796,56 @@ export default define({ } } -async function inspectOnIdle(sid: string) { +/** + * Re-arm the done-claim budgets when a genuinely new work cycle starts. + * + * The signal is an inbound user message, and what makes one genuine is that + * it is a *different* message — identified by id, not by timestamp. A + * re-delivered or replayed message has the same id and is not new work; + * comparing clock times instead would treat any drift between two reads of + * the history as a new request and hand the model a fresh budget every time + * it repeated itself, which is exactly what #26 was. Timestamps are only the + * fallback, for messages that carry no id. + * + * The history is already loaded, so this costs nothing extra. + * + * Doing this only here (and not on every busy cycle) is the #26 fix: a model + * that re-announces completion each turn would otherwise be handed a fresh + * budget each time it announced, and the nudge would never stop. + */ + function noteInboundUserMessage(sid: string, w: SessionWatch, messages: unknown[]): void { + let latest: { id?: string; at?: number } | null = null + for (let i = messages.length - 1; i >= 0; i--) { + const m = messages[i] as { + id?: string + type?: string + role?: string + time?: { created?: number } + info?: { time?: { created?: number }; role?: string } + } + const isUser = m?.type === "user" || m?.role === "user" || m?.info?.role === "user" + if (!isUser) continue + latest = { id: typeof m.id === "string" ? m.id : undefined, at: m.time?.created ?? m.info?.time?.created } + break + } + if (!latest) return + const isNew = latest.id + ? latest.id !== w.lastUserMessageID + : typeof latest.at === "number" && latest.at > w.lastUserMessageSeenAt + if (!isNew) return + w.lastUserMessageID = latest.id + if (typeof latest.at === "number") w.lastUserMessageSeenAt = latest.at + if (w.doneClaimAttempts > 0 || w.doneClaimOpenTodosAttempts > 0) { + dbg(`${short(sid)} new user message — re-arming the done-claim budgets`) + } + w.doneClaimAttempts = 0 + w.doneClaimOpenTodosAttempts = 0 + } + + async function inspectOnIdle(sid: string) { const w = ensureWatch(sid) const messages = await loadMessages(sid) + noteInboundUserMessage(sid, w, messages) // Judged first, and before anything that needs text: a stream that died // before delivering any text is precisely the case where every @@ -1700,10 +1876,28 @@ async function inspectOnIdle(sid: string) { // work finished, so nudging here would talk over a deliberate stop. // Latched rather than re-derived, because the next idle with no new text // would otherwise re-check the same turn forever. + // + // Cross-checked against the todo list, because the emoji alone is not + // trustworthy: a model that finishes early celebrates early, and latching + // on that turns a false positive into permanent silence. With work still + // listed, the celebration is a false positive and the right answer is to + // name what is unfinished. if (endsWithCelebration(text)) { + const open = getOpenTodos(await readTodos(sid)) + if (open.length > 0) { + // Deliberately not latching: the list is still open, so this turn is + // not a completion and the next one must be free to judge again. + await targetedRecovery( + sid, + "open-todos-celebration-false-positive", + buildOpenTodosReminder(open), + "todoNudgeAttempts", + ) + return + } if (!w.completionSignaled) { w.completionSignaled = true - log("info", `${short(sid)} turn ends with a celebration — latching completion, not nudging`) + log("info", `${short(sid)} turn ends with a celebration and no open todos — latching completion, not nudging`) } else { dbg(`${short(sid)} completion already latched — skipping`) } @@ -1731,16 +1925,26 @@ async function inspectOnIdle(sid: string) { await targetedRecovery(sid, "action-intent", opts.actionIntentPrompt ?? opts.continuePrompt ?? CONTINUE_PROMPT, "intentNudgeAttempts") return } - if (containsDoneClaimPattern(text, doneClaimPatterns) && w.doneClaimAttempts < maxRetries) { - // Ask once for the work report. v2 used to gate this on a 400-char - // length, which cannot tell "Task done." from a real summary and so - // both over-nudged terse reports and let short-but-real ones pass. - // containsWorkDescription is the same structural test v1 uses, and - // prompting again after a real report loops forever (#26). - if (!containsWorkDescription(text)) { - await targetedRecovery(sid, "done-claim-no-details", doneWithoutDetailsPrompt, "doneClaimAttempts") - } else { - dbg(`${short(sid)} done-claim carries a work description — skipping details prompt`) + if (containsDoneClaimPattern(text, doneClaimPatterns)) { + // Two different prompts for two different situations, which is why the + // budgets are separate: spending the details prompt must not silence the + // one that says work is still listed. + const open = getOpenTodos(await readTodos(sid)) + if (open.length > 0) { + await targetedRecovery(sid, "done-claim-open-todos", doneWithoutWorkPrompt, "doneClaimOpenTodosAttempts") + return + } + if (w.doneClaimAttempts < maxRetries) { + // Ask once for the work report. v2 used to gate this on a 400-char + // length, which cannot tell "Task done." from a real summary and so + // both over-nudged terse reports and let short-but-real ones pass. + // containsWorkDescription is the same structural test v1 uses, and + // prompting again after a real report loops forever (#26). + if (!containsWorkDescription(text)) { + await targetedRecovery(sid, "done-claim-no-details", doneWithoutDetailsPrompt, "doneClaimAttempts") + } else { + dbg(`${short(sid)} done-claim carries a work description — skipping details prompt`) + } } } From 7d7c8c3451c91d73974a7536aea62dfabf3f3895 Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 16:32:04 -0400 Subject: [PATCH 20/57] feat(v2): offer the model an explicit way to say it is finished MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v2 ships no built-in task_complete — nothing in the source mentions the name — so the plugin registers the tool itself through ctx.tool.transform. That makes this a feature rather than a compatibility shim: it is the strongest completion signal available, stronger than a trailing 🎉, because the model chose to call it. The escalation is v1's unchanged, and it is the part worth having. The ack is fed straight back into the model's turn, and a stuck model answers it by calling the tool again instead of ending with text; v1 logged 27 consecutive acked calls with zero new user input. So the first ack carries an explicit stop instruction, the second warns that further calls will be rejected, and the third throws, which surfaces as a tool error and forces the turn to end. The counter resets on a genuinely new inbound user message and on any other tool call. The second of those is the part that makes it count: a model that finishes, reports, does more work and reports again is doing its job, not looping, and the escalation must not reject it. Two guards. The todo gate is skipped for subagents — a child was never asked to do the parent's open items, so gating its report on them blocks it on work outside its scope. And a blocked call fires a visible nudge naming the blocking todos, because the tool result can collapse to an invisible one-liner in the transcript; skipped when the user holds the ball, since injecting then starts a step their real reply interrupts. The block stays bounded by maxRetries, because an unbounded block is its own kind of loop. Registration failure — no ctx.tool, or a registry that throws — is logged and swallowed. It must not take down the session recovery the rest of this plugin exists for, so the watchdog starts either way. Two tests cover that. This needed the todo read that landed two commits ago: without a real list there is nothing to gate on, and a gate that could not see open work would have been worse than no gate. 17 new tests with a control per behaviour, including controls for both resets. 721 pass, 0 fail. --- README.md | 8 +- docs/known-issues-v2.md | 39 +++ src/v2/index.task-complete.test.ts | 441 +++++++++++++++++++++++++++++ src/v2/index.ts | 215 ++++++++++++-- 4 files changed, 680 insertions(+), 23 deletions(-) create mode 100644 src/v2/index.task-complete.test.ts diff --git a/README.md b/README.md index 9912ff7..e62ab6a 100644 --- a/README.md +++ b/README.md @@ -219,11 +219,15 @@ _Motivated by:_ ### Explicit completion via `task_complete` -The agent can call the built-in `task_complete` tool to signal that all work is done. When invoked, the plugin stops sending any further `"continue"` prompts, clears all pending timers, and marks the session as complete. This replaces fragile text-based heuristics (emoji patterns, language detection) with a deterministic signal. +The agent can call the `task_complete` tool to signal that all work is done. When invoked, the plugin stops sending any further `"continue"` prompts, clears all pending timers, and marks the session as complete. This replaces fragile text-based heuristics (emoji patterns, language detection) with a deterministic signal. + +On v2 there is no built-in equivalent, so the plugin registers the tool itself through the v2 tool registry (`ctx.tool.transform`). It is registered before the first turn, and a registry that refuses the registration is logged and skipped rather than taking the watchdog down with it. If `task_complete` is called while open todos remain, the call is rejected (up to `maxRetries` times) with a message asking the agent to finish the remaining work first. -Repeat calls with no new user message in between are guarded: the first acknowledgement carries an explicit stop instruction, the second returns a repeat warning, and the third and subsequent calls are rejected as errors. This breaks the ack self-loop where the acknowledgement tool-result is fed back into the turn and a stuck model re-emits `task_complete` instead of ending with text. The counter resets on a genuine user message, on the block path above, or when any other tool runs in between. +Repeat calls with no new user message in between are guarded: the first acknowledgement carries an explicit stop instruction, the second returns a repeat warning, and the third and subsequent calls are rejected as errors. This breaks the ack self-loop where the acknowledgement tool-result is fed back into the turn and a stuck model re-emits `task_complete` instead of ending with text — v1 logged 27 consecutive acked calls with zero new user input. The counter resets on a genuine user message, on the block path above, or when any other tool runs in between. + +The todo gate is skipped for subagents: a child was never asked to do the parent's open items, so gating its report on them would block it on work outside its scope. --- diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index e0cff3b..00a770c 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -170,6 +170,45 @@ handed a fresh budget each time it announced, so the nudge never stopped. v2 still had that, and now does not. The open-todos nudge is the opposite case and does reset per cycle, because an open list is new information each turn. +## Explicit completion via `task_complete` + +v1 registers a `task_complete` tool and v2 has no built-in equivalent — nothing in +the v2 source mentions the name — so the plugin provides it through +`ctx.tool.transform`, the v2 tool registry. This is a feature rather than a +compatibility shim: it is the strongest completion signal available, stronger than +a trailing 🎉, because the model chose to call it. + +The ack escalation is the part that matters, and it is v1’s unchanged. The ack +tool-result is fed straight back into the model’s turn, and a stuck model answers +it by calling the tool again rather than ending with text — v1 logged **27 +consecutive acked calls with zero new user input**. So: + +- first call acks, and the ack carries an explicit stop instruction; +- second consecutive call warns, and says further calls will be rejected; +- third and beyond throw, which surfaces as a tool error and forces the turn to end. + +The counter is reset by a genuinely new inbound user message and by any other tool +call, so the escalation measures a stuck loop rather than several legitimate rounds +of work. Resetting it on tool calls rather than only on user messages is the v2 +addition: a model that finishes, reports, does more work and reports again is +doing its job, not looping. + +Two guards worth naming: + +- **The todo gate is skipped for subagents.** A child was never asked to do the + parent’s open items, so gating its report on them would block it on work outside + its scope. Same reasoning as v1. +- **A blocked call also fires a visible nudge** naming the blocking todos, because + the tool result can collapse to an invisible one-liner in the transcript. Skipped + when the user holds the ball, since injecting then starts a step their real reply + interrupts. The block itself is bounded by `maxRetries`: an unbounded block is its + own kind of loop, and a model that keeps saying "done" may know something the todo + list does not. + +Registration failure — no `ctx.tool`, or a registry that throws — is logged and +swallowed. It must not take down the session recovery that the rest of the plugin +exists for, so the watchdog starts either way. + ## Context saturation A session can fill its context window without ever looking stalled — it just keeps diff --git a/src/v2/index.task-complete.test.ts b/src/v2/index.task-complete.test.ts new file mode 100644 index 0000000..a773453 --- /dev/null +++ b/src/v2/index.task-complete.test.ts @@ -0,0 +1,441 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_task_complete" + +/** + * `task_complete`: the model's explicit "I am finished". + * + * v2 ships no built-in equivalent — nothing in the v2 source mentions the name — + * so the plugin registering it is the only way it exists. That makes this a + * feature rather than a shim: the model gets a completion signal stronger than a + * trailing 🎉, because it chose to call this. + * + * The escalation is the part worth testing. The ack is fed straight back into the + * turn, and a stuck model answers it by calling the tool again rather than ending + * with text; v1 logged 27 consecutive acked calls with zero new user input. So the + * first call acks with an explicit stop instruction, the second warns, and the + * third throws to force the turn to end. + * + * The counter is reset by a new user message and by any other tool call, so the + * escalation measures a stuck loop and not several legitimate rounds of work. + * + * Every group carries a control. + */ + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + debug: true, +} + +const userMessage = (text: string, at: number, id = `msg_u_${at}`) => ({ + type: "user", + id, + time: { created: at }, + content: [{ type: "text", text }], +}) + +/** An hour ago: outside the 5-minute active-user window. */ +const OLD = Date.now() - 60 * 60_000 + +type Registered = { + name: string + description: string + input: unknown + execute: (input: any, context: { sessionID: string }) => Promise<{ content: string }> +} + +type Harness = { + tool: Registered | null + injected: Array<{ text?: string }> + logs: string[] +} + +async function setup( + opts: { + todos?: unknown[] | undefined + toolRegistry?: boolean + extraMessages?: unknown[] + parentID?: string + maxRetriesOverride?: number + }, +): Promise { + const injected: Harness["injected"] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-tc-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + let registered: Registered | null = null + const history: unknown[] = [userMessage("do the work", OLD), ...(opts.extraMessages ?? [])] + + const ctx: any = { + event: stream, + options: { + ...OPTIONS, + logFile, + // Raise maxRetries so the "blocked call does not count toward the ack + // escalation" test can stay inside the block for several calls instead of + // walking into the bounded-block path. + ...(opts.maxRetriesOverride ? { maxRetries: opts.maxRetriesOverride } : {}), + }, + session: { + context: async () => history, + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ text: a?.text }), {}), + }, + client: { + session: { + get: async () => ({ data: opts.parentID ? { id: SID, parentID: opts.parentID } : { id: SID } }), + }, + }, + storage: { + get: async () => ({ todos: opts.todos ?? [], updatedAt: Date.now() }), + set: async () => {}, + remove: async () => {}, + }, + } + if (opts.toolRegistry !== false) { + ctx.tool = { + transform: async (cb: any) => { + cb({ + add: (t: Registered) => { + registered = t + }, + }) + return { dispose() {} } + }, + list: async () => (registered ? [registered] : []), + } + } + + const cleanup = await (plugin as any).setup(ctx) + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + return { + get tool() { + return registered + }, + injected, + logs, + // Exposed for the few tests that deliver an event or a message mid-session. + ...({ cleanup, logFile, streamRef: stream, historyRef: history } as any), + } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const call = (h: Harness, sid = SID) => h.tool!.execute({}, { sessionID: sid }) + +describe("v2: task_complete is offered to the model", () => { + test("CONTROL: the tool is registered, and takes no arguments", async () => { + const h = await setup({}) + expect(h.tool).not.toBeNull() + expect(h.tool!.name).toBe("task_complete") + expect(h.tool!.description).toContain("Signal that all work is complete") + // v1's `args: {}`: the signal is the call, not a payload. + expect(h.tool!.input).toMatchObject({ type: "object", properties: {} }) + await teardown(h) + }) + + test("CONTROL: a plain call is acknowledged and records completion", async () => { + const h = await setup({}) + const res = await call(h) + expect(res.content).toContain("Task completion acknowledged") + // The stop instruction is in the ack on purpose: the ack is fed straight + // back into the turn, so a stuck model calls again instead of ending. + expect(res.content).toContain("End your turn now") + expect(res.content).toContain("do not call task_complete") + await teardown(h) + }) + + test("CONTROL: the ack is repeatable by design — no latching on the first call", async () => { + // The counter only escalates from the second consecutive call. This is the + // control for the escalation tests below: if the ack threw or latched, they + // would pass for the wrong reason. + const h = await setup({}) + await call(h) + await teardown(h) + }) +}) + +describe("v2: task_complete escalation — the ack self-loop", () => { + test("a second consecutive call warns instead of acking again", async () => { + const h = await setup({}) + await call(h) + const second = await call(h) + expect(second.content).toContain("acknowledged already on your previous call") + expect(second.content).toContain("further repeat calls are rejected as errors") + await teardown(h) + }) + + test("a third consecutive call is rejected as an error, forcing the turn to end", async () => { + const h = await setup({}) + await call(h) + await call(h) + await expect(call(h)).rejects.toThrow(/acknowledged twice/) + await teardown(h) + }) + + test("a fourth call is still an error — the loop cannot re-enter through the ack", async () => { + const h = await setup({}) + await call(h) + await call(h) + await expect(call(h)).rejects.toThrow() + await expect(call(h)).rejects.toThrow() + await teardown(h) + }) + + test("CONTROL: one call, then real tool work, then another call is not a repeat", async () => { + // Several legitimate rounds of work must not trip the escalation. This is + // the difference between a stuck model and a model that did more work. + const h = await setup({}) + await call(h) + stream2(h).push(ev("session.tool.called", { tool: "read" })) + await wait(30) + const second = await call(h) + expect(second.content).toContain("Task completion acknowledged") + expect(second.content).not.toContain("acknowledged already") + await teardown(h) + }) + + test("CONTROL: a new user message also resets the loop counter", async () => { + const h = await setup({}) + await call(h) + await call(h) // now on the warning + // A new request is new work: the counter starts clean, or one bad stretch + // would poison the rest of the session. + // + // Driven by an idle rather than by the message appearing: the reset is + // observed during idle inspection, so a test that never goes idle is testing + // nothing. (The todo and dead-stream tests hit the same wall.) + pushUserMessage(h, "actually, also this", Date.now() - 10 * 60_000) + await goIdle(h) + const third = await call(h) + expect(third.content).toContain("Task completion acknowledged") + expect(third.content).not.toContain("acknowledged already") + await teardown(h) + }) +}) + +describe("v2: task_complete respects the todo list", () => { + const OPEN = [ + { content: "Write the migration guide", status: "pending", priority: "high" }, + { content: "Delete the temp fixtures", status: "in_progress", priority: "low" }, + ] + + test("CONTROL: a call with todos still open is blocked and names them", async () => { + const h = await setup({ todos: OPEN }) + const res = await call(h) + expect(res.content).toContain("Write the migration guide") + expect(res.content).toContain("Delete the temp fixtures") + expect(res.content).toContain("Mark any finished todos complete") + // Not an ack: the model has not been told its work is recorded. + expect(res.content).not.toContain("No further continuation will be sent") + await teardown(h) + }) + + test("a blocked call also fires a visible nudge naming the open items", async () => { + // The tool result can collapse to an invisible one-liner in the transcript, + // so the blocking todos are also injected. Mirrors the idle-resume path. + const h = await setup({ todos: OPEN }) + await call(h) + expect(h.injected.length).toBeGreaterThan(0) + expect(h.injected[0].text).toContain("Write the migration guide") + await teardown(h) + }) + + test("a blocked call does not count toward the ack escalation", async () => { + // Work remains, so a later completion is legitimate. Feeding the escalation + // here would reject a model that was simply told to finish more. + // maxRetries 1 so the first call is blocked and the second is let through: + // the point is what signal the let-through call carries, not how long the + // block lasts. + const h = await setup({ todos: OPEN, maxRetriesOverride: 1 }) + await call(h) + const second = await call(h) + expect(second.content).toContain("Task completion acknowledged") + expect(second.content).not.toContain("acknowledged already") + await teardown(h) + }) + + test("the block is bounded — a model insisting is eventually believed", async () => { + // An unbounded block is its own kind of loop. After maxRetries the call is + // honoured, because a model that keeps saying "done" may know something the + // todo list does not. maxRetries defaults to 3, so the fourth call is let + // through. + const h = await setup({ todos: OPEN }) + let res = await call(h) + for (let i = 0; i < 3; i++) res = await call(h) + expect(res.content).toContain("Task completion acknowledged") + expect(res.content).not.toContain("Mark any finished todos") + await teardown(h) + }) + + test("a call with every todo closed is acked", async () => { + const h = await setup({ + todos: [ + { content: "Done already", status: "completed", priority: "high" }, + { content: "Dropped", status: "cancelled", priority: "low" }, + ], + }) + const res = await call(h) + expect(res.content).toContain("Task completion acknowledged") + await teardown(h) + }) + + test("a call with no todo tool at all is acked, not blocked", async () => { + // The dangerous failure is a session with no list being told it has open + // work. An absent list must not read as an empty-but-authoritative one. + const h = await setup({}) + const res = await call(h) + expect(res.content).toContain("Task completion acknowledged") + expect(res.content).not.toContain("Mark any finished todos") + await teardown(h) + }) +}) + +describe("v2: task_complete on a subagent", () => { + test("a subagent reporting completion is not gated on the parent's todo list", async () => { + // A child was never asked to do the parent's open items, so gating its + // report on them would block it on work outside its scope. + const h = await setup({ + todos: [{ content: "Parent's own task", status: "pending", priority: "high" }], + parentID: "ses_parent", + }) + const res = await call(h) + expect(res.content).toContain("Task completion acknowledged") + await teardown(h) + }) +}) + +describe("v2: task_complete degrades rather than failing the watchdog", () => { + test("a host with no tool registry still runs the plugin", async () => { + const h = await setup({ toolRegistry: false }) + expect(h.tool).toBeNull() + // The rest of the plugin must be live regardless. + stream2(h).push(ev("session.execution.started")) + await wait(30) + stream2(h).push(ev("session.idle")) + await wait(200) + expect(h.logs.some((l) => l.includes("task_complete not offered"))).toBe(true) + await teardown(h) + }) + + test("a registry that throws does not stop the watchdog", async () => { + // Registration failure is the one thing that must never take down the + // session recovery that the rest of this plugin exists for. + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-tc-throw-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async () => [userMessage("go", OLD)], + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async () => ({}), + prompt: async () => ({}), + }, + client: { session: { get: async () => ({ data: { id: SID } }) } }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { + transform: async () => { + throw new Error("registry unavailable") + }, + list: async () => [], + }, + } + const cleanup = await (plugin as any).setup(ctx) + for (const e of [ev("session.execution.started"), ev("session.step.started"), ev("session.idle")]) { + stream.push(e) + await wait(10) + } + await wait(400) + ;(cleanup as (() => void) | undefined)?.() + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + expect(logs.some((l) => l.includes("task_complete registration failed"))).toBe(true) + // And the plugin is still running: it logged its startup line. + expect(logs.some((l) => l.includes("task_complete registration failed") && l.includes("continuing without it"))).toBe( + true, + ) + }) +}) + +/** The harness keeps the event stream private; these reach it for the tests + * that need to deliver a tool call or a user message mid-session. */ +function stream2(h: any) { + return h.streamRef as ReturnType +} + +function pushUserMessage(h: any, text: string, at: number) { + ;(h.historyRef as unknown[]).push(userMessage(text, at)) +} + +/** Drive a turn to idle, which is what makes the plugin look at the history. */ +async function goIdle(h: any) { + const stream = stream2(h) + for (const e of [ + ev("session.execution.started"), + ev("session.step.started"), + ev("session.step.ended"), + ev("session.idle"), + ]) { + stream.push(e) + await wait(10) + } + await wait(200) +} diff --git a/src/v2/index.ts b/src/v2/index.ts index a99556e..964975e 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -97,6 +97,15 @@ interface AutoResumePluginInput { * installed todo tool writes, so this only ever calls `get`. */ storage?: { get: (key: string) => Promise } + /** + * Tool registry (`ctx.tool.transform` / `ctx.tool.list` / `ctx.tool.hook`). + * v2 plugins register their own tools here, which is how `task_complete` + * exists: there is no built-in equivalent in v2, so the plugin provides it. + */ + tool?: { + transform: (cb: (editor: AutoResumeToolEditor) => void) => Promise + list: () => Promise + } /** Application logger, when the host provides one. */ app?: { log?: (level: string, message: string) => unknown } } @@ -117,6 +126,16 @@ interface V2Event { data?: Record } +/** The tool registry handle passed to `ctx.tool.transform`. */ +interface AutoResumeToolEditor { + add(tool: { + name: string + description: string + input: Record + execute: (input: any, context: { sessionID: string }) => Promise<{ content: string }> + }): void +} + interface SessionWatch { createdAt: number lastActivityAt: number @@ -163,6 +182,15 @@ interface SessionWatch { /** The session's todo list, read from the todo tool's storage key. */ todos: Todo[] todosFetchedAt: number + /** Consecutive `task_complete` acks with no new user message or other tool + * work between them. Guards the ack self-loop, where the ack tool-result is + * fed back and the model re-emits the tool instead of ending its turn. Persists + * across busy/idle cycles by design. */ + taskCompleteSignals: number + /** How many times a `task_complete` call was overridden because todos were + * still open. Bounded by `maxRetries` — past that the call is honoured, because + * an unbounded block is its own kind of loop. */ + taskCompleteOverrides: number /** Id of the newest inbound user message already acted on, so a replayed or * re-delivered message does not re-arm the done-claim budgets. Identity, not * clock time — see noteInboundUserMessage. */ @@ -480,6 +508,33 @@ const AWAITING_USER_TOOLS = new Set(["question", "permission", "ask", "confirm"] const CONTINUE_PROMPT = "continue" +/** + * The `task_complete` acknowledgement texts. + * + * The ack is fed straight back into the model's turn, so a stuck model re-emits + * `task_complete` instead of ending with text. That is not hypothetical: v1 + * recorded 27 consecutive acked calls with zero new user input. The first ack + * therefore carries an explicit stop instruction, the second warns, and the third + * throws so the turn is forced to end. + */ +const TASK_COMPLETE_ACK = + "Task completion acknowledged. No further continuation will be sent. " + + "End your turn now with a brief text reply — do not call task_complete or any other tool again " + + "unless the user sends a new message." + +const TASK_COMPLETE_REPEAT_WARNING = + "Task completion acknowledged already on your previous call — completion is recorded. " + + "End your turn now with a brief text reply. Do not call task_complete again; " + + "further repeat calls are rejected as errors." + +const TASK_COMPLETE_REPEAT_ERROR = + "task_complete already acknowledged twice with no new user message or tool work since — " + + "completion is recorded. End your turn with text and make no further tool calls." + +/** v1's wording, kept verbatim — the model has already seen this description. */ +const TASK_COMPLETE_DESCRIPTION = + "Signal that all work is complete and stop automatic continuation prompts. Call this ONLY after finishing everything requested. Call exactly once per completed round of work — repeat calls with no new user message in between are rejected." + const TOOL_TEXT_RECOVERY_PROMPT = "Your last message contained a raw tool call printed as text instead of being executed. " + "Please use the proper tool calling mechanism to execute it." @@ -1032,6 +1087,8 @@ export default define({ completionSignaled: false, todos: [], todosFetchedAt: 0, + taskCompleteSignals: 0, + taskCompleteOverrides: 0, lastUserMessageSeenAt: 0, lastTokenTotal: 0, contextWrapupAttempts: 0, @@ -1736,27 +1793,23 @@ export default define({ return true } - async function shouldStandDownForUser( - sid: string, - messages: unknown[], - activeUserWindowMs: number, - ): Promise { - try { - if (messages.length === 0) return false - const newest = messages[messages.length - 1] as { - type?: string - content?: { type?: string; name?: string; state?: { status?: string } }[] - time?: { created?: number } - info?: { time?: { created?: number } } - } - // (a) A tool call still awaiting its result means the model is waiting on the - // user — a synthetic nudge here would be interrupted by their real reply - // ("Step interrupted"). - // - // This used to test `state.status === "pending"`, which v2 never emits, so - // the branch was unreachable and the nudge fired with an unanswered - // `question` on screen. See TOOL_STATE_* above. - if (newest?.type === "assistant") { + /** + * True when the session is holding the ball: a tool call whose result is + * still outstanding. + * + * A synthetic nudge in that state starts a step the user's real reply then + * interrupts ("Step interrupted"), so every idle recovery path stands down. + * + * This used to test `state.status === "pending"`, which v2 never emits, so + * the branch was unreachable and the nudge fired with an unanswered + * `question` on screen. See TOOL_STATE_* above. + */ + function hasPendingUserInput(messages: unknown[]): boolean { + const newest = messages[messages.length - 1] as { + type?: string + content?: { type?: string; name?: string; state?: { status?: string } }[] + } + if (newest?.type !== "assistant") return false for (const part of newest.content ?? []) { const t = part?.type ?? "" if (!(t === "tool_use" || t === "tool" || t === "tool_call" || t.startsWith("tool"))) continue @@ -1767,7 +1820,18 @@ export default define({ // not that it was answered, so keep standing down. if (status !== TOOL_STATE_COMPLETED && AWAITING_USER_TOOLS.has(part?.name ?? "")) return true } + return false } + + async function shouldStandDownForUser( + sid: string, + messages: unknown[], + activeUserWindowMs: number, + ): Promise { + try { + if (messages.length === 0) return false + // (a) A tool call still awaiting its result — see hasPendingUserInput. + if (hasPendingUserInput(messages)) return true // (b) The user was recently active — they are mid-conversation, not stuck. // // v1 stamps `lastUserMessageAt` on ANY inbound user message and asks @@ -1840,6 +1904,102 @@ export default define({ } w.doneClaimAttempts = 0 w.doneClaimOpenTodosAttempts = 0 + // A new request is new work: the ack self-loop counter starts clean, + // because the model re-announcing completion after the user asked for + // more is not the stuck case this counts. + w.taskCompleteSignals = 0 + } + + /** + * `task_complete`: the model's explicit "I am finished". + * + * v2 has no built-in equivalent — `grep -rl task_complete packages` finds + * nothing — so registering it here is the only way it exists at all, and + * this is a real feature rather than a compatibility shim: it is the + * strongest completion signal available, stronger than a trailing 🎉, + * because the model chose to call it. + * + * Ported from v1 with its escalation intact. The ack tool-result is fed + * straight back into the turn, and a stuck model answers it by calling the + * tool again rather than ending with text — v1 logged 27 consecutive acked + * calls with zero new user input. So: first call acks with an explicit stop + * instruction, second warns, third throws and the turn is forced to end. + * + * The counter is reset by a new inbound user message and by any other tool + * call, so the escalation only ever measures a genuinely stuck loop rather + * than several legitimate rounds of work. + */ + async function registerTaskCompleteTool(): Promise { + if (!ctx.tool?.transform) { + dbg("tool registry unavailable — task_complete not offered") + return + } + try { + await ctx.tool.transform((editor) => { + editor.add({ + name: "task_complete", + description: TASK_COMPLETE_DESCRIPTION, + // No arguments. v1's `args: {}` is the same thing: the signal + // is the call, not a payload. + input: { type: "object", properties: {}, additionalProperties: false }, + execute: async (_input, toolCtx) => { + const sid = toolCtx?.sessionID + if (!sid) return { content: TASK_COMPLETE_ACK } + const w = ensureWatch(sid) + + // A subagent reporting completion is not a parent finishing, + // and gating a child's report on the parent's todo list would + // block it on work it was never asked to do. v1 skips the + // todo gate for subagents for the same reason. + if (!(await isSubAgentSession(sid))) { + const open = getOpenTodos(await readTodos(sid)) + if (open.length > 0 && w.taskCompleteOverrides < maxRetries) { + w.taskCompleteOverrides++ + // Work remains, so a later completion is legitimate: reset the + // repeat-signal counter rather than letting this override feed + // the escalation. + w.taskCompleteSignals = 0 + const reminder = buildOpenTodosReminder(open) + const blockMsg = `Mark any finished todos complete and do not redo completed work.\n${reminder}` + log( + "info", + `${short(sid)} task_complete blocked: ${open.length} open todos remain (override ${w.taskCompleteOverrides}/${maxRetries})`, + ) + // Also fire a visible nudge naming the blocking todos, in case + // the tool result collapses to an invisible one-liner. Skipped + // when the user holds the ball — injecting then would start a + // step their real reply interrupts. The tool result already + // carries the todo names either way. + let awaitingInput = false + try { + awaitingInput = hasPendingUserInput(await loadMessages(sid)) + } catch (e) { + dbg(`${short(sid)} task_complete block: awaiting-input check failed:`, e instanceof Error ? e.message : String(e)) + } + if (awaitingInput) { + log("info", `${short(sid)} task_complete blocked but user input pending — skipping visible nudge`) + } else { + await injectOnce(sid, blockMsg, "task-complete-blocked") + } + return { content: blockMsg } + } + } + + w.completionSignaled = true + log("info", `${short(sid)} task_complete called, ${(await isSubAgentSession(sid)) ? "subagent" : "agent"} done`) + w.taskCompleteSignals++ + if (w.taskCompleteSignals === 2) return { content: TASK_COMPLETE_REPEAT_WARNING } + if (w.taskCompleteSignals > 2) throw new Error(TASK_COMPLETE_REPEAT_ERROR) + return { content: TASK_COMPLETE_ACK } + }, + }) + }) + log("info", "registered the task_complete tool") + } catch (e) { + // A failed registration must not take the watchdog down with it. + const msg = e instanceof Error ? e.message : String(e) + log("warn", `task_complete registration failed (${msg}) — continuing without it`) + } } async function inspectOnIdle(sid: string) { @@ -2083,6 +2243,11 @@ export default define({ discoverSessions().catch(() => {}) }, discoveryDelayMs) + // Offered to the model before the first turn, so it can actually be called. + // Awaited, but guarded: a registry that refuses the registration must not + // stop the watchdog from starting. + await registerTaskCompleteTool() + // --------------------------------------------------------------------- // Context saturation // --------------------------------------------------------------------- @@ -2583,6 +2748,14 @@ export default define({ const w = ensureWatch(sid) const name = typeof ev.data?.tool === "string" ? ev.data.tool : (ev.data?.id as string | undefined) ?? "tool" w.lastWasTaskTool = isTaskToolCall(ev) + // Any real tool work between completions legitimises the next + // task_complete — reset the repeat-signal counter. task_complete + // itself is excluded; it manages the counter inside its execute. + // + // Without this, a model that finishes, calls the tool, then does one + // more piece of work and finishes again would be counted as a repeat, + // and the escalation would fire against a legitimate second round. + if (name !== "task_complete") w.taskCompleteSignals = 0 if (trackToolCall(w, name)) { void targetedRecovery(sid, "tool-loop", TOOL_LOOP_RECOVERY_PROMPT, "intentNudgeAttempts").catch(() => {}) } From 516c40fd9e0b7beda78da47ade3a8ccfad3bc70a Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 16:43:12 -0400 Subject: [PATCH 21/57] feat(v2): name a replacement when a model calls a tool that does not exist MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v1 asks ctx.client.tool.ids() for the registered names. v2 has no such route; ctx.tool.list() returns the same definitions, and only the effective names are needed. The effective name is the namespaced form when a tool sits in a namespace, which is exactly the string a model would have to call, so that is what the suggestion quotes back. The parts come from the message history rather than from events, and that is not a workaround: the thing being looked for is a part that already errored, and by the time the error lands the session.tool.called event for that part is long gone. This code only runs on idle anyway. Two v2 shapes differ from v1, and both differences point the same way — inert rather than wrong, which is the failure mode that does not announce itself. v1 reads the tool name from part.tool, which v2 has no field for; the name is in part.name, and a port keeping v1's read would compare against "" and never fire. v1 keys its already-examined set on part.id ?? part.callID, and v2's tool part has id and name but no callID. The state.status === "error" test carries over unchanged, because error really is one of v2's four tool states. The check abstains rather than guesses. No registry, an empty registry, or a registry that throws all mean no suggestion, because with nothing to compare against every name looks invented — and telling a model that a tool it plainly just used does not exist is the one failure mode that teaches it the registry is unreliable. One ordering fix fell out of testing. The per-request re-arm is derived from the message history, so it cannot be applied before the history is read, but the already-suggested guard ran before that read while the check runs alongside inspectOnIdle, whose ordering is not defined. On a re-armed turn the guard could see the previous turn's latch still standing and skip the very check that was supposed to re-arm it. The guard now runs after the read. 15 new tests, a control per behaviour. 736 pass, 0 fail. --- README.md | 4 +- docs/known-issues-v2.md | 51 ++++ src/v2/index.ts | 183 +++++++++++++ src/v2/index.unknown-tool.test.ts | 410 ++++++++++++++++++++++++++++++ 4 files changed, 647 insertions(+), 1 deletion(-) create mode 100644 src/v2/index.unknown-tool.test.ts diff --git a/README.md b/README.md index e62ab6a..405c0a5 100644 --- a/README.md +++ b/README.md @@ -58,7 +58,7 @@ Loop detection runs in two places. At idle, tool names are scanned from recent a When the model calls a tool that does not exist (a typo, a hallucinated name, or a tool from a different plugin that isn't loaded), OpenCode returns a `tool` part with `state.status = "error"`. If the same wrong tool name appears **2 times** in the session's message history, the plugin: -1. Fetches the list of available tools via `ctx.client.tool.ids()` (cached for 5 minutes). +1. Fetches the list of available tools — via `ctx.client.tool.ids()` on v1, or `ctx.tool.list()` on v2, cached for 5 minutes. On v2 the names are the registry's *effective* names, so a namespaced tool is quoted back in the form a model would have to call it. 2. Computes the closest match by Levenshtein distance (case-insensitive, threshold = half the wrong name's length). 3. Sends a continue prompt that names the wrong tool, states it does not exist, suggests the closest match (if one was found), and lists the first 20 available tools for reference. @@ -66,6 +66,8 @@ The suggestion fires once per busy cycle. A new user message resets the counter Threshold and cache are compile-time constants (`UNKNOWN_TOOL_THRESHOLD = 2`, `TOOL_IDS_CACHE_MS = 5 min`). The detection runs at idle, alongside the tool-text recovery check. +On v2 the tool list comes from the registry, so the check degrades rather than guesses: no registry, an empty registry, or a registry that throws all mean no suggestion, because with nothing to compare against every name would look invented. + _Motivated by:_ - [#22142](https://github.com/anomalyco/opencode/issues/22142) — Repetitive tool-call loops with alibaba-coding-plan-cn/qwen3.6-plus - [#16218](https://github.com/anomalyco/opencode/issues/16218) — Model repeats the same response in a loop after generating an answer diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 00a770c..5f4f493 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -170,6 +170,57 @@ handed a fresh budget each time it announced, so the nudge never stopped. v2 still had that, and now does not. The open-todos nudge is the opposite case and does reset per cycle, because an open list is new information each turn. +## Unknown tool names + +v1 asks `ctx.client.tool.ids()` for the registered tool names. v2 has no such +route, so the port uses `ctx.tool.list()`, which returns the same set of +definitions. Only the effective names are needed — nothing else in the record is +read — and the effective name is the namespaced form when a tool sits in a +namespace, which is exactly the string a model would have to call. That is what +the suggestion quotes back. + +The tool parts come from the message history, not from events, and that is not a +workaround: the thing being looked for is a part that already **errored**, and by +the time the error lands the `session.tool.called` event for that part is long +gone. This code only runs on idle anyway. + +Two shape differences from v1, both of which would have made the check inert +rather than wrong — worth naming because an inert detector is the failure mode +that does not announce itself: + +- v1 reads the tool name from `part.tool`. v2 has no such field; the name is in + `part.name`. A port that kept v1's read would compare against `""`, never + match, and silently never fire. +- v1 keys the "already examined" set on `part.id ?? part.callID`. v2's tool part + carries `id` (the call id) and `name`, and no `callID`. + +v2 tool-state statuses are `streaming`, `running`, `completed`, `error`, so the +`state.status === "error"` test v1 uses is correct on v2 unchanged. (The +`"pending"` status v1 also tests, elsewhere, does not exist in v2 — see the +TOOL_STATE_* note in the source.) + +Two behaviours are inherited from v1 rather than fixed here, and the second is +worth a decision: + +- **A re-armed turn re-names the oldest typo.** The per-request reset clears the + "already examined" set, so the walk restarts from the top of the history and + the oldest name still above threshold wins — a turn that introduced a *new* + invented name is told about the *previous* one. Suppressing a name once it has + been suggested is a behaviour change rather than a port, so it is documented + instead of made. +- **No registry, no check.** With no tool list every name looks invented, so the + check is skipped rather than accusing the model of tools that plainly exist. + Same for an empty registry and for a registry that throws, which is logged and + ignored: a check that guesses is worse than a check that abstains. + +One ordering fix the port needed. The per-request re-arm is derived from the +message history, so it cannot be applied before the history is read — but the +"already suggested" guard used to run before that read, and the check runs +alongside `inspectOnIdle`, whose ordering is not defined. On a re-armed turn the +guard could therefore see the previous turn's latch still standing and skip the +check that was supposed to re-arm it. The guard now runs after the read. + + ## Explicit completion via `task_complete` v1 registers a `task_complete` tool and v2 has no built-in equivalent — nothing in diff --git a/src/v2/index.ts b/src/v2/index.ts index 964975e..4a3b3b0 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -187,6 +187,15 @@ interface SessionWatch { * fed back and the model re-emits the tool instead of ending its turn. Persists * across busy/idle cycles by design. */ taskCompleteSignals: number + /** Failed tool calls by name, for the unknown-tool suggestion. Counted, not + * latched: one bad call is a typo, two is a model that will not self-correct. */ + unknownToolErrors: Map + /** Set once a suggestion has been injected, so it is sent at most once per + * user message rather than on every idle. */ + unknownToolSuggestionSent: boolean + /** Tool parts already examined. The history is re-walked on every idle, so + * without this the same error would be counted again each time. */ + checkedToolPartIDs: Set /** How many times a `task_complete` call was overridden because todos were * still open. Bounded by `maxRetries` — past that the call is honoured, because * an unbounded block is its own kind of loop. */ @@ -531,6 +540,16 @@ const TASK_COMPLETE_REPEAT_ERROR = "task_complete already acknowledged twice with no new user message or tool work since — " + "completion is recorded. End your turn with text and make no further tool calls." +/** How many times a model must call a tool that does not exist before we name a + * replacement. One is a typo; two is a model that will not correct itself. */ +const UNKNOWN_TOOL_THRESHOLD = 2 +/** The tool list changes only when a plugin reloads, so a short cache is enough + * and it keeps the check off the registry on every idle. */ +const TOOL_IDS_CACHE_MS = 5 * 60_000 +/** Cap on the tool list quoted back to the model. A full dump of every tool on + * the box is itself a way to fill the context. */ +const UNKNOWN_TOOL_LIST_LIMIT = 20 + /** v1's wording, kept verbatim — the model has already seen this description. */ const TASK_COMPLETE_DESCRIPTION = "Signal that all work is complete and stop automatic continuation prompts. Call this ONLY after finishing everything requested. Call exactly once per completed round of work — repeat calls with no new user message in between are rejected." @@ -822,6 +841,52 @@ function detectPatternLoop(recentTools: string[]): boolean { return false } +/** Edit distance, v1's implementation. Kept byte-for-byte in behaviour: the + * suggestion is only useful if it picks the same tool v1 would have. */ +function levenshtein(a: string, b: string): number { + const m = a.length + const n = b.length + if (m === 0) return n + if (n === 0) return m + let prev = new Array(n + 1) + let curr = new Array(n + 1) + for (let j = 0; j <= n; j++) prev[j] = j + for (let i = 1; i <= m; i++) { + curr[0] = i + for (let j = 1; j <= n; j++) { + const cost = a[i - 1] === b[j - 1] ? 0 : 1 + curr[j] = Math.min(prev[j] + 1, curr[j - 1] + 1, prev[j - 1] + cost) + } + const tmp = prev + prev = curr + curr = tmp + } + return prev[n] +} + +/** + * The closest registered tool name to one the model invented, or null. + * + * The threshold scales with the length of the wrong name — `max(2, len/2)` — so a + * short name is held to a near-exact match while a long one is allowed to be + * sloppy. Without that, a two-letter name would match almost anything and the + * suggestion would be noise. + */ +function suggestClosestTool(wrongName: string, available: string[]): string | null { + const lower = wrongName.toLowerCase() + let best: string | null = null + let bestDist = Infinity + for (const id of available) { + const dist = levenshtein(lower, id.toLowerCase()) + const threshold = Math.max(2, Math.floor(lower.length / 2)) + if (dist < bestDist && dist <= threshold) { + bestDist = dist + best = id + } + } + return best +} + function trackToolCall(w: SessionWatch, toolName: string): boolean { const now = Date.now() w.recentToolCalls = w.recentToolCalls.filter((c) => now - c.at < 120_000) @@ -1089,6 +1154,9 @@ export default define({ todosFetchedAt: 0, taskCompleteSignals: 0, taskCompleteOverrides: 0, + unknownToolErrors: new Map(), + unknownToolSuggestionSent: false, + checkedToolPartIDs: new Set(), lastUserMessageSeenAt: 0, lastTokenTotal: 0, contextWrapupAttempts: 0, @@ -1908,6 +1976,113 @@ export default define({ // because the model re-announcing completion after the user asked for // more is not the stuck case this counts. w.taskCompleteSignals = 0 + // The unknown-tool budget is scoped to one request for the same reason: + // a model told to do something new gets a fresh tool list, and one + // invented name in the previous round says nothing about this one. + if (w.unknownToolErrors.size > 0 || w.unknownToolSuggestionSent) { + dbg(`${short(sid)} new user message — re-arming the unknown-tool budget`) + } + w.unknownToolErrors.clear() + w.unknownToolSuggestionSent = false + w.checkedToolPartIDs.clear() + } + + /** Registered tool names, cached briefly — the registry only changes when a + * plugin reloads, and this runs on every idle. */ + let cachedToolIds: string[] | null = null + let cachedToolIdsAt = 0 + + async function getAvailableToolIds(): Promise { + if (cachedToolIds && Date.now() - cachedToolIdsAt < TOOL_IDS_CACHE_MS) return cachedToolIds + if (!ctx.tool?.list) return cachedToolIds ?? [] + try { + const listed = await ctx.tool.list() + // `id` is the effective name — the namespaced form when the tool sits + // in a namespace — which is exactly the string a model would have to + // call, so it is what the suggestion has to quote back. + const ids = (listed as Array<{ id?: string; name?: string }>) + .map((t) => (typeof t?.id === "string" && t.id ? t.id : typeof t?.name === "string" ? t.name : "")) + .filter((n) => n.length > 0) + cachedToolIds = ids + cachedToolIdsAt = Date.now() + return ids + } catch (e) { + log("warn", `failed to list tools: ${e instanceof Error ? e.message : String(e)}`) + // Stale names are better than none for a "does this exist" check: a + // tool added since the cache would be wrongly called unknown, but + // that costs one extra suggestion, while an empty list skips the + // check entirely and loses the feature. + return cachedToolIds ?? [] + } + } + + /** + * Name a replacement when a model keeps calling a tool that does not exist. + * + * v1 asks `ctx.client.tool.ids()`. v2 has no such route; `ctx.tool.list()` + * returns the same set of effective names, which is all this needs. + * + * The tool parts come from the message history rather than from events, + * because the whole point is a part that already *errored* — by the time the + * error lands, the `session.tool.called` event for that part is long gone, + * and this code only runs on idle anyway. + */ + async function checkForUnknownToolCalls(sid: string, w: SessionWatch): Promise { + if (w.userCancelled || w.completionSignaled) return false + try { + const messages = await loadMessages(sid) + // The history is read before the "already suggested" guard, and the + // per-request reset rides on that read. The guard used to come first, + // and since this runs alongside inspectOnIdle the ordering was undefined + // — on a re-armed turn the guard could see the previous turn's latch + // still standing and skip the check that was supposed to re-arm it. + noteInboundUserMessage(sid, w, messages) + if (w.unknownToolSuggestionSent) return false + const available = await getAvailableToolIds() + // No registry, or nothing registered: every name would look unknown. + if (available.length === 0) return false + for (const msg of messages) { + const content = (msg as { content?: unknown })?.content + if (!Array.isArray(content)) continue + for (const part of content as Array>) { + if (part?.type !== "tool") continue + // v2's tool part identifies itself by `id` and names the tool in + // `name` — v1 read `part.tool`, which v2 never emits, so the + // comparison there would have been against "" and never matched. + const partId = typeof part.id === "string" ? part.id : "" + if (!partId || w.checkedToolPartIDs.has(partId)) continue + w.checkedToolPartIDs.add(partId) + const state = part.state as { status?: string } | undefined + if (state?.status !== "error") continue + const toolName = typeof part.name === "string" ? part.name : "" + if (!toolName || available.includes(toolName)) continue + const count = (w.unknownToolErrors.get(toolName) ?? 0) + 1 + w.unknownToolErrors.set(toolName, count) + if (count < UNKNOWN_TOOL_THRESHOLD) continue + const suggestion = suggestClosestTool(toolName, available) + const toolList = available.slice(0, UNKNOWN_TOOL_LIST_LIMIT).join(", ") + const prompt = suggestion + ? `You tried to use the tool "${toolName}" ${count} times, but it does not exist. ` + + `The closest matching tool is "${suggestion}". ` + + `Please use "${suggestion}" instead and adjust your arguments accordingly. ` + + `Available tools include: ${toolList}.` + : `You tried to use the tool "${toolName}" ${count} times, but it does not exist. ` + + `Please check the available tools and use the correct one. ` + + `Available tools include: ${toolList}.` + w.unknownToolSuggestionSent = true + log( + "warn", + `${short(sid)} unknown tool "${toolName}" called ${count}x, suggesting "${suggestion ?? "(none)"}"`, + ) + await injectOnce(sid, prompt, "unknown-tool") + return true + } + } + return false + } catch (e) { + log("warn", `${short(sid)} unknown-tool check failed: ${e instanceof Error ? e.message : String(e)}`) + return false + } } /** @@ -2585,6 +2760,14 @@ export default define({ return } void inspectOnIdle(sid) + // Independent of the idle heuristics above: this looks at tool parts + // that already errored, which none of those read. Fire-and-forget + // so it cannot delay them. The latch check lives inside the function + // rather than here, because the per-request re-arm is derived from the + // history and only happens once the history has been read. + if (!w.completionSignaled && !w.userCancelled) { + void checkForUnknownToolCalls(sid, w) + } return } case "session.execution.interrupted": { diff --git a/src/v2/index.unknown-tool.test.ts b/src/v2/index.unknown-tool.test.ts new file mode 100644 index 0000000..19ccf8e --- /dev/null +++ b/src/v2/index.unknown-tool.test.ts @@ -0,0 +1,410 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_unknown_tool" + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + debug: true, +} + +const userMessage = (text: string, at: number, id = `msg_u_${at}`) => ({ + type: "user", + id, + time: { created: at }, + content: [{ type: "text", text }], +}) + +/** An hour ago: outside the 5-minute active-user window, so the idle path is not + * suppressed for "the user is mid-conversation". */ +const OLD = Date.now() - 60 * 60_000 + +/** + * A v2 tool part. Note `name` and `id`: v2 names the tool in `name`, where v1 + * read `part.tool`. `id` is the call id, which is what the "already examined" + * set keys on. + */ +const toolPart = (id: string, name: string, status: "error" | "completed" | "running" = "error") => ({ + type: "tool", + id, + name, + time: { created: OLD }, + state: + status === "error" + ? { status, input: {}, error: "tool not found" } + : status === "completed" + ? { status, input: {}, content: [{ type: "text", text: "ok" }] } + : { status, input: {}, metadata: {} }, +}) + +/** An assistant message carrying tool parts. It always has a text part too, and + * that is not incidental: a finished assistant message with no text part is a + * silent dead stream, and that detector fires first and would swallow the test. */ +const assistantWith = (...content: unknown[]) => ({ + type: "assistant", + id: "msg_a_here", + time: { created: OLD + 1_000 }, + content: [{ type: "text", text: "Let me look at that." }, ...content], +}) + +const TOOLS = [{ id: "read" }, { id: "write" }, { id: "glob" }, { id: "bash" }, { id: "task_complete" }] + +async function setup( + opts: { + history?: unknown[] + tools?: Array<{ id: string }> | undefined + toolRegistry?: boolean + toolListThrows?: boolean + parentID?: string + }, +): Promise { + const injected: Array<{ text?: string }> = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-ut-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const history: unknown[] = opts.history ?? [] + const listed = opts.tools ?? TOOLS + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async () => history, + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ text: a?.text }), {}), + }, + client: { + session: { get: async () => ({ data: opts.parentID ? { id: SID, parentID: opts.parentID } : { id: SID } }) }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + } + if (opts.toolRegistry !== false) { + ctx.tool = { + transform: async (cb: any) => { + cb({ add: () => {} }) + return { dispose() {} } + }, + list: async () => { + if (opts.toolListThrows) throw new Error("registry unavailable") + return listed + }, + } + } + + const cleanup = await (plugin as any).setup(ctx) + return { + injected, + history, + ...({ cleanup, logFile, streamRef: stream } as any), + } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +/** Drive a turn to idle, which is what makes the plugin look at the history. */ +async function goIdle(h: any) { + for (const e of [ev("session.execution.started"), ev("session.step.started"), ev("session.step.ended"), ev("session.idle")]) { + h.streamRef.push(e) + await wait(10) + } + await wait(250) +} + +const suggestions = (h: any) => h.injected.filter((i: any) => /does not exist/.test(i.text ?? "")) + +describe("v2: naming a replacement for a tool that does not exist", () => { + test("CONTROL: one invented tool call is a typo and gets no suggestion", async () => { + // The threshold is the whole design: a model that misspells one tool name + // normally self-corrects, and interrupting it teaches it nothing. + const h = await setup({ history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "globb"))] }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("a second invented call to the same name earns a suggestion", async () => { + const h = await setup({ + history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "globb")), assistantWith(toolPart("call_2", "globb"))], + }) + await goIdle(h) + const said = suggestions(h) + expect(said).toHaveLength(1) + expect(said[0].text).toContain('"globb"') + expect(said[0].text).toContain('The closest matching tool is "glob"') + // And the list, so a name that is not a near match is still recoverable. + expect(said[0].text).toContain("read, write, glob, bash") + await teardown(h) + }) + + test("a name with no near match gets the list without a wrong suggestion", async () => { + // Naming a bad near-match is worse than naming nothing: the model switches + // to it, fails again, and now believes the registry is unreliable. + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "zzqqxx")), + assistantWith(toolPart("call_2", "zzqqxx")), + ], + }) + await goIdle(h) + const said = suggestions(h) + expect(said).toHaveLength(1) + expect(said[0].text).toContain("Please check the available tools") + expect(said[0].text).not.toContain("The closest matching tool is") + await teardown(h) + }) + + test("the suggestion is sent once, not on every idle", async () => { + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb")), + assistantWith(toolPart("call_2", "globb")), + assistantWith(toolPart("call_3", "globb")), + ], + }) + await goIdle(h) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) + await teardown(h) + }) + + test("two different unknown names are counted separately", async () => { + // Sharing a budget across names would let an unrelated pair of typos silence + // the threshold for the name that actually matters. + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb")), + assistantWith(toolPart("call_2", "wriet")), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("a new user message re-arms the budget", async () => { + // The budget is scoped to one request. A model told to do something new has + // a fresh tool list in front of it, so the previous round's typos say nothing + // about this one. + // + // The second suggestion names the *old* typo, not the new one, and that is + // v1's behaviour reproduced rather than a bug in the port: the re-arm clears + // the "already examined" set, so the walk restarts from the top of the + // history and the oldest name still above threshold wins. Recorded in + // known-issues-v2.md as a quirk worth a decision, not silently changed — + // suppressing a name once it has been suggested is a behaviour change, and + // the project's bar is parity. + const h = await setup({ + history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "globb")), assistantWith(toolPart("call_2", "globb"))], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) + h.history.push(userMessage("now do the other thing", OLD + 2_000, "msg_u_second")) + h.history.push(assistantWith(toolPart("call_3", "wriet"))) + h.history.push(assistantWith(toolPart("call_4", "wriet"))) + await goIdle(h) + expect(suggestions(h)).toHaveLength(2) + expect(suggestions(h)[1].text).toContain('"globb"') + await teardown(h) + }) +}) + +describe("v2: what the unknown-tool check ignores", () => { + test("CONTROL: a tool that exists but errored is not an unknown tool", async () => { + // The dangerous false positive is telling a model a real tool does not + // exist. An error from a registered tool is a real problem, and the right + // response to it is not a rename. + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "read")), + assistantWith(toolPart("call_2", "read")), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: a tool that succeeded is not counted", async () => { + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb", "completed")), + assistantWith(toolPart("call_2", "globb", "completed")), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: a still-running tool is not counted", async () => { + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb", "running")), + assistantWith(toolPart("call_2", "globb", "running")), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("the same call id is never counted twice", async () => { + // The history is re-walked on every idle, so a part that is already counted + // would be counted again on the next pass and the threshold would be reached + // by one real mistake. + const part = toolPart("call_1", "globb") + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith(part), + assistantWith(structuredClone(part)), + assistantWith(structuredClone(part)), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: text and reasoning parts are ignored", async () => { + const h = await setup({ + history: [ + userMessage("go", OLD), + assistantWith({ type: "reasoning", text: "I should call globb" }), + assistantWith({ type: "text", text: "tool: globb" }), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) +}) + +describe("v2: the unknown-tool check degrades rather than guessing", () => { + test("CONTROL: no tool registry means no suggestion, ever", async () => { + // With no registry every name is unknown, so the check would accuse a model + // of inventing tools that plainly exist. + const h = await setup({ + toolRegistry: false, + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb")), + assistantWith(toolPart("call_2", "globb")), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: an empty registry means no suggestion", async () => { + const h = await setup({ + tools: [], + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb")), + assistantWith(toolPart("call_2", "globb")), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + await teardown(h) + }) + + test("a registry that throws is logged and skipped", async () => { + // Failing loudly here would mean a session full of one-line recovery + // prompts, which is worse than losing the feature for a few minutes. + const h = await setup({ + toolListThrows: true, + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb")), + assistantWith(toolPart("call_2", "globb")), + ], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(0) + const logs = existsSync(h.logFile) ? readFileSync(h.logFile, "utf8") : "" + expect(logs).toContain("failed to list tools") + await teardown(h) + }) + + test("a registry entry with no name at all is dropped, not quoted as empty", async () => { + // An unnamed entry would otherwise be joined into the tool list as an empty + // item, and the suggestion would read "... include: , read, write". + const h = await setup({ + tools: [{ id: "read" }, { id: "write" }, { id: "glob" }, {} as any], + history: [ + userMessage("go", OLD), + assistantWith(toolPart("call_1", "globb")), + assistantWith(toolPart("call_2", "globb")), + ], + }) + await goIdle(h) + const said = suggestions(h) + expect(said).toHaveLength(1) + expect(said[0].text).toContain("read, write, glob") + expect(said[0].text).not.toContain(", , ") + await teardown(h) + }) +}) From 3c153862b224dc11abacd7d658a2f7b0732a2013 Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 17:05:51 -0400 Subject: [PATCH 22/57] feat(v2): recover a parent left waiting on a subagent that will not answer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The failure is a parent that dispatches a subagent, the subagent dies without a terminal event, and the parent waits forever on a result that is not coming. Nothing is idle, nothing is busy in a way the ordinary watchdog questions, and every other path declines to act because the silence is shorter than chunkTimeoutMs. The watch is armed from session.idle, when the busy count drops from more than one to exactly one, and it runs ahead of the silence check for exactly that reason. v1's signal is kept; two of its conclusions are not, because v2 can check them. The survivor is asked whether it really has children. The busy-count drop alone is not evidence — two unrelated conversations open at once and one of them finishes looks identical to a parent whose child died, and v1 has no way to tell them apart, so it aborts the survivor either way. One session.list({ parentID }) at arm time turns a guess into a fact. The filter is requested *and* the returned rows are checked against it client-side: the only consequence of the filter being quietly ignored would be aborting real sessions. A child is judged on what it last did, not on whether the server still lists it. v1's check scans the global status map for a busy session and reports "idle" when it finds none — a reading that cannot work here, because this code is only ever reached after a child went idle, so a child being absent from the active set is the premise rather than the finding. Read that way every crashed subagent looks healthy, the child is never woken, and the parent is aborted on the first tick. So a child that said something inside the window is believed; one silent past the threshold with a tool call still outstanding gets a tripled window, because a five-minute build is not a dead model and killing the parent there loses the work; one silent past the threshold, or carrying an errored message, is worth a single attempt to wake. Waking the child is tried once per episode. v1's structure assumes the subagent goes busy again after the prompt, and a dead one never does, so an unbudgeted retry sends the same nudge on every watchdog tick. The sequence is: nudge once, wait a further subagentWaitMs, and abort the parent only if the child is still dead. v2 delivers the prompt through the plugin-scoped synthetic endpoint, which takes an explicit sessionID, so no cross-session route the plugin API does not expose is needed. In-flight tools are tracked from the tool lifecycle events — called increments, success and failed decrement — instead of a tool.execute.before hook, because v2 emits all three and the hook would be a second source of truth for the same fact. Both terminal events matter: a failed tool is finished work, and holding the slot on failure would make the watch defer on every session that has ever seen an error. A session that goes idle has its slots cleared, since a slot held there belongs to a tool that will never report. subagentWaitMs is the guard, and its default of 15s is what makes the abort tolerable: between a child finishing and the parent being interrupted, the parent is given that long to consume the child's result. A parent with a tool in flight is never interrupted, and neither is one waiting on the user. After maxRetries episodes the watch disarms and says so. 18 new tests, a control per behaviour. 755 pass, 0 fail. --- README.md | 4 +- docs/known-issues-v2.md | 67 +++- src/v2/index.options.test.ts | 23 +- src/v2/index.orphan.test.ts | 572 +++++++++++++++++++++++++++++++++++ src/v2/index.ts | 314 ++++++++++++++++++- 5 files changed, 969 insertions(+), 11 deletions(-) create mode 100644 src/v2/index.orphan.test.ts diff --git a/README.md b/README.md index 405c0a5..40466ba 100644 --- a/README.md +++ b/README.md @@ -79,7 +79,7 @@ _Motivated by:_ ### Orphan parent -A subagent finishes but the parent session stays stuck as "busy" forever. The plugin detects when `busyCount` drops from >1 to 1, waits `subagentWaitMs` + `gracePeriodMs` (18s default), probes the subagent (recovering a crashed child first if possible), then aborts and resumes the parent. +A subagent finishes but the parent session stays stuck as "busy" forever. The plugin detects when `busyCount` drops from >1 to 1, confirms the survivor really has subagents, waits `subagentWaitMs` + `gracePeriodMs` (18s default), probes the subagent (nudging a crashed child once, then waiting a further `subagentWaitMs` before giving up on it), and aborts and resumes the parent if it is still stuck. The parent is never interrupted while it has a tool in flight or is waiting on the user. _Motivated by:_ - [#35066](https://github.com/anomalyco/opencode/issues/35066) — notify parent when subagent sessions finish @@ -484,7 +484,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | -Accepted but **not applied** on v2: `subagentWaitMs`, `toolTextCheckDelayMs`. See +Accepted but **not applied** on v2: `toolTextCheckDelayMs`. See [docs/known-issues-v2.md](docs/known-issues-v2.md) for why. Message patterns are matched case-insensitively. Error names use exact match. diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 5f4f493..0807ede 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -12,7 +12,6 @@ config keeps loading unchanged. | Option | Why it is inert on v2 | | --- | --- | -| `subagentWaitMs` | The v1 orphan-watch timer that this delays has no v2 counterpart; the v2 port decides parent-vs-stalled from its own event-derived busy set. | | `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | `doneWithoutDetailsPrompt` and `doneWithoutWorkPrompt` **are** both applied on @@ -170,6 +169,72 @@ handed a fresh budget each time it announced, so the nudge never stopped. v2 still had that, and now does not. The open-todos nudge is the opposite case and does reset per cycle, because an open list is new information each turn. +## Orphan parent recovery + +v1 detects the busy-count drop from more than one session to exactly one and +treats the survivor as a parent whose subagent died. That is the right +*signal* and the wrong *conclusion* on a busy box, and v2 can do better because +v2 records the parent link. + +**The port asks whether the survivor actually has children before arming.** One +`session.list({ parentID })` call at arm time turns a guess into a fact. v1 has no +way to ask — it treats *any other busy session* as a subagent of the one in hand +— so on a box with two conversations open it aborts one of them when the other +finishes. The `parentID` filter is requested *and* the returned rows are checked +against it client-side, because the only consequence of the filter being quietly +ignored would be aborting real sessions. + +**A child is judged on what it last did, not on whether the server still lists +it.** v1's `checkSubagentStatus` scans the global status map for a *busy* +session and reports "idle" when it finds none. That reading cannot work here: +this function is only ever reached *after* a child went idle, so a child being +absent from the active set is the premise rather than the finding. Read that way, +every crashed subagent looks healthy, the child is never woken, and the parent is +aborted on the first tick. So the three cases are: + +- **recent** — whatever the server thinks, the child said something inside the + window, so believe it. If it is also active, keep waiting. +- **silent past the threshold, tool call outstanding** — a long build, not a + dead model. The window is tripled, and the parent is left alone. Killing the + parent there loses the work. +- **silent past the threshold, or an errored message** — it stopped without + reporting. Worth one attempt to wake it. + +**Waking the child is tried once per episode.** v1's structure assumes the subagent +goes busy again after the prompt; a dead one never does, so an unbudgeted +retry sends the same nudge on every watchdog tick. The sequence is: nudge once, +wait a further `subagentWaitMs`, and abort the parent only if the child is still +dead. v1 sends the prompt through the client route; v2 uses the plugin-scoped +`synthetic` endpoint, which takes an explicit `sessionID` and so needs no +cross-session route the plugin API does not expose. + +**In-flight tools are tracked from the tool lifecycle events** — `tool.called` +increments, `tool.success` and `tool.failed` decrement — rather than from a +`tool.execute.before` hook, because v2 emits all three and the hooks would be a +second source of truth for the same fact. Both terminal events matter: a *failed* +tool is finished work, and holding the slot on failure would make the watch defer +on every session that has ever seen an error. A session that goes idle has its +slots cleared, since a slot still held there belongs to a tool that will never +report. + +### What still bounds the risk + +`subagentWaitMs` is the guard. Between a child finishing and the parent being +aborted, the parent is given this long to consume the child's result — the +default is 15s. v1's logic aborts as soon as no subagent is active, and so does +this; the option is what makes that survivable, and its name is the warning. Set +it small and the watch becomes a blunt instrument. Lower it only if you have +watched a parent wait minutes on a dead child. + +Two further guards, both checked before anything is interrupted: a parent with a +tool in flight is never aborted, and neither is one whose newest message is +waiting on the user. + +A give-up path bounds the whole thing: after `maxRetries` episodes the watch +disarms and logs it, rather than retrying a parent that has already survived +several aborts. + + ## Unknown tool names v1 asks `ctx.client.tool.ids()` for the registered tool names. v2 has no such diff --git a/src/v2/index.options.test.ts b/src/v2/index.options.test.ts index e1d7343..3f5c0de 100644 --- a/src/v2/index.options.test.ts +++ b/src/v2/index.options.test.ts @@ -291,11 +291,11 @@ describe("v2: option reporting at startup", () => { }) test("the startup line names the accepted-but-inert options in use", async () => { - const { logs } = await replay([], { chunkTimeoutMs: 5000, subagentWaitMs: 15_000 }) + const { logs } = await replay([], { chunkTimeoutMs: 5000, toolTextCheckDelayMs: 3000 }) const ready = logs.filter((l) => l.includes("ready (opencode v2)")) expect(ready).toHaveLength(1) expect(ready[0]).toContain("accepted-but-inert=") - expect(ready[0]).toContain("subagentWaitMs") + expect(ready[0]).toContain("toolTextCheckDelayMs") }) test("discoveryDelayMs is live on v2, so it is no longer listed as inert", async () => { @@ -304,15 +304,30 @@ describe("v2: option reporting at startup", () => { const { logs } = await replay([], { chunkTimeoutMs: 5000, discoveryDelayMs: 5_000, - subagentWaitMs: 15_000, + toolTextCheckDelayMs: 3000, }) const ready = logs.filter((l) => l.includes("ready (opencode v2)")) expect(ready).toHaveLength(1) expect(ready[0]).toContain("accepted-but-inert=") - expect(ready[0]).toContain("subagentWaitMs") + expect(ready[0]).toContain("toolTextCheckDelayMs") expect(ready[0]).not.toContain("discoveryDelayMs") }) + test("subagentWaitMs is live on v2, so it is no longer listed as inert", async () => { + // The orphan watch reads this option, so a config carrying it must not be told + // it is being ignored — that warning is the only way a user finds out. + const { logs } = await replay([], { + chunkTimeoutMs: 5000, + subagentWaitMs: 15_000, + toolTextCheckDelayMs: 3000, + }) + const ready = logs.filter((l) => l.includes("ready (opencode v2)")) + expect(ready).toHaveLength(1) + expect(ready[0]).toContain("accepted-but-inert=") + expect(ready[0]).toContain("toolTextCheckDelayMs") + expect(ready[0]).not.toContain("subagentWaitMs") + }) + test("the startup line carries no accepted-but-inert list when none are set", async () => { const { logs } = await replay([], { chunkTimeoutMs: 5000 }) const ready = logs.filter((l) => l.includes("ready (opencode v2)")) diff --git a/src/v2/index.orphan.test.ts b/src/v2/index.orphan.test.ts new file mode 100644 index 0000000..d926d5f --- /dev/null +++ b/src/v2/index.orphan.test.ts @@ -0,0 +1,572 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_parent" +const CHILD = "ses_child" + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, sid: string, data: Record = {}) => ({ type, data: { sessionID: sid, ...data } }) + +/** Long timeouts: the orphan watch must fire well inside the silence check's own + * budget, and the point of several tests is that it does NOT wait for that. */ +const OPTIONS = { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + subagentWaitMs: 40, + debug: true, +} + +const userMessage = (text: string, at: number, id = `msg_u_${at}`) => ({ + type: "user", + id, + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantAt = (at: number, extra: unknown[] = [], err?: unknown) => ({ + type: "assistant", + id: "msg_a", + time: { created: at }, + ...(err !== undefined ? { error: err } : {}), + content: [{ type: "text", text: "working on it" }, ...extra], +}) + +type History = Record + +async function setup( + opts: { + history?: History + /** Session ids the server reports busy. */ + active?: string[] + /** Rows `session.list` returns. `parentID` is verified client-side. */ + listRows?: Array<{ id: string; parentID?: string }> + /** Drop `parentID` from list rows, simulating a host that ignores the filter. */ + listIgnoresFilter?: boolean + sessions?: Record + subagentWaitMs?: number + }, +): Promise { + const injected: Array<{ sid?: string; text?: string }> = [] + const interrupted: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-orphan-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const history: History = opts.history ?? { [SID]: [userMessage("go", Date.now() - 60_000)] } + const active = new Set(opts.active ?? []) + const rows = opts.listRows ?? [{ id: SID }, { id: CHILD, parentID: SID }] + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile, ...(opts.subagentWaitMs ? { subagentWaitMs: opts.subagentWaitMs } : {}) }, + session: { + context: async (a: any) => history[a?.sessionID ?? SID] ?? [], + active: async () => Object.fromEntries([...active].map((s) => [s, {}])), + interrupt: async (a: any) => (interrupted.push(a?.sessionID), {}), + synthetic: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + }, + client: { + session: { + get: async ({ path }: any) => ({ + data: { id: path?.id, ...(opts.sessions?.[path?.id] ?? {}) }, + }), + list: async (a: any) => { + // A host that honours parentID filters server-side; one that does + // not returns everything and lets the caller verify. + if (opts.listIgnoresFilter || !a?.parentID) return { data: rows } + return { data: rows.filter((r) => r.parentID === a.parentID) } + }, + }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + + const cleanup = await (plugin as any).setup(ctx) + // Age past the warmup window so the watchdog considers the session at all. + await wait(60) + return { + injected, + interrupted, + history, + active, + ...({ cleanup, logFile, streamRef: stream } as any), + } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const logs = (h: any) => (existsSync(h.logFile) ? readFileSync(h.logFile, "utf8") : "") + +const push = (h: any, type: string, sid: string, data: Record = {}) => h.streamRef.push(ev(type, sid, data)) + +/** Bring a session up busy: an execution and a step, no text. */ +async function makeBusy(h: any, sid: string) { + push(h, "session.execution.started", sid) + await wait(10) + push(h, "session.step.started", sid) + await wait(10) +} + +/** A child that goes busy and then quiet. */ +async function childGoesBusyThenIdle(h: any) { + await makeBusy(h, CHILD) + await wait(20) + // A finished session leaves the server active set. Leaving it in would have + // the watchdog re-mark the child busy, which is the server saying "still + // running" — and the whole premise of the watch is that it stopped. + h.active.delete(CHILD) + push(h, "session.idle", CHILD) + await wait(20) +} + +const A_LONG_WAY_AGO = Date.now() - 10 * 60_000 + +describe("v2: the orphan watch arms when a parent's subagents fall quiet", () => { + test("CONTROL: nothing is armed while only the parent is busy", async () => { + // The control for everything below: the watch is about subagents going + // quiet, not about a parent being busy. A parent alone must be left to the + // ordinary silence check. + const h = await setup({ active: [SID] }) + await makeBusy(h, SID) + await wait(400) + expect(h.interrupted).toHaveLength(0) + expect(logs(h)).not.toContain("orphan watch") + await teardown(h) + }) + + test("CONTROL: a session that outlived a busy session but has no subagents is left alone", async () => { + // Two unrelated conversations open at once and one of them finishes: from the + // busy-count transition alone that is indistinguishable from a parent whose + // child died. v1 cannot tell them apart and aborts the survivor. This one has + // no children at all, so the watch never arms. + const OTHER = "ses_stranger" + const h = await setup({ + active: [SID, OTHER], + history: { [SID]: [userMessage("go", A_LONG_WAY_AGO)] }, + listRows: [{ id: SID }], + }) + await makeBusy(h, SID) + await wait(20) + await makeBusy(h, OTHER) + await wait(20) + h.active.delete(OTHER) + push(h, "session.idle", OTHER) + await wait(500) + expect(logs(h)).not.toContain("orphan watch") + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: a subagent of a different parent does not arm this parent's watch", async () => { + // The child check is on the parentID link, not on "some child went quiet". A + // neighbour's fan-out finishing is not this session's business. + const OTHER_CHILD = "ses_other_child" + const h = await setup({ + active: [SID, OTHER_CHILD], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [OTHER_CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + listRows: [{ id: SID }, { id: OTHER_CHILD, parentID: "ses_their_parent" }], + }) + await makeBusy(h, SID) + await wait(20) + await makeBusy(h, OTHER_CHILD) + await wait(20) + h.active.delete(OTHER_CHILD) + push(h, "session.idle", OTHER_CHILD) + await wait(500) + expect(logs(h)).not.toContain("orphan watch") + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("the watch is armed when the last child goes idle and a parent stays busy", async () => { + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + expect(logs(h)).toContain("orphan watch") + await teardown(h) + }) + + test("the watch does not act before subagentWaitMs has elapsed", async () => { + // subagentWaitMs is the whole knob. Firing early would kill parents that are + // simply slow. + const h = await setup({ + active: [SID, CHILD], + subagentWaitMs: 60_000, + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(400) + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("a busy watch is not re-armed by the clock, so it still fires", async () => { + // If a deferral pushed the start time out instead of leaving it alone, a + // watch could be held off forever by its own deferrals. + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + expect(h.interrupted).toContain(SID) + await teardown(h) + }) +}) + +describe("v2: the orphan watch refuses to kill a working parent", () => { + test("a parent with a tool in flight is never aborted", async () => { + // The single most damaging thing this feature could do. A parent running a + // five-minute build looks exactly like a parent waiting on a dead child. + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + push(h, "session.tool.called", SID, { tool: "bash" }) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + expect(h.interrupted).toHaveLength(0) + expect(logs(h)).toContain("tool(s) in flight") + await teardown(h) + }) + + test("a finished tool releases the slot, and the watch then acts", async () => { + // The control for the test above: if the slot were never released the watch + // would defer forever and this feature would be inert. + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + push(h, "session.tool.called", SID, { tool: "bash" }) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(200) + expect(h.interrupted).toHaveLength(0) + push(h, "session.tool.success", SID, { tool: "bash" }) + await wait(400) + expect(h.interrupted).toContain(SID) + await teardown(h) + }) + + test("a failed tool also releases the slot", async () => { + // A failed tool is finished work. Holding the slot on failure would make the + // watch defer on every session that has ever seen an error. + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + push(h, "session.tool.called", SID, { tool: "bash" }) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(200) + push(h, "session.tool.failed", SID, { tool: "bash" }) + await wait(400) + expect(h.interrupted).toContain(SID) + await teardown(h) + }) + + test("a parent waiting on the user is never aborted", async () => { + // An open question is the user holding the ball, and injecting or aborting + // through that interrupts their answer. + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [ + userMessage("go", A_LONG_WAY_AGO), + { + type: "assistant", + id: "msg_q", + time: { created: A_LONG_WAY_AGO }, + content: [{ type: "tool", id: "c1", name: "question", state: { status: "running", input: {} } }], + }, + ], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("a subagent that is still running is waited for, not treated as dead", async () => { + // The most damaging thing this feature could do: abort a parent in the middle + // of a healthy fan-out. One child has finished, which is what armed the watch; + // the other is mid-run, and the parent is waiting on it. + const LIVE = "ses_child_live" + const h = await setup({ + active: [SID, CHILD, LIVE], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + [LIVE]: [assistantAt(Date.now() - 1_000)], + }, + listRows: [{ id: SID }, { id: CHILD, parentID: SID }, { id: LIVE, parentID: SID }], + }) + await makeBusy(h, SID) + await wait(20) + await makeBusy(h, LIVE) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("a subagent that finished a moment ago is not treated as crashed", async () => { + // The control for the test above, and the reason the stuck fuse is measured + // from the child's last message rather than from when it went idle: a child + // that just finished is a parent making progress, not a parent stuck. + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(Date.now() - 1_000)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(300) + // One attempt to wake it, because a fresh-but-gone child is not "crashed" + // and must not be reported as one. + expect(logs(h)).not.toContain("reported an error") + await teardown(h) + }) + + test("a slow subagent with a tool outstanding gets a longer fuse", async () => { + // The control for the stuck threshold: 90s of silence is past the 60s + // threshold and would read as dead, but the outstanding tool part means it is + // a long build, so the fuse is tripled and the parent is left alone. + const LIVE = "ses_child_slow" + const h = await setup({ + active: [SID, CHILD, LIVE], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + [LIVE]: [ + assistantAt(Date.now() - 90_000, [ + { type: "tool", id: "c9", name: "bash", state: { status: "running", input: {} } }, + ]), + ], + }, + listRows: [{ id: SID }, { id: CHILD, parentID: SID }, { id: LIVE, parentID: SID }], + }) + await makeBusy(h, SID) + await wait(20) + await makeBusy(h, LIVE) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + expect(h.interrupted).toHaveLength(0) + expect(logs(h)).not.toContain("recovery prompt sent") + await teardown(h) + }) +}) + +describe("v2: a dead subagent is woken before the parent is killed", () => { + test("CONTROL: a stuck subagent is woken before the parent is killed", async () => { + // Waking the child is much cheaper than killing the parent: the child is one + // nudge from finishing and the parent's whole turn survives. + // + // subagentWaitMs is raised so the two stages are separately observable. The + // sequence matters and is not "recover, never abort" — it is nudge once, then + // wait a further subagentWaitMs, and only abort the parent if the child is + // still dead. Without the second wait a nudge and an abort would land together. + const h = await setup({ + active: [SID, CHILD], + subagentWaitMs: 400, + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(700) + expect(h.interrupted).toHaveLength(0) + const nudged = h.injected.filter((i: any) => i.sid === CHILD && /stalled or timed out/.test(i.text ?? "")) + expect(nudged).toHaveLength(1) + // And only once, however many ticks pass. + await wait(400) + expect(h.injected.filter((i: any) => i.sid === CHILD)).toHaveLength(1) + await teardown(h) + }) + + test("a child that stays dead is escalated on the parent after a further wait", async () => { + // The control for the nudge: if the child never comes back, the parent is the + // only thing left to save, and it is waiting on a result that will not arrive. + const h = await setup({ + active: [SID, CHILD], + subagentWaitMs: 200, + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(900) + expect(h.interrupted).toContain(SID) + await teardown(h) + }) + + test("a crashed subagent with no way to reach it falls back to aborting the parent", async () => { + // A crashed child cannot be woken. The parent is the only remaining thing to + // save, and it is waiting on a result that will never arrive. + const h = await setup({ + active: [SID, CHILD], + sessions: { [CHILD]: {} }, + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [{ type: "assistant", id: "msg_err", time: { created: A_LONG_WAY_AGO }, error: "stream disconnected", content: [] }], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + expect(h.interrupted).toContain(SID) + await teardown(h) + }) +}) + +describe("v2: the subagent list is verified, not trusted", () => { + test("CONTROL: a host that ignores the parentID filter does not cause a wrong abort", async () => { + // The filter is requested AND checked. If it were only requested, a host that + // ignored it would hand back every session on the box and the watch would + // read a stranger's children as this parent's. + const h = await setup({ + active: [SID, CHILD], + listIgnoresFilter: true, + listRows: [ + { id: SID }, + { id: CHILD, parentID: SID }, + { id: "ses_unrelated" }, + ], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + // ses_unrelated is not this session's child, so the busy child here is CHILD + // and the verdict is the same either way — the point is that nothing else in + // the list was treated as a child and no stranger was touched. + expect(h.interrupted).not.toContain("ses_unrelated") + await teardown(h) + }) + + test("a list that returns nothing means the watch is never armed", async () => { + // No rows could mean the API is unavailable rather than childless. Arming on + // that reading would abort a parent on the strength of a failed query, so an + // empty list abstains — and abstaining only costs a missed recovery. + const h = await setup({ + active: [SID, CHILD], + listRows: [], + history: { + [SID]: [userMessage("go", A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + }, + }) + await makeBusy(h, SID) + await wait(20) + await childGoesBusyThenIdle(h) + await wait(500) + expect(h.interrupted).toHaveLength(0) + expect(logs(h)).toContain("has no subagents") + await teardown(h) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index 4a3b3b0..ac783f9 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -189,6 +189,20 @@ interface SessionWatch { taskCompleteSignals: number /** Failed tool calls by name, for the unknown-tool suggestion. Counted, not * latched: one bad call is a typo, two is a model that will not self-correct. */ + /** When the orphan watch was armed: the moment the last busy subagent of this + * session went idle. A parent still busy after that is the case this exists for. + * null means not armed. */ + orphanWatchStartAt: number | null + /** Last time the subagents of this session were polled, so the watchdog does not + * ask the server on every tick. */ + lastSubagentCheckAt: number + /** Whether this orphan episode has already tried waking its subagent. Without a + * budget the same prompt goes out on every watchdog tick — v1's structure + * assumed the subagent goes busy again, and a dead one never does. */ + orphanRecoveryTried: boolean + /** Tool calls started but not finished, tracked from the tool lifecycle events. + * The orphan watch must never abort a session that is legitimately working. */ + pendingTools: number unknownToolErrors: Map /** Set once a suggestion has been injected, so it is sent at most once per * user message rather than on every idle. */ @@ -387,6 +401,12 @@ const DEFAULT_WARMUP_MS = 15_000 const DEFAULT_MIN_ACTIVITY_GAP_MS = 1_000 const DEFAULT_TOOL_TEXT_CHECK_DELAY_MS = 3_000 const DEFAULT_SUBAGENT_WAIT_MS = 15_000 +/** How long a busy subagent may go without producing anything before it counts + * as stuck. v1 used the same number, and a tool call still outstanding triples + * it, because a long tool is not a hung model. */ +const SUBAGENT_STUCK_MS = 60_000 +const SUBAGENT_RECOVERY_PROMPT = + "It looks like you may have stalled or timed out. Please retry the last operation or continue with the task." const DEFAULT_SILENT_DEAD_STREAM_MIN_TOKENS = 200 const DEFAULT_CONTEXT_SATURATION_THRESHOLD = 0.85 /** How long a fetched todo list is trusted before being re-read. */ @@ -942,6 +962,9 @@ export default define({ const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS const logFile = opts.logFile ?? process.env.AUTO_RESUME_LOG_FILE ?? DEFAULT_LOG_FILE + // How long a parent may sit busy after its last subagent went idle before + // the orphan watch acts. v1 default, honoured for the first time here. + const subagentWaitMs = opts.subagentWaitMs ?? DEFAULT_SUBAGENT_WAIT_MS // ---- Ported from v1 (Mte90/opencode-auto-resume#33) ---- const rawBusyStallStrategy = opts.busyStallStrategy ?? "continue" @@ -1002,10 +1025,7 @@ export default define({ // is not ported. Listed (not warned) so an existing v1 config stays valid // and the gap is documented rather than surprising. See // docs/known-issues-v2.md. - const FEATURE_GATED_OPTIONS = [ - "subagentWaitMs", - "toolTextCheckDelayMs", - ] as const + const FEATURE_GATED_OPTIONS = ["toolTextCheckDelayMs"] as const // Options this build understands. Anything else in the user's config is // reported once at startup so a silent fallback is visible rather than @@ -1154,6 +1174,10 @@ export default define({ todosFetchedAt: 0, taskCompleteSignals: 0, taskCompleteOverrides: 0, + orphanWatchStartAt: null, + orphanRecoveryTried: false, + lastSubagentCheckAt: 0, + pendingTools: 0, unknownToolErrors: new Map(), unknownToolSuggestionSent: false, checkedToolPartIDs: new Set(), @@ -1207,6 +1231,10 @@ export default define({ // A new turn re-opens the question of whether the work is finished. w.completionSignaled = false w.gaveUp = false + // A new turn means the parent moved on, so whatever the previous orphan + // watch was about is no longer the question. Left armed, it would fire + // mid-turn and abort a session that is doing exactly what it should. + w.orphanWatchStartAt = null w.recentToolCalls = [] w.textParts.clear() w.lastAssistantText = "" @@ -1222,6 +1250,14 @@ export default define({ dbg(`${short(sid)} ${w.status} -> idle`) w.status = "idle" w.idleSince = Date.now() + // Idle means nothing is in flight. Any slot still held belongs to a tool + // that will never report — a subagent killed mid-call, for instance — + // and keeping it would make the orphan watch defer on this session + // forever. + w.pendingTools = 0 + // This session is not the parent the watch is about: the watch is armed on + // the session that *stays* busy, so an idle session's own watch is void. + w.orphanWatchStartAt = null } // NOTE: this used to clear `permissionPending`. The `session.idle` handler // calls markIdle() and then inspectOnIdle(), so the flag was always false by @@ -2318,6 +2354,245 @@ export default define({ } } + /** Busy sessions we are tracking, excluding cancelled ones. The orphan watch + * keys on this count: a subagent finishing is only interesting because it + * leaves exactly one session busy. */ + function busySessionsForOrphanWatch(): string[] { + const out: string[] = [] + for (const [sid, w] of sessions) { + if (w.status === "busy" && !w.userCancelled) out.push(sid) + } + return out + } + + /** + * The subagents of `parentSid`, by v2's own parent link. + * + * v1 had no way to ask, so it treated *any other busy session* as a subagent of + * the one in hand. That works on a single-project box and misfires on a busy + * one: a second unrelated session looks like a child, and the parent gets + * aborted for someone else's work. v2 stores `parent_id` on the session row, so + * the question can simply be asked. + * + * The filter is requested *and* verified client-side. `parentID` is part of + * the list input, but the only consequence of it being quietly ignored would be + * aborting real sessions, so the returned rows are checked rather than trusted. + */ + async function listSubagentIds(parentSid: string): Promise { + const listed = unwrapList(await callSessionApi("list", { parentID: parentSid })) + const out: string[] = [] + for (const row of listed) { + const sid = row?.id + if (typeof sid !== "string" || !sid.startsWith("ses_")) continue + if (sid === parentSid) continue + if ((row as { parentID?: unknown }).parentID !== parentSid) continue + out.push(sid) + } + return out + } + + type SubagentVerdict = { status: "crashed" | "idle" | "busy"; stuckSid?: string } + + /** + * What a parent's subagents are doing, in v1's three-way shape. + * + * v1 scans the global status map for any other *busy* session and treats a + * quiet one as "idle" — the parent is stuck with nothing to wait for. That + * reading cannot work on v2, and the reason is worth stating: this function is + * only ever reached *after* a child went idle, so a child being absent from + * the active set is the premise, not the finding. Reading that absence as + * "idle" would make every crashed subagent look like a healthy one, and the + * whole feature would abort the parent without ever trying to wake the child. + * + * So a child is judged on what it last did, not on whether the server still + * lists it: + * + * - **active, and recent** — running. Wait. + * - **active, and silent past the threshold** — hung. The fuse is tripled while + * a tool call is still outstanding, because a five-minute build is not a + * dead model and killing the parent there loses the work. + * - **not active, recent** — it finished. Nothing to wait for. + * - **not active, silent past the threshold** — it stopped without ever + * reporting. Waking it is worth one attempt. + */ + async function subagentVerdict(parentSid: string, activeIDs: Set): Promise { + try { + const children = await listSubagentIds(parentSid) + if (children.length === 0) return { status: "idle" } + const now = Date.now() + let sawBusy = false + for (const child of children) { + const messages = await loadMessages(child) + const last = messages[messages.length - 1] as + | { + type?: string + error?: unknown + time?: { created?: number } + content?: Array<{ type?: string; state?: { status?: string } }> + } + | undefined + if (last?.type === "assistant" && last.error !== undefined) { + dbg(`subagent ${short(child)} reported an error`) + return { status: "crashed" } + } + const msgTime = last?.time?.created + const silentFor = typeof msgTime === "number" ? now - msgTime : Infinity + // An unanswered tool part is the "still working, slowly" case. + const hasToolCall = (last?.content ?? []).some( + (p) => p?.type === "tool" && p.state?.status !== "completed" && p.state?.status !== "error", + ) + const limit = hasToolCall ? SUBAGENT_STUCK_MS * 3 : SUBAGENT_STUCK_MS + if (silentFor <= limit) { + // Recent enough to be believed, whatever the server thinks. + if (activeIDs.has(child)) sawBusy = true + continue + } + dbg( + `subagent ${short(child)} silent for ${Math.round(silentFor / 1000)}s (active=${activeIDs.has(child)})`, + ) + return { status: "crashed", stuckSid: child } + } + return sawBusy ? { status: "busy" } : { status: "idle" } + } catch (e) { + dbg(`subagent check failed for ${short(parentSid)}:`, e instanceof Error ? e.message : String(e)) + // Unknown is treated as busy: the cost of waiting is a later abort, and + // the cost of guessing wrong is a killed session. + return { status: "busy" } + } + } + + /** + * Nudge a stuck subagent directly. + * + * v1 uses the client prompt route. v2's plugin-scoped `session.synthetic` + * takes an explicit sessionID and appends the message to that session, which is + * the same effect without needing a cross-session route the plugin API does + * not expose. + * + * The general `injectOnce` refuses subagents, and correctly so: a child is + * not ours to recover. This is the one exception, and it is deliberate — the + * parent is stuck *because* the child is, so waking the child is cheaper than + * killing the parent and its whole turn. + */ + async function recoverStuckSubagent(sid: string): Promise { + try { + await callSessionApi("synthetic", { sessionID: sid, text: SUBAGENT_RECOVERY_PROMPT }) + log("info", `${short(sid)} recovery prompt sent to stuck subagent`) + return true + } catch (e) { + log("warn", `failed to recover subagent ${short(sid)}: ${e instanceof Error ? e.message : String(e)}`) + return false + } + } + + /** + * The orphan watch. + * + * The failure it exists for: a parent dispatches a subagent, the subagent dies + * without a terminal event, and the parent waits forever on a result that is + * never coming. Nothing is busy, nothing is idle-from-the-runtime's-point-of-view, + * and every ordinary watchdog path declines to act because the silence is shorter + * than `chunkTimeoutMs`. + * + * Armed from `session.idle`: when the busy count drops from more than one to + * exactly one, the survivor is a parent whose children have all gone quiet. + * After `subagentWaitMs` the watchdog asks what the children are doing and, if + * there is no live work left, aborts and resumes the parent. + */ + /** + * Arm the watch on a session that just outlived its subagents — but only if it + * really had some. + * + * The busy-count drop alone is not evidence. Two unrelated conversations open + * at once, one of them finishes, and the survivor looks identical to a parent + * whose child died: one session left, busy. v1 has no way to tell them apart + * and aborts the survivor either way, which on a busy box means killing + * somebody's conversation for a colleague's finishing turn. + * + * v2 can tell them apart, so it does: one listing call at arm time turns a + * guess into a fact. The cost is one query per 2-to-1 transition, and the + * alternative is an abort that cannot be undone. + */ + async function armOrphanWatch(sid: string): Promise { + const w = sessions.get(sid) + if (!w || w.orphanWatchStartAt !== null) return + if (w.status !== "busy" || w.userCancelled || w.completionSignaled) return + try { + if (await isSubAgentSession(sid)) return + const children = await listSubagentIds(sid) + // Re-checked after the await: the session may have gone idle, been + // cancelled, or been picked up by another arming while we were listing. + if (w.orphanWatchStartAt !== null || w.status !== "busy" || w.userCancelled) return + if (children.length === 0) { + dbg(`${short(sid)} outlived a busy session but has no subagents — not an orphan parent`) + return + } + w.orphanWatchStartAt = Date.now() + w.orphanRecoveryTried = false + log( + "info", + `subagents of ${short(sid)} fell quiet while it stayed busy. orphan watch (${subagentWaitMs / 1000}s, ${children.length} subagent(s))`, + ) + } catch (e) { + dbg(`orphan arming check failed for ${short(sid)}:`, e instanceof Error ? e.message : String(e)) + } + } + + async function runOrphanWatch(sid: string, w: SessionWatch, now: number, activeIDs: Set): Promise { + if (now - w.orphanWatchStartAt! < subagentWaitMs + gracePeriodMs) return + if (w.resumeAttempts >= maxRetries) { + if (!w.gaveUp) { + w.gaveUp = true + w.orphanWatchStartAt = null + log("warn", `${short(sid)} orphan watch gave up after ${w.resumeAttempts} attempts`) + } + return + } + // Never abort a parent that is running a tool. Two independent sources, + // because the counter can miss a tool that started before the plugin loaded: + // our own event-derived count, and the tool parts in the message history. + if (w.pendingTools > 0) { + dbg(`${short(sid)} parent has ${w.pendingTools} tool(s) in flight — deferring orphan abort`) + w.orphanWatchStartAt = now + return + } + if (hasPendingUserInput(await loadMessages(sid))) { + dbg(`${short(sid)} parent is waiting on the user — deferring orphan abort`) + w.orphanWatchStartAt = now + return + } + + const verdict = await subagentVerdict(sid, activeIDs) + if (verdict.status === "crashed") { + // Waking the child is cheaper than killing the parent, so it is tried + // first — but exactly once. If the child does not come back, the parent + // is the only thing left to save and the watch stops asking. + if (verdict.stuckSid && !w.orphanRecoveryTried) { + w.orphanRecoveryTried = true + if (await recoverStuckSubagent(verdict.stuckSid)) { + w.orphanWatchStartAt = now + return + } + } + log("info", `${short(sid)} subagent crashed and did not recover — aborting and resuming the parent`) + tryAbortAndResume(sid, w) + return + } + if (verdict.status === "busy") { + dbg(`${short(sid)} subagents still working — waiting`) + w.orphanWatchStartAt = now + return + } + // No subagent is active and none is stuck enough to name. v1 aborts here, + // and so does this — the thing standing between that and a wrong kill is + // subagentWaitMs, which is how long the parent was given to consume a + // finished child's result before this runs at all. Set it to something + // small and this becomes a blunt instrument; the option's own default is + // 15s, and its name is the warning. + log("info", `${short(sid)} stuck with no live subagents — aborting and resuming`) + tryAbortAndResume(sid, w) + } + async function checkActiveSessions() { const now = Date.now() @@ -2328,6 +2603,7 @@ export default define({ const w = ensureWatch(sid) if (w.status !== "busy") markBusy(sid) } + const activeSet = new Set(activeIDs) for (const [sid, w] of sessions) { if (w.status !== "busy" || w.userCancelled) continue @@ -2335,6 +2611,16 @@ export default define({ // emit anything. Without this a freshly-started turn can be declared // stalled while it is still queueing its first model call. if (now - w.createdAt < warmupMs) continue + // The orphan watch comes first, ahead of the silence check below, + // because that is the whole point of it: a parent waiting on a dead + // subagent has been quiet for less than chunkTimeoutMs, so every + // ordinary path would decline to act. + if (w.orphanWatchStartAt !== null) { + if (!w.aborting && !w.gaveUp && !w.completionSignaled) { + await runOrphanWatch(sid, w, now, activeSet) + } + continue + } // Stale compaction flag: a compaction that never reports // ended/failed within the TTL is wedged — clear the guard so // recovery can eventually intervene. @@ -2748,6 +3034,11 @@ export default define({ case "session.idle": { const sid = sidOf(ev) if (!sid) return + // Counted across every tracked session, and read *before* this one goes + // idle, because the transition is the signal: a session leaving a crowd + // of busy sessions behind it. Sampling after markIdle would make every + // idle look like a drop to zero and the arming condition unreachable. + const busyBefore = busySessionsForOrphanWatch().length markIdle(sid) const w = ensureWatch(sid) w.pendingRecoveryArmed = false @@ -2759,6 +3050,18 @@ export default define({ dbg(`${short(sid)} idle after plugin-initiated abort — skipping targeted recovery`) return } + // Arm the orphan watch. A subagent going idle is only interesting because + // of what it leaves behind: when the busy count drops from more than one + // to exactly one, the survivor is a parent whose children have all gone + // quiet, and it may be waiting for a result that is never coming. + // + // Re-arming is guarded: an already-armed watch keeps its start time, or a + // stream of child idles would keep pushing the deadline out. + const busyNow = busySessionsForOrphanWatch() + if (busyBefore > 1 && busyNow.length === 1) { + void armOrphanWatch(busyNow[0]) + } + void inspectOnIdle(sid) // Independent of the idle heuristics above: this looks at tool parts // that already errored, which none of those read. Fire-and-forget @@ -2931,6 +3234,7 @@ export default define({ const w = ensureWatch(sid) const name = typeof ev.data?.tool === "string" ? ev.data.tool : (ev.data?.id as string | undefined) ?? "tool" w.lastWasTaskTool = isTaskToolCall(ev) + w.pendingTools++ // Any real tool work between completions legitimises the next // task_complete — reset the repeat-signal counter. task_complete // itself is excluded; it manages the counter inside its execute. @@ -2990,6 +3294,7 @@ export default define({ if (!sid) return const w = ensureWatch(sid) w.lastWasTaskTool = false + w.pendingTools = Math.max(0, w.pendingTools - 1) touch(sid) return } @@ -2998,6 +3303,7 @@ export default define({ if (!sid) return const w = ensureWatch(sid) w.lastWasTaskTool = false + w.pendingTools = Math.max(0, w.pendingTools - 1) touch(sid) return } From 537b2d39fc0149241785a7bb20f78b2a3e107bb5 Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 17:30:00 -0400 Subject: [PATCH 23/57] feat(v2): judge a finished turn once its text has settled, not the instant it ends MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v1 never judged a finished turn on the `session.idle` event. It armed a timer for `toolTextCheckDelayMs` and ran the done/tool checks from there, and the option is that delay. The reason is a race, not a preference: `session.idle` can arrive while the assistant's closing text is still being written into the message history, so a check that reads too early sees a half-finished turn — missing the pattern it should have caught, or judging a turn that was not over. The v2 build judged on the idle event for most of its life, on the strength of the live delta buffer: the plugin sees text as it streams, so it usually has the final turn before `session.idle` arrives, which is more reliable than waiting rather than less. It does not remove the race. The buffer is empty when the plugin loads mid-turn, and the fallback in exactly that case is the history that may not have flushed. That is why this option was the last one still accepted and not applied. So the work splits in two, and the option tunes the half that needs it. The structural pass runs on the idle event: is this a dead stream, is it handing control back to the user, is the user already busy. None of those read the final text, and the first of them cannot wait — a stream that died before delivering any text is precisely the case where every pattern check has nothing to look at, so deferring it would only delay a recovery that is already certain. The pattern pass runs `toolTextCheckDelayMs` later and re-reads the history, which is the whole point: text that landed after the idle event is what it judges. It also re-runs the structural guards rather than trusting the first pass's verdict. Three seconds is long enough for the user to have replied, and a synthetic nudge fired through a real reply interrupts the step their reply started. This is the one place the split could have made things worse, because the guards are the reason a nudge is safe and the deferral is the reason they could be consulted too late. Three lifecycle details, because a timer that outlives what armed it is a new way to nudge a session that has moved on. A new turn cancels a pending pass: `markBusy` clears the timer, and the timer re-checks that the session is still idle, because a queued callback can outlive the turn that armed it. A second idle replaces the first rather than stacking — two passes on one turn would spend two attempts of one budget on one piece of text, and the second would be a nudge about a nudge. And stopping the plugin drops any pending pass, or a reload leaves a timer pointing at a session this build is no longer watching. `FEATURE_GATED_OPTIONS` is now empty: every option this build recognises is applied. The list is kept rather than deleted, because it is the only signal a user gets that an option is being ignored, and the next unported feature needs it more than this build does. With no production path left to it, the invariant is asserted on the source instead — a test fails if an option is added to the recognised set without being implemented, which is exactly how this build went from several inert options to none. 11 new tests. Every one is mutation-checked: removing the delay, the arming, the history re-read, either guard re-run, the cancel latch, the new-turn cancel or the cleanup turns the file red. 766 pass, 0 fail. --- README.md | 37 ++-- docs/known-issues-v2.md | 71 ++++++- src/v2/index.dead-stream.test.ts | 1 + src/v2/index.discovery.test.ts | 8 +- src/v2/index.open-shell.test.ts | 1 + src/v2/index.options.test.ts | 64 ++++--- src/v2/index.orphan.test.ts | 1 + src/v2/index.premature-stop.test.ts | 1 + src/v2/index.saturation.test.ts | 1 + src/v2/index.settle-delay.test.ts | 283 ++++++++++++++++++++++++++++ src/v2/index.stand-down.test.ts | 1 + src/v2/index.task-complete.test.ts | 1 + src/v2/index.thinking-tool.test.ts | 1 + src/v2/index.todo.test.ts | 1 + src/v2/index.ts | 90 ++++++++- src/v2/index.unknown-tool.test.ts | 1 + 16 files changed, 510 insertions(+), 53 deletions(-) create mode 100644 src/v2/index.settle-delay.test.ts diff --git a/README.md b/README.md index 40466ba..646c279 100644 --- a/README.md +++ b/README.md @@ -436,19 +436,35 @@ Disable via `"-auto-resume.v2"` in `plugins`. - **Migration notes (v1 → v2, stable validation):** [docs/v2/migration.md](docs/v2/migration.md) - **What the v2 port does and does not yet do:** [docs/known-issues-v2.md](docs/known-issues-v2.md) -The v2 build reads every option in the table below. A handful of them are -accepted without being applied yet, because the v2 feature they tune is not -ported — they are listed in the startup line as `accepted-but-inert=…` and -explained in [docs/known-issues-v2.md](docs/known-issues-v2.md). An option name -this build does not know at all produces a single warning at startup, so a typo -or an unported key is never silent. +### How a finished turn is judged + +`session.idle` does not decide anything by itself. The turn is inspected in two +passes, because `session.idle` can arrive while the assistant's closing text is +still being written into the message history: + +1. **Straight away** — is this a dead stream, is it handing control back to the + user, is the user already busy. None of these read the final text. +2. **After `toolTextCheckDelayMs`** (3s default) — the celebration, + tool-call-as-text, ready-to-continue, action-intent and done-claim detectors, + re-reading the history. This is the pass the option exists for, and it re-runs + the guards from step 1, because three seconds is long enough for the user to + have replied and a nudge sent over them interrupts their own step. + +A new turn cancels a pending second pass, and a second idle replaces it rather +than stacking another judgement on the same text. See +[docs/known-issues-v2.md](docs/known-issues-v2.md). + +The v2 build reads and applies every option in the table below — the last of +the gaps closed with `toolTextCheckDelayMs`. Any option it accepts but does not +apply is listed in the startup line as `accepted-but-inert=…` and explained in +[docs/known-issues-v2.md](docs/known-issues-v2.md); that list is currently empty. +An option name this build does not know at all produces a single warning at +startup, so a typo or a dropped port is never silent. ### Configurable options Defaults are the same on v1 and v2 unless a row says otherwise. -### Configurable options - | Option | Default | Description | |---|---|---| | `chunkTimeoutMs` | `45000` | Inactivity timeout before considering stream stalled | @@ -463,7 +479,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `streamingFailureErrorNames` | `["ProviderError","APIError","StreamError","ConnectionError","TimeoutError"]` | Error names that classify as streaming failures (exact match) | | `streamingFailureMessagePatterns` | `["streaming response failed","stream.*fail","connection.*reset","connection.*closed"]` | Regex patterns (case-insensitive) in error messages indicating streaming failure | | `maxRecoveryRetries` | `2` | Max streaming-failure recovery attempts before abort+resume escalation | -| `toolTextCheckDelayMs` | `3000` | Delay before scanning an idle session for tool-as-text; also the recovery watchdog delay | +| `toolTextCheckDelayMs` | `3000` | Settle delay before a finished turn's closing text is judged against the done/tool patterns; also the recovery watchdog delay. On v2 only the pattern half is deferred — dead streams are still caught on the idle event, because a stream that died before delivering any text has nothing for those patterns to read. Set `0` to judge on the idle event | | `minActivityGapMs` | `1000` | Skip recovery if the session was active within this gap | | `warmupMs` | `15000` | Action-intent detection disabled while a session is younger than this | | `debug` | `false` | Enable `[debug]` console diagnostics | @@ -484,9 +500,6 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | -Accepted but **not applied** on v2: `toolTextCheckDelayMs`. See -[docs/known-issues-v2.md](docs/known-issues-v2.md) for why. - Message patterns are matched case-insensitively. Error names use exact match. ### Internal constants (not configurable) diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 0807ede..e07b2cb 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -6,13 +6,19 @@ a config key that this build ignores is never silent. ## Options that are accepted but not applied -These keys are read without error, and are listed in the plugin's startup line as -`accepted-but-inert=…`. They stay in `AutoResumeOptions` so an existing v1 -config keeps loading unchanged. +**None.** Every option this build recognises is now applied, and the startup line +no longer carries an `accepted-but-inert=…` note. -| Option | Why it is inert on v2 | -| --- | --- | -| `toolTextCheckDelayMs` | The delayed raw-tool-call-as-text re-check is v1's polling shape. v2 evaluates the text once, on idle, from the authoritative message history. | +The machinery is kept rather than deleted. It is the only signal a user gets that +an option is being ignored, and the next unported feature needs it more than this +build does. `docs/known-issues-v2.md` and the README inert list are checked +against `FEATURE_GATED_OPTIONS` in the source, and a test asserts the list is +empty — so adding an option to the recognised set without implementing it cannot +pass quietly. + +Two gaps closed on this branch, in order: `subagentWaitMs` and +`discoveryDelayMs` (orphan-parent recovery and session discovery), then +`toolTextCheckDelayMs` (below). `doneWithoutDetailsPrompt` and `doneWithoutWorkPrompt` **are** both applied on v2, and they are the two halves of v1's done-claim handling: the first asks for a @@ -169,6 +175,59 @@ handed a fresh budget each time it announced, so the nudge never stopped. v2 still had that, and now does not. The open-todos nudge is the opposite case and does reset per cycle, because an open list is new information each turn. +## The settle delay before a turn is judged + +v1 never judges a finished turn on the `session.idle` event. It arms a timer for +`toolTextCheckDelayMs` and runs the done/tool checks from there, and the option +is the delay. That is not incidental: `session.idle` can arrive while the +assistant's closing text is still being written into the message history, so a +check that reads too early sees a half-finished turn — missing the pattern it +should have caught, or judging a turn that was not over. + +The v2 port judged on the idle event for most of its life, on the strength of the +live delta buffer: the plugin sees text as it streams, so it usually has the final +turn before `session.idle` arrives, which is more reliable than waiting. It does +not remove the race. The buffer is empty when the plugin loads mid-turn, and the +fallback in that case is precisely the history that may not have flushed. + +So the work is split in two, and the option tunes the half that needs it: + +| Pass | When | What it decides | +|---|---|---| +| structural | on `session.idle` | dead stream, user hand-off, user recently active | +| pattern | `+` `toolTextCheckDelayMs` | celebration, tool-call-as-text, ready-to-continue, action intent, done-claim | + +The structural pass is immediate on purpose. A stream that died before delivering +any text is exactly the case where every pattern check has nothing to look at, so +waiting would only delay a recovery that is already certain. + +The pattern pass re-reads the history rather than reusing the first read, which +is the whole point: text that landed after the idle event is what it judges. It +also **re-runs the structural guards** instead of trusting the first pass's +verdict, because the wait is long enough for the user to have replied — and a +synthetic nudge sent over a real reply is the "Step interrupted" bug those guards +exist to prevent. + +Three lifecycle details, all covered by tests: + +- A **new turn cancels** a pending pass. `markBusy` clears the timer, and the + timer itself re-checks that the session is still idle: a queued callback can + outlive the turn that armed it. +- A **second idle replaces** the first rather than stacking a second judgement on + the same text. Two passes on one turn would spend two attempts of one budget on + one piece of text, and the second would be a nudge about a nudge. +- **Stopping the plugin drops** any pending pass. Otherwise a reload leaves a + timer pointing at a session this build is no longer watching, and it fires + minutes later against a history nobody is reading. + +### One difference worth naming + +v1's `checkForToolCallAsText` bundles every text-based detector behind this one +timer. v2's detectors were separate before the option was honoured, so the delay +applies to the pattern pass as a whole rather than to the tool-call-as-text check +specifically. Nothing is lost by it — every detector that reads the closing text +gets the settle window, which is what the delay was for. + ## Orphan parent recovery v1 detects the busy-count drop from more than one session to exactly one and diff --git a/src/v2/index.dead-stream.test.ts b/src/v2/index.dead-stream.test.ts index b3218e8..2efca04 100644 --- a/src/v2/index.dead-stream.test.ts +++ b/src/v2/index.dead-stream.test.ts @@ -97,6 +97,7 @@ const toolStep = () => ({ const OPTIONS = { chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, checkIntervalMs: 20, gracePeriodMs: 0, warmupMs: 0, diff --git a/src/v2/index.discovery.test.ts b/src/v2/index.discovery.test.ts index b129d75..6828c07 100644 --- a/src/v2/index.discovery.test.ts +++ b/src/v2/index.discovery.test.ts @@ -120,7 +120,7 @@ async function run( const ctx: any = { event: stream, - options: { warmupMs: 0, discoveryDelayMs: 20, logFile, ...opts }, + options: { warmupMs: 0, discoveryDelayMs: 20, toolTextCheckDelayMs: 0, logFile, ...opts }, session, } if (client) ctx.client = client @@ -177,7 +177,7 @@ describe("v2: session discovery sweep", () => { const rows = [{ id: "not-a-session" }, { id: 42 }, {}, null] const ctx: any = { event: stream, - options: { warmupMs: 0, discoveryDelayMs: 20, logFile }, + options: { warmupMs: 0, discoveryDelayMs: 20, toolTextCheckDelayMs: 0, logFile }, session: { context: async () => [], active: async () => ({}), @@ -197,7 +197,7 @@ describe("v2: session discovery sweep", () => { const stream = makeEventStream() const ctx: any = { event: stream, - options: { warmupMs: 0, discoveryDelayMs: 20, logFile }, + options: { warmupMs: 0, discoveryDelayMs: 20, toolTextCheckDelayMs: 0, logFile }, session: { context: async () => [], active: async () => ({ [RUNNING]: { type: "running" } }), @@ -217,7 +217,7 @@ describe("v2: session discovery sweep", () => { const ctx: any = { event: stream, // Tight watchdog so a leaked interval is visible inside the wait. - options: { warmupMs: 0, discoveryDelayMs: 20, checkIntervalMs: 20, chunkTimeoutMs: 10_000, logFile }, + options: { warmupMs: 0, discoveryDelayMs: 20, checkIntervalMs: 20, chunkTimeoutMs: 10_000, toolTextCheckDelayMs: 0, logFile }, session: { context: async () => [], active: async () => { diff --git a/src/v2/index.open-shell.test.ts b/src/v2/index.open-shell.test.ts index 5318abc..4eb8385 100644 --- a/src/v2/index.open-shell.test.ts +++ b/src/v2/index.open-shell.test.ts @@ -95,6 +95,7 @@ async function replay(events: any[], messages?: any[]): Promise { const ctx: any = { event: stream, app: { log: () => {} }, + options: { toolTextCheckDelayMs: 0 }, session: { context: async () => messages ?? [oldUserTurn(), assistantTurn(READY_TEXT)], // Empty: no other session is active, so the `lastWasTaskTool` branch diff --git a/src/v2/index.options.test.ts b/src/v2/index.options.test.ts index 3f5c0de..ec79b4b 100644 --- a/src/v2/index.options.test.ts +++ b/src/v2/index.options.test.ts @@ -102,7 +102,10 @@ async function replay( // The plugin reads its config from `ctx.options`; the second argument to // `setup()` is ignored. Passing options the other way makes every // negative assertion below pass for the wrong reason. - options: { ...opts, logFile }, + // Before the spread, so it applies to every test here unless one asks for a + // different value. The deferred pattern pass defaults to 3s, which is not what + // these tests measure — the nudge they assert on would land after the wait. + options: { toolTextCheckDelayMs: 0, ...opts, logFile }, session: { context: async () => [oldUserTurn(), assistantTurn(text)], // Empty: no other session is active, so the `lastWasTaskTool` branch @@ -290,42 +293,45 @@ describe("v2: option reporting at startup", () => { expect(logs.filter((l) => l.includes("unrecognised"))).toEqual([]) }) - test("the startup line names the accepted-but-inert options in use", async () => { - const { logs } = await replay([], { chunkTimeoutMs: 5000, toolTextCheckDelayMs: 3000 }) - const ready = logs.filter((l) => l.includes("ready (opencode v2)")) - expect(ready).toHaveLength(1) - expect(ready[0]).toContain("accepted-but-inert=") - expect(ready[0]).toContain("toolTextCheckDelayMs") - }) - - test("discoveryDelayMs is live on v2, so it is no longer listed as inert", async () => { - // Control: a genuinely inert option, set the same way, IS still listed. - // Without this the assertion below would pass for the wrong reason. + test("the startup line names no inert options, because none are inert", async () => { + // Every v1 option this build understands is now applied, so the inert list has + // nothing to put in it — including for the three options that were inert when + // this branch started. A config carrying them must not be told they are being + // ignored: that note is the only way a user finds out. const { logs } = await replay([], { chunkTimeoutMs: 5000, - discoveryDelayMs: 5_000, + subagentWaitMs: 15_000, toolTextCheckDelayMs: 3000, + discoveryDelayMs: 5_000, }) const ready = logs.filter((l) => l.includes("ready (opencode v2)")) expect(ready).toHaveLength(1) - expect(ready[0]).toContain("accepted-but-inert=") - expect(ready[0]).toContain("toolTextCheckDelayMs") + expect(ready[0]).not.toContain("accepted-but-inert=") + expect(ready[0]).not.toContain("subagentWaitMs") + expect(ready[0]).not.toContain("toolTextCheckDelayMs") expect(ready[0]).not.toContain("discoveryDelayMs") }) - test("subagentWaitMs is live on v2, so it is no longer listed as inert", async () => { - // The orphan watch reads this option, so a config carrying it must not be told - // it is being ignored — that warning is the only way a user finds out. - const { logs } = await replay([], { - chunkTimeoutMs: 5000, - subagentWaitMs: 15_000, - toolTextCheckDelayMs: 3000, - }) - const ready = logs.filter((l) => l.includes("ready (opencode v2)")) - expect(ready).toHaveLength(1) - expect(ready[0]).toContain("accepted-but-inert=") - expect(ready[0]).toContain("toolTextCheckDelayMs") - expect(ready[0]).not.toContain("subagentWaitMs") + test("CONTROL: nothing is listed as feature-gated in the source", async () => { + // The structural counterpart to the assertion above, and the one that actually + // holds the line: adding an option to RECOGNISED_OPTIONS without implementing it + // is exactly how this build went from "several options inert" to "none", and it + // is the failure the list exists to make visible. With the list empty there is + // no behavioural test left for it — no production path reaches it — so it is + // asserted on the source instead. + const block = SOURCE.match(/const FEATURE_GATED_OPTIONS = \[([^\]]*)\]/) + expect(block).not.toBeNull() + const gated = [...(block![1].match(/"([a-zA-Z][a-zA-Z0-9]*)"/g) ?? [])].map((x) => x.replace(/"/g, "")) + expect(gated).toEqual([]) + }) + + test("the inert reporting is kept, so the next gap has somewhere to be listed", async () => { + // Deliberately not deleted when the last entry left the list. The reporting is + // the only signal a user gets that an option is being ignored, and the next + // unported feature needs it more than this build does. + expect(SOURCE).toContain("FEATURE_GATED_OPTIONS") + expect(SOURCE).toContain("accepted-but-inert=") + expect(SOURCE).toContain("gatedInUse") }) test("the startup line carries no accepted-but-inert list when none are set", async () => { @@ -401,7 +407,7 @@ describe("v2: contract assertions on source", () => { expect(block).not.toBeNull() const recognised = new Set([...(block![1].match(/"([a-zA-Z][a-zA-Z0-9]*)"/g) ?? [])].map((s) => s.replace(/"/g, ""))) // Sampled across every category: prompts, timing, pattern lists, the - // v1-only alias, and the three that v2 recognises but does not act on. + // v1-only alias, and the two that were inert until this branch ported them. for (const key of [ "continuePrompt", "actionIntentPrompt", diff --git a/src/v2/index.orphan.test.ts b/src/v2/index.orphan.test.ts index d926d5f..fcaaa71 100644 --- a/src/v2/index.orphan.test.ts +++ b/src/v2/index.orphan.test.ts @@ -50,6 +50,7 @@ const ev = (type: string, sid: string, data: Record = {}) => ({ * budget, and the point of several tests is that it does NOT wait for that. */ const OPTIONS = { chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, checkIntervalMs: 20, gracePeriodMs: 0, warmupMs: 0, diff --git a/src/v2/index.premature-stop.test.ts b/src/v2/index.premature-stop.test.ts index ccaa307..858cd85 100644 --- a/src/v2/index.premature-stop.test.ts +++ b/src/v2/index.premature-stop.test.ts @@ -100,6 +100,7 @@ const OPTIONS = { // Long enough that the stall watchdog never fires: these tests are about the // idle path only, and a stall injection would mask which path produced a nudge. chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, checkIntervalMs: 20, gracePeriodMs: 0, warmupMs: 0, diff --git a/src/v2/index.saturation.test.ts b/src/v2/index.saturation.test.ts index 5df393f..6693aca 100644 --- a/src/v2/index.saturation.test.ts +++ b/src/v2/index.saturation.test.ts @@ -138,6 +138,7 @@ async function replay(h: HarnessOpts) { event: stream, options: { chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, checkIntervalMs: 20, gracePeriodMs: 0, warmupMs: 0, diff --git a/src/v2/index.settle-delay.test.ts b/src/v2/index.settle-delay.test.ts new file mode 100644 index 0000000..ee90d50 --- /dev/null +++ b/src/v2/index.settle-delay.test.ts @@ -0,0 +1,283 @@ +import { describe, test, expect } from "bun:test" +import { rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_settle" + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +/** An hour old, so nothing is suppressed for "the user is mid-conversation". */ +const OLD = Date.now() - 60 * 60_000 + +/** Trips `ready-to-continue` and is not a user hand-off. */ +const READY_TEXT = "Ready to continue with task" +/** Trips `ready-to-continue` *and* is a hand-off — the guards must win. + * Deliberately the control's text plus a bare "?", so the sentence still matches + * the pattern and the only reason to stay quiet is the hand-off check. */ +const READY_HANDOFF_TEXT = "Ready to continue with task?" + +const userTurn = () => ({ type: "user", id: "msg_u", time: { created: OLD }, content: [{ type: "text", text: "go" }] }) + +/** + * A finished assistant message. `noText` gives it reasoning but no text part and + * enough output tokens, which is what a stream that died mid-response looks like. + */ +const assistantTurn = (text: string | null) => ({ + type: "assistant", + id: "msg_a", + time: { created: OLD + 1_000 }, + content: text === null ? [{ type: "reasoning", text: "thinking about the answer" }] : [{ type: "text", text }], + finish: "stop", + tokens: { input: 100, output: 400, reasoning: 0, cache: { read: 0, write: 0 } }, +}) + +async function setup(opts: { text?: string | null; toolTextCheckDelayMs?: number }): Promise { + const injected: Array<{ text?: string; description?: string }> = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-settle-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + // Mutable on purpose: the point of several tests below is what the deferred + // pass reads, so the history has to be able to change after the first idle. + let history: unknown[] = [userTurn(), assistantTurn(opts.text === undefined ? "Working on it." : opts.text)] + + const ctx: any = { + event: stream, + options: { + chunkTimeoutMs: 600_000, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + toolTextCheckDelayMs: 0, + logFile, + ...(opts.toolTextCheckDelayMs === undefined ? {} : { toolTextCheckDelayMs: opts.toolTextCheckDelayMs }), + }, + session: { + context: async () => history, + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ text: a?.text, description: a?.description }), {}), + prompt: async (a: any) => (injected.push({ text: a?.text, description: a?.description }), {}), + }, + client: { session: { get: async () => ({ data: { id: SID } }), list: async () => ({ data: [{ id: SID }] }) } }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { + transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), + list: async () => [], + }, + } + + const cleanup = await (plugin as any).setup(ctx) + return { + injected, + setText: (t: string) => { + history = [userTurn(), assistantTurn(t)] + }, + /** The user replies *during* the delay: a fresh message, not the old one. */ + setFreshUser: (t: string) => { + history = [ + { type: "user", id: "msg_u2", time: { created: Date.now() }, content: [{ type: "text", text: "hold on" }] }, + assistantTurn(t), + ] + }, + cleanup, + logFile, + streamRef: stream, + } +} + +async function teardown(h: any) { + h.cleanup?.() + rmSync(h.logFile, { force: true }) +} + +async function goIdle(h: any) { + for (const e of [ev("session.execution.started"), ev("session.step.started"), ev("session.step.ended"), ev("session.idle")]) { + h.streamRef.push(e) + await wait(10) + } +} + +describe("v2: the settle delay lets a finished turn's text arrive before it is judged", () => { + test("CONTROL: a zero delay judges the turn straight away", async () => { + // The control for every timing assertion below. Without it, a test that + // "passed because the nudge never came" would be indistinguishable from one + // that proved the delay works. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 0 }) + await goIdle(h) + await wait(250) + expect(h.injected).toHaveLength(1) + await teardown(h) + }) + + test("a long delay holds the nudge back", async () => { + // The option's whole effect. v1 judges from a timer rather than on the idle + // event, and a check that reads too early sees a half-finished turn. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 3_000 }) + await goIdle(h) + await wait(300) + expect(h.injected).toHaveLength(0) + await teardown(h) + }) + + test("and releases it once the delay has passed", async () => { + // The other half of the control above: a deferral that never fires would + // look exactly like a working delay from the test before it. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 300 }) + await goIdle(h) + await wait(800) + expect(h.injected).toHaveLength(1) + await teardown(h) + }) + + test("the deferred pass sees text that only landed after the first idle", async () => { + // The reason the delay exists, stated as a test. At the idle event the last + // message said something else entirely; by the time the pass runs, the + // assistant's real closing turn is in the history. Judging on idle alone + // would have seen the wrong turn and stayed silent. + const h = await setup({ text: "Working on it.", toolTextCheckDelayMs: 300 }) + await goIdle(h) + h.setText(READY_TEXT) + await wait(800) + expect(h.injected).toHaveLength(1) + await teardown(h) + }) + + test("a turn that starts during the delay cancels the pending judgement", async () => { + // Otherwise the deferred pass would read the live delta buffer, which a new + // turn has already emptied, and judge the new turn against the old turn's + // verdict. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 400 }) + await goIdle(h) + await wait(50) + h.streamRef.push(ev("session.execution.started")) + await wait(20) + h.streamRef.push(ev("session.step.started")) + await wait(600) + expect(h.injected).toHaveLength(0) + await teardown(h) + }) + + test("a second idle inside the window replaces the first, rather than stacking", async () => { + // Two passes on one turn would spend two attempts of one budget on one piece + // of text, and the second attempt would be a nudge about a nudge. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 250 }) + await goIdle(h) + await wait(40) + await goIdle(h) + await wait(700) + expect(h.injected).toHaveLength(1) + await teardown(h) + }) +}) + +describe("v2: the settle delay defers the patterns, not the structural checks", () => { + test("CONTROL: a dead stream is still caught immediately", async () => { + // The reason the split exists. A stream that died before delivering any + // text is precisely the case where every text-based check has nothing to + // look at, so waiting would only delay a recovery that is certain. + const h = await setup({ text: null, toolTextCheckDelayMs: 3_000 }) + await goIdle(h) + await wait(300) + expect(h.injected).toHaveLength(1) + await teardown(h) + }) + + test("a turn that became a hand-off during the delay is not nudged", async () => { + // The pattern phase re-runs the guards instead of trusting the structural + // pass's verdict, because the wait is long enough for the user to have + // replied. Setting this up as a *change* is what makes it a test of that: + // with the hand-off present from the start, the structural pass refuses and + // never arms anything, so the assertion would hold whether or not the + // pattern phase looked at the text again. + // + // The closing line is a bare question mark, which is the only difference + // between this and the control's text — so the sentence still matches + // ready-to-continue, and it stays quiet because of the guard. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 200 }) + await goIdle(h) + h.setText(READY_HANDOFF_TEXT) + await wait(700) + expect(h.injected).toHaveLength(0) + await teardown(h) + }) + + test("a user who replies during the delay stops the nudge", async () => { + // The second reason the pattern phase re-runs the guards. The structural + // pass saw a user message from an hour ago and judged the turn fair game; + // three seconds later the user is mid-conversation, and a nudge sent over + // them is the "Step interrupted" bug every guard in this file exists for. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 300 }) + await goIdle(h) + h.setFreshUser(READY_TEXT) + await wait(800) + expect(h.injected).toHaveLength(0) + await teardown(h) + }) + + test("a cancelled session is not nudged after the delay", async () => { + // The user interrupting mid-wait is the loudest possible "stop". + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 300 }) + await goIdle(h) + await wait(30) + h.streamRef.push(ev("session.execution.interrupted")) + await wait(800) + expect(h.injected).toHaveLength(0) + await teardown(h) + }) + + test("stopping the plugin drops a pending judgement", async () => { + // Otherwise a reload would leave a timer pointing at a session this build is + // no longer watching, and it would fire minutes later against a history + // nobody is reading. + const h = await setup({ text: READY_TEXT, toolTextCheckDelayMs: 400 }) + await goIdle(h) + await wait(30) + h.cleanup() + await wait(800) + expect(h.injected).toHaveLength(0) + rmSync(h.logFile, { force: true }) + }) +}) diff --git a/src/v2/index.stand-down.test.ts b/src/v2/index.stand-down.test.ts index 16371b7..0052d55 100644 --- a/src/v2/index.stand-down.test.ts +++ b/src/v2/index.stand-down.test.ts @@ -102,6 +102,7 @@ async function replay(opts: { messages: any[]; events: any[] }): Promise {} }, + options: { toolTextCheckDelayMs: 0 }, session: { // v2 returns a plain ARRAY here, not { messages }. Asserted separately below. context: async () => opts.messages, diff --git a/src/v2/index.task-complete.test.ts b/src/v2/index.task-complete.test.ts index a773453..68686d8 100644 --- a/src/v2/index.task-complete.test.ts +++ b/src/v2/index.task-complete.test.ts @@ -67,6 +67,7 @@ const ev = (type: string, data: Record = {}) => ({ type, data: const OPTIONS = { chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, checkIntervalMs: 20, gracePeriodMs: 0, warmupMs: 0, diff --git a/src/v2/index.thinking-tool.test.ts b/src/v2/index.thinking-tool.test.ts index 0c57de8..ebff4be 100644 --- a/src/v2/index.thinking-tool.test.ts +++ b/src/v2/index.thinking-tool.test.ts @@ -105,6 +105,7 @@ const TEXT_WITH_TOOL_CALL = [{ type: "text", text: " | null /** Tool calls started but not finished, tracked from the tool lifecycle events. * The orphan watch must never abort a session that is legitimately working. */ pendingTools: number @@ -965,6 +969,10 @@ export default define({ // How long a parent may sit busy after its last subagent went idle before // the orphan watch acts. v1 default, honoured for the first time here. const subagentWaitMs = opts.subagentWaitMs ?? DEFAULT_SUBAGENT_WAIT_MS + // How long to let a finished turn's text settle before judging it against the + // done/tool patterns. v1's default, and the reason v1 does not judge on idle + // at all — see inspectOnIdle. + const toolTextCheckDelayMs = opts.toolTextCheckDelayMs ?? DEFAULT_TOOL_TEXT_CHECK_DELAY_MS // ---- Ported from v1 (Mte90/opencode-auto-resume#33) ---- const rawBusyStallStrategy = opts.busyStallStrategy ?? "continue" @@ -1025,7 +1033,13 @@ export default define({ // is not ported. Listed (not warned) so an existing v1 config stays valid // and the gap is documented rather than surprising. See // docs/known-issues-v2.md. - const FEATURE_GATED_OPTIONS = ["toolTextCheckDelayMs"] as const + // + // Empty: every v1 option this build understands is now applied. Kept as a + // list rather than deleted because the reporting below, and the + // accepted-but-inert= note in the startup line, are the only things a user + // has to go on when an option is quietly ignored — so the next one that + // arrives has somewhere to be listed. + const FEATURE_GATED_OPTIONS = [] as const // Options this build understands. Anything else in the user's config is // reported once at startup so a silent fallback is visible rather than @@ -1176,6 +1190,7 @@ export default define({ taskCompleteOverrides: 0, orphanWatchStartAt: null, orphanRecoveryTried: false, + toolTextTimer: null, lastSubagentCheckAt: 0, pendingTools: 0, unknownToolErrors: new Map(), @@ -1228,6 +1243,12 @@ export default define({ // same thing. w.todoNudgeAttempts = 0 w.intentNudgeAttempts = 0 + // A new turn means the text a pending pattern pass was going to judge is + // superseded, and markBusy has just emptied the buffer it would have read. + if (w.toolTextTimer) { + clearTimeout(w.toolTextTimer) + w.toolTextTimer = null + } // A new turn re-opens the question of whether the work is finished. w.completionSignaled = false w.gaveUp = false @@ -1694,6 +1715,32 @@ export default define({ } } + /** + * Arm the deferred pattern pass, replacing any already pending. + * + * Replacement rather than stacking is the point: two idles inside one delay + * window would otherwise judge the same turn twice, and a turn that matches + * would spend two attempts of its budget on one piece of text. + */ + function schedulePatternPass(sid: string): void { + const w = ensureWatch(sid) + if (w.toolTextTimer) clearTimeout(w.toolTextTimer) + w.toolTextTimer = setTimeout(() => { + w.toolTextTimer = null + // v1's own guard, and it is doing real work here: a turn that started + // during the delay has its own idle, and its own armed pass. Judging + // this one now would read the live buffer, which markBusy has already + // emptied for the new turn. + if (w.status !== "idle") { + dbg(`${short(sid)} pattern pass skipped — session is ${w.status}, not idle`) + return + } + if (w.userCancelled) return + void inspectOnIdle(sid, "pattern") + }, toolTextCheckDelayMs) + dbg(`${short(sid)} pattern pass armed for +${toolTextCheckDelayMs}ms`) + } + async function targetedRecovery( sid: string, kind: string, @@ -2213,7 +2260,35 @@ export default define({ } } - async function inspectOnIdle(sid: string) { + /** + * Judge a finished turn — in two phases, because the text may not have settled. + * + * v1 evaluates the done/tool patterns from a timer armed on the idle event + * (`toolTextCheckDelayMs`, 3s by default) rather than on the event itself, and + * the delay is the feature: `session.idle` can arrive while the assistant's + * final text is still being written into the message history, and a check that + * reads too early sees a half-finished turn and either misses the pattern or + * judges a turn that was not over. + * + * v2 already avoids most of that race with the live delta buffer, which is + * why this port was able to judge on idle for as long as it did. It does not + * remove it: the buffer is empty when the plugin loaded mid-turn, and the + * fallback is exactly the history that may not have flushed. So the option is + * honoured by splitting the work: + * + * - **structural** (immediately): is this a dead stream, is it handing control + * to the user, is the user already busy. None of these depend on the final + * text, and a dead stream must be caught before anything that reads text. + * - **pattern** (after the delay): the celebration, tool-call-as-text, ready-to- + * continue, action-intent and done-claim detectors. These are exactly the + * ones that read the last text and so are exactly the ones the settle delay + * exists for. + * + * The guards above the split run again in the pattern phase, deliberately: three + * seconds is long enough for the user to have replied, and a nudge fired through + * their reply is the bug the hand-off and stand-down guards exist to prevent. + */ + async function inspectOnIdle(sid: string, phase: "structural" | "pattern" = "structural") { const w = ensureWatch(sid) const messages = await loadMessages(sid) noteInboundUserMessage(sid, w, messages) @@ -2243,6 +2318,11 @@ export default define({ return } + if (phase === "structural") { + schedulePatternPass(sid) + return + } + // The model's own completion signal. A trailing 🎉 means it considers the // work finished, so nudging here would talk over a deliberate stop. // Latched rather than re-derived, because the next idle with no new text @@ -3422,6 +3502,12 @@ export default define({ clearInterval(watchdog) clearInterval(discoveryTimer) clearTimeout(initialDiscovery) + // A pending pattern pass outlives the plugin otherwise, and would judge a + // session against a history this build is no longer watching. + for (const w of sessions.values()) { + if (w.toolTextTimer) clearTimeout(w.toolTextTimer) + w.toolTextTimer = null + } sessions.clear() log("info", "stopped") } diff --git a/src/v2/index.unknown-tool.test.ts b/src/v2/index.unknown-tool.test.ts index 19ccf8e..9779572 100644 --- a/src/v2/index.unknown-tool.test.ts +++ b/src/v2/index.unknown-tool.test.ts @@ -47,6 +47,7 @@ const ev = (type: string, data: Record = {}) => ({ type, data: const OPTIONS = { chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, checkIntervalMs: 20, gracePeriodMs: 0, warmupMs: 0, From 263a8e8f7617f97b6ade0aca0a2de60c92b4e91e Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 17:51:27 -0400 Subject: [PATCH 24/57] docs(v2): the install guide was describing a plugin this is not MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `docs/v2/installing.md` opened with a mandatory install step: bun add @opencode/plugin@2.0.5 That was true of the beta port, which imported the API package. This one does not. `src/v2/index.ts` imports only `node:fs`, `node:os` and `node:path`, and declares its own `{ id, setup }` helper, so the built artefact is a single self-contained file and a local-file install needs no install step at all. A user following the old guide was told to install a package the plugin never loads, and the troubleshooting table carried a row for a `Cannot find package '@opencode/plugin'` failure this build cannot produce. So: the install step is gone, the requirement that the package "must be resolvable by opencode" is gone, the impossible troubleshooting row is gone, and `@opencode/plugin` is now described where it still matters — as the types the source is checked against in `bunx tsc --noEmit`. Three more things in those two files had stopped being true: - `chunkTimeoutMs` was documented as 45000 in the config example, the options table and the expected startup banner. It has been 180000 on both builds since upstream `49957b2`; the port had made the same change independently. - The options table listed 14 of 33 options with nothing pointing at the rest. It now carries the ones that matter *on v2* — the settle delay, and the two v2-only options that exist because v2 removed a capability (`logFile`, since there is no server log sink, and `injectIntervalMs`) — and links the README for the full table. It also says the port reads and applies all 31 of v1's options, which is now true. - `docs/v2/migration.md` still claimed a 12-test suite. It is 13 files and 188 tests, and it now says what those tests are for: the two vacuous ones in the settle-delay file, and why every test in the port is mutation-checked. The README's v2 section gets the same correction, since it is the page most people read instead of the guide. --- README.md | 2 +- docs/v2/installing.md | 85 +++++++++++++++++++------------------------ docs/v2/migration.md | 70 ++++++++++++++++++++++++----------- 3 files changed, 86 insertions(+), 71 deletions(-) diff --git a/README.md b/README.md index 38ae453..0f74124 100644 --- a/README.md +++ b/README.md @@ -412,7 +412,7 @@ With options: ### OpenCode v2 (stable) -The v2 plugin uses the `Plugin.define` API with `ctx.event.subscribe()` (AsyncIterable) instead of the v1 hooks-object pattern, and targets the stable **`@opencode/plugin` 2.0.5** API (opencode v2.0.5+). Add to your `opencode.json`: +The v2 plugin uses the `{ id, setup }` plugin shape with `ctx.event.subscribe()` (AsyncIterable) instead of the v1 hooks-object pattern, and is written against the stable v2 API (opencode 2.0.5+). It has **no runtime dependencies** — it defines the plugin helper locally and imports only `node:fs`, `node:os` and `node:path` — so there is nothing to `bun add`. Add it to your `opencode.json`: ```jsonc { diff --git a/docs/v2/installing.md b/docs/v2/installing.md index f82263a..0be5f92 100644 --- a/docs/v2/installing.md +++ b/docs/v2/installing.md @@ -9,33 +9,19 @@ plugin API (`@opencode/plugin` 2.0.5, shipped with `opencode` v2.0.5). ## Requirements - **opencode v2.0.5 or newer** (`opencode --version`). -- **`@opencode/plugin` must be resolvable by opencode.** A local plugin is loaded - with a normal ESM import, so the API package has to be installed next to it - (see [Step 1](#step-1--install-the-plugin-api-package)). Without it opencode - logs `failed to load plugin … Cannot find package '@opencode/plugin'`. -- Node.js/Bun is only needed to install the API package and run the - test/typecheck tooling; opencode itself loads the plugin. -- The plugin source: [`src/v2/index.ts`](../../src/v2/index.ts) from this repo. +- **No runtime dependencies.** `src/v2/index.ts` imports only `node:fs`, + `node:os` and `node:path`, and defines its own `{ id, setup }` helper rather + than importing one, so the built file is self-contained. `@opencode/plugin` is + not needed to *run* the plugin — only to typecheck the source against the + official types (see [Development](#development)). +- The plugin source: [`src/v2/index.ts`](../../src/v2/index.ts) from this repo, or + the built `dist/v2/index.js`. ## Install -### Step 1 — install the plugin API package +There is nothing to install first. Copy the file, and opencode loads it. -Global plugins resolve imports from `~/.config/opencode`, so install the stable -API package there once: - -```sh -cd ~/.config/opencode -bun add @opencode/plugin@2.0.5 # or: npm install @opencode/plugin@2.0.5 -``` - -For a per-project plugin, install it in the project instead: - -```sh -npm install @opencode/plugin@2.0.5 -``` - -### Step 2 — add the plugin +### Step 1 — add the plugin #### Option A — drop-in file (no config) @@ -70,7 +56,7 @@ pin a published package. Add an entry to the `plugins` array: { "package": "./plugins/auto-resume-v2.ts", "options": { - "chunkTimeoutMs": 45000, + "chunkTimeoutMs": 180000, "maxRetries": 3 } } @@ -81,32 +67,28 @@ pin a published package. Add an entry to the `plugins` array: Both `.opencode/plugin/` (v1 directory name) and `.opencode/plugins/` are discovered; use `.opencode/plugins/` for v2 files. -### Step 3 — restart opencode +### Step 2 — restart opencode -Restart opencode after installing. The plugin loader resolves plugin -dependencies when the server starts, so a hot file reload is **not** enough to -pick up a newly installed `@opencode/plugin` package. +Restart opencode after copying the file. A hot plugin reload is **not** +reliable on every version, so restart when in doubt — a missing banner is the +symptom to check for. ## Options -All options are optional and are read from `ctx.options` in `setup`. +Every option is optional and is read from `ctx.options` in `setup`. The port +reads **all** of v1's options, and applies all of them — there are none left in an +"accepted but not applied" state. The full table with every default is in the +[README](../../README.md#configurable-options); the ones worth knowing about on +v2 specifically: | Option | Default | Description | |---|---|---| -| `chunkTimeoutMs` | `45000` | Silence on a **busy** session before recovery is considered. | -| `gracePeriodMs` | `3000` | Extra grace added to the timeout before acting. | -| `checkIntervalMs` | `5000` | Watchdog polling interval. | -| `maxRetries` | `3` | Resume attempts per stall before escalating. | -| `baseBackoffMs` | `1000` | Base delay for exponential backoff between attempts. | -| `maxBackoffMs` | `8000` | Ceiling for the backoff delay. | -| `loopMaxContinues` | `3` | Continues allowed inside `loopWindowMs` before forcing interrupt + resume. | -| `loopWindowMs` | `600000` | Window (ms) for the hallucination-loop guard (default 10 min). | -| `maxRecoveryRetries` | `2` | Cap for targeted recovery prompts (tool-as-text / intent nudges). | -| `continuePrompt` | `"continue"` | Prompt used for a plain resume / ready-to-continue nudge. | -| `toolTextRecoveryPrompt` | _(built-in)_ | Prompt used when a tool call is printed as text instead of executed. | -| `doneWithoutWorkPrompt` | _(built-in)_ | Prompt used to verify a suspicious terse "done" claim. | -| `actionIntentPrompt` | _(falls back to `continuePrompt`)_ | Prompt used when the model ends with an unexecuted intent. | -| `debug` | `false` | Verbose `[auto-resume:debug]` logging. | +| `chunkTimeoutMs` | `180000` | Silence on a **busy** session before recovery is considered. Matches v1 since `49957b2`. | +| `toolTextCheckDelayMs` | `3000` | Settle delay before a finished turn's closing text is judged against the done/tool patterns. Dead streams are still caught on the idle event, because a stream that died before delivering any text has nothing for those patterns to read. `0` judges on the idle event. | +| `logFile` | `~/.local/state/opencode-v2/auto-resume.log` | Where this build logs. **v2 removed v1's server log endpoint**, so without this the plugin has no log at all; `AUTO_RESUME_LOG_FILE` in the environment overrides it. Size-capped at 2 MB. | +| `injectIntervalMs` | `15000` | Minimum gap between recovery injections for one session. No v1 equivalent. | +| `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.compact()` for a saturated subagent, instead of leaving it alone. | +| `debug` | `false` | Verbose `[auto-resume:debug]` logging, appended to `logFile`. | Example: @@ -123,16 +105,17 @@ Example: ## Verify it loaded -Start opencode; the plugin logs its banner at load: +Start opencode; the plugin logs its banner at load, into `logFile`: ``` -[auto-resume] ready (opencode v2). timeout=45000ms interval=5000ms retries=3 loop=3/600s +[auto-resume] ready (opencode v2). timeout=180000ms interval=5000ms retries=3 loop=3/600s warmup=15000ms stall=continue ``` To see an intervention in action, let a session go quiet past `chunkTimeoutMs`; the plugin injects a visible **synthetic** message in the session timeline (`auto-resume: …`) and resumes the turn. Every recovery is -also appended to the opencode log with an `[auto-resume]` prefix. +appended to `logFile` with an `[auto-resume]` prefix — the opencode log no longer +has a plugin sink to write to, which is why `logFile` exists. ## Disable / uninstall @@ -146,7 +129,7 @@ also appended to the opencode log with an `[auto-resume]` prefix. | Symptom | Check | |---|---| | No `[auto-resume] ready …` banner | File is in a `plugins/` dir opencode scans, or listed in `plugins`; restart opencode. | -| Logs say `failed to load plugin … Cannot find package '@opencode/plugin'` | Install the API package next to the plugin (Step 1) **and restart** opencode. | +| Logs say `failed to load plugin` with no reason | The file is not a valid ES module, or a syntax error. Run `bunx tsc --noEmit` on it (see [Development](#development)). | | Nothing happens on a stall | Increase verbosity with `"debug": true`; confirm `chunkTimeoutMs` isn't larger than your real stall. | | Recovers but you don't see a notice | Your model/provider may reject `session.synthetic()`; the plugin falls back to `session.prompt()` (no TUI banner). | | Never recovers a parent waiting on a subagent | Intentional: parent sessions blocked on a running subagent are left alone. | @@ -154,11 +137,17 @@ also appended to the opencode log with an `[auto-resume]` prefix. ## Development +The source has no imports from `@opencode/plugin` — the `{ id, setup }` helper +is defined locally — so the package is only needed to check the file against the +official types: + ```sh -# typecheck the v2 port against the stable types bun add -d @opencode/plugin@2.0.5 typescript bunx tsc --noEmit --strict --target ESNext --module ESNext \ --moduleResolution bundler --skipLibCheck src/v2/index.ts + +# the distributable is a single self-contained file +bun build src/v2/index.ts --outfile dist/v2/index.js --target bun ``` See [`migration.md`](./migration.md) for the full v1→v2 mapping and the diff --git a/docs/v2/migration.md b/docs/v2/migration.md index 907bfee..191173d 100644 --- a/docs/v2/migration.md +++ b/docs/v2/migration.md @@ -2,7 +2,10 @@ > Drop this into the PR description (or keep as docs/v2-migration.md upstream). > Target: https://github.com/Mte90/opencode-auto-resume -> Runtime: **opencode v2.0.5 (stable)** with **@opencode/plugin 2.0.5**. +> Runtime: **opencode v2.0.5 (stable)**. The port has no runtime dependency — +> `src/v2/index.ts` imports only `node:fs`, `node:os` and `node:path`, and +> defines its own `{ id, setup }` helper. `@opencode/plugin@2.0.5` is used to +> typecheck the source, not to run it. > Originally ported against opencode2 v0.0.0-beta-18050 / @opencode-ai/plugin > 0.0.0-next-17403; re-validated against the stable release (see §7). @@ -11,9 +14,9 @@ Ports the plugin from the v1 hooks API (`Plugin` factory returning a hooks object) to the v2 promise-plugin API (`Plugin.define({ id, setup })` + `ctx.event.subscribe()`). All detection/recovery features are preserved. -Strict-mode typechecked against the real `@opencode/plugin@2.0.5` types; -validated by a mocked-context runtime suite covering every recovery -path (12/12 checks). +Strict-mode typechecked against the real `@opencode/plugin@2.0.5` types, and +covered by 188 tests across 13 files driven through the real event stream — one +file per feature area, 774 passing repo-wide including v1. ## 1. Config key renamed: `plugin` → `plugins` @@ -150,19 +153,29 @@ respected. ## Options -Unchanged names/defaults from v1: `chunkTimeoutMs` (45000), `gracePeriodMs` -(3000), `checkIntervalMs` (5000), `maxRetries` (3), `baseBackoffMs` (1000), -`maxBackoffMs` (8000), `loopMaxContinues` (3), `loopWindowMs` (600000), -`maxRecoveryRetries` (2), `debug` (false), plus the prompt overrides -`continuePrompt`, `toolTextRecoveryPrompt`, `doneWithoutWorkPrompt`, -`actionIntentPrompt`. +The port reads **all 31 of v1's options** and applies **all 31**. Names and +defaults are unchanged, and the full table is in the +[README](../../README.md#configurable-options). Two defaults moved after both +builds picked them up from upstream: `chunkTimeoutMs` to 180000 (`49957b2`) and +`activeUserWindowMs` to 300000 (`e1b8374`). + +Two options are **v2-only**, with no v1 equivalent, because v2 removed the +capability they configure: + +- `logFile` — v2 has no server log endpoint, so the plugin writes its own + (2 MB cap, `AUTO_RESUME_LOG_FILE` overrides). +- `injectIntervalMs` — a floor on how often one session may be nudged. + +For the record, this list is checked against the source in both directions: a test +fails if an option joins the recognised set without being implemented, and the +doc check fails if the table names one that is not gated. ```jsonc { "plugins": [ { "package": "./plugins/auto-resume-v2.ts", - "options": { "chunkTimeoutMs": 45000, "maxRetries": 3 } + "options": { "chunkTimeoutMs": 180000, "maxRetries": 3 } } ] } @@ -173,23 +186,36 @@ Disable by id without touching other plugins: add `"-auto-resume.v2"`. ## Testing - `tsc --strict --noEmit` clean against **`@opencode/plugin@2.0.5`** (stable). -- Mocked-context runtime suite (bun, 12 tests) against the stable package: +- Runtime suite (bun, 13 files / 188 tests) against the stable API: definition shape · `subscribe({ signal })` · abort-on-cleanup · stall watchdog → `synthetic` with visible `description` + `resume` · idle forensics via the `session.context()` fallback · healthy idle turn is a no-op · permission-hold blocks recovery · `interrupted` `reason="user"` disables recovery · `reason="inactivity"` does **not** · execution-failure recovery · `synthetic`→`prompt` fallback · `session.deleted` drops state — - **12/12 passing**. + Each later feature area has its own file: unknown tool suggestions, + silent dead streams, premature stop, context saturation, the todo list, explicit + `task_complete`, reasoning-tool recovery, orphan parent recovery, the settle + delay, session discovery, the options surface, and the hand-off and stand-down + guards. + + Two tests in the settle-delay file were **vacuous** on the first pass — they + passed whether or not the code under test ran. Every test in the v2 port is now + mutation-checked: removing the delay, the arming, the history re-read, either + guard re-run, the cancel latch, the new-turn cancel or the cleanup each turns + the file red. A negative test that never fails is indistinguishable from a + detector that never runs. ## 7. Re-validated against stable v2 (2.0.5) The port was originally written against the beta (`@opencode-ai/plugin` `0.0.0-next-17403`). Re-checked against the stable release: -- **Package renamed**: the dependency is now **`@opencode/plugin`** (stable - `2.0.5`), published from the opencode repo. The beta `@opencode-ai/plugin` - package is not the stable API. Imports must use `@opencode/plugin`. +- **Package renamed**: the API types are now **`@opencode/plugin`** (stable + `2.0.5`), published from the opencode repo; the beta `@opencode-ai/plugin` is + not the stable API. The port typechecks against those types but does not import + them — it declares `{ id, setup }` locally, so a **local-file install needs no + `bun add` at all** and the built artefact is a single self-contained file. Config/discovery is unchanged (`plugins` key, object form, and plugins under `.opencode/plugin/` **or** `.opencode/plugins/` are auto-loaded). - **Every event the plugin matches still exists** in 2.0.5, with the same names @@ -211,9 +237,9 @@ The port was originally written against the beta (`@opencode-ai/plugin` `ctx.session.interrupt()` input is `{ sessionID, resume? }`. - **Docs recommend `ctx.event.subscribe({ signal })`** and aborting the stream during cleanup; the port now does this with an `AbortController`. -- **Local-file installs need the API package resolvable.** opencode loads a - local `.ts` plugin with a normal ESM import and snapshots plugin - dependencies at server startup, so `@opencode/plugin` must be installed in - the config dir (`~/.config/opencode`) and opencode restarted. A *published* - plugin should instead declare `@opencode/plugin` in its `dependencies` - (replacing `@opencode-ai/plugin`). +- **Local-file installs need nothing installed.** opencode loads a local `.ts` + plugin with a normal ESM import, so *a plugin that imports the API package* + would have to have it resolvable and would need a restart. This port does not + import it, which removes that constraint — copy the file and restart. A + *published* plugin that later does import it should declare + `@opencode/plugin` in its `dependencies` (replacing `@opencode-ai/plugin`). From 805aca436d53d18d3915a22df1306cd8aaca73de Mon Sep 17 00:00:00 2001 From: famewolf Date: Thu, 1 Oct 2026 18:52:27 -0400 Subject: [PATCH 25/57] Drop the session.reverted case: v1 never emitted it The case was justified by a comment asserting that v1's legacy revert event was `session.reverted`. Verified live on 2026-10-01 against opencode v1.18.34: v1 emits no revert-named event at all. A revert arrives as `session.updated` whose `properties.info.revert` is populated with { messageID, partID?, snapshot?, diff? }, preceded by `session.diff`. The `session.revert*` strings in the v1 binary are HTTP route identifiers (`identifier: "session.revert"`, `"session.unrevert"`), not bus events, and a plugin watching the bus sees session.created / updated / deleted / diff / error / idle / status and nothing revert-shaped. The case was therefore unreachable, and the comment was wrong. Removed, and the reverts section now records the v1 mechanism that was actually verified. The v2 revert family handling is unchanged. --- src/v2/index.ts | 19 +++++++++---------- 1 file changed, 9 insertions(+), 10 deletions(-) diff --git a/src/v2/index.ts b/src/v2/index.ts index 8971afc..edb4fef 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -3183,6 +3183,15 @@ export default define({ return } // --- reverts --- + // v1.18.34 (verified live, 2026-10-01, on an arch LXC with the real + // v1 plugin loaded): v1 has NO revert-named event. A revert arrives as + // `session.updated` whose `properties.info.revert` is populated with + // { messageID, partID?, snapshot?, diff? }, preceded by `session.diff`. + // So on v1 a plugin that only switches on the event *name* never sees a + // rewind at all — which is exactly why the v1 port never dropped its + // watch state on one. (An earlier comment here claimed v1 emitted + // `session.reverted`; that was wrong, and the case it defended is gone.) + // // v2 emits a three-stage revert family. `staged` is intermediate // (the user can still clear it), while `cleared` and `committed` // are terminal; drop our watch state on the terminal stages so a @@ -3202,16 +3211,6 @@ export default define({ sessions.delete(sid) return } - // v1 legacy: not emitted in v2 (reverts now surface as - // `session.revert.cleared` / `session.revert.committed`). Kept - // defensively so older runtimes still drop their watch state. - case "session.reverted": { - const sid = sidOf(ev) - if (!sid) return - forgetShells(sid) - sessions.delete(sid) - return - } // --- compaction (NEVER interrupt a session that is compacting) --- case "session.compaction.started": { From a83e158dba8b9fd6d29a94a46367567a7c59b0d5 Mon Sep 17 00:00:00 2001 From: famewolf Date: Sat, 3 Oct 2026 00:52:15 -0400 Subject: [PATCH 26/57] fix(v2): read todos from the session message log, not the dead storage key MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v2 emits no `todo.updated` event and exposes no todo route, and the v1-shaped `ctx.storage` key `todos/` is never written — so the reader added in 16fdc92 always missed. auto-resume concluded "no todos" on sessions with work listed and fired false done-claim nudges at them. The real store is the session message log: every `todowrite` call is persisted as a tool part whose input carries the whole list, so the newest COMPLETED call is the current list. Read it via `ctx.client.session.message.list`, falling back to a loopback `GET /api/session/{id}/message?limit=200` — that endpoint answers `{data, cursor}`, never a bare array, and orders results newest first, so a last-match-wins parser selects the OLDEST list. Ranking is by timestamp, with encounter order as the tiebreak. `ctx.storage` is kept only as a last-resort fallback: in v2 that key is empty, not authoritative. This needs host access beyond the plugin context: `readFileSync` on `/proc/self/cmdline` for the local server's `--port`, plus a loopback GET. Under OpenChamber the server is spawned as `opencode serve --hostname ... --port ` and exports no port variable, so an env-only candidate list comes back empty and the HTTP source is never attempted at all. Both are best-effort — a host that refuses them degrades to "cannot read todos", never a crash — and non-Linux falls back to env discovery. Documented in docs/v2/installing.md and docs/v2/migration.md. 6 new tests in index.todo.test.ts, each mutation-checked: disabling the message-log read, switching to last-match-wins, and dropping the completed-status filter each turn the file red on the right test. Full suite 780 pass / 0 fail, tsc --strict clean. --- README.md | 16 ++- docs/v2/installing.md | 8 ++ docs/v2/migration.md | 19 ++++ src/v2/index.todo.test.ts | 148 ++++++++++++++++++++++++- src/v2/index.ts | 227 +++++++++++++++++++++++++++++++++----- 5 files changed, 384 insertions(+), 34 deletions(-) diff --git a/README.md b/README.md index 0f74124..456b73b 100644 --- a/README.md +++ b/README.md @@ -328,9 +328,19 @@ The diagram below is v1's: it is drawn against the v1 SSE event stream, the v1 `todo.updated` event and the v1 `session.todo()` API. The v2 build reads the same information through different doors — `session.usage.updated` for tokens, `ctx.model.get()` for the window, `ctx.plugin.list()` to detect magic-context, -`ctx.session.context()` for message history, and `ctx.storage` for the todo list -(the v2 `todo` table has no route and emits no event). See -[docs/known-issues-v2.md](docs/known-issues-v2.md) for each substitution. +and `ctx.session.context()` for message history. + +The todo list is the substitution with teeth. The v2 `todo` table has no route and +emits no event, and the v1-shaped `ctx.storage` key `todos/` is never +written in v2 — so a reader that trusts storage concludes "no todos" and fires +false done-claim nudges at a session with work still listed. The real store is the +**session message log**: every `todowrite` call is persisted as a tool part whose +input carries the entire list, so the newest *completed* call is the current list. +It is read through `ctx.client.session.message.list` when the host offers it, and +otherwise over loopback HTTP (`GET /api/session/{id}/message`, page capped at 200 +— the endpoint answers `{ data, cursor }`, never a bare array, and orders results +newest first). `ctx.storage` survives only as a last-resort fallback: in v2 that +key is empty, not authoritative. ``` Any SSE Event diff --git a/docs/v2/installing.md b/docs/v2/installing.md index 0be5f92..ae3552f 100644 --- a/docs/v2/installing.md +++ b/docs/v2/installing.md @@ -14,6 +14,14 @@ plugin API (`@opencode/plugin` 2.0.5, shipped with `opencode` v2.0.5). than importing one, so the built file is self-contained. `@opencode/plugin` is not needed to *run* the plugin — only to typecheck the source against the official types (see [Development](#development)). +- **Host access beyond the plugin context.** Reading the todo list needs the + local server's address, and that is not exposed as an environment variable under + OpenChamber. The plugin reads `/proc/self/cmdline` for the `--port` flag (Linux) + and may issue a loopback `GET` to `127.0.0.1`; where `/proc` is unavailable it + falls back to `OPENCODE_SERVER_URL` / `OPENCODE_SERVER_PORT` / + `OPENCODE_PORT` / `PORT`. If `OPENCODE_SERVER_PASSWORD` (or `OPENCODE_PASSWORD`) + is set, it is sent as HTTP Basic on that request. All of it is best-effort: a + host that refuses any of it degrades to "cannot read todos", never a crash. - The plugin source: [`src/v2/index.ts`](../../src/v2/index.ts) from this repo, or the built `dist/v2/index.js`. diff --git a/docs/v2/migration.md b/docs/v2/migration.md index 191173d..c31da1a 100644 --- a/docs/v2/migration.md +++ b/docs/v2/migration.md @@ -126,6 +126,25 @@ Two shape details the port relies on: - Recovery notifications use `ctx.session.synthetic({ sessionID, text, description, resume })` so the intervention is visible in the TUI, falling back to `session.prompt({ sessionID, text })`. +- **The todo list is read from the session message log, not from storage.** v2 + emits no `todo.updated` event and exposes no todo route, and the v1-shaped + `ctx.storage` key `todos/` is never written — so a storage-first + reader concludes "no todos" and fires false done-claim nudges. Every `todowrite` + call is persisted as a tool part whose input carries the whole list, so the + newest **completed** call is the current list. It is read via + `ctx.client.session.message.list`, falling back to a loopback + `GET /api/session/{id}/message?limit=200` (that endpoint answers + `{ data, cursor }`, never a bare array; 200 is its ceiling; results are ordered + newest first, so a last-match-wins parser would select the OLDEST list). + `ctx.storage` is now only a last-resort fallback. +- **New host access beyond the plugin context.** `readFileSync` on + `/proc/self/cmdline` to discover the local server's `--port`, plus a loopback + HTTP GET. Env vars alone are not enough: under OpenChamber the server is spawned + as `opencode serve --hostname 127.0.0.1 --port ` and exports no port variable, + so an env-only candidate list comes back empty and the HTTP source is never + attempted at all. Every one of these is best-effort — a host that refuses them + degrades to "cannot read todos", never a crash — and where `/proc` is + unavailable the plugin falls back to env-var discovery. - `ctx.client.app.log(...)` → prefixed console logging. - Status polling demoted: session state comes primarily from lifecycle events; a low-frequency interval cross-checks active sessions for silence diff --git a/src/v2/index.todo.test.ts b/src/v2/index.todo.test.ts index b499266..aeb6a48 100644 --- a/src/v2/index.todo.test.ts +++ b/src/v2/index.todo.test.ts @@ -10,10 +10,16 @@ const SID = "ses_todo" /** * The todo list as the plugin under test reads it. * - * auto-resume does not own a todo list — it reads the one the installed todo tool - * writes, under the key `todos/`. So these tests supply that key and - * nothing else, which is the point: they would pass unchanged against a real - * `todowrite` tool, and they assert the plugin never writes there itself. + * auto-resume does not own a todo list — it reads one. In v2 the real store is the + * SESSION MESSAGE LOG: every `todowrite` call is persisted as a tool part whose + * input carries the whole list, so the newest completed call IS the current list. + * The v1-shaped `todos/` storage key is kept as a last-resort fallback + * (nothing in v2 writes it, so it always misses). + * + * The `replay` harness therefore takes both: `todos` populates the storage + * fallback, `messages` supplies the message log. Left undefined, `messages` + * resolves to an empty page so the storage cases keep testing storage — which is + * the point, because the fallback must stay alive. * * Three things are under test: * @@ -126,6 +132,7 @@ async function replay( todos: unknown[] | undefined, opts: Record = {}, storageShape: { omitStorage?: boolean } = {}, + messages: unknown[] | undefined = undefined, ): Promise { const injected: Harness["injected"] = [] const storageWrites: string[] = [] @@ -147,7 +154,16 @@ async function replay( synthetic: async (a: any) => (injected.push({ text: a?.text }), {}), prompt: async (a: any) => (injected.push({ text: a?.text }), {}), }, - client: { session: { get: async () => ({ data: {} }) } }, + client: { + session: { + get: async () => ({ data: {} }), + // The message list endpoint answers with a `{ data, cursor }` envelope, + // never a bare array — handing the envelope straight to the parser trips + // its Array.isArray guard and silently resolves nothing. Reproduced here + // so that trap is exercised by every test in the group below. + message: { list: async () => ({ data: messages ?? [], cursor: null }) }, + }, + }, } if (!storageShape.omitStorage) { ctx.storage = { @@ -332,3 +348,125 @@ describe("v2: the todo list, read from the tool that owns it", () => { expect(injected[0].text).toBe("Your todo list still has open items.") }) }) + + +describe("v2: the todo list is read from the session message log", () => { + /** A completed `todowrite` tool part, stamped so ordering is unambiguous. */ + const part = (todos: unknown[], at: number) => ({ + type: "tool", + name: "todowrite", + state: { status: "completed", input: { todos }, time: { start: at, end: at } }, + }) + const msg = (parts: unknown[]) => ({ role: "user", content: parts }) + /** A `todowrite` part whose `input` is `input` verbatim, for malformed shapes. */ + const rawPart = (input: unknown, at: number) => ({ + type: "tool", + name: "todowrite", + state: { status: "completed", input, time: { start: at, end: at } }, + }) + + /** Distinct from OPEN/CLOSED so an assertion cannot pass on the wrong list. */ + const NEWEST = [{ content: "newest open item", status: "pending", priority: "high" }] + const OLDER = [{ content: "older open item", status: "pending", priority: "low" }] + + test("CONTROL: a todowrite in the log is enough, with the storage key left empty", async () => { + // The bug this group exists for. A storage-only reader finds an empty key, + // concludes "no todos", and fires a false done-claim nudge on a session with + // work still listed — which is how /todo printed [0/0] beside 14 todos. + const { injected, logs, storageReads } = await replay( + [{ text: CELEBRATED }], + [], + {}, + {}, + [msg([part(OPEN, Date.now())])], + ) + expect(injected).toHaveLength(1) + expect(injected[0].text).toContain("Write the migration guide") + // Not read at all: the log answered, so the dead key is never consulted. + expect(storageReads).not.toContain(`todos/${SID}`) + expect(logs.some((l) => l.includes("todos via ctx.client.session.message.list"))).toBe(true) + }) + + test("the NEWEST completed todowrite wins, not the oldest", async () => { + // `/api/session/{id}/message` returns NEWEST FIRST, so a last-match-wins loop + // selects the OLDEST list. This shipped in two places before it was caught, + // and it is silent: an older list is a plausible-looking list. + const now = Date.now() + const { injected } = await replay( + [{ text: CELEBRATED }], + [], + {}, + {}, + // Newest first, exactly as the endpoint orders it. + [msg([part(NEWEST, now)]), msg([part(OLDER, now - 60_000)])], + ) + expect(injected).toHaveLength(1) + expect(injected[0].text).toContain("newest open item") + expect(injected[0].text).not.toContain("older open item") + }) + + test("an unfinished todowrite is ignored, not half-read", async () => { + // Only a COMPLETED call replaced the list. A pending or failed one carries a + // partial input, and treating that as the list is how a half-written todo + // becomes "everything is done". + const { injected, logs } = await replay( + [{ text: CELEBRATED }], + [], + {}, + {}, + [ + msg([ + { + type: "tool", + name: "todowrite", + state: { status: "error", input: { todos: NEWEST }, time: { end: Date.now() } }, + }, + ]), + ], + ) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("no open todos — latching completion"))).toBe(true) + }) + + test("a malformed list yields 'cannot conclude', never an empty list", async () => { + // The dangerous direction is "unreadable" becoming "nothing is open". An + // unreadable list must fall through to the fallback, exactly like no list. + for (const bad of [{ todos: "nope" }, { todos: [null, 7] }, {}]) { + const { injected } = await replay( + [{ text: CELEBRATED }], + [], + {}, + {}, + [msg([rawPart(bad, Date.now())])], + ) + expect(injected).toEqual([]) + } + }) + + test("the storage fallback still answers when the log has no todowrite", async () => { + // The fallback is not dead code we can drop: it is the only source on a host + // whose log the plugin cannot read. + const { injected, storageReads } = await replay( + [{ text: CELEBRATED }], + OPEN, + {}, + {}, + [msg([{ type: "text", text: "no tool calls here" }])], + ) + expect(storageReads).toContain(`todos/${SID}`) + expect(injected).toHaveLength(1) + expect(injected[0].text).toContain("Write the migration guide") + }) + + test("a host with neither source degrades instead of crashing", async () => { + const { injected, logs } = await replay( + [{ text: CELEBRATED }], + [], + {}, + { omitStorage: true }, + [], + ) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("latching completion"))).toBe(true) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index edb4fef..57df820 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -54,7 +54,7 @@ type AutoResumePlugin = { const define = (plugin: T): T => plugin -import { appendFileSync, existsSync, mkdirSync, statSync, writeFileSync } from "node:fs" +import { appendFileSync, readFileSync, existsSync, mkdirSync, statSync, writeFileSync } from "node:fs" import { homedir } from "node:os" import { dirname, join } from "node:path" @@ -787,6 +787,119 @@ function buildOpenTodosReminder(todos: Todo[]): string { return `You have ${open.length} unfinished task${plural}:\n${list}\n\nPlease continue working on ${thisWord} ${taskWord}.` } + +/** Candidate local server base URLs: env vars first, then /proc/self/cmdline. + * `cmdline` is injectable so the port parsing is unit-testable without spawning + * a process whose argv looks like a server invocation. + * + * The /proc read is not belt-and-braces: under OpenChamber the server is spawned + * as `opencode serve --hostname 127.0.0.1 --port ` and exports NO port env + * var, so an env-only list comes back EMPTY and the HTTP source is never + * attempted at all. Same mechanism the prompt-polisher uses, which is why its + * HTTP DELETE reaches this same server. + */ +function serverBaseUrls(cmdline?: string): string[] { + const out: string[] = [] + const push = (v: string | undefined) => { + if (typeof v !== "string" || !v) return + const url = v.startsWith("http") ? v : `http://127.0.0.1:${v.replace(/^:/, "")}` + if (!out.includes(url)) out.push(url.replace(/\/$/, "")) + } + const env = process.env + push(env.OPENCODE_SERVER_URL) + push(env.OPENCODE_URL) + push(env.OPENCODE_SERVER_PORT) + push(env.OPENCODE_PORT) + push(env.PORT) + try { + const raw = cmdline ?? readFileSync("/proc/self/cmdline", "utf8") + const argv = raw.split("\0") + const portFlag = argv.indexOf("--port") + if (portFlag !== -1 && argv[portFlag + 1]) { + const hostFlag = argv.indexOf("--hostname") + const host = hostFlag !== -1 && argv[hostFlag + 1] ? argv[hostFlag + 1] : "127.0.0.1" + push(`http://${host}:${argv[portFlag + 1]}`) + } + } catch { + // /proc unavailable (non-Linux): env vars only + } + return out +} +/** HTTP Basic for the local server, same derivation the prompt-polisher uses. */ +function serverHeaders(): Record { + const pw = process.env.OPENCODE_SERVER_PASSWORD || process.env.OPENCODE_PASSWORD + if (!pw) return {} + return { Authorization: `Basic ${Buffer.from(`opencode:${pw}`).toString("base64")}` } +} +/** + * Normalize a `todowrite` tool input into a todo list, or `undefined` when it is + * not a usable list. + * + * An entry with an unrecognized status is KEPT (cast, not dropped): `isOpenTodo` + * treats only `pending`/`in_progress` as open, so an unknown status reads as + * closed. Dropping it would shrink the list, and a shrunken list is the + * dangerous direction - it makes real open work look finished. + */ +function normalizeTodoList(input: unknown): Todo[] | undefined { + if (input === null || typeof input !== "object") return undefined + const todos = (input as { todos?: unknown }).todos + if (!Array.isArray(todos)) return undefined + const out: Todo[] = [] + for (const entry of todos) { + if (entry === null || typeof entry !== "object") continue + const e = entry as { content?: unknown; status?: unknown; priority?: unknown } + if (typeof e.content !== "string" || typeof e.status !== "string") continue + const todo: Todo = { content: e.content, status: e.status as Todo["status"], priority: "medium" } + if (typeof e.priority === "string") todo.priority = e.priority as Todo["priority"] + out.push(todo) + } + return out.length > 0 ? out : undefined +} +/** + * Canonical parser, mirroring opencode-todo-fork's `latestTodosFromMessages`: walk + * the messages, and for every COMPLETED `todowrite` tool part keep its normalized + * input. `todowrite` REPLACES the whole list, so the newest such call IS the + * current list. + * + * Order note: `/api/session/{id}/message` returns NEWEST FIRST, so a + * last-match-wins loop selects the OLDEST list - that bug shipped in two places + * before it was caught. Rank by timestamp instead, falling back to encounter + * order, and skip a malformed historical call while keeping the previous valid + * list. + * + * Returns `undefined`, never `[]`, so "no todos exist" stays distinguishable from + * "we could not look" - that distinction is what keeps this diagnosable instead of + * silently degrading. + */ +function todosFromMessages(messages: unknown): Todo[] | undefined { + if (!Array.isArray(messages)) return undefined + const candidates: { list: Todo[]; at: number | null; order: number }[] = [] + let order = 0 + for (const message of messages) { + if (message === null || typeof message !== "object") continue + const content = (message as { content?: unknown }).content + if (!Array.isArray(content)) continue + for (const part of content) { + if (part === null || typeof part !== "object") continue + const p = part as { type?: unknown; name?: unknown; state?: Record; time?: any } + if (p.type !== "tool" || p.name !== "todowrite") continue + if (p.state?.status !== "completed") continue + const list = normalizeTodoList(p.state.input) + if (!list) continue + const t = p.state?.time ?? p.time ?? (message as { time?: any }).time ?? {} + const at = [t?.end, t?.start, t?.created].find((v: unknown) => typeof v === "number") + candidates.push({ list, at: typeof at === "number" ? at : null, order: order++ }) + } + } + if (candidates.length === 0) return undefined + candidates.sort((a, b) => { + if (a.at !== null && b.at !== null && a.at !== b.at) return b.at - a.at + if ((a.at === null) !== (b.at === null)) return a.at === null ? 1 : -1 + return a.order - b.order + }) + return candidates[0]?.list +} + /** Model ends with ":" announcing intent without executing. */ function containsActionIntent(text: string): boolean { if (text.length <= 15) return false @@ -1674,47 +1787,109 @@ export default define({ } /** Targeted recovery prompts (tool-as-text, done-claims, intent nudges). */ + /** * Read the session's todo list. * - * v1 tracked todos from a `todo.updated` event and a server API. v2 has - * neither: the todo table exists in the v2 database but no route reaches - * it and nothing emits an event for it. What v2 does have is the storage - * domain, and the installed todo tool already writes the list there under - * a stable, per-session key. + * v1 tracked todos from a `todo.updated` event and a server API. v2 has neither: + * nothing emits a todo event and no route reaches the todo table. + * + * What v2 DOES have is the session message log. Every `todowrite` call is + * persisted as a tool part whose input carries the entire list, so the newest + * completed call IS the current list. That is the real store, so it is read first + * - reading anything else first is what made auto-resume conclude "no todos" + * while the list was sitting in the log. * - * So this reads the key rather than owning a list. That is the difference - * between a second copy that can drift and the real one — and it means - * auto-resume works with whichever todo tool is installed instead of - * requiring its own. + * `ctx.storage.get("todos/")` is kept LAST, as a fallback only. It is a + * v1-shaped key that nothing in v2 writes, so it always misses; do not promote it + * back to primary. In v2 the key is empty, not authoritative. * - * Cached for a short TTL: this is read on every idle inspection, and the - * list can only change while the model is working, which is not when we - * ask. Returns `[]` on any failure — every caller treats "no list" as - * "cannot conclude", never as "nothing is open". + * Cached for a short TTL: this is read on every idle inspection, and the list can + * only change while the model is working, which is not when we ask. Returns `[]` on + * any failure - every caller treats "no list" as "cannot conclude", never as + * "nothing is open". */ async function readTodos(sid: string): Promise { const w = ensureWatch(sid) const now = Date.now() if (w.todosFetchedAt && now - w.todosFetchedAt < TODO_CACHE_TTL_MS) return w.todos + const fromMessages = await todosFromMessageSources(sid) + if (fromMessages) { + w.todos = fromMessages + w.todosFetchedAt = now + return w.todos + } if (!ctx.storage) return w.todos try { - const raw = await ctx.storage.get(`todos/${sid}`) - const record = raw as { todos?: unknown; updatedAt?: unknown } | null | undefined - const list = Array.isArray(record?.todos) ? record.todos : [] - w.todos = list.filter( - (t): t is Todo => - !!t && typeof t === "object" && typeof (t as Todo).content === "string" && - typeof (t as Todo).status === "string", - ) - w.todosFetchedAt = now - return w.todos + const raw = await ctx.storage.get(`todos/${sid}`) + const record = raw as { todos?: unknown; updatedAt?: unknown } | null | undefined + const list = Array.isArray(record?.todos) ? record.todos : [] + w.todos = list.filter( + (t): t is Todo => + !!t && typeof t === "object" && typeof (t as Todo).content === "string" && + typeof (t as Todo).status === "string", + ) + w.todosFetchedAt = now + return w.todos + } catch (e) { + dbg(`${short(sid)} todo read failed:`, e instanceof Error ? e.message : String(e)) + return w.todos + } + } + /** + * Try every plausible source of a session's messages, in order, returning the + * parsed todo list from the first that yields a `todowrite` tool part. + * + * Requires host access beyond the plugin ctx: `readFileSync` on + * `/proc/self/cmdline`, and a loopback HTTP GET. Both are best-effort - a host + * that refuses either degrades to the storage fallback rather than throwing. + */ + async function todosFromMessageSources(sid: string): Promise { + // 1. The SDK client, if this plugin ctx was given one. Unwrap the `{data,cursor}` + // envelope: handing it straight to the parser trips its Array.isArray guard and + // silently resolves nothing. + try { + const message = (ctx.client as any)?.session?.message + if (typeof message?.list === "function") { + const res = await message.list.call(message, { path: { id: sid } }) + const got = todosFromMessages(res?.data ?? res) + if (got) { + dbg(`${short(sid)} todos via ctx.client.session.message.list (${got.length})`) + return got + } + } } catch (e) { - dbg(`${short(sid)} todo read failed:`, e instanceof Error ? e.message : String(e)) - return w.todos + dbg(`${short(sid)} client message.list failed:`, e instanceof Error ? e.message : String(e)) + } + // 2. The local HTTP server. Same shape the prompt-polisher already uses, so the + // auth and base-URL derivation are proven rather than guessed. + for (const base of serverBaseUrls()) { + try { + // `limit` matters: the default page is 50 messages and 200 is the server's + // ceiling (probed - 250 returns 400). Newest first, so a recent `todowrite` is + // always inside the page. + const res = await fetch(`${base}/api/session/${encodeURIComponent(sid)}/message?limit=200`, { + headers: serverHeaders(), + }) + if (!res.ok) { + dbg(`${short(sid)} HTTP message read ${res.status} from ${base}`) + continue + } + // Same envelope trap as source 1: `{data, cursor}`, never a bare array. + const body = await res.json() + const got = todosFromMessages(body?.data ?? body) + if (got) { + dbg(`${short(sid)} todos via HTTP ${base} (${got.length})`) + return got } + } catch (e) { + dbg(`${short(sid)} HTTP message read failed:`, e instanceof Error ? e.message : String(e)) + } + } + return undefined } + /** * Arm the deferred pattern pass, replacing any already pending. * From 801e22e55335982d4196c15bcd4b67c8bc760823 Mon Sep 17 00:00:00 2001 From: famewolf Date: Sat, 3 Oct 2026 01:18:53 -0400 Subject: [PATCH 27/57] docs: correct why the storage fallback cannot be promoted The previous wording said the todos/ key is "a v1-shaped key that nothing in v2 writes", so it "always misses" and is "empty, not authoritative". That is false. A todo plugin does write that key, in its todowrite tool: ctx.storage.set(storageKey(sessionID), { todos, updatedAt }) The reason the read still cannot see it is that ctx.storage is namespaced per plugin, so the two writes land in different files. Verified against a live server on two independent surfaces: - the route carries the plugin id as a path segment: /api/plugin/storage// - the on-disk tree is storage/plugin//.json auto-resume asking for todos/ resolves to storage/plugin/auto-resume.v2/todos/.json; the todo plugin's write is under its own id. Cross-plugin visibility is impossible by construction, so the read misses regardless of who is writing - which is a stronger and more accurate reason to keep the message log primary than the one that was written down. This matters beyond wording: "nothing writes it" invites the fix "so let auto-resume write it too", which would be wrong twice over - a reader cannot surface another plugin's data by writing a key, and a second writer for one logical list races the first. Both the code comment and the docs now say so explicitly. Also records a measured detail the ranking depends on: the todowrite tool parts carry no state.time, so ranking has to fall back to the enclosing message's time.created. Under newest-first ordering that fallback is load-bearing, not cosmetic. Comment-only change. No behaviour differs. 20 todo tests pass, tsc --strict clean. --- README.md | 27 +++++++++++++++++---------- docs/v2/migration.md | 27 ++++++++++++++++++++------- src/v2/index.ts | 16 +++++++++++++--- 3 files changed, 50 insertions(+), 20 deletions(-) diff --git a/README.md b/README.md index 456b73b..82ed1e0 100644 --- a/README.md +++ b/README.md @@ -331,16 +331,23 @@ information through different doors — `session.usage.updated` for tokens, and `ctx.session.context()` for message history. The todo list is the substitution with teeth. The v2 `todo` table has no route and -emits no event, and the v1-shaped `ctx.storage` key `todos/` is never -written in v2 — so a reader that trusts storage concludes "no todos" and fires -false done-claim nudges at a session with work still listed. The real store is the -**session message log**: every `todowrite` call is persisted as a tool part whose -input carries the entire list, so the newest *completed* call is the current list. -It is read through `ctx.client.session.message.list` when the host offers it, and -otherwise over loopback HTTP (`GET /api/session/{id}/message`, page capped at 200 -— the endpoint answers `{ data, cursor }`, never a bare array, and orders results -newest first). `ctx.storage` survives only as a last-resort fallback: in v2 that -key is empty, not authoritative. +emits no event, and **`ctx.storage` cannot be used to read another plugin's todo +list** — it is namespaced per plugin. `ctx.storage.get("todos/" + id)` from this +plugin resolves to `storage/plugin/auto-resume.v2/todos/.json`; a todo plugin +writing that same key lands in its own directory. Verified against a live server: +the route is `/api/plugin/storage//` and the on-disk tree is +`storage/plugin//.json`. So a storage-first reader concludes "no +todos" no matter who is writing, and fires false done-claim nudges at a session +with work still listed. + +The store both plugins can agree on is the **session message log**: every +`todowrite` call is persisted as a tool part whose input carries the entire list, +so the newest *completed* call is the current list. It is read through +`ctx.client.session.message.list` when the host offers it, and otherwise over +loopback HTTP (`GET /api/session/{id}/message`, page capped at 200 — the endpoint +answers `{ data, cursor }`, never a bare array, and orders results newest first). +`ctx.storage` survives only as a last-resort fallback, for a host that writes the +list itself. ``` Any SSE Event diff --git a/docs/v2/migration.md b/docs/v2/migration.md index c31da1a..3b444c1 100644 --- a/docs/v2/migration.md +++ b/docs/v2/migration.md @@ -127,16 +127,29 @@ Two shape details the port relies on: description, resume })` so the intervention is visible in the TUI, falling back to `session.prompt({ sessionID, text })`. - **The todo list is read from the session message log, not from storage.** v2 - emits no `todo.updated` event and exposes no todo route, and the v1-shaped - `ctx.storage` key `todos/` is never written — so a storage-first - reader concludes "no todos" and fires false done-claim nudges. Every `todowrite` - call is persisted as a tool part whose input carries the whole list, so the - newest **completed** call is the current list. It is read via + emits no `todo.updated` event and exposes no todo route — and, decisively, + **`ctx.storage` is namespaced per plugin**, so it cannot read another plugin's + todo list at all. `ctx.storage.get("todos/")` resolves to + `storage/plugin/auto-resume.v2/todos/.json`; a todo plugin writing that key + lands under its own plugin id. Verified live: the route is + `/api/plugin/storage//` and the on-disk tree is + `storage/plugin//.json`. A storage-first reader therefore + concludes "no todos" regardless of who writes, and fires false done-claim + nudges. + + Every `todowrite` call is persisted as a tool part whose input carries the whole + list, so the newest **completed** call is the current list. It is read via `ctx.client.session.message.list`, falling back to a loopback `GET /api/session/{id}/message?limit=200` (that endpoint answers `{ data, cursor }`, never a bare array; 200 is its ceiling; results are ordered - newest first, so a last-match-wins parser would select the OLDEST list). - `ctx.storage` is now only a last-resort fallback. + newest first, so a last-match-wins parser would select the OLDEST list). One + measured detail worth keeping: the tool parts carry no `state.time`, so ranking + must fall back to the enclosing message's `time.created` — under newest-first + ordering that fallback is load-bearing, not cosmetic. + + `ctx.storage` remains only as a last-resort fallback. Do not "fix" the namespace + mismatch by having auto-resume write the key: a reader cannot surface another + plugin's data, and a second writer would race the first. - **New host access beyond the plugin context.** `readFileSync` on `/proc/self/cmdline` to discover the local server's `--port`, plus a loopback HTTP GET. Env vars alone are not enough: under OpenChamber the server is spawned diff --git a/src/v2/index.ts b/src/v2/index.ts index 57df820..3ddc909 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -1800,9 +1800,19 @@ export default define({ * - reading anything else first is what made auto-resume conclude "no todos" * while the list was sitting in the log. * - * `ctx.storage.get("todos/")` is kept LAST, as a fallback only. It is a - * v1-shaped key that nothing in v2 writes, so it always misses; do not promote it - * back to primary. In v2 the key is empty, not authoritative. + * `ctx.storage.get("todos/")` is kept LAST, and it is NOT a source of + * another plugin's todos. `ctx.storage` is namespaced per plugin: the key + * resolves to `storage/plugin/auto-resume.v2/todos/.json`, while a todo + * plugin writing the same key lands in its OWN directory + * (`storage/plugin//todos/.json`). Verified against the live + * server: `/api/plugin/storage//` carries the plugin id as a + * path segment, and the on-disk tree is `storage/plugin//.json`. + * So no key auto-resume does not write itself can ever hold a todo list here. + * + * Do not promote it back to primary, and do not "fix" the mismatch by writing + * the key: a reader cannot make another plugin's data appear in its namespace, + * and duplicating that data would create a second writer racing the first. The + * message log above is the only store both plugins can agree on. * * Cached for a short TTL: this is read on every idle inspection, and the list can * only change while the model is working, which is not when we ask. Returns `[]` on From a355db38fe24a6ab250e4e09d3e8c5a0636fd570 Mon Sep 17 00:00:00 2001 From: famewolf Date: Sat, 3 Oct 2026 01:24:31 -0400 Subject: [PATCH 28/57] docs: say why the todo parser is a copy and not an import The block was annotated "Canonical parser, mirroring opencode-todo-fork's latestTodosFromMessages", which tells a reader it is a copy but not why it is one. Without the reason it reads as an oversight, and the obvious next step looks like an improvement - import the sibling plugin instead. There is no cross-plugin import contract, so that would tie auto-resume's ability to read a todo list to a sibling plugin being installed at a fixed path. auto-resume has to work without it. The real cost is drift between two copies, which is why the note now says both repos test the ranking. Also names the function as it is actually exported (todosFromMessages); latestTodosFromMessages is the name used in that repo's src/tui-data.ts. Comment-only. tsc --strict clean, 20 todo tests pass. --- src/v2/index.ts | 18 ++++++++++++++---- 1 file changed, 14 insertions(+), 4 deletions(-) diff --git a/src/v2/index.ts b/src/v2/index.ts index 3ddc909..9f24302 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -856,10 +856,20 @@ function normalizeTodoList(input: unknown): Todo[] | undefined { return out.length > 0 ? out : undefined } /** - * Canonical parser, mirroring opencode-todo-fork's `latestTodosFromMessages`: walk - * the messages, and for every COMPLETED `todowrite` tool part keep its normalized - * input. `todowrite` REPLACES the whole list, so the newest such call IS the - * current list. + * Canonical parser, mirroring opencode-todo-fork's `todosFromMessages` (the same + * routine is called `latestTodosFromMessages` in that repo's `src/tui-data.ts`): + * walk the messages, and for every COMPLETED `todowrite` tool part keep its + * normalized input. `todowrite` REPLACES the whole list, so the newest such call + * IS the current list. + * + * This is a deliberate copy rather than an import, and the reason is worth + * stating because it looks like an oversight otherwise. Plugins are loaded + * independently and there is no cross-plugin import contract: reaching into a + * sibling plugin's directory would tie auto-resume's ability to read a todo + * list to that plugin being installed at that exact path, and auto-resume has to + * work without it. The cost is that a fix to one copy is not automatically a fix + * to the other - so both copies carry the ordering note below, and both repos + * test the ranking rather than trusting it to stay in step. * * Order note: `/api/session/{id}/message` returns NEWEST FIRST, so a * last-match-wins loop selects the OLDEST list - that bug shipped in two places From 5f3fb0abcd118224bbf01ec7c68a1701ba568599 Mon Sep 17 00:00:00 2001 From: famewolf Date: Sat, 3 Oct 2026 02:38:41 -0400 Subject: [PATCH 29/57] docs: sync todo storage section to namespaced reality, 780 count --- docs/known-issues-v2.md | 25 +++++++++++++++++-------- docs/v2/migration.md | 2 +- 2 files changed, 18 insertions(+), 9 deletions(-) diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index e07b2cb..3207b69 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -142,17 +142,26 @@ API. v2 has neither: the `todo` table is created in the v2 database (migration `20260127222353_familiar_lady_ursula.ts`) but no route reaches it and nothing emits an event for it, so there is no way to observe the list changing. -What v2 does have is the storage domain, and the installed todo tool already -writes the list there under a stable per-session key — -`ctx.storage.set("todos/", { todos, updatedAt })`. auto-resume reads -that key instead of owning a list of its own. +What v2 does have is the **session message log**: every `todowrite` call is +persisted as a tool part whose input carries the whole list, so the newest +completed call *is* the current list. auto-resume reads that, instead of owning a +list of its own. + +`ctx.storage` looks like the obvious store and is not one. It is namespaced per +plugin: a key resolves to `storage/plugin//.json`, and the route is +`/api/plugin/storage//`. A todo plugin writing +`todos/` therefore lands in its own directory, so this plugin reading +the same key is reading a key nothing wrote — regardless of who writes what. A +storage-first reader concludes "no todos" while the list sits in the message log, +and fires false done-claim nudges at a session that still has work listed. The consequence worth stating: auto-resume is a **consumer**, not a second owner. It works with whichever todo tool is installed rather than requiring its own, and -there is no copy to drift. It only ever calls `get`. With no `ctx.storage` at all, -or with a record it cannot parse, it falls back to "no list" — which means the -older behaviour (latch on the emoji, ask for details on a bare done-claim), never -a nudge on the strength of a list it failed to read. +there is no copy to drift. It calls `get` on storage only after the message log has +come up empty. With no `ctx.storage` at all, or with a record it cannot parse, it +falls back to "no list" — which means the older behaviour (latch on the emoji, ask +for details on a bare done-claim), never a nudge on the strength of a list it failed +to read. Three places consume it: diff --git a/docs/v2/migration.md b/docs/v2/migration.md index 3b444c1..9aad2e8 100644 --- a/docs/v2/migration.md +++ b/docs/v2/migration.md @@ -16,7 +16,7 @@ object) to the v2 promise-plugin API (`Plugin.define({ id, setup })` + `ctx.event.subscribe()`). All detection/recovery features are preserved. Strict-mode typechecked against the real `@opencode/plugin@2.0.5` types, and covered by 188 tests across 13 files driven through the real event stream — one -file per feature area, 774 passing repo-wide including v1. +file per feature area, 780 passing repo-wide including v1. ## 1. Config key renamed: `plugin` → `plugins` From ef25e3965a90c0a8a6b473391fa1368e48ab188c Mon Sep 17 00:00:00 2001 From: famewolf Date: Sat, 3 Oct 2026 03:16:28 -0400 Subject: [PATCH 30/57] test: gate back-to-back idles on observed nudge, not 10ms hope --- src/v2/index.options.test.ts | 32 ++++++++++++++++++++++++++++---- 1 file changed, 28 insertions(+), 4 deletions(-) diff --git a/src/v2/index.options.test.ts b/src/v2/index.options.test.ts index ec79b4b..8340832 100644 --- a/src/v2/index.options.test.ts +++ b/src/v2/index.options.test.ts @@ -85,6 +85,15 @@ async function replay( events: any[], opts: Record = {}, harnessOpts: { sendFails?: boolean; text?: string } = {}, + /** Index of an event that must wait for the first nudge to land before it + * is pushed. Back-to-back idles with a 0ms pattern delay rely on the first + * 0ms timer firing inside the 10ms inter-event gap; under parallel-suite + * load it does not, and the second idle's schedulePatternPass then + * *replaces* the first pass (by design — see schedulePatternPass) instead + * of following it, so only one nudge is ever logged. Gating on the observed + * line removes the wall-clock race. Times out silently after 10s: the + * test's own assertion then reports the miss with the natural evidence. */ + waitForNudgeBeforeEvent?: number, ): Promise { const injected: Harness["injected"] = [] const interrupts: string[] = [] @@ -128,10 +137,24 @@ async function replay( }, } +/** Poll the scratch log until `needle` appears (see `waitForNudgeBeforeEvent`). */ +async function waitForLogLine(path: string, needle: string, timeoutMs: number): Promise { + const start = Date.now() + while (Date.now() - start < timeoutMs) { + try { + if (readFileSync(path, "utf8").includes(needle)) return + } catch { + // Not written yet — the plugin creates it on first log line. + } + await wait(20) + } +} + const cleanup = await (plugin as any).setup(ctx) - for (const e of events) { - stream.push(e) - await wait(10) + for (let i = 0; i < events.length; i++) { + stream.push(events[i]) + if (i === waitForNudgeBeforeEvent) await waitForLogLine(logFile, "ready-to-continue detected", 10_000) + else await wait(10) } await wait(600) // handleEvent is sync; its work is async ;(cleanup as (() => void) | undefined)?.() @@ -359,7 +382,7 @@ describe("v2: a nudge that was never delivered does not consume a retry", () => ] test("CONTROL: both sends succeed -> each nudge is 1/3, because markBusy resets the per-turn budget", async () => { - const { logs } = await replay(twoIdles, { maxRetries: 3, loopMaxContinues: 99, injectIntervalMs: 0 }) + const { logs } = await replay(twoIdles, { maxRetries: 3, loopMaxContinues: 99, injectIntervalMs: 0 }, {}, 3) const nudges = logs.filter((l) => l.includes("ready-to-continue detected")) expect(nudges.length).toBeGreaterThanOrEqual(2) // This is what makes the test below meaningful: on the success path the @@ -373,6 +396,7 @@ describe("v2: a nudge that was never delivered does not consume a retry", () => twoIdles, { maxRetries: 3, loopMaxContinues: 99, injectIntervalMs: 0 }, { sendFails: true }, + 3, ) expect(injected).toEqual([]) const nudges = logs.filter((l) => l.includes("ready-to-continue detected")) From f773ea6ad10dd85135bb2eb257670f0aa3c50c00 Mon Sep 17 00:00:00 2001 From: famewolf Date: Sat, 3 Oct 2026 11:36:43 -0400 Subject: [PATCH 31/57] docs: recommend opencode-todo-fork as the confirmed todo plugin --- README.md | 5 +++++ docs/known-issues-v2.md | 5 +++-- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 82ed1e0..b03947d 100644 --- a/README.md +++ b/README.md @@ -349,6 +349,11 @@ answers `{ data, cursor }`, never a bare array, and orders results newest first) `ctx.storage` survives only as a last-resort fallback, for a host that writes the list itself. +Confirmed working setup: `opencode-todo-fork` — verified live (list resolved +from the log, `/todo` round-trip green). Other todo plugins are unconfirmed: +anything that persists `todowrite` calls to the message log should read, but +only the fork has been tested. + ``` Any SSE Event ├─ has sessionID? → touchSession(sid) — reset only that session's timer diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 3207b69..db7df25 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -156,8 +156,9 @@ storage-first reader concludes "no todos" while the list sits in the message log and fires false done-claim nudges at a session that still has work listed. The consequence worth stating: auto-resume is a **consumer**, not a second owner. -It works with whichever todo tool is installed rather than requiring its own, and -there is no copy to drift. It calls `get` on storage only after the message log has +It works with whichever todo tool is installed rather than requiring its own. +Confirmed against `opencode-todo-fork`; other todo plugins are unconfirmed. +There is no copy to drift. It calls `get` on storage only after the message log has come up empty. With no `ctx.storage` at all, or with a record it cannot parse, it falls back to "no list" — which means the older behaviour (latch on the emoji, ask for details on a bare done-claim), never a nudge on the strength of a list it failed From 2052c079178a6c0afb3682a6009fbd08f31fcf3e Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:51:57 -0400 Subject: [PATCH 32/57] v2: cross-instance duplicate check via session log MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from f99ddd8 (auto-resume/v2-todo-message-log): stacked watchdogs each hold private counters, so the in-memory skip cannot see a sibling's prod — the session log can. Test coverage lands in src/v2/index.duplicate-prod.test.ts (index.visible-continue.test.ts no longer exists on this branch). --- src/v2/index.duplicate-prod.test.ts | 165 ++++++++++++++++++++++++++++ src/v2/index.ts | 89 ++++++++++++++- 2 files changed, 253 insertions(+), 1 deletion(-) create mode 100644 src/v2/index.duplicate-prod.test.ts diff --git a/src/v2/index.duplicate-prod.test.ts b/src/v2/index.duplicate-prod.test.ts new file mode 100644 index 0000000..0a9afa7 --- /dev/null +++ b/src/v2/index.duplicate-prod.test.ts @@ -0,0 +1,165 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +/** + * v2: cross-instance duplicate check (shared log). + * + * Ported from f99ddd8 (todo branch): stacked watchdogs (reload churn) each + * hold private counters, so the in-memory skip cannot see a sibling's prod — + * the session log can. If our exact text is already the newest user message + * within the recency window, stand down. Fail-open: fetch problems mean + * "no info", never "duplicate". + */ + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_dup" +let counter = 2000 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +type Harness = { + injected: Array<{ kind: string; text?: string }> + logs: string[] +} + +/** A turn that starts and then goes silent: a stall candidate. */ +const busyStallEvents = [ev("session.execution.started")] + +const FAST = { + chunkTimeoutMs: 50, + gracePeriodMs: 0, + checkIntervalMs: 20, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + loopMaxContinues: 99, + injectIntervalMs: 0, +} + +async function replay( + events: any[], + opts: Record = {}, + waitMs = 600, + recentMessages: unknown[] | undefined = undefined, +): Promise { + const injected: Harness["injected"] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-dup-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + event: stream, + options: { ...FAST, ...opts, logFile, debug: true }, + session: { + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ kind: "synthetic", text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ kind: "prompt", text: a?.text }), {}), + }, + client: { + session: { + get: async () => ({ data: {} }), + message: { + list: async () => ({ data: recentMessages ?? [], cursor: null }), + }, + }, + }, + storage: { + get: async () => ({ todos: [], updatedAt: Date.now() }), + set: async () => {}, + remove: async () => {}, + }, + } + + const cleanup = await (plugin as any).setup(ctx) + const started = Date.now() + for (const e of events) { + stream.push(e) + await wait(10) + } + while (Date.now() - started < waitMs) await wait(10) + await wait(50) + ;(cleanup as (() => void) | undefined)?.() + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { injected, logs } +} + +const userMsg = (text: string, at: number) => ({ + role: "user", + content: [{ type: "text", text }], + time: { created: at }, +}) + +describe("v2: cross-instance duplicate check (shared log)", () => { + test("a sibling's identical prod in the log suppresses ours", async () => { + // Two stacked watchdogs, one session: the first instance's prod is a + // user message in the log, so the second instance must stand down even + // though its private counters know nothing. + const { injected, logs } = await replay( + busyStallEvents, + { maxRetries: 5, continuePrompt: "go" }, + 900, + [userMsg("go", Date.now())], + ) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("identical prod already in session log"))).toBe(true) + }) + + test("a newer user message is progress, not a duplicate", async () => { + const { injected } = await replay( + busyStallEvents, + { maxRetries: 1, continuePrompt: "go" }, + 600, + [userMsg("actually, also this", Date.now())], + ) + expect(injected.length).toBeGreaterThan(0) + }) + + test("a stale identical prod does not suppress", async () => { + const { injected } = await replay( + busyStallEvents, + { maxRetries: 1, continuePrompt: "go" }, + 600, + [userMsg("go", Date.now() - 10 * 60_000)], + ) + expect(injected.length).toBeGreaterThan(0) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index 9f24302..8321240 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -153,6 +153,9 @@ interface SessionWatch { interruptWindowStart: number /** Timestamp of the last recovery injection, for `injectIntervalMs` debouncing. */ lastInjectAt: number + /** Last injected prod text + assistant snapshot at prod time, for duplicate suppression. */ + lastProdText: string + prodAssistantSnapshot: string recovering: boolean /** Latched when a failure carries an OOC error that `continue` can never clear; recovery (continue + abort+resume) is refused until a genuine user/agent turn. */ oocLocked: boolean @@ -1293,6 +1296,8 @@ export default define({ interruptsThisWindow: 0, interruptWindowStart: 0, lastInjectAt: 0, + lastProdText: "", + prodAssistantSnapshot: "", recovering: false, oocLocked: false, oocLockReason: null, @@ -1532,6 +1537,75 @@ export default define({ * nothing to restart it, which is the failure this whole fix targets. * It stays rate-limited by `w.aborting` and `MAX_INTERRUPTS_PER_WINDOW`. */ + /** + * Cross-instance duplicate check: is our exact text already the latest + * user message in the session log, posted within RECENT_PROD_WINDOW_MS? + * Stacked watchdogs (reload churn) each hold private counters, so the + * in-memory soft skip cannot see a sibling's prod — the log can. + * Fail-open: any fetch problem means "no info", never "duplicate". + */ + const RECENT_PROD_WINDOW_MS = 60_000 + async function recentOwnProdInLog(sid: string, text: string): Promise { + const pages: unknown[] = [] + try { + const message = (ctx.client as any)?.session?.message + if (typeof message?.list === "function") { + const res = await message.list.call(message, { path: { id: sid } }) + const data = (res as { data?: unknown } | undefined)?.data ?? res + if (Array.isArray(data)) pages.push(...data) + } + } catch { + // Fall through to HTTP below. + } + if (pages.length === 0) { + for (const base of serverBaseUrls()) { + try { + const res = await fetch( + `${base}/api/session/${encodeURIComponent(sid)}/message?limit=20`, + { headers: serverHeaders() }, + ) + if (!res.ok) continue + const body = ((await res.json()) as { data?: unknown } | undefined)?.data + if (Array.isArray(body)) { + pages.push(...body) + break + } + } catch { + // Next candidate base. + } + } + } + if (pages.length === 0) return false + const now = Date.now() + for (const m of pages) { + if (m === null || typeof m !== "object") continue + const msg = m as { role?: unknown; text?: unknown; content?: unknown; time?: unknown } + if (msg.role !== "user") continue + let t = "" + if (typeof msg.text === "string") { + t = msg.text + } else if (Array.isArray(msg.content)) { + t = (msg.content as unknown[]) + .filter( + (p): p is { type: string; text: string } => + !!p && + typeof p === "object" && + (p as { type?: unknown }).type === "text" && + typeof (p as { text?: unknown }).text === "string", + ) + .map((p) => p.text) + .join("") + } + // Newest-first: the first user message decides. Anything newer + // than our prod (user typed after it) is progress, not a dupe. + if (t !== text) return false + const created = (msg.time as { created?: unknown } | undefined)?.created + if (typeof created === "number" && now - created > RECENT_PROD_WINDOW_MS) return false + return true + } + return false + } + async function injectOnce( sid: string, text: string, @@ -1559,7 +1633,20 @@ export default define({ return false } w.lastInjectAt = Date.now() - return notifyAndPrompt(sid, text, notification) + // Cross-instance backstop for the in-memory skip above: stacked + // watchdogs cannot see each other's counters, but they share the + // log. Applies to every caller except the abort+resume escalation, + // which must go through by construction. + if (!allowDuringSelfAbort && (await recentOwnProdInLog(sid, text))) { + dbg(`${short(sid)} duplicate continue suppressed — identical prod already in session log`) + return false + } + const sent = await notifyAndPrompt(sid, text, notification) + if (sent) { + w.lastProdText = text + w.prodAssistantSnapshot = w.lastAssistantText + } + return sent } function cleanupIdleSessions() { From 7fb1b9fe3833813d366e9a543a54671f5de3be35 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:52:31 -0400 Subject: [PATCH 33/57] v2: one live instance per process, and a per-session inject mutex MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from d74debe (auto-resume/v2-todo-message-log). Test-only delta vs the original: none — src/v2/index.duplicate-instance.test.ts and src/v2/index.inject-lock.test.ts apply as-is. Docs hunk dropped (docs/v2/backport-to-v1.md no longer exists on this branch). --- src/v2/index.duplicate-instance.test.ts | 155 ++++++++++++++++++++++++ src/v2/index.inject-lock.test.ts | 113 +++++++++++++++++ src/v2/index.ts | 92 +++++++++++++- 3 files changed, 357 insertions(+), 3 deletions(-) create mode 100644 src/v2/index.duplicate-instance.test.ts create mode 100644 src/v2/index.inject-lock.test.ts diff --git a/src/v2/index.duplicate-instance.test.ts b/src/v2/index.duplicate-instance.test.ts new file mode 100644 index 0000000..24bab80 --- /dev/null +++ b/src/v2/index.duplicate-instance.test.ts @@ -0,0 +1,155 @@ +import { describe, test, expect } from "bun:test" +import { readFileSync, rmSync } from "node:fs" +import { existsSync } from "node:fs" +import { join } from "node:path" +import { tmpdir } from "node:os" +import plugin from "./index" + +const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_dupinst" +let counter = 0 + +/** The real v2 seam: `event.subscribe` returns an async iterable. */ +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const FAST = { + chunkTimeoutMs: 50, + gracePeriodMs: 0, + checkIntervalMs: 20, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + loopMaxContinues: 99, + injectIntervalMs: 0, + visibleContinue: true, +} + +/** + * Boot the plugin `instances` times against ONE host ctx, exactly as the loader + * does — the live log shows six `ready` lines per server start, all from a + * single entrypoint and a single server process. Then let ONE stalled turn + * elapse and report every injection produced. + */ +async function bootStormed(instances: number, opts: Record = {}): Promise<{ injected: any[]; logs: string[] }> { + const injected: any[] = [] + const streams = Array.from({ length: instances }, () => makeEventStream()) + let next = 0 + const logFile = join(tmpdir(), `auto-resume-dup-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + // ONE event source shared by every setup: `subscribe` hands out a fresh + // async iterable per caller, which is what the host does. + event: { + subscribe: ({ signal }: { signal?: AbortSignal } = {}) => { + const s = streams[next++] ?? streams[streams.length - 1] + signal?.addEventListener("abort", () => s.close()) + return s + }, + }, + options: { ...FAST, ...opts, logFile, debug: true }, + session: { + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ kind: "synthetic", text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ kind: "prompt", text: a?.text }), {}), + }, + client: { + session: { + get: async () => ({ data: {} }), + message: { list: async () => ({ data: [], cursor: null }) }, + }, + }, + storage: { get: async () => ({ todos: [], updatedAt: Date.now() }), set: async () => {}, remove: async () => {} }, + } + + const cleanups: Array<(() => void) | undefined> = [] + for (let i = 0; i < instances; i++) cleanups.push(await (plugin as any).setup(ctx)) + + // ONE turn that starts and then goes silent. Every live instance subscribes to + // the same host stream, so all of them see it — that fan-out IS the storm. + streams.forEach((s) => s.push(ev("session.execution.started"))) + await wait(700) + for (const c of cleanups) (c as (() => void) | undefined)?.() + + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { injected, logs } +} + +describe("v2: N live setups must not become N identical injections", () => { + test("CONTROL: ONE setup on a stalled turn injects exactly once", async () => { + const { injected } = await bootStormed(1, { maxRetries: 1 }) + expect(injected.length).toBe(1) + }, 20_000) + + // The storm. setup() re-runs on every config reload; the log shows six `ready` + // lines with no `stopped` between them. Each setup built a private `sessions` + // map, so each owned a private resumeAttempts counter: six watchdogs, six + // "attempt 1/3" lines in the same millisecond, six identical continues with + // no wait between them. + test("SIX live setups -> ONE injection, not six", async () => { + const { injected } = await bootStormed(6, { maxRetries: 1 }) + expect(injected.length).toBe(1) + }, 25_000) + + // The behavioural test above is the real proof. These two pin the MECHANISM so + // the fix cannot be undone by a refactor that keeps the symptom away by luck. + test("MECHANISM: only the LAST ready is left running, and only it stalls", async () => { + const { logs } = await bootStormed(6, { maxRetries: 1 }) + const ready = logs.filter((l) => l.includes("ready (opencode v2)")).length + const stopped = logs.filter((l) => l.includes("[auto-resume] stopped")).length + const stalls = logs.filter((l) => l.includes("stall detected")).length + // Six setups still each announce themselves (the loader asked for six)... + expect(ready).toBe(6) + // ...but five are torn down on arrival, leaving one watchdog. + expect(stopped).toBeGreaterThanOrEqual(5) + // One live instance, therefore one counter, therefore one injection. + expect(stalls).toBe(1) + }, 25_000) + + test("structural: a module-scope singleton guards setup() against stacking", () => { + const singletonAt = SOURCE.indexOf("let activeInstance:") + expect(singletonAt).toBeGreaterThan(-1) + const setupAt = SOURCE.indexOf("setup: async (ctx: AutoResumePluginInput) => {") + // Declared BEFORE setup opens, and consulted inside it: that ordering is the fix. + expect(singletonAt).toBeLessThan(setupAt) + expect(SOURCE).toContain("activeInstance.dispose()") + expect(SOURCE).toContain("activeInstance = { dispose }") + }) +}) \ No newline at end of file diff --git a/src/v2/index.inject-lock.test.ts b/src/v2/index.inject-lock.test.ts new file mode 100644 index 0000000..fb03fa1 --- /dev/null +++ b/src/v2/index.inject-lock.test.ts @@ -0,0 +1,113 @@ +import { describe, test, expect } from "bun:test" +import { readFileSync } from "node:fs" +import { join } from "node:path" +import plugin from "./index" + +const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_lock" + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const FAST = { + chunkTimeoutMs: 50, + gracePeriodMs: 0, + checkIntervalMs: 20, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + loopMaxContinues: 99, + injectIntervalMs: 0, + visibleContinue: true, +} + +/** + * A ctx whose `session.prompt` takes `slowMs` to resolve. Every caller that has + * passed its guards is then sitting in that await at the same time, which is + * exactly the window the mutex has to close. + */ +async function run(slowMs: number) { + const injected: any[] = [] + const stream = makeEventStream() + const ctx: any = { + event: stream, + options: { ...FAST, maxRetries: 1, debug: true }, + session: { + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ kind: "synthetic", text: a?.text }), {}), + prompt: async (a: any) => { + // Record the send FIRST, then stall: the send is what must not repeat. + injected.push({ kind: "prompt", text: a?.text }) + await wait(slowMs) + return {} + }, + }, + client: { + session: { get: async () => ({ data: {} }), message: { list: async () => ({ data: [], cursor: null }) } }, + }, + storage: { get: async () => ({ todos: [], updatedAt: Date.now() }), set: async () => {}, remove: async () => {} }, + } + const cleanup = await (plugin as any).setup(ctx) + stream.push(ev("session.execution.started")) + await wait(800) + ;(cleanup as (() => void) | undefined)?.() + return injected +} + +describe("v2: injectOnce is serialized per session", () => { + test("CONTROL: one stalled turn still injects exactly once", async () => { + expect((await run(0)).length).toBe(1) + }, 20_000) + + // Same-ms volleys survived the cross-instance log check because it is + // check-then-act: `recentOwnProdInLog` read the log, then awaited, then sent. + // Six interleaved callers all read the same empty log. The mutex closes that + // window — it is why the duplicate guard alone was not enough. + test("a SLOW send must not let a concurrent caller through", async () => { + expect((await run(250)).length).toBe(1) + }, 25_000) + + test("structural: the mutex wraps injectOnce, and is per-session keyed", () => { + expect(SOURCE).toContain("const injectLocks = new Map>()") + // injectOnce must be a thin wrapper that takes the lock; the body must not. + expect(SOURCE).toContain("return withInjectLock(sid, () =>") + expect(SOURCE).toContain("async function injectOnceLocked(") + // The lock must be released in a finally, or one failed send wedges the session. + expect(SOURCE).toMatch(/finally \{[\s\S]{0,200}release\(\)/) + }) +}) \ No newline at end of file diff --git a/src/v2/index.ts b/src/v2/index.ts index 8321240..d70a09f 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -1072,10 +1072,39 @@ function isTaskToolCall(ev: V2Event): boolean { // Plugin // --------------------------------------------------------------------------- +/** + * The one live instance of this plugin in this process, if any. + * + * `setup()` is re-run by the loader on every config reload — the live log shows + * 2523 loads of this ONE entrypoint against a single server process, and bursts + * of SIX `ready` lines with no `stopped` between them. Every setup call builds + * its own `sessions` map, its own watchdog interval and its own event pump, so + * six live instances meant six private `resumeAttempts` counters: one stall + * produced six `resume attempt 1/3` lines in the same millisecond and six + * identical continues with no wait between them (2026-10-03, session + * ses_efcc2ade…kJmjQR3o). The counter was never broken — each copy had its own. + * + * Last-wins: a fresh setup disposes the previous instance before arming its own, + * so exactly one watchdog and one counter exist no matter how often the loader + * re-runs setup. Disposal is deliberately not left to the host, because the + * storms prove it does not always happen. + */ +let activeInstance: { dispose: () => void } | null = null + export default define({ id: "auto-resume.v2", setup: async (ctx: AutoResumePluginInput) => { + // Supersede any instance the loader left running. Wrapped: a throwing + // predecessor must not block the new one from arming. + if (activeInstance) { + try { + activeInstance.dispose() + } catch { + /* predecessor already torn down */ + } + activeInstance = null + } const opts = (ctx.options ?? {}) as AutoResumeOptions const chunkTimeoutMs = opts.chunkTimeoutMs ?? DEFAULT_CHUNK_TIMEOUT_MS @@ -1606,11 +1635,50 @@ export default define({ return false } - async function injectOnce( + /** + * Per-session async mutex around the whole injection. Bun is + * single-threaded, so a promise chain serializes overlapping callers: + * stacked watchdogs interleave at every await, and without this the + * log check below runs six-wide against the same empty log before any + * send lands. First holder checks-then-sends; the rest re-check after + * and stand down on sight of the first prod. + */ + const injectLocks = new Map>() + async function withInjectLock(sid: string, fn: () => Promise): Promise { + const prev = injectLocks.get(sid) ?? Promise.resolve() + let release!: () => void + const current = new Promise((resolve) => { + release = resolve + }) + // The chain tail is what the NEXT caller waits on, so it must resolve only + // after this holder's work finishes — hence `prev.then(() => current)`. + const tail = prev.then(() => current) + injectLocks.set(sid, tail) + await prev + try { + return await fn() + } finally { + release() + // Drop the key once we are the last holder, so a long-lived process + // does not accumulate one entry per session it ever watched. + if (injectLocks.get(sid) === tail) injectLocks.delete(sid) + } + } + + /** + * The injection body. Split out of `injectOnce` so the per-session mutex + * wraps the WHOLE check-then-act: every guard above (subagent, self-abort, + * busy, debounce, duplicate) is a read, and the send is the write. Stacked + * callers interleave at every `await`, so six watchdogs could each pass all + * six reads against pre-send state and then all six send. Serialized here, + * the first caller to finish sends and the rest re-read its result. + */ + async function injectOnceLocked( sid: string, text: string, notification: string, - allowDuringSelfAbort = false, + allowDuringSelfAbort: boolean, + checkDuplicate: boolean, ): Promise { const w = ensureWatch(sid) // A subagent is not ours to recover. Checked first, before every other @@ -1649,6 +1717,19 @@ export default define({ return sent } + /** `injectOnceLocked` serialized per session; see the mutex's own comment. */ + async function injectOnce( + sid: string, + text: string, + notification: string, + allowDuringSelfAbort = false, + checkDuplicate = false, + ): Promise { + return withInjectLock(sid, () => + injectOnceLocked(sid, text, notification, allowDuringSelfAbort, checkDuplicate), + ) + } + function cleanupIdleSessions() { const now = Date.now() const busy = new Set() @@ -3777,7 +3858,7 @@ export default define({ ) // Cleanup: stop timers and the event pump; OpenCode awaits this on disable/reload/shutdown. - return () => { + const dispose = () => { running = false eventAbort.abort() clearInterval(watchdog) @@ -3790,7 +3871,12 @@ export default define({ w.toolTextTimer = null } sessions.clear() + // Only clear the singleton if it is still OURS: a later setup may already + // have superseded us, and unregistering it would strand a live watchdog. + if (activeInstance?.dispose === dispose) activeInstance = null log("info", "stopped") } + activeInstance = { dispose } + return dispose }, }) From 4ccbf80c6d043e52b1e9fd94fc2d5ebf15f51ee8 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:53:10 -0400 Subject: [PATCH 34/57] =?UTF-8?q?v2:=20put=20the=20singleton=20on=20global?= =?UTF-8?q?This=20=E2=80=94=20module=20scope=20cannot=20see=20re-evaluatio?= =?UTF-8?q?ns?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from d756146 (auto-resume/v2-todo-message-log). Deliberate delta vs the original: the ready line keeps only the mod= instance tag — the visibleContinue= interpolation is dropped because the visibleContinue option does not exist on this branch. src/v2/index.singleton-registry.test.ts applies as-is. --- src/v2/index.singleton-registry.test.ts | 132 ++++++++++++++++++++++++ src/v2/index.ts | 77 +++++++++++--- 2 files changed, 196 insertions(+), 13 deletions(-) create mode 100644 src/v2/index.singleton-registry.test.ts diff --git a/src/v2/index.singleton-registry.test.ts b/src/v2/index.singleton-registry.test.ts new file mode 100644 index 0000000..0c34545 --- /dev/null +++ b/src/v2/index.singleton-registry.test.ts @@ -0,0 +1,132 @@ +import { describe, test, expect } from "bun:test" +import { readFileSync } from "node:fs" +import { join } from "node:path" +import plugin from "./index" + +const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_registry" + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const FAST = { + chunkTimeoutMs: 50, + gracePeriodMs: 0, + checkIntervalMs: 20, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + loopMaxContinues: 99, + injectIntervalMs: 0, + visibleContinue: true, +} + +type Boot = { injected: any[]; registryLive: unknown; registryKey: string } + +async function boot(instances: number): Promise { + const injected: any[] = [] + const streams = Array.from({ length: instances }, () => makeEventStream()) + let next = 0 + const ctx: any = { + event: { + subscribe: ({ signal }: { signal?: AbortSignal } = {}) => { + const s = streams[next++] ?? streams[streams.length - 1] + signal?.addEventListener("abort", () => s.close()) + return s + }, + }, + options: { ...FAST, maxRetries: 1 }, + session: { + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ kind: "synthetic", text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ kind: "prompt", text: a?.text }), {}), + }, + client: { + session: { get: async () => ({ data: {} }), message: { list: async () => ({ data: [], cursor: null }) } }, + }, + storage: { get: async () => ({ todos: [], updatedAt: Date.now() }), set: async () => {}, remove: async () => {} }, + } + const cleanups: Array<(() => void) | undefined> = [] + for (let i = 0; i < instances; i++) cleanups.push(await (plugin as any).setup(ctx)) + + const key = `__auto_resume_singleton__:${process.pid}` + const afterSetup = (globalThis as any)[key]?.live + + streams.forEach((s) => s.push(ev("session.execution.started"))) + await wait(700) + const duringStall = (globalThis as any)[key]?.live + for (const c of cleanups) (c as (() => void) | undefined)?.() + + return { injected, registryLive: duringStall ?? afterSetup, registryKey: key } +} + +describe("v2: the singleton lives on globalThis, so re-evaluations share it", () => { + test("CONTROL: one setup on a stalled turn injects once", async () => { + const { injected } = await boot(1) + expect(injected.length).toBe(1) + }, 20_000) + + test("six setups in one process -> ONE registry entry, ONE injection", async () => { + const { injected, registryLive } = await boot(6) + // The registry must name exactly one live instance... + expect(registryLive).toBeDefined() + expect(typeof (registryLive as any).dispose).toBe("function") + // ...and only that one may inject. + expect(injected.length).toBe(1) + }, 25_000) + + test("the registry key is process-scoped, not module-scoped", async () => { + const { registryKey } = await boot(1) + expect(registryKey).toBe(`__auto_resume_singleton__:${process.pid}`) + }) + + // The shipped bug: `let activeInstance` in module scope. Six evaluations of + // the bundle each hold their own, so each believes it is the only live one and + // the storms continued (8 lines, same millisecond, 19:54:19). + test("structural: the singleton is reached THROUGH globalThis", () => { + expect(SOURCE).toContain("globalThis as unknown as Record") + expect(SOURCE).toContain("function singletonRegistry()") + // No module-scope `activeInstance` may remain: it cannot see other copies. + expect(SOURCE).not.toMatch(/^let activeInstance/m) + // The teardown must go through the registry too, and must clear the slot + // only when it still holds OUR disposer — otherwise a late teardown from a + // superseded copy unregisters the live one. + expect(SOURCE).toMatch(/if \(reg\.live\?\.dispose === dispose\) reg\.live = undefined/) + expect(SOURCE).toContain("registry.live = { dispose }") + }) +}) \ No newline at end of file diff --git a/src/v2/index.ts b/src/v2/index.ts index d70a09f..8a3b5a5 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -1089,21 +1089,69 @@ function isTaskToolCall(ev: V2Event): boolean { * re-runs setup. Disposal is deliberately not left to the host, because the * storms prove it does not always happen. */ -let activeInstance: { dispose: () => void } | null = null +/** + * The one live instance of this plugin in this PROCESS, shared across every + * evaluation of this bundle. + * + * It lives on `globalThis`, not in module scope, and that is the whole fix. The + * loader re-EVALUATES the plugin file on every config reload: the live log shows + * one entrypoint URL, one pid, and a fresh module identity each time + * (`mod=1812055.c6d4mu`, `.72a9tl`, `.dt27x7`, … six distinct copies inside one + * server process). Module scope cannot help, because each copy gets its own + * `activeInstance` — a singleton fix shipped that way and the storms continued + * (19:54:19, eight identical `resume attempt 1/3` lines in 5ms). + * + * `globalThis` is shared by all module instances in a process (verified on both + * node and bun), so a key here is the one channel that survives re-evaluation. + * Each setup finds its predecessor through the registry and disposes it before + * arming itself: last wins, so exactly one watchdog and one `resumeAttempts` + * counter exist however often the loader reloads us. Disposal cannot be left to + * the host, because the storms prove it does not always happen. + * + * Keyed by pid + plugin id: two servers in one process cannot happen, but two + * config trees (v1 and v2) can load different builds, and they must not dispose + * each other. + */ +const REGISTRY_KEY = `__auto_resume_singleton__:${process.pid}` + +type SingletonRegistry = { live?: { dispose: () => void } } + +function singletonRegistry(): SingletonRegistry { + const g = globalThis as unknown as Record + let reg = g[REGISTRY_KEY] + if (!reg) { + reg = {} + g[REGISTRY_KEY] = reg + } + return reg +} + +/** + * Identity of THIS module evaluation, stamped into the ready line. + * + * `pid` separates processes; `eval` separates module evaluations inside one + * process. Six distinct values under one pid is what exposed the real cause — + * keep this until the storms are gone, because it is the only way to tell a + * re-evaluation apart from a genuine second server. + */ +const MODULE_INSTANCE = `${process.pid}.${Math.random().toString(36).slice(2, 8)}` export default define({ id: "auto-resume.v2", setup: async (ctx: AutoResumePluginInput) => { - // Supersede any instance the loader left running. Wrapped: a throwing - // predecessor must not block the new one from arming. - if (activeInstance) { + // Supersede any instance the loader left running — found through the + // process-wide registry, so it also catches instances belonging to OTHER + // evaluations of this bundle. Wrapped: a throwing predecessor must not + // block the new one from arming. + const registry = singletonRegistry() + if (registry.live) { try { - activeInstance.dispose() + registry.live.dispose() } catch { /* predecessor already torn down */ } - activeInstance = null + registry.live = undefined } const opts = (ctx.options ?? {}) as AutoResumeOptions @@ -3853,7 +3901,7 @@ export default define({ log( "info", - `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s warmup=${warmupMs}ms stall=${busyStallStrategy}` + + `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s warmup=${warmupMs}ms stall=${busyStallStrategy} mod=${MODULE_INSTANCE}` + (gatedInUse.length > 0 ? ` accepted-but-inert=${gatedInUse.join(",")}` : ""), ) @@ -3871,12 +3919,15 @@ export default define({ w.toolTextTimer = null } sessions.clear() - // Only clear the singleton if it is still OURS: a later setup may already - // have superseded us, and unregistering it would strand a live watchdog. - if (activeInstance?.dispose === dispose) activeInstance = null - log("info", "stopped") - } - activeInstance = { dispose } + // Only clear the registry slot if it is still OURS: a later setup may + // already have superseded us, and unregistering it would strand a live + // watchdog. The identity check matters across copies too, since they + // share one registry. + const reg = singletonRegistry() + if (reg.live?.dispose === dispose) reg.live = undefined + log("info", `stopped mod=${MODULE_INSTANCE}`) + } + registry.live = { dispose } return dispose }, }) From 8a34aa5c24d4f9a40fef13b7c51a93cb2d5ddebd Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:53:31 -0400 Subject: [PATCH 35/57] =?UTF-8?q?v2:=20thinking/tool-call=20turns=20are=20?= =?UTF-8?q?alive=20=E2=80=94=20dead-stream=20detector=20no=20longer=20fire?= =?UTF-8?q?s=20into=20active=20work?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from 2047d7c (auto-resume/v2-todo-message-log): applies cleanly, no deltas. 13 dead-stream tests pass. --- src/v2/index.dead-stream.test.ts | 33 ++++++++++++++++++++++++++- src/v2/index.ts | 38 ++++++++++++++++++++++++++++---- 2 files changed, 66 insertions(+), 5 deletions(-) diff --git a/src/v2/index.dead-stream.test.ts b/src/v2/index.dead-stream.test.ts index 2efca04..3d1666d 100644 --- a/src/v2/index.dead-stream.test.ts +++ b/src/v2/index.dead-stream.test.ts @@ -86,6 +86,16 @@ const finished = (opts: { finish?: string; text?: string; output?: number; reaso tokens: { input: 100, output: opts.output ?? 400, reasoning: 0, cache: { read: 0, write: 0 } }, }) +/** A finished turn that issued tool calls but no chatter text: active work. */ +const finishedWithTools = () => ({ + type: "assistant", + id: "msg_a3", + time: { created: Date.now() - 60_000 }, + content: [{ type: "tool", id: "call_9", name: "bash", state: { status: "completed" } }], + finish: "stop", + tokens: { input: 100, output: 400, reasoning: 0, cache: { read: 0, write: 0 } }, +}) + /** An intermediate tool-call step: no finish reason, so the walk skips it. */ const toolStep = () => ({ type: "assistant", @@ -114,7 +124,7 @@ type Harness = { injected: Array<{ text?: string }>; logs: string[] } async function replay( messages: unknown[], opts: Record = {}, - extra: { serverRunning?: boolean } = {}, + extra: { serverRunning?: boolean; preIdleEvents?: Array<{ type: string; data?: Record }> } = {}, ): Promise { const injected: Harness["injected"] = [] const stream = makeEventStream() @@ -147,6 +157,7 @@ async function replay( for (const e of [ ev("session.execution.started"), ev("session.step.started"), + ...(extra.preIdleEvents ?? []).map((p) => ev(p.type, p.data ?? {})), ev("session.step.ended"), ev("session.idle"), ]) { @@ -207,6 +218,26 @@ describe("v2: silent dead stream", () => { expect(injected).toHaveLength(1) }) + test("a finished turn carrying tool calls is working, not a dead stream", async () => { + // ses_efaec2f99ffexosULGDJJ8i6sA 2026-10-04: a thinking model doing tool + // work ends turns with finish=stop, hundreds of output tokens, and no + // text parts. Judging that "silent" fires a visible continue into + // active work. Tool calls are delivered work — not silence. + const { injected, logs } = await replay([finishedWithTools()]) + expect(injected).toEqual([]) + expect(logs.some((l) => l.includes("silent dead stream"))).toBe(false) + }) + + test("tools still in flight veto the dead-stream inject", async () => { + // Belt and braces for the race the test above cannot see: the finished + // message predates the tool events, so the message looks dead while the + // calls it issued have not answered yet. + const { injected } = await replay([finished({ output: 400 })], {}, { + preIdleEvents: [{ type: "session.tool.called", data: { tool: "bash", id: "call_9" } }], + }) + expect(injected).toEqual([]) + }) + test("a session the server still reports as running is left alone", async () => { // A provider that is quietly retrying looks identical from the event // stream. Injecting into that would turn a recovering session into a diff --git a/src/v2/index.ts b/src/v2/index.ts index 8a3b5a5..9dec2da 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -210,6 +210,12 @@ interface SessionWatch { /** Tool calls started but not finished, tracked from the tool lifecycle events. * The orphan watch must never abort a session that is legitimately working. */ pendingTools: number + /** Snapshot of `pendingTools` taken at the `session.idle` transition, before + * `markIdle()` zeroes the live counter. Idle does not mean nothing is in + * flight — the event fires between steps while tools run — so any idle-path + * check that reads the live counter always sees zero. Same class of bug the + * `busyBefore` sample above guards against. */ + toolsInFlightAtIdle: number unknownToolErrors: Map /** Set once a suggestion has been injected, so it is sent at most once per * user message rather than on every idle. */ @@ -1398,6 +1404,7 @@ export default define({ toolTextTimer: null, lastSubagentCheckAt: 0, pendingTools: 0, + toolsInFlightAtIdle: 0, unknownToolErrors: new Map(), unknownToolSuggestionSent: false, checkedToolPartIDs: new Set(), @@ -2294,8 +2301,12 @@ export default define({ * finish, and walking back past a delivered answer to one of those would * recover a session that just used a tool. * - * Returns `null` when the newest finished message carried text, or when - * there is no finished message at all. + * Returns `null` when the newest finished message carried text — or when it + * carried tool calls, which are delivered work, not silence. A thinking + * model doing tool work ends turns with a finish reason, spent output + * tokens, and no text parts; judging that "silent" fires a continue + * into active work (ses_efaec2f99ffexosULGDJJ8i6sA, 2026-10-04). + * Returns `null` as well when there is no finished message at all. */ function lastSilentDeadStream(messages: unknown[]): { finish: string; outputTokens: number } | null { for (let i = messages.length - 1; i >= 0; i--) { @@ -2308,9 +2319,16 @@ export default define({ if (msg?.type !== "assistant") continue const finish = typeof msg.finish === "string" ? msg.finish : undefined if (!finish) continue - const hasText = Array.isArray(msg.content) && - msg.content.some((p) => p?.type === "text" && typeof p.text === "string" && p.text.length > 0) + const parts = Array.isArray(msg.content) ? msg.content : [] + const hasText = parts.some((p) => p?.type === "text" && typeof p.text === "string" && p.text.length > 0) if (hasText) return null + // Same tool-part convention as hasPendingUserInput: a finished + // turn that issued tool calls did work, even with no chatter. + const hasToolCalls = parts.some((p) => { + const t = p?.type ?? "" + return t === "tool_use" || t === "tool" || t === "tool_call" || t.startsWith("tool") + }) + if (hasToolCalls) return null return { finish, outputTokens: posNum(msg.tokens?.output) } } return null @@ -2338,6 +2356,14 @@ export default define({ return false } const w = ensureWatch(sid) + // Tools still awaiting results are working, not stalled — even when + // the finished message predates the tool events and looks dead. + // Reads the at-idle snapshot, not the live counter: markIdle zeroes + // it on the transition, and idle fires while tools run. + if (w.toolsInFlightAtIdle > 0) { + dbg(`${short(sid)} silent dead stream, but ${w.toolsInFlightAtIdle} tool(s) still in flight — working, not stalled`) + return true + } // Ask the server before injecting, not just our own flag. A provider // that is quietly retrying looks exactly like a dead stream from the // event stream, and the event may not have arrived yet when the turn @@ -3530,6 +3556,10 @@ export default define({ // of busy sessions behind it. Sampling after markIdle would make every // idle look like a drop to zero and the arming condition unreachable. const busyBefore = busySessionsForOrphanWatch().length + // Snapshot before the transition for the same reason: markIdle + // zeroes the live counter, and idle fires while tools run. + const wPre = ensureWatch(sid) + wPre.toolsInFlightAtIdle = wPre.pendingTools markIdle(sid) const w = ensureWatch(sid) w.pendingRecoveryArmed = false From 7fdc472f66e710b824bef84a0d8036c2b68b1300 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:53:58 -0400 Subject: [PATCH 36/57] v2: own prompts don't re-arm budgets; bash->shell alias before fuzzy match Ported from 808fb80 (auto-resume/v2-todo-message-log): whitespace-only conflict at the ensureWatch initializer resolved to HEAD indentation. 18 unknown-tool tests pass. --- src/v2/index.ts | 45 ++++++++++++++++++++++++-- src/v2/index.unknown-tool.test.ts | 54 +++++++++++++++++++++++++++++++ 2 files changed, 96 insertions(+), 3 deletions(-) diff --git a/src/v2/index.ts b/src/v2/index.ts index 9dec2da..cea8d3a 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -156,6 +156,11 @@ interface SessionWatch { /** Last injected prod text + assistant snapshot at prod time, for duplicate suppression. */ lastProdText: string prodAssistantSnapshot: string + /** Texts this plugin injected (visible channel posts them as real user + * messages). `noteInboundUserMessage` must not read them as new user + * instructions, or every injection re-arms the budgets it just spent and + * the same errors refire on the next idle. Capped; identity is exact text. */ + ownPromptTexts: string[] recovering: boolean /** Latched when a failure carries an OOC error that `continue` can never clear; recovery (continue + abort+resume) is refused until a genuine user/agent turn. */ oocLocked: boolean @@ -1028,6 +1033,23 @@ function levenshtein(a: string, b: string): number { * sloppy. Without that, a two-letter name would match almost anything and the * suggestion would be noise. */ +/** + * Names models invent for tools that exist under another name. Edit distance + * cannot bridge these ("bash"→"shell" is 4 against a threshold of 2), so the + * map is consulted before the fuzzy matcher. The target must be registered — + * an unguarded guess names a second nonexistent tool and teaches the model + * the registry lies. + */ +const TOOL_NAME_ALIASES: Record = { + bash: "shell", +} + +function resolveToolSuggestion(wrongName: string, available: string[]): string | null { + const alias = TOOL_NAME_ALIASES[wrongName.toLowerCase()] + if (alias && available.includes(alias)) return alias + return suggestClosestTool(wrongName, available) +} + function suggestClosestTool(wrongName: string, available: string[]): string | null { const lower = wrongName.toLowerCase() let best: string | null = null @@ -1381,6 +1403,7 @@ export default define({ lastInjectAt: 0, lastProdText: "", prodAssistantSnapshot: "", + ownPromptTexts: [], recovering: false, oocLocked: false, oocLockReason: null, @@ -1768,6 +1791,8 @@ export default define({ if (sent) { w.lastProdText = text w.prodAssistantSnapshot = w.lastAssistantText + w.ownPromptTexts.push(text) + if (w.ownPromptTexts.length > 10) w.ownPromptTexts.shift() } return sent } @@ -2466,7 +2491,7 @@ export default define({ * budget each time it announced, and the nudge would never stop. */ function noteInboundUserMessage(sid: string, w: SessionWatch, messages: unknown[]): void { - let latest: { id?: string; at?: number } | null = null + let latest: { id?: string; at?: number; text?: string } | null = null for (let i = messages.length - 1; i >= 0; i--) { const m = messages[i] as { id?: string @@ -2474,10 +2499,14 @@ export default define({ role?: string time?: { created?: number } info?: { time?: { created?: number }; role?: string } + content?: Array<{ type?: string; text?: string }> } const isUser = m?.type === "user" || m?.role === "user" || m?.info?.role === "user" if (!isUser) continue - latest = { id: typeof m.id === "string" ? m.id : undefined, at: m.time?.created ?? m.info?.time?.created } + const text = Array.isArray(m?.content) + ? m.content.filter((p) => p?.type === "text" && typeof p.text === "string").map((p) => p.text as string).join("\n") + : undefined + latest = { id: typeof m.id === "string" ? m.id : undefined, at: m.time?.created ?? m.info?.time?.created, text } break } if (!latest) return @@ -2487,6 +2516,16 @@ export default define({ if (!isNew) return w.lastUserMessageID = latest.id if (typeof latest.at === "number") w.lastUserMessageSeenAt = latest.at + // Our own injections travel the visible channel as real user messages. + // They start a new turn but carry no new instructions: re-arming on + // them clears the budgets just spent and the same errors refire on + // the next idle (ses_ef81e8561ffeXyx3jAzKl5lltv, 2026-10-04 — four + // identical unknown-tool prompts, each "2x"). Tracking above stays + // truthful; only the clearing is skipped. + if (typeof latest.text === "string" && latest.text.length > 0 && w.ownPromptTexts.includes(latest.text)) { + dbg(`${short(sid)} newest user message is our own prompt — not re-arming budgets`) + return + } if (w.doneClaimAttempts > 0 || w.doneClaimOpenTodosAttempts > 0) { dbg(`${short(sid)} new user message — re-arming the done-claim budgets`) } @@ -2579,7 +2618,7 @@ export default define({ const count = (w.unknownToolErrors.get(toolName) ?? 0) + 1 w.unknownToolErrors.set(toolName, count) if (count < UNKNOWN_TOOL_THRESHOLD) continue - const suggestion = suggestClosestTool(toolName, available) + const suggestion = resolveToolSuggestion(toolName, available) const toolList = available.slice(0, UNKNOWN_TOOL_LIST_LIMIT).join(", ") const prompt = suggestion ? `You tried to use the tool "${toolName}" ${count} times, but it does not exist. ` + diff --git a/src/v2/index.unknown-tool.test.ts b/src/v2/index.unknown-tool.test.ts index 9779572..8b67164 100644 --- a/src/v2/index.unknown-tool.test.ts +++ b/src/v2/index.unknown-tool.test.ts @@ -264,6 +264,60 @@ describe("v2: naming a replacement for a tool that does not exist", () => { expect(suggestions(h)[1].text).toContain('"globb"') await teardown(h) }) + + test("our own suggestion is not a new request and does not re-arm", async () => { + // ses_ef81e8561ffeXyx3jAzKl5lltv 2026-10-04: the visible channel posts + // our suggestion as a real user message with a new id. The next idle + // read it as new instructions, cleared the error map, the latch and + // the examined-parts set, recounted the SAME parts back to threshold, + // and suggested again — four identical "(none)" prompts, each "2x". + const tools = [{ id: "read" }, { id: "shell" }, { id: "glob" }] + const h = await setup({ + history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "bash")), assistantWith(toolPart("call_2", "bash"))], + tools, + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) + const own = suggestions(h)[0] + h.history.push({ + type: "user", + id: "msg_own_suggestion", + time: { created: OLD + 3_000 }, + content: [{ type: "text", text: own.text }], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) + await teardown(h) + }) + + test("a v1 tool name maps to its v2 rename instead of (none)", async () => { + // "bash" is edit-distance 4 from "shell" against a threshold of 2, so + // the fuzzy matcher can never bridge it. A static alias tried first can. + const h = await setup({ + history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "bash")), assistantWith(toolPart("call_2", "bash"))], + tools: [{ id: "read" }, { id: "shell" }, { id: "glob" }], + }) + await goIdle(h) + const said = suggestions(h) + expect(said).toHaveLength(1) + expect(said[0].text).toContain('The closest matching tool is "shell"') + await teardown(h) + }) + + test("CONTROL: an alias whose target is not registered falls back to the list", async () => { + // The alias must never name a tool that does not exist: a wrong guess + // costs a second failure round and teaches the model the registry lies. + const h = await setup({ + history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "bash")), assistantWith(toolPart("call_2", "bash"))], + tools: [{ id: "read" }, { id: "write" }, { id: "glob" }], + }) + await goIdle(h) + const said = suggestions(h) + expect(said).toHaveLength(1) + expect(said[0].text).toContain("Please check the available tools") + expect(said[0].text).not.toContain("The closest matching tool is") + await teardown(h) + }) }) describe("v2: what the unknown-tool check ignores", () => { From 333a4e1a42c8b2a19ee2e4bf83d74259eab5c2a7 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:54:37 -0400 Subject: [PATCH 37/57] v2: quota failures stand down on a gate ladder instead of retrying into the ban MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from aaac082 (auto-resume/v2-todo-message-log). Deliberate delta vs the original: the inert configProbe option (interface, setup const, RECOGNISED_OPTIONS, README row) is NOT ported — it belongs to the excluded probe/docs scope. 6 rate-limit tests pass. --- README.md | 1 + src/v2/index.rate-limit.test.ts | 232 ++++++++++++++++++++++++++++++++ src/v2/index.ts | 112 ++++++++++++++- 3 files changed, 341 insertions(+), 4 deletions(-) create mode 100644 src/v2/index.rate-limit.test.ts diff --git a/README.md b/README.md index b03947d..3659f42 100644 --- a/README.md +++ b/README.md @@ -522,6 +522,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | +| `rateLimitCooldownsMs` | v2 only | Per-attempt cooldowns (ms) before a rate-limited session may be retried, evaluated as gates on each failure and watchdog tick (no armed timers to lose on reload). Default `[900000, 1800000, 3600000, 7200000 ×5]` ≈ 12h coverage; past the ladder the session stays silent until a genuine user turn. Quota hits never consume the normal retry budget | Message patterns are matched case-insensitively. Error names use exact match. diff --git a/src/v2/index.rate-limit.test.ts b/src/v2/index.rate-limit.test.ts new file mode 100644 index 0000000..b2f1b91 --- /dev/null +++ b/src/v2/index.rate-limit.test.ts @@ -0,0 +1,232 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_ratelimit" + +/** + * Rate-limit / quota failures. + * + * Observed 2026-10-04: `session.step.failed: provider.quota Rate limit + * exceeded. Please try again later.` fired a visible continue 1s later, twice + * in two seconds — retrying straight into the ban. The failure handler + * recovered from every non-abort, non-OOC error with no rate-limit + * classification at all. + * + * The rule: a quota hit stands down on a gate ladder (gates, not timers — + * armed timeouts do not survive a plugin reload), evaluated on each failure + * and each watchdog tick. Attempts consume their own budget; past it, silence + * until a genuine user turn. A fresh turn resets the ladder. + * + * Every group carries a control. + */ + +let counter = 0 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +const QUOTA = { error: { type: "provider.quota", message: "Rate limit exceeded. Please try again later." } } +const STREAM_ERR = { error: { type: "StreamError", message: "connection reset" } } + +const userTurn = () => ({ + type: "user", + id: "msg_u0", + time: { created: Date.now() - 60 * 60_000 }, + content: [{ type: "text", text: "do the thing" }], +}) + +const OPTIONS = { + chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + debug: true, +} + +type Harness = { injected: Array<{ text?: string }>; logs: string[]; stream: any; cleanup?: () => void; logFile: string; failTurn: (err: unknown) => Promise; startTurn: () => Promise; failStep: (err: unknown) => Promise; idleTurn: () => Promise } + +async function replay(opts: Record = {}): Promise { + const injected: Harness["injected"] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-ratelimit-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile, ...opts }, + session: { + context: async () => [userTurn()], + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => { + injected.push({ text: a?.text }) + return {} + }, + prompt: async (a: any) => { + injected.push({ text: a?.text }) + return {} + }, + }, + client: { session: { get: async () => ({ data: {} }) } }, + } + + const cleanup = await (plugin as any).setup(ctx) + // Split on purpose: a new execution re-arms the ladder (fresh turn = + // activity = quota available again), while repeated step failures inside + // one execution share a single ladder episode. + const startTurn = async () => { + for (const e of [ev("session.execution.started"), ev("session.step.started")]) { + stream.push(e) + await wait(10) + } + } + const failStep = async (err: unknown) => { + stream.push(ev("session.step.failed", err as Record)) + await wait(10) + } + const idleTurn = async () => { + stream.push(ev("session.idle")) + await wait(50) + } + const failTurn = async (err: unknown) => { + await startTurn() + await failStep(err) + } + return { + injected, + logs: [], + stream, + cleanup, + logFile, + failTurn, + startTurn, + failStep, + idleTurn, + } +} + +function readLogs(h: Harness): string[] { + const logs = existsSync(h.logFile) ? readFileSync(h.logFile, "utf8").split("\n") : [] + rmSync(h.logFile, { force: true }) + return logs +} + +describe("v2: rate-limit failures stand down on a ladder", () => { + test("CONTROL: a non-rate failure recovers immediately", async () => { + const h = await replay() + await h.failTurn(STREAM_ERR) + await wait(600) + expect(h.injected.length).toBeGreaterThanOrEqual(1) + h.cleanup?.() + readLogs(h) + }) + + test("a quota hit injects nothing inside the cooldown", async () => { + const h = await replay({ rateLimitCooldownsMs: [60_000] }) + await h.failTurn(QUOTA) + await wait(300) + expect(h.injected).toEqual([]) + const logs = readLogs(h) + expect(logs.some((l) => l.includes("rate limited") && l.includes("standing down"))).toBe(true) + expect(logs.some((l) => l.includes("session.step.failed") && l.includes("resume attempt"))).toBe(false) + h.cleanup?.() + }) + + test("a second quota hit inside the cooldown still injects nothing", async () => { + const h = await replay({ rateLimitCooldownsMs: [60_000] }) + await h.startTurn() + await h.failStep(QUOTA) + await h.failStep(QUOTA) + await wait(300) + expect(h.injected).toEqual([]) + h.cleanup?.() + readLogs(h) + }) + + test("a served cooldown retries through the normal path", async () => { + const h = await replay({ rateLimitCooldownsMs: [50] }) + await h.failTurn(QUOTA) + await wait(500) + expect(h.injected).toHaveLength(1) + const logs = readLogs(h) + expect(logs.some((l) => l.includes("cooldown served"))).toBe(true) + h.cleanup?.() + }) + + test("past the ladder the session goes silent until a user turn", async () => { + const h = await replay({ rateLimitCooldownsMs: [20] }) + await h.startTurn() + await h.failStep(QUOTA) + await wait(300) + expect(h.injected).toHaveLength(1) + // Budget spent (1 rung): the next hit must not inject. + await h.failStep(QUOTA) + await wait(300) + expect(h.injected).toHaveLength(1) + const logs = readLogs(h) + expect(logs.some((l) => l.includes("budget exhausted"))).toBe(true) + h.cleanup?.() + }) + + test("a fresh turn resets the ladder", async () => { + const h = await replay({ rateLimitCooldownsMs: [20] }) + await h.startTurn() + await h.failStep(QUOTA) + await wait(300) + expect(h.injected).toHaveLength(1) + await h.failStep(QUOTA) + await wait(200) + // New turn = activity = quota available again: fresh ladder, so the + // next quota hit stands down with a new cooldown instead of silence. + // The idle first ends our recovery turn (consuming the self-cause + // flag), so the new execution reads as genuine. + await h.idleTurn() + await h.startTurn() + await h.failStep(QUOTA) + await wait(200) + const logs = readLogs(h) + expect(logs.filter((l) => l.includes("next attempt in")).length).toBeGreaterThanOrEqual(2) + h.cleanup?.() + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index cea8d3a..d96260c 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -221,6 +221,13 @@ interface SessionWatch { * check that reads the live counter always sees zero. Same class of bug the * `busyBefore` sample above guards against. */ toolsInFlightAtIdle: number + /** Rate-limit ladder state. `rateLimitedAt` is the last quota decision + * point, `rateLimitAttempts` counts served attempts in this episode, and + * `awaitingQuotaRetry` arms the watchdog-tick gate. A fresh turn resets + * all three: new activity means quota is available again. */ + rateLimitedAt: number + rateLimitAttempts: number + awaitingQuotaRetry: boolean unknownToolErrors: Map /** Set once a suggestion has been injected, so it is sent at most once per * user message rather than on every idle. */ @@ -332,6 +339,13 @@ export interface AutoResumeOptions { * variable sets the same thing and wins over neither. */ logFile?: string + /** + * Per-attempt cooldowns (ms) before a rate-limited session may be retried. + * Gates, not timers: evaluated on each failure and each watchdog tick, so + * nothing is lost to a reload. Past the ladder the session stays silent + * until a genuine user turn. Default spans ~12h for overnight coverage. + */ + rateLimitCooldownsMs?: number[] } // --------------------------------------------------------------------------- @@ -498,6 +512,15 @@ const SHELL_OPEN_MAX_MS = 30 * 60_000 /** OOC (out-of-context) errors that `continue` can never clear — recovery is locked out on these. */ const OOC_ERROR_RE = /exceeds the available context size|context size \(\d+\)|too large to compact|too many tokens|prompt is too long/i +/** Quota/rate-limit errors (observed: `{type:"provider.quota", message:"Rate + * limit exceeded. Please try again later."}`). An immediate continue retries + * straight into the ban, so these stand down on a gate ladder instead of the + * normal backoff. Gates, not timers: armed timeouts do not survive a reload. */ +const RATE_LIMIT_RE = /rate limit exceeded|too many requests|\b429\b|quota exceeded|provider\.quota/i +/** Per-attempt cooldowns before a rate-limited session may be retried: 15m, + * 30m, 1h, then 2h out to ~12h of unattended coverage. Past the ladder the + * session stays silent until a genuine user turn. */ +const DEFAULT_RATE_LIMIT_COOLDOWNS_MS = [15, 30, 60, 120, 120, 120, 120, 120].map((m) => m * 60_000) /** * Window after `session.interrupt()` during which any abort-shaped event is @@ -1197,6 +1220,11 @@ export default define({ const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS const logFile = opts.logFile ?? process.env.AUTO_RESUME_LOG_FILE ?? DEFAULT_LOG_FILE + const rateLimitCooldownsMs = + Array.isArray(opts.rateLimitCooldownsMs) && opts.rateLimitCooldownsMs.length > 0 && + opts.rateLimitCooldownsMs.every((n) => typeof n === "number" && n > 0) + ? (opts.rateLimitCooldownsMs as number[]) + : DEFAULT_RATE_LIMIT_COOLDOWNS_MS // How long a parent may sit busy after its last subagent went idle before // the orphan watch acts. v1 default, honoured for the first time here. const subagentWaitMs = opts.subagentWaitMs ?? DEFAULT_SUBAGENT_WAIT_MS @@ -1310,6 +1338,7 @@ export default define({ "thinkingToolRecoveryPrompt", "doneWithoutWorkPrompt", "logFile", + "rateLimitCooldownsMs", ]) const unknownOptions = Object.keys(opts).filter((key) => !RECOGNISED_OPTIONS.has(key)) if (unknownOptions.length > 0) { @@ -1428,6 +1457,9 @@ export default define({ lastSubagentCheckAt: 0, pendingTools: 0, toolsInFlightAtIdle: 0, + rateLimitedAt: 0, + rateLimitAttempts: 0, + awaitingQuotaRetry: false, unknownToolErrors: new Map(), unknownToolSuggestionSent: false, checkedToolPartIDs: new Set(), @@ -1570,6 +1602,43 @@ export default define({ return ABORT_ERROR_TYPE_RE.test(errType.trim()) || ABORT_ERROR_MSG_RE.test(errMsg) } + function isRateLimitError(errType: string, errMsg: string): boolean { + return RATE_LIMIT_RE.test(`${errType} ${errMsg}`) + } + + /** + * Stand down on quota/rate-limit failures instead of retrying into the + * ban. Gates, not timers: the decision is re-evaluated on each failure + * and each watchdog tick against `rateLimitedAt`, so nothing is lost + * to a reload. Served cooldowns flow into the normal `recover()` path + * (budgets, loop guard and inject guards all apply); past the ladder + * the session stays silent until a genuine user turn. + */ + function handleRateLimitFailure(sid: string, w: SessionWatch, errType: string, errMsg: string): void { + const ladder = rateLimitCooldownsMs + const n = w.rateLimitAttempts + markIdle(sid) + if (n >= ladder.length) { + w.awaitingQuotaRetry = false + log("warn", `${short(sid)} rate-limit budget exhausted (${ladder.length} attempts) — standing down until user turn`) + return + } + const now = Date.now() + if (w.rateLimitedAt > 0 && now - w.rateLimitedAt >= ladder[n]) { + w.rateLimitAttempts = n + 1 + w.rateLimitedAt = now + w.awaitingQuotaRetry = false + w.pendingRecoveryArmed = true + log("info", `${short(sid)} rate-limit cooldown served — retrying (attempt ${n + 1}/${ladder.length})`) + void recover(sid, "rate-limit cooldown served") + return + } + if (w.rateLimitedAt === 0) w.rateLimitedAt = now + w.awaitingQuotaRetry = true + const waitMs = Math.max(0, ladder[n] - (now - w.rateLimitedAt)) + log("warn", `${short(sid)} rate limited (${errType || "error"}) — standing down, next attempt in ${Math.ceil(waitMs / 1000)}s (${n + 1}/${ladder.length})`) + } + /** Remaining plugin-initiated interrupts allowed in the current window. */ function interruptBudget(w: SessionWatch): number { if (w.interruptWindowStart === 0 || Date.now() - w.interruptWindowStart > INTERRUPT_WINDOW_MS) { @@ -3237,6 +3306,27 @@ export default define({ await recover(sid, `no activity for ${Math.ceil(silence / 1000)}s`) } } + // Rate-limit gate: sessions stood down on quota get one attempt per + // served cooldown, evaluated here so no armed timer can be lost to + // a reload. Guards mirror the failure handler; `recover()` applies + // its own budget, loop and inject guards on top. + for (const [rsid, rw] of sessions) { + if (!rw.awaitingQuotaRetry) continue + const rn = rw.rateLimitAttempts + if (rn >= rateLimitCooldownsMs.length) { + rw.awaitingQuotaRetry = false + continue + } + if (rw.userCancelled || rw.gaveUp || rw.compacting || rw.permissionPending) continue + if (selfAbortActive(rw)) continue + if (Date.now() - rw.rateLimitedAt < rateLimitCooldownsMs[rn]) continue + rw.awaitingQuotaRetry = false + rw.rateLimitAttempts = rn + 1 + rw.rateLimitedAt = Date.now() + rw.pendingRecoveryArmed = true + log("info", `${short(rsid)} rate-limit cooldown served — retrying (attempt ${rn + 1}/${rateLimitCooldownsMs.length})`) + void recover(rsid, "rate-limit cooldown served") + } cleanupIdleSessions() } @@ -3563,10 +3653,18 @@ export default define({ const w = ensureWatch(sid) if (w.selfRecovery) { w.selfRecovery = false - } else if (w.oocLocked) { - w.oocLocked = false - w.oocLockReason = null - log("info", `${short(sid)} genuine execution — clearing OOC lock`) + } else { + if (w.oocLocked) { + w.oocLocked = false + w.oocLockReason = null + log("info", `${short(sid)} genuine execution — clearing OOC lock`) + } + // A genuine turn is activity, and activity means quota is + // available again: the rate-limit ladder starts over. Our + // own recovery turns (selfRecovery) do not reset it. + w.rateLimitedAt = 0 + w.rateLimitAttempts = 0 + w.awaitingQuotaRetry = false } markBusy(sid) return @@ -3927,6 +4025,12 @@ export default define({ w.pendingRecoveryArmed = false return } + // Quota/rate-limit failures stand down on the gate ladder — an + // immediate continue retries straight into the ban. + if (isRateLimitError(errType, errMsg)) { + handleRateLimitFailure(sid, w, errType, errMsg) + return + } maybeLockOoc(sid, errMsg) markIdle(sid) w.pendingRecoveryArmed = true // our delayed recovery must survive this idle transition From 0fef903791e50217cb91cb9be1385e55f9195cab Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:55:02 -0400 Subject: [PATCH 38/57] v2: own-prompt re-arm exclusion survives null-body projections via inject recency Ported from 9cd4471 (auto-resume/v2-todo-message-log): applies cleanly, no deltas. 19 unknown-tool tests pass. --- src/v2/index.ts | 20 ++++++++++++++++++-- src/v2/index.unknown-tool.test.ts | 18 ++++++++++++++++++ 2 files changed, 36 insertions(+), 2 deletions(-) diff --git a/src/v2/index.ts b/src/v2/index.ts index d96260c..2603515 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -548,6 +548,12 @@ const INTERRUPT_WINDOW_MS = 10 * 60_000 * injection per interval. */ const DEFAULT_INJECT_INTERVAL_MS = 15_000 +/** A user message landing within this of our last inject is treated as our + * own prompt for budget re-arm purposes (see noteInboundUserMessage). Same + * box, same clock; our prompt lands in about a second. A genuine user message + * coincidentally inside the window costs one request's worth of stale + * budgets — no loop — versus the recount loop this prevents. */ +const OWN_PROMPT_TIME_MS = 30_000 const MAX_IDLE_SESSIONS = 50 const IDLE_CLEANUP_MS = 10 * 60_000 @@ -2590,8 +2596,18 @@ export default define({ // them clears the budgets just spent and the same errors refire on // the next idle (ses_ef81e8561ffeXyx3jAzKl5lltv, 2026-10-04 — four // identical unknown-tool prompts, each "2x"). Tracking above stays - // truthful; only the clearing is skipped. - if (typeof latest.text === "string" && latest.text.length > 0 && w.ownPromptTexts.includes(latest.text)) { + // truthful; only the clearing is skipped. Two signals: exact text + // match, and recency to our last inject — the live projection can + // carry user messages with no readable body (content null), which + // defeats text matching, but our prompt always lands within seconds + // of the inject on the same box clock. + const ownByText = typeof latest.text === "string" && latest.text.length > 0 && w.ownPromptTexts.includes(latest.text) + const ownByTime = + typeof latest.at === "number" && + w.lastInjectAt > 0 && + latest.at >= w.lastInjectAt && + latest.at - w.lastInjectAt < OWN_PROMPT_TIME_MS + if (ownByText || ownByTime) { dbg(`${short(sid)} newest user message is our own prompt — not re-arming budgets`) return } diff --git a/src/v2/index.unknown-tool.test.ts b/src/v2/index.unknown-tool.test.ts index 8b67164..aa216cb 100644 --- a/src/v2/index.unknown-tool.test.ts +++ b/src/v2/index.unknown-tool.test.ts @@ -290,6 +290,24 @@ describe("v2: naming a replacement for a tool that does not exist", () => { await teardown(h) }) + test("own prompt with an unreadable body still does not re-arm", async () => { + // The live projection can carry user messages with no readable text + // (observed: content null), which defeats text matching. Recency to + // our own inject is the backstop: our prompt always lands within + // seconds of it. + const tools = [{ id: "read" }, { id: "shell" }, { id: "glob" }] + const h = await setup({ + history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "bash")), assistantWith(toolPart("call_2", "bash"))], + tools, + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) + h.history.push({ type: "user", id: "msg_own_bare", time: { created: Date.now() }, content: null }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) + await teardown(h) + }) + test("a v1 tool name maps to its v2 rename instead of (none)", async () => { // "bash" is edit-distance 4 from "shell" against a threshold of 2, so // the fuzzy matcher can never bridge it. A static alias tried first can. From 05838f76d990e27b124b0ee374820063d817c454 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:55:19 -0400 Subject: [PATCH 39/57] =?UTF-8?q?v2:=20re-arm=20keeps=20examined=20parts?= =?UTF-8?q?=20=E2=80=94=20stale=20errors=20never=20re-suggest?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from 7dca80d (auto-resume/v2-todo-message-log): applies cleanly, no deltas. 20 unknown-tool tests pass. --- docs/known-issues-v2.md | 11 +++++------ src/v2/index.ts | 6 +++++- src/v2/index.unknown-tool.test.ts | 30 ++++++++++++++++++++++-------- 3 files changed, 32 insertions(+), 15 deletions(-) diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index db7df25..329cd4b 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -336,12 +336,11 @@ TOOL_STATE_* note in the source.) Two behaviours are inherited from v1 rather than fixed here, and the second is worth a decision: -- **A re-armed turn re-names the oldest typo.** The per-request reset clears the - "already examined" set, so the walk restarts from the top of the history and - the oldest name still above threshold wins — a turn that introduced a *new* - invented name is told about the *previous* one. Suppressing a name once it has - been suggested is a behaviour change rather than a port, so it is documented - instead of made. +- **A re-armed turn names the newest typo.** The per-request reset clears + counts and the latch but keeps the "already examined" set: already-reported + errors are never recounted (2026-10-04 changed this — recounting made every + genuine user message re-suggest the same stale errors). A turn that + introduces a *new* invented name is told about that one. - **No registry, no check.** With no tool list every name looks invented, so the check is skipped rather than accusing the model of tools that plainly exist. Same for an empty registry and for a registry that throws, which is logged and diff --git a/src/v2/index.ts b/src/v2/index.ts index 2603515..b4a228d 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -2628,7 +2628,11 @@ export default define({ } w.unknownToolErrors.clear() w.unknownToolSuggestionSent = false - w.checkedToolPartIDs.clear() + // Deliberately NOT clearing checkedToolPartIDs: already-reported + // errors must never be recounted. Clearing it made every genuine + // user message re-suggest the same stale errors (2026-10-04: one + // bash suggestion per message, long after the model moved to shell). + // New error parts still count fresh toward the threshold. } /** Registered tool names, cached briefly — the registry only changes when a diff --git a/src/v2/index.unknown-tool.test.ts b/src/v2/index.unknown-tool.test.ts index aa216cb..27985cc 100644 --- a/src/v2/index.unknown-tool.test.ts +++ b/src/v2/index.unknown-tool.test.ts @@ -244,13 +244,10 @@ describe("v2: naming a replacement for a tool that does not exist", () => { // a fresh tool list in front of it, so the previous round's typos say nothing // about this one. // - // The second suggestion names the *old* typo, not the new one, and that is - // v1's behaviour reproduced rather than a bug in the port: the re-arm clears - // the "already examined" set, so the walk restarts from the top of the - // history and the oldest name still above threshold wins. Recorded in - // known-issues-v2.md as a quirk worth a decision, not silently changed — - // suppressing a name once it has been suggested is a behaviour change, and - // the project's bar is parity. + // The re-arm resets counts and the latch, but NOT the examined-parts set: + // recounting already-suggested errors is what nagged a live session with + // the same suggestion on every user message (2026-10-04). New errors still + // earn a suggestion — and it names the NEW name, not the old one. const h = await setup({ history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "globb")), assistantWith(toolPart("call_2", "globb"))], }) @@ -261,7 +258,24 @@ describe("v2: naming a replacement for a tool that does not exist", () => { h.history.push(assistantWith(toolPart("call_4", "wriet"))) await goIdle(h) expect(suggestions(h)).toHaveLength(2) - expect(suggestions(h)[1].text).toContain('"globb"') + expect(suggestions(h)[1].text).toContain('"wriet"') + await teardown(h) + }) + + test("a genuine new request does not re-suggest already-reported errors", async () => { + // The 3:40 PM incident: every real user message re-armed, cleared the + // examined set, and recounted the SAME stale bash errors — one suggestion + // per user message while the model had long moved on. + const h = await setup({ + history: [userMessage("go", OLD), assistantWith(toolPart("call_1", "bash")), assistantWith(toolPart("call_2", "bash"))], + tools: [{ id: "read" }, { id: "shell" }, { id: "glob" }], + }) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) + h.history.push(userMessage("stop ignoring; report repeats", OLD + 2_000, "msg_u_second")) + await goIdle(h) + await goIdle(h) + expect(suggestions(h)).toHaveLength(1) await teardown(h) }) From c36c56041947e0e46b8755a2122da244d6e21705 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 12:56:06 -0400 Subject: [PATCH 40/57] v2: skip busy-stall recovery while a parent waits on live subagents MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from a69376a (auto-resume/v2-todo-message-log). Deliberate delta vs the original: the rich stall-text builder (buildStallContinueText) does not exist on this branch, so recover() keeps its plain continue text — only the parent-wait guard (detection + inject time) is ported. 3 parent-wait tests pass. --- src/v2/index.parent-wait.test.ts | 184 +++++++++++++++++++++++++++++++ src/v2/index.ts | 38 +++++++ 2 files changed, 222 insertions(+) create mode 100644 src/v2/index.parent-wait.test.ts diff --git a/src/v2/index.parent-wait.test.ts b/src/v2/index.parent-wait.test.ts new file mode 100644 index 0000000..1f8ad92 --- /dev/null +++ b/src/v2/index.parent-wait.test.ts @@ -0,0 +1,184 @@ +import { describe, test, expect } from "bun:test" +import { rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_parent" +const CHILD = "ses_child" + +let counter = 1000 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, sid: string, data: Record = {}) => ({ type, data: { sessionID: sid, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 300, + toolTextCheckDelayMs: 0, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + subagentWaitMs: 40, + debug: false, +} + +const userMessage = (text: string, at: number, id = `msg_u_${at}`) => ({ + type: "user", + id, + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantAt = (at: number) => ({ + type: "assistant", + id: "msg_a", + time: { created: at }, + content: [{ type: "text", text: "working on it" }], +}) + +type History = Record + +async function setup(opts: { history?: History; active?: string[] }): Promise { + const injected: Array<{ sid?: string; text?: string }> = [] + const interrupted: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-parentwait-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const history: History = opts.history ?? {} + const active = new Set(opts.active ?? []) + const rows = [{ id: SID }, { id: CHILD, parentID: SID }] + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async (a: any) => history[a?.sessionID ?? SID] ?? [], + active: async () => Object.fromEntries([...active].map((s) => [s, {}])), + interrupt: async (a: any) => (interrupted.push(a?.sessionID), {}), + synthetic: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + }, + client: { + session: { + get: async ({ path }: any) => ({ data: { id: path?.id } }), + list: async (a: any) => { + if (!a?.parentID) return { data: rows } + return { data: rows.filter((r) => (r as any).parentID === a.parentID) } + }, + }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + + const cleanup = await (plugin as any).setup(ctx) + await wait(60) + return { injected, interrupted, history, active, cleanup, logFile, streamRef: stream } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const push = (h: any, type: string, sid: string, data: Record = {}) => h.streamRef.push(ev(type, sid, data)) + +async function makeBusy(h: any, sid: string) { + push(h, "session.execution.started", sid) + await wait(10) + push(h, "session.step.started", sid) + await wait(10) +} + +const parentInjects = (h: any) => h.injected.filter((i: any) => i.sid === SID) + +describe("v2: busy-stall skips a parent waiting on a live subagent", () => { + test("no stall recovery while a live child works, even with no task-tool event", async () => { + // ses_ef73a206 shape: parent blocked on a subagent, silent past + // chunkTimeoutMs, never emitted session.tool.called for the dispatch + // (lastWasTaskTool false, pendingTools 0). The parentID link + recent + // child activity is the only evidence — and it must be enough. + const now = Date.now() + const h = await setup({ + active: [SID, CHILD], + history: { [SID]: [userMessage("go", now - 60_000)], [CHILD]: [assistantAt(now - 5_000)] }, + }) + await makeBusy(h, SID) + await wait(900) + expect(parentInjects(h)).toHaveLength(0) + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("no stall recovery while tools are held and children are unresolved", async () => { + // tool.called fired (pendingTools 1, lastWasTaskTool true) but the + // child is absent from the server active set — the old + // lastWasTaskTool+others guard misses, the parentID link must catch it. + const now = Date.now() + const h = await setup({ + active: [SID], + history: { [SID]: [userMessage("go", now - 60_000)], [CHILD]: [assistantAt(now - 5_000)] }, + }) + await makeBusy(h, SID) + push(h, "session.tool.called", SID, { tool: "subagent", input: { description: "research" } }) + await wait(900) + expect(parentInjects(h)).toHaveLength(0) + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: a truly stalled parent with no live children still recovers", async () => { + // No children at all on the parentID link — ordinary stall, must fire. + const now = Date.now() + const h = await setup({ + active: [SID], + history: { [SID]: [userMessage("go", now - 60_000)] }, + }) + // NOTE: default rows still link CHILD to SID; override by pushing the + // child stale AND out of the active set is covered by the next test. + // Here we keep the link but the child is long dead. + h.history[CHILD] = [assistantAt(now - 10 * 60_000)] + await makeBusy(h, SID) + await wait(900) + expect(parentInjects(h).length).toBeGreaterThan(0) + await teardown(h) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index b4a228d..8ad9bd8 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -2104,6 +2104,13 @@ export default define({ dbg(`${short(sid)} still busy and live at inject time (${Math.round((Date.now() - w.lastActivityAt) / 1000)}s since last event) — not interrupting`) return } + // Re-check the parent-wait guard: a child may have gone live + // between detection and this delayed inject. Interrupting a + // parent that is waiting on a subagent kills real work. + if (await parentWaitingOnSubagents(sid, new Set(await getActiveSessions()))) { + dbg(`${short(sid)} waiting on subagents at inject time — not interrupting`) + return + } const ok = await injectOnce(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "stalled — retrying") w.pendingRecoveryArmed = false if (ok) { @@ -3238,6 +3245,29 @@ export default define({ tryAbortAndResume(sid, w) } + /** + * Parent-wait guard for the busy-stall path. A parent blocked on a live + * subagent is quiet on its own sessionID by construction — the child's + * events belong to the child — so "no activity for Ns" must not by itself + * trigger recovery. Deliberately independent of the `lastWasTaskTool` + * heuristic below: native dispatches may never emit `session.tool.called`, + * and the server active set may lag the parentID link. + * + * Waits when the verdict confirms a live child, or when the parent still + * holds tools in flight against children that are neither confirmed live + * nor proven dead. A crashed verdict, or tools held with no children on + * the link at all (wedged bare tool), falls through to normal recovery. + */ + async function parentWaitingOnSubagents(sid: string, activeIDs: Set): Promise { + const verdict = await subagentVerdict(sid, activeIDs) + if (verdict.status === "busy") return true + if (verdict.status !== "idle") return false + const w = ensureWatch(sid) + if (w.pendingTools <= 0) return false + const children = await listSubagentIds(sid) + return children.length > 0 + } + async function checkActiveSessions() { const now = Date.now() @@ -3315,6 +3345,14 @@ export default define({ continue } } + // A parent waiting on a live subagent is working, not stalled. The + // heuristic above misses native dispatches (no tool.called event) + // and active-set lag; the parentID link does not (see + // parentWaitingOnSubagents, ses_ef73a206 2026-10-04). + if (await parentWaitingOnSubagents(sid, activeSet)) { + dbg(`${short(sid)} waiting on subagents — skipping stall detection`) + continue + } // busyStallStrategy picks how a stall is answered. "abort" interrupts // the wedged step before continuing, because a stream that went // silent mid-generation will not pick the prompt up on its own. From a38d6815a75bcc9dc9fac67fb3bba2ae126801d4 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:07:36 -0400 Subject: [PATCH 41/57] test: drop the module-scope assertion superseded by the globalThis registry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from b2fb63f (auto-resume/v2-todo-message-log) — required companion to the d74debe+d756146 ports: the structural test asserted the module-scope singleton that d756146 replaced with the globalThis registry (covered by index.singleton-registry.test.ts instead). --- src/v2/index.duplicate-instance.test.ts | 11 ----------- 1 file changed, 11 deletions(-) diff --git a/src/v2/index.duplicate-instance.test.ts b/src/v2/index.duplicate-instance.test.ts index 24bab80..e7c8c42 100644 --- a/src/v2/index.duplicate-instance.test.ts +++ b/src/v2/index.duplicate-instance.test.ts @@ -5,7 +5,6 @@ import { join } from "node:path" import { tmpdir } from "node:os" import plugin from "./index" -const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) const SID = "ses_dupinst" let counter = 0 @@ -142,14 +141,4 @@ describe("v2: N live setups must not become N identical injections", () => { // One live instance, therefore one counter, therefore one injection. expect(stalls).toBe(1) }, 25_000) - - test("structural: a module-scope singleton guards setup() against stacking", () => { - const singletonAt = SOURCE.indexOf("let activeInstance:") - expect(singletonAt).toBeGreaterThan(-1) - const setupAt = SOURCE.indexOf("setup: async (ctx: AutoResumePluginInput) => {") - // Declared BEFORE setup opens, and consulted inside it: that ordering is the fix. - expect(singletonAt).toBeLessThan(setupAt) - expect(SOURCE).toContain("activeInstance.dispose()") - expect(SOURCE).toContain("activeInstance = { dispose }") - }) }) \ No newline at end of file From 67437d8ecdb13dcb4f824912ea332b0f9b9828a7 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:19:24 -0400 Subject: [PATCH 42/57] v2: visible stall continue + rich prod text + duplicate anti-repeat Ported from 8073ba1 (auto-resume/v2-todo-message-log). Deliberate deltas: kept HEAD's newer injectOnceLocked signature (required params from the mutex port) and HEAD's ownPromptTexts tracking; src/v2/index.visible- continue.test.ts restored with its 7 tests (all pass). --- README.md | 2 + src/v2/index.ts | 81 ++++++++++- src/v2/index.visible-continue.test.ts | 195 ++++++++++++++++++++++++++ 3 files changed, 277 insertions(+), 1 deletion(-) create mode 100644 src/v2/index.visible-continue.test.ts diff --git a/README.md b/README.md index 3659f42..f6497f5 100644 --- a/README.md +++ b/README.md @@ -521,6 +521,8 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `activeUserWindowMs` | `300000` | Inbound-user-message recency window during which idle nudges stand down (user likely composing) | | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | +| `visibleContinue` | `false`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Checks that decide nothing stay silent either way | +| `richContinuePrompt` | `true`, v2 only | Stall continue names the stall reason, attempt count, and remaining todos instead of bare `"continue"`. A custom `continuePrompt` always wins verbatim | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | | `rateLimitCooldownsMs` | v2 only | Per-attempt cooldowns (ms) before a rate-limited session may be retried, evaluated as gates on each failure and watchdog tick (no armed timers to lose on reload). Default `[900000, 1800000, 3600000, 7200000 ×5]` ≈ 12h coverage; past the ladder the session stays silent until a genuine user turn. Quota hits never consume the normal retry budget | diff --git a/src/v2/index.ts b/src/v2/index.ts index 8ad9bd8..8637077 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -289,6 +289,20 @@ export interface AutoResumeOptions { /** Minimum gap between recovery injections for one session. */ injectIntervalMs?: number continuePrompt?: string + /** + * When true, a stall continue is sent via `ctx.session.prompt()` — a real + * user message visible in session history (cattleprod-style) — instead of + * the hidden `ctx.session.synthetic({resume:true})`. Default false: same + * channel as before. Checks that decide nothing stay silent either way; + * only the action-driving continue becomes visible. + */ + visibleContinue?: boolean + /** + * When true (default), a stall continue names the stall reason, attempt + * count, and remaining todos instead of sending bare "continue". A custom + * `continuePrompt` always wins verbatim. Same channel and guards either way. + */ + richContinuePrompt?: boolean toolTextRecoveryPrompt?: string doneWithoutWorkPrompt?: string actionIntentPrompt?: string @@ -1225,6 +1239,8 @@ export default define({ const debug = opts.debug ?? DEFAULT_DEBUG const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS + const visibleContinue = opts.visibleContinue ?? false + const richContinuePrompt = opts.richContinuePrompt ?? true const logFile = opts.logFile ?? process.env.AUTO_RESUME_LOG_FILE ?? DEFAULT_LOG_FILE const rateLimitCooldownsMs = Array.isArray(opts.rateLimitCooldownsMs) && opts.rateLimitCooldownsMs.length > 0 && @@ -1322,6 +1338,8 @@ export default define({ "debug", "activeUserWindowMs", "injectIntervalMs", + "visibleContinue", + "richContinuePrompt", "continuePrompt", "toolTextRecoveryPrompt", "actionIntentPrompt", @@ -1853,6 +1871,25 @@ export default define({ dbg(`${short(sid)} injection debounced — ${Math.round(since / 1000)}s since last (min ${injectIntervalMs / 1000}s)`) return false } + // Exact-duplicate anti-repeat (cattleprod-style soft skip, opt-in per + // caller): an immediate re-fire with identical text and zero model + // progress since the last successful prod means the previous nudge + // changed nothing — skip it and let the sliding-window counter / + // escalation own the loop. No message fetch needed: progress is + // assistant-text growth plus tool completions. Scoped to the stall + // path by default: the targeted nudges below own per-kind budgets + // with re-arm semantics this must not second-guess. + if ( + checkDuplicate && + !allowDuringSelfAbort && + w.lastProdText !== "" && + text === w.lastProdText && + w.pendingTools <= 0 && + w.lastAssistantText === w.prodAssistantSnapshot + ) { + dbg(`${short(sid)} duplicate continue suppressed — identical text, no progress since last prod`) + return false + } w.lastInjectAt = Date.now() // Cross-instance backstop for the in-memory skip above: stacked // watchdogs cannot see each other's counters, but they share the @@ -1959,6 +1996,18 @@ export default define({ */ async function notifyAndPrompt(sid: string, text: string, notification: string, resume = true): Promise { ensureWatch(sid).selfRecovery = true + if (visibleContinue) { + // Cattleprod-style: a real user message in session history, so the + // intervention — and any loop — is self-evident in the transcript. + // Falls back to synthetic only if the prompt call throws. + try { + await ctx.session.prompt({ sessionID: sid, text }) + return true + } catch (err) { + const msg = err instanceof Error ? err.message : String(err) + log("warn", `${short(sid)} visible prompt failed: ${msg}`) + } + } try { await ctx.session.synthetic({ sessionID: sid, @@ -2043,6 +2092,35 @@ export default define({ log("warn", `${short(sid)} OOC error latched — recovery locked out until a genuine turn: ${errMsg.slice(0, 120)}`) } + /** + * Build the stall-continue text. A custom `continuePrompt` always wins + * verbatim; otherwise, when `richContinuePrompt` is on (default), the + * text names the stall reason, attempt count, and remaining todos + * (cattleprod-style) instead of bare "continue". Same channel and + * guards either way — only the content changes. + */ + async function buildStallContinueText(sid: string, reason: string, attempt: number): Promise { + if (opts.continuePrompt !== undefined && opts.continuePrompt !== "") return opts.continuePrompt + if (!richContinuePrompt) return CONTINUE_PROMPT + let suffix = "Todo list unavailable." + try { + const todos = await readTodos(sid) + const open = todos.filter((t) => t.status !== "completed" && t.status !== "cancelled") + if (open.length > 0) { + const lines = open + .slice(0, 5) + .map((t) => `• [${t.status}] ${String(t.content).slice(0, 120)}`) + if (open.length > 5) lines.push(`• …and ${open.length - 5} more`) + suffix = `Remaining todos:\n${lines.join("\n")}` + } else { + suffix = "No open todos recorded." + } + } catch { + // Best effort: a todo-read failure must never block recovery. + } + return `continue — stalled (${reason}; attempt ${attempt}/${maxRetries}).\n${suffix}`.slice(0, 2000) + } + /** * Core recovery ladder for a stuck/failed session. * plain continue with backoff -> more attempts -> abort+resume escalation. @@ -2111,7 +2189,8 @@ export default define({ dbg(`${short(sid)} waiting on subagents at inject time — not interrupting`) return } - const ok = await injectOnce(sid, opts.continuePrompt ?? CONTINUE_PROMPT, "stalled — retrying") + const stallText = await buildStallContinueText(sid, reason, attempt) + const ok = await injectOnce(sid, stallText, "stalled — retrying", false, true) w.pendingRecoveryArmed = false if (ok) { w.lastRetryAt = Date.now() diff --git a/src/v2/index.visible-continue.test.ts b/src/v2/index.visible-continue.test.ts new file mode 100644 index 0000000..120aa1a --- /dev/null +++ b/src/v2/index.visible-continue.test.ts @@ -0,0 +1,195 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +/** + * v2: visible stall continue (cattleprod-style) + rich prod text + + * exact-duplicate anti-repeat. + * + * Every group carries a control. The harness drives a genuinely stalled busy + * session (started, then silent) so the watchdog fires `recover()`, and tags + * each delivery by channel (`synthetic` vs `prompt`) — the distinction the + * feature exists to make. + */ + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_vis" +let counter = 0 + +const OPEN = [ + { content: "Write the migration guide", status: "pending", priority: "high" }, + { content: "Delete the temp fixtures", status: "in_progress", priority: "low" }, +] + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, data: Record = {}) => ({ type, data: { sessionID: SID, ...data } }) + +type Harness = { + injected: Array<{ kind: string; text?: string }> + logs: string[] +} + +/** A turn that starts and then goes silent: a stall candidate. */ +const busyStallEvents = [ev("session.execution.started")] + +/** Timings small enough that the watchdog fires inside the harness wait. */ +const FAST = { + chunkTimeoutMs: 50, + gracePeriodMs: 0, + checkIntervalMs: 20, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + loopMaxContinues: 99, + injectIntervalMs: 0, +} + +async function replay( + events: any[], + opts: Record = {}, + todos: unknown[] | undefined = undefined, + extraEvents: Array<{ at: number; event: any }> = [], + waitMs = 600, +): Promise { + const injected: Harness["injected"] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-vis-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + event: stream, + options: { ...FAST, ...opts, logFile, debug: true }, + session: { + active: async () => ({}), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ kind: "synthetic", text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ kind: "prompt", text: a?.text }), {}), + }, + client: { session: { get: async () => ({ data: {} }) } }, + storage: { + get: async () => ({ todos: todos ?? [], updatedAt: Date.now() }), + set: async () => {}, + remove: async () => {}, + }, + } + + const cleanup = await (plugin as any).setup(ctx) + const started = Date.now() + let extraIdx = 0 + for (const e of events) { + stream.push(e) + await wait(10) + } + // Interleave extra events at wall-clock offsets while the watchdog works. + while (Date.now() - started < waitMs) { + while (extraIdx < extraEvents.length && Date.now() - started >= extraEvents[extraIdx].at) { + stream.push(extraEvents[extraIdx].event) + extraIdx++ + } + await wait(10) + } + for (; extraIdx < extraEvents.length; extraIdx++) stream.push(extraEvents[extraIdx].event) + await wait(50) + ;(cleanup as (() => void) | undefined)?.() + const logs = existsSync(logFile) ? readFileSync(logFile, "utf8").split("\n") : [] + rmSync(logFile, { force: true }) + return { injected, logs } +} + +describe("v2: stall-continue channel", () => { + test("CONTROL: default channel is hidden synthetic", async () => { + const { injected } = await replay(busyStallEvents, { maxRetries: 1 }) + expect(injected.length).toBeGreaterThan(0) + expect(injected.every((i) => i.kind === "synthetic")).toBe(true) + }) + + test("visibleContinue:true sends a real prompt message instead", async () => { + const { injected } = await replay(busyStallEvents, { maxRetries: 1, visibleContinue: true }) + expect(injected.length).toBeGreaterThan(0) + expect(injected.every((i) => i.kind === "prompt")).toBe(true) + }) +}) + +describe("v2: rich stall-continue text", () => { + test("CONTROL: default names the stall and the open todos", async () => { + const { injected } = await replay(busyStallEvents, { maxRetries: 1 }, OPEN) + expect(injected.length).toBeGreaterThan(0) + expect(injected[0].text).toContain("continue — stalled") + expect(injected[0].text).toContain("no activity for") + expect(injected[0].text).toContain("Write the migration guide") + expect(injected[0].text).toContain("Delete the temp fixtures") + }) + + test("richContinuePrompt:false restores bare 'continue'", async () => { + const { injected } = await replay(busyStallEvents, { maxRetries: 1, richContinuePrompt: false }, OPEN) + expect(injected.length).toBeGreaterThan(0) + expect(injected[0].text).toBe("continue") + }) + + test("a custom continuePrompt always wins verbatim", async () => { + const { injected } = await replay(busyStallEvents, { maxRetries: 1, continuePrompt: "go" }, OPEN) + expect(injected.length).toBeGreaterThan(0) + expect(injected[0].text).toBe("go") + }) +}) + +describe("v2: exact-duplicate anti-repeat", () => { + test("an identical re-fire with zero progress is suppressed", async () => { + // Verbatim custom text, so every attempt builds the same string; no + // assistant text and no tool work ever lands, so attempts 2+ are dupes. + const { injected, logs } = await replay(busyStallEvents, { maxRetries: 5, continuePrompt: "go" }, [], [], 900) + expect(injected).toHaveLength(1) + expect(injected[0].text).toBe("go") + expect(logs.some((l) => l.includes("duplicate continue suppressed"))).toBe(true) + }) + + test("new model output re-arms it", async () => { + // Fresh assistant text between windows is progress: the same prod text + // must go out again instead of being swallowed. Headroom past the + // progress event matters: attempts burn ~1/70ms, so maxRetries 5 would + // be spent before the delta lands and the session gives up. + const { injected } = await replay( + busyStallEvents, + { maxRetries: 10, continuePrompt: "go" }, + [], + [{ at: 250, event: ev("session.text.delta", { messageID: "msg_a1", delta: "still working" }) }], + 1200, + ) + expect(injected.length).toBeGreaterThanOrEqual(2) + }) +}) From 90113a7a1f8fe13ca232839fd5d6c4bd6debf44d Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:21:56 -0400 Subject: [PATCH 43/57] v2: AUTO_RESUME_VISIBLE_CONTINUE env fallback + startup line reports channel Ported from a2d3897 (auto-resume/v2-todo-message-log): ready line keeps both tags (visibleContinue + mod= instance tag from the globalThis port). 8 visible-continue tests pass. --- README.md | 2 +- src/v2/index.ts | 13 +++++++++---- src/v2/index.visible-continue.test.ts | 11 +++++++++++ 3 files changed, 21 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index f6497f5..0a4b268 100644 --- a/README.md +++ b/README.md @@ -521,7 +521,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `activeUserWindowMs` | `300000` | Inbound-user-message recency window during which idle nudges stand down (user likely composing) | | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | -| `visibleContinue` | `false`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Checks that decide nothing stay silent either way | +| `visibleContinue` | `false`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Env fallback `AUTO_RESUME_VISIBLE_CONTINUE=1` (set it in the top-level `env` block; the `plugin` array has no options slot). Checks that decide nothing stay silent either way | | `richContinuePrompt` | `true`, v2 only | Stall continue names the stall reason, attempt count, and remaining todos instead of bare `"continue"`. A custom `continuePrompt` always wins verbatim | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | | `rateLimitCooldownsMs` | v2 only | Per-attempt cooldowns (ms) before a rate-limited session may be retried, evaluated as gates on each failure and watchdog tick (no armed timers to lose on reload). Default `[900000, 1800000, 3600000, 7200000 ×5]` ≈ 12h coverage; past the ladder the session stays silent until a genuine user turn. Quota hits never consume the normal retry budget | diff --git a/src/v2/index.ts b/src/v2/index.ts index 8637077..605fcb7 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -293,8 +293,11 @@ export interface AutoResumeOptions { * When true, a stall continue is sent via `ctx.session.prompt()` — a real * user message visible in session history (cattleprod-style) — instead of * the hidden `ctx.session.synthetic({resume:true})`. Default false: same - * channel as before. Checks that decide nothing stay silent either way; - * only the action-driving continue becomes visible. + * channel as before. Env fallback `AUTO_RESUME_VISIBLE_CONTINUE=1` (the + * `opencode.jsonc` `plugin` array has no options slot on this box, so env + * via the top-level `env` block is the live knob). Checks that decide + * nothing stay silent either way; only the action-driving continue + * becomes visible. */ visibleContinue?: boolean /** @@ -1239,7 +1242,9 @@ export default define({ const debug = opts.debug ?? DEFAULT_DEBUG const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS - const visibleContinue = opts.visibleContinue ?? false + const visibleContinue = + opts.visibleContinue ?? + /^(1|true|yes)$/i.test(process.env.AUTO_RESUME_VISIBLE_CONTINUE ?? "") const richContinuePrompt = opts.richContinuePrompt ?? true const logFile = opts.logFile ?? process.env.AUTO_RESUME_LOG_FILE ?? DEFAULT_LOG_FILE const rateLimitCooldownsMs = @@ -4211,7 +4216,7 @@ export default define({ log( "info", - `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s warmup=${warmupMs}ms stall=${busyStallStrategy} mod=${MODULE_INSTANCE}` + + `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s warmup=${warmupMs}ms stall=${busyStallStrategy} visibleContinue=${visibleContinue} mod=${MODULE_INSTANCE}` + (gatedInUse.length > 0 ? ` accepted-but-inert=${gatedInUse.join(",")}` : ""), ) diff --git a/src/v2/index.visible-continue.test.ts b/src/v2/index.visible-continue.test.ts index 120aa1a..e2e31cc 100644 --- a/src/v2/index.visible-continue.test.ts +++ b/src/v2/index.visible-continue.test.ts @@ -143,6 +143,17 @@ describe("v2: stall-continue channel", () => { expect(injected.length).toBeGreaterThan(0) expect(injected.every((i) => i.kind === "prompt")).toBe(true) }) + + test("AUTO_RESUME_VISIBLE_CONTINUE=1 enables the visible channel", async () => { + process.env.AUTO_RESUME_VISIBLE_CONTINUE = "1" + try { + const { injected } = await replay(busyStallEvents, { maxRetries: 1 }) + expect(injected.length).toBeGreaterThan(0) + expect(injected.every((i) => i.kind === "prompt")).toBe(true) + } finally { + delete process.env.AUTO_RESUME_VISIBLE_CONTINUE + } + }) }) describe("v2: rich stall-continue text", () => { From 3df233936a4cfb4ca5d8c5d73ddfec2221afca1b Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:22:12 -0400 Subject: [PATCH 44/57] v2: visibleContinue defaults true (config options dropped by parser) Ported from bf4e8c8 (auto-resume/v2-todo-message-log): applies cleanly, no deltas. 10 visible-continue tests pass. --- README.md | 2 +- src/v2/index.ts | 19 ++++++++++++------- src/v2/index.visible-continue.test.ts | 19 ++++++++++++++++++- 3 files changed, 31 insertions(+), 9 deletions(-) diff --git a/README.md b/README.md index 0a4b268..6698267 100644 --- a/README.md +++ b/README.md @@ -521,7 +521,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `activeUserWindowMs` | `300000` | Inbound-user-message recency window during which idle nudges stand down (user likely composing) | | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | -| `visibleContinue` | `false`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Env fallback `AUTO_RESUME_VISIBLE_CONTINUE=1` (set it in the top-level `env` block; the `plugin` array has no options slot). Checks that decide nothing stay silent either way | +| `visibleContinue` | `true`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Set `false` for the old channel. Explicit option wins, else `AUTO_RESUME_VISIBLE_CONTINUE` (`0`/`false` = hidden) wins over the default. (Config `plugin`-entry options are dropped by this box's parser, so code default is the live knob.) | | `richContinuePrompt` | `true`, v2 only | Stall continue names the stall reason, attempt count, and remaining todos instead of bare `"continue"`. A custom `continuePrompt` always wins verbatim | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | | `rateLimitCooldownsMs` | v2 only | Per-attempt cooldowns (ms) before a rate-limited session may be retried, evaluated as gates on each failure and watchdog tick (no armed timers to lose on reload). Default `[900000, 1800000, 3600000, 7200000 ×5]` ≈ 12h coverage; past the ladder the session stays silent until a genuine user turn. Quota hits never consume the normal retry budget | diff --git a/src/v2/index.ts b/src/v2/index.ts index 605fcb7..746b467 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -292,12 +292,14 @@ export interface AutoResumeOptions { /** * When true, a stall continue is sent via `ctx.session.prompt()` — a real * user message visible in session history (cattleprod-style) — instead of - * the hidden `ctx.session.synthetic({resume:true})`. Default false: same - * channel as before. Env fallback `AUTO_RESUME_VISIBLE_CONTINUE=1` (the - * `opencode.jsonc` `plugin` array has no options slot on this box, so env - * via the top-level `env` block is the live knob). Checks that decide - * nothing stay silent either way; only the action-driving continue - * becomes visible. + * the hidden `ctx.session.synthetic({resume:true})`. Default true: the + * intervention — and any loop — is self-evident in the transcript, which + * is the point. Set false for the old hidden channel. Explicit option + * wins; else `AUTO_RESUME_VISIBLE_CONTINUE` (`1`/`true`/`yes` = visible, + * anything else set = hidden) wins over the default. (Options set via + * the config `plugin` entry are dropped by this box's parser and the + * top-level `env` block never reaches the plugin process, so code + * default is the only live knob.) */ visibleContinue?: boolean /** @@ -1242,9 +1244,12 @@ export default define({ const debug = opts.debug ?? DEFAULT_DEBUG const activeUserWindowMs = opts.activeUserWindowMs ?? DEFAULT_ACTIVE_USER_WINDOW_MS const injectIntervalMs = opts.injectIntervalMs ?? DEFAULT_INJECT_INTERVAL_MS + const envVisible = process.env.AUTO_RESUME_VISIBLE_CONTINUE const visibleContinue = opts.visibleContinue ?? - /^(1|true|yes)$/i.test(process.env.AUTO_RESUME_VISIBLE_CONTINUE ?? "") + (envVisible === undefined || envVisible === "" + ? true + : /^(1|true|yes)$/i.test(envVisible)) const richContinuePrompt = opts.richContinuePrompt ?? true const logFile = opts.logFile ?? process.env.AUTO_RESUME_LOG_FILE ?? DEFAULT_LOG_FILE const rateLimitCooldownsMs = diff --git a/src/v2/index.visible-continue.test.ts b/src/v2/index.visible-continue.test.ts index e2e31cc..047c139 100644 --- a/src/v2/index.visible-continue.test.ts +++ b/src/v2/index.visible-continue.test.ts @@ -132,9 +132,15 @@ async function replay( } describe("v2: stall-continue channel", () => { - test("CONTROL: default channel is hidden synthetic", async () => { + test("CONTROL: default channel is the visible prompt", async () => { const { injected } = await replay(busyStallEvents, { maxRetries: 1 }) expect(injected.length).toBeGreaterThan(0) + expect(injected.every((i) => i.kind === "prompt")).toBe(true) + }) + + test("visibleContinue:false restores hidden synthetic", async () => { + const { injected } = await replay(busyStallEvents, { maxRetries: 1, visibleContinue: false }) + expect(injected.length).toBeGreaterThan(0) expect(injected.every((i) => i.kind === "synthetic")).toBe(true) }) @@ -154,6 +160,17 @@ describe("v2: stall-continue channel", () => { delete process.env.AUTO_RESUME_VISIBLE_CONTINUE } }) + + test("AUTO_RESUME_VISIBLE_CONTINUE=0 opts back out to hidden", async () => { + process.env.AUTO_RESUME_VISIBLE_CONTINUE = "0" + try { + const { injected } = await replay(busyStallEvents, { maxRetries: 1 }) + expect(injected.length).toBeGreaterThan(0) + expect(injected.every((i) => i.kind === "synthetic")).toBe(true) + } finally { + delete process.env.AUTO_RESUME_VISIBLE_CONTINUE + } + }) }) describe("v2: rich stall-continue text", () => { From 0497165fa58e2b0450ee5d00eb7b026adbfd5507 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:22:24 -0400 Subject: [PATCH 45/57] test: stand-down CONTROL follows visible-default channel Ported from 6e20464 (auto-resume/v2-todo-message-log): applies cleanly, no deltas. 13 stand-down tests pass. --- src/v2/index.stand-down.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/v2/index.stand-down.test.ts b/src/v2/index.stand-down.test.ts index 0052d55..92bc1c6 100644 --- a/src/v2/index.stand-down.test.ts +++ b/src/v2/index.stand-down.test.ts @@ -138,7 +138,7 @@ describe("v2: stand down while the user owes us an answer", () => { events: turnEvents(READY_TEXT), }) expect(injected.length).toBeGreaterThan(0) - expect(injected[0].kind).toBe("synthetic") + expect(injected[0].kind).toBe("prompt") }) // The v2 message projection never writes "pending" — it writes "running". From a9f2425554e2807a5f407d23f9e56b30d2776701 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:22:41 -0400 Subject: [PATCH 46/57] test: keep test sessions out of the LIVE auto-resume log Ported from af87ab6 (auto-resume/v2-todo-message-log): applies cleanly, no deltas. inject-lock + singleton-registry suites now write a private tmpdir logFile instead of the production default. --- src/v2/index.inject-lock.test.ts | 19 +++++++++++++++++-- src/v2/index.singleton-registry.test.ts | 19 +++++++++++++++++-- 2 files changed, 34 insertions(+), 4 deletions(-) diff --git a/src/v2/index.inject-lock.test.ts b/src/v2/index.inject-lock.test.ts index fb03fa1..5cc4ea3 100644 --- a/src/v2/index.inject-lock.test.ts +++ b/src/v2/index.inject-lock.test.ts @@ -1,11 +1,24 @@ import { describe, test, expect } from "bun:test" -import { readFileSync } from "node:fs" +import { readFileSync, rmSync } from "node:fs" import { join } from "node:path" +import { tmpdir } from "node:os" import plugin from "./index" const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) const SID = "ses_lock" +let counter = 0 + +/** + * A private log file. Without `logFile` the plugin writes to its DEFAULT path, + * which is the LIVE server's log — test sessions then show up interleaved with + * real ones and the live log stops being usable as forensics. + */ +function privateLog(tag: string): string { + const f = join(tmpdir(), `auto-resume-${tag}-${process.pid}-${counter++}.log`) + rmSync(f, { force: true }) + return f +} function makeEventStream() { const queue: any[] = [] @@ -63,9 +76,10 @@ const FAST = { async function run(slowMs: number) { const injected: any[] = [] const stream = makeEventStream() + const logFile = privateLog("lock") const ctx: any = { event: stream, - options: { ...FAST, maxRetries: 1, debug: true }, + options: { ...FAST, maxRetries: 1, debug: true, logFile }, session: { active: async () => ({}), interrupt: async () => ({}), @@ -86,6 +100,7 @@ async function run(slowMs: number) { stream.push(ev("session.execution.started")) await wait(800) ;(cleanup as (() => void) | undefined)?.() + rmSync(logFile, { force: true }) return injected } diff --git a/src/v2/index.singleton-registry.test.ts b/src/v2/index.singleton-registry.test.ts index 0c34545..95e30d7 100644 --- a/src/v2/index.singleton-registry.test.ts +++ b/src/v2/index.singleton-registry.test.ts @@ -1,11 +1,24 @@ import { describe, test, expect } from "bun:test" -import { readFileSync } from "node:fs" +import { readFileSync, rmSync } from "node:fs" import { join } from "node:path" +import { tmpdir } from "node:os" import plugin from "./index" const SOURCE = readFileSync(join(import.meta.dir, "index.ts"), "utf8") const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) const SID = "ses_registry" +let counter = 0 + +/** + * A private log file. Without `logFile` the plugin writes to its DEFAULT path, + * which is the LIVE server's log — test sessions then show up interleaved with + * real ones and the live log stops being usable as forensics. + */ +function privateLog(tag: string): string { + const f = join(tmpdir(), `auto-resume-${tag}-${process.pid}-${counter++}.log`) + rmSync(f, { force: true }) + return f +} function makeEventStream() { const queue: any[] = [] @@ -60,6 +73,7 @@ type Boot = { injected: any[]; registryLive: unknown; registryKey: string } async function boot(instances: number): Promise { const injected: any[] = [] const streams = Array.from({ length: instances }, () => makeEventStream()) + const logFile = privateLog("registry") let next = 0 const ctx: any = { event: { @@ -69,7 +83,7 @@ async function boot(instances: number): Promise { return s }, }, - options: { ...FAST, maxRetries: 1 }, + options: { ...FAST, maxRetries: 1, logFile }, session: { active: async () => ({}), interrupt: async () => ({}), @@ -91,6 +105,7 @@ async function boot(instances: number): Promise { await wait(700) const duringStall = (globalThis as any)[key]?.live for (const c of cleanups) (c as (() => void) | undefined)?.() + rmSync(logFile, { force: true }) return { injected, registryLive: duringStall ?? afterSetup, registryKey: key } } From 963a9fd477e5df028e6935d9fd7599faf8660f25 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:23:35 -0400 Subject: [PATCH 47/57] docs: dead-stream tool-call exception, plugins-vs-plugin delivery Ported from 2f99b9b (auto-resume/v2-todo-message-log) minus the configProbe knob row (probe scope excluded): dead-stream paragraph + silentDeadStreamMinTokens row + visibleContinue plugins-key note + known-issues dead-stream amendment. --- README.md | 6 +++--- docs/known-issues-v2.md | 6 ++++++ 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 6698267..6c9bb2e 100644 --- a/README.md +++ b/README.md @@ -176,7 +176,7 @@ _Motivated by:_ ### Silent dead-stream recovery -The model stream can die after emitting only reasoning — no text part, no tool call — finalizing with `finish: "unknown"`. OpenCode treats the message as completed and the session goes idle, so no error or stall path triggers. On idle, if the **newest** assistant message has a finish reason, zero text parts, and at least `silentDeadStreamMinTokens` output tokens, the plugin sends a recovery prompt. Only the newest assistant message is evaluated — a delivered text answer means normal completion, and older tool-call steps are never misread as dead streams. Recovery is also skipped if the session has gone busy/retry again before the prompt is sent (race guard). +The model stream can die after emitting only reasoning — no text part, no tool call — finalizing with a finish reason such as `finish: "unknown"` (any finish reason qualifies, including `"stop"`). OpenCode treats the message as completed and the session goes idle, so no error or stall path triggers. On idle, if the **newest** assistant message has a finish reason, zero text parts, **zero tool-call parts**, and at least `silentDeadStreamMinTokens` output tokens, the plugin sends a recovery prompt. A finished turn that issued tool calls is working, not silent — thinking models routinely end tool-working turns with no chatter text (2026-10-04: this fired twice into an active session before the exception existed). As a second guard, recovery is refused while tool calls are still in flight, read from a snapshot taken at the idle transition (`markIdle` zeroes the live counter, and idle fires while tools run). Only the newest assistant message is evaluated — a delivered text answer means normal completion, and older tool-call steps are never misread as dead streams. Recovery is also skipped if the session has gone busy/retry again before the prompt is sent (race guard). ### Context saturation → magic-context wrapup @@ -515,13 +515,13 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `doneWithoutDetailsPrompt` | `DONE_WITHOUT_DETAILS_PROMPT` | Override the done-claim-with-no-todos report prompt | | `doneClaimPatterns` | `DONE_CLAIM_PATTERNS` | Array of regex strings overriding the default done-claim detection patterns (case-insensitive, multiline). Invalid regexes are skipped. Empty array falls back to defaults. | | `readyToContinuePatterns` | `READY_TO_CONTINUE_PATTERNS` | Array of regex strings overriding the default ready-to-continue detection patterns (case-insensitive). Invalid regexes are skipped. Empty array falls back to defaults. | -| `silentDeadStreamMinTokens` | `200` | Min output tokens to treat a textless `finish:"unknown"` message as a dead stream | +| `silentDeadStreamMinTokens` | `200` | Min output tokens to treat a textless, tool-less finished message as a dead stream. Finished turns carrying tool calls are working, not silent, and tools in flight veto the inject regardless | | `busyStallStrategy` | `"continue"` | Busy-stall response: `"continue"`, `"abort"` (abort-first), or `"off"` (disabled) | | `contextSaturationThreshold` | `0.85` | Ratio of used/usable context that routes a saturated session to reclamation: a parent to magic-context `ctx-wrapup` (only when magic-context is installed), a subagent to native compaction (only when `subagentNativeCompactionEnabled`) | | `activeUserWindowMs` | `300000` | Inbound-user-message recency window during which idle nudges stand down (user likely composing) | | `subagentNativeCompactionEnabled` | `false` | Opt-in native `session.summarize()` for saturated subagent sessions (no magic-context detection required) | | `injectIntervalMs` | v2 only | Minimum gap between recovery injections for one session. No v1 equivalent | -| `visibleContinue` | `true`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Set `false` for the old channel. Explicit option wins, else `AUTO_RESUME_VISIBLE_CONTINUE` (`0`/`false` = hidden) wins over the default. (Config `plugin`-entry options are dropped by this box's parser, so code default is the live knob.) | +| `visibleContinue` | `true`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Set `false` for the old channel. Explicit option wins, else `AUTO_RESUME_VISIBLE_CONTINUE` (`0`/`false` = hidden) wins over the default. (Options must sit under the plural `plugins` key — entries under singular `plugin` load but their options are never delivered.) | | `richContinuePrompt` | `true`, v2 only | Stall continue names the stall reason, attempt count, and remaining todos instead of bare `"continue"`. A custom `continuePrompt` always wins verbatim | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | | `rateLimitCooldownsMs` | v2 only | Per-attempt cooldowns (ms) before a rate-limited session may be retried, evaluated as gates on each failure and watchdog tick (no armed timers to lose on reload). Default `[900000, 1800000, 3600000, 7200000 ×5]` ≈ 12h coverage; past the ladder the session stays silent until a genuine user turn. Quota hits never consume the normal retry budget | diff --git a/docs/known-issues-v2.md b/docs/known-issues-v2.md index 329cd4b..0caec7a 100644 --- a/docs/known-issues-v2.md +++ b/docs/known-issues-v2.md @@ -433,6 +433,12 @@ failure fires, and the stall watchdog — which only looks at *busy* sessions never sees it. v1's rule is unchanged on v2: walk back to the newest assistant message that **has** a finish reason; if it carried no text and generated at least `silentDeadStreamMinTokens` output tokens, the stream died mid-response. +Two 2026-10-04 amendments narrow "no text": a finished message carrying +**tool-call parts** is a working turn, not a dead stream (thinking models end +tool-working turns with no chatter), and recovery is refused while tool calls +are still in flight — read from a snapshot taken at the idle transition, +because `markIdle` zeroes the live `pendingTools` counter and idle fires +while tools run. The walk skips messages with no finish reason on purpose. An intermediate tool-call step has none, and stopping at it would report a dead stream for every From e89bd0bcfe2b6a7d793cbda8a67f45e9af88a8c1 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:25:12 -0400 Subject: [PATCH 48/57] docs: v1 backport research (rich text, dup guard, startup log, silent stub) Ported from b0b5100 (auto-resume/v2-todo-message-log): restores docs/v2/backport-to-v1.md verbatim ( HEAD had deleted it). Line-numbered research mapping v2 features onto v1's src/index.ts. --- docs/v2/backport-to-v1.md | 223 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 223 insertions(+) create mode 100644 docs/v2/backport-to-v1.md diff --git a/docs/v2/backport-to-v1.md b/docs/v2/backport-to-v1.md new file mode 100644 index 0000000..6adc6e9 --- /dev/null +++ b/docs/v2/backport-to-v1.md @@ -0,0 +1,223 @@ +# v2 → v1 auto-resume backport research + +Branch: `auto-resume/v2-todo-message-log` (HEAD a2d3897; v2 changes in 8073ba1 + a2d3897) +Subject: `src/index.ts` (v1 plugin, `@opencode-ai/plugin` v1 hooks style, ~3000+ lines) +Reference: `src/v2/index.ts` (v2 plugin) + +## v1 mechanism today + +### Continue channel (confirmed visible) + +- `sendContinuePrompt(sid, text, w)` at **src/index.ts:837** is the single choke point for every continue. It calls `ctx.client.session.prompt({path:{id}, body:{parts:[{type:'text',text}],agent,model}})` at **914-921** (retry path **933-936**). This is a **visible user message** — confirmed per user directive. +- Agent/model are inferred from the last user message at **869-907** (via `getSessionMessages`, **1067-1080**). +- Guards inside `sendContinuePrompt`: `w.continuing && !watchdogRetryGuard` skip (**838-841**), `userCancelled || completionSignaled` (**842**), `gaveUp` latch (**848-851**), `oocLocked` latch (**852-856**). +- `recordContinue(sid)` at **927** (success) and **938** (retry success); `w.lastRetryAt` set at **928/939**; `finally` block resets `continuing`/`todoCheckAttempts`/clears toolTextTimer at **945-950**. +- Deferred watchdog `setTimeout` at **954-1025**: retry backoff → `sendContinuePrompt(sid, continuePrompt, w)` at **976** → escalate `tryAbortAndResume` → `gaveUp` latch at **1004**. + +### Continue prompt content today (BARE) + +- `continuePrompt` option at **530-531**: `const continuePrompt: string = (options?.continuePrompt as string) ?? "continue"` — **bare text today**. +- `actionIntentPrompt` at **532-533** defaults to `continuePrompt`. +- Other fixed prompts: `toolTextRecoveryPrompt` (**534**), `thinkingToolRecoveryPrompt` (**536**), `doneWithoutWorkPrompt` (**538**), `doneWithoutDetailsPrompt` (**540**), `TOOL_LOOP_RECOVERY_PROMPT` (**130**), `SUBAGENT_RECOVERY_PROMPT` (**1240**). +- `buildOpenTodosReminder(todos)` at **482-491** builds a richer "You have N unfinished tasks..." text (already used by idle-open-todos paths). + +### All `sendContinuePrompt` call sites (all send bare `continuePrompt` or a fixed prompt) + +| Line | Context | Prompt sent | +|---|---|---| +| 976 | deferred watchdog retry (WP-05) | `continuePrompt` (bare) | +| 1138 | unknown-tool suggestion | custom `prompt` (tool name) | +| 1903 | tool-text recovery best-candidate | `bestCandidate.prompt` | +| 1963 | abort+continue | `continuePrompt` (bare) | +| 2016 | `tryResume` generic send | `prompt ?? continuePrompt` | +| 2244 | pending-recovery (WP-04) | `continuePrompt` (bare) | +| 2650 | action-intent (idle) | `actionIntentPrompt` | +| 2732 | action-intent (busy) | `actionIntentPrompt` | +| 2953 | task_complete blocked | `blockMsg` (fixed) | +| 3061 | tool-loop recovery | `TOOL_LOOP_RECOVERY_PROMPT` (fixed) | + +### `tryResume` (the main stall path) + +- Signature at **1984**: `tryResume(sid, w, reason, prompt?)`. +- Backoff check at **1991-1992** (`backoffMs(w.resumeAttempts, ...)` at **447-453**). +- Hallucination-loop check at **1994-2009** (inflight tools → skip; else `tryAbortAndResume`). +- `w.resumeAttempts++` at **2011**; log at **2013**. +- Send at **2016**: `sendContinuePrompt(sid, prompt ?? continuePrompt, w)` — **bare unless caller passes explicit prompt**. +- Callers of `tryResume`: + - **2203**: stream-stall (`"Stream stall"`, no explicit prompt → bare) + - **2302**: periodic idle-open-todos (explicit `reminder` from `buildOpenTodosReminder`) + - **2486-2492**: streaming-failure on idle (explicit `continuePrompt`) + - **2533-2538**: silent-dead-stream (explicit `continuePrompt`) + - **2607-2612**: idle-open-todos celebration false-positive (explicit `reminder`) + - **2613-2616**: idle-open-todos normal (explicit `reminder`) + +### Loop guards (existing) + +- `recordContinue(sid)` at **573-577** (pushes timestamp into `w.continueTimestamps`). +- `isHallucinationLoop(sid)` at **583-588**: `continueTimestamps.length >= loopMaxContinues` (default **3**, option at **508-509**) within `loopWindowMs` (default **600s**, option at **510-511**) → true. +- `backoffMs` at **447-453**: exponential, capped. +- `w.gaveUp` latch: set at **1004** (watchdog exhausted), cleared at **1435**/`resetSessionFlags` and **1463**/`resetBusyFlags` (on new user message). +- `w.oocLocked` at **2876-2880** (OOC error → permanent halt). +- `minActivityGapMs` check at **1882-1886** (skip if session was active < gap ago). +- `todoNudgeAttempts` budget at **1863-1865** and **2602-2616**. +- `maxRetries` (option at **500-501**) and `maxRecoveryRetries` (option at **514-515**). +- No env-var reads in v1 today (all options via 2nd factory arg at **493**). +- `debug` option at **522-523**. + +### Assistant text availability + +- v1 has **no rolling `lastAssistantText`** accumulation (v2 uses `textParts` map maintained in `appendText` at src/v2/index.ts:1669-1677). +- v1 can get assistant text via `getSessionMessages(sid)` at **1067-1080** (SDK `session.messages`). The pattern is shown in `lastAssistantEndsWithCelebration` at **1155-1178** (iterate last assistant msg, concatenate text parts). +- Todo state: `w.todos`, `buildOpenTodosReminder` at **482-491**, `fetchSessionTodos` at **1206-1236**. + +### Startup log + +- Lines **2982-2985**: `if (!initialised) { initialised = true; log("info", \`opencode-auto-resume ready. timeout=${chunkTimeoutMs}ms, orphan=${subagentWaitMs}ms, loop=${loopMaxContinues}x/${loopWindowMs/1000}s\`) }` — no channel/rich/dup info today. + +--- + +## V2 changes under review (reference: src/v2/index.ts) + +1. **`visibleContinue`** + `AUTO_RESUME_VISIBLE_CONTINUE` env (v2 opts **1115-1117**; option doc **282-288**): stall continue via `ctx.session.prompt` (visible) instead of synthetic, fallback to synthetic on throw. **NOT needed for v1** (v1 already visible). +2. **`richContinuePrompt`** default `true` (v2 **1118**; doc **293-296**): `buildStallContinueText` at v2 **1799-1821** — custom `continuePrompt` wins verbatim; else `continue — stalled (${reason}; attempt ${attempt}/${maxRetries}).\n${todoSuffix}` where `todoSuffix` lists up to 5 open todos `• [status] content` or "No open todos recorded." / "Todo list unavailable."; `slice(0,2000)`. Called from `recover()` at v2 **1882**. +3. **Exact-duplicate anti-repeat**: v2 watch fields `lastProdText`/`prodAssistantSnapshot` (**159-162**) + rolling `lastAssistantText` (**172**, maintained **1669-1677**); `injectOnce` gate at v2 **1603-1609**: `checkDuplicate && !selfAbort && lastProdText!=="" && text===lastProdText && pendingTools<=0 && lastAssistantText===prodAssistantSnapshot` → soft skip (dbg only). On success updates at v2 **1616-1617**. `recover()` passes `checkDuplicate=true`. +4. **Startup log line** at v2 **3786-3791**: includes `visibleContinue=${visibleContinue}` (+ accepted-but-inert list). + +V2 option-reading style: `const opts = (ctx.options ?? {}) as AutoResumeOptions` at v2 **1100** (v1 uses 2nd factory arg — parity = `options` arg). + +--- + +## Per-item verdicts (ranked by value) + +### Item 2: `richContinuePrompt` — **YES, backport** (highest value) + +**Why:** v1 sends bare `"continue"` from many stall sites. A rich prompt (reason + attempt + open todos) dramatically improves model compliance on stall recovery, matching v2 behavior. + +**Insertion point (v1):** +- New function `buildStallContinueText(sid, reason, attempt)` — insert near `buildOpenTodosReminder` at **~492** (after line 491). Reuses `fetchSessionTodos` (**1206-1236**) for todo suffix. Logic mirrors v2 **1799-1821**. +- New option at **~531** (after `continuePrompt`): `const richContinuePrompt: boolean = (options?.richContinuePrompt as boolean) ?? true`. +- **Call sites to change** (stall path only, scope per user direction): + - `tryResume` at **2016**: `sendContinuePrompt(sid, prompt ?? await buildStallContinueText(sid, reason, w.resumeAttempts), w)` — but only when `prompt` is undefined (caller didn't pass explicit). Note: `tryResume` is async so `await` is fine. + - Line **1963** (abort+continue): `sendContinuePrompt(sid, await buildStallContinueText(sid, "abort+resume", w.resumeAttempts), w)` + - Line **2244** (pending-recovery): `sendContinuePrompt(sid, await buildStallContinueText(sid, w.pendingRecoveryReason ?? "recovery", w.recoveryAttempts), w)` + - Line **976** (watchdog retry): `sendContinuePrompt(sid, await buildStallContinueText(sid, "recovery", w.recoveryAttempts), w)` + - Lines **2486-2492** (streaming-failure) and **2533-2538** (silent-dead-stream): these pass `continuePrompt` explicitly to `tryResume`; change to pass `undefined` so `tryResume` builds the rich text. + - **Do NOT change**: 1138 (unknown-tool, custom prompt), 1903 (tool-text recovery, bestCandidate), 2650/2732 (actionIntent), 2953 (task_complete block), 3061 (TOOL_LOOP_RECOVERY_PROMPT) — these have their own per-kind prompts/budgets. +- **Conflicts:** None with existing guards. `tryResume` is already async. The `prompt` param at **1984** is optional; callers passing explicit prompts (idle-todos at 2302/2607/2611) are unaffected. `minActivityGapMs` check at **1882-1886** is in the tool-text path, not `tryResume`. + +**Test file:** `src/index.continue.test.ts` (548 lines, best fit — already tests continue prompt sending). Add tests: rich prompt includes reason+attempt+todo suffix; custom `continuePrompt` wins verbatim; `richContinuePrompt: false` reverts to bare. + +--- + +### Item 3: Exact-duplicate anti-repeat — **YES, backport** (medium value, prevents redundant messages) + +**Why:** Prevents sending the same continue text when the model made zero progress (no new assistant text, no tools in flight). Soft skip only; loop window counter stays the backstop. + +**Insertion point (v1):** +- New `SessionWatch` fields at end of interface (line 68, after `checkedToolPartIDs` at line 67): `lastProdText: string`, `prodAssistantSnapshot: string`, `lastAssistantText: string`. +- New function `snapshotAssistantText(sid)` — mirrors `lastAssistantEndsWithCelebration` pattern at **1155-1178**: `getSessionMessages(sid)`, find last assistant msg, concatenate text parts, return trimmed string. Insert near **~1180**. +- Gate inside `sendContinuePrompt` at **~855** (after `oocLocked` check, before `w.continuing = true`): + ``` + if (w.lastProdText !== "" && text === w.lastProdText && w.pendingTools <= 0) { + const snap = await snapshotAssistantText(sid) + if (snap === w.prodAssistantSnapshot) { + dbg(`${short(sid)} duplicate continue suppressed`) + return + } + } + ``` +- On successful send (after **927** `recordContinue`): update `w.lastProdText = text; w.prodAssistantSnapshot = await snapshotAssistantText(sid)`. +- **Cost note:** `snapshotAssistantText` calls `getSessionMessages` (SDK round-trip). `sendContinuePrompt` already calls `getSessionMessages` at **869** for agent/model inference — **reuse that fetch**: extract the last assistant text from the same `msgs` array (pattern at **1155-1178**) instead of a second call. +- **Conflicts:** None. `w.pendingTools` already exists on `SessionWatch` (**line 56**). The gate is a soft skip (no state change beyond dbg), so it cannot interfere with `gaveUp`/`oocLocked`/backoff logic. The `w.continuing` guard at **838** is checked before the dup gate. + +**Test file:** `src/index.watchdog.test.ts` (549 lines, tests the stall/retry path). Add: duplicate continue suppressed when no progress; NOT suppressed when assistant text grew; NOT suppressed when tools in flight. + +--- + +### Item 4: Startup log line — **YES, trivial** (low effort, nice-to-have) + +**Why:** Reports resolved channel + rich/dup settings, matching v2 diagnostic parity. + +**Insertion point (v1):** Line **2984** — append to the existing `log("info", ...)` string: +- ` visible=${true}` (v1 always visible; informational) +- ` rich=${richContinuePrompt}` +- ` dupGuard=${true}` (always on once backported) + +No new option needed — just report the resolved values. + +**Test file:** `src/index.plugin.test.ts` (655 lines, "Plugin Lifecycle" section at line 28). Add: startup log contains `rich=true` / `rich=false` depending on option. + +--- + +### New item: `silentContinue` opt-out — **PARTIAL / DEFER** (parity stub only) + +**User direction:** v1 continues are already visible; v1 stays visible-by-default with an opt-out to silence. + +**Problem:** v1 has **only one channel** — `client.session.prompt` = visible. There is no hidden/synthetic channel in v1 (v2 has `ctx.session.hook("message", ...)` for silent). A true "silent continue" in v1 would require either (a) a no-op + log-only (defeats the purpose) or (b) a new hidden mechanism that doesn't exist in the v1 plugin API. + +**Recommendation:** Accept the option name for config parity (`silentContinue?: boolean`, default `false`), but **implement as a no-op with a warn log** if set to `true`: `log("warn", "silentContinue=true requested but v1 has no hidden channel; continues remain visible")`. This documents intent, prevents config-porting confusion, and is honest about the limitation. **Assumption:** user accepts this limitation; if true silence is needed, it requires a v1 API extension beyond this backport. + +**Insertion point:** Option at **~531** (alongside `richContinuePrompt`); warn-log at **~2984** (startup) or in `sendContinuePrompt` at **~840** (once, via `initialised`-style flag). + +**Test file:** `src/index.plugin.test.ts` — assert warn log emitted when `silentContinue: true`. + +--- + +## Suggested option names + defaults (v1 parity) + +| Option | Default | v2 equivalent | Notes | +|---|---|---|---| +| `richContinuePrompt` | `true` | v2 `richContinuePrompt` (1118) | Stall-path rich text; custom `continuePrompt` wins verbatim | +| `silentContinue` | `false` | v2 `visibleContinue` (inverted) | No-op + warn in v1 (no hidden channel); parity stub | +| `continuePrompt` | `"continue"` | v2 `continuePrompt` | Existing, unchanged; wins verbatim over rich text | + +No env-var reads in v1 (all via `options` 2nd factory arg at line 493). V1 has no `AUTO_RESUME_*` env today. If env parity is desired, add `process.env.AUTO_RESUME_RICH_CONTINUE_PROMPT` fallback — but this deviates from v1's current style (zero env reads); **recommend options-only for v1**. + +--- + +## Test plan + +All tests use `bun:test` (existing pattern in `src/index.*.test.ts`). + +### Item 2 (rich prompt) → extend `src/index.continue.test.ts` +1. **Rich text includes reason + attempt + todo suffix:** trigger a stall (`tryResume`), assert `promptCalls[0].body` contains `"stalled"` + `"attempt 1/3"` + open todo content. +2. **Custom `continuePrompt` wins verbatim:** set `options.continuePrompt = "keep going"`, assert prompt body === `"keep going"` (not rich). +3. **`richContinuePrompt: false` reverts to bare:** set `options.richContinuePrompt = false`, assert prompt body === `"continue"`. +4. **Todo suffix capped at 5 + "…and N more":** 7 open todos → suffix has 5 bullets + `• …and 2 more`. +5. **No open todos → "No open todos recorded."** +6. **Todo fetch fails → "Todo list unavailable."** (mock `fetchSessionTodos` to throw). +7. **Rich text sliced to 2000 chars.** + +### Item 3 (anti-dup) → extend `src/index.watchdog.test.ts` +1. **Duplicate suppressed:** send continue, no assistant-text growth, no tools → second identical continue skipped (no new `promptCalls`). +2. **NOT suppressed when text grew:** assistant text changed since last prod → second continue sent. +3. **NOT suppressed when tools in flight:** `w.pendingTools > 0` → second continue sent. +4. **First send always goes through:** `lastProdText === ""` → no gate. +5. **Gate does not affect `gaveUp`/`oocLocked`/backoff:** those checks still fire independently. + +### Item 4 (startup log) → extend `src/index.plugin.test.ts` +1. **Default:** startup log contains `rich=true`, `dupGuard=true`. +2. **`richContinuePrompt: false`:** startup log contains `rich=false`. +3. **`silentContinue: true`:** startup log contains `silentContinue=true` + warn. + +### New: `silentContinue` → extend `src/index.plugin.test.ts` +1. **`silentContinue: true`:** warn log "no hidden channel" emitted once at startup. +2. **`silentContinue: false` (default):** no warn. + +### Regression +- Run existing `src/index.continue.test.ts`, `src/index.watchdog.test.ts`, `src/index.plugin.test.ts` suites unchanged. +- Run `src/index.pending-recovery.test.ts`, `src/index.silent-dead-stream.test.ts`, `src/index.streaming-failure.test.ts` (stall-path callers affected by item 2). + +--- + +## Summary + +| Item | Verdict | Effort | Value | +|---|---|---|---| +| 1. `visibleContinue` | **NO** (v1 already visible) | — | — | +| 2. `richContinuePrompt` | **YES** | Medium (~80 lines) | High | +| 3. Anti-dup gate | **YES** | Medium (~50 lines) | Medium | +| 4. Startup log | **YES** | Trivial (~5 lines) | Low | +| New: `silentContinue` | **PARTIAL** (no-op stub) | Trivial (~10 lines) | Low (parity) | + +Total estimated effort: ~150 lines of v1 changes + ~150 lines of tests. From 27add46d871847a411c2beafd7e82607900eb51c Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:27:54 -0400 Subject: [PATCH 49/57] docs: refresh v2 suite figures and feature list in migration.md v2 suite is 233 tests across 20 files (was: 188 across 13); feature list now names the ported areas (dup suppression, mutex/singleton, quota ladder, parent-wait, visible channel). Figures from executed runs, 2026-10-05. --- docs/v2/migration.md | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/docs/v2/migration.md b/docs/v2/migration.md index 9aad2e8..f4a30e7 100644 --- a/docs/v2/migration.md +++ b/docs/v2/migration.md @@ -15,8 +15,9 @@ Ports the plugin from the v1 hooks API (`Plugin` factory returning a hooks object) to the v2 promise-plugin API (`Plugin.define({ id, setup })` + `ctx.event.subscribe()`). All detection/recovery features are preserved. Strict-mode typechecked against the real `@opencode/plugin@2.0.5` types, and -covered by 188 tests across 13 files driven through the real event stream — one -file per feature area, 780 passing repo-wide including v1. +covered by 233 tests across 20 files driven through the real event stream — one +file per feature area (figures rechecked 2026-10-05 after the todo-branch fix +ports; repo-wide total in the suite footer below counts v1+v2). ## 1. Config key renamed: `plugin` → `plugins` @@ -218,7 +219,7 @@ Disable by id without touching other plugins: add `"-auto-resume.v2"`. ## Testing - `tsc --strict --noEmit` clean against **`@opencode/plugin@2.0.5`** (stable). -- Runtime suite (bun, 13 files / 188 tests) against the stable API: +- Runtime suite (bun, 20 files / 233 tests, rechecked 2026-10-05) against the stable API: definition shape · `subscribe({ signal })` · abort-on-cleanup · stall watchdog → `synthetic` with visible `description` + `resume` · idle forensics via the `session.context()` fallback · healthy idle turn is a no-op · permission-hold @@ -229,7 +230,9 @@ Disable by id without touching other plugins: add `"-auto-resume.v2"`. silent dead streams, premature stop, context saturation, the todo list, explicit `task_complete`, reasoning-tool recovery, orphan parent recovery, the settle delay, session discovery, the options surface, and the hand-off and stand-down - guards. + guards — plus cross-instance duplicate suppression, the inject mutex and + singleton registry, the quota ladder, parent-wait on live subagents, and the + visible channel with rich stall text. Two tests in the settle-delay file were **vacuous** on the first pass — they passed whether or not the code under test ran. Every test in the v2 port is now From 7cc07149e0fe94b7db52c85ec9d8378c550bc689 Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 13:34:26 -0400 Subject: [PATCH 50/57] =?UTF-8?q?v2:=20inert=20configProbe=20option=20?= =?UTF-8?q?=E2=80=94=20proves=20plugins[].options=20delivery=20without=20t?= =?UTF-8?q?ouching=20a=20live=20knob?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ported from 1064641 (auto-resume/v2-todo-message-log): applies with neighborhood merges only. README row included (was stripped from the earlier quota/docs ports pending this decision). 23 options tests pass. --- README.md | 1 + src/v2/index.ts | 12 +++++++++++- 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 6c9bb2e..cf43af1 100644 --- a/README.md +++ b/README.md @@ -524,6 +524,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `visibleContinue` | `true`, v2 only | Send stall continue via `session.prompt()` — a real user message visible in session history (cattleprod-style) — instead of hidden `session.synthetic({resume:true})`. Set `false` for the old channel. Explicit option wins, else `AUTO_RESUME_VISIBLE_CONTINUE` (`0`/`false` = hidden) wins over the default. (Options must sit under the plural `plugins` key — entries under singular `plugin` load but their options are never delivered.) | | `richContinuePrompt` | `true`, v2 only | Stall continue names the stall reason, attempt count, and remaining todos instead of bare `"continue"`. A custom `continuePrompt` always wins verbatim | | `logFile` | v2 only | Where this build appends its log. v2 removed v1's server log endpoint, so without this the plugin is silent. Defaults to `~/.local/state/opencode-v2/auto-resume.log` | +| `configProbe` | v2 only, inert | Delivery probe: echoed as `probe=` in the `ready` startup line and read nowhere else. Bump the token to prove `plugins[].options` reaches the plugin without touching a live knob | | `rateLimitCooldownsMs` | v2 only | Per-attempt cooldowns (ms) before a rate-limited session may be retried, evaluated as gates on each failure and watchdog tick (no armed timers to lose on reload). Default `[900000, 1800000, 3600000, 7200000 ×5]` ≈ 12h coverage; past the ladder the session stays silent until a genuine user turn. Quota hits never consume the normal retry budget | Message patterns are matched case-insensitively. Error names use exact match. diff --git a/src/v2/index.ts b/src/v2/index.ts index 746b467..185a8ac 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -365,6 +365,12 @@ export interface AutoResumeOptions { * until a genuine user turn. Default spans ~12h for overnight coverage. */ rateLimitCooldownsMs?: number[] + /** + * Inert delivery probe. Never affects behaviour; echoed in the `ready` + * line so a config-options change can be verified without touching a + * live knob. Bump the value to test delivery. + */ + configProbe?: string | number } // --------------------------------------------------------------------------- @@ -1257,6 +1263,9 @@ export default define({ opts.rateLimitCooldownsMs.every((n) => typeof n === "number" && n > 0) ? (opts.rateLimitCooldownsMs as number[]) : DEFAULT_RATE_LIMIT_COOLDOWNS_MS + // Inert probe: proves `plugins[].options` reaches `ctx.options` without + // touching a live knob. Echoed in the `ready` line only. + const configProbe = opts.configProbe ?? null // How long a parent may sit busy after its last subagent went idle before // the orphan watch acts. v1 default, honoured for the first time here. const subagentWaitMs = opts.subagentWaitMs ?? DEFAULT_SUBAGENT_WAIT_MS @@ -1373,6 +1382,7 @@ export default define({ "doneWithoutWorkPrompt", "logFile", "rateLimitCooldownsMs", + "configProbe", ]) const unknownOptions = Object.keys(opts).filter((key) => !RECOGNISED_OPTIONS.has(key)) if (unknownOptions.length > 0) { @@ -4221,7 +4231,7 @@ export default define({ log( "info", - `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s warmup=${warmupMs}ms stall=${busyStallStrategy} visibleContinue=${visibleContinue} mod=${MODULE_INSTANCE}` + + `ready (opencode v2). timeout=${chunkTimeoutMs}ms interval=${checkIntervalMs}ms retries=${maxRetries} loop=${loopMaxContinues}/${loopWindowMs / 1000}s warmup=${warmupMs}ms stall=${busyStallStrategy} visibleContinue=${visibleContinue} probe=${configProbe ?? "-"} mod=${MODULE_INSTANCE}` + (gatedInUse.length > 0 ? ` accepted-but-inert=${gatedInUse.join(",")}` : ""), ) From 3c7ce7fbe7ad14e48b8003ab7b2d5637fc6bcf7d Mon Sep 17 00:00:00 2001 From: famewolf Date: Mon, 5 Oct 2026 15:04:17 -0400 Subject: [PATCH 51/57] v2: busy-stall stands down while a turn waits on the user MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ses_ef72c5f3 2026-10-05: the watchdog injected a rich stall continue into a session parked on a question tool awaiting the user (186s event silence during the Q&A round-trip). The busy-stall path had no user-wait guard — only the idle path stands down. New isWaitingOnUser() covers only user-awaiting tools (question/permission/ask/confirm), so a wedged bare tool still recovers; wired at detection (checkActiveSessions) and inject time (recover). Covered by src/v2/index.user-wait.test.ts (1 repro + 2 controls). --- src/v2/index.ts | 40 +++++++ src/v2/index.user-wait.test.ts | 189 +++++++++++++++++++++++++++++++++ 2 files changed, 229 insertions(+) create mode 100644 src/v2/index.user-wait.test.ts diff --git a/src/v2/index.ts b/src/v2/index.ts index 185a8ac..a8c2094 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -2209,6 +2209,12 @@ export default define({ dbg(`${short(sid)} waiting on subagents at inject time — not interrupting`) return } + // Same re-check for user input: a question asked between + // detection and this delayed inject parks the turn. + if (isWaitingOnUser(await loadMessages(sid))) { + dbg(`${short(sid)} waiting on user input at inject time — not interrupting`) + return + } const stallText = await buildStallContinueText(sid, reason, attempt) const ok = await injectOnce(sid, stallText, "stalled — retrying", false, true) w.pendingRecoveryArmed = false @@ -2617,6 +2623,32 @@ export default define({ return false } + /** + * Narrow user-wait check for the busy-stall path. Unlike + * `hasPendingUserInput` (any unfinished tool means the idle turn + * isn't over), a busy session with a wedged bare tool and zero events + * IS a stall — so only tools whose completion is the user answering + * (`question`, `permission`, `ask`, `confirm`) count. A turn parked + * on one of those is waiting, not stalled (ses_ef72c5f3 2026-10-05: a + * stall continue fired into a session parked on a `question`). + */ + function isWaitingOnUser(messages: unknown[]): boolean { + const newest = messages[messages.length - 1] as { + type?: string + content?: { type?: string; name?: string; state?: { status?: string } }[] + } + if (newest?.type !== "assistant") return false + for (const part of newest.content ?? []) { + const t = part?.type ?? "" + if (!(t === "tool_use" || t === "tool" || t === "tool_call" || t.startsWith("tool"))) continue + if (!AWAITING_USER_TOOLS.has(part?.name ?? "")) continue + // As in `hasPendingUserInput`: "error" means interrupted, not + // answered — only a completed user-awaiting tool ends the wait. + if (part?.state?.status !== TOOL_STATE_COMPLETED) return true + } + return false + } + async function shouldStandDownForUser( sid: string, messages: unknown[], @@ -3452,6 +3484,14 @@ export default define({ dbg(`${short(sid)} waiting on subagents — skipping stall detection`) continue } + // A turn parked on user input (a `question` awaiting an answer, + // an unanswered permission beyond the event flow) is waiting, + // not stalled. Narrower than the idle-path check on purpose: a + // wedged bare tool with zero events is still a stall. + if (isWaitingOnUser(await loadMessages(sid))) { + dbg(`${short(sid)} waiting on user input — skipping stall detection`) + continue + } // busyStallStrategy picks how a stall is answered. "abort" interrupts // the wedged step before continuing, because a stream that went // silent mid-generation will not pick the prompt up on its own. diff --git a/src/v2/index.user-wait.test.ts b/src/v2/index.user-wait.test.ts new file mode 100644 index 0000000..3b519a2 --- /dev/null +++ b/src/v2/index.user-wait.test.ts @@ -0,0 +1,189 @@ +import { describe, test, expect } from "bun:test" +import { rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +/** + * v2: busy-stall stands down while a turn waits on the user. + * + * Repro: ses_ef72c5f3 (2026-10-05) — the watchdog injected + * `continue — stalled (no activity for 186s)` into a session parked on a + * `question` tool awaiting the user. The busy-stall path had no user-wait + * guard (only the idle path stands down via `shouldStandDownForUser`). + * A turn awaiting the user is waiting, not stalled — but a wedged bare + * tool (or an answered question) must still recover. + */ + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_parent" + +let counter = 2000 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, sid: string, data: Record = {}) => ({ type, data: { sessionID: sid, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 300, + toolTextCheckDelayMs: 0, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + subagentWaitMs: 40, + debug: false, +} + +const userMessage = (text: string, at: number) => ({ + type: "user", + id: "msg_u", + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantWithTool = (name: string, status: string, at: number) => ({ + type: "assistant", + id: "msg_a", + time: { created: at }, + content: [{ type: "tool", name, state: { status } }], +}) + +type History = Record + +async function setup(history: History, active: string[]): Promise { + const injected: Array<{ sid?: string; text?: string }> = [] + const interrupted: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-userwait-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const activeSet = new Set(active) + const rows: Array<{ id: string; parentID?: string }> = [{ id: SID }] + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async (a: any) => history[a?.sessionID ?? SID] ?? [], + active: async () => Object.fromEntries([...activeSet].map((s) => [s, {}])), + interrupt: async (a: any) => (interrupted.push(a?.sessionID), {}), + synthetic: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + }, + client: { + session: { + get: async ({ path }: any) => ({ data: { id: path?.id } }), + list: async (a: any) => { + if (!a?.parentID) return { data: rows } + return { data: rows.filter((r) => (r as any).parentID === a.parentID) } + }, + }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + + const cleanup = await (plugin as any).setup(ctx) + await wait(60) + return { injected, interrupted, cleanup, logFile, streamRef: stream } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const push = (h: any, type: string, sid: string, data: Record = {}) => h.streamRef.push(ev(type, sid, data)) + +async function makeBusy(h: any, sid: string) { + push(h, "session.execution.started", sid) + await wait(10) + push(h, "session.step.started", sid) + await wait(10) +} + +const parentInjects = (h: any) => h.injected.filter((i: any) => i.sid === SID) + +describe("v2: busy-stall stands down while a turn waits on the user", () => { + test("no stall recovery while a question tool is still running", async () => { + const now = Date.now() + const h = await setup( + { + [SID]: [userMessage("go", now - 60_000), assistantWithTool("question", "running", now - 5_000)], + }, + [SID], + ) + await makeBusy(h, SID) + await wait(900) + expect(parentInjects(h)).toHaveLength(0) + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: a wedged bare tool still recovers", async () => { + // A non-user tool stuck running with no events is exactly what the + // stall path exists for — the user-wait guard must not swallow it. + const now = Date.now() + const h = await setup( + { + [SID]: [userMessage("go", now - 60_000), assistantWithTool("shell", "running", now - 5_000)], + }, + [SID], + ) + await makeBusy(h, SID) + await wait(900) + expect(parentInjects(h).length).toBeGreaterThan(0) + await teardown(h) + }) + + test("CONTROL: an answered question still recovers", async () => { + // The user answered (tool completed) but the session never moved: + // that IS a stall, not a wait. + const now = Date.now() + const h = await setup( + { + [SID]: [userMessage("go", now - 60_000), assistantWithTool("question", "completed", now - 5_000)], + }, + [SID], + ) + await makeBusy(h, SID) + await wait(900) + expect(parentInjects(h).length).toBeGreaterThan(0) + await teardown(h) + }) +}) From 43246e6d3797bd59172d6160341a1492277f7c9e Mon Sep 17 00:00:00 2001 From: famewolf Date: Tue, 6 Oct 2026 16:16:07 -0400 Subject: [PATCH 52/57] v2: subagent protection without ctx.client + cancel stall timers on dispose MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ses_eed6d1b5 + ses_eeeb3532 (2026-10-06): continue nudges landed in working subagents. Root cause, probed live: prod ctx carries NO client and NO session list/get/active surface, so isSubAgentSession always degraded to false (zero refusals in all of log history) and listSubagentIds always missed. GET /api/session answers with parentID rows under message-read auth, so both now fall back to a short-TTL loopback list. Second defect in the same incident: recover()'s backoff timer was untracked, so a disposed instance still fired (twin identical prods 5ms apart) — the handle is now tracked and cleared on dispose, with a running-check in the timer body. Covered by src/v2/index.child-guard.test.ts (2 repro, both fail-before/pass-after). --- src/v2/index.child-guard.test.ts | 215 +++++++++++++++++++++++++++++++ src/v2/index.ts | 72 ++++++++++- 2 files changed, 286 insertions(+), 1 deletion(-) create mode 100644 src/v2/index.child-guard.test.ts diff --git a/src/v2/index.child-guard.test.ts b/src/v2/index.child-guard.test.ts new file mode 100644 index 0000000..12d235c --- /dev/null +++ b/src/v2/index.child-guard.test.ts @@ -0,0 +1,215 @@ +import { describe, test, expect, afterEach } from "bun:test" +import { rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +/** + * v2: subagent protection without ctx.client. + * + * Production ctx carries no client and no session list/get/active surface + * (probed live 2026-10-06: every subagent check silently degraded to false; + * zero "injection refused" lines in all of log history). Two injections + * landed in child ses_eed4521bdffeHLLmpomhiB0KXx the same millisecond from + * stacked setups whose backoff timers survived disposal. So: + * 1. subagent identity resolves over the loopback session list + * (GET /api/session carries parentID rows), and + * 2. dispose cancels pending recover timers (plus a running-check in the + * timer body), so a superseded instance stays silent. + */ + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_parent" +const CHILD = "ses_eed4521bdffeHLLmpomhiB0KXx" + +let counter = 3000 +const realFetch = globalThis.fetch +const realServerPort = process.env.OPENCODE_SERVER_PORT + +afterEach(() => { + globalThis.fetch = realFetch + if (realServerPort === undefined) delete process.env.OPENCODE_SERVER_PORT + else process.env.OPENCODE_SERVER_PORT = realServerPort +}) + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, sid: string, data: Record = {}) => ({ type, data: { sessionID: sid, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 300, + toolTextCheckDelayMs: 0, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + subagentWaitMs: 40, + debug: false, +} + +const userMessage = (text: string, at: number) => ({ + type: "user", + id: "msg_u", + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantText = (text: string, at: number) => ({ + type: "assistant", + id: "msg_a", + time: { created: at }, + content: [{ type: "text", text }], +}) + +type History = Record + +async function setup(opts: { + history?: History + active?: string[] + withClient?: boolean + backoffMs?: number +}): Promise { + const injected: Array<{ sid?: string; text?: string }> = [] + const interrupted: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-childguard-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const history: History = opts.history ?? {} + const active = new Set(opts.active ?? []) + + const ctx: any = { + event: stream, + options: { + ...OPTIONS, + logFile, + ...(opts.backoffMs ? { baseBackoffMs: opts.backoffMs, maxBackoffMs: opts.backoffMs } : {}), + }, + session: { + context: async (a: any) => history[a?.sessionID ?? SID] ?? [], + active: async () => Object.fromEntries([...active].map((s) => [s, {}])), + interrupt: async (a: any) => (interrupted.push(a?.sessionID), {}), + synthetic: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + if (opts.withClient !== false) { + ctx.client = { + session: { + get: async ({ path }: any) => ({ data: { id: path?.id } }), + list: async () => ({ data: [] }), + }, + } + } + + const cleanup = await (plugin as any).setup(ctx) + await wait(60) + return { injected, interrupted, cleanup, logFile, streamRef: stream } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const push = (h: any, type: string, sid: string, data: Record = {}) => h.streamRef.push(ev(type, sid, data)) + +async function makeBusy(h: any, sid: string) { + push(h, "session.execution.started", sid) + await wait(10) + push(h, "session.step.started", sid) + await wait(10) +} + +/** Loopback session list stub: the shape prod GET /api/session returns. */ +function stubSessionList(rows: Array<{ id: string; parentID?: string }>) { + globalThis.fetch = (async (url: any) => { + const u = String(url) + if (u.includes("/api/session?") || u.endsWith("/api/session")) { + return { ok: true, json: async () => ({ data: rows }) } as any + } + return { ok: false, status: 404, json: async () => ({}) } as any + }) as any +} + +describe("v2: subagent protection without ctx.client", () => { + test("a stalled child is refused injection via the loopback list", async () => { + // Prod shape: no ctx.client at all. The child goes quiet past the + // stall threshold; the ONLY thing that knows it is a child is the + // loopback session list. Before the HTTP fallback this injected. + // serverBaseUrls() needs a port to build a base — prod has one via + // env/argv, the test env does not, so pin one here. + process.env.OPENCODE_SERVER_PORT = "18234" + stubSessionList([ + { id: SID }, + { id: CHILD, parentID: SID }, + ]) + const now = Date.now() + const h = await setup( + { + history: { [CHILD]: [userMessage("go", now - 60_000), assistantText("working", now - 5_000)] }, + active: [CHILD], + withClient: false, + }, + ) + await makeBusy(h, CHILD) + await wait(900) + expect(h.injected.filter((i: any) => i.sid === CHILD)).toHaveLength(0) + expect(h.interrupted).toHaveLength(0) + await teardown(h) + }) + + test("dispose cancels a scheduled stall inject", async () => { + // The 5ms-apart twin prods: instance A schedules its backoff timer, + // setup B disposes A, A's timer still fires. With a 5s backoff there + // is ample room to dispose first: nothing may land afterwards. + stubSessionList([{ id: SID }]) + const now = Date.now() + const h = await setup({ + history: { [SID]: [userMessage("go", now - 60_000), assistantText("working", now - 5_000)] }, + active: [SID], + backoffMs: 1500, + }) + await makeBusy(h, SID) + await wait(700) + await teardown(h) + await wait(2200) + expect(h.injected).toHaveLength(0) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index a8c2094..efb5832 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -212,6 +212,11 @@ interface SessionWatch { * it rather than stack a second judgement on the same turn, and so a new turn * can cancel it. */ toolTextTimer: ReturnType | null + /** The pending deferred stall-recovery inject, if any. Held so dispose + * can cancel it: without this, a reloaded-away instance fires its + * backoff timer after disposal and injects from a dead sessions map + * (observed 2026-10-06: twin identical prods 5ms apart). */ + recoverTimer: ReturnType | null /** Tool calls started but not finished, tracked from the tool lifecycle events. * The orphan watch must never abort a session that is legitimately working. */ pendingTools: number @@ -1498,6 +1503,7 @@ export default define({ orphanWatchStartAt: null, orphanRecoveryTried: false, toolTextTimer: null, + recoverTimer: null, lastSubagentCheckAt: 0, pendingTools: 0, toolsInFlightAtIdle: 0, @@ -1706,6 +1712,39 @@ export default define({ return out } + /** + * Loopback session list with parentID rows. `ctx` offers no list/get/ + * active surface and no client in production (verified 2026-10-06: + * every subagent check silently degraded to false, zero refusals in + * log history) — but `GET /api/session` answers with parentID rows + * under the same auth as the message reads. Short-TTL cache: verdicts + * and guards call this several times per watchdog pass. Fail-open + * toward the pre-guard behavior (empty) on any fetch problem. + */ + type SessionRow = { id?: unknown; parentID?: unknown } + let sessionRowsCache: { at: number; rows: SessionRow[] } | null = null + const SESSION_ROWS_TTL_MS = 10_000 + async function listSessionsHttp(): Promise { + const now = Date.now() + if (sessionRowsCache && now - sessionRowsCache.at < SESSION_ROWS_TTL_MS) return sessionRowsCache.rows + for (const base of serverBaseUrls()) { + try { + const res = await fetch(`${base}/api/session?limit=500`, { headers: serverHeaders() }) + if (!res.ok) continue + const body = (await res.json()) as unknown + const data = (body as { data?: unknown } | undefined)?.data ?? body + if (Array.isArray(data)) { + const rows = (data as SessionRow[]).filter((r) => r !== null && typeof r === "object") + sessionRowsCache = { at: now, rows } + return rows + } + } catch { + // Next base; exhaustion degrades to stale/empty below. + } + } + return sessionRowsCache?.rows ?? [] + } + /** * Is this session itself a subagent? Definitive test: ask the server for * our own record and look at `parentID`. @@ -1736,6 +1775,17 @@ export default define({ } catch { sub = false } + if (!sub) { + // HTTP fallback: prod ctx has no client (verified 2026-10-06), + // so the lookup above always misses there. Same degraded-false + // contract on any fetch problem. + try { + const rows = await listSessionsHttp() + sub = rows.some((r) => r?.id === sid && typeof r?.parentID === "string") + } catch { + sub = false + } + } w.isSubAgent = sub return sub } @@ -2182,7 +2232,13 @@ export default define({ "info", `${short(sid)} stall detected (${reason}) — resume attempt ${attempt}/${maxRetries} in ${delay}ms`, ) - setTimeout(async () => { + if (w.recoverTimer) clearTimeout(w.recoverTimer) + w.recoverTimer = setTimeout(async () => { + w.recoverTimer = null + // Disposed instances must stay silent: a superseded setup's + // backoff outlives its own dispose (which is exactly how twin + // identical prods landed milliseconds apart). + if (!running) return try { // Skip only if the session genuinely turned healthy again // (a normal completion clears pendingRecoveryArmed) or the @@ -3171,6 +3227,18 @@ export default define({ if ((row as { parentID?: unknown }).parentID !== parentSid) continue out.push(sid) } + if (out.length === 0) { + // HTTP fallback: prod ctx has no list surface, so the call + // above always misses there (verified 2026-10-06). Same + // client-side verification as above. + for (const row of await listSessionsHttp()) { + const sid = row?.id + if (typeof sid !== "string" || !sid.startsWith("ses_")) continue + if (sid === parentSid) continue + if (row?.parentID !== parentSid) continue + out.push(sid) + } + } return out } @@ -4287,6 +4355,8 @@ export default define({ for (const w of sessions.values()) { if (w.toolTextTimer) clearTimeout(w.toolTextTimer) w.toolTextTimer = null + if (w.recoverTimer) clearTimeout(w.recoverTimer) + w.recoverTimer = null } sessions.clear() // Only clear the registry slot if it is still OURS: a later setup may From 0c07d32c3dffa0cd926b23a73d6cb2f4626a9e92 Mon Sep 17 00:00:00 2001 From: famewolf Date: Tue, 6 Oct 2026 16:25:57 -0400 Subject: [PATCH 53/57] v2: verify delivery before fallback sends (no twin prods) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ses_eed4521b (2026-10-06): twin identical stall continues 5ms apart. A prompt() that delivers and THEN throws (transport flakiness was in the log minutes earlier) fell through to the synthetic fallback — same text twice from one instance, no second setup needed. After any send throw, the plugin now checks the session log (delivered prompt or synthetic reads back as the newest user message) and treats it as delivered instead of sending again. Genuine failures (nothing in the log) still fall back as before. Covered by src/v2/index.delivery-once.test.ts (repro + control, fail-before shown on unpatched code). --- src/v2/index.delivery-once.test.ts | 193 +++++++++++++++++++++++++++++ src/v2/index.ts | 14 +++ 2 files changed, 207 insertions(+) create mode 100644 src/v2/index.delivery-once.test.ts diff --git a/src/v2/index.delivery-once.test.ts b/src/v2/index.delivery-once.test.ts new file mode 100644 index 0000000..50e0789 --- /dev/null +++ b/src/v2/index.delivery-once.test.ts @@ -0,0 +1,193 @@ +import { describe, test, expect } from "bun:test" +import { rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +/** + * v2: a send that throws after delivering must not send again. + * + * Observed 2026-10-06: twin identical `continue — stalled` prods 5ms apart + * in child ses_eed4521bdffeHLLmpomhiB0KXx. One sufficient mechanism needs no + * second instance at all: `ctx.session.prompt()` delivers, then throws + * (transport flakiness was in the log minutes earlier), and the catch falls + * through to `synthetic` — same text twice. After any throw, the plugin + * verifies delivery against the session log (our text already the newest + * user message) and only then falls back. + */ + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_parent" + +let counter = 4000 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, sid: string, data: Record = {}) => ({ type, data: { sessionID: sid, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 300, + maxRetries: 1, + toolTextCheckDelayMs: 0, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + injectIntervalMs: 0, + subagentWaitMs: 40, + debug: false, + visibleContinue: true, +} + +const userMessage = (text: string, at: number) => ({ + type: "user", + id: "msg_u", + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantText = (text: string, at: number) => ({ + type: "assistant", + id: "msg_a", + time: { created: at }, + content: [{ type: "text", text }], +}) + +async function setup(opts: { + deliverPrompt?: boolean + failPrompt?: boolean + failSynthetic?: boolean + logMessages?: unknown[] +}): Promise { + const sends: Array<{ channel: string; text?: string }> = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-delivery-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const logMessages: unknown[] = opts.logMessages ?? [] + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async () => [], + active: async () => ({ [SID]: {} }), + interrupt: async () => ({}), + synthetic: async (a: any) => { + sends.push({ channel: "synthetic", text: a?.text }) + if (opts.failSynthetic) throw new Error("synthetic transport blew up after send") + return {} + }, + prompt: async (a: any) => { + // deliver-then-throw: the transport error lands after the + // message is already in the log, like the 19:59 transport + // retry storm around the observed twin. + if (opts.deliverPrompt !== false) { + sends.push({ channel: "prompt", text: a?.text }) + // Newest-first log: a delivery lands at the front. + logMessages.unshift({ + role: "user", + time: { created: Date.now() }, + content: [{ type: "text", text: a?.text }], + }) + } + if (opts.failPrompt) throw new Error("prompt transport blew up after send") + return {} + }, + }, + client: { + session: { + get: async ({ path }: any) => ({ data: { id: path?.id } }), + list: async () => ({ data: [] }), + message: { list: async () => logMessages }, + }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + + const cleanup = await (plugin as any).setup(ctx) + await wait(60) + return { sends, cleanup, logFile, streamRef: stream } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const push = (h: any, type: string, sid: string, data: Record = {}) => h.streamRef.push(ev(type, sid, data)) + +async function makeBusyStalled(h: any) { + push(h, "session.execution.started", SID) + await wait(10) + push(h, "session.step.started", SID) + await wait(10) + await wait(900) +} + +describe("v2: a send that throws after delivering sends nothing more", () => { + test("prompt delivers-then-throws: no synthetic fallback", async () => { + const now = Date.now() + const h = await setup({ + deliverPrompt: true, + failPrompt: true, + logMessages: [ + { role: "assistant", time: { created: now - 5_000 }, content: [{ type: "text", text: "working" }] }, + { role: "user", time: { created: now - 60_000 }, content: [{ type: "text", text: "go" }] }, + ], + }) + await makeBusyStalled(h) + expect(h.sends).toHaveLength(1) + expect(h.sends[0].channel).toBe("prompt") + await teardown(h) + }) + + test("CONTROL: prompt fails undelivered: synthetic fallback still fires", async () => { + const now = Date.now() + const h = await setup({ + deliverPrompt: false, + failPrompt: true, + logMessages: [ + { role: "assistant", time: { created: now - 5_000 }, content: [{ type: "text", text: "working" }] }, + { role: "user", time: { created: now - 60_000 }, content: [{ type: "text", text: "go" }] }, + ], + }) + await makeBusyStalled(h) + expect(h.sends).toHaveLength(1) + expect(h.sends[0].channel).toBe("synthetic") + await teardown(h) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index efb5832..d390594 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -2076,6 +2076,14 @@ export default define({ } catch (err) { const msg = err instanceof Error ? err.message : String(err) log("warn", `${short(sid)} visible prompt failed: ${msg}`) + // A transport error can land AFTER the message was delivered + // (observed: twin identical prods 5ms apart). The delivered + // prompt is a user message in the log, so verify before + // falling back — a second send of the same text is the dupe. + if (await recentOwnProdInLog(sid, text)) { + dbg(`${short(sid)} prompt threw but the text is already the newest user message — treating as delivered`) + return true + } } } try { @@ -2089,6 +2097,12 @@ export default define({ } catch (err) { const msg = err instanceof Error ? err.message : String(err) log("warn", `${short(sid)} synthetic failed: ${msg}`) + // Same verify-before-retry: a delivered synthetic also acts as + // a user turn, so it reads back through the same log check. + if (await recentOwnProdInLog(sid, text)) { + dbg(`${short(sid)} synthetic threw but the text is already the newest user message — treating as delivered`) + return true + } // Last resort: a plain prompt still resumes the session (just // without the visible synthetic notification in the TUI). try { From 5f85adbc7f8d828ce5c912598bb8ca1085050984 Mon Sep 17 00:00:00 2001 From: famewolf Date: Tue, 6 Oct 2026 18:47:16 -0400 Subject: [PATCH 54/57] v2: quiet subagents wait inside a patience window; orphan aborts terminate ses_ef1c822c (2026-10-06): the orphan watch called a healthy long-thinking coder worker crashed after 60s quiet, interrupted the parent twice, then looped every 5s forever. Two bounded corrections (a hung model must still terminate): - subagentVerdict gains waiting: quiet past the stuck window but inside subagentDeadMs (new option, default 30m, tripled with a tool outstanding) with no error evidence waits; only error evidence or quiet past the window reads as crashed. runOrphanWatch and parentWaitingOnSubagents both wait on it; error evidence still wins immediately over a merely-quiet sibling. - runOrphanWatch counts each abort toward the gaveUp budget (both abort sites), so the watch terminates instead of cycling indefinitely. 5 orphan fixtures moved from 10m-quiet (now waiting) to past-dead quiet to keep testing crashed handling; new src/v2/index.subagent-patience.test.ts covers wait / past-dead abort / termination (fail-before shown on unpatched code). --- README.md | 1 + src/v2/index.orphan.test.ts | 13 +- src/v2/index.subagent-patience.test.ts | 222 +++++++++++++++++++++++++ src/v2/index.ts | 52 +++++- 4 files changed, 276 insertions(+), 12 deletions(-) create mode 100644 src/v2/index.subagent-patience.test.ts diff --git a/README.md b/README.md index cf43af1..70e0022 100644 --- a/README.md +++ b/README.md @@ -496,6 +496,7 @@ Defaults are the same on v1 and v2 unless a row says otherwise. | `baseBackoffMs` | `1000` | First retry delay (doubles each attempt) | | `maxBackoffMs` | `8000` | Backoff cap | | `subagentWaitMs` | `15000` | Wait before treating orphan parent as stuck | +| `subagentDeadMs` | `1800000` | Outer bound: quiet past this with no error evidence reads as dead (v2 only). Inside it, a quiet child is waited on, not aborted — 30 min default for long-thinking models | | `loopMaxContinues` | `3` | Continues in window before triggering abort | | `loopWindowMs` | `600000` | Hallucination loop detection window (10 min) | | `streamingFailureErrorNames` | `["ProviderError","APIError","StreamError","ConnectionError","TimeoutError"]` | Error names that classify as streaming failures (exact match) | diff --git a/src/v2/index.orphan.test.ts b/src/v2/index.orphan.test.ts index fcaaa71..c856772 100644 --- a/src/v2/index.orphan.test.ts +++ b/src/v2/index.orphan.test.ts @@ -170,6 +170,9 @@ async function childGoesBusyThenIdle(h: any) { } const A_LONG_WAY_AGO = Date.now() - 10 * 60_000 +// Past the 30m dead window: quiet this long with no error reads as crashed, +// while A_LONG_WAY_AGO (10m) now reads as waiting. Crashed-path tests use this. +const PAST_DEAD = Date.now() - 40 * 60_000 describe("v2: the orphan watch arms when a parent's subagents fall quiet", () => { test("CONTROL: nothing is armed while only the parent is busy", async () => { @@ -272,7 +275,7 @@ describe("v2: the orphan watch arms when a parent's subagents fall quiet", () => active: [SID, CHILD], history: { [SID]: [userMessage("go", A_LONG_WAY_AGO)], - [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(PAST_DEAD)], }, }) await makeBusy(h, SID) @@ -313,7 +316,7 @@ describe("v2: the orphan watch refuses to kill a working parent", () => { active: [SID, CHILD], history: { [SID]: [userMessage("go", A_LONG_WAY_AGO)], - [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(PAST_DEAD)], }, }) await makeBusy(h, SID) @@ -336,7 +339,7 @@ describe("v2: the orphan watch refuses to kill a working parent", () => { active: [SID, CHILD], history: { [SID]: [userMessage("go", A_LONG_WAY_AGO)], - [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(PAST_DEAD)], }, }) await makeBusy(h, SID) @@ -466,7 +469,7 @@ describe("v2: a dead subagent is woken before the parent is killed", () => { subagentWaitMs: 400, history: { [SID]: [userMessage("go", A_LONG_WAY_AGO)], - [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(PAST_DEAD)], }, }) await makeBusy(h, SID) @@ -490,7 +493,7 @@ describe("v2: a dead subagent is woken before the parent is killed", () => { subagentWaitMs: 200, history: { [SID]: [userMessage("go", A_LONG_WAY_AGO)], - [CHILD]: [assistantAt(A_LONG_WAY_AGO)], + [CHILD]: [assistantAt(PAST_DEAD)], }, }) await makeBusy(h, SID) diff --git a/src/v2/index.subagent-patience.test.ts b/src/v2/index.subagent-patience.test.ts new file mode 100644 index 0000000..aef4158 --- /dev/null +++ b/src/v2/index.subagent-patience.test.ts @@ -0,0 +1,222 @@ +import { describe, test, expect } from "bun:test" +import { existsSync, readFileSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +/** + * v2: a quiet subagent is waited on within a patience window, not killed. + * + * ses_ef1c822c (2026-10-06): the orphan watch declared a healthy + * long-thinking coder worker "crashed" after 60s of quiet, interrupted the + * parent twice, then looped every 5s forever (orphan aborts never counted + * toward gaveUp). Two corrections, both bounded so a hung model can't sit + * forever: + * 1. Quiet without error evidence waits inside an outer dead-window + * (default 30m); only error evidence or quiet past the window reads + * as crashed. + * 2. Each orphan abort counts toward the gaveUp budget, so the watch + * terminates instead of cycling indefinitely. + */ + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_parent" +const CHILD = "ses_child" + +let counter = 5000 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, sid: string, data: Record = {}) => ({ type, data: { sessionID: sid, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 600_000, + toolTextCheckDelayMs: 0, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + maxRetries: 3, + injectIntervalMs: 0, + subagentWaitMs: 40, + debug: false, +} + +const userMessage = (text: string, at: number) => ({ + type: "user", + id: "msg_u", + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantAt = (at: number) => ({ + type: "assistant", + id: "msg_a", + time: { created: at }, + content: [{ type: "text", text: "working on it" }], +}) + +type History = Record + +async function setup(opts: { + history?: History + active?: string[] + maxRetries?: number +}): Promise { + const injected: Array<{ sid?: string; text?: string }> = [] + const interrupted: string[] = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-patience-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const history: History = opts.history ?? {} + const active = new Set(opts.active ?? []) + const rows = [{ id: SID }, { id: CHILD, parentID: SID }] + + const ctx: any = { + event: stream, + options: { + ...OPTIONS, + logFile, + ...(opts.maxRetries !== undefined ? { maxRetries: opts.maxRetries } : {}), + }, + session: { + context: async (a: any) => history[a?.sessionID ?? SID] ?? [], + active: async () => Object.fromEntries([...active].map((s) => [s, {}])), + interrupt: async (a: any) => (interrupted.push(a?.sessionID), {}), + synthetic: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + }, + client: { + session: { + get: async ({ path }: any) => ({ data: { id: path?.id } }), + list: async (a: any) => { + if (!a?.parentID) return { data: rows } + return { data: rows.filter((r) => (r as any).parentID === a.parentID) } + }, + }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + + const cleanup = await (plugin as any).setup(ctx) + await wait(60) + return { injected, interrupted, logFile, streamRef: stream, cleanup } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const logs = (h: any) => (existsSync(h.logFile) ? readFileSync(h.logFile, "utf8") : "") +const push = (h: any, type: string, sid: string, data: Record = {}) => h.streamRef.push(ev(type, sid, data)) + +async function makeBusy(h: any, sid: string) { + push(h, "session.execution.started", sid) + await wait(10) + push(h, "session.step.started", sid) + await wait(10) +} + +/** Parent busy, child goes quiet: arms the orphan watch. */ +async function armWatch(h: any) { + await makeBusy(h, SID) + await wait(20) + await makeBusy(h, CHILD) + await wait(20) + push(h, "session.idle", CHILD) + await wait(20) +} + +describe("v2: quiet subagents wait inside a patience window", () => { + test("a healthy quiet worker is waited on, not aborted", async () => { + // 5 minutes of quiet: past the 60s stuck window, inside the 30m + // dead window, no error. Old verdict: crashed -> interrupt + prod. + const now = Date.now() + const h = await setup({ + active: [SID, CHILD], + history: { + [SID]: [userMessage("go", now - 60_000)], + [CHILD]: [assistantAt(now - 5 * 60_000)], + }, + }) + await armWatch(h) + await wait(600) + expect(h.interrupted).toHaveLength(0) + expect(h.injected.filter((i: any) => i.sid === SID)).toHaveLength(0) + await teardown(h) + }) + + test("CONTROL: quiet past the dead window still aborts", async () => { + // 40 minutes of quiet with no error: a hung model must not sit + // forever. Crashed handling is preserved. + const now = Date.now() + const h = await setup({ + active: [SID], + history: { + [SID]: [userMessage("go", now - 60_000)], + [CHILD]: [assistantAt(now - 40 * 60_000)], + }, + }) + await armWatch(h) + await wait(600) + expect(h.interrupted.length).toBeGreaterThan(0) + await teardown(h) + }) + + test("orphan aborts terminate instead of cycling forever", async () => { + // maxRetries 1: one orphan abort spends the budget; the next tick + // must give up, not abort again. + const now = Date.now() + const h = await setup({ + active: [SID], + maxRetries: 1, + history: { + [SID]: [userMessage("go", now - 60_000)], + [CHILD]: [assistantAt(now - 40 * 60_000)], + }, + }) + await armWatch(h) + // The abort path holds `aborting` for 2s (interrupt window), during + // which the watch skips; gaveUp lands on the first tick after that. + await wait(3000) + expect(h.interrupted).toHaveLength(1) + expect(logs(h)).toContain("gave up") + await teardown(h) + }) +}) diff --git a/src/v2/index.ts b/src/v2/index.ts index d390594..bca10a9 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -340,6 +340,8 @@ export interface AutoResumeOptions { toolTextCheckDelayMs?: number /** Wait for a subagent to report before treating it as stalled. */ subagentWaitMs?: number + /** Outer bound: quiet past this with no error evidence reads as dead. */ + subagentDeadMs?: number /** Token floor for the silent-dead-stream heuristic. */ silentDeadStreamMinTokens?: number /** Error names that mark a streaming failure. */ @@ -467,6 +469,12 @@ const DEFAULT_SUBAGENT_WAIT_MS = 15_000 * as stuck. v1 used the same number, and a tool call still outstanding triples * it, because a long tool is not a hung model. */ const SUBAGENT_STUCK_MS = 60_000 +/** Outer bound: quiet past this with no error evidence reads as dead, not + * merely stuck. Between the stuck window and this bound a quiet child is + * waited on — a hung model must not sit forever, but a long-thinking one + * (coder blocks run many minutes on one shared GPU) must not read as + * crashed at 60s either. ses_ef1c822c 2026-10-06. */ +const DEFAULT_SUBAGENT_DEAD_MS = 30 * 60_000 const SUBAGENT_RECOVERY_PROMPT = "It looks like you may have stalled or timed out. Please retry the last operation or continue with the task." const DEFAULT_SILENT_DEAD_STREAM_MIN_TOKENS = 200 @@ -1274,6 +1282,7 @@ export default define({ // How long a parent may sit busy after its last subagent went idle before // the orphan watch acts. v1 default, honoured for the first time here. const subagentWaitMs = opts.subagentWaitMs ?? DEFAULT_SUBAGENT_WAIT_MS + const subagentDeadMs = opts.subagentDeadMs ?? DEFAULT_SUBAGENT_DEAD_MS // How long to let a finished turn's text settle before judging it against the // done/tool patterns. v1's default, and the reason v1 does not judge on idle // at all — see inspectOnIdle. @@ -1377,6 +1386,7 @@ export default define({ "minActivityGapMs", "toolTextCheckDelayMs", "subagentWaitMs", + "subagentDeadMs", "silentDeadStreamMinTokens", "streamingFailureErrorNames", "streamingFailureMessagePatterns", @@ -3256,7 +3266,7 @@ export default define({ return out } - type SubagentVerdict = { status: "crashed" | "idle" | "busy"; stuckSid?: string } + type SubagentVerdict = { status: "crashed" | "idle" | "busy" | "waiting"; stuckSid?: string } /** * What a parent's subagents are doing, in v1's three-way shape. @@ -3286,6 +3296,7 @@ export default define({ if (children.length === 0) return { status: "idle" } const now = Date.now() let sawBusy = false + let sawWaiting: string | null = null for (const child of children) { const messages = await loadMessages(child) const last = messages[messages.length - 1] as @@ -3307,17 +3318,33 @@ export default define({ (p) => p?.type === "tool" && p.state?.status !== "completed" && p.state?.status !== "error", ) const limit = hasToolCall ? SUBAGENT_STUCK_MS * 3 : SUBAGENT_STUCK_MS + // Outer bound: quiet past the stuck window but inside the dead + // window, with no error evidence, is a long thinker — waited + // on, not killed. Only past-dead quiet reads as crashed, so + // a hung model still terminates (bounded wait, ses_ef1c822c). + // Error evidence above already returned; scan on so a later + // crashed sibling is still surfaced. + const deadLimit = hasToolCall ? subagentDeadMs * 3 : subagentDeadMs if (silentFor <= limit) { // Recent enough to be believed, whatever the server thinks. if (activeIDs.has(child)) sawBusy = true continue } + if (silentFor <= deadLimit) { + dbg( + `subagent ${short(child)} quiet for ${Math.round(silentFor / 1000)}s, inside ${Math.round(deadLimit / 1000)}s patience — waiting, not crashed (active=${activeIDs.has(child)})`, + ) + if (sawWaiting === null) sawWaiting = child + continue + } dbg( `subagent ${short(child)} silent for ${Math.round(silentFor / 1000)}s (active=${activeIDs.has(child)})`, ) return { status: "crashed", stuckSid: child } } - return sawBusy ? { status: "busy" } : { status: "idle" } + if (sawBusy) return { status: "busy" } + if (sawWaiting !== null) return { status: "waiting", stuckSid: sawWaiting } + return { status: "idle" } } catch (e) { dbg(`subagent check failed for ${short(parentSid)}:`, e instanceof Error ? e.message : String(e)) // Unknown is treated as busy: the cost of waiting is a later abort, and @@ -3440,9 +3467,18 @@ export default define({ } } log("info", `${short(sid)} subagent crashed and did not recover — aborting and resuming the parent`) + // Count orphan aborts toward the gaveUp budget (checked at this + // function's top): without this the watch cycles forever, because + // nothing else increments resumeAttempts on this path. + w.resumeAttempts++ tryAbortAndResume(sid, w) return } + if (verdict.status === "waiting") { + dbg(`${short(sid)} subagents quiet but inside patience — waiting, not aborting`) + w.orphanWatchStartAt = now + return + } if (verdict.status === "busy") { dbg(`${short(sid)} subagents still working — waiting`) w.orphanWatchStartAt = now @@ -3456,6 +3492,7 @@ export default define({ // 15s, and its name is the warning. log("info", `${short(sid)} stuck with no live subagents — aborting and resuming`) tryAbortAndResume(sid, w) + w.resumeAttempts++ } /** @@ -3466,14 +3503,15 @@ export default define({ * heuristic below: native dispatches may never emit `session.tool.called`, * and the server active set may lag the parentID link. * - * Waits when the verdict confirms a live child, or when the parent still - * holds tools in flight against children that are neither confirmed live - * nor proven dead. A crashed verdict, or tools held with no children on - * the link at all (wedged bare tool), falls through to normal recovery. + * Waits when the verdict confirms a live child or a quiet one inside + * the patience window, or when the parent still holds tools in flight + * against children that are neither confirmed live nor proven dead. + * A crashed verdict, or tools held with no children on the link at + * all (wedged bare tool), falls through to normal recovery. */ async function parentWaitingOnSubagents(sid: string, activeIDs: Set): Promise { const verdict = await subagentVerdict(sid, activeIDs) - if (verdict.status === "busy") return true + if (verdict.status === "busy" || verdict.status === "waiting") return true if (verdict.status !== "idle") return false const w = ensureWatch(sid) if (w.pendingTools <= 0) return false From 511d64d67a26f45f56d7b913d0ad4f3ad4c8cbc3 Mon Sep 17 00:00:00 2001 From: famewolf Date: Tue, 6 Oct 2026 19:09:41 -0400 Subject: [PATCH 55/57] test: parent-wait CONTROL drops the child link entirely MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A merely-quiet child now reads as waiting (patience window), so the CONTROL's old fixture (linked but quiet child) no longer exercises the recover path. The setup takes an explicit listRows override and the CONTROL runs with no child rows at all — matching its name. --- src/v2/index.parent-wait.test.ts | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/src/v2/index.parent-wait.test.ts b/src/v2/index.parent-wait.test.ts index 1f8ad92..2b39444 100644 --- a/src/v2/index.parent-wait.test.ts +++ b/src/v2/index.parent-wait.test.ts @@ -75,7 +75,13 @@ const assistantAt = (at: number) => ({ type History = Record -async function setup(opts: { history?: History; active?: string[] }): Promise { +async function setup(opts: { + history?: History + /** Session ids the server reports busy. */ + active?: string[] + /** Rows `session.list` returns. Defaults to parent + one linked child. */ + listRows?: Array<{ id: string; parentID?: string }> +}): Promise { const injected: Array<{ sid?: string; text?: string }> = [] const interrupted: string[] = [] const stream = makeEventStream() @@ -84,7 +90,7 @@ async function setup(opts: { history?: History; active?: string[] }): Promise { test("CONTROL: a truly stalled parent with no live children still recovers", async () => { // No children at all on the parentID link — ordinary stall, must fire. + // (A merely-quiet child now reads as waiting, not dead, so the link + // itself must be absent for this control.) const now = Date.now() const h = await setup({ active: [SID], + listRows: [{ id: SID }], history: { [SID]: [userMessage("go", now - 60_000)] }, }) - // NOTE: default rows still link CHILD to SID; override by pushing the - // child stale AND out of the active set is covered by the next test. - // Here we keep the link but the child is long dead. - h.history[CHILD] = [assistantAt(now - 10 * 60_000)] await makeBusy(h, SID) await wait(900) expect(parentInjects(h).length).toBeGreaterThan(0) From 22d8f2807ade8e98047f55a31bfd50930f82116e Mon Sep 17 00:00:00 2001 From: famewolf Date: Tue, 6 Oct 2026 19:44:33 -0400 Subject: [PATCH 56/57] v2: dead instances send nothing (liveness checks on every send path) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Twin identical prods 5ms apart (ses_eed4521b, 2026-10-06) also fit a second mechanism needing no second setup: the backoff timer fires, then a slow transport await lets a superseding setup dispose this one mid-flight, and the send proceeds anyway — clearTimeout cannot stop an already-fired timer. Every send path now re-checks liveness: entry of injectOnceLocked (before ensureWatch, so dead copies cannot resurrect watch state), both notifyAndPrompt fallback edges, recoverStuckSubagent entry, and tryAbortAndResume after its 2s settle sleep. Covered by the new disposed-instance test (fail-before shown); the delivered-then-throw twin is covered by index.delivery-once.test.ts. --- src/v2/index.delivery-once.test.ts | 54 +++++++++++++++++++++++++++++- src/v2/index.ts | 15 ++++++++- 2 files changed, 67 insertions(+), 2 deletions(-) diff --git a/src/v2/index.delivery-once.test.ts b/src/v2/index.delivery-once.test.ts index 50e0789..57e12d3 100644 --- a/src/v2/index.delivery-once.test.ts +++ b/src/v2/index.delivery-once.test.ts @@ -158,7 +158,59 @@ async function makeBusyStalled(h: any) { await wait(900) } -describe("v2: a send that throws after delivering sends nothing more", () => { +describe("v2: a disposed instance sends nothing more", () => { + test("prompt rejects post-dispose: no synthetic fallback", async () => { + // The twin mechanism: timer fired and the prompt await is in flight + // when a newer setup disposes this one. The rejection must not + // trigger the fallback send — the live instance owns the session now. + let releasePrompt!: (err: unknown) => void + const promptGate = new Promise((_, reject) => { + releasePrompt = reject + }) + const sent: string[] = [] + const now = Date.now() + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-disposed-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile, maxRetries: 1 }, + session: { + context: async () => [ + { type: "assistant", time: { created: now - 5_000 }, content: [{ type: "text", text: "working" }] }, + ], + active: async () => ({ [SID]: {} }), + interrupt: async () => ({}), + synthetic: async (a: any) => (sent.push(`synthetic:${a?.text}`), {}), + prompt: async (a: any) => { + sent.push(`prompt:${a?.text}`) + await promptGate + return {} + }, + }, + client: { + session: { + get: async ({ path }: any) => ({ data: { id: path?.id } }), + list: async () => ({ data: [] }), + message: { list: async () => [] }, + }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + const cleanup = await (plugin as any).setup(ctx) + await wait(60) + stream.push({ type: "session.execution.started", data: { sessionID: SID } }) + await wait(10) + stream.push({ type: "session.step.started", data: { sessionID: SID } }) + await wait(700) + ;(cleanup as () => void)() + releasePrompt(new Error("transport blew up")) + await wait(300) + expect(sent.filter((s) => s.startsWith("synthetic:"))).toHaveLength(0) + rmSync(logFile, { force: true }) + }) + test("prompt delivers-then-throws: no synthetic fallback", async () => { const now = Date.now() const h = await setup({ diff --git a/src/v2/index.ts b/src/v2/index.ts index bca10a9..97ce0d5 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -1931,6 +1931,10 @@ export default define({ allowDuringSelfAbort: boolean, checkDuplicate: boolean, ): Promise { + // Dead instances stay silent across every injection path, not + // just the stall timer that schedules most of them. Checked + // before ensureWatch so a dead copy cannot resurrect watch state. + if (!running) return false const w = ensureWatch(sid) // A subagent is not ours to recover. Checked first, before every other // guard, so no code path below can reach a child. @@ -2094,6 +2098,10 @@ export default define({ dbg(`${short(sid)} prompt threw but the text is already the newest user message — treating as delivered`) return true } + // Dead instances stay silent: a superseded setup's in-flight + // send may reject after disposal — the live instance owns the + // session now, and any fallback from here would double-send. + if (!running) return false } } try { @@ -2107,6 +2115,7 @@ export default define({ } catch (err) { const msg = err instanceof Error ? err.message : String(err) log("warn", `${short(sid)} synthetic failed: ${msg}`) + if (!running) return false // Same verify-before-retry: a delivered synthetic also acts as // a user turn, so it reads back through the same log check. if (await recentOwnProdInLog(sid, text)) { @@ -2157,8 +2166,11 @@ export default define({ const msg = err instanceof Error ? err.message : String(err) log("warn", `${short(sid)} interrupt failed: ${msg}`) } - // Give the runtime a beat to settle the interrupted turn + // Give the runtime a beat to settle the interrupted turn. + // A superseded setup must not resume after this gap: liveness is + // re-checked because disposal can land mid-sleep. await new Promise((r) => setTimeout(r, 2_000)) + if (!running) return false w.aborting = false w.resumeAttempts = 0 // Deliberately does not recordContinue(): our own escalation must not feed @@ -3367,6 +3379,7 @@ export default define({ * killing the parent and its whole turn. */ async function recoverStuckSubagent(sid: string): Promise { + if (!running) return false try { await callSessionApi("synthetic", { sessionID: sid, text: SUBAGENT_RECOVERY_PROMPT }) log("info", `${short(sid)} recovery prompt sent to stuck subagent`) From 4eeaed815c78fef51a027028dd28d704c2111765 Mon Sep 17 00:00:00 2001 From: famewolf Date: Wed, 7 Oct 2026 13:50:33 -0400 Subject: [PATCH 57/57] v2: genuine user message re-arms after user interrupt (timestamp-gated) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ses_eeb02d4b (2026-10-07): after an interrupt, auto-resume never got the session going again — userCancelled latched permanently (set once, cleared nowhere) while every recovery path checks it first. A genuine new inbound user message now clears it alongside the other budget re-arms (own prompts excluded as before), so manual interrupts still stand down but handing the session back re-enables recovery. The clearing requires message time >= latch time: a pre-interrupt message the plugin simply hadn't observed yet must not re-arm. Covered by src/v2/index.user-rearm.test.ts (re-arm + stood-down control, fail-before shown). --- src/v2/index.ts | 44 ++++++++- src/v2/index.user-rearm.test.ts | 168 ++++++++++++++++++++++++++++++++ 2 files changed, 207 insertions(+), 5 deletions(-) create mode 100644 src/v2/index.user-rearm.test.ts diff --git a/src/v2/index.ts b/src/v2/index.ts index 97ce0d5..6f80610 100644 --- a/src/v2/index.ts +++ b/src/v2/index.ts @@ -141,6 +141,10 @@ interface SessionWatch { lastActivityAt: number status: "busy" | "idle" | "unknown" userCancelled: boolean + /** When the user-cancel latch was set. Re-arm requires an inbound user + * message newer than this — otherwise a pre-interrupt message the plugin + * simply hadn't observed yet would clear a latch it postdates. */ + userCancelledAt: number resumeAttempts: number lastRetryAt: number gaveUp: boolean @@ -1481,6 +1485,7 @@ export default define({ lastActivityAt: Date.now(), status: "unknown", userCancelled: false, + userCancelledAt: 0, resumeAttempts: 0, lastRetryAt: 0, gaveUp: false, @@ -2303,7 +2308,7 @@ export default define({ } // Same re-check for user input: a question asked between // detection and this delayed inject parks the turn. - if (isWaitingOnUser(await loadMessages(sid))) { + if (isWaitingOnUser(sid, w, await loadMessages(sid))) { dbg(`${short(sid)} waiting on user input at inject time — not interrupting`) return } @@ -2724,7 +2729,11 @@ export default define({ * on one of those is waiting, not stalled (ses_ef72c5f3 2026-10-05: a * stall continue fired into a session parked on a `question`). */ - function isWaitingOnUser(messages: unknown[]): boolean { + function isWaitingOnUser(sid: string, w: SessionWatch, messages: unknown[]): boolean { + // A genuine new user message re-arms here too (not just on the + // idle path): this is what clears a user-interrupt stand-down + // when the user hands the session back. + noteInboundUserMessage(sid, w, messages) const newest = messages[messages.length - 1] as { type?: string content?: { type?: string; name?: string; state?: { status?: string } }[] @@ -2846,6 +2855,17 @@ export default define({ } w.doneClaimAttempts = 0 w.doneClaimOpenTodosAttempts = 0 + // A genuine new user message hands the session back: clear a + // user-interrupt stand-down (ses_eeb02d4b, 2026-10-07 — the latch + // otherwise holds until session cleanup and even an explicit + // "continue" is refused). Only a message newer than the latch + // itself re-arms: a pre-interrupt message the plugin simply + // hadn't observed yet must not clear it. Manual interrupts + // themselves are untouched; only future recovery re-arms. + if (w.userCancelled && typeof latest.at === "number" && latest.at >= w.userCancelledAt) { + dbg(`${short(sid)} new user message after interrupt — re-arming recovery`) + w.userCancelled = false + } // A new request is new work: the ack self-loop counter starts clean, // because the model re-announcing completion after the user asked for // more is not the stuck case this counts. @@ -3545,7 +3565,18 @@ export default define({ const activeSet = new Set(activeIDs) for (const [sid, w] of sessions) { - if (w.status !== "busy" || w.userCancelled) continue + if (w.status !== "busy" || w.userCancelled) { + // A stood-down session still gets its inbound mail read: a + // genuine new user message re-arms (clears userCancelled + // inside noteInboundUserMessage), so the latch cannot block + // its own clearing path. Skipped otherwise. + if (w.status === "busy" && w.userCancelled) { + noteInboundUserMessage(sid, w, await loadMessages(sid)) + if (w.userCancelled) continue + } else { + continue + } + } // Warmup: a session that only just went busy has not had a chance to // emit anything. Without this a freshly-started turn can be declared // stalled while it is still queueing its first model call. @@ -3621,7 +3652,7 @@ export default define({ // an unanswered permission beyond the event flow) is waiting, // not stalled. Narrower than the idle-path check on purpose: a // wedged bare tool with zero events is still a stall. - if (isWaitingOnUser(await loadMessages(sid))) { + if (isWaitingOnUser(sid, w, await loadMessages(sid))) { dbg(`${short(sid)} waiting on user input — skipping stall detection`) continue } @@ -4075,7 +4106,10 @@ export default define({ // userCancelled; `shutdown`/`superseded`/`inactivity` must not // permanently disable recovery for the session. const mine = w.aborting || selfAbortActive(w) - if (!mine) w.userCancelled = reason === undefined || reason === "user" + if (!mine) { + w.userCancelled = reason === undefined || reason === "user" + if (w.userCancelled) w.userCancelledAt = Date.now() + } // A user interrupt also ends any in-flight compaction. w.compacting = false w.compactionStartedAt = null diff --git a/src/v2/index.user-rearm.test.ts b/src/v2/index.user-rearm.test.ts new file mode 100644 index 0000000..8457bca --- /dev/null +++ b/src/v2/index.user-rearm.test.ts @@ -0,0 +1,168 @@ +import { describe, test, expect } from "bun:test" +import { rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import plugin from "./index" + +/** + * v2: a genuine new user message re-arms recovery after a user interrupt. + * + * ses_eeb02d4b (2026-10-07): after an interrupt, auto-resume never got the + * session going again — not even after an explicit user "continue". + * `userCancelled` latched permanently: set once on a non-plugin interrupt + * and cleared nowhere (not on new user messages, not on new turns), while + * every recovery path checks it first. A new inbound user message is the + * hand-back signal and re-arms alongside the other budgets — own injected + * prompts excluded, as with every other re-arm. + */ + +const wait = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const SID = "ses_parent" + +let counter = 6000 + +function makeEventStream() { + const queue: any[] = [] + const waiters: ((ev: any) => void)[] = [] + let closed = false + const stream = { + push(ev: any) { + if (waiters.length) waiters.shift()!(ev) + else queue.push(ev) + }, + close() { + closed = true + while (waiters.length) waiters.shift()!(null) + }, + subscribe: () => stream, + [Symbol.asyncIterator]() { + return { + next: () => + new Promise((resolve) => { + if (queue.length) return resolve({ value: queue.shift(), done: false }) + if (closed) return resolve({ value: undefined, done: true }) + waiters.push((ev) => + resolve(ev === null ? { value: undefined, done: true } : { value: ev, done: false }), + ) + }), + return: () => { + closed = true + return Promise.resolve({ value: undefined, done: true }) + }, + } + }, + } + return stream +} + +const ev = (type: string, sid: string, data: Record = {}) => ({ type, data: { sessionID: sid, ...data } }) + +const OPTIONS = { + chunkTimeoutMs: 300, + toolTextCheckDelayMs: 0, + checkIntervalMs: 20, + gracePeriodMs: 0, + warmupMs: 0, + baseBackoffMs: 1, + maxBackoffMs: 2, + maxRetries: 3, + injectIntervalMs: 0, + subagentWaitMs: 40, + debug: false, +} + +const userMessage = (text: string, at: number, id = "msg_u") => ({ + type: "user", + id, + time: { created: at }, + content: [{ type: "text", text }], +}) + +const assistantText = (text: string, at: number) => ({ + type: "assistant", + id: "msg_a", + time: { created: at }, + content: [{ type: "text", text }], +}) + +async function setup(history: Record): Promise { + const injected: Array<{ sid?: string; text?: string }> = [] + const stream = makeEventStream() + const logFile = join(tmpdir(), `auto-resume-rearm-${process.pid}-${counter++}.log`) + rmSync(logFile, { force: true }) + + const ctx: any = { + event: stream, + options: { ...OPTIONS, logFile }, + session: { + context: async (a: any) => history[a?.sessionID ?? SID] ?? [], + active: async () => ({ [SID]: {} }), + interrupt: async () => ({}), + synthetic: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + prompt: async (a: any) => (injected.push({ sid: a?.sessionID, text: a?.text }), {}), + }, + client: { + session: { + get: async ({ path }: any) => ({ data: { id: path?.id } }), + list: async () => ({ data: [{ id: SID }] }), + }, + }, + storage: { get: async () => ({ todos: [] }), set: async () => {}, remove: async () => {} }, + tool: { transform: async (cb: any) => (cb({ add: () => {} }), { dispose() {} }), list: async () => [] }, + } + + const cleanup = await (plugin as any).setup(ctx) + await wait(60) + return { injected, cleanup, logFile, streamRef: stream, history } +} + +async function teardown(h: any) { + ;(h.cleanup as (() => void) | undefined)?.() + rmSync(h.logFile, { force: true }) +} + +const push = (h: any, type: string, sid: string, data: Record = {}) => h.streamRef.push(ev(type, sid, data)) + +describe("v2: genuine user message re-arms after user interrupt", () => { + test("interrupt latches, then a new user message re-arms recovery", async () => { + const now = Date.now() + const history: Record = { + [SID]: [userMessage("go", now - 60_000, "msg_u1")], + } + const h = await setup(history) + push(h, "session.execution.started", SID) + await wait(10) + push(h, "session.execution.interrupted", SID, { reason: "user" }) + await wait(50) + // Genuine new inbound user message, then a fresh turn that stalls. + history[SID] = [ + userMessage("go", now - 60_000, "msg_u1"), + userMessage("continue", Date.now(), "msg_u2"), + assistantText("working", Date.now()), + ] + push(h, "session.execution.started", SID) + await wait(10) + push(h, "session.step.started", SID) + await wait(900) + expect(h.injected.filter((i: any) => i.sid === SID).length).toBeGreaterThan(0) + await teardown(h) + }) + + test("CONTROL: interrupt with no new user message stays stood down", async () => { + const now = Date.now() + const history: Record = { + [SID]: [userMessage("go", now - 60_000, "msg_u1")], + } + const h = await setup(history) + push(h, "session.execution.started", SID) + await wait(10) + push(h, "session.execution.interrupted", SID, { reason: "user" }) + await wait(50) + push(h, "session.execution.started", SID) + await wait(10) + push(h, "session.step.started", SID) + await wait(900) + expect(h.injected.filter((i: any) => i.sid === SID)).toHaveLength(0) + await teardown(h) + }) +})