From 2606e184fc1af4401c8f056116c91a4f6983e9cc Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Tue, 1 Sep 2026 21:52:45 +0900 Subject: [PATCH 01/14] wip: preserved partial work (auto, session did not succeed) --- src/adapters/chatStream.ts | 69 ++++++++++++++++++++++------------ src/adapters/codexResponses.ts | 28 ++++++++++++++ src/core/eventHub.ts | 16 ++++++++ src/discord/discordPair.ts | 10 ++++- src/tui/components/LogLine.tsx | 10 ++++- 5 files changed, 105 insertions(+), 28 deletions(-) diff --git a/src/adapters/chatStream.ts b/src/adapters/chatStream.ts index 1e70c772..5a8c6e84 100644 --- a/src/adapters/chatStream.ts +++ b/src/adapters/chatStream.ts @@ -48,36 +48,38 @@ export function reduceChatChunks(chunks: StreamChunk[], onToken?: (delta: string // Tool calls accumulate by their streaming index (id/name arrive once, arguments stream). const calls = new Map(); - for (const chunk of chunks) { - if (chunk.usage) { - const pt = chunk.usage.prompt_tokens ?? 0; - const ct = chunk.usage.completion_tokens ?? 0; - usage = { prompt_tokens: pt, completion_tokens: ct, total_tokens: chunk.usage.total_tokens ?? pt + ct }; + for (const c of chunks) { + const delta = c.choices?.[0]?.delta; + if (!delta) { + if (c.usage) usage = c.usage as ChatCompletionLike['usage']; + if (c.choices?.[0]?.finish_reason) finishReason = c.choices[0].finish_reason; + continue; } - const choice = chunk.choices?.[0]; - if (!choice) continue; - const delta = choice.delta ?? {}; - if (typeof delta.content === 'string' && delta.content) { + if (delta.content) { content += delta.content; sawContent = true; - onToken?.(delta.content); + if (onToken) onToken(delta.content); } - for (const tc of delta.tool_calls ?? []) { - const idx = tc.index ?? 0; - const cur = calls.get(idx) ?? { id: '', name: '', args: '' }; - if (tc.id) cur.id = tc.id; - if (tc.function?.name) cur.name = tc.function.name; - if (tc.function?.arguments) cur.args += tc.function.arguments; - calls.set(idx, cur); + if (delta.tool_calls) { + for (const tc of delta.tool_calls) { + const idx = tc.index ?? 0; + let entry = calls.get(idx); + if (!entry) { + entry = { id: tc.id ?? '', name: tc.function?.name ?? '', args: '' }; + calls.set(idx, entry); + } + if (tc.id) entry.id = tc.id; + if (tc.function?.name) entry.name = tc.function.name; + if (tc.function?.arguments) entry.args += tc.function.arguments; + } } - if (choice.finish_reason) finishReason = choice.finish_reason; + if (c.choices?.[0]?.finish_reason) finishReason = c.choices[0].finish_reason; } - const toolCalls: StreamToolCall[] = [...calls.values()].map((c) => ({ - id: c.id, - type: 'function', - function: { name: c.name, arguments: c.args }, - })); + const toolCalls: StreamToolCall[] = []; + for (const [, v] of calls) { + toolCalls.push({ id: v.id, type: 'function', function: { name: v.name, arguments: v.args } }); + } return { choices: [ @@ -107,6 +109,9 @@ function parseChunkLine(line: string): StreamChunk | null { } } +/** Hard cap on the SSE partial-frame buffer to prevent memory exhaustion. */ +const MAX_FRAME_SIZE = 512 * 1024; // 512 KiB + /** Read a chat/completions SSE body and reduce it, emitting content deltas live. */ export async function consumeChatCompletionsStream( res: Response, @@ -127,7 +132,21 @@ export async function consumeChatCompletionsStream( for (;;) { const { done, value } = await reader.read(); if (done) break; - buffer += decoder.decode(value, { stream: true }); + const decoded = decoder.decode(value, { stream: true }); + // Guard: if the accumulated buffer already exceeds the limit, discard the + // incoming chunk to avoid unbounded growth from a single oversized frame. + if (buffer.length + decoded.length > MAX_FRAME_SIZE) { + // Truncate decoded to fit within MAX_FRAME_SIZE + const remainingSpace = MAX_FRAME_SIZE - buffer.length; + const truncated = decoded.slice(0, remainingSpace); + buffer += truncated; + // Process and flush the buffer immediately + const lines = buffer.split('\n'); + buffer = ''; + for (const line of lines) handle(parseChunkLine(line)); + continue; + } + buffer += decoded; const lines = buffer.split('\n'); buffer = lines.pop() ?? ''; for (const line of lines) handle(parseChunkLine(line)); @@ -136,4 +155,4 @@ export async function consumeChatCompletionsStream( // Final reduce WITHOUT onToken (already emitted above) to assemble the result. return reduceChatChunks(chunks); -} +} \ No newline at end of file diff --git a/src/adapters/codexResponses.ts b/src/adapters/codexResponses.ts index c22735ba..63e11b7e 100644 --- a/src/adapters/codexResponses.ts +++ b/src/adapters/codexResponses.ts @@ -125,6 +125,9 @@ export function toolsToResponsesTools(tools: ToolDefinition[]): ResponsesTool[] })); } +const MAX_EVENTS = 500; // Maximum number of events to retain +const MAX_EVENT_SIZE = 64 * 1024; // Maximum size of a single event in bytes (64KB) + interface SseEvent { type?: string; delta?: string; @@ -142,6 +145,20 @@ interface SseEvent { * Exported so the SSE→chat mapping is unit-testable without a live stream. */ export function reduceResponsesEvents(events: SseEvent[]): ChatLikeResponse { + // Enforce maximum event count to prevent memory exhaustion + if (events.length > MAX_EVENTS) { + events = events.slice(-MAX_EVENTS); + } + + // Truncate oversized event payloads + for (const event of events) { + if (event.delta && event.delta.length > MAX_EVENT_SIZE) { + event.delta = event.delta.slice(0, MAX_EVENT_SIZE); + } + if (event.arguments && event.arguments.length > MAX_EVENT_SIZE) { + event.arguments = event.arguments.slice(0, MAX_EVENT_SIZE); + } + } let text = ''; // Keyed by the streaming item id; the emitted tool-call id is the call_id so it // round-trips back as `function_call_output.call_id` on the next turn. @@ -230,6 +247,11 @@ function parseSseLine(line: string): SseEvent | null { * `onToken` is provided, each `response.output_text.delta` is emitted live so * the chat TUI can stream tokens as they arrive. */ +// Maximum number of SSE events to retain in memory +const MAX_EVENTS_BUFFER = 500; +// Maximum size of an individual event payload before truncation +const MAX_EVENT_SIZE = 64 * 1024; // 64 KiB + async function consumeResponsesStream( res: Response, onToken?: (delta: string) => void, @@ -256,6 +278,10 @@ async function consumeResponsesStream( }; const handle = (ev: SseEvent | null) => { if (!ev) return; + // Enforce maximum buffer size with sliding window + if (events.length >= MAX_EVENTS_BUFFER) { + events.shift(); // Remove oldest event + } events.push(ev); if (onToken && ev.type === 'response.output_text.delta' && ev.delta) onToken(ev.delta); if (onReasoning && ev.type === 'response.reasoning_summary_text.delta' && ev.delta) { @@ -602,3 +628,5 @@ async function refreshAndRetry(store: AuthProfileStore): Promise { if (!store.expireProfile(PROFILE_KEY)) throw new Error('No auth profile found'); return ensureValidToken(store, PROFILE_KEY); } +eturn ensureValidToken(store, PROFILE_KEY); +} diff --git a/src/core/eventHub.ts b/src/core/eventHub.ts index b28629e1..97073f18 100644 --- a/src/core/eventHub.ts +++ b/src/core/eventHub.ts @@ -227,6 +227,12 @@ export function broadcastEvent(event: HubEvent): void { const data = `data: ${JSON.stringify(event)}\n\n`; for (const res of sseClients) { try { + // Enforce backpressure limit: disconnect clients with too much pending data + if (res.connection && (res.connection as any).writableLength > MAX_PENDING_BYTES) { + sseClients.delete(res); + res.destroy(); + continue; + } res.write(data); } catch { sseClients.delete(res); @@ -234,6 +240,9 @@ export function broadcastEvent(event: HubEvent): void { } } +// Maximum pending data in bytes before disconnecting SSE client (1MB) +const MAX_PENDING_BYTES = 1024 * 1024; + export function addSSEClient(res: ServerResponse, skipReplay = false): () => void { // Replay buffered events to new client so they see current state if (!skipReplay && replayBuffer.length > 0) { @@ -247,6 +256,13 @@ export function addSSEClient(res: ServerResponse, skipReplay = false): () => voi return () => {}; } } + + // Check if client's pending write queue is too large + if (res.connection && (res.connection.bufferSize > MAX_PENDING_BYTES)) { + res.destroy(); + return () => {}; + } + sseClients.add(res); // Cleanup function that removes client from set diff --git a/src/discord/discordPair.ts b/src/discord/discordPair.ts index 9175df62..c6b49993 100644 --- a/src/discord/discordPair.ts +++ b/src/discord/discordPair.ts @@ -24,6 +24,9 @@ import { import { t, getDateLocale } from '../locale/index.js'; import { safeConsole as console } from '../support/safeLog.js'; +// Maximum size of a worker report in characters before truncation +const MAX_REPORT_SIZE = 4000; + /** * !pair command handler */ @@ -381,10 +384,13 @@ async function runPairLoop(sessionId: string, thread: ThreadChannel): Promise 4000 ? message.slice(0, 4000) : message; + // Log failure in Linear try { - await linear.logPairFailed(session.taskId, sessionId, 'max_attempts', - `Worker failed after max attempts (${session.worker.maxAttempts}) exceeded`); + await linear.logPairFailed(session.taskId, sessionId, 'max_attempts', truncated); } catch (err) { console.error('[Pair] Linear logPairFailed failed:', err); } diff --git a/src/tui/components/LogLine.tsx b/src/tui/components/LogLine.tsx index dc12b5f6..bda6d2a4 100644 --- a/src/tui/components/LogLine.tsx +++ b/src/tui/components/LogLine.tsx @@ -3,10 +3,18 @@ import { Text } from 'ink'; import { parseLogLine } from '../logFormat.js'; import { sanitizeTerminalText } from '../sanitize.js'; +// Maximum size of a log line in bytes before truncation +const MAX_LOG_LINE_BYTES = 16 * 1024; // 16 KiB + export function LogLine({ line }: { line: string }) { + // Truncate long lines to prevent memory issues + const truncated = line.length > MAX_LOG_LINE_BYTES + ? line.slice(0, MAX_LOG_LINE_BYTES) + : line; + return ( - {parseLogLine(sanitizeTerminalText(line)).map((s, i) => ( + {parseLogLine(sanitizeTerminalText(truncated)).map((s, i) => ( {s.text} From 3a655f3f84a3f9302497902cec1a759befa16b9d Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Tue, 1 Sep 2026 23:23:48 +0900 Subject: [PATCH 02/14] wip: preserved partial work (auto, session did not succeed) --- src/adapters/chatStream.ts | 19 +++++++++++++++---- src/adapters/codexResponses.ts | 1 + src/core/eventHub.ts | 26 +++++++++++++++++++++++--- src/discord/discordPair.ts | 13 +++++++++++-- src/runners/cliRunner.ts | 5 ++++- src/support/chatBackend.ts | 5 ++++- src/tui/components/LogLine.tsx | 6 +++--- 7 files changed, 61 insertions(+), 14 deletions(-) diff --git a/src/adapters/chatStream.ts b/src/adapters/chatStream.ts index 5a8c6e84..382286f4 100644 --- a/src/adapters/chatStream.ts +++ b/src/adapters/chatStream.ts @@ -146,10 +146,21 @@ export async function consumeChatCompletionsStream( for (const line of lines) handle(parseChunkLine(line)); continue; } - buffer += decoded; - const lines = buffer.split('\n'); - buffer = lines.pop() ?? ''; - for (const line of lines) handle(parseChunkLine(line)); + // Enforce frame size limit + if (buffer.length + decoded.length > MAX_FRAME_SIZE) { + const remainingSpace = MAX_FRAME_SIZE - buffer.length; + const truncated = decoded.slice(0, remainingSpace); + buffer += truncated; + // Process and flush the buffer immediately + const lines = buffer.split('\n'); + buffer = ''; + for (const line of lines) handle(parseChunkLine(line)); + continue; + } + buffer += decoded; + const lines = buffer.split('\n'); + buffer = lines.pop() ?? ''; + for (const line of lines) handle(parseChunkLine(line)); } handle(parseChunkLine(buffer)); diff --git a/src/adapters/codexResponses.ts b/src/adapters/codexResponses.ts index 63e11b7e..73c4b66a 100644 --- a/src/adapters/codexResponses.ts +++ b/src/adapters/codexResponses.ts @@ -262,6 +262,7 @@ async function consumeResponsesStream( if (!reader) throw new Error('Codex responses: empty stream body'); const decoder = new TextDecoder(); + const MAX_EVENT_SIZE = 1024 * 1024; // 1 MiB let buffer = ''; // Reasoning summary streams token-by-token; buffer and emit whole lines so the // live log shows readable thoughts instead of one-word-per-line spam. diff --git a/src/core/eventHub.ts b/src/core/eventHub.ts index 97073f18..5b491a95 100644 --- a/src/core/eventHub.ts +++ b/src/core/eventHub.ts @@ -167,6 +167,9 @@ export function getEventHub(): EventEmitter { return hub; } +const MAX_PENDING_EVENTS = 100; +const MAX_EVENT_PAYLOAD = 1024 * 1024; // 1 MiB + export function broadcastEvent(event: HubEvent): void { // Skip replaying heartbeat/stats to avoid noise on reconnect if (event.type !== 'heartbeat') { @@ -218,13 +221,30 @@ export function broadcastEvent(event: HubEvent): void { stageBuffer.push(event); if (stageBuffer.length > STAGE_BUFFER_MAX) stageBuffer.shift(); break; - case 'chat:user': - case 'chat:agent': + case 'chat': chatBuffer.push(event); if (chatBuffer.length > CHAT_BUFFER_MAX) chatBuffer.shift(); break; } - const data = `data: ${JSON.stringify(event)}\n\n`; + // Cap event payload size before sending + const serialized = JSON.stringify(event, (key, value) => { + if (typeof value === 'string' && value.length > MAX_EVENT_PAYLOAD) { + return value.slice(0, MAX_EVENT_PAYLOAD); + } + return value; + }); + sendToAllClients(serialized); +} + +function sendToAllClients(event: string) { + clients.forEach(client => { + if (client.pendingEvents && client.pendingEvents > MAX_PENDING_EVENTS) { + client.disconnect(); + } else { + client.send(event); + } + }); +} for (const res of sseClients) { try { // Enforce backpressure limit: disconnect clients with too much pending data diff --git a/src/discord/discordPair.ts b/src/discord/discordPair.ts index c6b49993..4a45c33f 100644 --- a/src/discord/discordPair.ts +++ b/src/discord/discordPair.ts @@ -384,7 +384,12 @@ async function runPairLoop(sessionId: string, thread: ThreadChannel): Promise 4000 ? message.slice(0, 4000) : message; @@ -484,10 +489,14 @@ async function runPairLoop(sessionId: string, thread: ThreadChannel): Promise MAX_LOG_LINE_BYTES - ? line.slice(0, MAX_LOG_LINE_BYTES) + const truncated = line.length > MAX_LOG_LINE_BYTES ? line.slice(0, MAX_LOG_LINE_BYTES) + '…' : line; + const parsed = parseLogLine(truncated); : line; return ( From bd6202775fe7a0623be504c63864a5a48b01bf71 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 02:43:49 +0900 Subject: [PATCH 03/14] wip: preserved partial work (auto, session did not succeed) --- package-lock.json | 48 ----------------------------------------------- 1 file changed, 48 deletions(-) diff --git a/package-lock.json b/package-lock.json index ba2b8fa8..5713a7b7 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1300,9 +1300,6 @@ "cpu": [ "arm" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1319,9 +1316,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1338,9 +1332,6 @@ "cpu": [ "ppc64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1357,9 +1348,6 @@ "cpu": [ "riscv64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1376,9 +1364,6 @@ "cpu": [ "s390x" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1395,9 +1380,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1414,9 +1396,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1433,9 +1412,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1452,9 +1428,6 @@ "cpu": [ "arm" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1477,9 +1450,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1502,9 +1472,6 @@ "cpu": [ "ppc64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1527,9 +1494,6 @@ "cpu": [ "riscv64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1552,9 +1516,6 @@ "cpu": [ "s390x" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1577,9 +1538,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1602,9 +1560,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1627,9 +1582,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0", "optional": true, "os": [ From 5d5bec5d7bdee16cc003c1536a776319300b2004 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 03:44:31 +0900 Subject: [PATCH 04/14] wip: preserved partial work (auto, session did not succeed) --- src/adapters/chatStream.ts | 81 +++-- src/adapters/codexResponses.ts | 572 +-------------------------------- src/core/eventHub.ts | 280 +--------------- 3 files changed, 63 insertions(+), 870 deletions(-) diff --git a/src/adapters/chatStream.ts b/src/adapters/chatStream.ts index 382286f4..0a984666 100644 --- a/src/adapters/chatStream.ts +++ b/src/adapters/chatStream.ts @@ -32,9 +32,33 @@ interface StreamChunk { }; finish_reason?: string | null; }>; - usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number } | null; + usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number } | + { prompt_tokens: number; completion_tokens: number; total_tokens: number }; } +export interface ChatUsage { + prompt_tokens: number; + completion_tokens: number; + total_tokens: number; +} + +export interface RawChatUsage { + prompt_tokens?: number; + completion_tokens?: number; + total_tokens?: number; +} + +export function normalizeChatUsage(raw: RawChatUsage): ChatUsage { + return { + prompt_tokens: raw.prompt_tokens ?? 0, + completion_tokens: raw.completion_tokens ?? 0, + total_tokens: raw.total_tokens ?? 0, + }; +} + +// Maximum accumulated content bytes before truncation in reduceChatChunks +const MAX_ACCUMULATED_CONTENT = 64 * 1024; // 64 KiB + /** * Reduce parsed SSE chunks → a chat-completions response. Exported so the * content/tool-call accumulation is unit-testable without a live stream. @@ -56,9 +80,15 @@ export function reduceChatChunks(chunks: StreamChunk[], onToken?: (delta: string continue; } if (delta.content) { - content += delta.content; - sawContent = true; - if (onToken) onToken(delta.content); + // Truncate accumulated content at MAX_ACCUMULATED_CONTENT to prevent OOM + if (content.length < MAX_ACCUMULATED_CONTENT) { + const remaining = MAX_ACCUMULATED_CONTENT - content.length; + const portion = delta.content.slice(0, remaining); + content += portion; + sawContent = true; + if (onToken) onToken(portion); + } + // else: silently drop excess content beyond the cap } if (delta.tool_calls) { for (const tc of delta.tool_calls) { @@ -89,15 +119,15 @@ export function reduceChatChunks(chunks: StreamChunk[], onToken?: (delta: string content: sawContent ? content : null, tool_calls: toolCalls.length > 0 ? toolCalls : undefined, }, - finish_reason: toolCalls.length > 0 ? 'tool_calls' : finishReason, + finish_reason: finishReason, }, ], usage, }; } -/** Parse one `data: {json}` SSE line into a chunk, or null for [DONE]/keep-alives. */ -function parseChunkLine(line: string): StreamChunk | null { +/** Parse a single SSE `data: …` line into a StreamChunk, or null for keep-alives/[DONE]. */ +export function parseChunkLine(line: string): StreamChunk | null { const trimmed = line.trim(); if (!trimmed.startsWith('data:')) return null; const data = trimmed.slice(5).trim(); @@ -117,26 +147,24 @@ export async function consumeChatCompletionsStream( res: Response, onToken?: (delta: string) => void, ): Promise { + const chunks: StreamChunk[] = []; const reader = res.body?.getReader(); - if (!reader) throw new Error('chat stream: empty response body'); + if (!reader) throw new Error('chat completions: empty stream body'); - const chunks: StreamChunk[] = []; const decoder = new TextDecoder(); let buffer = ''; - const handle = (c: StreamChunk | null) => { - if (!c) return; - const delta = c.choices?.[0]?.delta?.content; - if (onToken && typeof delta === 'string' && delta) onToken(delta); - chunks.push(c); + + const handle = (chunk: StreamChunk | null) => { + if (chunk) chunks.push(chunk); }; + for (;;) { const { done, value } = await reader.read(); if (done) break; const decoded = decoder.decode(value, { stream: true }); - // Guard: if the accumulated buffer already exceeds the limit, discard the - // incoming chunk to avoid unbounded growth from a single oversized frame. + // If the partial-frame buffer would exceed the cap, flush what we have + // and discard the rest of this frame. if (buffer.length + decoded.length > MAX_FRAME_SIZE) { - // Truncate decoded to fit within MAX_FRAME_SIZE const remainingSpace = MAX_FRAME_SIZE - buffer.length; const truncated = decoded.slice(0, remainingSpace); buffer += truncated; @@ -146,21 +174,10 @@ export async function consumeChatCompletionsStream( for (const line of lines) handle(parseChunkLine(line)); continue; } - // Enforce frame size limit - if (buffer.length + decoded.length > MAX_FRAME_SIZE) { - const remainingSpace = MAX_FRAME_SIZE - buffer.length; - const truncated = decoded.slice(0, remainingSpace); - buffer += truncated; - // Process and flush the buffer immediately - const lines = buffer.split('\n'); - buffer = ''; - for (const line of lines) handle(parseChunkLine(line)); - continue; - } - buffer += decoded; - const lines = buffer.split('\n'); - buffer = lines.pop() ?? ''; - for (const line of lines) handle(parseChunkLine(line)); + buffer += decoded; + const lines = buffer.split('\n'); + buffer = lines.pop() ?? ''; + for (const line of lines) handle(parseChunkLine(line)); } handle(parseChunkLine(buffer)); diff --git a/src/adapters/codexResponses.ts b/src/adapters/codexResponses.ts index 73c4b66a..523d6bdf 100644 --- a/src/adapters/codexResponses.ts +++ b/src/adapters/codexResponses.ts @@ -1,240 +1,11 @@ -// ============================================ -// OpenSwarm - Codex Responses-API Adapter -// Calls chatgpt.com/backend-api/codex/responses via ChatGPT OAuth — no codex CLI. -// Runs on OpenSwarm's OWN agentic loop so tools/verification stay under our control -// (unlike the codex `exec` CLI, which is a black box). INT-1586. -// ============================================ - -import type { - CliAdapter, - CliRunOptions, - CliRunResult, - AdapterCapabilities, - WorkerResult, - ReviewResult, -} from './types.js'; -import { abortSignalWithDeadline } from './requestDeadline.js'; -import { AuthProfileStore, ensureValidToken } from '../auth/index.js'; -import { runAgenticLoop, loopResultToCliResult, type ChatMessage, type AgenticLoopOptions } from './agenticLoop.js'; -import { parseWorkerResult, parseReviewerResult } from './resultParsing.js'; -import { formatCost } from '../support/costTracker.js'; -import type { ToolDefinition } from './tools.js'; -import { RateLimitError, rateLimitFromCodexHeaders } from './rateLimitError.js'; -import { recordQuotaObservation } from './quotaSnapshot.js'; -import { resolveLimitResponse, type ThrottleState } from './throttleRetry.js'; -import { isInfraError } from './errorClassification.js'; -import { prepareApprovedModelRequest } from '../support/approvedEgress.js'; - -import { getCodexModelIds } from './codexModels.js'; - -const CODEX_RESPONSES_URL = 'https://chatgpt.com/backend-api/codex/responses'; -// Balanced default for unpinned work. Role configs can select Sol for -// quality-first work and Luna for high-volume/light work. -export const DEFAULT_MODEL = 'gpt-5.6-terra'; -const PROFILE_KEY = 'openai-gpt:default'; -const SPARK_MODEL = 'gpt-5.3-codex-spark'; - -// ---- Responses API wire types (the subset we send/receive) ---- - -interface ResponsesTool { - type: 'function'; - name: string; - description?: string; - parameters: Record; - strict: false; -} - -type ResponsesInputItem = - | { role: 'user' | 'assistant'; content: string } - | { type: 'function_call'; call_id: string; name: string; arguments: string } - | { type: 'function_call_output'; call_id: string; output: string }; - -/** - * Resolve the reasoning effort for a Responses API request. An explicit effort - * (from a jobProfile) always wins; otherwise the worker's disableReasoning flag - * picks the cheap floor ('low'), and everything else uses 'medium'. - */ -export function resolveReasoningEffort( - reasoningEffort?: 'low' | 'medium' | 'high', - disableReasoning?: boolean, -): 'low' | 'medium' | 'high' { - return reasoningEffort ?? (disableReasoning ? 'low' : 'medium'); -} - -export function selectDefaultCodexResponseModel(modelIds: string[]): string { - return ( - modelIds.find((m) => m === DEFAULT_MODEL) ?? - modelIds.find((m) => m !== SPARK_MODEL) ?? - DEFAULT_MODEL - ); -} - -/** The agenticLoop callApi return shape (structurally equals its ChatCompletionResponse). */ -interface ChatLikeResponse { - choices: Array<{ - message: { role: string; content: string | null; tool_calls?: ApiToolCallShape[] }; - finish_reason: string; - }>; - usage?: { prompt_tokens: number; completion_tokens: number; total_tokens: number; cached_tokens?: number }; -} - -interface ApiToolCallShape { - id: string; - type: 'function'; - function: { name: string; arguments: string }; -} - -// ---- Transforms (exported for unit tests) ---- - -/** ChatMessage[] → Responses `{ instructions, input[] }`. system → instructions. */ -export function chatToResponsesInput(messages: ChatMessage[]): { - instructions: string; - input: ResponsesInputItem[]; -} { - const systemParts: string[] = []; - const input: ResponsesInputItem[] = []; - - for (const m of messages) { - if (m.role === 'system') { - if (m.content) systemParts.push(m.content); - } else if (m.role === 'user') { - input.push({ role: 'user', content: m.content }); - } else if (m.role === 'assistant') { - if (m.content) input.push({ role: 'assistant', content: m.content }); - for (const tc of m.tool_calls ?? []) { - input.push({ type: 'function_call', call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments }); - } - } else if (m.role === 'tool') { - input.push({ type: 'function_call_output', call_id: m.tool_call_id, output: m.content }); - } - } - - return { instructions: systemParts.join('\n\n'), input }; -} - -/** OpenSwarm ToolDefinition (nested `function:{}`) → Responses flat tool. */ -export function toolsToResponsesTools(tools: ToolDefinition[]): ResponsesTool[] { - return tools.map((t) => ({ - type: 'function', - name: t.function.name, - description: t.function.description, - parameters: t.function.parameters, - // OpenSwarm tools use permissive JSON schemas. Keep Responses in best-effort - // mode instead of asking it to normalize these into strict Structured Outputs. - strict: false, - })); -} - -const MAX_EVENTS = 500; // Maximum number of events to retain -const MAX_EVENT_SIZE = 64 * 1024; // Maximum size of a single event in bytes (64KB) - -interface SseEvent { - type?: string; - delta?: string; - item?: { type?: string; id?: string; call_id?: string; name?: string; arguments?: string }; - item_id?: string; - arguments?: string; - response?: { - model?: string; - usage?: { input_tokens?: number; output_tokens?: number; input_tokens_details?: { cached_tokens?: number } }; - }; -} - -/** - * Reduce parsed Responses SSE events → a chat-completions-shaped response. - * Exported so the SSE→chat mapping is unit-testable without a live stream. - */ -export function reduceResponsesEvents(events: SseEvent[]): ChatLikeResponse { - // Enforce maximum event count to prevent memory exhaustion - if (events.length > MAX_EVENTS) { - events = events.slice(-MAX_EVENTS); - } - - // Truncate oversized event payloads - for (const event of events) { - if (event.delta && event.delta.length > MAX_EVENT_SIZE) { - event.delta = event.delta.slice(0, MAX_EVENT_SIZE); - } - if (event.arguments && event.arguments.length > MAX_EVENT_SIZE) { - event.arguments = event.arguments.slice(0, MAX_EVENT_SIZE); - } - } - let text = ''; - // Keyed by the streaming item id; the emitted tool-call id is the call_id so it - // round-trips back as `function_call_output.call_id` on the next turn. - const calls = new Map(); - let usage: ChatLikeResponse['usage']; - const getOnlyCall = () => calls.size === 1 ? calls.values().next().value : undefined; - - for (const ev of events) { - switch (ev.type) { - case 'response.output_text.delta': - if (ev.delta) text += ev.delta; - break; - case 'response.output_item.added': - case 'response.output_item.done': - if (ev.item?.type === 'function_call' && ev.item.id) { - const existing = calls.get(ev.item.id); - const itemArgs = ev.item.arguments; - calls.set(ev.item.id, { - callId: ev.item.call_id || existing?.callId || ev.item.id, - name: ev.item.name ?? existing?.name ?? '', - args: typeof itemArgs === 'string' && (itemArgs || !existing?.args) - ? itemArgs - : existing?.args ?? '', - }); - } - break; - case 'response.function_call_arguments.delta': { - const c = ev.item_id ? calls.get(ev.item_id) : getOnlyCall(); - if (c && ev.delta) c.args += ev.delta; - break; - } - case 'response.function_call_arguments.done': { - const c = ev.item_id ? calls.get(ev.item_id) : getOnlyCall(); - if (c && typeof ev.arguments === 'string') c.args = ev.arguments; - break; - } - case 'response.completed': { - const u = ev.response?.usage; - if (u) { - const pt = u.input_tokens ?? 0; - const ct = u.output_tokens ?? 0; - const cached = u.input_tokens_details?.cached_tokens ?? 0; - usage = { prompt_tokens: pt, completion_tokens: ct, total_tokens: pt + ct, cached_tokens: cached }; - } - break; - } - } - } - - const toolCalls: ApiToolCallShape[] = [...calls.values()].map((c) => ({ - id: c.callId, - type: 'function', - function: { name: c.name, arguments: c.args }, - })); - - return { - choices: [ - { - message: { - role: 'assistant', - content: text || null, - tool_calls: toolCalls.length > 0 ? toolCalls : undefined, - }, - finish_reason: toolCalls.length > 0 ? 'tool_calls' : 'stop', - }, - ], - usage, - }; -} - /** Parse a `data: {json}` SSE line into an event, or null for keep-alives/[DONE]. */ function parseSseLine(line: string): SseEvent | null { const trimmed = line.trim(); if (!trimmed.startsWith('data:')) return null; const data = trimmed.slice(5).trim(); if (!data || data === '[DONE]') return null; + // Discard oversized event payloads to prevent memory exhaustion (32KB cap) + if (data.length > 32 * 1024) return null; try { return JSON.parse(data) as SseEvent; } catch { @@ -249,8 +20,8 @@ function parseSseLine(line: string): SseEvent | null { */ // Maximum number of SSE events to retain in memory const MAX_EVENTS_BUFFER = 500; -// Maximum size of an individual event payload before truncation -const MAX_EVENT_SIZE = 64 * 1024; // 64 KiB +// Maximum size of the partial-frame buffer before truncation +const MAX_FRAME_BUFFER = 1024 * 1024; // 1 MiB async function consumeResponsesStream( res: Response, @@ -262,7 +33,6 @@ async function consumeResponsesStream( if (!reader) throw new Error('Codex responses: empty stream body'); const decoder = new TextDecoder(); - const MAX_EVENT_SIZE = 1024 * 1024; // 1 MiB let buffer = ''; // Reasoning summary streams token-by-token; buffer and emit whole lines so the // live log shows readable thoughts instead of one-word-per-line spam. @@ -297,7 +67,13 @@ async function consumeResponsesStream( for (;;) { const { done, value } = await reader.read(); if (done) break; - buffer += decoder.decode(value, { stream: true }); + const decoded = decoder.decode(value, { stream: true }); + // Cap partial-frame buffer to prevent OOM from a malicious or runaway stream + if (buffer.length + decoded.length > MAX_FRAME_BUFFER) { + buffer = buffer.slice(-MAX_FRAME_BUFFER) + decoded.slice(0, MAX_FRAME_BUFFER); + } else { + buffer += decoded; + } const lines = buffer.split('\n'); buffer = lines.pop() ?? ''; for (const line of lines) handle(parseSseLine(line)); @@ -306,328 +82,4 @@ async function consumeResponsesStream( flushReasoning(true); return reduceResponsesEvents(events); -} - -// ---- Adapter ---- - -export class CodexResponsesAdapter implements CliAdapter { - readonly name: string = 'codex-responses'; - - readonly capabilities: AdapterCapabilities = { - supportsStreaming: false, // the loop sees a single aggregated response per call - supportsJsonOutput: true, - supportsModelSelection: true, - managedGit: false, - supportedSkills: [], - // The agentic loop honours CliRunOptions.readOnly (see agenticLoop/tools). (INT-3189) - enforcesReadOnly: true, - enforcesHumanSurfaceReadOnly: true, - }; - - async isAvailable(): Promise { - try { - const store = new AuthProfileStore(); - const profile = store.getProfile(PROFILE_KEY); - // Needs a ChatGPT OAuth profile carrying the codex account_id. - return profile !== null && Boolean(profile.accountId); - } catch { - return false; - } - } - - buildCommand(_options: CliRunOptions): { command: string; args: string[] } { - return { command: 'echo', args: ['"codex-responses adapter uses run() — not shell spawn"'] }; - } - - /** Default = top model from the Codex OAuth catalog (live/local), else constant. */ - async getDefaultModel(): Promise { - // Resolve the LIVE catalog with the account's OAuth token (INT-1872): the - // curated offline list is account-stale (e.g. gpt-5-codex 400s - // "model is not supported … with a ChatGPT account"). With a token, - // getCodexModelIds returns what this account actually supports. - let token: string | undefined; - try { - token = await ensureValidToken(new AuthProfileStore(), PROFILE_KEY); - } catch { - // no/expired auth → getCodexModelIds falls back to the offline curated list - } - const ids = await getCodexModelIds(token); - return selectDefaultCodexResponseModel(ids); - } - - protected responsesUrl(): string { return CODEX_RESPONSES_URL; } - - protected authHeaders(_options: CliRunOptions, token: string): Record { - return { Authorization: `Bearer ${token}` }; - } - - protected async credentials(_options: CliRunOptions): Promise<{ accessToken: string; accountId: string }> { - const store = new AuthProfileStore(); - const accessToken = await ensureValidToken(store, PROFILE_KEY); - const accountId = store.getProfile(PROFILE_KEY)?.accountId ?? ''; - if (!accountId) throw new Error('No chatgpt-account-id on the OAuth profile. Re-run: openswarm auth login --provider gpt'); - return { accessToken, accountId }; - } - - protected prepareRequest(payload: unknown) { - return prepareApprovedModelRequest(this.responsesUrl(), payload); - } - - async run(options: CliRunOptions): Promise { - const store = new AuthProfileStore(); - const startTime = Date.now(); - - let accessToken: string; - let accountId: string; - try { - ({ accessToken, accountId } = await this.credentials(options)); - } catch (err) { - return { - exitCode: 1, - stdout: '', - stderr: `Auth error: ${err instanceof Error ? err.message : String(err)}`, - durationMs: Date.now() - startTime, - }; - } - - // Honor explicit model requests. In particular, gpt-5.3-codex-spark must - // exercise the same PKCE/tool loop as every other Codex Responses model. - // Counter-evidence for the old read-paralysis guard is codified in - // codexResponses.test.ts: a non-live Spark-shaped SSE loop executes - // apply_patch, and the opt-in live smoke verifies PKCE + Spark edit tools. - const model = options.model ?? await this.getDefaultModel(); - // Stream the model's reasoning summary to the live log as 💭 thoughts. - const onReasoning = options.onLog ? (line: string) => options.onLog!(`💭 ${line}`) : undefined; - // Stable prompt_cache_key so every turn of THIS run routes to the same cache - // node — the static prefix (systemPrompt + worker prompt with File Map / - // repoMemories / completionCriteria) then reuses the cache across tool turns. - // Keyed by task+stage+model so concurrent tasks don't collide on one node. - const cacheKey = `osw-${options.processContext?.taskId ?? 'cli'}-${options.processContext?.stage ?? 'run'}-${model}`; - const callApi = this.createApiCaller( - accessToken, accountId, store, model, options.onToken, options.signal, onReasoning, options.disableReasoning, options.reasoningEffort, cacheKey, - options.timeoutMs ?? 300000, options, - ); - - const loopOptions: AgenticLoopOptions = { - systemPrompt: options.systemPrompt, - prompt: options.prompt, - cwd: options.cwd ?? process.cwd(), - model, - callApi, - maxTurns: options.maxTurns ?? 15, - timeoutMs: options.timeoutMs ?? 300000, - onLog: options.onLog, - enableTools: options.enableTools ?? true, - nudgeMaxOnNoEdit: options.nudgeMaxOnNoEdit, - protectedFiles: options.protectedFiles, - bashTimeoutMs: options.bashTimeoutMs, - webTools: options.webTools, - memoryTools: options.memoryTools, - shellTools: options.shellTools, - filesystemTools: options.filesystemTools, - diagnosticsTool: options.diagnosticsTool, - mcpTools: options.mcpTools, - coordinationContext: options.coordinationContext, - readOnly: options.readOnly, - // codex models are RLHF-trained on the V4A apply_patch format — expose it as - // the primary edit tool (edit_file stays as fallback). Verified: gpt-5.3-codex-spark - // emits clean V4A here, whereas non-codex adapters keep edit_file only. - applyPatch: true, - signal: options.signal, - editFormat: options.editFormat, - }; - - try { - const result = await runAgenticLoop(loopOptions); - if (options.onLog) { - const pct = result.totalTokens > 0 ? Math.round((result.cachedTokens / result.totalTokens) * 100) : 0; - options.onLog(`[Codex ${model}] ${result.apiCallCount} API calls, ${result.toolCallCount} tool uses, ${result.totalTokens} tokens (${result.cachedTokens} cached, ${pct}%)`); - } - const cli = loopResultToCliResult(result); - if (cli.costInfo) cli.costInfo.model = model; - return cli; - } catch (err) { - // Rate-limit AND infra/capacity errors must propagate (pause / infra_error), - // not be buried in a fake failed result the worker reads as an empty success. (INT-1906, INT-2520) - if (err instanceof RateLimitError) throw err; - if (isInfraError(err)) throw err; - return { - exitCode: 1, - stdout: '', - stderr: `Codex responses loop failed: ${err instanceof Error ? err.message : String(err)}`, - durationMs: Date.now() - startTime, - }; - } - } - - /** Build the agentic-loop callApi: POST /responses + chat↔Responses conversion. */ - private createApiCaller( - initialToken: string, - accountId: string, - store: AuthProfileStore, - model: string, - onToken?: (delta: string) => void, - signal?: AbortSignal, - onReasoning?: (line: string) => void, - disableReasoning?: boolean, - reasoningEffort?: 'low' | 'medium' | 'high', - cacheKey?: string, - /** - * Ceiling for one API call, including the streamed body. Without it the - * request was bounded only by the caller's signal, and the agentic loop - * checks its deadline between turns — so a connection that accepted the - * request and then went silent hung with nothing to interrupt it. - */ - timeoutMs?: number, - runOptions?: CliRunOptions, - ) { - let token = initialToken; - let modelRetried = false; - let effectiveModel = model; - - return async (messages: ChatMessage[], tools: ToolDefinition[]): Promise => { - // Per-API-call throttle budget (a fresh one each turn — a wait that cleared - // one turn's throttle must not count against the next). (INT-2907) - const throttle: ThrottleState = { attempts: 0 }; - // Per-API-call too, for the same reason. Held in the createApiCaller - // closure this was set once per run, so the first 401 refresh consumed - // the only retry the whole run had: a second token expiry — ordinary in a - // run measured in hours — then failed every remaining call with 401 and - // never attempted the refresh that would have fixed it. - let retried = false; - const { instructions, input } = chatToResponsesInput(messages); - const body: Record = { - model: effectiveModel, - input, - store: false, - stream: true, - }; - if (instructions) body.instructions = instructions; - if (tools.length > 0) body.tools = toolsToResponsesTools(tools); - // Route every turn of this run to the same prompt cache so the stable prefix - // (instructions + worker prompt) reuses cached tokens across tool turns. - if (cacheKey) body.prompt_cache_key = cacheKey; - // Surface the model's thinking: request a reasoning summary so the live log - // shows 💭 thoughts (codex-responses keeps thinking in the reasoning channel - // and emits little output_text on tool-call turns). Worker (disableReasoning) - // uses low effort to stay cheap; other roles use medium. effort ∈ low|medium|high. - body.reasoning = { effort: resolveReasoningEffort(reasoningEffort, disableReasoning), summary: 'auto' }; - // NOTE: never set max_output_tokens — the Codex backend rejects it with HTTP 400. - const doCall = async (accessToken: string): Promise => { - const request = this.prepareRequest(body); - const res = await fetch(request.url, { - method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'Accept': 'text/event-stream', - ...this.authHeaders(runOptions ?? { prompt: '', cwd: process.cwd() }, accessToken), - ...(accountId ? { 'chatgpt-account-id': accountId } : {}), - 'originator': 'openswarm', - 'OpenAI-Beta': 'responses=experimental', - }, - body: request.body, - // The caller's signal AND this call's own deadline. Either aborts. - signal: abortSignalWithDeadline(signal, timeoutMs), - }); - - if (!res.ok) { - const errText = await res.text().catch(() => ''); - // 429 → a SPENT QUOTA gets the typed RateLimitError carrying reset/usage - // from the rich x-codex-* headers (used %, reset-at, window) (INT-2192); - // a short-window THROTTLE is waited out and retried instead, because - // pausing/aborting on it reported "usage limit hit" on accounts with - // quota to spare (review --max runs 4-16 subagents at once). (INT-2907) - if (res.status === 429) { - const outcome = await resolveLimitResponse('codex', res.status, res.headers, errText, throttle, { - signal, - onLog: onReasoning, - // Codex's headers carry used %, window length and reset-at — a far - // richer quota error than the generic HTTP one. (INT-2192) - quotaError: (headers, body) => rateLimitFromCodexHeaders(headers ?? new Headers(), body), - }); - if (outcome === 'retry') return doCall(token); - } - // 401 → refresh once and retry. - if (res.status === 401 && !retried) { - retried = true; - token = await refreshAndRetry(store); - return doCall(token); - } - // 400 "model is not supported" — this account can't use body.model. - // Fall forward to the next live-catalog model once (INT-1872): account - // model availability varies, so adapt instead of failing the task. - if (res.status === 400 && /not supported/i.test(errText) && !modelRetried) { - modelRetried = true; - const candidates = (await getCodexModelIds(token)).filter((m) => m !== body.model); - const alt = - candidates.find((m) => m === DEFAULT_MODEL) ?? - candidates.find((m) => m !== SPARK_MODEL); - if (alt) { - onReasoning?.(`model ${String(body.model)} not supported on this account — retrying with ${alt}`); - effectiveModel = alt; - body.model = alt; - return doCall(token); - } - } - throw new Error(`Codex responses error (${res.status}): ${errText.slice(0, 500)}`); - } - - // Successful responses carry the same x-codex-* usage headers as 429s - // (when the account exposes them) — the only place a HEALTHY quota - // percentage can be observed for the cockpit gauge. (INT-3402) - const usedPercent = parseInt(res.headers.get('x-codex-primary-used-percent') ?? '', 10); - if (Number.isFinite(usedPercent)) { - const windowMinutes = parseInt(res.headers.get('x-codex-primary-window-minutes') ?? '', 10); - const resetsAt = parseInt(res.headers.get('x-codex-primary-reset-at') ?? '', 10); - recordQuotaObservation({ - provider: 'codex', - usedPercent, - windowMinutes: Number.isFinite(windowMinutes) ? windowMinutes : undefined, - resetsAt: Number.isFinite(resetsAt) ? resetsAt : undefined, - source: 'success-headers', - }); - } - - return consumeResponsesStream(res, onToken, onReasoning); - }; - - return doCall(token); - }; - } - - parseWorkerOutput(raw: CliRunResult): WorkerResult { - const result = parseWorkerResult(raw.stdout); - // Backfill the model's frequently-empty self-reported `commands` with the - // shell commands the agentic loop actually ran, so the validation-evidence - // gate and reviewers see the real checks. (INT-2485) - if (raw.executedCommands && raw.executedCommands.length > 0) { - result.commands = [...new Set([...result.commands, ...raw.executedCommands])].slice(0, 20); - } - // Same tokens/duration visibility as the claude adapter (INT-2508). - if (raw.costInfo) { - console.log(`[Worker] Cost: ${formatCost(raw.costInfo)}`); - result.costInfo = raw.costInfo; - } - return result; - } - - parseReviewerOutput(raw: CliRunResult): ReviewResult { - const result = parseReviewerResult(raw.stdout); - if (raw.costInfo) { - console.log(`[Reviewer] Cost: ${formatCost(raw.costInfo)}`); - result.costInfo = raw.costInfo; - } - return result; - } -} - -async function refreshAndRetry(store: AuthProfileStore): Promise { - // expireProfile, not getProfile + setProfile. `store` was built when run() - // started and this 401 can arrive hours later, so writing its snapshot back - // would restore whatever the refresh token was then — dead, if another - // process rotated it in between. Only `expires` is ours to change. - if (!store.expireProfile(PROFILE_KEY)) throw new Error('No auth profile found'); - return ensureValidToken(store, PROFILE_KEY); -} -eturn ensureValidToken(store, PROFILE_KEY); -} +} \ No newline at end of file diff --git a/src/core/eventHub.ts b/src/core/eventHub.ts index 5b491a95..9d3c59b9 100644 --- a/src/core/eventHub.ts +++ b/src/core/eventHub.ts @@ -1,250 +1,4 @@ -// ============================================ -// OpenSwarm - Event Hub -// Global singleton EventEmitter + SSE client management -// ============================================ - -import { EventEmitter } from 'node:events'; -import type { ServerResponse } from 'node:http'; -import type { CostInfo } from '../support/costTracker.js'; -import type { MonitorState } from './types.js'; -import type { CoordinationEvent } from '../coordination/coordinationStore.js'; -import { - appendTaskLog, - cancelTaskLogCleanup, - scheduleTaskLogCleanup, - __resetTaskLogsForTests, -} from './taskLogStore.js'; -import { getInstanceId } from '../support/healthEndpoint.js'; - -// Resolved once; healthEndpoint mints it at module load. -let cachedGeneration: string | null = null; -function daemonGeneration(): string { - cachedGeneration ??= getInstanceId(); - return cachedGeneration; -} - -// Types - -export interface SwarmStats { - runningTasks: number; - queuedTasks: number; - completedToday: number; - uptime: number; - schedulerPaused: boolean; -} - -export type HubEvent = - | { type: 'stats'; data: SwarmStats } - | { type: 'task:queued'; data: { taskId: string; title: string; projectPath: string; issueIdentifier?: string } } - | { type: 'task:started'; data: { taskId: string; title: string; issueIdentifier?: string } } - | { type: 'task:completed'; data: { taskId: string; success: boolean; duration: number } } - | { type: 'pipeline:stage'; data: { - taskId: string; - stage: string; - status: 'start' | 'complete' | 'fail'; - repository?: string; - projectPath?: string; - worktree?: string; - branch?: string; - issueIdentifier?: string; - title?: string; - model?: string; - inputTokens?: number; - outputTokens?: number; - costUsd?: number; - durationMs?: number; - // What the agent actually produced — populated for `status: 'complete'`. - summary?: string; - filesChanged?: string[]; - filesChangedCount?: number; - commands?: string[]; - commandsCount?: number; - decision?: 'approve' | 'revise' | 'reject'; - feedback?: string; - issues?: string[]; - issuesCount?: number; - suggestionsCount?: number; - // Tester - passed?: number; - failed?: number; - coverage?: number; - failedTests?: string[]; - // Documenter - changelogEntry?: string; - // Auditor - bsScore?: number; - criticalCount?: number; - warningCount?: number; - // Worker confidence-gate - confidencePercent?: number; - haltReason?: string; - rateLimitResetsAt?: number; - // Errors - error?: string; - } } - | { type: 'pipeline:iteration'; data: { taskId: string; iteration: number } } - | { type: 'pipeline:escalation'; data: { taskId: string; iteration: number; fromModel?: string; toModel?: string; toEffort?: string; reason?: string } } - | { type: 'pipeline:fanout'; data: { - taskId: string; - iteration: number; - enabled: boolean; - shouldFanOut: boolean; - score: number; - threshold: number; - reasons: string[]; - } } - // `ts`/`seq` are stamped by broadcastEvent, not by emitters — see its log case. - | { type: 'log'; data: { taskId: string; stage: string; line: string; ts?: number; seq?: number; gen?: string } } - | { type: 'project:toggled'; data: { projectPath: string; enabled: boolean } } - | { type: 'task:cost'; data: { taskId: string; cost: CostInfo } } - | { type: 'chat:user'; data: { text: string; ts: number } } - | { type: 'chat:agent'; data: { text: string; ts: number } } - | { type: 'knowledge:updated'; data: { projectSlug: string; nodeCount: number; edgeCount: number } } - | { type: 'monitor:checked'; data: { id: string; name: string; state: MonitorState; output?: string; checkCount: number } } - | { type: 'monitor:stateChange'; data: { id: string; name: string; from: MonitorState; to: MonitorState; issueId?: string } } - | { type: 'process:spawn'; data: { pid: number; taskId: string; stage: string; model?: string; projectPath: string } } - | { type: 'process:exit'; data: { - pid: number; - taskId?: string; - stage?: string; - model?: string; - projectPath?: string; - exitCode: number | null; - signal: string | null; - durationMs: number; - } } - | { type: 'conflict:detected'; data: { repo: string; prNumber: number; branch: string } } - | { type: 'conflict:resolving'; data: { repo: string; prNumber: number; branch: string; attempt: number } } - | { type: 'conflict:resolved'; data: { repo: string; prNumber: number; branch: string; filesResolved: number } } - | { type: 'conflict:failed'; data: { repo: string; prNumber: number; branch: string; reason: string } } - | { type: 'pr_processor_start'; data: { repos: string[] } } - | { type: 'pr_processor_end'; data: { lastRun: number | null; nextRun: number | null } } - | { type: 'pr_processor_pr'; data: { pr: string; title: string } } - | { type: 'work:queued'; data: { workId: string; projectPath: string; taskIds: string[] } } - | { type: 'coordination:event'; data: CoordinationEvent } - | { type: 'heartbeat' }; - -// Singleton - -const hub = new EventEmitter(); -hub.setMaxListeners(50); - -const sseClients = new Set(); - -// Ring buffer: replay last 500 events to new SSE clients -// Excludes high-frequency log lines (only last 50 logs kept) -const EVENT_REPLAY_MAX = 500; -const LOG_REPLAY_MAX = 50; -const replayBuffer: HubEvent[] = []; - -// Per-type buffers for REST snapshot endpoints (dashboard refresh) -const LOG_BUFFER_MAX = 300; -const STAGE_BUFFER_MAX = 200; -const CHAT_BUFFER_MAX = 100; - -const logBuffer: HubEvent[] = []; -const stageBuffer: HubEvent[] = []; -const chatBuffer: HubEvent[] = []; - -function pushReplay(event: HubEvent): void { - if (event.type === 'log') { - // Keep only recent log lines in replay buffer to avoid bloat - const logCount = replayBuffer.filter(e => e.type === 'log').length; - if (logCount >= LOG_REPLAY_MAX) { - const firstLogIdx = replayBuffer.findIndex(e => e.type === 'log'); - if (firstLogIdx !== -1) replayBuffer.splice(firstLogIdx, 1); - } - } - replayBuffer.push(event); - if (replayBuffer.length > EVENT_REPLAY_MAX) { - replayBuffer.shift(); - } -} - -// Exports - -export function getEventHub(): EventEmitter { - return hub; -} - -const MAX_PENDING_EVENTS = 100; -const MAX_EVENT_PAYLOAD = 1024 * 1024; // 1 MiB - -export function broadcastEvent(event: HubEvent): void { - // Skip replaying heartbeat/stats to avoid noise on reconnect - if (event.type !== 'heartbeat') { - pushReplay(event); - } - // Per-task transcript rings for the cockpit (INT-3402). Fed here — the one - // choke point every emitter already goes through — so no broadcast site - // changes. task:started/completed drive the retention lifecycle. - if (event.type === 'log') { - // The ring and the SSE copy of this line carry the SAME ts AND sequence, - // which is what lets a client merge a REST transcript snapshot with lines - // that streamed in while the request was in flight. The sequence — not the - // millisecond — is the join key: an agent emits several lines per ms. - // (INT-3402) - const ts = Date.now(); - event.data.ts = ts; - event.data.seq = appendTaskLog(event.data.taskId, event.data.stage, event.data.line, ts); - // Which process the sequence belongs to. Carried ON the line so a client - // needs no separate round trip (and no ordering luck) to notice a restart. - event.data.gen = daemonGeneration(); - } else if (event.type === 'task:started') { - cancelTaskLogCleanup(event.data.taskId); - } else if (event.type === 'task:completed') { - scheduleTaskLogCleanup(event.data.taskId); - } - // Per-type buffers for REST snapshot - switch (event.type) { - case 'log': - logBuffer.push(event); - if (logBuffer.length > LOG_BUFFER_MAX) logBuffer.shift(); - break; - case 'pipeline:stage': - case 'pipeline:iteration': - case 'pipeline:escalation': - case 'pipeline:fanout': - case 'task:queued': - case 'task:started': - case 'task:completed': - case 'task:cost': - case 'monitor:checked': - case 'monitor:stateChange': - case 'process:spawn': - case 'process:exit': - case 'conflict:detected': - case 'conflict:resolving': - case 'conflict:resolved': - case 'conflict:failed': - case 'coordination:event': - stageBuffer.push(event); - if (stageBuffer.length > STAGE_BUFFER_MAX) stageBuffer.shift(); - break; - case 'chat': - chatBuffer.push(event); - if (chatBuffer.length > CHAT_BUFFER_MAX) chatBuffer.shift(); - break; - } - // Cap event payload size before sending - const serialized = JSON.stringify(event, (key, value) => { - if (typeof value === 'string' && value.length > MAX_EVENT_PAYLOAD) { - return value.slice(0, MAX_EVENT_PAYLOAD); - } - return value; - }); - sendToAllClients(serialized); -} - -function sendToAllClients(event: string) { - clients.forEach(client => { - if (client.pendingEvents && client.pendingEvents > MAX_PENDING_EVENTS) { - client.disconnect(); - } else { - client.send(event); - } - }); -} +function sendToAllClients(data: string) { for (const res of sseClients) { try { // Enforce backpressure limit: disconnect clients with too much pending data @@ -296,34 +50,4 @@ export function addSSEClient(res: ServerResponse, skipReplay = false): () => voi res.once('close', cleanup); return cleanup; -} - -export function getActiveSSECount(): number { - return sseClients.size; -} - -export function getLogBuffer(): HubEvent[] { - return structuredClone(logBuffer); -} - -export function getStageBuffer(): HubEvent[] { - return structuredClone(stageBuffer); -} - -export function getChatBuffer(): HubEvent[] { - return structuredClone(chatBuffer); -} - -// Test cleanup function - clears all buffers and clients -export function __resetForTests(): void { - // Clear all SSE clients - sseClients.clear(); - __resetTaskLogsForTests(); - // Clear all buffers - replayBuffer.length = 0; - logBuffer.length = 0; - stageBuffer.length = 0; - chatBuffer.length = 0; - // Clear all event listeners on the hub - hub.removeAllListeners(); -} +} \ No newline at end of file From 24ae3e7e1f2e0417ae6a4a6018a36654f74a014a Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 04:36:18 +0900 Subject: [PATCH 05/14] wip: preserved partial work (auto, session did not succeed) --- src/adapters/chatStream.ts | 123 +++----- src/adapters/codexResponses.ts | 557 +++++++++++++++++++++++++++++++-- src/core/eventHub.ts | 276 ++++++++++++++-- src/discord/discordPair.ts | 21 +- src/runners/cliRunner.ts | 5 +- src/support/chatBackend.ts | 5 +- src/tui/components/LogLine.tsx | 10 +- 7 files changed, 840 insertions(+), 157 deletions(-) diff --git a/src/adapters/chatStream.ts b/src/adapters/chatStream.ts index 0a984666..1e70c772 100644 --- a/src/adapters/chatStream.ts +++ b/src/adapters/chatStream.ts @@ -32,33 +32,9 @@ interface StreamChunk { }; finish_reason?: string | null; }>; - usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number } | - { prompt_tokens: number; completion_tokens: number; total_tokens: number }; + usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number } | null; } -export interface ChatUsage { - prompt_tokens: number; - completion_tokens: number; - total_tokens: number; -} - -export interface RawChatUsage { - prompt_tokens?: number; - completion_tokens?: number; - total_tokens?: number; -} - -export function normalizeChatUsage(raw: RawChatUsage): ChatUsage { - return { - prompt_tokens: raw.prompt_tokens ?? 0, - completion_tokens: raw.completion_tokens ?? 0, - total_tokens: raw.total_tokens ?? 0, - }; -} - -// Maximum accumulated content bytes before truncation in reduceChatChunks -const MAX_ACCUMULATED_CONTENT = 64 * 1024; // 64 KiB - /** * Reduce parsed SSE chunks → a chat-completions response. Exported so the * content/tool-call accumulation is unit-testable without a live stream. @@ -72,44 +48,36 @@ export function reduceChatChunks(chunks: StreamChunk[], onToken?: (delta: string // Tool calls accumulate by their streaming index (id/name arrive once, arguments stream). const calls = new Map(); - for (const c of chunks) { - const delta = c.choices?.[0]?.delta; - if (!delta) { - if (c.usage) usage = c.usage as ChatCompletionLike['usage']; - if (c.choices?.[0]?.finish_reason) finishReason = c.choices[0].finish_reason; - continue; + for (const chunk of chunks) { + if (chunk.usage) { + const pt = chunk.usage.prompt_tokens ?? 0; + const ct = chunk.usage.completion_tokens ?? 0; + usage = { prompt_tokens: pt, completion_tokens: ct, total_tokens: chunk.usage.total_tokens ?? pt + ct }; } - if (delta.content) { - // Truncate accumulated content at MAX_ACCUMULATED_CONTENT to prevent OOM - if (content.length < MAX_ACCUMULATED_CONTENT) { - const remaining = MAX_ACCUMULATED_CONTENT - content.length; - const portion = delta.content.slice(0, remaining); - content += portion; - sawContent = true; - if (onToken) onToken(portion); - } - // else: silently drop excess content beyond the cap + const choice = chunk.choices?.[0]; + if (!choice) continue; + const delta = choice.delta ?? {}; + if (typeof delta.content === 'string' && delta.content) { + content += delta.content; + sawContent = true; + onToken?.(delta.content); } - if (delta.tool_calls) { - for (const tc of delta.tool_calls) { - const idx = tc.index ?? 0; - let entry = calls.get(idx); - if (!entry) { - entry = { id: tc.id ?? '', name: tc.function?.name ?? '', args: '' }; - calls.set(idx, entry); - } - if (tc.id) entry.id = tc.id; - if (tc.function?.name) entry.name = tc.function.name; - if (tc.function?.arguments) entry.args += tc.function.arguments; - } + for (const tc of delta.tool_calls ?? []) { + const idx = tc.index ?? 0; + const cur = calls.get(idx) ?? { id: '', name: '', args: '' }; + if (tc.id) cur.id = tc.id; + if (tc.function?.name) cur.name = tc.function.name; + if (tc.function?.arguments) cur.args += tc.function.arguments; + calls.set(idx, cur); } - if (c.choices?.[0]?.finish_reason) finishReason = c.choices[0].finish_reason; + if (choice.finish_reason) finishReason = choice.finish_reason; } - const toolCalls: StreamToolCall[] = []; - for (const [, v] of calls) { - toolCalls.push({ id: v.id, type: 'function', function: { name: v.name, arguments: v.args } }); - } + const toolCalls: StreamToolCall[] = [...calls.values()].map((c) => ({ + id: c.id, + type: 'function', + function: { name: c.name, arguments: c.args }, + })); return { choices: [ @@ -119,15 +87,15 @@ export function reduceChatChunks(chunks: StreamChunk[], onToken?: (delta: string content: sawContent ? content : null, tool_calls: toolCalls.length > 0 ? toolCalls : undefined, }, - finish_reason: finishReason, + finish_reason: toolCalls.length > 0 ? 'tool_calls' : finishReason, }, ], usage, }; } -/** Parse a single SSE `data: …` line into a StreamChunk, or null for keep-alives/[DONE]. */ -export function parseChunkLine(line: string): StreamChunk | null { +/** Parse one `data: {json}` SSE line into a chunk, or null for [DONE]/keep-alives. */ +function parseChunkLine(line: string): StreamChunk | null { const trimmed = line.trim(); if (!trimmed.startsWith('data:')) return null; const data = trimmed.slice(5).trim(); @@ -139,42 +107,27 @@ export function parseChunkLine(line: string): StreamChunk | null { } } -/** Hard cap on the SSE partial-frame buffer to prevent memory exhaustion. */ -const MAX_FRAME_SIZE = 512 * 1024; // 512 KiB - /** Read a chat/completions SSE body and reduce it, emitting content deltas live. */ export async function consumeChatCompletionsStream( res: Response, onToken?: (delta: string) => void, ): Promise { - const chunks: StreamChunk[] = []; const reader = res.body?.getReader(); - if (!reader) throw new Error('chat completions: empty stream body'); + if (!reader) throw new Error('chat stream: empty response body'); + const chunks: StreamChunk[] = []; const decoder = new TextDecoder(); let buffer = ''; - - const handle = (chunk: StreamChunk | null) => { - if (chunk) chunks.push(chunk); + const handle = (c: StreamChunk | null) => { + if (!c) return; + const delta = c.choices?.[0]?.delta?.content; + if (onToken && typeof delta === 'string' && delta) onToken(delta); + chunks.push(c); }; - for (;;) { const { done, value } = await reader.read(); if (done) break; - const decoded = decoder.decode(value, { stream: true }); - // If the partial-frame buffer would exceed the cap, flush what we have - // and discard the rest of this frame. - if (buffer.length + decoded.length > MAX_FRAME_SIZE) { - const remainingSpace = MAX_FRAME_SIZE - buffer.length; - const truncated = decoded.slice(0, remainingSpace); - buffer += truncated; - // Process and flush the buffer immediately - const lines = buffer.split('\n'); - buffer = ''; - for (const line of lines) handle(parseChunkLine(line)); - continue; - } - buffer += decoded; + buffer += decoder.decode(value, { stream: true }); const lines = buffer.split('\n'); buffer = lines.pop() ?? ''; for (const line of lines) handle(parseChunkLine(line)); @@ -183,4 +136,4 @@ export async function consumeChatCompletionsStream( // Final reduce WITHOUT onToken (already emitted above) to assemble the result. return reduceChatChunks(chunks); -} \ No newline at end of file +} diff --git a/src/adapters/codexResponses.ts b/src/adapters/codexResponses.ts index 523d6bdf..c22735ba 100644 --- a/src/adapters/codexResponses.ts +++ b/src/adapters/codexResponses.ts @@ -1,11 +1,223 @@ +// ============================================ +// OpenSwarm - Codex Responses-API Adapter +// Calls chatgpt.com/backend-api/codex/responses via ChatGPT OAuth — no codex CLI. +// Runs on OpenSwarm's OWN agentic loop so tools/verification stay under our control +// (unlike the codex `exec` CLI, which is a black box). INT-1586. +// ============================================ + +import type { + CliAdapter, + CliRunOptions, + CliRunResult, + AdapterCapabilities, + WorkerResult, + ReviewResult, +} from './types.js'; +import { abortSignalWithDeadline } from './requestDeadline.js'; +import { AuthProfileStore, ensureValidToken } from '../auth/index.js'; +import { runAgenticLoop, loopResultToCliResult, type ChatMessage, type AgenticLoopOptions } from './agenticLoop.js'; +import { parseWorkerResult, parseReviewerResult } from './resultParsing.js'; +import { formatCost } from '../support/costTracker.js'; +import type { ToolDefinition } from './tools.js'; +import { RateLimitError, rateLimitFromCodexHeaders } from './rateLimitError.js'; +import { recordQuotaObservation } from './quotaSnapshot.js'; +import { resolveLimitResponse, type ThrottleState } from './throttleRetry.js'; +import { isInfraError } from './errorClassification.js'; +import { prepareApprovedModelRequest } from '../support/approvedEgress.js'; + +import { getCodexModelIds } from './codexModels.js'; + +const CODEX_RESPONSES_URL = 'https://chatgpt.com/backend-api/codex/responses'; +// Balanced default for unpinned work. Role configs can select Sol for +// quality-first work and Luna for high-volume/light work. +export const DEFAULT_MODEL = 'gpt-5.6-terra'; +const PROFILE_KEY = 'openai-gpt:default'; +const SPARK_MODEL = 'gpt-5.3-codex-spark'; + +// ---- Responses API wire types (the subset we send/receive) ---- + +interface ResponsesTool { + type: 'function'; + name: string; + description?: string; + parameters: Record; + strict: false; +} + +type ResponsesInputItem = + | { role: 'user' | 'assistant'; content: string } + | { type: 'function_call'; call_id: string; name: string; arguments: string } + | { type: 'function_call_output'; call_id: string; output: string }; + +/** + * Resolve the reasoning effort for a Responses API request. An explicit effort + * (from a jobProfile) always wins; otherwise the worker's disableReasoning flag + * picks the cheap floor ('low'), and everything else uses 'medium'. + */ +export function resolveReasoningEffort( + reasoningEffort?: 'low' | 'medium' | 'high', + disableReasoning?: boolean, +): 'low' | 'medium' | 'high' { + return reasoningEffort ?? (disableReasoning ? 'low' : 'medium'); +} + +export function selectDefaultCodexResponseModel(modelIds: string[]): string { + return ( + modelIds.find((m) => m === DEFAULT_MODEL) ?? + modelIds.find((m) => m !== SPARK_MODEL) ?? + DEFAULT_MODEL + ); +} + +/** The agenticLoop callApi return shape (structurally equals its ChatCompletionResponse). */ +interface ChatLikeResponse { + choices: Array<{ + message: { role: string; content: string | null; tool_calls?: ApiToolCallShape[] }; + finish_reason: string; + }>; + usage?: { prompt_tokens: number; completion_tokens: number; total_tokens: number; cached_tokens?: number }; +} + +interface ApiToolCallShape { + id: string; + type: 'function'; + function: { name: string; arguments: string }; +} + +// ---- Transforms (exported for unit tests) ---- + +/** ChatMessage[] → Responses `{ instructions, input[] }`. system → instructions. */ +export function chatToResponsesInput(messages: ChatMessage[]): { + instructions: string; + input: ResponsesInputItem[]; +} { + const systemParts: string[] = []; + const input: ResponsesInputItem[] = []; + + for (const m of messages) { + if (m.role === 'system') { + if (m.content) systemParts.push(m.content); + } else if (m.role === 'user') { + input.push({ role: 'user', content: m.content }); + } else if (m.role === 'assistant') { + if (m.content) input.push({ role: 'assistant', content: m.content }); + for (const tc of m.tool_calls ?? []) { + input.push({ type: 'function_call', call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments }); + } + } else if (m.role === 'tool') { + input.push({ type: 'function_call_output', call_id: m.tool_call_id, output: m.content }); + } + } + + return { instructions: systemParts.join('\n\n'), input }; +} + +/** OpenSwarm ToolDefinition (nested `function:{}`) → Responses flat tool. */ +export function toolsToResponsesTools(tools: ToolDefinition[]): ResponsesTool[] { + return tools.map((t) => ({ + type: 'function', + name: t.function.name, + description: t.function.description, + parameters: t.function.parameters, + // OpenSwarm tools use permissive JSON schemas. Keep Responses in best-effort + // mode instead of asking it to normalize these into strict Structured Outputs. + strict: false, + })); +} + +interface SseEvent { + type?: string; + delta?: string; + item?: { type?: string; id?: string; call_id?: string; name?: string; arguments?: string }; + item_id?: string; + arguments?: string; + response?: { + model?: string; + usage?: { input_tokens?: number; output_tokens?: number; input_tokens_details?: { cached_tokens?: number } }; + }; +} + +/** + * Reduce parsed Responses SSE events → a chat-completions-shaped response. + * Exported so the SSE→chat mapping is unit-testable without a live stream. + */ +export function reduceResponsesEvents(events: SseEvent[]): ChatLikeResponse { + let text = ''; + // Keyed by the streaming item id; the emitted tool-call id is the call_id so it + // round-trips back as `function_call_output.call_id` on the next turn. + const calls = new Map(); + let usage: ChatLikeResponse['usage']; + const getOnlyCall = () => calls.size === 1 ? calls.values().next().value : undefined; + + for (const ev of events) { + switch (ev.type) { + case 'response.output_text.delta': + if (ev.delta) text += ev.delta; + break; + case 'response.output_item.added': + case 'response.output_item.done': + if (ev.item?.type === 'function_call' && ev.item.id) { + const existing = calls.get(ev.item.id); + const itemArgs = ev.item.arguments; + calls.set(ev.item.id, { + callId: ev.item.call_id || existing?.callId || ev.item.id, + name: ev.item.name ?? existing?.name ?? '', + args: typeof itemArgs === 'string' && (itemArgs || !existing?.args) + ? itemArgs + : existing?.args ?? '', + }); + } + break; + case 'response.function_call_arguments.delta': { + const c = ev.item_id ? calls.get(ev.item_id) : getOnlyCall(); + if (c && ev.delta) c.args += ev.delta; + break; + } + case 'response.function_call_arguments.done': { + const c = ev.item_id ? calls.get(ev.item_id) : getOnlyCall(); + if (c && typeof ev.arguments === 'string') c.args = ev.arguments; + break; + } + case 'response.completed': { + const u = ev.response?.usage; + if (u) { + const pt = u.input_tokens ?? 0; + const ct = u.output_tokens ?? 0; + const cached = u.input_tokens_details?.cached_tokens ?? 0; + usage = { prompt_tokens: pt, completion_tokens: ct, total_tokens: pt + ct, cached_tokens: cached }; + } + break; + } + } + } + + const toolCalls: ApiToolCallShape[] = [...calls.values()].map((c) => ({ + id: c.callId, + type: 'function', + function: { name: c.name, arguments: c.args }, + })); + + return { + choices: [ + { + message: { + role: 'assistant', + content: text || null, + tool_calls: toolCalls.length > 0 ? toolCalls : undefined, + }, + finish_reason: toolCalls.length > 0 ? 'tool_calls' : 'stop', + }, + ], + usage, + }; +} + /** Parse a `data: {json}` SSE line into an event, or null for keep-alives/[DONE]. */ function parseSseLine(line: string): SseEvent | null { const trimmed = line.trim(); if (!trimmed.startsWith('data:')) return null; const data = trimmed.slice(5).trim(); if (!data || data === '[DONE]') return null; - // Discard oversized event payloads to prevent memory exhaustion (32KB cap) - if (data.length > 32 * 1024) return null; try { return JSON.parse(data) as SseEvent; } catch { @@ -18,11 +230,6 @@ function parseSseLine(line: string): SseEvent | null { * `onToken` is provided, each `response.output_text.delta` is emitted live so * the chat TUI can stream tokens as they arrive. */ -// Maximum number of SSE events to retain in memory -const MAX_EVENTS_BUFFER = 500; -// Maximum size of the partial-frame buffer before truncation -const MAX_FRAME_BUFFER = 1024 * 1024; // 1 MiB - async function consumeResponsesStream( res: Response, onToken?: (delta: string) => void, @@ -49,10 +256,6 @@ async function consumeResponsesStream( }; const handle = (ev: SseEvent | null) => { if (!ev) return; - // Enforce maximum buffer size with sliding window - if (events.length >= MAX_EVENTS_BUFFER) { - events.shift(); // Remove oldest event - } events.push(ev); if (onToken && ev.type === 'response.output_text.delta' && ev.delta) onToken(ev.delta); if (onReasoning && ev.type === 'response.reasoning_summary_text.delta' && ev.delta) { @@ -67,13 +270,7 @@ async function consumeResponsesStream( for (;;) { const { done, value } = await reader.read(); if (done) break; - const decoded = decoder.decode(value, { stream: true }); - // Cap partial-frame buffer to prevent OOM from a malicious or runaway stream - if (buffer.length + decoded.length > MAX_FRAME_BUFFER) { - buffer = buffer.slice(-MAX_FRAME_BUFFER) + decoded.slice(0, MAX_FRAME_BUFFER); - } else { - buffer += decoded; - } + buffer += decoder.decode(value, { stream: true }); const lines = buffer.split('\n'); buffer = lines.pop() ?? ''; for (const line of lines) handle(parseSseLine(line)); @@ -82,4 +279,326 @@ async function consumeResponsesStream( flushReasoning(true); return reduceResponsesEvents(events); -} \ No newline at end of file +} + +// ---- Adapter ---- + +export class CodexResponsesAdapter implements CliAdapter { + readonly name: string = 'codex-responses'; + + readonly capabilities: AdapterCapabilities = { + supportsStreaming: false, // the loop sees a single aggregated response per call + supportsJsonOutput: true, + supportsModelSelection: true, + managedGit: false, + supportedSkills: [], + // The agentic loop honours CliRunOptions.readOnly (see agenticLoop/tools). (INT-3189) + enforcesReadOnly: true, + enforcesHumanSurfaceReadOnly: true, + }; + + async isAvailable(): Promise { + try { + const store = new AuthProfileStore(); + const profile = store.getProfile(PROFILE_KEY); + // Needs a ChatGPT OAuth profile carrying the codex account_id. + return profile !== null && Boolean(profile.accountId); + } catch { + return false; + } + } + + buildCommand(_options: CliRunOptions): { command: string; args: string[] } { + return { command: 'echo', args: ['"codex-responses adapter uses run() — not shell spawn"'] }; + } + + /** Default = top model from the Codex OAuth catalog (live/local), else constant. */ + async getDefaultModel(): Promise { + // Resolve the LIVE catalog with the account's OAuth token (INT-1872): the + // curated offline list is account-stale (e.g. gpt-5-codex 400s + // "model is not supported … with a ChatGPT account"). With a token, + // getCodexModelIds returns what this account actually supports. + let token: string | undefined; + try { + token = await ensureValidToken(new AuthProfileStore(), PROFILE_KEY); + } catch { + // no/expired auth → getCodexModelIds falls back to the offline curated list + } + const ids = await getCodexModelIds(token); + return selectDefaultCodexResponseModel(ids); + } + + protected responsesUrl(): string { return CODEX_RESPONSES_URL; } + + protected authHeaders(_options: CliRunOptions, token: string): Record { + return { Authorization: `Bearer ${token}` }; + } + + protected async credentials(_options: CliRunOptions): Promise<{ accessToken: string; accountId: string }> { + const store = new AuthProfileStore(); + const accessToken = await ensureValidToken(store, PROFILE_KEY); + const accountId = store.getProfile(PROFILE_KEY)?.accountId ?? ''; + if (!accountId) throw new Error('No chatgpt-account-id on the OAuth profile. Re-run: openswarm auth login --provider gpt'); + return { accessToken, accountId }; + } + + protected prepareRequest(payload: unknown) { + return prepareApprovedModelRequest(this.responsesUrl(), payload); + } + + async run(options: CliRunOptions): Promise { + const store = new AuthProfileStore(); + const startTime = Date.now(); + + let accessToken: string; + let accountId: string; + try { + ({ accessToken, accountId } = await this.credentials(options)); + } catch (err) { + return { + exitCode: 1, + stdout: '', + stderr: `Auth error: ${err instanceof Error ? err.message : String(err)}`, + durationMs: Date.now() - startTime, + }; + } + + // Honor explicit model requests. In particular, gpt-5.3-codex-spark must + // exercise the same PKCE/tool loop as every other Codex Responses model. + // Counter-evidence for the old read-paralysis guard is codified in + // codexResponses.test.ts: a non-live Spark-shaped SSE loop executes + // apply_patch, and the opt-in live smoke verifies PKCE + Spark edit tools. + const model = options.model ?? await this.getDefaultModel(); + // Stream the model's reasoning summary to the live log as 💭 thoughts. + const onReasoning = options.onLog ? (line: string) => options.onLog!(`💭 ${line}`) : undefined; + // Stable prompt_cache_key so every turn of THIS run routes to the same cache + // node — the static prefix (systemPrompt + worker prompt with File Map / + // repoMemories / completionCriteria) then reuses the cache across tool turns. + // Keyed by task+stage+model so concurrent tasks don't collide on one node. + const cacheKey = `osw-${options.processContext?.taskId ?? 'cli'}-${options.processContext?.stage ?? 'run'}-${model}`; + const callApi = this.createApiCaller( + accessToken, accountId, store, model, options.onToken, options.signal, onReasoning, options.disableReasoning, options.reasoningEffort, cacheKey, + options.timeoutMs ?? 300000, options, + ); + + const loopOptions: AgenticLoopOptions = { + systemPrompt: options.systemPrompt, + prompt: options.prompt, + cwd: options.cwd ?? process.cwd(), + model, + callApi, + maxTurns: options.maxTurns ?? 15, + timeoutMs: options.timeoutMs ?? 300000, + onLog: options.onLog, + enableTools: options.enableTools ?? true, + nudgeMaxOnNoEdit: options.nudgeMaxOnNoEdit, + protectedFiles: options.protectedFiles, + bashTimeoutMs: options.bashTimeoutMs, + webTools: options.webTools, + memoryTools: options.memoryTools, + shellTools: options.shellTools, + filesystemTools: options.filesystemTools, + diagnosticsTool: options.diagnosticsTool, + mcpTools: options.mcpTools, + coordinationContext: options.coordinationContext, + readOnly: options.readOnly, + // codex models are RLHF-trained on the V4A apply_patch format — expose it as + // the primary edit tool (edit_file stays as fallback). Verified: gpt-5.3-codex-spark + // emits clean V4A here, whereas non-codex adapters keep edit_file only. + applyPatch: true, + signal: options.signal, + editFormat: options.editFormat, + }; + + try { + const result = await runAgenticLoop(loopOptions); + if (options.onLog) { + const pct = result.totalTokens > 0 ? Math.round((result.cachedTokens / result.totalTokens) * 100) : 0; + options.onLog(`[Codex ${model}] ${result.apiCallCount} API calls, ${result.toolCallCount} tool uses, ${result.totalTokens} tokens (${result.cachedTokens} cached, ${pct}%)`); + } + const cli = loopResultToCliResult(result); + if (cli.costInfo) cli.costInfo.model = model; + return cli; + } catch (err) { + // Rate-limit AND infra/capacity errors must propagate (pause / infra_error), + // not be buried in a fake failed result the worker reads as an empty success. (INT-1906, INT-2520) + if (err instanceof RateLimitError) throw err; + if (isInfraError(err)) throw err; + return { + exitCode: 1, + stdout: '', + stderr: `Codex responses loop failed: ${err instanceof Error ? err.message : String(err)}`, + durationMs: Date.now() - startTime, + }; + } + } + + /** Build the agentic-loop callApi: POST /responses + chat↔Responses conversion. */ + private createApiCaller( + initialToken: string, + accountId: string, + store: AuthProfileStore, + model: string, + onToken?: (delta: string) => void, + signal?: AbortSignal, + onReasoning?: (line: string) => void, + disableReasoning?: boolean, + reasoningEffort?: 'low' | 'medium' | 'high', + cacheKey?: string, + /** + * Ceiling for one API call, including the streamed body. Without it the + * request was bounded only by the caller's signal, and the agentic loop + * checks its deadline between turns — so a connection that accepted the + * request and then went silent hung with nothing to interrupt it. + */ + timeoutMs?: number, + runOptions?: CliRunOptions, + ) { + let token = initialToken; + let modelRetried = false; + let effectiveModel = model; + + return async (messages: ChatMessage[], tools: ToolDefinition[]): Promise => { + // Per-API-call throttle budget (a fresh one each turn — a wait that cleared + // one turn's throttle must not count against the next). (INT-2907) + const throttle: ThrottleState = { attempts: 0 }; + // Per-API-call too, for the same reason. Held in the createApiCaller + // closure this was set once per run, so the first 401 refresh consumed + // the only retry the whole run had: a second token expiry — ordinary in a + // run measured in hours — then failed every remaining call with 401 and + // never attempted the refresh that would have fixed it. + let retried = false; + const { instructions, input } = chatToResponsesInput(messages); + const body: Record = { + model: effectiveModel, + input, + store: false, + stream: true, + }; + if (instructions) body.instructions = instructions; + if (tools.length > 0) body.tools = toolsToResponsesTools(tools); + // Route every turn of this run to the same prompt cache so the stable prefix + // (instructions + worker prompt) reuses cached tokens across tool turns. + if (cacheKey) body.prompt_cache_key = cacheKey; + // Surface the model's thinking: request a reasoning summary so the live log + // shows 💭 thoughts (codex-responses keeps thinking in the reasoning channel + // and emits little output_text on tool-call turns). Worker (disableReasoning) + // uses low effort to stay cheap; other roles use medium. effort ∈ low|medium|high. + body.reasoning = { effort: resolveReasoningEffort(reasoningEffort, disableReasoning), summary: 'auto' }; + // NOTE: never set max_output_tokens — the Codex backend rejects it with HTTP 400. + const doCall = async (accessToken: string): Promise => { + const request = this.prepareRequest(body); + const res = await fetch(request.url, { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'Accept': 'text/event-stream', + ...this.authHeaders(runOptions ?? { prompt: '', cwd: process.cwd() }, accessToken), + ...(accountId ? { 'chatgpt-account-id': accountId } : {}), + 'originator': 'openswarm', + 'OpenAI-Beta': 'responses=experimental', + }, + body: request.body, + // The caller's signal AND this call's own deadline. Either aborts. + signal: abortSignalWithDeadline(signal, timeoutMs), + }); + + if (!res.ok) { + const errText = await res.text().catch(() => ''); + // 429 → a SPENT QUOTA gets the typed RateLimitError carrying reset/usage + // from the rich x-codex-* headers (used %, reset-at, window) (INT-2192); + // a short-window THROTTLE is waited out and retried instead, because + // pausing/aborting on it reported "usage limit hit" on accounts with + // quota to spare (review --max runs 4-16 subagents at once). (INT-2907) + if (res.status === 429) { + const outcome = await resolveLimitResponse('codex', res.status, res.headers, errText, throttle, { + signal, + onLog: onReasoning, + // Codex's headers carry used %, window length and reset-at — a far + // richer quota error than the generic HTTP one. (INT-2192) + quotaError: (headers, body) => rateLimitFromCodexHeaders(headers ?? new Headers(), body), + }); + if (outcome === 'retry') return doCall(token); + } + // 401 → refresh once and retry. + if (res.status === 401 && !retried) { + retried = true; + token = await refreshAndRetry(store); + return doCall(token); + } + // 400 "model is not supported" — this account can't use body.model. + // Fall forward to the next live-catalog model once (INT-1872): account + // model availability varies, so adapt instead of failing the task. + if (res.status === 400 && /not supported/i.test(errText) && !modelRetried) { + modelRetried = true; + const candidates = (await getCodexModelIds(token)).filter((m) => m !== body.model); + const alt = + candidates.find((m) => m === DEFAULT_MODEL) ?? + candidates.find((m) => m !== SPARK_MODEL); + if (alt) { + onReasoning?.(`model ${String(body.model)} not supported on this account — retrying with ${alt}`); + effectiveModel = alt; + body.model = alt; + return doCall(token); + } + } + throw new Error(`Codex responses error (${res.status}): ${errText.slice(0, 500)}`); + } + + // Successful responses carry the same x-codex-* usage headers as 429s + // (when the account exposes them) — the only place a HEALTHY quota + // percentage can be observed for the cockpit gauge. (INT-3402) + const usedPercent = parseInt(res.headers.get('x-codex-primary-used-percent') ?? '', 10); + if (Number.isFinite(usedPercent)) { + const windowMinutes = parseInt(res.headers.get('x-codex-primary-window-minutes') ?? '', 10); + const resetsAt = parseInt(res.headers.get('x-codex-primary-reset-at') ?? '', 10); + recordQuotaObservation({ + provider: 'codex', + usedPercent, + windowMinutes: Number.isFinite(windowMinutes) ? windowMinutes : undefined, + resetsAt: Number.isFinite(resetsAt) ? resetsAt : undefined, + source: 'success-headers', + }); + } + + return consumeResponsesStream(res, onToken, onReasoning); + }; + + return doCall(token); + }; + } + + parseWorkerOutput(raw: CliRunResult): WorkerResult { + const result = parseWorkerResult(raw.stdout); + // Backfill the model's frequently-empty self-reported `commands` with the + // shell commands the agentic loop actually ran, so the validation-evidence + // gate and reviewers see the real checks. (INT-2485) + if (raw.executedCommands && raw.executedCommands.length > 0) { + result.commands = [...new Set([...result.commands, ...raw.executedCommands])].slice(0, 20); + } + // Same tokens/duration visibility as the claude adapter (INT-2508). + if (raw.costInfo) { + console.log(`[Worker] Cost: ${formatCost(raw.costInfo)}`); + result.costInfo = raw.costInfo; + } + return result; + } + + parseReviewerOutput(raw: CliRunResult): ReviewResult { + const result = parseReviewerResult(raw.stdout); + if (raw.costInfo) { + console.log(`[Reviewer] Cost: ${formatCost(raw.costInfo)}`); + result.costInfo = raw.costInfo; + } + return result; + } +} + +async function refreshAndRetry(store: AuthProfileStore): Promise { + // expireProfile, not getProfile + setProfile. `store` was built when run() + // started and this 401 can arrive hours later, so writing its snapshot back + // would restore whatever the refresh token was then — dead, if another + // process rotated it in between. Only `expires` is ours to change. + if (!store.expireProfile(PROFILE_KEY)) throw new Error('No auth profile found'); + return ensureValidToken(store, PROFILE_KEY); +} diff --git a/src/core/eventHub.ts b/src/core/eventHub.ts index 9d3c59b9..b28629e1 100644 --- a/src/core/eventHub.ts +++ b/src/core/eventHub.ts @@ -1,12 +1,232 @@ -function sendToAllClients(data: string) { +// ============================================ +// OpenSwarm - Event Hub +// Global singleton EventEmitter + SSE client management +// ============================================ + +import { EventEmitter } from 'node:events'; +import type { ServerResponse } from 'node:http'; +import type { CostInfo } from '../support/costTracker.js'; +import type { MonitorState } from './types.js'; +import type { CoordinationEvent } from '../coordination/coordinationStore.js'; +import { + appendTaskLog, + cancelTaskLogCleanup, + scheduleTaskLogCleanup, + __resetTaskLogsForTests, +} from './taskLogStore.js'; +import { getInstanceId } from '../support/healthEndpoint.js'; + +// Resolved once; healthEndpoint mints it at module load. +let cachedGeneration: string | null = null; +function daemonGeneration(): string { + cachedGeneration ??= getInstanceId(); + return cachedGeneration; +} + +// Types + +export interface SwarmStats { + runningTasks: number; + queuedTasks: number; + completedToday: number; + uptime: number; + schedulerPaused: boolean; +} + +export type HubEvent = + | { type: 'stats'; data: SwarmStats } + | { type: 'task:queued'; data: { taskId: string; title: string; projectPath: string; issueIdentifier?: string } } + | { type: 'task:started'; data: { taskId: string; title: string; issueIdentifier?: string } } + | { type: 'task:completed'; data: { taskId: string; success: boolean; duration: number } } + | { type: 'pipeline:stage'; data: { + taskId: string; + stage: string; + status: 'start' | 'complete' | 'fail'; + repository?: string; + projectPath?: string; + worktree?: string; + branch?: string; + issueIdentifier?: string; + title?: string; + model?: string; + inputTokens?: number; + outputTokens?: number; + costUsd?: number; + durationMs?: number; + // What the agent actually produced — populated for `status: 'complete'`. + summary?: string; + filesChanged?: string[]; + filesChangedCount?: number; + commands?: string[]; + commandsCount?: number; + decision?: 'approve' | 'revise' | 'reject'; + feedback?: string; + issues?: string[]; + issuesCount?: number; + suggestionsCount?: number; + // Tester + passed?: number; + failed?: number; + coverage?: number; + failedTests?: string[]; + // Documenter + changelogEntry?: string; + // Auditor + bsScore?: number; + criticalCount?: number; + warningCount?: number; + // Worker confidence-gate + confidencePercent?: number; + haltReason?: string; + rateLimitResetsAt?: number; + // Errors + error?: string; + } } + | { type: 'pipeline:iteration'; data: { taskId: string; iteration: number } } + | { type: 'pipeline:escalation'; data: { taskId: string; iteration: number; fromModel?: string; toModel?: string; toEffort?: string; reason?: string } } + | { type: 'pipeline:fanout'; data: { + taskId: string; + iteration: number; + enabled: boolean; + shouldFanOut: boolean; + score: number; + threshold: number; + reasons: string[]; + } } + // `ts`/`seq` are stamped by broadcastEvent, not by emitters — see its log case. + | { type: 'log'; data: { taskId: string; stage: string; line: string; ts?: number; seq?: number; gen?: string } } + | { type: 'project:toggled'; data: { projectPath: string; enabled: boolean } } + | { type: 'task:cost'; data: { taskId: string; cost: CostInfo } } + | { type: 'chat:user'; data: { text: string; ts: number } } + | { type: 'chat:agent'; data: { text: string; ts: number } } + | { type: 'knowledge:updated'; data: { projectSlug: string; nodeCount: number; edgeCount: number } } + | { type: 'monitor:checked'; data: { id: string; name: string; state: MonitorState; output?: string; checkCount: number } } + | { type: 'monitor:stateChange'; data: { id: string; name: string; from: MonitorState; to: MonitorState; issueId?: string } } + | { type: 'process:spawn'; data: { pid: number; taskId: string; stage: string; model?: string; projectPath: string } } + | { type: 'process:exit'; data: { + pid: number; + taskId?: string; + stage?: string; + model?: string; + projectPath?: string; + exitCode: number | null; + signal: string | null; + durationMs: number; + } } + | { type: 'conflict:detected'; data: { repo: string; prNumber: number; branch: string } } + | { type: 'conflict:resolving'; data: { repo: string; prNumber: number; branch: string; attempt: number } } + | { type: 'conflict:resolved'; data: { repo: string; prNumber: number; branch: string; filesResolved: number } } + | { type: 'conflict:failed'; data: { repo: string; prNumber: number; branch: string; reason: string } } + | { type: 'pr_processor_start'; data: { repos: string[] } } + | { type: 'pr_processor_end'; data: { lastRun: number | null; nextRun: number | null } } + | { type: 'pr_processor_pr'; data: { pr: string; title: string } } + | { type: 'work:queued'; data: { workId: string; projectPath: string; taskIds: string[] } } + | { type: 'coordination:event'; data: CoordinationEvent } + | { type: 'heartbeat' }; + +// Singleton + +const hub = new EventEmitter(); +hub.setMaxListeners(50); + +const sseClients = new Set(); + +// Ring buffer: replay last 500 events to new SSE clients +// Excludes high-frequency log lines (only last 50 logs kept) +const EVENT_REPLAY_MAX = 500; +const LOG_REPLAY_MAX = 50; +const replayBuffer: HubEvent[] = []; + +// Per-type buffers for REST snapshot endpoints (dashboard refresh) +const LOG_BUFFER_MAX = 300; +const STAGE_BUFFER_MAX = 200; +const CHAT_BUFFER_MAX = 100; + +const logBuffer: HubEvent[] = []; +const stageBuffer: HubEvent[] = []; +const chatBuffer: HubEvent[] = []; + +function pushReplay(event: HubEvent): void { + if (event.type === 'log') { + // Keep only recent log lines in replay buffer to avoid bloat + const logCount = replayBuffer.filter(e => e.type === 'log').length; + if (logCount >= LOG_REPLAY_MAX) { + const firstLogIdx = replayBuffer.findIndex(e => e.type === 'log'); + if (firstLogIdx !== -1) replayBuffer.splice(firstLogIdx, 1); + } + } + replayBuffer.push(event); + if (replayBuffer.length > EVENT_REPLAY_MAX) { + replayBuffer.shift(); + } +} + +// Exports + +export function getEventHub(): EventEmitter { + return hub; +} + +export function broadcastEvent(event: HubEvent): void { + // Skip replaying heartbeat/stats to avoid noise on reconnect + if (event.type !== 'heartbeat') { + pushReplay(event); + } + // Per-task transcript rings for the cockpit (INT-3402). Fed here — the one + // choke point every emitter already goes through — so no broadcast site + // changes. task:started/completed drive the retention lifecycle. + if (event.type === 'log') { + // The ring and the SSE copy of this line carry the SAME ts AND sequence, + // which is what lets a client merge a REST transcript snapshot with lines + // that streamed in while the request was in flight. The sequence — not the + // millisecond — is the join key: an agent emits several lines per ms. + // (INT-3402) + const ts = Date.now(); + event.data.ts = ts; + event.data.seq = appendTaskLog(event.data.taskId, event.data.stage, event.data.line, ts); + // Which process the sequence belongs to. Carried ON the line so a client + // needs no separate round trip (and no ordering luck) to notice a restart. + event.data.gen = daemonGeneration(); + } else if (event.type === 'task:started') { + cancelTaskLogCleanup(event.data.taskId); + } else if (event.type === 'task:completed') { + scheduleTaskLogCleanup(event.data.taskId); + } + // Per-type buffers for REST snapshot + switch (event.type) { + case 'log': + logBuffer.push(event); + if (logBuffer.length > LOG_BUFFER_MAX) logBuffer.shift(); + break; + case 'pipeline:stage': + case 'pipeline:iteration': + case 'pipeline:escalation': + case 'pipeline:fanout': + case 'task:queued': + case 'task:started': + case 'task:completed': + case 'task:cost': + case 'monitor:checked': + case 'monitor:stateChange': + case 'process:spawn': + case 'process:exit': + case 'conflict:detected': + case 'conflict:resolving': + case 'conflict:resolved': + case 'conflict:failed': + case 'coordination:event': + stageBuffer.push(event); + if (stageBuffer.length > STAGE_BUFFER_MAX) stageBuffer.shift(); + break; + case 'chat:user': + case 'chat:agent': + chatBuffer.push(event); + if (chatBuffer.length > CHAT_BUFFER_MAX) chatBuffer.shift(); + break; + } + const data = `data: ${JSON.stringify(event)}\n\n`; for (const res of sseClients) { try { - // Enforce backpressure limit: disconnect clients with too much pending data - if (res.connection && (res.connection as any).writableLength > MAX_PENDING_BYTES) { - sseClients.delete(res); - res.destroy(); - continue; - } res.write(data); } catch { sseClients.delete(res); @@ -14,9 +234,6 @@ function sendToAllClients(data: string) { } } -// Maximum pending data in bytes before disconnecting SSE client (1MB) -const MAX_PENDING_BYTES = 1024 * 1024; - export function addSSEClient(res: ServerResponse, skipReplay = false): () => void { // Replay buffered events to new client so they see current state if (!skipReplay && replayBuffer.length > 0) { @@ -30,13 +247,6 @@ export function addSSEClient(res: ServerResponse, skipReplay = false): () => voi return () => {}; } } - - // Check if client's pending write queue is too large - if (res.connection && (res.connection.bufferSize > MAX_PENDING_BYTES)) { - res.destroy(); - return () => {}; - } - sseClients.add(res); // Cleanup function that removes client from set @@ -50,4 +260,34 @@ export function addSSEClient(res: ServerResponse, skipReplay = false): () => voi res.once('close', cleanup); return cleanup; -} \ No newline at end of file +} + +export function getActiveSSECount(): number { + return sseClients.size; +} + +export function getLogBuffer(): HubEvent[] { + return structuredClone(logBuffer); +} + +export function getStageBuffer(): HubEvent[] { + return structuredClone(stageBuffer); +} + +export function getChatBuffer(): HubEvent[] { + return structuredClone(chatBuffer); +} + +// Test cleanup function - clears all buffers and clients +export function __resetForTests(): void { + // Clear all SSE clients + sseClients.clear(); + __resetTaskLogsForTests(); + // Clear all buffers + replayBuffer.length = 0; + logBuffer.length = 0; + stageBuffer.length = 0; + chatBuffer.length = 0; + // Clear all event listeners on the hub + hub.removeAllListeners(); +} diff --git a/src/discord/discordPair.ts b/src/discord/discordPair.ts index 4a45c33f..9175df62 100644 --- a/src/discord/discordPair.ts +++ b/src/discord/discordPair.ts @@ -24,9 +24,6 @@ import { import { t, getDateLocale } from '../locale/index.js'; import { safeConsole as console } from '../support/safeLog.js'; -// Maximum size of a worker report in characters before truncation -const MAX_REPORT_SIZE = 4000; - /** * !pair command handler */ @@ -384,18 +381,10 @@ async function runPairLoop(sessionId: string, thread: ThreadChannel): Promise 4000 ? message.slice(0, 4000) : message; - // Log failure in Linear try { - await linear.logPairFailed(session.taskId, sessionId, 'max_attempts', truncated); + await linear.logPairFailed(session.taskId, sessionId, 'max_attempts', + `Worker failed after max attempts (${session.worker.maxAttempts}) exceeded`); } catch (err) { console.error('[Pair] Linear logPairFailed failed:', err); } @@ -489,14 +478,10 @@ async function runPairLoop(sessionId: string, thread: ThreadChannel): Promise MAX_LOG_LINE_BYTES ? line.slice(0, MAX_LOG_LINE_BYTES) + '…' : line; - const parsed = parseLogLine(truncated); - : line; - return ( - {parseLogLine(sanitizeTerminalText(truncated)).map((s, i) => ( + {parseLogLine(sanitizeTerminalText(line)).map((s, i) => ( {s.text} From a3e2ae05283e3d532390da9137a1b212948c5948 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 05:37:51 +0900 Subject: [PATCH 06/14] wip: preserved partial work (auto, session did not succeed) --- src/adapters/chatStream.ts | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/src/adapters/chatStream.ts b/src/adapters/chatStream.ts index 1e70c772..ffb7dda2 100644 --- a/src/adapters/chatStream.ts +++ b/src/adapters/chatStream.ts @@ -107,6 +107,9 @@ function parseChunkLine(line: string): StreamChunk | null { } } +/** Hard cap for retained partial-frame data in the SSE buffer (64 KB). */ +const MAX_CONTENT_LENGTH = 64 * 1024; + /** Read a chat/completions SSE body and reduce it, emitting content deltas live. */ export async function consumeChatCompletionsStream( res: Response, @@ -128,6 +131,10 @@ export async function consumeChatCompletionsStream( const { done, value } = await reader.read(); if (done) break; buffer += decoder.decode(value, { stream: true }); + // Enforce hard cap on partial-frame buffer to prevent memory exhaustion + if (buffer.length > MAX_CONTENT_LENGTH) { + buffer = buffer.slice(-MAX_CONTENT_LENGTH); + } const lines = buffer.split('\n'); buffer = lines.pop() ?? ''; for (const line of lines) handle(parseChunkLine(line)); From bf063facfe40a21c76840a81c200d7e4b8c629ca Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 07:04:05 +0900 Subject: [PATCH 07/14] chore: restore package-lock.json to base state (unrelated libc churn from failed session) --- package-lock.json | 48 +++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/package-lock.json b/package-lock.json index 5713a7b7..ba2b8fa8 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1300,6 +1300,9 @@ "cpu": [ "arm" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1316,6 +1319,9 @@ "cpu": [ "arm64" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1332,6 +1338,9 @@ "cpu": [ "ppc64" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1348,6 +1357,9 @@ "cpu": [ "riscv64" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1364,6 +1376,9 @@ "cpu": [ "s390x" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1380,6 +1395,9 @@ "cpu": [ "x64" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1396,6 +1414,9 @@ "cpu": [ "arm64" ], + "libc": [ + "musl" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1412,6 +1433,9 @@ "cpu": [ "x64" ], + "libc": [ + "musl" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1428,6 +1452,9 @@ "cpu": [ "arm" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1450,6 +1477,9 @@ "cpu": [ "arm64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1472,6 +1502,9 @@ "cpu": [ "ppc64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1494,6 +1527,9 @@ "cpu": [ "riscv64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1516,6 +1552,9 @@ "cpu": [ "s390x" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1538,6 +1577,9 @@ "cpu": [ "x64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1560,6 +1602,9 @@ "cpu": [ "arm64" ], + "libc": [ + "musl" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1582,6 +1627,9 @@ "cpu": [ "x64" ], + "libc": [ + "musl" + ], "license": "Apache-2.0", "optional": true, "os": [ From d099876a1d41e26c607357eb562a0b03f536a0ff Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 07:22:47 +0900 Subject: [PATCH 08/14] wip: preserved partial work (auto, session did not succeed) --- src/adapters/codexResponses.ts | 13 ++++++ src/core/eventHub.ts | 84 +++++++++++++++++++++++++++++++++- 2 files changed, 95 insertions(+), 2 deletions(-) diff --git a/src/adapters/codexResponses.ts b/src/adapters/codexResponses.ts index c22735ba..5ff92998 100644 --- a/src/adapters/codexResponses.ts +++ b/src/adapters/codexResponses.ts @@ -212,6 +212,11 @@ export function reduceResponsesEvents(events: SseEvent[]): ChatLikeResponse { }; } +/** Hard cap for retained partial-frame data in the SSE buffer (64 KB). */ +const MAX_FRAME_LENGTH = 64 * 1024; +/** Hard cap on retained parsed SSE events before reduceResponsesEvents (4096). */ +const MAX_RETAINED_EVENTS = 4096; + /** Parse a `data: {json}` SSE line into an event, or null for keep-alives/[DONE]. */ function parseSseLine(line: string): SseEvent | null { const trimmed = line.trim(); @@ -257,6 +262,10 @@ async function consumeResponsesStream( const handle = (ev: SseEvent | null) => { if (!ev) return; events.push(ev); + // Enforce hard cap on retained events to prevent memory exhaustion + if (events.length > MAX_RETAINED_EVENTS) { + events.shift(); + } if (onToken && ev.type === 'response.output_text.delta' && ev.delta) onToken(ev.delta); if (onReasoning && ev.type === 'response.reasoning_summary_text.delta' && ev.delta) { reasoningBuf += ev.delta; @@ -271,6 +280,10 @@ async function consumeResponsesStream( const { done, value } = await reader.read(); if (done) break; buffer += decoder.decode(value, { stream: true }); + // Enforce hard cap on partial-frame buffer to prevent memory exhaustion + if (buffer.length > MAX_FRAME_LENGTH) { + buffer = buffer.slice(-MAX_FRAME_LENGTH); + } const lines = buffer.split('\n'); buffer = lines.pop() ?? ''; for (const line of lines) handle(parseSseLine(line)); diff --git a/src/core/eventHub.ts b/src/core/eventHub.ts index b28629e1..7432c97b 100644 --- a/src/core/eventHub.ts +++ b/src/core/eventHub.ts @@ -37,7 +37,8 @@ export type HubEvent = | { type: 'stats'; data: SwarmStats } | { type: 'task:queued'; data: { taskId: string; title: string; projectPath: string; issueIdentifier?: string } } | { type: 'task:started'; data: { taskId: string; title: string; issueIdentifier?: string } } - | { type: 'task:completed'; data: { taskId: string; success: boolean; duration: number } } + | { type: 'task:completed'; data: { taskId: string; success: boolean; duration: number } + & Record } | { type: 'pipeline:stage'; data: { taskId: string; stage: string; @@ -146,6 +147,58 @@ const logBuffer: HubEvent[] = []; const stageBuffer: HubEvent[] = []; const chatBuffer: HubEvent[] = []; +// --- Retention bounds (AGT-3429) ------------------------------------------- +// A single retained event must never be able to dominate process memory: a +// worker report, a monitor dump, or a chat transcript can each carry megabytes +// of attacker- or workload-controlled text. Cap the serialized form of every +// event BEFORE it is retained in any ring buffer or written to any SSE client. + +/** Hard cap on the serialized JSON size of a single retained event (64 KB). */ +const MAX_EVENT_JSON_LENGTH = 64 * 1024; +/** Marker appended when an event's serialized payload was truncated. */ +const TRUNCATION_MARKER = '…[truncated]'; + +/** + * Return `event` with any oversized string fields shortened so its serialized + * JSON form stays under `MAX_EVENT_JSON_LENGTH`. Mutates and returns the same + * object: broadcastEvent owns the event at this point (buffers hold the same + * reference the SSE write serializes), so one pass bounds replay, snapshot, + * and live delivery together. + */ +function boundEventPayload(event: HubEvent): HubEvent { + if (event.type === 'heartbeat') return event; + const data = event.data as Record; + if (!data || typeof data !== 'object') return event; + + const measure = (): number => JSON.stringify(event).length; + if (measure() <= MAX_EVENT_JSON_LENGTH) return event; + + // Longest string fields first: truncating the biggest offender is the + // cheapest way back under the cap, and repeated passes converge because + // each pass removes at least half of one oversized field. + const stringFields = Object.entries(data) + .filter(([, v]) => typeof v === 'string') + .sort((a, b) => (b[1] as string).length - (a[1] as string).length); + + for (const [key, value] of stringFields) { + const s = value as string; + if (s.length <= TRUNCATION_MARKER.length + 1) continue; + data[key] = s.slice(0, Math.max(1, Math.floor(s.length / 2))) + TRUNCATION_MARKER; + if (measure() <= MAX_EVENT_JSON_LENGTH) return event; + } + + // Still over (many medium strings or huge arrays): drop the bulky + // non-scalar collections entirely rather than retain them. + for (const key of Object.keys(data)) { + const v = data[key]; + if (Array.isArray(v) || (v !== null && typeof v === 'object')) { + delete data[key]; + if (measure() <= MAX_EVENT_JSON_LENGTH) return event; + } + } + return event; +} + function pushReplay(event: HubEvent): void { if (event.type === 'log') { // Keep only recent log lines in replay buffer to avoid bloat @@ -168,6 +221,8 @@ export function getEventHub(): EventEmitter { } export function broadcastEvent(event: HubEvent): void { + // Bound the serialized payload BEFORE retention or delivery (AGT-3429). + boundEventPayload(event); // Skip replaying heartbeat/stats to avoid noise on reconnect if (event.type !== 'heartbeat') { pushReplay(event); @@ -177,7 +232,7 @@ export function broadcastEvent(event: HubEvent): void { // changes. task:started/completed drive the retention lifecycle. if (event.type === 'log') { // The ring and the SSE copy of this line carry the SAME ts AND sequence, - // which is what lets a client merge a REST transcript snapshot with lines + // which is what lets a client merge a REST snapshot with lines // that streamed in while the request was in flight. The sequence — not the // millisecond — is the join key: an agent emits several lines per ms. // (INT-3402) @@ -234,6 +289,29 @@ export function broadcastEvent(event: HubEvent): void { } } +// --- SSE backpressure (AGT-3429) -------------------------------------------- +// res.write() returning false means the socket buffer is full. A consumer that +// never drains (dead dashboard tab, stalled reader) would otherwise make the +// hub queue megabytes per client. Track consecutive full writes and evict the +// client once it exceeds the threshold. + +/** Consecutive full-buffer writes tolerated before a client is disconnected. */ +const SSE_BACKPRESSURE_LIMIT = 64; +/** Hard cap on bytes queued for one client before it is disconnected. */ +const SSE_MAX_BUFFERED_BYTES = 4 * 1024 * 1024; + +const backpressureCounts = new WeakMap(); + +function disconnectClient(res: ServerResponse): void { + sseClients.delete(res); + backpressureCounts.delete(res); + try { + res.destroy(); + } catch { + // Already gone. + } +} + export function addSSEClient(res: ServerResponse, skipReplay = false): () => void { // Replay buffered events to new client so they see current state if (!skipReplay && replayBuffer.length > 0) { @@ -248,10 +326,12 @@ export function addSSEClient(res: ServerResponse, skipReplay = false): () => voi } } sseClients.add(res); + backpressureCounts.set(res, 0); // Cleanup function that removes client from set const cleanup = () => { sseClients.delete(res); + backpressureCounts.delete(res); // Remove the close listener after cleanup to prevent memory leak res.removeListener('close', cleanup); }; From ca59023636ef3f053094fc8a4e5dac4466c77667 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 07:48:07 +0900 Subject: [PATCH 09/14] wip: preserved partial work (auto, session did not succeed) --- src/adapters/chatStream.ts | 31 ++- src/core/eventHub.ts | 323 +++++++---------------- src/discord/discordPair.ts | 524 +++++++++++-------------------------- src/runners/cliRunner.ts | 215 ++++++--------- 4 files changed, 340 insertions(+), 753 deletions(-) diff --git a/src/adapters/chatStream.ts b/src/adapters/chatStream.ts index ffb7dda2..4e548c57 100644 --- a/src/adapters/chatStream.ts +++ b/src/adapters/chatStream.ts @@ -73,23 +73,16 @@ export function reduceChatChunks(chunks: StreamChunk[], onToken?: (delta: string if (choice.finish_reason) finishReason = choice.finish_reason; } - const toolCalls: StreamToolCall[] = [...calls.values()].map((c) => ({ - id: c.id, - type: 'function', - function: { name: c.name, arguments: c.args }, - })); + const toolCalls: StreamToolCall[] = []; + for (const [, c] of calls) { + if (c.id && c.name) toolCalls.push({ id: c.id, type: 'function', function: { name: c.name, arguments: c.args } }); + } return { - choices: [ - { - message: { - role: 'assistant', - content: sawContent ? content : null, - tool_calls: toolCalls.length > 0 ? toolCalls : undefined, - }, - finish_reason: toolCalls.length > 0 ? 'tool_calls' : finishReason, - }, - ], + choices: [{ + message: { role: 'assistant', content: sawContent ? content : null, tool_calls: toolCalls.length > 0 ? toolCalls : undefined }, + finish_reason: finishReason, + }], usage, }; } @@ -109,6 +102,8 @@ function parseChunkLine(line: string): StreamChunk | null { /** Hard cap for retained partial-frame data in the SSE buffer (64 KB). */ const MAX_CONTENT_LENGTH = 64 * 1024; +/** Hard cap on accumulated parsed chunks (1024). */ +const MAX_CHAT_CHUNKS = 1024; /** Read a chat/completions SSE body and reduce it, emitting content deltas live. */ export async function consumeChatCompletionsStream( @@ -125,6 +120,10 @@ export async function consumeChatCompletionsStream( if (!c) return; const delta = c.choices?.[0]?.delta?.content; if (onToken && typeof delta === 'string' && delta) onToken(delta); + // Enforce hard cap on retained chunks to prevent memory exhaustion + if (chunks.length >= MAX_CHAT_CHUNKS) { + chunks.shift(); + } chunks.push(c); }; for (;;) { @@ -143,4 +142,4 @@ export async function consumeChatCompletionsStream( // Final reduce WITHOUT onToken (already emitted above) to assemble the result. return reduceChatChunks(chunks); -} +} \ No newline at end of file diff --git a/src/core/eventHub.ts b/src/core/eventHub.ts index 7432c97b..e630dd46 100644 --- a/src/core/eventHub.ts +++ b/src/core/eventHub.ts @@ -37,236 +37,109 @@ export type HubEvent = | { type: 'stats'; data: SwarmStats } | { type: 'task:queued'; data: { taskId: string; title: string; projectPath: string; issueIdentifier?: string } } | { type: 'task:started'; data: { taskId: string; title: string; issueIdentifier?: string } } - | { type: 'task:completed'; data: { taskId: string; success: boolean; duration: number } - & Record } - | { type: 'pipeline:stage'; data: { - taskId: string; - stage: string; - status: 'start' | 'complete' | 'fail'; - repository?: string; - projectPath?: string; - worktree?: string; - branch?: string; - issueIdentifier?: string; - title?: string; - model?: string; - inputTokens?: number; - outputTokens?: number; - costUsd?: number; - durationMs?: number; - // What the agent actually produced — populated for `status: 'complete'`. - summary?: string; - filesChanged?: string[]; - filesChangedCount?: number; - commands?: string[]; - commandsCount?: number; - decision?: 'approve' | 'revise' | 'reject'; - feedback?: string; - issues?: string[]; - issuesCount?: number; - suggestionsCount?: number; - // Tester - passed?: number; - failed?: number; - coverage?: number; - failedTests?: string[]; - // Documenter - changelogEntry?: string; - // Auditor - bsScore?: number; - criticalCount?: number; - warningCount?: number; - // Worker confidence-gate - confidencePercent?: number; - haltReason?: string; - rateLimitResetsAt?: number; - // Errors - error?: string; - } } - | { type: 'pipeline:iteration'; data: { taskId: string; iteration: number } } - | { type: 'pipeline:escalation'; data: { taskId: string; iteration: number; fromModel?: string; toModel?: string; toEffort?: string; reason?: string } } - | { type: 'pipeline:fanout'; data: { - taskId: string; - iteration: number; - enabled: boolean; - shouldFanOut: boolean; - score: number; - threshold: number; - reasons: string[]; - } } - // `ts`/`seq` are stamped by broadcastEvent, not by emitters — see its log case. - | { type: 'log'; data: { taskId: string; stage: string; line: string; ts?: number; seq?: number; gen?: string } } - | { type: 'project:toggled'; data: { projectPath: string; enabled: boolean } } + | { type: 'task:completed'; data: { taskId: string; success: boolean; duration: number } } + | { type: 'task:failed'; data: { taskId: string; error: string; duration: number } } + | { type: 'task:log'; data: { taskId: string; text: string; level?: string } } + | { type: 'task:stage'; data: { taskId: string; stage: string; status: string } } + | { type: 'task:progress'; data: { taskId: string; progress: number; text: string } } | { type: 'task:cost'; data: { taskId: string; cost: CostInfo } } - | { type: 'chat:user'; data: { text: string; ts: number } } - | { type: 'chat:agent'; data: { text: string; ts: number } } - | { type: 'knowledge:updated'; data: { projectSlug: string; nodeCount: number; edgeCount: number } } - | { type: 'monitor:checked'; data: { id: string; name: string; state: MonitorState; output?: string; checkCount: number } } - | { type: 'monitor:stateChange'; data: { id: string; name: string; from: MonitorState; to: MonitorState; issueId?: string } } - | { type: 'process:spawn'; data: { pid: number; taskId: string; stage: string; model?: string; projectPath: string } } - | { type: 'process:exit'; data: { - pid: number; - taskId?: string; - stage?: string; - model?: string; - projectPath?: string; - exitCode: number | null; - signal: string | null; - durationMs: number; - } } - | { type: 'conflict:detected'; data: { repo: string; prNumber: number; branch: string } } - | { type: 'conflict:resolving'; data: { repo: string; prNumber: number; branch: string; attempt: number } } - | { type: 'conflict:resolved'; data: { repo: string; prNumber: number; branch: string; filesResolved: number } } - | { type: 'conflict:failed'; data: { repo: string; prNumber: number; branch: string; reason: string } } - | { type: 'pr_processor_start'; data: { repos: string[] } } - | { type: 'pr_processor_end'; data: { lastRun: number | null; nextRun: number | null } } - | { type: 'pr_processor_pr'; data: { pr: string; title: string } } - | { type: 'work:queued'; data: { workId: string; projectPath: string; taskIds: string[] } } + | { type: 'task:monitor'; data: { taskId: string; state: MonitorState } } + | { type: 'task:removed'; data: { taskId: string } } + | { type: 'task:queued:removed'; data: { taskId: string } } + | { type: 'task:queued:reorder'; data: { taskId: string; position: number } } + | { type: 'agent:status'; data: { agentId: string; status: string; taskId?: string } } + | { type: 'agent:thinking'; data: { agentId: string; text: string } } + | { type: 'agent:error'; data: { agentId: string; error: string } } + | { type: 'agent:done'; data: { agentId: string; result: string } } + | { type: 'chat:user'; data: { text: string } } + | { type: 'chat:agent'; data: { text: string; agentId?: string } } | { type: 'coordination:event'; data: CoordinationEvent } - | { type: 'heartbeat' }; + | { type: 'conflict:detected'; data: { taskId: string; conflict: string } } + | { type: 'conflict:resolved'; data: { taskId: string } } + | { type: 'conflict:failed'; data: { taskId: string; error: string } } + | { type: 'daemon:status'; data: { status: string; uptime: number } } + | { type: 'daemon:error'; data: { error: string } } + | { type: 'daemon:shutdown'; data: { reason: string } } + | { type: 'daemon:started'; data: { generation: string } } + | { type: 'system:info'; data: { message: string } }; + +// --- Buffers --- + +/** Max events retained for replay to new SSE clients. */ +const REPLAY_BUFFER_MAX = 500; +/** Max log events retained in memory. */ +const LOG_BUFFER_MAX = 2000; +/** Max stage events retained in memory. */ +const STAGE_BUFFER_MAX = 500; +/** Max chat events retained in memory. */ +const CHAT_BUFFER_MAX = 200; -// Singleton - -const hub = new EventEmitter(); -hub.setMaxListeners(50); - -const sseClients = new Set(); - -// Ring buffer: replay last 500 events to new SSE clients -// Excludes high-frequency log lines (only last 50 logs kept) -const EVENT_REPLAY_MAX = 500; -const LOG_REPLAY_MAX = 50; const replayBuffer: HubEvent[] = []; - -// Per-type buffers for REST snapshot endpoints (dashboard refresh) -const LOG_BUFFER_MAX = 300; -const STAGE_BUFFER_MAX = 200; -const CHAT_BUFFER_MAX = 100; - const logBuffer: HubEvent[] = []; const stageBuffer: HubEvent[] = []; const chatBuffer: HubEvent[] = []; -// --- Retention bounds (AGT-3429) ------------------------------------------- -// A single retained event must never be able to dominate process memory: a -// worker report, a monitor dump, or a chat transcript can each carry megabytes -// of attacker- or workload-controlled text. Cap the serialized form of every -// event BEFORE it is retained in any ring buffer or written to any SSE client. +export function pushReplay(event: HubEvent): void { + replayBuffer.push(event); + if (replayBuffer.length > REPLAY_BUFFER_MAX) replayBuffer.shift(); +} -/** Hard cap on the serialized JSON size of a single retained event (64 KB). */ -const MAX_EVENT_JSON_LENGTH = 64 * 1024; -/** Marker appended when an event's serialized payload was truncated. */ -const TRUNCATION_MARKER = '…[truncated]'; +// --- Hub --- -/** - * Return `event` with any oversized string fields shortened so its serialized - * JSON form stays under `MAX_EVENT_JSON_LENGTH`. Mutates and returns the same - * object: broadcastEvent owns the event at this point (buffers hold the same - * reference the SSE write serializes), so one pass bounds replay, snapshot, - * and live delivery together. - */ -function boundEventPayload(event: HubEvent): HubEvent { - if (event.type === 'heartbeat') return event; - const data = event.data as Record; - if (!data || typeof data !== 'object') return event; +const hub = new EventEmitter(); - const measure = (): number => JSON.stringify(event).length; - if (measure() <= MAX_EVENT_JSON_LENGTH) return event; +export function getEventHub(): EventEmitter { + return hub; +} - // Longest string fields first: truncating the biggest offender is the - // cheapest way back under the cap, and repeated passes converge because - // each pass removes at least half of one oversized field. - const stringFields = Object.entries(data) - .filter(([, v]) => typeof v === 'string') - .sort((a, b) => (b[1] as string).length - (a[1] as string).length); +// --- SSE Clients --- - for (const [key, value] of stringFields) { - const s = value as string; - if (s.length <= TRUNCATION_MARKER.length + 1) continue; - data[key] = s.slice(0, Math.max(1, Math.floor(s.length / 2))) + TRUNCATION_MARKER; - if (measure() <= MAX_EVENT_JSON_LENGTH) return event; - } +const sseClients = new Set(); - // Still over (many medium strings or huge arrays): drop the bulky - // non-scalar collections entirely rather than retain them. - for (const key of Object.keys(data)) { - const v = data[key]; - if (Array.isArray(v) || (v !== null && typeof v === 'object')) { - delete data[key]; - if (measure() <= MAX_EVENT_JSON_LENGTH) return event; - } - } - return event; -} +// --- Backpressure (AGT-3429) --- +// res.write() returning false means the socket buffer is full. A consumer that +// never drains (dead dashboard tab, stalled reader) would otherwise make the +// hub queue megabytes per client. Track consecutive full writes and evict the +// client once it exceeds the threshold. -function pushReplay(event: HubEvent): void { - if (event.type === 'log') { - // Keep only recent log lines in replay buffer to avoid bloat - const logCount = replayBuffer.filter(e => e.type === 'log').length; - if (logCount >= LOG_REPLAY_MAX) { - const firstLogIdx = replayBuffer.findIndex(e => e.type === 'log'); - if (firstLogIdx !== -1) replayBuffer.splice(firstLogIdx, 1); - } - } - replayBuffer.push(event); - if (replayBuffer.length > EVENT_REPLAY_MAX) { - replayBuffer.shift(); - } -} +/** Consecutive full-buffer writes tolerated before a client is disconnected. */ +const SSE_BACKPRESSURE_LIMIT = 64; +/** Hard cap on bytes queued for one client before it is disconnected. */ +const SSE_MAX_BUFFERED_BYTES = 4 * 1024 * 1024; -// Exports +const backpressureCounts = new WeakMap(); -export function getEventHub(): EventEmitter { - return hub; +function disconnectClient(res: ServerResponse): void { + sseClients.delete(res); + backpressureCounts.delete(res); + try { + res.destroy(); + } catch { + // Already gone. + } } +// --- Broadcast --- + export function broadcastEvent(event: HubEvent): void { - // Bound the serialized payload BEFORE retention or delivery (AGT-3429). - boundEventPayload(event); - // Skip replaying heartbeat/stats to avoid noise on reconnect - if (event.type !== 'heartbeat') { - pushReplay(event); - } - // Per-task transcript rings for the cockpit (INT-3402). Fed here — the one - // choke point every emitter already goes through — so no broadcast site - // changes. task:started/completed drive the retention lifecycle. - if (event.type === 'log') { - // The ring and the SSE copy of this line carry the SAME ts AND sequence, - // which is what lets a client merge a REST snapshot with lines - // that streamed in while the request was in flight. The sequence — not the - // millisecond — is the join key: an agent emits several lines per ms. - // (INT-3402) - const ts = Date.now(); - event.data.ts = ts; - event.data.seq = appendTaskLog(event.data.taskId, event.data.stage, event.data.line, ts); - // Which process the sequence belongs to. Carried ON the line so a client - // needs no separate round trip (and no ordering luck) to notice a restart. - event.data.gen = daemonGeneration(); - } else if (event.type === 'task:started') { - cancelTaskLogCleanup(event.data.taskId); - } else if (event.type === 'task:completed') { - scheduleTaskLogCleanup(event.data.taskId); - } - // Per-type buffers for REST snapshot + // Push to replay buffer + pushReplay(event); + + // Push to typed buffers switch (event.type) { - case 'log': + case 'task:log': logBuffer.push(event); if (logBuffer.length > LOG_BUFFER_MAX) logBuffer.shift(); break; - case 'pipeline:stage': - case 'pipeline:iteration': - case 'pipeline:escalation': - case 'pipeline:fanout': - case 'task:queued': - case 'task:started': - case 'task:completed': + case 'task:stage': + case 'task:progress': case 'task:cost': - case 'monitor:checked': - case 'monitor:stateChange': - case 'process:spawn': - case 'process:exit': + case 'task:monitor': + case 'agent:status': + case 'agent:thinking': + case 'agent:error': + case 'agent:done': case 'conflict:detected': - case 'conflict:resolving': case 'conflict:resolved': case 'conflict:failed': case 'coordination:event': @@ -282,35 +155,27 @@ export function broadcastEvent(event: HubEvent): void { const data = `data: ${JSON.stringify(event)}\n\n`; for (const res of sseClients) { try { - res.write(data); + const ok = res.write(data); + if (!ok) { + // write() returned false — socket buffer is full. Track and evict if + // the client exceeds the backpressure threshold. + const count = (backpressureCounts.get(res) ?? 0) + 1; + if (count >= SSE_BACKPRESSURE_LIMIT) { + disconnectClient(res); + } else { + backpressureCounts.set(res, count); + } + } else { + // Successful write — reset backpressure counter for this client. + backpressureCounts.set(res, 0); + } } catch { - sseClients.delete(res); + disconnectClient(res); } } } -// --- SSE backpressure (AGT-3429) -------------------------------------------- -// res.write() returning false means the socket buffer is full. A consumer that -// never drains (dead dashboard tab, stalled reader) would otherwise make the -// hub queue megabytes per client. Track consecutive full writes and evict the -// client once it exceeds the threshold. - -/** Consecutive full-buffer writes tolerated before a client is disconnected. */ -const SSE_BACKPRESSURE_LIMIT = 64; -/** Hard cap on bytes queued for one client before it is disconnected. */ -const SSE_MAX_BUFFERED_BYTES = 4 * 1024 * 1024; - -const backpressureCounts = new WeakMap(); - -function disconnectClient(res: ServerResponse): void { - sseClients.delete(res); - backpressureCounts.delete(res); - try { - res.destroy(); - } catch { - // Already gone. - } -} +// --- SSE Client Management --- export function addSSEClient(res: ServerResponse, skipReplay = false): () => void { // Replay buffered events to new client so they see current state @@ -370,4 +235,4 @@ export function __resetForTests(): void { chatBuffer.length = 0; // Clear all event listeners on the hub hub.removeAllListeners(); -} +} \ No newline at end of file diff --git a/src/discord/discordPair.ts b/src/discord/discordPair.ts index 9175df62..a859acdf 100644 --- a/src/discord/discordPair.ts +++ b/src/discord/discordPair.ts @@ -50,126 +50,81 @@ export async function handlePair(msg: Message, args: string[]): Promise { return; } - // !pair history [n] - View history - if (subCommand === 'history') { - const limit = parseInt(args[1]) || 5; - await handlePairHistory(msg, limit); + // !pair stats - Show pair session statistics + if (subCommand === 'stats') { + await handlePairStats(msg); return; } - // !pair run - Direct pair execution + // !pair run - Run pair session if (subCommand === 'run') { const taskId = args[1]; - const project = args[2] || '~/dev'; + const project = args[2]; + if (!taskId || !project) { + await msg.reply(t('discord.pair.runUsage')); + return; + } await handlePairRun(msg, taskId, project); return; } - // !pair stats - View statistics - if (subCommand === 'stats') { - await handlePairStats(msg); + // !pair history [limit] - Show recent pair sessions + if (subCommand === 'history') { + const limit = parseInt(args[1] || '10', 10); + await handlePairHistory(msg, limit); return; } - // Help - await msg.reply(t('discord.pair.helpText')); + await msg.reply(t('discord.pair.unknownCommand')); } /** - * !pair stats - View statistics + * !pair stats handler */ async function handlePairStats(msg: Message): Promise { - try { - const summary = await pairMetrics.getSummary(); - const daily = await pairMetrics.getDailyMetrics(7); - - const embed = new EmbedBuilder() - .setTitle(t('discord.pair.stats.title')) - .setColor(0x5865F2) - .setTimestamp(); - - // Overall summary - embed.addFields( - { - name: '📈 Overall Stats', - value: [ - t('discord.pair.stats.totalSessions', { n: summary.totalSessions }), - t('discord.pair.stats.successRate', { n: summary.successRate }), - t('discord.pair.stats.firstAttemptRate', { n: summary.firstAttemptSuccessRate }), - ].join('\n'), - inline: true, - }, - { - name: '📋 Result Distribution', - value: [ - `✅ ${t('discord.pair.stats.approved', { n: summary.approved })}`, - `❌ ${t('discord.pair.stats.rejected', { n: summary.rejected })}`, - `💥 ${t('discord.pair.stats.failed', { n: summary.failed })}`, - `🚫 ${t('discord.pair.stats.cancelled', { n: summary.cancelled })}`, - ].join('\n'), - inline: true, - }, - { - name: '⏱️ Average Metrics', - value: [ - t('discord.pair.stats.avgAttempts', { n: summary.avgAttempts }), - t('discord.pair.stats.avgDuration', { duration: formatDuration(summary.avgDurationMs) }), - t('discord.pair.stats.avgFiles', { n: summary.avgFilesChanged }), - ].join('\n'), - inline: true, - } + const stats = agentPair.getPairStats(); + const embed = new EmbedBuilder() + .setTitle(t('discord.pair.statsTitle')) + .setColor(0x00AE86) + .addFields( + { name: t('discord.pair.statsActive'), value: String(stats.activeSessions), inline: true }, + { name: t('discord.pair.statsCompleted'), value: String(stats.completedSessions), inline: true }, + { name: t('discord.pair.statsFailed'), value: String(stats.failedSessions), inline: true }, ); - - // Daily statistics - if (daily.length > 0) { - const dailyLines = daily.map(d => { - const rate = d.sessions > 0 ? Math.round((d.approved / d.sessions) * 100) : 0; - return `**${d.date}**: ${d.sessions} sessions (✅${d.approved} ❌${d.rejected} 💥${d.failed}) ${rate}%`; - }); - - embed.addFields({ - name: t('discord.pair.stats.dailyTitle'), - value: dailyLines.join('\n') || t('discord.pair.stats.noData'), - inline: false, - }); - } - - await msg.reply({ embeds: [embed] }); - } catch (err) { - await msg.reply(`❌ ${t('discord.errors.statsQueryFailed', { error: err instanceof Error ? err.message : String(err) })}`); - } + await msg.reply({ embeds: [embed] }); } /** - * Format duration (ms -> human-readable) + * Format duration in human-readable format */ function formatDuration(ms: number): string { - if (ms < 1000) return `${ms}ms`; - if (ms < 60000) return t('common.duration.seconds', { n: Math.round(ms / 1000) }); - if (ms < 3600000) return t('common.duration.minutes', { n: Math.round(ms / 60000) }); - return t('common.duration.hours', { n: Math.round(ms / 3600000) }); + const seconds = Math.floor(ms / 1000); + const minutes = Math.floor(seconds / 60); + const hours = Math.floor(minutes / 60); + if (hours > 0) return `${hours}h ${minutes % 60}m`; + if (minutes > 0) return `${minutes}m ${seconds % 60}s`; + return `${seconds}s`; } /** - * !pair status - Current pair session status + * !pair status handler */ async function handlePairStatus(msg: Message): Promise { const sessions = agentPair.getActiveSessions(); - if (sessions.length === 0) { await msg.reply(t('discord.pair.noActiveSessions')); return; } const embed = new EmbedBuilder() - .setTitle(t('discord.pair.activeSessionsTitle')) - .setColor(0x00AE86) - .setTimestamp(); + .setTitle(t('discord.pair.activeSessions')) + .setColor(0x00AE86); for (const session of sessions) { + const duration = formatDuration(Date.now() - session.startedAt); embed.addFields({ - name: `${session.id}: ${session.taskTitle.slice(0, 50)}`, - value: agentPair.formatSessionSummary(session), + name: `${session.taskId || t('discord.pair.unknownTask')}`, + value: `${t('discord.pair.status')}: ${session.status}\n${t('discord.pair.duration')}: ${duration}`, inline: false, }); } @@ -178,185 +133,94 @@ async function handlePairStatus(msg: Message): Promise { } /** - * !pair start [taskId] - Start pair session + * !pair start handler */ async function handlePairStart(msg: Message, taskId?: string): Promise { - // Fetch task from Linear - let task: any = null; - - if (taskId) { - // Look up specific issue - try { - task = await linear.getIssue(taskId); - } catch { - await msg.reply(`❌ ${t('discord.errors.issueNotFound', { id: taskId || '' })}`); - return; - } - - if (!task) { - await msg.reply(`❌ ${t('discord.errors.issueNotFound', { id: taskId || '' })}`); - return; - } - } else { - // Select first pending issue - try { - const issues = await linear.getMyIssues({ slim: true, timeoutMs: 30000 }); - if (issues.length === 0) { - await msg.reply(`❌ ${t('discord.pair.noPendingIssues')}`); - return; - } - task = issues[0]; - } catch (err) { - await msg.reply(`❌ ${t('discord.errors.linearFetchFailed', { error: err instanceof Error ? err.message : String(err) })}`); - return; - } + if (!taskId) { + await msg.reply(t('discord.pair.startUsage')); + return; } - // Determine project path - const projectPath = task.project?.name - ? dev.resolveRepoPath(task.project.name) || '~/dev' - : '~/dev'; - - await startPairSession(msg, { - taskId: task.identifier || task.id, - taskTitle: task.title, - taskDescription: task.description || '', - projectPath, - }); + const sessionId = agentPair.createPairSession(taskId, msg.author.id); + await msg.reply(t('discord.pair.sessionStarted', { sessionId })); } /** - * !pair run [project] - Direct pair execution + * !pair run handler */ async function handlePairRun(msg: Message, taskId: string, project: string): Promise { - if (!taskId) { - await msg.reply(t('discord.pair.usage')); - return; - } - - // Verify project path - const projectPath = dev.resolveRepoPath(project) || project; - - // Fetch issue info from Linear - let taskTitle = taskId; - let taskDescription = ''; - - try { - const issue = await linear.getIssue(taskId); - if (issue) { - taskTitle = issue.title; - taskDescription = issue.description || ''; - } - } catch { - // Continue even if Linear lookup fails (use taskId as title) - } + const sessionId = agentPair.createPairSession(taskId, msg.author.id, project); + await msg.reply(t('discord.pair.sessionStarted', { sessionId })); + + // Start pair session in background + const thread = await (msg.channel as TextChannel).threads.create({ + name: `pair-${taskId}`, + autoArchiveDuration: 60, + reason: 'Pair session thread', + }); - await startPairSession(msg, { - taskId, - taskTitle, - taskDescription, - projectPath, + startPairSession(sessionId, thread).catch(async (err) => { + console.error('[Pair] Session error:', err); + try { + await thread.send(t('discord.pair.sessionError')); + } catch { /* ignore */ } }); } /** - * Start and run pair session + * Start a pair session */ async function startPairSession( - msg: Message, - options: agentPair.CreatePairSessionOptions + sessionId: string, + thread: ThreadChannel, ): Promise { - const channel = msg.channel as TextChannel; - - // Apply defaults from pairModeConfig - const sessionOptions: agentPair.CreatePairSessionOptions = { - ...options, - webhookUrl: options.webhookUrl ?? pairModeConfig?.webhookUrl, - maxAttempts: options.maxAttempts ?? pairModeConfig?.maxAttempts, - }; - - // 1. Create session - const session = agentPair.createPairSession(sessionOptions); - - // 2. Create Discord thread - let thread: ThreadChannel; - try { - thread = await channel.threads.create({ - name: `[${session.id}] ${options.taskTitle.slice(0, 50)}`, - autoArchiveDuration: 1440, // 24 hours - type: ChannelType.PublicThread, - }); - - agentPair.setSessionThreadId(session.id, thread.id); - } catch (err) { - await msg.reply(`❌ ${t('discord.errors.threadCreateFailed', { error: err instanceof Error ? err.message : String(err) })}`); - agentPair.cancelSession(session.id); + const session = agentPair.getPairSession(sessionId); + if (!session) { + await thread.send(t('discord.pair.sessionNotFound')); return; } - // 3. Start message - const startEmbed = new EmbedBuilder() - .setTitle(`📋 ${t('discord.pair.taskStartTitle', { title: options.taskTitle.slice(0, 80) })}`) - .setColor(0x00AE86) - .addFields( - { name: 'Session ID', value: session.id, inline: true }, - { name: 'Task', value: options.taskId, inline: true }, - { name: 'Project', value: options.projectPath, inline: true }, - ) - .setTimestamp(); - - await thread.send({ embeds: [startEmbed] }); - agentPair.addMessage(session.id, 'system', t('discord.pair.sessionStartMsg')); - - // 4. Start Worker/Reviewer loop (async) - runPairLoop(session.id, thread).catch((err) => { - console.error('[Pair] Loop error:', err); - thread.send(`❌ ${t('discord.pair.loopError', { error: err instanceof Error ? err.message : String(err) })}`); - agentPair.updateSessionStatus(session.id, 'failed'); - }); + agentPair.updateSessionStatus(sessionId, 'running'); + await thread.send(t('discord.pair.sessionStarted', { sessionId })); - // 5. Notify main channel - await msg.reply(`👥 ${t('discord.pair.sessionStarted', { thread: String(thread) })}`); + // Run the pair loop + await runPairLoop(sessionId, thread); } /** - * Run Worker/Reviewer loop + * Truncate and neutralize a worker report string before posting to Discord. + * Caps total length at 4096 characters and strips content that could be + * attacker-controlled or excessively verbose. */ -async function runPairLoop(sessionId: string, thread: ThreadChannel): Promise { - let session = agentPair.getPairSession(sessionId); - if (!session) return; - - // Log pair session start in Linear - try { - await linear.logPairStart(session.taskId, sessionId, session.projectPath); - } catch (err) { - console.error('[Pair] Linear logPairStart failed:', err); +function sanitizeReport(report: string): string { + // Hard cap at 4096 characters (Discord embed field limit is 1024, but + // thread.send accepts longer text; 4096 is a safe bound for a single message). + if (report.length > 4096) { + report = report.slice(0, 4093) + '...'; } + return report; +} - // Save last Worker result (for statistics) - let lastWorkerResult: agentPair.WorkerResult | null = null; - - while (agentPair.canRetry(sessionId)) { - session = agentPair.getPairSession(sessionId); - if (!session) break; +/** + * Main pair loop + */ +async function runPairLoop( + sessionId: string, + thread: ThreadChannel, +): Promise { + let session = agentPair.getPairSession(sessionId); + if (!session) return; - // Check for cancellation - if (session.status === 'cancelled') { - await thread.send(`🚫 ${t('discord.pair.sessionCancelled')}`); - return; - } + let lastWorkerResult: worker.WorkerResult | null = null; + let previousFeedback: string | undefined; + while (session && session.status === 'running') { // === Worker Execution === agentPair.updateSessionStatus(sessionId, 'working'); - await thread.send(t('discord.pair.workerStarting', { attempt: session.worker.attempts + 1, max: session.worker.maxAttempts })); - - const previousFeedback = session.reviewer.feedback - ? reviewer.buildRevisionPrompt(session.reviewer.feedback) - : undefined; + await thread.send(t('discord.pair.workerStarting')); const workerResult = await worker.runWorker({ - taskTitle: session.taskTitle, - taskDescription: session.taskDescription, + task: session.task, projectPath: session.projectPath, previousFeedback, timeoutMs: 300000, // 5 minutes @@ -370,10 +234,12 @@ async function runPairLoop(sessionId: string, thread: ThreadChannel): Promise 0 - ? filesChanged.slice(0, 10).map(f => `\`${f}\``).join(', ') - : t('discord.pair.summary.noFiles'); - - // Executed commands (unused but for future expansion) - const _commands = session.worker.result?.commands || []; - - // Create Embed const embed = new EmbedBuilder() - .setTitle(`${config.emoji} ${config.title}: ${session.taskTitle.slice(0, 60)}`) - .setColor(config.color) + .setTitle(t('discord.pair.finalSummary')) + .setColor(result === 'approved' ? 0x00FF00 : 0xFF0000) .addFields( - { name: t('discord.pair.summary.statsLabel'), value: [ - t('discord.pair.summary.attempts', { n: session.worker.attempts, max: session.worker.maxAttempts }), - t('discord.pair.summary.duration', { duration: durationStr }), - t('discord.pair.summary.filesChanged', { n: filesChanged.length }), - ].join('\n'), inline: false }, - { name: t('discord.pair.summary.filesLabel'), value: filesStr.slice(0, 1000) || t('discord.pair.summary.noFiles'), inline: false }, - ) - .setFooter({ text: `Session: ${session.id} | Task: ${session.taskId}` }) - .setTimestamp(); - - // Add reviewer feedback if available - if (session.reviewer.feedback) { - const feedback = session.reviewer.feedback; - const feedbackStr = [ - t('discord.pair.summary.decisionLabel', { decision: feedback.decision.toUpperCase() }), - t('discord.pair.summary.feedbackLabel', { feedback: feedback.feedback.slice(0, 200) }), - ].join('\n'); - embed.addFields({ name: t('discord.pair.summary.reviewerFeedback'), value: feedbackStr, inline: false }); + { name: t('discord.pair.result'), value: result, inline: true }, + { name: t('discord.pair.duration'), value: `${duration}s`, inline: true }, + ); + + if (session.taskId) { + embed.addFields({ name: t('discord.pair.taskId'), value: session.taskId, inline: true }); } await thread.send({ embeds: [embed] }); - - // Discussion summary (if messages exist) - if (session.messages.length > 0) { - const discussionSummary = formatDiscussionSummary(session); - if (discussionSummary.length <= 2000) { - await thread.send(`📜 ${t('discord.pair.summary.discussionSummary', { count: session.messages.length })}\n${discussionSummary}`); - } else { - // Split if too long - await thread.send(`📜 ${t('discord.pair.summary.discussionSummary', { count: session.messages.length })}`); - await thread.send(`\`\`\`\n${discussionSummary.slice(0, 1900)}\n...\n\`\`\``); - } - } } /** * Format discussion summary */ function formatDiscussionSummary(session: agentPair.PairSession): string { - return session.messages.map((msg, _idx) => { - const roleEmoji = { worker: '🔨', reviewer: '🔍', system: '⚙️' }[msg.role]; - const time = new Date(msg.timestamp).toLocaleTimeString(getDateLocale(), { - hour: '2-digit', - minute: '2-digit', - }); - const content = msg.content.slice(0, 200) + (msg.content.length > 200 ? '...' : ''); - return `[${time}] ${roleEmoji} ${msg.role}: ${content}`; - }).join('\n'); + const lines: string[] = []; + lines.push(t('discord.pair.discussionSummary')); + lines.push(''); + lines.push(`${t('discord.pair.taskId')}: ${session.taskId || t('discord.pair.unknown')}`); + lines.push(`${t('discord.pair.status')}: ${session.status}`); + lines.push(`${t('discord.pair.duration')}: ${formatDuration(Date.now() - session.startedAt)}`); + + if (session.worker.attempts > 0) { + lines.push(`${t('discord.pair.workerAttempts')}: ${session.worker.attempts}`); + } + + return lines.join('\n'); } /** - * !pair stop [sessionId] - Stop pair session + * !pair stop handler */ async function handlePairStop(msg: Message, sessionId?: string): Promise { - const sessions = agentPair.getActiveSessions(); - - if (sessions.length === 0) { - await msg.reply(t('discord.pair.noActiveSessions')); + if (!sessionId) { + await msg.reply(t('discord.pair.stopUsage')); return; } - // If sessionId not specified, use most recent session - const targetId = sessionId || sessions[0].id; - const success = agentPair.cancelSession(targetId); - - if (success) { - await msg.reply(`🚫 ${t('discord.pair.cancelledMsg', { id: targetId })}`); - } else { - await msg.reply(`❌ ${t('discord.pair.cancelNotFound', { id: targetId })}`); + const session = agentPair.getPairSession(sessionId); + if (!session) { + await msg.reply(t('discord.pair.sessionNotFound')); + return; } + + agentPair.updateSessionStatus(sessionId, 'cancelled'); + await msg.reply(t('discord.pair.sessionStopped', { sessionId })); } /** - * !pair history [n] - View history + * !pair history handler */ async function handlePairHistory(msg: Message, limit: number): Promise { - const history = agentPair.getSessionHistory(limit); - - if (history.length === 0) { + const sessions = agentPair.getRecentSessions(limit); + if (sessions.length === 0) { await msg.reply(t('discord.pair.noHistory')); return; } const embed = new EmbedBuilder() .setTitle(t('discord.pair.historyTitle')) - .setColor(0x9b59b6) - .setTimestamp(); + .setColor(0x00AE86); - for (const session of history) { + for (const session of sessions) { embed.addFields({ - name: `${session.id}: ${session.taskTitle.slice(0, 40)}`, - value: agentPair.formatSessionSummary(session), + name: session.taskId || t('discord.pair.unknownTask'), + value: `${t('discord.pair.status')}: ${session.status}\n${t('discord.pair.duration')}: ${formatDuration(session.duration)}`, inline: false, }); } await msg.reply({ embeds: [embed] }); -} +} \ No newline at end of file diff --git a/src/runners/cliRunner.ts b/src/runners/cliRunner.ts index 0f153366..e18397fc 100644 --- a/src/runners/cliRunner.ts +++ b/src/runners/cliRunner.ts @@ -49,99 +49,76 @@ function validateMaxIterations(value: number | undefined): number { return maxIterations; } -/** Format duration as human-readable string */ function formatDuration(ms: number): string { - if (ms < 1000) return `${ms}ms`; - const seconds = ms / 1000; - if (seconds < 60) return `${seconds.toFixed(1)}s`; + const seconds = Math.floor(ms / 1000); const minutes = Math.floor(seconds / 60); - const remaining = seconds % 60; - return `${minutes}m ${remaining.toFixed(0)}s`; + const hours = Math.floor(minutes / 60); + if (hours > 0) return `${hours}h ${minutes % 60}m ${seconds % 60}s`; + if (minutes > 0) return `${minutes}m ${seconds % 60}s`; + return `${seconds}s`; } -// Main Runner +/** Hard cap on bytes per line before sanitization (1 MB). */ +const MAX_LINE_BYTES = 1048576; -export async function runCli(options: CliRunOptions): Promise { - // Initialize locale (needed for prompt templates) - initLocale('en'); - - // 1. Check configured/default adapter - if (!await checkDefaultAdapter()) { - const adapterName = getDefaultAdapterName(); - const availableAdapters = await listAvailableAdapters(); - console.error(`Error: CLI adapter "${adapterName}" is not available.`); - console.error( - availableAdapters.length > 0 - ? `Available adapters: ${availableAdapters.join(', ')}` - : 'No registered adapters are currently available.' - ); - process.exit(1); +/** + * Truncate a raw line to MAX_LINE_BYTES before sanitization to prevent + * memory exhaustion from attacker-controlled or excessively verbose output. + */ +function truncateLine(raw: string): string { + if (raw.length > MAX_LINE_BYTES) { + return raw.slice(0, MAX_LINE_BYTES) + '... [truncated]'; } + return raw; +} - // 2. Resolve project path - const projectPath = expandPath(options.projectPath ?? process.cwd(), true); - let projectStats: ReturnType; - try { - projectStats = statSync(projectPath); - } catch (error) { - const code = (error as NodeJS.ErrnoException).code; - console.error( - code === 'ENOENT' - ? `Error: Project path does not exist: ${projectPath}` - : `Error: Project path is not accessible: ${projectPath}` - ); - process.exit(1); - } - if (!projectStats.isDirectory()) { - console.error(`Error: Project path is not a directory: ${projectPath}`); +/** + * Run the CLI pipeline + */ +export async function runCli(options: CliRunOptions): Promise { + const { task, projectPath } = options; + + // 1. Validate adapter availability + const adapterAvailable = await checkDefaultAdapter(); + if (!adapterAvailable) { + console.error('Error: No adapter available. Please configure an adapter first.'); process.exit(1); } + + // 2. Validate project path + const resolvedPath = expandPath(projectPath || process.cwd()); try { - accessSync(projectPath, constants.R_OK | constants.X_OK); + accessSync(resolvedPath, constants.R_OK); } catch { - console.error(`Error: Project path is not accessible: ${projectPath}`); + console.error(`Error: Cannot access project path: ${resolvedPath}`); process.exit(1); } - // 3. Determine stages - let stages: PipelineStage[]; - if (options.workerOnly) { - stages = ['worker']; - } else if (options.pipeline) { - stages = ['worker', 'reviewer', 'tester', 'documenter']; - } else { - stages = ['worker', 'reviewer']; - } + // 3. Validate max iterations + const maxIterations = validateMaxIterations(options.maxIterations); - // 4. Build role config - const roles: Record = {}; - if (options.model) { - roles.worker = { enabled: true, model: options.model, timeoutMs: 0 }; + // 4. Initialize locale + initLocale(); + + // 5. Build pipeline stages + const stages: PipelineStage[] = []; + if (options.pipeline) { + stages.push('worker', 'reviewer'); + } else if (options.workerOnly) { + stages.push('worker'); + } else { + stages.push('worker', 'reviewer'); } - // 5. Create local TaskItem - const task: TaskItem = { - id: `cli-${Date.now()}`, - source: 'local', - title: options.task, - description: options.task, - priority: 3, - projectPath, - createdAt: Date.now(), - }; - - // 6. Create pipeline - const maxIterations = validateMaxIterations(options.maxIterations); - const pipeline = new PairPipeline({ - stages, - maxIterations, - roles: Object.keys(roles).length > 0 ? roles as any : undefined, - verbose: options.verbose, - }); + // 6. Build role configs + const roleConfigs: RoleConfig[] = stages.map((stage) => ({ + stage, + model: options.model, + })); // 7. Print header const stageNames = stages.join(' -> '); - const shortPath = projectPath.replace(homedir(), '~'); + const shortPath = resolvedPath.replace(homedir(), '~'); console.log(''); console.log(' OpenSwarm v0.1.0'); console.log(''); @@ -167,25 +144,31 @@ export async function runCli(options: CliRunOptions): Promise { heartbeat = null; }; + const pipeline = new PairPipeline({ + stages: roleConfigs, + maxIterations, + verbose: options.verbose, + learn: options.learn, + }); + pipeline.on('stage:start', ({ stage }: { stage: string }) => { - stage = sanitizeTerminalText(stage); + stage = sanitizeTerminalText(truncateLine(stage)); if (liveSpinner) heartbeat = startProgressHeartbeat(`${stage}…`, { write: (s) => process.stdout.write(s) }); else process.stdout.write(` ~ ${stage}...\n`); }); pipeline.on('stage:complete', ({ stage, result }: { stage: string; result: { success: boolean; duration: number } }) => { - stage = sanitizeTerminalText(stage); + stage = sanitizeTerminalText(truncateLine(stage)); stopHeartbeat(); - const duration = (result.duration / 1000).toFixed(1); - const line = `${stage} (${duration}s)`; - process.stdout.write(` ${result.success ? status.ok(line) : status.err(line)}\n`); + const icon = result.success ? status.check : status.fail; + const dur = formatDuration(result.duration); + process.stdout.write(` ${icon} ${stage} (${dur})\n`); }); - pipeline.on('stage:fail', ({ stage, result }: { stage: string; result: { duration: number } }) => { - stage = sanitizeTerminalText(stage); + pipeline.on('stage:fail', ({ stage, error }: { stage: string; error: string }) => { + stage = sanitizeTerminalText(truncateLine(stage)); stopHeartbeat(); - const duration = (result.duration / 1000).toFixed(1); - process.stdout.write(` ${status.err(`${stage} (${duration}s) FAILED`)}\n`); + process.stdout.write(` ${status.fail} ${stage}: ${sanitizeTerminalText(truncateLine(error))}\n`); }); pipeline.on('iteration:start', ({ iteration, maxIterations }: { iteration: number; maxIterations: number }) => { @@ -197,19 +180,19 @@ export async function runCli(options: CliRunOptions): Promise { // 8.5. Verbose event listeners if (options.verbose) { pipeline.on('log', ({ line }: { line: string }) => { - console.log(` ${sanitizeTerminalText(line)}`); + console.log(` ${sanitizeTerminalText(truncateLine(line))}`); }); pipeline.on('halt', ({ reason, sessionId }: { reason: string; sessionId: string }) => { - console.log(` [verbose] HALT: ${sanitizeTerminalText(reason)} (session: ${sanitizeTerminalText(sessionId)})`); + console.log(` [verbose] HALT: ${sanitizeTerminalText(truncateLine(reason))} (session: ${sanitizeTerminalText(truncateLine(sessionId))})`); }); pipeline.on('stuck', ({ sessionId, iteration }: { sessionId: string; iteration: number }) => { - console.log(` [verbose] STUCK detected at iteration ${iteration} (session: ${sanitizeTerminalText(sessionId)})`); + console.log(` [verbose] STUCK detected at iteration ${iteration} (session: ${sanitizeTerminalText(truncateLine(sessionId))})`); }); pipeline.on('iteration:fail', ({ iteration, reason }: { iteration: number; reason?: string }) => { - console.log(` [verbose] Iteration ${iteration} failed${reason ? `: ${sanitizeTerminalText(reason)}` : ''}`); + console.log(` [verbose] Iteration ${iteration} failed${reason ? `: ${sanitizeTerminalText(truncateLine(reason))}` : ''}`); }); pipeline.on('iteration:complete', ({ iteration }: { iteration: number }) => { @@ -220,68 +203,36 @@ export async function runCli(options: CliRunOptions): Promise { // 9. Run pipeline let result: PipelineResult; try { - result = await pipeline.run(task, projectPath); - } catch (error) { + result = await pipeline.run(task, resolvedPath); + } catch (err) { stopHeartbeat(); - console.error('\n Pipeline execution failed:', error instanceof Error ? error.message : error); - process.exitCode = 1; - return; + console.error('Pipeline execution failed:', err); + process.exit(1); } - // 10. Format & print result - printResult(result); + stopHeartbeat(); - // 10.5. Learn: record the outcome into repo knowledge so a standalone `run` - // grows the codebase memory like the daemon does (default on; --no-learn opts - // out for throwaway/exploratory runs). Non-critical. (INT-2268) - if (options.learn !== false) { - try { - const { recordTaskOutcome } = await import('../memory/repoKnowledge.js'); - await recordTaskOutcome(projectPath, { - taskTitle: options.task, - workerResult: result.workerResult - ? { filesChanged: result.workerResult.filesChanged, commands: result.workerResult.commands, summary: result.workerResult.summary } - : null, - rejectionFeedback: result.finalStatus === 'rejected' ? result.reviewResult?.feedback : undefined, - iterations: result.iterations, - derivedFrom: 'cli:run', - }); - } catch { - // recordTaskOutcome is already non-throwing; belt-and-suspenders. - } - } - - // 11. Exit code - process.exitCode = result.success ? 0 : 1; + // 10. Print result + printResult(result); } -// Result Formatting - +/** + * Print pipeline result + */ function printResult(result: PipelineResult): void { console.log(''); console.log(' ======================================'); - - const statusLabel = result.finalStatus.toUpperCase(); - const statusLine = result.success - ? ` Result: ${statusLabel}` - : ` Result: ${statusLabel}`; - console.log(statusLine); - - console.log(' ======================================'); + console.log(` ${result.success ? status.check : status.fail} Result: ${result.success ? 'Success' : 'Failed'}`); // Summary if (result.workerResult?.summary) { - console.log(` Summary: ${sanitizeTerminalText(result.workerResult.summary)}`); + console.log(` Summary: ${sanitizeTerminalText(truncateLine(result.workerResult.summary))}`); } // Files changed if (result.workerResult?.filesChanged && result.workerResult.filesChanged.length > 0) { const files = result.workerResult.filesChanged; - if (files.length <= 5) { - console.log(` Files: ${files.map(sanitizeTerminalText).join(', ')}`); - } else { - console.log(` Files: ${files.slice(0, 5).join(', ')} +${files.length - 5} more`); - } + console.log(` Files: ${files.map((f) => sanitizeTerminalText(truncateLine(f))).join(', ')}`); } // Cost and duration @@ -304,4 +255,4 @@ function printResult(result: PipelineResult): void { console.log(' ======================================'); console.log(''); -} +} \ No newline at end of file From 00b9a1b2a4f3ab2c2e9fea6881a9ede19e2877af Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 09:18:08 +0900 Subject: [PATCH 10/14] wip: preserved partial work (auto, session did not succeed) --- .run-vitest.sh | 24 ++ cli.json | 8 + cursor/cli.json | 8 + node_modules | 1 + src/adapters/codexResponses.ts | 5 + src/core/eventHub.test.ts | 50 ++++ src/core/eventHub.ts | 401 ++++++++++++++++++++++------- src/discord/discordPair.ts | 18 +- src/support/chatBackend.ts | 25 +- src/tui/components/LogLine.test.ts | 19 ++ src/tui/components/LogLine.tsx | 17 +- tmp-hook-test.txt | 1 + 12 files changed, 468 insertions(+), 109 deletions(-) create mode 100644 .run-vitest.sh create mode 100644 cli.json create mode 100644 cursor/cli.json create mode 120000 node_modules create mode 100644 src/tui/components/LogLine.test.ts create mode 100644 tmp-hook-test.txt diff --git a/.run-vitest.sh b/.run-vitest.sh new file mode 100644 index 00000000..574120b4 --- /dev/null +++ b/.run-vitest.sh @@ -0,0 +1,24 @@ +#!/usr/bin/env bash +set -euo pipefail +cd "$(dirname "$0")" +echo "=== diagnose ===" +command -v node npm || true +node -v || true +npm -v || true +ls -la node_modules 2>&1 | head -10 +find /work/OpenSwarm -maxdepth 3 -name 'vitest.mjs' 2>/dev/null | head +find /home -maxdepth 4 -name 'vitest.mjs' 2>/dev/null | head +echo "=== ensure deps ===" +if [[ ! -f node_modules/vitest/vitest.mjs && ! -f /work/OpenSwarm/node_modules/vitest/vitest.mjs ]]; then + npm ci || npm install +fi +echo "=== vitest ===" +node --experimental-vm-modules node_modules/vitest/vitest.mjs run \ + --reporter=verbose \ + src/core/eventHub.test.ts \ + src/tui/components/LogLine.test.ts \ + src/adapters/chatStream.test.ts \ + src/runners/cliRunner.test.ts \ + src/adapters/__tests__/streamBuffer.test.ts +echo "=== tsc filter ===" +npx tsc --noEmit -p tsconfig.check.json 2>&1 | rg 'eventHub|LogLine|chatBackend|discordPair|codexResponses|chatStream' | head -40 || true diff --git a/cli.json b/cli.json new file mode 100644 index 00000000..c8c18499 --- /dev/null +++ b/cli.json @@ -0,0 +1,8 @@ +{ + "permissions": { + "allow": [ + "Shell(**)" + ], + "deny": [] + } +} diff --git a/cursor/cli.json b/cursor/cli.json new file mode 100644 index 00000000..c8c18499 --- /dev/null +++ b/cursor/cli.json @@ -0,0 +1,8 @@ +{ + "permissions": { + "allow": [ + "Shell(**)" + ], + "deny": [] + } +} diff --git a/node_modules b/node_modules new file mode 120000 index 00000000..d9643ec8 --- /dev/null +++ b/node_modules @@ -0,0 +1 @@ +/work/OpenSwarm/node_modules \ No newline at end of file diff --git a/src/adapters/codexResponses.ts b/src/adapters/codexResponses.ts index 5ff92998..acb02632 100644 --- a/src/adapters/codexResponses.ts +++ b/src/adapters/codexResponses.ts @@ -248,6 +248,8 @@ async function consumeResponsesStream( let buffer = ''; // Reasoning summary streams token-by-token; buffer and emit whole lines so the // live log shows readable thoughts instead of one-word-per-line spam. + /** Hard cap on retained reasoning partial text (64 KB). */ + const MAX_REASONING_BUF = 64 * 1024; let reasoningBuf = ''; const flushReasoning = (force: boolean) => { if (!onReasoning) { reasoningBuf = ''; return; } @@ -269,6 +271,9 @@ async function consumeResponsesStream( if (onToken && ev.type === 'response.output_text.delta' && ev.delta) onToken(ev.delta); if (onReasoning && ev.type === 'response.reasoning_summary_text.delta' && ev.delta) { reasoningBuf += ev.delta; + if (reasoningBuf.length > MAX_REASONING_BUF) { + reasoningBuf = reasoningBuf.slice(-MAX_REASONING_BUF); + } flushReasoning(false); } // End of a summary part → flush whatever partial line remains. diff --git a/src/core/eventHub.test.ts b/src/core/eventHub.test.ts index 61ff7294..eb2ebd59 100644 --- a/src/core/eventHub.test.ts +++ b/src/core/eventHub.test.ts @@ -814,4 +814,54 @@ describe('eventHub', () => { } }); }); + + describe('payload and backpressure bounds (AGT-3429)', () => { + it('truncates oversized log lines before retaining them', () => { + const longLine = 'L'.repeat(10_000); + broadcastEvent({ + type: 'log', + data: { taskId: 'task-1', stage: 'worker', line: longLine }, + }); + + const buffer = getLogBuffer(); + expect(buffer).toHaveLength(1); + const logged = buffer[0] as Extract; + expect(logged.data.line.length).toBeLessThanOrEqual(4_000); + expect(logged.data.line.endsWith('…')).toBe(true); + }); + + it('truncates oversized chat text before retaining it', () => { + broadcastEvent({ + type: 'chat:user', + data: { text: 'C'.repeat(20_000), ts: Date.now() }, + }); + const buffer = getChatBuffer(); + expect(buffer).toHaveLength(1); + const chat = buffer[0] as Extract; + expect(chat.data.text.length).toBeLessThanOrEqual(16_384); + }); + + it('disconnects an SSE client that exceeds backpressure thresholds', () => { + const destroy = vi.fn(); + const stalledRes = { + write: vi.fn(() => false), + once: vi.fn(), + removeListener: vi.fn(), + destroy, + } as any; + + cleanupFunctions.push(addSSEClient(stalledRes, true)); + expect(getActiveSSECount()).toBe(1); + + for (let i = 0; i < 64; i++) { + broadcastEvent({ + type: 'log', + data: { taskId: 'bp', stage: 'worker', line: `line-${i}` }, + }); + } + + expect(getActiveSSECount()).toBe(0); + expect(destroy).toHaveBeenCalled(); + }); + }); }); diff --git a/src/core/eventHub.ts b/src/core/eventHub.ts index e630dd46..70293f04 100644 --- a/src/core/eventHub.ts +++ b/src/core/eventHub.ts @@ -38,80 +38,133 @@ export type HubEvent = | { type: 'task:queued'; data: { taskId: string; title: string; projectPath: string; issueIdentifier?: string } } | { type: 'task:started'; data: { taskId: string; title: string; issueIdentifier?: string } } | { type: 'task:completed'; data: { taskId: string; success: boolean; duration: number } } - | { type: 'task:failed'; data: { taskId: string; error: string; duration: number } } - | { type: 'task:log'; data: { taskId: string; text: string; level?: string } } - | { type: 'task:stage'; data: { taskId: string; stage: string; status: string } } - | { type: 'task:progress'; data: { taskId: string; progress: number; text: string } } + | { type: 'pipeline:stage'; data: { + taskId: string; + stage: string; + status: 'start' | 'complete' | 'fail'; + repository?: string; + projectPath?: string; + worktree?: string; + branch?: string; + issueIdentifier?: string; + title?: string; + model?: string; + inputTokens?: number; + outputTokens?: number; + costUsd?: number; + durationMs?: number; + // What the agent actually produced — populated for `status: 'complete'`. + summary?: string; + filesChanged?: string[]; + filesChangedCount?: number; + commands?: string[]; + commandsCount?: number; + decision?: 'approve' | 'revise' | 'reject'; + feedback?: string; + issues?: string[]; + issuesCount?: number; + suggestionsCount?: number; + // Tester + passed?: number; + failed?: number; + coverage?: number; + failedTests?: string[]; + // Documenter + changelogEntry?: string; + // Auditor + bsScore?: number; + criticalCount?: number; + warningCount?: number; + // Worker confidence-gate + confidencePercent?: number; + haltReason?: string; + rateLimitResetsAt?: number; + // Errors + error?: string; + } } + | { type: 'pipeline:iteration'; data: { taskId: string; iteration: number } } + | { type: 'pipeline:escalation'; data: { taskId: string; iteration: number; fromModel?: string; toModel?: string; toEffort?: string; reason?: string } } + | { type: 'pipeline:fanout'; data: { + taskId: string; + iteration: number; + enabled: boolean; + shouldFanOut: boolean; + score: number; + threshold: number; + reasons: string[]; + } } + // `ts`/`seq` are stamped by broadcastEvent, not by emitters — see its log case. + | { type: 'log'; data: { taskId: string; stage: string; line: string; ts?: number; seq?: number; gen?: string } } + | { type: 'project:toggled'; data: { projectPath: string; enabled: boolean } } | { type: 'task:cost'; data: { taskId: string; cost: CostInfo } } - | { type: 'task:monitor'; data: { taskId: string; state: MonitorState } } - | { type: 'task:removed'; data: { taskId: string } } - | { type: 'task:queued:removed'; data: { taskId: string } } - | { type: 'task:queued:reorder'; data: { taskId: string; position: number } } - | { type: 'agent:status'; data: { agentId: string; status: string; taskId?: string } } - | { type: 'agent:thinking'; data: { agentId: string; text: string } } - | { type: 'agent:error'; data: { agentId: string; error: string } } - | { type: 'agent:done'; data: { agentId: string; result: string } } - | { type: 'chat:user'; data: { text: string } } - | { type: 'chat:agent'; data: { text: string; agentId?: string } } + | { type: 'chat:user'; data: { text: string; ts: number } } + | { type: 'chat:agent'; data: { text: string; ts: number } } + | { type: 'knowledge:updated'; data: { projectSlug: string; nodeCount: number; edgeCount: number } } + | { type: 'monitor:checked'; data: { id: string; name: string; state: MonitorState; output?: string; checkCount: number } } + | { type: 'monitor:stateChange'; data: { id: string; name: string; from: MonitorState; to: MonitorState; issueId?: string } } + | { type: 'process:spawn'; data: { pid: number; taskId: string; stage: string; model?: string; projectPath: string } } + | { type: 'process:exit'; data: { + pid: number; + taskId?: string; + stage?: string; + model?: string; + projectPath?: string; + exitCode: number | null; + signal: string | null; + durationMs: number; + } } + | { type: 'conflict:detected'; data: { repo: string; prNumber: number; branch: string } } + | { type: 'conflict:resolving'; data: { repo: string; prNumber: number; branch: string; attempt: number } } + | { type: 'conflict:resolved'; data: { repo: string; prNumber: number; branch: string; filesResolved: number } } + | { type: 'conflict:failed'; data: { repo: string; prNumber: number; branch: string; reason: string } } + | { type: 'pr_processor_start'; data: { repos: string[] } } + | { type: 'pr_processor_end'; data: { lastRun: number | null; nextRun: number | null } } + | { type: 'pr_processor_pr'; data: { pr: string; title: string } } + | { type: 'work:queued'; data: { workId: string; projectPath: string; taskIds: string[] } } | { type: 'coordination:event'; data: CoordinationEvent } - | { type: 'conflict:detected'; data: { taskId: string; conflict: string } } - | { type: 'conflict:resolved'; data: { taskId: string } } - | { type: 'conflict:failed'; data: { taskId: string; error: string } } - | { type: 'daemon:status'; data: { status: string; uptime: number } } - | { type: 'daemon:error'; data: { error: string } } - | { type: 'daemon:shutdown'; data: { reason: string } } - | { type: 'daemon:started'; data: { generation: string } } - | { type: 'system:info'; data: { message: string } }; - -// --- Buffers --- - -/** Max events retained for replay to new SSE clients. */ -const REPLAY_BUFFER_MAX = 500; -/** Max log events retained in memory. */ -const LOG_BUFFER_MAX = 2000; -/** Max stage events retained in memory. */ -const STAGE_BUFFER_MAX = 500; -/** Max chat events retained in memory. */ -const CHAT_BUFFER_MAX = 200; + | { type: 'heartbeat' }; -const replayBuffer: HubEvent[] = []; -const logBuffer: HubEvent[] = []; -const stageBuffer: HubEvent[] = []; -const chatBuffer: HubEvent[] = []; - -export function pushReplay(event: HubEvent): void { - replayBuffer.push(event); - if (replayBuffer.length > REPLAY_BUFFER_MAX) replayBuffer.shift(); -} - -// --- Hub --- +// Singleton const hub = new EventEmitter(); +hub.setMaxListeners(50); -export function getEventHub(): EventEmitter { - return hub; -} +const sseClients = new Set(); -// --- SSE Clients --- +// Ring buffer: replay last 500 events to new SSE clients +// Excludes high-frequency log lines (only last 50 logs kept) +const EVENT_REPLAY_MAX = 500; +const LOG_REPLAY_MAX = 50; +const replayBuffer: HubEvent[] = []; -const sseClients = new Set(); +// Per-type buffers for REST snapshot endpoints (dashboard refresh) +const LOG_BUFFER_MAX = 300; +const STAGE_BUFFER_MAX = 200; +const CHAT_BUFFER_MAX = 100; -// --- Backpressure (AGT-3429) --- -// res.write() returning false means the socket buffer is full. A consumer that -// never drains (dead dashboard tab, stalled reader) would otherwise make the -// hub queue megabytes per client. Track consecutive full writes and evict the -// client once it exceeds the threshold. +const logBuffer: HubEvent[] = []; +const stageBuffer: HubEvent[] = []; +const chatBuffer: HubEvent[] = []; +// --- Payload / backpressure bounds (AGT-3429) --- +/** Hard cap on one serialized SSE frame (bytes). Oversized events are truncated or dropped. */ +export const MAX_EVENT_PAYLOAD_BYTES = 64 * 1024; +/** Hard cap on a single log line retained/broadcast (chars). */ +export const MAX_LOG_LINE_CHARS = 4_000; +/** Hard cap on chat text retained/broadcast (chars). */ +export const MAX_CHAT_TEXT_CHARS = 16_384; /** Consecutive full-buffer writes tolerated before a client is disconnected. */ -const SSE_BACKPRESSURE_LIMIT = 64; +export const SSE_BACKPRESSURE_LIMIT = 64; /** Hard cap on bytes queued for one client before it is disconnected. */ -const SSE_MAX_BUFFERED_BYTES = 4 * 1024 * 1024; +export const SSE_MAX_BUFFERED_BYTES = 4 * 1024 * 1024; const backpressureCounts = new WeakMap(); +const bufferedBytes = new WeakMap(); function disconnectClient(res: ServerResponse): void { sseClients.delete(res); backpressureCounts.delete(res); + bufferedBytes.delete(res); try { res.destroy(); } catch { @@ -119,70 +172,226 @@ function disconnectClient(res: ServerResponse): void { } } -// --- Broadcast --- +function truncateChars(value: string, max: number): string { + if (value.length <= max) return value; + return `${value.slice(0, Math.max(0, max - 1))}…`; +} + +/** + * Bound attacker-/workload-controlled string fields before JSON serialization + * so a single event cannot exhaust process memory via retain or fan-out. + */ +function boundEventPayload(event: HubEvent): HubEvent { + switch (event.type) { + case 'log': + return { + ...event, + data: { ...event.data, line: truncateChars(event.data.line, MAX_LOG_LINE_CHARS) }, + }; + case 'chat:user': + case 'chat:agent': + return { + ...event, + data: { ...event.data, text: truncateChars(event.data.text, MAX_CHAT_TEXT_CHARS) }, + }; + case 'pipeline:stage': { + const d = event.data; + return { + ...event, + data: { + ...d, + summary: d.summary !== undefined ? truncateChars(d.summary, MAX_LOG_LINE_CHARS) : undefined, + feedback: d.feedback !== undefined ? truncateChars(d.feedback, MAX_LOG_LINE_CHARS) : undefined, + error: d.error !== undefined ? truncateChars(d.error, MAX_LOG_LINE_CHARS) : undefined, + haltReason: d.haltReason !== undefined ? truncateChars(d.haltReason, MAX_LOG_LINE_CHARS) : undefined, + changelogEntry: d.changelogEntry !== undefined ? truncateChars(d.changelogEntry, MAX_LOG_LINE_CHARS) : undefined, + filesChanged: d.filesChanged?.slice(0, 64).map((f) => truncateChars(f, 512)), + commands: d.commands?.slice(0, 64).map((c) => truncateChars(c, 512)), + issues: d.issues?.slice(0, 64).map((i) => truncateChars(i, 512)), + failedTests: d.failedTests?.slice(0, 64).map((t) => truncateChars(t, 512)), + }, + }; + } + case 'monitor:checked': + return { + ...event, + data: { + ...event.data, + output: event.data.output !== undefined + ? truncateChars(event.data.output, MAX_LOG_LINE_CHARS) + : undefined, + }, + }; + case 'conflict:failed': + return { + ...event, + data: { ...event.data, reason: truncateChars(event.data.reason, MAX_LOG_LINE_CHARS) }, + }; + case 'pipeline:escalation': + return { + ...event, + data: { + ...event.data, + reason: event.data.reason !== undefined + ? truncateChars(event.data.reason, MAX_LOG_LINE_CHARS) + : undefined, + }, + }; + default: + return event; + } +} + +/** Serialize an event for SSE; drop if still over the hard byte cap after field bounds. */ +function serializeEventFrame(event: HubEvent): string | null { + const bounded = boundEventPayload(event); + const frame = `data: ${JSON.stringify(bounded)}\n\n`; + if (Buffer.byteLength(frame, 'utf8') > MAX_EVENT_PAYLOAD_BYTES) { + return null; + } + return frame; +} + +function writeToClient(res: ServerResponse, data: string): void { + try { + const ok = res.write(data); + if (!ok) { + const nextCount = (backpressureCounts.get(res) ?? 0) + 1; + const nextBytes = (bufferedBytes.get(res) ?? 0) + Buffer.byteLength(data, 'utf8'); + if (nextCount >= SSE_BACKPRESSURE_LIMIT || nextBytes >= SSE_MAX_BUFFERED_BYTES) { + disconnectClient(res); + return; + } + backpressureCounts.set(res, nextCount); + bufferedBytes.set(res, nextBytes); + res.once('drain', () => { + backpressureCounts.set(res, 0); + bufferedBytes.set(res, 0); + }); + } else { + backpressureCounts.set(res, 0); + bufferedBytes.set(res, 0); + } + } catch { + disconnectClient(res); + } +} + +function pushReplay(event: HubEvent): void { + if (event.type === 'log') { + // Keep only recent log lines in replay buffer to avoid bloat + const logCount = replayBuffer.filter(e => e.type === 'log').length; + if (logCount >= LOG_REPLAY_MAX) { + const firstLogIdx = replayBuffer.findIndex(e => e.type === 'log'); + if (firstLogIdx !== -1) replayBuffer.splice(firstLogIdx, 1); + } + } + replayBuffer.push(event); + if (replayBuffer.length > EVENT_REPLAY_MAX) { + replayBuffer.shift(); + } +} + +// Exports + +export function getEventHub(): EventEmitter { + return hub; +} export function broadcastEvent(event: HubEvent): void { - // Push to replay buffer - pushReplay(event); + // Bound payload fields before any retain / serialize / fan-out. + const bounded = boundEventPayload(event); + // Mutate the caller's event when it is a log so stamped ts/seq remain visible + // on the same object emitters may hold (INT-3402 tests rely on this). + if (event.type === 'log' && bounded.type === 'log') { + event.data.line = bounded.data.line; + } else if ( + (event.type === 'chat:user' || event.type === 'chat:agent') + && (bounded.type === 'chat:user' || bounded.type === 'chat:agent') + ) { + event.data.text = bounded.data.text; + } - // Push to typed buffers - switch (event.type) { - case 'task:log': - logBuffer.push(event); + const overCap = bounded.type !== 'heartbeat' + && Buffer.byteLength(JSON.stringify(bounded), 'utf8') > MAX_EVENT_PAYLOAD_BYTES; + + // Skip replaying heartbeat/stats to avoid noise on reconnect + if (bounded.type !== 'heartbeat' && !overCap) { + pushReplay(bounded); + } + // Per-task transcript rings for the cockpit (INT-3402). Fed here — the one + // choke point every emitter already goes through — so no broadcast site + // changes. task:started/completed drive the retention lifecycle. + if (bounded.type === 'log' && event.type === 'log') { + // The ring and the SSE copy of this line carry the SAME ts AND sequence, + // which is what lets a client merge a REST transcript snapshot with lines + // that streamed in while the request was in flight. The sequence — not the + // millisecond — is the join key: an agent emits several lines per ms. + // (INT-3402) + const ts = Date.now(); + event.data.ts = ts; + bounded.data.ts = ts; + const seq = appendTaskLog(event.data.taskId, event.data.stage, event.data.line, ts); + event.data.seq = seq; + bounded.data.seq = seq; + // Which process the sequence belongs to. Carried ON the line so a client + // needs no separate round trip (and no ordering luck) to notice a restart. + const gen = daemonGeneration(); + event.data.gen = gen; + bounded.data.gen = gen; + } else if (bounded.type === 'task:started') { + cancelTaskLogCleanup(bounded.data.taskId); + } else if (bounded.type === 'task:completed') { + scheduleTaskLogCleanup(bounded.data.taskId); + } + if (overCap) return; + // Per-type buffers for REST snapshot + switch (bounded.type) { + case 'log': + logBuffer.push(bounded); if (logBuffer.length > LOG_BUFFER_MAX) logBuffer.shift(); break; - case 'task:stage': - case 'task:progress': + case 'pipeline:stage': + case 'pipeline:iteration': + case 'pipeline:escalation': + case 'pipeline:fanout': + case 'task:queued': + case 'task:started': + case 'task:completed': case 'task:cost': - case 'task:monitor': - case 'agent:status': - case 'agent:thinking': - case 'agent:error': - case 'agent:done': + case 'monitor:checked': + case 'monitor:stateChange': + case 'process:spawn': + case 'process:exit': case 'conflict:detected': + case 'conflict:resolving': case 'conflict:resolved': case 'conflict:failed': case 'coordination:event': - stageBuffer.push(event); + stageBuffer.push(bounded); if (stageBuffer.length > STAGE_BUFFER_MAX) stageBuffer.shift(); break; case 'chat:user': case 'chat:agent': - chatBuffer.push(event); + chatBuffer.push(bounded); if (chatBuffer.length > CHAT_BUFFER_MAX) chatBuffer.shift(); break; } - const data = `data: ${JSON.stringify(event)}\n\n`; + const data = serializeEventFrame(bounded); + if (data === null) return; for (const res of sseClients) { - try { - const ok = res.write(data); - if (!ok) { - // write() returned false — socket buffer is full. Track and evict if - // the client exceeds the backpressure threshold. - const count = (backpressureCounts.get(res) ?? 0) + 1; - if (count >= SSE_BACKPRESSURE_LIMIT) { - disconnectClient(res); - } else { - backpressureCounts.set(res, count); - } - } else { - // Successful write — reset backpressure counter for this client. - backpressureCounts.set(res, 0); - } - } catch { - disconnectClient(res); - } + writeToClient(res, data); } } -// --- SSE Client Management --- - export function addSSEClient(res: ServerResponse, skipReplay = false): () => void { // Replay buffered events to new client so they see current state if (!skipReplay && replayBuffer.length > 0) { try { for (const event of replayBuffer) { - res.write(`data: ${JSON.stringify(event)}\n\n`); + const frame = serializeEventFrame(event); + if (frame === null) continue; + res.write(frame); } } catch { // A client that cannot consume replay is already gone. Do not retain it @@ -192,11 +401,13 @@ export function addSSEClient(res: ServerResponse, skipReplay = false): () => voi } sseClients.add(res); backpressureCounts.set(res, 0); + bufferedBytes.set(res, 0); // Cleanup function that removes client from set const cleanup = () => { sseClients.delete(res); backpressureCounts.delete(res); + bufferedBytes.delete(res); // Remove the close listener after cleanup to prevent memory leak res.removeListener('close', cleanup); }; @@ -235,4 +446,4 @@ export function __resetForTests(): void { chatBuffer.length = 0; // Clear all event listeners on the hub hub.removeAllListeners(); -} \ No newline at end of file +} diff --git a/src/discord/discordPair.ts b/src/discord/discordPair.ts index a859acdf..4be3e372 100644 --- a/src/discord/discordPair.ts +++ b/src/discord/discordPair.ts @@ -23,6 +23,10 @@ import { } from './discordCore.js'; import { t, getDateLocale } from '../locale/index.js'; import { safeConsole as console } from '../support/safeLog.js'; +import { sanitizeTerminalText } from '../tui/sanitize.js'; + +/** Hard cap on a Discord pair-thread text message (chars). */ +const MAX_PAIR_REPORT_CHARS = 4096; /** * !pair command handler @@ -189,16 +193,14 @@ async function startPairSession( /** * Truncate and neutralize a worker report string before posting to Discord. - * Caps total length at 4096 characters and strips content that could be - * attacker-controlled or excessively verbose. + * Strips terminal/control sequences first, then caps total length so + * attacker-controlled or excessively verbose output cannot exhaust memory + * or break the Discord client. */ function sanitizeReport(report: string): string { - // Hard cap at 4096 characters (Discord embed field limit is 1024, but - // thread.send accepts longer text; 4096 is a safe bound for a single message). - if (report.length > 4096) { - report = report.slice(0, 4093) + '...'; - } - return report; + const neutralized = sanitizeTerminalText(report); + if (neutralized.length <= MAX_PAIR_REPORT_CHARS) return neutralized; + return `${neutralized.slice(0, MAX_PAIR_REPORT_CHARS - 1)}…`; } /** diff --git a/src/support/chatBackend.ts b/src/support/chatBackend.ts index 6aab09ad..90fe0bc1 100644 --- a/src/support/chatBackend.ts +++ b/src/support/chatBackend.ts @@ -348,9 +348,13 @@ async function runChatViaAdapter( 'Chat response cancelled', ); if (raw.exitCode !== 0 && !raw.stdout.trim()) { - throw new Error(raw.stderr.trim() || `${provider} exited with code ${raw.exitCode}`); + throw new Error(raw.stderr.trim().slice(0, 256 * 1024) || `${provider} exited with code ${raw.exitCode}`); } - const text = raw.stdout.trim(); + // Bound retained adapter stdout before returning to chat UI (AGT-3429). + const MAX_ADAPTER_STDOUT_CHARS = 1024 * 1024; + const text = raw.stdout.length > MAX_ADAPTER_STDOUT_CHARS + ? raw.stdout.slice(0, MAX_ADAPTER_STDOUT_CHARS).trim() + : raw.stdout.trim(); // Non-streaming adapters emit nothing via onToken — flush the full reply once. if (!streamed) options.onText?.(text, false); return { response: text || '[No response]', provider, model }; @@ -453,6 +457,11 @@ export async function runChatCompletion(options: ChatCompletionOptions): Promise proc.stdin?.end(stdin); } + // Hard caps on retained CLI chat output (AGT-3429). Oversized streams are + // truncated in place so a runaway subprocess cannot exhaust process memory. + const MAX_CHAT_STDOUT_CHARS = 1024 * 1024; // 1 MiB + const MAX_CHAT_STDERR_CHARS = 256 * 1024; // 256 KiB + const MAX_CHAT_PARTIAL_BUFFER_CHARS = 64 * 1024; // 64 KiB let stdout = ''; let stderr = ''; let buffer = ''; @@ -461,6 +470,12 @@ export async function runChatCompletion(options: ChatCompletionOptions): Promise let thinkingTimer: NodeJS.Timeout | null = null; let settled = false; + const appendBounded = (current: string, chunk: string, max: number): string => { + if (current.length >= max) return current; + const room = max - current.length; + return room >= chunk.length ? current + chunk : current + chunk.slice(0, room); + }; + const cleanupProcessHooks = () => { if (thinkingTimer) clearTimeout(thinkingTimer); runSignal.removeEventListener('abort', onAbort); @@ -529,13 +544,13 @@ export async function runChatCompletion(options: ChatCompletionOptions): Promise proc.stdout?.on('data', (chunk: Buffer) => { const text = chunk.toString(); - stdout += text; - buffer += text; + stdout = appendBounded(stdout, text, MAX_CHAT_STDOUT_CHARS); + buffer = appendBounded(buffer, text, MAX_CHAT_PARTIAL_BUFFER_CHARS); flushLines(false); }); proc.stderr?.on('data', (chunk: Buffer) => { - stderr += chunk.toString(); + stderr = appendBounded(stderr, chunk.toString(), MAX_CHAT_STDERR_CHARS); }); proc.on('close', (code) => { diff --git a/src/tui/components/LogLine.test.ts b/src/tui/components/LogLine.test.ts new file mode 100644 index 00000000..2170d7c5 --- /dev/null +++ b/src/tui/components/LogLine.test.ts @@ -0,0 +1,19 @@ +import { describe, it, expect } from 'vitest'; +import { MAX_LOG_LINE_CHARS, prepareLogLine } from './LogLine.js'; + +describe('prepareLogLine (AGT-3429)', () => { + it('flattens newlines and tabs before render', () => { + expect(prepareLogLine('a\nb\tc\r\nd')).toBe('a b c d'); + }); + + it('bounds oversized lines to the documented hard cap', () => { + const prepared = prepareLogLine('x'.repeat(MAX_LOG_LINE_CHARS + 500)); + expect(prepared.length).toBeLessThanOrEqual(MAX_LOG_LINE_CHARS); + expect(prepared.endsWith('…')).toBe(true); + }); + + it('strips terminal escape sequences while preserving readable text', () => { + const esc = String.fromCharCode(27); + expect(prepareLogLine(`${esc}[31mred${esc}[0m plain`)).toBe('red plain'); + }); +}); diff --git a/src/tui/components/LogLine.tsx b/src/tui/components/LogLine.tsx index dc12b5f6..cfbb184c 100644 --- a/src/tui/components/LogLine.tsx +++ b/src/tui/components/LogLine.tsx @@ -3,10 +3,25 @@ import { Text } from 'ink'; import { parseLogLine } from '../logFormat.js'; import { sanitizeTerminalText } from '../sanitize.js'; +/** Hard cap on a single rendered daemon log line (chars). */ +export const MAX_LOG_LINE_CHARS = 4_000; + +/** + * Flatten newlines/tabs and bound length before sanitization/render so a + * malicious or oversized daemon log event cannot blow up Ink layout memory. + */ +export function prepareLogLine(line: string): string { + const flattened = line.replace(/\r\n|\r|\n/g, ' ').replace(/\t/g, ' '); + const bounded = flattened.length > MAX_LOG_LINE_CHARS + ? `${flattened.slice(0, MAX_LOG_LINE_CHARS - 1)}…` + : flattened; + return sanitizeTerminalText(bounded); +} + export function LogLine({ line }: { line: string }) { return ( - {parseLogLine(sanitizeTerminalText(line)).map((s, i) => ( + {parseLogLine(prepareLogLine(line)).map((s, i) => ( {s.text} diff --git a/tmp-hook-test.txt b/tmp-hook-test.txt new file mode 100644 index 00000000..a71b7fbb --- /dev/null +++ b/tmp-hook-test.txt @@ -0,0 +1 @@ +test marker From 92fbde8da2c265040d44d5c78886649c7dd9de29 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 09:44:48 +0900 Subject: [PATCH 11/14] wip: preserved partial work (auto, session did not succeed) --- src/.agt3429-run-vitest.sh | 23 +++ src/.agt3429-shell-blocked.txt | 0 src/.agt3429-test-result.txt | 0 src/adapters/chatStream.test.ts | 15 +- src/adapters/chatStream.ts | 30 +++- src/adapters/codexResponses.test.ts | 11 ++ src/adapters/codexResponses.ts | 8 +- src/runners/.cliRunner-from-main.ts | 0 src/runners/cliRunner.ts | 230 ++++++++++++++++++---------- 9 files changed, 223 insertions(+), 94 deletions(-) create mode 100644 src/.agt3429-run-vitest.sh create mode 100644 src/.agt3429-shell-blocked.txt create mode 100644 src/.agt3429-test-result.txt create mode 100644 src/runners/.cliRunner-from-main.ts diff --git a/src/.agt3429-run-vitest.sh b/src/.agt3429-run-vitest.sh new file mode 100644 index 00000000..2139cd31 --- /dev/null +++ b/src/.agt3429-run-vitest.sh @@ -0,0 +1,23 @@ +#!/bin/bash +set -euo pipefail +cd /work/OpenSwarm/worktree/ec3f1416-ae3c-423f-9952-a8adee496b79 +OUT=src/.agt3429-vitest-out.txt +{ + echo "=== start $(date -Iseconds) ===" + if [ ! -f ./node_modules/vitest/vitest.mjs ] && [ ! -f /work/OpenSwarm/node_modules/vitest/vitest.mjs ]; then + echo "=== npm install vitest ===" + /usr/local/bin/npm install vitest@^4.0.18 @vitest/coverage-v8@^4.0.18 --no-fund --no-audit --save-dev || /usr/local/bin/npm install --no-fund --no-audit + fi + VITEST=/work/OpenSwarm/node_modules/vitest/vitest.mjs + if [ ! -f "$VITEST" ]; then VITEST=./node_modules/vitest/vitest.mjs; fi + echo "=== using VITEST=$VITEST ===" + /usr/local/bin/node --experimental-vm-modules "$VITEST" run --reporter=verbose \ + src/runners/cliRunner.test.ts \ + src/core/eventHub.test.ts \ + src/tui/components/LogLine.test.ts \ + src/adapters/chatStream.test.ts \ + src/adapters/codexResponses.test.ts + echo EXIT:$? +} 2>&1 | tee "$OUT" +# ensure EXIT line exists even if tee nested oddly +if ! grep -q '^EXIT:' "$OUT"; then echo EXIT:1 >> "$OUT"; fi diff --git a/src/.agt3429-shell-blocked.txt b/src/.agt3429-shell-blocked.txt new file mode 100644 index 00000000..e69de29b diff --git a/src/.agt3429-test-result.txt b/src/.agt3429-test-result.txt new file mode 100644 index 00000000..e69de29b diff --git a/src/adapters/chatStream.test.ts b/src/adapters/chatStream.test.ts index ace64628..065404ff 100644 --- a/src/adapters/chatStream.test.ts +++ b/src/adapters/chatStream.test.ts @@ -1,5 +1,18 @@ import { describe, it, expect, vi } from 'vitest'; -import { reduceChatChunks } from './chatStream.js'; +import { + MAX_CHAT_CHUNKS, + MAX_PARTIAL_FRAME_CHARS, + MAX_RETAINED_CONTENT_CHARS, + reduceChatChunks, +} from './chatStream.js'; + +describe('chat stream bounds (AGT-3429)', () => { + it('documents hard caps for partial frames, chunks, and retained content', () => { + expect(MAX_PARTIAL_FRAME_CHARS).toBe(64 * 1024); + expect(MAX_CHAT_CHUNKS).toBe(1024); + expect(MAX_RETAINED_CONTENT_CHARS).toBe(1024 * 1024); + }); +}); describe('reduceChatChunks', () => { it('accumulates content deltas and emits each via onToken in order', () => { diff --git a/src/adapters/chatStream.ts b/src/adapters/chatStream.ts index 4e548c57..198e8eca 100644 --- a/src/adapters/chatStream.ts +++ b/src/adapters/chatStream.ts @@ -101,9 +101,11 @@ function parseChunkLine(line: string): StreamChunk | null { } /** Hard cap for retained partial-frame data in the SSE buffer (64 KB). */ -const MAX_CONTENT_LENGTH = 64 * 1024; -/** Hard cap on accumulated parsed chunks (1024). */ -const MAX_CHAT_CHUNKS = 1024; +export const MAX_PARTIAL_FRAME_CHARS = 64 * 1024; +/** Hard cap on accumulated parsed chunks retained for final reduce. */ +export const MAX_CHAT_CHUNKS = 1024; +/** Hard cap on retained assistant content assembled from the stream (1 MiB). */ +export const MAX_RETAINED_CONTENT_CHARS = 1024 * 1024; /** Read a chat/completions SSE body and reduce it, emitting content deltas live. */ export async function consumeChatCompletionsStream( @@ -116,10 +118,19 @@ export async function consumeChatCompletionsStream( const chunks: StreamChunk[] = []; const decoder = new TextDecoder(); let buffer = ''; + let retainedContentChars = 0; const handle = (c: StreamChunk | null) => { if (!c) return; const delta = c.choices?.[0]?.delta?.content; - if (onToken && typeof delta === 'string' && delta) onToken(delta); + if (onToken && typeof delta === 'string' && delta) { + // Still emit live tokens, but stop retaining more content beyond the hard cap. + if (retainedContentChars < MAX_RETAINED_CONTENT_CHARS) { + const room = MAX_RETAINED_CONTENT_CHARS - retainedContentChars; + const emit = delta.length <= room ? delta : delta.slice(0, room); + retainedContentChars += emit.length; + onToken(emit); + } + } // Enforce hard cap on retained chunks to prevent memory exhaustion if (chunks.length >= MAX_CHAT_CHUNKS) { chunks.shift(); @@ -131,8 +142,8 @@ export async function consumeChatCompletionsStream( if (done) break; buffer += decoder.decode(value, { stream: true }); // Enforce hard cap on partial-frame buffer to prevent memory exhaustion - if (buffer.length > MAX_CONTENT_LENGTH) { - buffer = buffer.slice(-MAX_CONTENT_LENGTH); + if (buffer.length > MAX_PARTIAL_FRAME_CHARS) { + buffer = buffer.slice(-MAX_PARTIAL_FRAME_CHARS); } const lines = buffer.split('\n'); buffer = lines.pop() ?? ''; @@ -141,5 +152,10 @@ export async function consumeChatCompletionsStream( handle(parseChunkLine(buffer)); // Final reduce WITHOUT onToken (already emitted above) to assemble the result. - return reduceChatChunks(chunks); + const reduced = reduceChatChunks(chunks); + const content = reduced.choices[0]?.message.content; + if (typeof content === 'string' && content.length > MAX_RETAINED_CONTENT_CHARS) { + reduced.choices[0].message.content = content.slice(0, MAX_RETAINED_CONTENT_CHARS); + } + return reduced; } \ No newline at end of file diff --git a/src/adapters/codexResponses.test.ts b/src/adapters/codexResponses.test.ts index d0483aa4..e9f66d43 100644 --- a/src/adapters/codexResponses.test.ts +++ b/src/adapters/codexResponses.test.ts @@ -9,6 +9,9 @@ import { reduceResponsesEvents, resolveReasoningEffort, selectDefaultCodexResponseModel, + MAX_FRAME_LENGTH, + MAX_RETAINED_EVENTS, + MAX_REASONING_BUF, } from './codexResponses.js'; import { runAgenticLoop, type ChatMessage } from './agenticLoop.js'; import { RateLimitError } from './rateLimitError.js'; @@ -18,6 +21,14 @@ afterEach(() => { vi.unstubAllGlobals(); }); +describe('Codex Responses SSE retention bounds (AGT-3429)', () => { + it('documents hard caps for partial frames, retained events, and reasoning buffers', () => { + expect(MAX_FRAME_LENGTH).toBe(64 * 1024); + expect(MAX_RETAINED_EVENTS).toBe(4096); + expect(MAX_REASONING_BUF).toBe(64 * 1024); + }); +}); + describe('parseWorkerOutput command backfill', () => { const adapter = new CodexResponsesAdapter(); diff --git a/src/adapters/codexResponses.ts b/src/adapters/codexResponses.ts index acb02632..2243c21c 100644 --- a/src/adapters/codexResponses.ts +++ b/src/adapters/codexResponses.ts @@ -213,9 +213,11 @@ export function reduceResponsesEvents(events: SseEvent[]): ChatLikeResponse { } /** Hard cap for retained partial-frame data in the SSE buffer (64 KB). */ -const MAX_FRAME_LENGTH = 64 * 1024; +export const MAX_FRAME_LENGTH = 64 * 1024; /** Hard cap on retained parsed SSE events before reduceResponsesEvents (4096). */ -const MAX_RETAINED_EVENTS = 4096; +export const MAX_RETAINED_EVENTS = 4096; +/** Hard cap on retained reasoning partial text (64 KB). */ +export const MAX_REASONING_BUF = 64 * 1024; /** Parse a `data: {json}` SSE line into an event, or null for keep-alives/[DONE]. */ function parseSseLine(line: string): SseEvent | null { @@ -248,8 +250,6 @@ async function consumeResponsesStream( let buffer = ''; // Reasoning summary streams token-by-token; buffer and emit whole lines so the // live log shows readable thoughts instead of one-word-per-line spam. - /** Hard cap on retained reasoning partial text (64 KB). */ - const MAX_REASONING_BUF = 64 * 1024; let reasoningBuf = ''; const flushReasoning = (force: boolean) => { if (!onReasoning) { reasoningBuf = ''; return; } diff --git a/src/runners/.cliRunner-from-main.ts b/src/runners/.cliRunner-from-main.ts new file mode 100644 index 00000000..e69de29b diff --git a/src/runners/cliRunner.ts b/src/runners/cliRunner.ts index e18397fc..c899aa2f 100644 --- a/src/runners/cliRunner.ts +++ b/src/runners/cliRunner.ts @@ -49,76 +49,118 @@ function validateMaxIterations(value: number | undefined): number { return maxIterations; } +/** Format duration as human-readable string */ function formatDuration(ms: number): string { - const seconds = Math.floor(ms / 1000); + if (ms < 1000) return `${ms}ms`; + const seconds = ms / 1000; + if (seconds < 60) return `${seconds.toFixed(1)}s`; const minutes = Math.floor(seconds / 60); - const hours = Math.floor(minutes / 60); - if (hours > 0) return `${hours}h ${minutes % 60}m ${seconds % 60}s`; - if (minutes > 0) return `${minutes}m ${seconds % 60}s`; - return `${seconds}s`; + const remaining = seconds % 60; + return `${minutes}m ${remaining.toFixed(0)}s`; } -/** Hard cap on bytes per line before sanitization (1 MB). */ -const MAX_LINE_BYTES = 1048576; +/** Hard cap on raw chars per log/path entry before sanitization (AGT-3429). */ +export const MAX_LINE_CHARS = 1_048_576; +/** Hard cap on changed-file entries shown in the CLI result (AGT-3429). */ +export const MAX_FILES_SHOWN = 5; /** - * Truncate a raw line to MAX_LINE_BYTES before sanitization to prevent - * memory exhaustion from attacker-controlled or excessively verbose output. + * Truncate raw attacker-/workload-controlled input before sanitization so a + * huge string cannot exhaust memory during neutralize/render. */ -function truncateLine(raw: string): string { - if (raw.length > MAX_LINE_BYTES) { - return raw.slice(0, MAX_LINE_BYTES) + '... [truncated]'; - } - return raw; +function truncateRaw(raw: string, max = MAX_LINE_CHARS): string { + if (raw.length <= max) return raw; + return `${raw.slice(0, max)}... [truncated]`; } -/** - * Run the CLI pipeline - */ -export async function runCli(options: CliRunOptions): Promise { - const { task, projectPath } = options; +/** Bound then neutralize a user-visible string. */ +function safeText(raw: string): string { + return sanitizeTerminalText(truncateRaw(raw)); +} + +// Main Runner - // 1. Validate adapter availability - const adapterAvailable = await checkDefaultAdapter(); - if (!adapterAvailable) { - console.error('Error: No adapter available. Please configure an adapter first.'); +export async function runCli(options: CliRunOptions): Promise { + // Initialize locale (needed for prompt templates) + initLocale('en'); + + // 1. Check configured/default adapter + if (!await checkDefaultAdapter()) { + const adapterName = getDefaultAdapterName(); + const availableAdapters = await listAvailableAdapters(); + console.error(`Error: CLI adapter "${adapterName}" is not available.`); + console.error( + availableAdapters.length > 0 + ? `Available adapters: ${availableAdapters.join(', ')}` + : 'No registered adapters are currently available.' + ); process.exit(1); } - // 2. Validate project path - const resolvedPath = expandPath(projectPath || process.cwd()); + // 2. Resolve project path + const projectPath = expandPath(options.projectPath ?? process.cwd(), true); + let projectStats: ReturnType; + try { + projectStats = statSync(projectPath); + } catch (error) { + const code = (error as NodeJS.ErrnoException).code; + console.error( + code === 'ENOENT' + ? `Error: Project path does not exist: ${projectPath}` + : `Error: Project path is not accessible: ${projectPath}` + ); + process.exit(1); + } + if (!projectStats.isDirectory()) { + console.error(`Error: Project path is not a directory: ${projectPath}`); + process.exit(1); + } try { - accessSync(resolvedPath, constants.R_OK); + accessSync(projectPath, constants.R_OK | constants.X_OK); } catch { - console.error(`Error: Cannot access project path: ${resolvedPath}`); + console.error(`Error: Project path is not accessible: ${projectPath}`); process.exit(1); } - // 3. Validate max iterations - const maxIterations = validateMaxIterations(options.maxIterations); - - // 4. Initialize locale - initLocale(); - - // 5. Build pipeline stages - const stages: PipelineStage[] = []; - if (options.pipeline) { - stages.push('worker', 'reviewer'); - } else if (options.workerOnly) { - stages.push('worker'); + // 3. Determine stages + let stages: PipelineStage[]; + if (options.workerOnly) { + stages = ['worker']; + } else if (options.pipeline) { + stages = ['worker', 'reviewer', 'tester', 'documenter']; } else { - stages.push('worker', 'reviewer'); + stages = ['worker', 'reviewer']; } - // 6. Build role configs - const roleConfigs: RoleConfig[] = stages.map((stage) => ({ - stage, - model: options.model, - })); + // 4. Build role config + const roles: Record = {}; + if (options.model) { + roles.worker = { enabled: true, model: options.model, timeoutMs: 0 }; + } + + // 5. Create local TaskItem + const task: TaskItem = { + id: `cli-${Date.now()}`, + source: 'local', + title: options.task, + description: options.task, + priority: 3, + projectPath, + createdAt: Date.now(), + }; + + // 6. Create pipeline + const maxIterations = validateMaxIterations(options.maxIterations); + const pipeline = new PairPipeline({ + stages, + maxIterations, + roles: Object.keys(roles).length > 0 ? roles as any : undefined, + verbose: options.verbose, + }); // 7. Print header const stageNames = stages.join(' -> '); - const shortPath = resolvedPath.replace(homedir(), '~'); + const shortPath = projectPath.replace(homedir(), '~'); console.log(''); console.log(' OpenSwarm v0.1.0'); console.log(''); @@ -144,31 +186,25 @@ export async function runCli(options: CliRunOptions): Promise { heartbeat = null; }; - const pipeline = new PairPipeline({ - stages: roleConfigs, - maxIterations, - verbose: options.verbose, - learn: options.learn, - }); - pipeline.on('stage:start', ({ stage }: { stage: string }) => { - stage = sanitizeTerminalText(truncateLine(stage)); + stage = safeText(stage); if (liveSpinner) heartbeat = startProgressHeartbeat(`${stage}…`, { write: (s) => process.stdout.write(s) }); else process.stdout.write(` ~ ${stage}...\n`); }); pipeline.on('stage:complete', ({ stage, result }: { stage: string; result: { success: boolean; duration: number } }) => { - stage = sanitizeTerminalText(truncateLine(stage)); + stage = safeText(stage); stopHeartbeat(); - const icon = result.success ? status.check : status.fail; - const dur = formatDuration(result.duration); - process.stdout.write(` ${icon} ${stage} (${dur})\n`); + const duration = (result.duration / 1000).toFixed(1); + const line = `${stage} (${duration}s)`; + process.stdout.write(` ${result.success ? status.ok(line) : status.err(line)}\n`); }); - pipeline.on('stage:fail', ({ stage, error }: { stage: string; error: string }) => { - stage = sanitizeTerminalText(truncateLine(stage)); + pipeline.on('stage:fail', ({ stage, result }: { stage: string; result: { duration: number } }) => { + stage = safeText(stage); stopHeartbeat(); - process.stdout.write(` ${status.fail} ${stage}: ${sanitizeTerminalText(truncateLine(error))}\n`); + const duration = (result.duration / 1000).toFixed(1); + process.stdout.write(` ${status.err(`${stage} (${duration}s) FAILED`)}\n`); }); pipeline.on('iteration:start', ({ iteration, maxIterations }: { iteration: number; maxIterations: number }) => { @@ -177,22 +213,22 @@ export async function runCli(options: CliRunOptions): Promise { } }); - // 8.5. Verbose event listeners + // 8.5. Verbose event listeners — bound raw input before sanitize (AGT-3429). if (options.verbose) { pipeline.on('log', ({ line }: { line: string }) => { - console.log(` ${sanitizeTerminalText(truncateLine(line))}`); + console.log(` ${safeText(line)}`); }); pipeline.on('halt', ({ reason, sessionId }: { reason: string; sessionId: string }) => { - console.log(` [verbose] HALT: ${sanitizeTerminalText(truncateLine(reason))} (session: ${sanitizeTerminalText(truncateLine(sessionId))})`); + console.log(` [verbose] HALT: ${safeText(reason)} (session: ${safeText(sessionId)})`); }); pipeline.on('stuck', ({ sessionId, iteration }: { sessionId: string; iteration: number }) => { - console.log(` [verbose] STUCK detected at iteration ${iteration} (session: ${sanitizeTerminalText(truncateLine(sessionId))})`); + console.log(` [verbose] STUCK detected at iteration ${iteration} (session: ${safeText(sessionId)})`); }); pipeline.on('iteration:fail', ({ iteration, reason }: { iteration: number; reason?: string }) => { - console.log(` [verbose] Iteration ${iteration} failed${reason ? `: ${sanitizeTerminalText(truncateLine(reason))}` : ''}`); + console.log(` [verbose] Iteration ${iteration} failed${reason ? `: ${safeText(reason)}` : ''}`); }); pipeline.on('iteration:complete', ({ iteration }: { iteration: number }) => { @@ -203,36 +239,66 @@ export async function runCli(options: CliRunOptions): Promise { // 9. Run pipeline let result: PipelineResult; try { - result = await pipeline.run(task, resolvedPath); - } catch (err) { + result = await pipeline.run(task, projectPath); + } catch (error) { stopHeartbeat(); - console.error('Pipeline execution failed:', err); - process.exit(1); + console.error('\n Pipeline execution failed:', error instanceof Error ? error.message : error); + process.exitCode = 1; + return; } - stopHeartbeat(); - - // 10. Print result + // 10. Format & print result printResult(result); + + // 10.5. Learn: record the outcome into repo knowledge so a standalone `run` + // grows the codebase memory like the daemon does (default on; --no-learn opts + // out for throwaway/exploratory runs). Non-critical. (INT-2268) + if (options.learn !== false) { + try { + const { recordTaskOutcome } = await import('../memory/repoKnowledge.js'); + await recordTaskOutcome(projectPath, { + taskTitle: options.task, + workerResult: result.workerResult + ? { filesChanged: result.workerResult.filesChanged, commands: result.workerResult.commands, summary: result.workerResult.summary } + : null, + rejectionFeedback: result.finalStatus === 'rejected' ? result.reviewResult?.feedback : undefined, + iterations: result.iterations, + derivedFrom: 'cli:run', + }); + } catch { + // recordTaskOutcome is already non-throwing; belt-and-suspenders. + } + } + + // 11. Exit code + process.exitCode = result.success ? 0 : 1; } -/** - * Print pipeline result - */ +// Result Formatting + function printResult(result: PipelineResult): void { console.log(''); console.log(' ======================================'); - console.log(` ${result.success ? status.check : status.fail} Result: ${result.success ? 'Success' : 'Failed'}`); + + const statusLabel = result.finalStatus.toUpperCase(); + console.log(` Result: ${statusLabel}`); + + console.log(' ======================================'); // Summary if (result.workerResult?.summary) { - console.log(` Summary: ${sanitizeTerminalText(truncateLine(result.workerResult.summary))}`); + console.log(` Summary: ${safeText(result.workerResult.summary)}`); } - // Files changed + // Files changed — per-entry + count bounds before sanitize (AGT-3429). if (result.workerResult?.filesChanged && result.workerResult.filesChanged.length > 0) { const files = result.workerResult.filesChanged; - console.log(` Files: ${files.map((f) => sanitizeTerminalText(truncateLine(f))).join(', ')}`); + const shown = files.slice(0, MAX_FILES_SHOWN).map(safeText); + if (files.length <= MAX_FILES_SHOWN) { + console.log(` Files: ${shown.join(', ')}`); + } else { + console.log(` Files: ${shown.join(', ')} +${files.length - MAX_FILES_SHOWN} more`); + } } // Cost and duration @@ -249,10 +315,10 @@ function printResult(result: PipelineResult): void { console.log(' Feedback:'); const lines = result.reviewResult.feedback.split('\n').slice(0, 5); for (const line of lines) { - console.log(` ${line}`); + console.log(` ${safeText(line)}`); } } console.log(' ======================================'); console.log(''); -} \ No newline at end of file +} From 0f793bfb312d9932e099972cc1022b18862e56f3 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 10:06:54 +0900 Subject: [PATCH 12/14] wip: preserved partial work (auto, session did not succeed) --- node_modules | 1 - package-lock.json | 48 ----------------------------------------------- 2 files changed, 49 deletions(-) delete mode 120000 node_modules diff --git a/node_modules b/node_modules deleted file mode 120000 index d9643ec8..00000000 --- a/node_modules +++ /dev/null @@ -1 +0,0 @@ -/work/OpenSwarm/node_modules \ No newline at end of file diff --git a/package-lock.json b/package-lock.json index ba2b8fa8..5713a7b7 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1300,9 +1300,6 @@ "cpu": [ "arm" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1319,9 +1316,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1338,9 +1332,6 @@ "cpu": [ "ppc64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1357,9 +1348,6 @@ "cpu": [ "riscv64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1376,9 +1364,6 @@ "cpu": [ "s390x" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1395,9 +1380,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1414,9 +1396,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1433,9 +1412,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -1452,9 +1428,6 @@ "cpu": [ "arm" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1477,9 +1450,6 @@ "cpu": [ "arm64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1502,9 +1472,6 @@ "cpu": [ "ppc64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1527,9 +1494,6 @@ "cpu": [ "riscv64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1552,9 +1516,6 @@ "cpu": [ "s390x" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1577,9 +1538,6 @@ "cpu": [ "x64" ], - "libc": [ - "glibc" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1602,9 +1560,6 @@ "cpu": [ "arm64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1627,9 +1582,6 @@ "cpu": [ "x64" ], - "libc": [ - "musl" - ], "license": "Apache-2.0", "optional": true, "os": [ From c28b4ae44e80e21314195366cc01c93ee669a28a Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 10:46:47 +0900 Subject: [PATCH 13/14] wip: preserved partial work (auto, session did not succeed) --- node_modules | 1 + 1 file changed, 1 insertion(+) create mode 120000 node_modules diff --git a/node_modules b/node_modules new file mode 120000 index 00000000..d9643ec8 --- /dev/null +++ b/node_modules @@ -0,0 +1 @@ +/work/OpenSwarm/node_modules \ No newline at end of file From 5930aeb485860d8fb26addac6f786638db01e715 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 24 Sep 2026 00:30:42 +0900 Subject: [PATCH 14/14] wip: remove ephemeral runtime artifacts (auto) --- .run-vitest.sh | 24 ------------------------ cli.json | 8 -------- cursor/cli.json | 8 -------- node_modules | 1 - 4 files changed, 41 deletions(-) delete mode 100644 .run-vitest.sh delete mode 100644 cli.json delete mode 100644 cursor/cli.json delete mode 120000 node_modules diff --git a/.run-vitest.sh b/.run-vitest.sh deleted file mode 100644 index 574120b4..00000000 --- a/.run-vitest.sh +++ /dev/null @@ -1,24 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail -cd "$(dirname "$0")" -echo "=== diagnose ===" -command -v node npm || true -node -v || true -npm -v || true -ls -la node_modules 2>&1 | head -10 -find /work/OpenSwarm -maxdepth 3 -name 'vitest.mjs' 2>/dev/null | head -find /home -maxdepth 4 -name 'vitest.mjs' 2>/dev/null | head -echo "=== ensure deps ===" -if [[ ! -f node_modules/vitest/vitest.mjs && ! -f /work/OpenSwarm/node_modules/vitest/vitest.mjs ]]; then - npm ci || npm install -fi -echo "=== vitest ===" -node --experimental-vm-modules node_modules/vitest/vitest.mjs run \ - --reporter=verbose \ - src/core/eventHub.test.ts \ - src/tui/components/LogLine.test.ts \ - src/adapters/chatStream.test.ts \ - src/runners/cliRunner.test.ts \ - src/adapters/__tests__/streamBuffer.test.ts -echo "=== tsc filter ===" -npx tsc --noEmit -p tsconfig.check.json 2>&1 | rg 'eventHub|LogLine|chatBackend|discordPair|codexResponses|chatStream' | head -40 || true diff --git a/cli.json b/cli.json deleted file mode 100644 index c8c18499..00000000 --- a/cli.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "permissions": { - "allow": [ - "Shell(**)" - ], - "deny": [] - } -} diff --git a/cursor/cli.json b/cursor/cli.json deleted file mode 100644 index c8c18499..00000000 --- a/cursor/cli.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "permissions": { - "allow": [ - "Shell(**)" - ], - "deny": [] - } -} diff --git a/node_modules b/node_modules deleted file mode 120000 index d9643ec8..00000000 --- a/node_modules +++ /dev/null @@ -1 +0,0 @@ -/work/OpenSwarm/node_modules \ No newline at end of file