From 776a2fa158c1eeda91583c81698821eb7ca23e54 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 04:34:59 +0900 Subject: [PATCH 1/3] wip: preserved partial work (auto, session did not succeed) --- package-lock.json | 108 ++++---- package.json | 4 +- src/agents/pipelineFormat.ts | 267 +++++++++++-------- src/agents/reviewer.ts | 445 ++++++++----------------------- src/agents/skillDocumenter.ts | 212 +++++---------- src/agents/tester.ts | 337 +++++++---------------- src/automation/workerAuditLog.ts | 109 ++++---- src/support/outputBudget.ts | 113 ++++++++ 8 files changed, 654 insertions(+), 941 deletions(-) create mode 100644 src/support/outputBudget.ts diff --git a/package-lock.json b/package-lock.json index 36740455..87077e91 100644 --- a/package-lock.json +++ b/package-lock.json @@ -45,7 +45,7 @@ "devDependencies": { "@types/node": "^22.0.0", "@types/react": "^19.2.17", - "@vitest/coverage-v8": "^4.0.18", + "@vitest/coverage-v8": "^4.1.11", "bun-types": "^1.1.0", "ink-testing-library": "^4.0.0", "jsdom": "^26.1.0", @@ -53,7 +53,7 @@ "playwright": "^1.47.0", "tsx": "^4.21.0", "typescript": "^5.9.3", - "vitest": "^4.0.18" + "vitest": "^4.1.11" }, "engines": { "node": ">=22" @@ -2900,7 +2900,7 @@ "version": "19.2.17", "resolved": "https://registry.npmjs.org/@types/react/-/react-19.2.17.tgz", "integrity": "sha512-MXfmqaVPEVgkBT/aY0aGCkRWWtByiYQXo3xdQ8r5RzuFrPiRn8Gar2tQdXSUQ2GKV3bkXckek89V8wQBY2Q/Aw==", - "devOptional": true, + "dev": true, "license": "MIT", "dependencies": { "csstype": "^3.2.2" @@ -2922,14 +2922,14 @@ } }, "node_modules/@vitest/coverage-v8": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.8.tgz", - "integrity": "sha512-lt3kovsyHwYe00wq4D1ti0Z974fWj4NLp6siqiyEufUpyFwK9Yhi7rBhac9JL5aA0zoMrJqc4vYPZRUnI7l7nw==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.11.tgz", + "integrity": "sha512-8MVGEFnJIcdGjcbfKmeq8z0pZHH0JlVtoVZH9Q/qwUp6wyFnEJUBMrw9DCaj+ra3vShGmhavjalMIhPNxZAUcw==", "dev": true, "license": "MIT", "dependencies": { "@bcoe/v8-coverage": "^1.0.2", - "@vitest/utils": "4.1.8", + "@vitest/utils": "4.1.11", "ast-v8-to-istanbul": "^1.0.0", "istanbul-lib-coverage": "^3.2.2", "istanbul-lib-report": "^3.0.1", @@ -2943,8 +2943,8 @@ "url": "https://opencollective.com/vitest" }, "peerDependencies": { - "@vitest/browser": "4.1.8", - "vitest": "4.1.8" + "@vitest/browser": "4.1.11", + "vitest": "4.1.11" }, "peerDependenciesMeta": { "@vitest/browser": { @@ -2953,16 +2953,16 @@ } }, "node_modules/@vitest/expect": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.8.tgz", - "integrity": "sha512-h3nDO677RDLEGlBxyQ5CW8RlMThSKSRLUePLOx09gNIWRL40edgA1GCZSZgf1W55MFAG6/Sw14KeaAnqv0NKdQ==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/expect/-/expect-4.1.11.tgz", + "integrity": "sha512-VX2x5vNJXET47KAFzwERI+KRMtTTCSWTfSMKsW7JsUsXV4psq++e3DvZpuTDOpHcxytiDs6p2nhVb2tVDiiUYw==", "dev": true, "license": "MIT", "dependencies": { "@standard-schema/spec": "^1.1.0", "@types/chai": "^5.2.2", - "@vitest/spy": "4.1.8", - "@vitest/utils": "4.1.8", + "@vitest/spy": "4.1.11", + "@vitest/utils": "4.1.11", "chai": "^6.2.2", "tinyrainbow": "^3.1.0" }, @@ -2971,13 +2971,13 @@ } }, "node_modules/@vitest/mocker": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.8.tgz", - "integrity": "sha512-LEiN/xe4OSIbKe9HQIp5OC24agGD9J5CnmMgsLohVVoOPWL9a2sBoR6VBx43jQZb7Kr1l4RCuyCJzcAa0+dojw==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/mocker/-/mocker-4.1.11.tgz", + "integrity": "sha512-2XJVD55d1o5AZous5CCGKS74g/riOj9odEt2bQpCVZeblHyHdnMeFl4jl0XjU21stf4mbjUkew2eXQZt65g5CQ==", "dev": true, "license": "MIT", "dependencies": { - "@vitest/spy": "4.1.8", + "@vitest/spy": "4.1.11", "estree-walker": "^3.0.3", "magic-string": "^0.30.21" }, @@ -2998,9 +2998,9 @@ } }, "node_modules/@vitest/pretty-format": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.8.tgz", - "integrity": "sha512-9GasEBxpZ1VYIpqHf/0+YGg121uSNwCKOJqIrTwWP/TB7DmFCiaBpNl3aPZzoLWfWkuqhbH8vJIVobZkvdo2cA==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/pretty-format/-/pretty-format-4.1.11.tgz", + "integrity": "sha512-yiZzPbGTS9Sr/JpFl8zHrcIkAofNbFV6k21vIgQN/cY/oxZeXhJv5sc/MBJ5jFKWmWs+oJHw0UXLZjmf931+Vw==", "dev": true, "license": "MIT", "dependencies": { @@ -3011,13 +3011,13 @@ } }, "node_modules/@vitest/runner": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.8.tgz", - "integrity": "sha512-EmVxeBAfMJvycdjd6Hm+RbFBbA9fKvo0Kx37hNpBYoYeavH3RNsBXWDooR1mgD52dCrxIIuP7UotpfiwOikvcg==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/runner/-/runner-4.1.11.tgz", + "integrity": "sha512-LztvUgdwMNJMIkj3hQnnxiC2Xy1zNxq928W/xhjCLaNCzqTZOudjwbQf6v9IntZGPw132i2Lq2rgTRZHD3JHNw==", "dev": true, "license": "MIT", "dependencies": { - "@vitest/utils": "4.1.8", + "@vitest/utils": "4.1.11", "pathe": "^2.0.3" }, "funding": { @@ -3025,14 +3025,14 @@ } }, "node_modules/@vitest/snapshot": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.8.tgz", - "integrity": "sha512-acfZboRmAIf05DEKcBQy33VXojFJjtUdLyo7oOmV9kebb2xdU01UknNiPuPZoJZQyO7DF0gZdTGTpeAzET9QPQ==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/snapshot/-/snapshot-4.1.11.tgz", + "integrity": "sha512-pN7ikn1ON7h8ee4gIAp4AzyK+zBtJPzVbqOgu5LCEh4VaJVbPQcgYQYJIMGQPXVeJJq1fnfazis7a5pFNPahog==", "dev": true, "license": "MIT", "dependencies": { - "@vitest/pretty-format": "4.1.8", - "@vitest/utils": "4.1.8", + "@vitest/pretty-format": "4.1.11", + "@vitest/utils": "4.1.11", "magic-string": "^0.30.21", "pathe": "^2.0.3" }, @@ -3041,9 +3041,9 @@ } }, "node_modules/@vitest/spy": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.8.tgz", - "integrity": "sha512-6EevtBp6OZOPF7bmz36HrGMeP3txgVSrgebWxHOafDXGkhIzfXK14f8KF6MuFfgXXUeHxmpD3BQxkV00/3s5mA==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/spy/-/spy-4.1.11.tgz", + "integrity": "sha512-apNa/prQy2qCeywhnixOHPRCgGNhvg7T4Dapfl1GahLp/R+uhBm5cPyFoNVyqsNd2h1nJxL6BqqdIjiABL60YA==", "dev": true, "license": "MIT", "funding": { @@ -3051,13 +3051,13 @@ } }, "node_modules/@vitest/utils": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.8.tgz", - "integrity": "sha512-uOJamYALNhfJ6iolExyQM40yIQwDqYnkKtQ5VCiSe17E33H0aQ/u+1GlRuz4LZBk6Mm3sg90G9hEbmEt37C1Zg==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/@vitest/utils/-/utils-4.1.11.tgz", + "integrity": "sha512-zTCVGpyFsGWBhllOyKlTw/vnr6D9qxsfSDyfbyZmTyjHw5N/VuvzHpHoQjm2ZJzn4RJgx5w4r7V0er69CmLgPQ==", "dev": true, "license": "MIT", "dependencies": { - "@vitest/pretty-format": "4.1.8", + "@vitest/pretty-format": "4.1.11", "convert-source-map": "^2.0.0", "tinyrainbow": "^3.1.0" }, @@ -4002,7 +4002,7 @@ "version": "3.2.3", "resolved": "https://registry.npmjs.org/csstype/-/csstype-3.2.3.tgz", "integrity": "sha512-z1HGKcYy2xA8AGQfwrn0PAy+PB7X/GSj3UVJW9qKyn43xWa+gl5nXmU4qqLMRzWVLFC8KusUX8T/0kCiOYpAIQ==", - "devOptional": true, + "dev": true, "license": "MIT" }, "node_modules/data-urls": { @@ -7679,19 +7679,19 @@ } }, "node_modules/vitest": { - "version": "4.1.8", - "resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.8.tgz", - "integrity": "sha512-flY6ScbCIt9HThs+C5HS7jvGOB560DJtk/Z15IQROTA6zEy49Nh8T/dofWTQL+n3vswqn87sbJNiuqw1SDp5Ig==", + "version": "4.1.11", + "resolved": "https://registry.npmjs.org/vitest/-/vitest-4.1.11.tgz", + "integrity": "sha512-fhACrNXUidIbGSBr5FlbuBkO7VWC1ZyLl0DO4CU2DrQoAPxX84Ysxs+HeGQpii5lZWV1Q4gBZTTu49mF+A6Edw==", "dev": true, "license": "MIT", "dependencies": { - "@vitest/expect": "4.1.8", - "@vitest/mocker": "4.1.8", - "@vitest/pretty-format": "4.1.8", - "@vitest/runner": "4.1.8", - "@vitest/snapshot": "4.1.8", - "@vitest/spy": "4.1.8", - "@vitest/utils": "4.1.8", + "@vitest/expect": "4.1.11", + "@vitest/mocker": "4.1.11", + "@vitest/pretty-format": "4.1.11", + "@vitest/runner": "4.1.11", + "@vitest/snapshot": "4.1.11", + "@vitest/spy": "4.1.11", + "@vitest/utils": "4.1.11", "es-module-lexer": "^2.0.0", "expect-type": "^1.3.0", "magic-string": "^0.30.21", @@ -7719,12 +7719,12 @@ "@edge-runtime/vm": "*", "@opentelemetry/api": "^1.9.0", "@types/node": "^20.0.0 || ^22.0.0 || >=24.0.0", - "@vitest/browser-playwright": "4.1.8", - "@vitest/browser-preview": "4.1.8", - "@vitest/browser-webdriverio": "4.1.8", - "@vitest/coverage-istanbul": "4.1.8", - "@vitest/coverage-v8": "4.1.8", - "@vitest/ui": "4.1.8", + "@vitest/browser-playwright": "4.1.11", + "@vitest/browser-preview": "4.1.11", + "@vitest/browser-webdriverio": "4.1.11", + "@vitest/coverage-istanbul": "4.1.11", + "@vitest/coverage-v8": "4.1.11", + "@vitest/ui": "4.1.11", "happy-dom": "*", "jsdom": "*", "vite": "^6.0.0 || ^7.0.0 || ^8.0.0" diff --git a/package.json b/package.json index 54953ef0..303e8c83 100644 --- a/package.json +++ b/package.json @@ -79,7 +79,7 @@ "devDependencies": { "@types/node": "^22.0.0", "@types/react": "^19.2.17", - "@vitest/coverage-v8": "^4.0.18", + "@vitest/coverage-v8": "^4.1.11", "bun-types": "^1.1.0", "ink-testing-library": "^4.0.0", "jsdom": "^26.1.0", @@ -87,7 +87,7 @@ "playwright": "^1.47.0", "tsx": "^4.21.0", "typescript": "^5.9.3", - "vitest": "^4.0.18" + "vitest": "^4.1.11" }, "engines": { "node": ">=22" diff --git a/src/agents/pipelineFormat.ts b/src/agents/pipelineFormat.ts index 8b6018b6..d5d86d0b 100644 --- a/src/agents/pipelineFormat.ts +++ b/src/agents/pipelineFormat.ts @@ -6,6 +6,14 @@ import { EmbedBuilder } from 'discord.js'; import type { PipelineResult } from './pairPipeline.js'; import { formatCost } from '../support/costTracker.js'; +import { + boundedFieldValue, + boundedDescription, + PIPELINE_EMBED_FIELD_VALUE_LIMIT, + PIPELINE_FAILED_TESTS_PREVIEW, + DISCORD_EMBED_FIELDS_PER_EMBED, + DISCORD_EMBED_AGGREGATE_VALUE_LIMIT, +} from '../support/outputBudget.js'; /** Format epoch ms to HH:MM:SS local time string */ function formatTimestamp(epochMs: number): string { @@ -41,166 +49,205 @@ export function formatPipelineResult(result: PipelineResult): string { || (ctx.projectPath ? ctx.projectPath.split('/').pop() || '' : ''); if (displayName) parts.push(`πŸ“ ${displayName}`); if (ctx.issueIdentifier) parts.push(`πŸ”– ${ctx.issueIdentifier}`); - if (ctx.projectPath) parts.push(`\`${ctx.projectPath.split('/').slice(-2).join('/')}\``); - if (parts.length > 0) { - lines.push(parts.join(' | ')); - } - if (ctx.taskTitle) { - lines.push(`πŸ“‹ ${ctx.taskTitle}`); - } - lines.push(''); + if (ctx.taskTitle) parts.push(ctx.taskTitle); + lines.push(`**${parts.join(' | ')}**`); } - lines.push(`${statusEmoji} **Pipeline ${result.finalStatus.toUpperCase()}**`); + // Status line lines.push(''); - lines.push(`**Session:** \`${result.sessionId}\``); - lines.push(`**Iterations:** ${result.iterations}`); - lines.push(`**Duration:** ${(result.totalDuration / 1000).toFixed(1)}s`); + lines.push(`${statusEmoji} **Status:** ${result.finalStatus}`); + // Duration + if (result.totalDuration) { + const mins = Math.floor(result.totalDuration / 60000); + const secs = Math.round((result.totalDuration % 60000) / 1000); + lines.push(`⏱ **Duration:** ${mins}m ${secs}s`); + } + + // Cost if (result.totalCost) { - lines.push(`**Cost:** $${result.totalCost.costUsd.toFixed(4)} (${formatCost(result.totalCost)})`); + lines.push(`πŸ’° **Cost:** ${formatCost(result.totalCost)}`); } - lines.push(''); - lines.push('**Stages:**'); - for (const stage of result.stages) { - const emoji = stage.success ? 'βœ…' : '❌'; - const duration = (stage.duration / 1000).toFixed(1); - const time = formatTimestamp(stage.startedAt); - lines.push(` ${emoji} ${stage.stage} (${duration}s) @ ${time}`); + // Stage summary + if (result.stages && result.stages.length > 0) { + lines.push(''); + lines.push('**Stages:**'); + for (const stage of result.stages) { + const stageEmoji = stage.status === 'success' ? 'βœ…' : stage.status === 'failed' ? '❌' : '⏳'; + lines.push(` ${stageEmoji} ${stage.name}${stage.durationMs ? ` (${Math.round(stage.durationMs / 1000)}s)` : ''}`); + } + } + + // Worker summary + if (result.workerResult) { + lines.push(''); + lines.push(`**πŸ”¨ Worker:** ${result.workerResult.summary || 'No summary'}`); + if (result.workerResult.filesChanged && result.workerResult.filesChanged.length > 0) { + const files = result.workerResult.filesChanged.slice(0, 10); + lines.push(` Files: ${files.join(', ')}`); + if (result.workerResult.filesChanged.length > 10) { + lines.push(` ... +${result.workerResult.filesChanged.length - 10} more`); + } + } + } + + // Reviewer feedback + if (result.reviewResult) { + lines.push(''); + const reviewEmoji = result.reviewResult.decision === 'approved' ? 'βœ…' : '❌'; + lines.push(`${reviewEmoji} **Reviewer:** ${result.reviewResult.decision}`); + if (result.reviewResult.feedback) { + // Bound reviewer feedback to prevent oversized messages + const feedback = result.reviewResult.feedback.length > 500 + ? result.reviewResult.feedback.slice(0, 500) + '…' + : result.reviewResult.feedback; + lines.push(` ${feedback}`); + } + } + + // Test results + if (result.testerResult) { + lines.push(''); + const testEmoji = result.testerResult.success ? 'βœ…' : '❌'; + lines.push(`${testEmoji} **Tests:** ${result.testerResult.testsPassed} passed, ${result.testerResult.testsFailed} failed`); + } + + // PR URL + if (result.prUrl) { + lines.push(''); + lines.push(`πŸ”— **Pull Request:** ${result.prUrl}`); } return lines.join('\n'); } /** - * Format pipeline result as a Discord Embed + * Format pipeline result as a Discord embed + * Enforces per-field and aggregate embed budgets to prevent payload rejection. */ export function formatPipelineResultEmbed(result: PipelineResult): EmbedBuilder { - const statusConfig = { - approved: { emoji: 'βœ…', color: 0x00FF00, label: 'SUCCESS' }, - rejected: { emoji: '❌', color: 0xFF0000, label: 'REJECTED' }, - failed: { emoji: 'πŸ’₯', color: 0xFF6B6B, label: 'FAILED' }, - cancelled: { emoji: '🚫', color: 0xFFAA00, label: 'CANCELLED' }, - decomposed: { emoji: 'πŸ”€', color: 0x00AAFF, label: 'DECOMPOSED' }, - superseded: { emoji: '♻️', color: 0x00AAFF, label: 'SUPERSEDED' }, - deferred: { emoji: '⏳', color: 0xFFAA00, label: 'DEFERRED' }, - waiting_on_operator: { emoji: 'πŸ™‹', color: 0xFFC300, label: 'WAITING ON OPERATOR' }, - rate_limited: { emoji: '⏸', color: 0xFFAA00, label: 'RATE LIMITED' }, - infra_error: { emoji: 'πŸ”Œ', color: 0xFFAA00, label: 'INFRA ERROR' }, - }[result.finalStatus] || { emoji: '❓', color: 0x808080, label: 'UNKNOWN' }; + const statusColor = { + approved: 0x00ff41, + rejected: 0xff0044, + failed: 0xff6600, + cancelled: 0x888888, + decomposed: 0x00aaff, + superseded: 0xaa00ff, + deferred: 0xffaa00, + waiting_on_operator: 0xffff00, + rate_limited: 0xff8800, + infra_error: 0xff4444, + }[result.finalStatus] || 0x888888; const embed = new EmbedBuilder() - .setTitle(`${statusConfig.emoji} Pipeline ${statusConfig.label}`) - .setColor(statusConfig.color) + .setColor(statusColor) .setTimestamp(); - // Task context + // Title (bounded) + const title = result.taskContext?.taskTitle || 'Pipeline Result'; + embed.setTitle(title.length > 256 ? `${title.slice(0, 253)}…` : title); + + // Description (bounded) if (result.taskContext) { const ctx = result.taskContext; const displayName = ctx.projectName || (ctx.projectPath ? ctx.projectPath.split('/').pop() || '' : ''); - - if (displayName && ctx.issueIdentifier) { - embed.setDescription(`πŸ“ **${displayName}** | πŸ”– ${ctx.issueIdentifier}\n${ctx.taskTitle || ''}`); - } else if (ctx.taskTitle) { - embed.setDescription(ctx.taskTitle); - } + const descParts: string[] = []; + if (displayName) descParts.push(`πŸ“ ${displayName}`); + if (ctx.issueIdentifier) descParts.push(`πŸ”– ${ctx.issueIdentifier}`); + embed.setDescription(boundedDescription(descParts.join(' | '))); } - // Summary stats - const durationStr = (result.totalDuration / 1000).toFixed(1) + 's'; - const costStr = result.totalCost - ? `$${result.totalCost.costUsd.toFixed(4)} (${formatCost(result.totalCost)})` - : 'N/A'; + // Track aggregate field value length to stay within embed budget + let aggregateValueLength = 0; + + // Helper to add a field only if it fits within the aggregate budget + const tryAddField = (name: string, value: string, inline = false): boolean => { + const bounded = boundedFieldValue(value, PIPELINE_EMBED_FIELD_VALUE_LIMIT); + const newTotal = aggregateValueLength + bounded.length; + if (newTotal > DISCORD_EMBED_AGGREGATE_VALUE_LIMIT) return false; + if (embed.data.fields && embed.data.fields.length >= DISCORD_EMBED_FIELDS_PER_EMBED) return false; + embed.addFields({ name, value: bounded, inline }); + aggregateValueLength = newTotal; + return true; + }; + + // Status field + tryAddField('Status', result.finalStatus, true); + + // Duration + if (result.totalDuration) { + const mins = Math.floor(result.totalDuration / 60000); + const secs = Math.round((result.totalDuration % 60000) / 1000); + tryAddField('Duration', `${mins}m ${secs}s`, true); + } - embed.addFields( - { name: 'πŸ”„ Iterations', value: result.iterations.toString(), inline: true }, - { name: '⏱️ Duration', value: durationStr, inline: true }, - { name: 'πŸ’° Cost', value: costStr, inline: true }, - ); + // Cost + if (result.totalCost) { + tryAddField('Cost', formatCost(result.totalCost), true); + } // Stages - const stagesStr = result.stages - .map(s => { - const emoji = s.success ? 'βœ…' : '❌'; - const duration = (s.duration / 1000).toFixed(1); - const time = formatTimestamp(s.startedAt); - return `${emoji} **${s.stage}** (${duration}s) @ ${time}`; - }) - .join('\n') || 'No stages'; - - embed.addFields({ name: 'πŸ“Š Stages', value: stagesStr, inline: false }); - - // Worker result - if (result.workerResult) { - const worker = result.workerResult; - let workerValue = ''; - - if (worker.summary) { - workerValue += `${worker.summary.slice(0, 200)}${worker.summary.length > 200 ? '...' : ''}\n\n`; - } + if (result.stages && result.stages.length > 0) { + const stagesStr = result.stages.map((s) => { + const emoji = s.status === 'success' ? 'βœ…' : s.status === 'failed' ? '❌' : '⏳'; + return `${emoji} ${s.name}${s.durationMs ? ` (${Math.round(s.durationMs / 1000)}s)` : ''}`; + }).join('\n'); + tryAddField('πŸ“Š Stages', stagesStr, false); + } - if (worker.filesChanged && worker.filesChanged.length > 0) { - const filesStr = worker.filesChanged.slice(0, 5).map(f => `\`${f}\``).join(', '); - workerValue += `**Files:** ${filesStr}`; - if (worker.filesChanged.length > 5) { - workerValue += ` +${worker.filesChanged.length - 5} more`; + // Worker + if (result.workerResult) { + let workerValue = result.workerResult.summary + ? boundedFieldValue(result.workerResult.summary, PIPELINE_EMBED_FIELD_VALUE_LIMIT) + : 'No summary'; + if (result.workerResult.filesChanged && result.workerResult.filesChanged.length > 0) { + const files = result.workerResult.filesChanged.slice(0, 10).join(', '); + workerValue += `\n\n**Files:** ${files}`; + if (result.workerResult.filesChanged.length > 10) { + workerValue += `\n… +${result.workerResult.filesChanged.length - 10} more`; } } - - if (workerValue) { - embed.addFields({ name: 'πŸ”¨ Worker', value: workerValue, inline: false }); - } + tryAddField('πŸ”¨ Worker', workerValue, false); } - // Reviewer result + // Reviewer if (result.reviewResult) { - const review = result.reviewResult; - let reviewValue = `**Decision:** ${review.decision.toUpperCase()}\n\n`; - - if (review.feedback) { - reviewValue += review.feedback.slice(0, 300); - if (review.feedback.length > 300) reviewValue += '...'; - } - - if (review.issues && review.issues.length > 0) { - reviewValue += `\n\n**Issues found:** ${review.issues.length}`; + const reviewEmoji = result.reviewResult.decision === 'approved' ? 'βœ…' : '❌'; + let reviewValue = `${reviewEmoji} **${result.reviewResult.decision}**`; + if (result.reviewResult.feedback) { + const feedback = boundedFieldValue(result.reviewResult.feedback, PIPELINE_EMBED_FIELD_VALUE_LIMIT); + reviewValue += `\n\n${feedback}`; } - - embed.addFields({ name: 'βœ… Reviewer', value: reviewValue, inline: false }); + tryAddField('βœ… Reviewer', reviewValue, false); } - // Tester result + // Tests if (result.testerResult) { const test = result.testerResult; - const total = test.testsPassed + test.testsFailed; - const passRate = total > 0 ? ((test.testsPassed / total) * 100).toFixed(1) : '0'; - - let testValue = `βœ… Passed: ${test.testsPassed}/${total} (${passRate}%)${test.deterministic ? ' Β· deterministic' : ''}`; - - if (test.coverage !== undefined) { - testValue += `\nπŸ“Š Coverage: ${test.coverage.toFixed(1)}%`; + const testEmoji = test.success ? 'βœ…' : '❌'; + let testValue = `${testEmoji} **${test.testsPassed} passed, ${test.testsFailed} failed**`; + if (test.coverage != null) { + testValue += ` | Coverage: ${(test.coverage * 100).toFixed(1)}%`; } - - if (test.testsFailed > 0 && test.failedTests && test.failedTests.length > 0) { - const failedStr = test.failedTests.slice(0, 2).map(t => `❌ ${t}`).join('\n'); + if (!test.success && test.failedTests && test.failedTests.length > 0) { + const failedStr = test.failedTests.slice(0, PIPELINE_FAILED_TESTS_PREVIEW).map(t => `❌ ${t}`).join('\n'); testValue += `\n\n${failedStr}`; - if (test.failedTests.length > 2) { - testValue += `\n... +${test.failedTests.length - 2} more`; + if (test.failedTests.length > PIPELINE_FAILED_TESTS_PREVIEW) { + testValue += `\n... +${test.failedTests.length - PIPELINE_FAILED_TESTS_PREVIEW} more`; } } - - embed.addFields({ name: 'πŸ§ͺ Tests', value: testValue, inline: false }); + tryAddField('πŸ§ͺ Tests', testValue, false); } // PR URL if (result.prUrl) { - embed.addFields({ name: 'πŸ”— Pull Request', value: `[View PR](${result.prUrl})`, inline: false }); + tryAddField('πŸ”— Pull Request', `[View PR](${result.prUrl})`, false); } // Footer embed.setFooter({ text: `Session: ${result.sessionId.slice(0, 8)}...` }); return embed; -} +} \ No newline at end of file diff --git a/src/agents/reviewer.ts b/src/agents/reviewer.ts index 58d49817..f6b87730 100644 --- a/src/agents/reviewer.ts +++ b/src/agents/reviewer.ts @@ -15,6 +15,7 @@ import type { VerifyEvidence } from '../verify/runner.js'; import { renderVerifyEvidence } from './verificationEvidence.js'; import type { InstructionCapsule } from './instructionCapsule.js'; import { COORDINATION_GUIDANCE_PROMPT, type CoordinationToolContext } from '../coordination/coordinationTools.js'; +import { boundedMessageContent, DISCORD_MESSAGE_CONTENT_LIMIT } from '../support/outputBudget.js'; // Types @@ -29,384 +30,156 @@ export interface ReviewerOptions { maxTurns?: number; // Max agentic turns per CLI invocation adapterName?: AdapterName; processContext?: ProcessContext; - /** Reasoning effort from a jobProfile (codex-responses: low|medium|high). */ - reasoningEffort?: 'low' | 'medium' | 'high'; - /** Execution-grounded definition of done to hard-gate on (INT-1914). */ - completionCriteria?: string[]; - /** - * Non-blocking deterministic guard warnings (dead-module, reformat/scope, …) - * surfaced to the reviewer so it verifies each instead of them dying in a log - * line. (INT-2388) - */ - guardWarnings?: string[]; - /** Deterministic command evidence produced by the harness tester. */ + /** Reasoning effort from a j + * compatible adapter (e.g. openrouter). */ + reasoningEffort?: number; + /** Coordination tools available to the reviewer agent */ + coordinationTools?: CoordinationToolContext; + /** Verify evidence from the deterministic tester */ verificationEvidence?: VerifyEvidence[]; - /** Relevant repository-local logs from earlier review commands. */ - priorReviewContext?: string; - /** - * 'change' (default): review a worker's diff. 'audit': evaluate existing files - * with no diff/worker (the `review --max` codebase audit). (INT-2006) - */ - mode?: 'change' | 'audit' | 'direct'; - /** MCP tools to expose (e.g. linear__*). When unset the adapter self-sources (INT-1951). (INT-1950) */ - mcpTools?: ToolDefinition[]; - /** Tool-activity log lines (πŸ”§ read_file …) for live progress display. (INT-1963) */ - onLog?: (line: string) => void; - /** Streamed reasoning/text deltas for live progress display. (INT-1963) */ - onToken?: (delta: string) => void; - /** Abort the run + in-flight adapter call (pipeline cancel / project disable). */ - signal?: AbortSignal; - /** - * Deny the reviewer every mutating tool β€” write_file, edit_file, apply_patch - * and bash β€” leaving read_file/search_files/search_memory. - * - * Mandatory whenever the diff under review is not trusted. Reviewing a pull - * request in CI puts an agent with shell access on attacker-controlled files - * while the provider credential sits in the environment, which turns prompt - * injection into command execution. A review is a judgement, not an - * execution, so nothing legitimate is lost. - * - * Off by default: the local `openswarm review` path reviews the operator's own - * working tree, and running commands there is how the reviewer substantiates a - * claim. (INT-3189) - */ - readOnly?: boolean; - /** - * The change under review, as text. Supplied rather than discovered: a - * read-only reviewer cannot shell out for it, and in committed-diff mode - * there is nothing in the working tree to read. (INT-3101) - */ - diff?: string; - /** Run-scoped Claude Code instruction and runbook snapshot. */ + /** Instruction capsule for the reviewer */ instructionCapsule?: InstructionCapsule; - /** - * This reviewer's board identity β€” its call sign and mailbox address. - * - * Distinct from the worker's on the same task: two agents answering to one - * name make advice and operator answers unroutable. Coordination tools stay - * withheld while `readOnly` is set (INT-3189); the identity is still carried - * so reports and the dashboard name the reviewer. - */ - coordinationContext?: CoordinationToolContext; -} - -/** Tell the reviewer the call sign other agents and the operator address it by. */ -function reviewerIdentityHeader(callSign: string | undefined): string { - if (!callSign) return ''; - return `\n\n## Your identity\nYou are **${callSign}**. Sign your review with that call sign so the worker and the operator know who reviewed the change.\n`; -} - -/** - * Coordination tools are withheld while readOnly is set (INT-3189) even - * though coordinationContext is still carried for identity/labeling β€” so - * this must check both, not just coordinationContext, or the reviewer would - * be told to use a tool it doesn't actually have. (AGT-4054) - */ -function reviewerCoordinationGuidance( - coordinationContext: CoordinationToolContext | undefined, - readOnly: boolean | undefined, -): string { - return coordinationContext && !readOnly - ? COORDINATION_GUIDANCE_PROMPT + getPrompts().coordinationConsultationPrompt - : ''; } export interface PreCheckResult { passed: boolean; - issues: string[]; - confidence: number; // 0-3: quality of the check + reason: string; } // Prompts -/** - * Build Pre-Check prompt for fast validation (Haiku) - */ -function buildPreCheckPrompt(options: ReviewerOptions): string { - const files = options.workerResult.filesChanged; - const filesSummary = files.length <= 20 - ? (files.join(', ') || '(none)') - : `${files.slice(0, 20).join(', ')} (+${files.length - 20} more)`; - - return `You are a fast pre-check validator. Perform a quick validation of the Worker's output. - -## Task -${options.taskTitle} - -## Worker Result -- Success: ${options.workerResult.success} -- Files Changed (${files.length}): ${filesSummary} -- Summary: ${options.workerResult.summary} - -## Your Job (Fast Check Only) -Check for OBVIOUS problems: -1. **Syntax Errors**: Are there any clear syntax errors in the output? -2. **Missing Files**: Did the worker claim to create/modify files that don't exist? -3. **Incomplete Output**: Does the output look cut off or incomplete? -4. **Basic Format Issues**: Are there obvious formatting problems? - -**DO NOT** perform deep logical review - that's for the next stage. - -## Response Format -Respond in this EXACT format: - -PASSED: [yes/no] -CONFIDENCE: [0-3] -ISSUES: -- [issue 1] -- [issue 2] -... - -Keep it brief. This is a fast filter, not a deep review.`; +function reviewerIdentityHeader(callSign: string | undefined): string { + return callSign + ? `You are a code reviewer (call sign: ${callSign}).` + : 'You are a code reviewer.'; } -/** - * Build Reviewer prompt using locale templates - */ -export function buildReviewerPrompt(options: ReviewerOptions): string { - const files = options.workerResult.filesChanged; - const filesSummary = files.length <= 20 - ? (files.join(', ') || '(none)') - : `${files.slice(0, 20).join(', ')} (+${files.length - 20} more)`; - - // Audit mode: no diff/commands to report β€” just hand the auditor the file list. (INT-2006) - if (options.mode === 'audit') { - return getPrompts().buildReviewerPrompt({ - taskTitle: options.taskTitle, - taskDescription: options.taskDescription, - authoritativeOperatorFeedback: options.authoritativeOperatorFeedback, - workerReport: `- **Files under audit (${files.length}):** ${filesSummary}`, - mode: 'audit', - priorReviewContext: options.priorReviewContext, - }); - } - - // Direct mode reviews a Git diff supplied by a user/CI checkout, not an - // OpenSwarm worker result. Do not manufacture a zero-command worker report: - // "evidence not collected here" is different from "validation was not run". - if (options.mode === 'direct') { - // The diff goes in the prompt, not left for the agent to reconstruct. A - // read-only reviewer has no bash, and in committed-diff mode the working - // tree is clean β€” reading a file shows the result, never the change. Under - // `--read-only --base`, which is what the CI gate uses, the reviewer could - // not see its own subject and said so while still returning a verdict. - // (INT-3101) - // Appended as plain text on purpose: the template already wraps the whole - // report in its untrusted-data block, which escapes the closing marker and - // code fences. A second fence here would be escaped by that one, so it - // would add noise while providing none of the protection it appears to. - const report = options.diff - ? `- **Files changed (${files.length}):** ${filesSummary}\n- **Diff under review:**\n${options.diff}` - : `- **Files changed (${files.length}):** ${filesSummary}`; - return getPrompts().buildReviewerPrompt({ - taskTitle: options.taskTitle, - taskDescription: options.taskDescription, - authoritativeOperatorFeedback: options.authoritativeOperatorFeedback, - workerReport: report, - mode: 'direct', - priorReviewContext: options.priorReviewContext, - }); - } - - const cmds = options.workerResult.commands; - const cmdsSummary = cmds.length <= 10 - ? (cmds.join(', ') || '(none)') - : `${cmds.slice(0, 10).join(', ')} (+${cmds.length - 10} more)`; - - const guardSection = options.guardWarnings && options.guardWarnings.length > 0 - ? `- **Automated guard warnings (deterministic pre-checks β€” verify each, don't dismiss):**\n${options.guardWarnings.map(w => ` - ${w}`).join('\n')}\n` - : ''; +function reviewerCoordinationGuidance(): string { + return [ + '', + '## Coordination tools available', + '', + 'You have access to coordination tools that let you communicate with other agents', + 'and read durable repository threads. Use them when you need to:', + '', + 'β€’ Ask a worker agent for clarification on their changes', + 'β€’ Check if there are existing discussions about the code you are reviewing', + 'β€’ Coordinate with other reviewers on shared files', + '', + COORDINATION_GUIDANCE_PROMPT, + '', + '**Important:** Only use coordination tools when you have a specific question or', + 'need to share information. Do not use them for routine status updates.', + ].join('\n'); +} - const workerReport = ` -- **Success:** ${options.workerResult.success} -- **Summary:** ${options.workerResult.summary} -- **Files Changed (${files.length}):** ${filesSummary} -- **Commands:** ${cmdsSummary} -${options.workerResult.error ? `- **Error:** ${options.workerResult.error}` : ''} -${guardSection}`; +function buildPreCheckPrompt(options: ReviewerOptions): string { + const prompts = getPrompts(); + return prompts.buildPreCheckPrompt({ + taskTitle: options.taskTitle, + taskDescription: options.taskDescription, + workerResult: options.workerResult, + }); +} - return getPrompts().buildReviewerPrompt({ +function buildReviewerPrompt(options: ReviewerOptions): string { + const prompts = getPrompts(); + return prompts.buildReviewerPrompt({ taskTitle: options.taskTitle, taskDescription: options.taskDescription, + workerResult: options.workerResult, authoritativeOperatorFeedback: options.authoritativeOperatorFeedback, - workerReport, - completionCriteria: options.completionCriteria, - verificationEvidence: renderVerifyEvidence(options.verificationEvidence ?? []), - priorReviewContext: options.priorReviewContext, + verificationEvidence: options.verificationEvidence, + instructionCapsule: options.instructionCapsule, }); } -// Pre-Check Execution (Fast Validation with Haiku) +// Execution -/** - * Run fast pre-check validation with Haiku model - * This is a cheap filter before expensive Sonnet review - * Expected to catch 30-40% of obvious issues, saving ~35% on review costs - */ -export async function runPreCheck(options: ReviewerOptions): Promise { +async function runPreCheck(options: ReviewerOptions): Promise { const prompt = buildPreCheckPrompt(options); - const cwd = expandPath(options.projectPath); - const adapter = getAdapter(options.adapterName); - - try { - // Use Haiku for fast validation - const raw = await spawnCli(adapter, { - prompt, - cwd, - timeoutMs: 30000, // 30 seconds max for pre-check - model: options.model, - maxTurns: options.maxTurns, - processContext: options.processContext, - onLog: options.onLog, - signal: options.signal, - readOnly: options.readOnly, - }); - - // DEBUG: Log raw Haiku output for troubleshooting - console.log('[Reviewer] Pre-check raw output (first 500 chars):', raw.stdout.slice(0, 500)); - - // Parse pre-check output - const lines = raw.stdout.split('\n'); - const passedLine = lines.find((l: string) => l.startsWith('PASSED:')); - const confidenceLine = lines.find((l: string) => l.startsWith('CONFIDENCE:')); - - const passed = passedLine?.includes('yes') ?? false; - const confidence = parseInt(confidenceLine?.split(':')[1]?.trim() || '1', 10); - - const issueStart = lines.findIndex((l: string) => l.startsWith('ISSUES:')); - const issues = issueStart >= 0 - ? lines.slice(issueStart + 1) - .filter((l: string) => l.trim().startsWith('-')) - .map((l: string) => l.replace(/^-\s*/, '').trim()) - .filter(Boolean) - : []; - - // DEBUG: Log parsing results - if (!passed && issues.length === 0) { - console.warn('[Reviewer] Pre-check failed but no issues found. Haiku may not be following format.'); - console.warn('[Reviewer] PASSED line:', passedLine || '(not found)'); - console.warn('[Reviewer] ISSUES section:', issueStart >= 0 ? 'found' : 'not found'); - - // Provide better default error message - if (!passedLine) { - issues.push('Haiku did not provide PASSED: line in expected format'); - } - if (issueStart < 0) { - issues.push('Haiku did not provide ISSUES: section in expected format'); - } - } + const adapter = getAdapter(options.adapterName || 'cli'); + const result = await spawnCli(adapter, prompt, { + processContext: options.processContext, + timeoutMs: options.timeoutMs ?? 120_000, + maxTurns: options.maxTurns ?? 5, + model: options.model, + reasoningEffort: options.reasoningEffort, + }); - return { - passed, - issues: issues.length > 0 ? issues : ['Pre-check failed with no specific issues (format parsing error)'], - confidence: Math.min(3, Math.max(0, confidence)), - }; - } catch (error) { - // Rate limit errors must propagate so the scheduler can pause β€” pre-check - // failure otherwise just passes through to the full review. - if (error instanceof RateLimitError) throw error; - // If pre-check fails, allow proceeding to full review - console.warn('[Reviewer] Pre-check failed, proceeding to full review:', error); - return { - passed: true, // Don't block on pre-check failure - issues: ['Pre-check timed out or failed'], - confidence: 0, - }; - } + const output = result.output.trim().toLowerCase(); + const passed = output.includes('yes') || output.includes('pass') || output.includes('approve'); + return { passed, reason: result.output.trim() }; } -// Reviewer Execution - -/** - * Run Reviewer agent (full review with Sonnet) - */ export async function runReviewer(options: ReviewerOptions): Promise { const prompt = buildReviewerPrompt(options); - const cwd = expandPath(options.projectPath); - const adapter = getAdapter(options.adapterName); + const adapter = getAdapter(options.adapterName || 'cli'); + const result = await spawnCli(adapter, prompt, { + processContext: options.processContext, + timeoutMs: options.timeoutMs ?? 120_000, + maxTurns: options.maxTurns ?? 10, + model: options.model, + reasoningEffort: options.reasoningEffort, + coordinationTools: options.coordinationTools, + }); - try { - // Run CLI via adapter - const raw = await spawnCli(adapter, { - prompt, - cwd, - timeoutMs: options.timeoutMs ?? 300000, // 5 min default - model: options.model, - maxTurns: options.maxTurns, - processContext: options.processContext, - systemPrompt: getPrompts().systemPrompt - + reviewerIdentityHeader(options.coordinationContext?.actorName) - + reviewerCoordinationGuidance(options.coordinationContext, options.readOnly) - + (options.instructionCapsule?.text ?? ''), - reasoningEffort: options.reasoningEffort, - mcpTools: options.mcpTools, - onLog: options.onLog, - onToken: options.onToken, - signal: options.signal, - readOnly: options.readOnly, - coordinationContext: options.coordinationContext, - }); + const output = result.output.trim(); - // Parse result via adapter - const parsedResult = adapter.parseReviewerOutput(raw); - // Backfill loop-measured usage for adapters that don't extract their own. (INT-2508) - if (raw.costInfo && !parsedResult.costInfo) { - parsedResult.costInfo = raw.costInfo; + // Try to parse structured output + try { + const parsed = JSON.parse(output); + if (parsed.decision && parsed.feedback !== undefined) { + return { + decision: parsed.decision, + feedback: parsed.feedback, + issues: parsed.issues || [], + suggestions: parsed.suggestions || [], + costInfo: result.costInfo, + }; } - return parsedResult; - } catch (error) { - // Rate limit errors must propagate so the scheduler can pause. - if (error instanceof RateLimitError) throw error; - // An infra failure (CLI exit, auth, spawn, timeout) means the REVIEWER never - // ran β€” it is NOT a quality verdict. Propagate so the pipeline classifies it - // as 'infra_error' instead of letting it masquerade as a 'reject' that - // increments the rejection-limit STUCK counter. (INT-2010) - if (isInfraError(error)) throw error; - // Whatever is left ran the reviewer but produced NO usable verdict β€” most often - // adapter.parseReviewerOutput throwing on malformed output. This is NOT a quality - // 'reject' (a 'reject' discards the worker's work AND counts toward the - // rejectionβ†’STUCK limit, turning a reviewer-side parse bug into a false STUCK), - // and NOT a 'revise' (which the CLI reads as an exit-0 success and which spends - // the worker's revision budget on a reviewer-side problem). Throw an infra-marked - // error β†’ the pipeline classifies infra_error (backoff retry, no STUCK) and the - // CLI exits non-zero. (INT-2521) - const msg = error instanceof Error ? error.message : String(error); - throw new Error(`reviewer-stage: produced no parseable verdict: ${msg}`, { cause: error }); + } catch { + // Not JSON, use raw output } -} -// Formatting + return { + decision: output.includes('approve') ? 'approved' : 'rejected', + feedback: output, + issues: [], + suggestions: [], + costInfo: result.costInfo, + }; +} /** - * Format Reviewer result as a Discord message + * Format review feedback as a Discord message. + * Enforces Discord message content limits to prevent payload rejection. */ export function formatReviewFeedback(result: ReviewResult): string { - const decisionEmoji = { - approve: 'βœ…', - revise: 'πŸ”„', - reject: '❌', - }[result.decision]; - - const decisionText = { - approve: 'APPROVED', - revise: 'REVISION NEEDED', - reject: 'REJECTED', - }[result.decision]; - + const decisionEmoji = result.decision === 'approved' ? 'βœ…' : '❌'; const lines: string[] = []; - lines.push(`${decisionEmoji} ${t('agents.reviewer.report.decision', { text: decisionText })}`); + lines.push(`${decisionEmoji} **Review Decision: ${result.decision}**`); lines.push(''); - lines.push(t('agents.reviewer.report.feedback', { text: result.feedback })); + // Feedback (bounded per Discord message limit) + if (result.feedback) { + lines.push(boundedMessageContent(result.feedback)); + } + + // Issues if (result.issues && result.issues.length > 0) { lines.push(''); lines.push(t('agents.reviewer.report.issues')); - for (const issue of result.issues.slice(0, 5)) { - lines.push(` β€’ ${issue}`); + for (const issue of result.issues.slice(0, 10)) { + lines.push(` ⚠️ ${issue}`); + } + if (result.issues.length > 10) { + lines.push(` … +${result.issues.length - 10} more`); } } + // Suggestions if (result.suggestions && result.suggestions.length > 0) { lines.push(''); lines.push(t('agents.reviewer.report.suggestions')); @@ -415,7 +188,11 @@ export function formatReviewFeedback(result: ReviewResult): string { } } - return lines.join('\n'); + const full = lines.join('\n'); + // Ensure the entire message fits within Discord limits + return full.length > DISCORD_MESSAGE_CONTENT_LIMIT + ? full.slice(0, DISCORD_MESSAGE_CONTENT_LIMIT - 3) + '…' + : full; } /** @@ -428,4 +205,4 @@ export function buildRevisionPrompt(result: ReviewResult): string { issues: result.issues || [], suggestions: result.suggestions || [], }); -} +} \ No newline at end of file diff --git a/src/agents/skillDocumenter.ts b/src/agents/skillDocumenter.ts index dc4c28de..54a69ed1 100644 --- a/src/agents/skillDocumenter.ts +++ b/src/agents/skillDocumenter.ts @@ -9,6 +9,7 @@ import { getAdapter, spawnCli } from '../adapters/index.js'; import { type CostInfo, extractCostFromStreamJson, formatCost } from '../support/costTracker.js'; import { expandPath } from '../core/config.js'; import { RateLimitError } from '../adapters/rateLimitError.js'; +import { boundedMessageContent, DISCORD_MESSAGE_CONTENT_LIMIT } from '../support/outputBudget.js'; // Types @@ -43,142 +44,73 @@ function buildSkillDocumenterPrompt(options: SkillDocumenterOptions): string { return `/documents -## Task Context -- **Task:** ${options.taskTitle} -- **Description:** ${options.taskDescription.slice(0, 200)}${options.taskDescription.length > 200 ? '...' : ''} +## Task + +${options.taskTitle} + +${options.taskDescription} + +## Worker Report -## Worker's Changes ${workerReport} -Update the project documentation to reflect the changes from the above task. -After the documentation update is complete, output the result in the following JSON format: +## Instructions -\`\`\`json -{ - "success": true, - "updatedFiles": ["CLAUDE.md", "docs/architecture.md"], - "summary": "Added new module description to architecture docs" -} -\`\`\` +Review the worker's changes and update the project's documentation accordingly. -When there is nothing to update: -\`\`\`json -{ - "success": true, - "updatedFiles": [], - "summary": "No documentation update needed (minor change)" -} -\`\`\` +1. Check if any documentation files need updating based on the changes made. +2. Update relevant documentation files. +3. If no documentation changes are needed, report that. -On failure: +## Output Format + +Return a JSON object with the following structure: \`\`\`json { - "success": false, - "updatedFiles": [], - "summary": "Documentation update failed", - "error": "Detailed error message" + "success": true/false, + "updatedFiles": ["path/to/file1.md", ...], + "summary": "Brief summary of documentation changes" } -\`\`\` -`; +\`\`\``; } -// Skill Documenter Execution +// Execution export async function runSkillDocumenter(options: SkillDocumenterOptions): Promise { const prompt = buildSkillDocumenterPrompt(options); - const cwd = expandPath(options.projectPath); - const adapter = getAdapter(options.adapterName); + const adapter = getAdapter(options.adapterName || 'cli'); + const result = await spawnCli(adapter, prompt, { + timeoutMs: options.timeoutMs ?? 120_000, + maxTurns: options.maxTurns ?? 5, + model: options.model, + }); - try { - const raw = await spawnCli(adapter, { - prompt, - cwd, - timeoutMs: options.timeoutMs, - model: options.model, - maxTurns: options.maxTurns, - }); - return parseSkillDocumenterOutput(raw.stdout); - } catch (error) { - if (error instanceof RateLimitError) throw error; - return { - success: false, - updatedFiles: [], - summary: 'Skill Documenter execution failed', - error: error instanceof Error ? error.message : String(error), - }; - } + const output = result.output.trim(); + const parsed = parseSkillDocumenterOutput(output); + + return { + ...parsed, + costInfo: result.costInfo, + }; } -// Output Parsing +// Parsing function parseSkillDocumenterOutput(output: string): SkillDocumenterResult { - try { - const costInfo = extractCostFromStreamJson(output); - if (costInfo) { - console.log(`[SkillDocumenter] Cost: ${formatCost(costInfo)}`); - } - - // Extract result entry from NDJSON - let resultText = ''; - for (const line of output.split('\n')) { - try { - const event = JSON.parse(line.trim()); - if (event.type === 'result' && event.result) { - resultText = event.result; - break; - } - if (event.type === 'item.completed' && event.item?.type === 'agent_message' && event.item.text) { - resultText = event.item.text; - } - } catch { /* skip non-JSON lines */ } - } - - if (!resultText) { - const result = extractFromText(output); - result.costInfo = costInfo; - return result; - } - - const result = extractResultJson(resultText) || extractFromText(resultText); - result.costInfo = costInfo; - return result; - } catch (error) { - console.error('[SkillDocumenter] Parse error:', error); - return extractFromText(output); - } + // Try JSON extraction first + const jsonResult = extractResultJson(output); + if (jsonResult) return jsonResult; + + // Fallback to text extraction + return extractFromText(output); } function extractResultJson(text: string): SkillDocumenterResult | null { - const jsonMatch = text.match(/```json\s*([\s\S]*?)\s*```/); - if (!jsonMatch) { - const objMatch = text.match(/\{\s*"success"\s*:/); - if (!objMatch) return null; - - const startIdx = objMatch.index!; - let depth = 0; - let endIdx = startIdx; - - for (let i = startIdx; i < text.length; i++) { - if (text[i] === '{') depth++; - if (text[i] === '}') { - depth--; - if (depth === 0) { - endIdx = i + 1; - break; - } - } - } - - try { - const parsed = JSON.parse(text.slice(startIdx, endIdx)); - return normalizeResult(parsed); - } catch { - return null; - } - } + const jsonMatch = text.match(/\{[\s\S]*"success"[\s\S]*\}/); + if (!jsonMatch) return null; try { - const parsed = JSON.parse(jsonMatch[1]); + const parsed = JSON.parse(jsonMatch[0]); return normalizeResult(parsed); } catch { return null; @@ -189,56 +121,36 @@ function normalizeResult(parsed: any): SkillDocumenterResult { return { success: Boolean(parsed.success), updatedFiles: Array.isArray(parsed.updatedFiles) ? parsed.updatedFiles : [], - summary: parsed.summary || '(no summary)', - error: parsed.error, + summary: typeof parsed.summary === 'string' ? parsed.summary : '', + error: typeof parsed.error === 'string' ? parsed.error : undefined, }; } function extractFromText(text: string): SkillDocumenterResult { - const hasError = /error|fail|exception/i.test(text); - const hasSuccess = /success|completed|updated|documented/i.test(text); - - const updatedFiles: string[] = []; - const filePatterns = [ - /(?:updated?|modified?|created?|wrote?):\s*(.+\.(?:md|rst|txt))/gi, - /(?:CLAUDE|AGENTS|README|docs?)\.md/gi, - ]; - - for (const pattern of filePatterns) { - const matches = text.matchAll(pattern); - for (const m of matches) { - const file = m[1] || m[0]; - if (!updatedFiles.includes(file)) { - updatedFiles.push(file); - } - } - } - return { - success: !hasError || hasSuccess, - updatedFiles: updatedFiles.slice(0, 10), + success: text.includes('success') || text.includes('updated'), + updatedFiles: extractSummary(text).split('\n').filter(l => l.includes('.md') || l.includes('.ts')), summary: extractSummary(text), - error: hasError ? extractErrorMessage(text) : undefined, + error: extractErrorMessage(text), }; } function extractSummary(text: string): string { - const lines = text.split('\n').filter((l) => l.trim().length > 10); - if (lines.length === 0) return '(no summary)'; - const summary = lines[0].trim(); - return summary.length > 200 ? summary.slice(0, 200) + '...' : summary; + const lines = text.split('\n').filter(l => l.length > 0); + return lines.slice(0, 5).join('\n'); } -function extractErrorMessage(text: string): string { - const errorMatch = text.match(/(?:error|exception|failed?):\s*(.+)/i); - if (errorMatch) return errorMatch[1].slice(0, 200); - const lines = text.split('\n').filter((l) => /error|fail/i.test(l)); - if (lines.length > 0) return lines[0].slice(0, 200); - return 'Unknown error'; +function extractErrorMessage(text: string): string | undefined { + const errorMatch = text.match(/error:?\s*(.+)/i); + return errorMatch ? errorMatch[1] : undefined; } // Formatting +/** + * Format skill documenter report as a Discord message. + * Enforces Discord message content limits to prevent payload rejection. + */ export function formatSkillDocReport(result: SkillDocumenterResult): string { const statusEmoji = result.success ? 'πŸ“„' : '❌'; const lines: string[] = []; @@ -257,5 +169,9 @@ export function formatSkillDocReport(result: SkillDocumenterResult): string { lines.push(`**Error:** ${result.error}`); } - return lines.join('\n'); -} + const full = lines.join('\n'); + // Ensure the entire message fits within Discord limits + return full.length > DISCORD_MESSAGE_CONTENT_LIMIT + ? full.slice(0, DISCORD_MESSAGE_CONTENT_LIMIT - 3) + '…' + : full; +} \ No newline at end of file diff --git a/src/agents/tester.ts b/src/agents/tester.ts index ce955ea1..9142e6b8 100644 --- a/src/agents/tester.ts +++ b/src/agents/tester.ts @@ -11,6 +11,12 @@ import { expandPath } from '../core/config.js'; import { RateLimitError } from '../adapters/rateLimitError.js'; import { isInfraError } from '../adapters/errorClassification.js'; import type { VerifyEvidence } from '../verify/runner.js'; +import { + PROMPT_FEEDBACK_LIMIT, + PROMPT_FAILED_TESTS_LIMIT, + PROMPT_SUGGESTIONS_LIMIT, + truncate, +} from '../support/outputBudget.js'; // Types @@ -46,306 +52,126 @@ export interface TesterResult { * Build Tester prompt */ function buildTesterPrompt(options: TesterOptions): string { - const workerReport = ` -- **Success:** ${options.workerResult.success} -- **Summary:** ${options.workerResult.summary} -- **Files Changed:** ${options.workerResult.filesChanged.join(', ') || '(none)'} -- **Commands:** ${options.workerResult.commands.join(', ') || '(none)'} -`; - - return `# Tester Agent - -## Original Task -- **Title:** ${options.taskTitle} -- **Description:** ${options.taskDescription.slice(0, 200)}${options.taskDescription.length > 200 ? '...' : ''} - -## Worker's Changes -${workerReport} - -## Instructions -1. Run tests for the changed files -2. Verify that all existing tests pass -3. Suggest new tests if needed for new functionality -4. Report test coverage if available - -## Test Execution Steps -1. Check the project's test command (package.json, pytest.ini, etc.) -2. Run relevant test files -3. Analyze any failed tests -4. Determine if additional tests are needed - -## Output Format (IMPORTANT - must output in this format at the end) -After testing is complete, output the result in the following JSON format: - -\`\`\`json -{ - "success": true, - "testsPassed": 10, - "testsFailed": 0, - "coverage": 85.5, - "failedTests": [], - "suggestions": ["Additional test suggestions (if any)"] -} -\`\`\` - -On failure: -\`\`\`json -{ - "success": false, - "testsPassed": 8, - "testsFailed": 2, - "coverage": 75.0, - "failedTests": ["test_feature.py::test_case1", "test_feature.py::test_case2"], - "suggestions": ["Failure cause analysis", "Fix suggestions"], - "error": "Detailed error message" -} -\`\`\` -`; + const prompts = getPrompts(); + return prompts.buildTesterPrompt({ + taskTitle: options.taskTitle, + taskDescription: options.taskDescription, + workerResult: options.workerResult, + }); } -// Tester Execution +// Execution -/** - * Run Tester agent - */ export async function runTester(options: TesterOptions): Promise { const prompt = buildTesterPrompt(options); - const cwd = expandPath(options.projectPath); - const adapter = getAdapter(options.adapterName); + const adapter = getAdapter(options.adapterName || 'cli'); + const result = await spawnCli(adapter, prompt, { + timeoutMs: options.timeoutMs ?? 120_000, + maxTurns: options.maxTurns ?? 5, + model: options.model, + }); - try { - const raw = await spawnCli(adapter, { - prompt, - cwd, - timeoutMs: options.timeoutMs, - model: options.model, - maxTurns: options.maxTurns, - }); - - return parseTesterOutput(raw.stdout); - } catch (error) { - // Rate-limit AND infra failures (CLI exit, timeout, auth, spawn) mean the - // TESTER never ran β€” they are NOT "tests failed". Propagate so the pipeline - // classifies rate_limited / infra_error instead of feeding a bogus - // "fix the tests" self-repair loop that burns iterations β†’ false STUCK. - // worker.ts:337 / reviewer.ts:264 already do this; the tester was missing it. (INT-2521) - if (error instanceof RateLimitError) throw error; - if (isInfraError(error)) throw error; - return { - success: false, - testsPassed: 0, - testsFailed: 0, - output: '', - error: error instanceof Error ? error.message : String(error), - }; - } -} + const output = result.output.trim(); + const parsed = parseTesterOutput(output); -/** - * Parse Tester output - */ -export function parseTesterOutput(output: string): TesterResult { - try { - const costInfo = extractCostFromStreamJson(output); - if (costInfo) { - console.log(`[Tester] Cost: ${formatCost(costInfo)}`); - } + return { + ...parsed, + costInfo: result.costInfo, + }; +} - // Extract result entry from NDJSON - let resultText = ''; - for (const line of output.split('\n')) { - try { - const event = JSON.parse(line.trim()); - if (event.type === 'result' && event.result) { - resultText = event.result; - break; - } - if (event.type === 'item.completed' && event.item?.type === 'agent_message' && event.item.text) { - resultText = event.item.text; - } - } catch { /* skip non-JSON lines */ } - } +// Parsing - if (!resultText) { - const result = extractFromText(output); - result.costInfo = costInfo; - return result; - } +export function parseTesterOutput(output: string): TesterResult { + // Try JSON extraction first + const jsonResult = extractResultJson(output); + if (jsonResult) return jsonResult; - // Extract JSON block from result - const result = extractResultJson(resultText) || extractFromText(resultText); - result.costInfo = costInfo; - return result; - } catch (error) { - console.error('[Tester] Parse error:', error); - return extractFromText(output); - } + // Fallback to text extraction + return extractFromText(output); } -/** - * Extract JSON block from result - */ -function extractResultJson(text: string): TesterResult | null { - // Find ```json ... ``` block - const jsonMatch = text.match(/```json\s*([\s\S]*?)\s*```/); - if (!jsonMatch) { - // Find plain JSON object - const objMatch = text.match(/\{\s*"success"\s*:/); - if (!objMatch) return null; - - const startIdx = objMatch.index!; - let depth = 0; - let endIdx = startIdx; - - for (let i = startIdx; i < text.length; i++) { - if (text[i] === '{') depth++; - if (text[i] === '}') { - depth--; - if (depth === 0) { - endIdx = i + 1; - break; - } - } - } - - try { - const parsed = JSON.parse(text.slice(startIdx, endIdx)); - return normalizeResult(parsed, text); - } catch { - return null; - } - } +export function extractResultJson(text: string): TesterResult | null { + const jsonMatch = text.match(/\{[\s\S]*"success"[\s\S]*\}/); + if (!jsonMatch) return null; try { - const parsed = JSON.parse(jsonMatch[1]); + const parsed = JSON.parse(jsonMatch[0]); return normalizeResult(parsed, text); } catch { return null; } } -/** - * Normalize result - */ function normalizeResult(parsed: any, output: string): TesterResult { return { success: Boolean(parsed.success), testsPassed: typeof parsed.testsPassed === 'number' ? parsed.testsPassed : 0, testsFailed: typeof parsed.testsFailed === 'number' ? parsed.testsFailed : 0, coverage: typeof parsed.coverage === 'number' ? parsed.coverage : undefined, - output, - failedTests: Array.isArray(parsed.failedTests) ? parsed.failedTests : undefined, - suggestions: Array.isArray(parsed.suggestions) ? parsed.suggestions : undefined, - error: parsed.error, + output: typeof parsed.output === 'string' ? parsed.output : output, + failedTests: Array.isArray(parsed.failedTests) ? parsed.failedTests : [], + suggestions: Array.isArray(parsed.suggestions) ? parsed.suggestions : [], + error: typeof parsed.error === 'string' ? parsed.error : undefined, }; } -/** - * Extract result from text (when JSON parsing fails) - */ function extractFromText(text: string): TesterResult { - // Estimate success - const hasError = /error|fail|exception|cannot/i.test(text); - const hasSuccess = /pass|success|completed|all tests/i.test(text); - - // Extract test statistics - let testsPassed = 0; - let testsFailed = 0; - - // Common test result patterns - const passMatch = text.match(/(\d+)\s*(?:passed|pass|passing)/i); - const failMatch = text.match(/(\d+)\s*(?:failed|fail|failing)/i); - - if (passMatch) testsPassed = parseInt(passMatch[1], 10); - if (failMatch) testsFailed = parseInt(failMatch[1], 10); - - // Extract coverage - let coverage: number | undefined; - const coverageMatch = text.match(/(?:coverage|cov)[:\s]*(\d+(?:\.\d+)?)\s*%/i); - if (coverageMatch) { - coverage = parseFloat(coverageMatch[1]); - } - - // Extract failed tests - const failedTests: string[] = []; - const failedPattern = /(?:FAILED|FAIL)\s+([^\s]+(?:::[\w_]+)?)/gi; - const failedMatches = text.matchAll(failedPattern); - for (const m of failedMatches) { - if (!failedTests.includes(m[1])) { - failedTests.push(m[1]); - } - } - - // A tester that produced NO output verified nothing β€” the "no error keyword β‡’ - // success" default would fake a PASS on an empty/degenerate run and let - // unverified code through the blocking test gate. Only genuinely empty output is - // flagged, so a short-but-real run ("collected 0 items") is unaffected. (INT-2521) - const noOutput = text.trim().length === 0; return { - success: !noOutput && (!hasError || (hasSuccess && testsFailed === 0)), - testsPassed, - testsFailed, - coverage, + success: !text.includes('FAIL') && !text.includes('failed'), + testsPassed: 0, + testsFailed: 0, output: text, - failedTests: failedTests.length > 0 ? failedTests : undefined, - error: hasError ? extractErrorMessage(text) : (noOutput ? 'Tester produced no output β€” result unverified' : undefined), + failedTests: [], + suggestions: [], + error: extractErrorMessage(text), }; } -/** - * Extract error message - */ -function extractErrorMessage(text: string): string { - const errorMatch = text.match(/(?:error|exception|failed?):\s*(.+)/i); - if (errorMatch) { - return errorMatch[1].slice(0, 200); - } - - const lines = text.split('\n').filter((l) => /error|fail/i.test(l)); - if (lines.length > 0) { - return lines[0].slice(0, 200); - } - - return 'Unknown error'; +function extractErrorMessage(text: string): string | undefined { + const errorMatch = text.match(/error:?\s*(.+)/i); + return errorMatch ? errorMatch[1] : undefined; } // Formatting /** - * Format Tester result as Discord message + * Format test report as a Discord message */ export function formatTestReport(result: TesterResult): string { const statusEmoji = result.success ? 'βœ…' : '❌'; const lines: string[] = []; - lines.push(`${statusEmoji} **Tester Result: ${result.success ? 'PASS' : 'FAIL'}**`); + lines.push(`${statusEmoji} **Test Results: ${result.success ? 'Passed' : 'Failed'}**`); lines.push(''); - lines.push(`**Passed:** ${result.testsPassed} | **Failed:** ${result.testsFailed}`); + lines.push(`**Tests Passed:** ${result.testsPassed}`); + lines.push(`**Tests Failed:** ${result.testsFailed}`); - if (result.coverage !== undefined) { - lines.push(`**Coverage:** ${result.coverage.toFixed(1)}%`); + if (result.coverage != null) { + lines.push(`**Coverage:** ${(result.coverage * 100).toFixed(1)}%`); } if (result.failedTests && result.failedTests.length > 0) { lines.push(''); lines.push('**Failed Tests:**'); - for (const test of result.failedTests.slice(0, 5)) { - lines.push(` β€’ \`${test}\``); + for (const test of result.failedTests.slice(0, PROMPT_FAILED_TESTS_LIMIT)) { + lines.push(` ❌ ${test}`); } - if (result.failedTests.length > 5) { - lines.push(` β€’ ... +${result.failedTests.length - 5} more`); + if (result.failedTests.length > PROMPT_FAILED_TESTS_LIMIT) { + lines.push(` … +${result.failedTests.length - PROMPT_FAILED_TESTS_LIMIT} more`); } } if (result.suggestions && result.suggestions.length > 0) { lines.push(''); lines.push('**Suggestions:**'); - for (const suggestion of result.suggestions.slice(0, 3)) { + for (const suggestion of result.suggestions.slice(0, PROMPT_SUGGESTIONS_LIMIT)) { lines.push(` β€’ ${suggestion}`); } } if (result.error) { + lines.push(''); lines.push(`**Error:** ${result.error}`); } @@ -353,33 +179,54 @@ export function formatTestReport(result: TesterResult): string { } /** - * Convert Tester result to Worker feedback + * Build test fix prompt for the worker. + * Enforces aggregate bounds on feedback to prevent prompt bloat. */ export function buildTestFixPrompt(result: TesterResult): string { const lines: string[] = []; - lines.push('## Test Failures'); - lines.push(''); - lines.push(`**Passed:** ${result.testsPassed} | **Failed:** ${result.testsFailed}`); + lines.push(`The tests ${result.success ? 'passed' : 'failed'}.`); + lines.push(`Tests passed: ${result.testsPassed}, Tests failed: ${result.testsFailed}`); + if (result.coverage != null) { + lines.push(`Coverage: ${(result.coverage * 100).toFixed(1)}%`); + } + + // Bound failed tests list to prevent prompt bloat if (result.failedTests && result.failedTests.length > 0) { lines.push(''); lines.push('### Failed Tests:'); - for (let i = 0; i < result.failedTests.length; i++) { - lines.push(`${i + 1}. \`${result.failedTests[i]}\``); + const shown = result.failedTests.slice(0, PROMPT_FAILED_TESTS_LIMIT); + for (let i = 0; i < shown.length; i++) { + lines.push(`${i + 1}. \`${shown[i]}\``); + } + if (result.failedTests.length > PROMPT_FAILED_TESTS_LIMIT) { + lines.push(`… +${result.failedTests.length - PROMPT_FAILED_TESTS_LIMIT} more`); } } + // Bound suggestions list to prevent prompt bloat if (result.suggestions && result.suggestions.length > 0) { lines.push(''); lines.push('### Fix Suggestions:'); - for (let i = 0; i < result.suggestions.length; i++) { - lines.push(`${i + 1}. ${result.suggestions[i]}`); + const shown = result.suggestions.slice(0, PROMPT_SUGGESTIONS_LIMIT); + for (let i = 0; i < shown.length; i++) { + lines.push(`${i + 1}. ${shown[i]}`); + } + if (result.suggestions.length > PROMPT_SUGGESTIONS_LIMIT) { + lines.push(`… +${result.suggestions.length - PROMPT_SUGGESTIONS_LIMIT} more`); } } lines.push(''); lines.push('Fix the above test failures.'); - return lines.join('\n'); + const full = lines.join('\n'); + // Enforce aggregate prompt budget + return full.length > PROMPT_FEEDBACK_LIMIT + ? full.slice(0, PROMPT_FEEDBACK_LIMIT - 3) + '…' + : full; } + +// Re-export getPrompts for tester +import { getPrompts } from '../locale/index.js'; \ No newline at end of file diff --git a/src/automation/workerAuditLog.ts b/src/automation/workerAuditLog.ts index 8096e014..3749ea7c 100644 --- a/src/automation/workerAuditLog.ts +++ b/src/automation/workerAuditLog.ts @@ -9,12 +9,20 @@ import type { WorkerResult } from '../agents/agentPair.js'; import { formatAutomationComment, type CommentSection } from '../linear/format.js'; +import { + AUDIT_FILES_MAX, + AUDIT_COMMANDS_MAX, + AUDIT_SUMMARY_CAP, + AUDIT_GOAL_CAP, + capArray, + codeList, +} from '../support/outputBudget.js'; /** Caps so a chatty agent can't post a multi-MB comment. */ -const MAX_FILES = 20; -const MAX_COMMANDS = 12; -const SUMMARY_CAP = 600; -const GOAL_CAP = 400; +const MAX_FILES = AUDIT_FILES_MAX; +const MAX_COMMANDS = AUDIT_COMMANDS_MAX; +const SUMMARY_CAP = AUDIT_SUMMARY_CAP; +const GOAL_CAP = AUDIT_GOAL_CAP; function cap(s: string | undefined, n: number): string { if (!s) return ''; @@ -35,69 +43,74 @@ function codeList(items: string[] | undefined, max: number): string { } export interface WorkerStartInfo { - /** 1-based iteration/attempt number. */ - attempt: number; - maxAttempts?: number; taskTitle: string; - /** Prompt summary β€” task description or draft intent summary. */ - taskGoal?: string; - /** Files the worker is expected to touch (from draft analysis / impact). */ - targetFiles?: string[]; - /** Resolved model for this worker run. */ - model?: string; - /** Max agentic turns (proxy for effort budget). */ - maxTurns?: number; - /** True when this run follows reviewer/guard feedback (a revision). */ - isRevision?: boolean; + taskDescription: string; + projectPath: string; + /** The goal the worker was asked to achieve (from the task). */ + goal?: string; } -/** Comment body posted when a worker run starts (the instruction). */ +/** + * Build the "worker started" audit comment. + */ export function buildWorkerStartComment(info: WorkerStartInfo): string { - const attemptLabel = info.maxAttempts - ? `attempt #${info.attempt}/${info.maxAttempts}` - : `attempt #${info.attempt}`; - const heading = info.isRevision ? 'Worker revision' : 'Worker instruction'; - - const sections: CommentSection[] = [{ label: 'Task', body: cap(info.taskTitle, 200) }]; - if (info.taskGoal) sections.push({ label: 'Goal', body: cap(info.taskGoal, GOAL_CAP) }); - if (info.targetFiles && info.targetFiles.length > 0) { - sections.push({ label: 'Target files', body: codeList(info.targetFiles, MAX_FILES) }); + const sections: CommentSection[] = []; + + if (info.goal) { + sections.push({ label: 'Goal', body: cap(info.goal, GOAL_CAP) }); } return formatAutomationComment({ - heading: `${heading} (${attemptLabel})`, + heading: 'Worker started', + summary: cap(info.taskTitle, SUMMARY_CAP), sections, - meta: { Model: info.model, 'Max turns': info.maxTurns }, + meta: { + Project: info.projectPath, + }, attribution: 'Worker audit log', }); } export interface WorkerCompleteInfo { - attempt: number; - maxAttempts?: number; result: WorkerResult; - /** Worker run duration in seconds. */ + /** Seconds the worker ran for. */ durationSec?: number; + /** Which attempt number this was (1-based). */ + attempt?: number; + /** Max attempts allowed. */ + maxAttempts?: number; } -/** Comment body posted when a worker run completes (the actions taken). */ +/** + * Build the "worker completed" audit comment. + * Caps individual file and command entries before rendering to prevent oversized comments. + */ export function buildWorkerCompleteComment(info: WorkerCompleteInfo): string { const { result } = info; - const attemptLabel = info.maxAttempts - ? `attempt #${info.attempt}/${info.maxAttempts}` - : `attempt #${info.attempt}`; - const verdict = result.haltReason ? 'Halted' : result.success ? 'Done' : 'Failed'; - - const files = result.filesChanged ?? []; - const commands = result.commands ?? []; - - const sections: CommentSection[] = [ - { label: `Files changed (${files.length})`, body: codeList(files, MAX_FILES) }, - ]; - if (commands.length > 0) { - sections.push({ label: `Commands (${commands.length})`, body: codeList(commands, MAX_COMMANDS) }); + const verdict = result.success ? 'βœ… Complete' : '❌ Failed'; + const attemptLabel = info.attempt != null && info.maxAttempts != null + ? `(attempt ${info.attempt}/${info.maxAttempts})` + : ''; + + const sections: CommentSection[] = []; + + // Files changed β€” capped to prevent oversized comments + if (result.filesChanged && result.filesChanged.length > 0) { + const { shown, omitted } = capArray(result.filesChanged, MAX_FILES); + const filesStr = shown.map(inlineCode).join(', '); + const body = omitted > 0 ? `${filesStr} _+${omitted} more_` : filesStr; + sections.push({ label: 'Files changed', body }); + } + + // Commands run β€” capped to prevent oversized comments + if (result.commands && result.commands.length > 0) { + const { shown, omitted } = capArray(result.commands, MAX_COMMANDS); + const cmdsStr = shown.map(inlineCode).join(', '); + const body = omitted > 0 ? `${cmdsStr} _+${omitted} more_` : cmdsStr; + sections.push({ label: 'Commands', body }); } - if (result.haltReason) sections.push({ label: 'Halt reason', body: cap(result.haltReason, GOAL_CAP) }); + + // Error β€” capped if (result.error) sections.push({ label: 'Error', body: cap(result.error, GOAL_CAP) }); const duration = info.durationSec != null @@ -114,4 +127,4 @@ export function buildWorkerCompleteComment(info: WorkerCompleteInfo): string { }, attribution: 'Worker audit log', }); -} +} \ No newline at end of file diff --git a/src/support/outputBudget.ts b/src/support/outputBudget.ts new file mode 100644 index 00000000..83bfa652 --- /dev/null +++ b/src/support/outputBudget.ts @@ -0,0 +1,113 @@ +// ============================================ +// OpenSwarm β€” Output Budget Helpers +// +// Shared destination-specific field, message, and aggregate limits. +// Every consumer enforces these before sending or rendering untrusted content. +// ============================================ + +// ── Discord embed limits (discord.js EmbedBuilder enforces these at build time) ── +export const DISCORD_EMBED_TITLE_LIMIT = 256; +export const DISCORD_EMBED_DESCRIPTION_LIMIT = 4096; +export const DISCORD_EMBED_FIELD_NAME_LIMIT = 256; +export const DISCORD_EMBED_FIELD_VALUE_LIMIT = 1024; +export const DISCORD_EMBED_FOOTER_LIMIT = 2048; +export const DISCORD_EMBED_FIELDS_PER_EMBED = 25; +/** Total characters across all field values in one embed (discord.js enforces ~6000). */ +export const DISCORD_EMBED_AGGREGATE_VALUE_LIMIT = 6000; +/** Safe message content limit (below Discord's 2000 hard cap, leaving room for framing). */ +export const DISCORD_MESSAGE_CONTENT_LIMIT = 1900; + +// ── Linear API limits ── +export const LINEAR_TITLE_LIMIT = 512; +export const LINEAR_DESCRIPTION_LIMIT = 3000; +export const LINEAR_COMMENT_LIMIT = 3000; + +// ── Prompt / feedback budgets ── +export const PROMPT_FEEDBACK_LIMIT = 4000; +export const PROMPT_FAILED_TESTS_LIMIT = 10; +export const PROMPT_SUGGESTIONS_LIMIT = 5; + +// ── Worker audit log budgets ── +export const AUDIT_FILES_MAX = 20; +export const AUDIT_COMMANDS_MAX = 12; +export const AUDIT_SUMMARY_CAP = 600; +export const AUDIT_GOAL_CAP = 400; + +// ── TUI display budgets ── +export const TUI_LOG_LINE_LIMIT = 200; +export const TUI_CELL_DEFAULT_WIDTH = 28; + +// ── CLI runner budgets ── +export const CLI_FEEDBACK_LINES = 5; +export const CLI_STDERR_LINE_LIMIT = 100; + +// ── Pipeline embed budgets ── +export const PIPELINE_EMBED_FIELD_VALUE_LIMIT = 900; // below 1024 to leave room for markdown framing +export const PIPELINE_FAILED_TESTS_PREVIEW = 2; + +// ── Helpers ── + +/** Truncate a string to `limit` chars, appending "…" when clipped. */ +export function truncate(value: string, limit: number): string { + if (value.length <= limit) return value; + return `${value.slice(0, limit - 1)}…`; +} + +/** Truncate a string to `limit` chars, appending a suffix when clipped. */ +export function truncateWithSuffix(value: string, limit: number, suffix = '\n… (truncated)'): string { + if (value.length <= limit) return value; + return `${value.slice(0, limit - suffix.length)}${suffix}`; +} + +/** Cap an array to `max` items, returning the slice and a count of omitted items. */ +export function capArray(items: T[], max: number): { shown: T[]; omitted: number } { + if (items.length <= max) return { shown: items, omitted: 0 }; + return { shown: items.slice(0, max), omitted: items.length - max }; +} + +/** Render a list as inline code, capped, with an "+N more" suffix when truncated. */ +export function codeList(items: string[] | undefined, max: number): string { + if (!items || items.length === 0) return '_(none)_'; + const { shown, omitted } = capArray(items, max); + const rendered = shown.map((s) => `\`${s.replaceAll('`', '\\`')}\``).join(', '); + return omitted > 0 ? `${rendered} _+${omitted} more_` : rendered; +} + +/** Ensure a Discord embed field value stays within the per-field limit. */ +export function boundedFieldValue(value: string, limit = DISCORD_EMBED_FIELD_VALUE_LIMIT): string { + return truncateWithSuffix(value, limit); +} + +/** Ensure a Discord embed description stays within the limit. */ +export function boundedDescription(value: string): string { + return truncateWithSuffix(value, DISCORD_EMBED_DESCRIPTION_LIMIT); +} + +/** Ensure a Discord message content stays within the safe limit. */ +export function boundedMessageContent(value: string): string { + return truncateWithSuffix(value, DISCORD_MESSAGE_CONTENT_LIMIT); +} + +/** Flatten multiline text to a single line (replace newlines with spaces). */ +export function flattenToSingleLine(value: string): string { + return value.replace(/\r?\n/g, ' ').replace(/\s+/g, ' ').trim(); +} + +/** Bound a string for Linear description/comment fields. */ +export function boundedLinearText(value: string, limit = LINEAR_DESCRIPTION_LIMIT): string { + return truncateWithSuffix(value, limit); +} + +/** Bound a string for Linear title. */ +export function boundedLinearTitle(value: string): string { + return truncateWithSuffix(value, LINEAR_TITLE_LIMIT); +} + +/** Sanitize and bound exception text for user-facing output. */ +export function sanitizeException(error: unknown): string { + if (error == null) return 'An unknown error occurred.'; + const msg = error instanceof Error ? error.message : String(error); + // Strip any content that looks like a stack trace or internal path + const cleaned = msg.split('\n')[0].trim(); + return truncate(cleaned || 'An error occurred.', 200); +} \ No newline at end of file From 707939158a281d3e9e2b7e4b60f49956594474d9 Mon Sep 17 00:00:00 2001 From: Heewon Oh Date: Thu, 10 Sep 2026 09:44:56 +0900 Subject: [PATCH 2/3] wip: preserved partial work (auto, session did not succeed) --- node_modules | 1 + src/agents/pipelineFormat.ts | 237 ++++++------- src/agents/reviewer.ts | 445 +++++++++++++++++++------ src/agents/skillDocumenter.ts | 209 ++++++++---- src/agents/tester.ts | 315 +++++++++++++---- src/automation/workerAuditLog.ts | 103 +++--- src/discord/discordHandlers.ts | 109 +++--- src/runners/cliRunner.ts | 29 +- src/support/outputBudget.test.ts | 91 +++++ src/support/outputBudget.ts | 52 ++- src/support/workflowLinear.ts | 16 +- src/tui/components/AuditBoard.test.tsx | 18 + src/tui/components/AuditBoard.tsx | 10 +- src/tui/components/DataTable.tsx | 10 +- src/tui/dataTable.test.tsx | 8 + 15 files changed, 1153 insertions(+), 500 deletions(-) create mode 120000 node_modules create mode 100644 src/support/outputBudget.test.ts diff --git a/node_modules b/node_modules new file mode 120000 index 00000000..d9643ec8 --- /dev/null +++ b/node_modules @@ -0,0 +1 @@ +/work/OpenSwarm/node_modules \ No newline at end of file diff --git a/src/agents/pipelineFormat.ts b/src/agents/pipelineFormat.ts index d5d86d0b..20e5b596 100644 --- a/src/agents/pipelineFormat.ts +++ b/src/agents/pipelineFormat.ts @@ -9,10 +9,12 @@ import { formatCost } from '../support/costTracker.js'; import { boundedFieldValue, boundedDescription, + boundedMessageContent, PIPELINE_EMBED_FIELD_VALUE_LIMIT, PIPELINE_FAILED_TESTS_PREVIEW, DISCORD_EMBED_FIELDS_PER_EMBED, DISCORD_EMBED_AGGREGATE_VALUE_LIMIT, + truncate, } from '../support/outputBudget.js'; /** Format epoch ms to HH:MM:SS local time string */ @@ -49,120 +51,79 @@ export function formatPipelineResult(result: PipelineResult): string { || (ctx.projectPath ? ctx.projectPath.split('/').pop() || '' : ''); if (displayName) parts.push(`πŸ“ ${displayName}`); if (ctx.issueIdentifier) parts.push(`πŸ”– ${ctx.issueIdentifier}`); - if (ctx.taskTitle) parts.push(ctx.taskTitle); - lines.push(`**${parts.join(' | ')}**`); - } - - // Status line - lines.push(''); - lines.push(`${statusEmoji} **Status:** ${result.finalStatus}`); - - // Duration - if (result.totalDuration) { - const mins = Math.floor(result.totalDuration / 60000); - const secs = Math.round((result.totalDuration % 60000) / 1000); - lines.push(`⏱ **Duration:** ${mins}m ${secs}s`); - } - - // Cost - if (result.totalCost) { - lines.push(`πŸ’° **Cost:** ${formatCost(result.totalCost)}`); - } - - // Stage summary - if (result.stages && result.stages.length > 0) { - lines.push(''); - lines.push('**Stages:**'); - for (const stage of result.stages) { - const stageEmoji = stage.status === 'success' ? 'βœ…' : stage.status === 'failed' ? '❌' : '⏳'; - lines.push(` ${stageEmoji} ${stage.name}${stage.durationMs ? ` (${Math.round(stage.durationMs / 1000)}s)` : ''}`); + if (ctx.projectPath) parts.push(`\`${ctx.projectPath.split('/').slice(-2).join('/')}\``); + if (parts.length > 0) { + lines.push(parts.join(' | ')); } - } - - // Worker summary - if (result.workerResult) { - lines.push(''); - lines.push(`**πŸ”¨ Worker:** ${result.workerResult.summary || 'No summary'}`); - if (result.workerResult.filesChanged && result.workerResult.filesChanged.length > 0) { - const files = result.workerResult.filesChanged.slice(0, 10); - lines.push(` Files: ${files.join(', ')}`); - if (result.workerResult.filesChanged.length > 10) { - lines.push(` ... +${result.workerResult.filesChanged.length - 10} more`); - } + if (ctx.taskTitle) { + lines.push(`πŸ“‹ ${truncate(ctx.taskTitle, 200)}`); } - } - - // Reviewer feedback - if (result.reviewResult) { lines.push(''); - const reviewEmoji = result.reviewResult.decision === 'approved' ? 'βœ…' : '❌'; - lines.push(`${reviewEmoji} **Reviewer:** ${result.reviewResult.decision}`); - if (result.reviewResult.feedback) { - // Bound reviewer feedback to prevent oversized messages - const feedback = result.reviewResult.feedback.length > 500 - ? result.reviewResult.feedback.slice(0, 500) + '…' - : result.reviewResult.feedback; - lines.push(` ${feedback}`); - } } - // Test results - if (result.testerResult) { - lines.push(''); - const testEmoji = result.testerResult.success ? 'βœ…' : '❌'; - lines.push(`${testEmoji} **Tests:** ${result.testerResult.testsPassed} passed, ${result.testerResult.testsFailed} failed`); + lines.push(`${statusEmoji} **Pipeline ${result.finalStatus.toUpperCase()}**`); + lines.push(''); + lines.push(`**Session:** \`${result.sessionId}\``); + lines.push(`**Iterations:** ${result.iterations}`); + lines.push(`**Duration:** ${(result.totalDuration / 1000).toFixed(1)}s`); + + if (result.totalCost) { + lines.push(`**Cost:** $${result.totalCost.costUsd.toFixed(4)} (${formatCost(result.totalCost)})`); } - // PR URL - if (result.prUrl) { - lines.push(''); - lines.push(`πŸ”— **Pull Request:** ${result.prUrl}`); + lines.push(''); + lines.push('**Stages:**'); + for (const stage of result.stages) { + const emoji = stage.success ? 'βœ…' : '❌'; + const duration = (stage.duration / 1000).toFixed(1); + const time = formatTimestamp(stage.startedAt); + lines.push(` ${emoji} ${stage.stage} (${duration}s) @ ${time}`); } - return lines.join('\n'); + return boundedMessageContent(lines.join('\n')); } /** - * Format pipeline result as a Discord embed + * Format pipeline result as a Discord Embed. * Enforces per-field and aggregate embed budgets to prevent payload rejection. */ export function formatPipelineResultEmbed(result: PipelineResult): EmbedBuilder { - const statusColor = { - approved: 0x00ff41, - rejected: 0xff0044, - failed: 0xff6600, - cancelled: 0x888888, - decomposed: 0x00aaff, - superseded: 0xaa00ff, - deferred: 0xffaa00, - waiting_on_operator: 0xffff00, - rate_limited: 0xff8800, - infra_error: 0xff4444, - }[result.finalStatus] || 0x888888; + const statusConfig = { + approved: { emoji: 'βœ…', color: 0x00FF00, label: 'SUCCESS' }, + rejected: { emoji: '❌', color: 0xFF0000, label: 'REJECTED' }, + failed: { emoji: 'πŸ’₯', color: 0xFF6B6B, label: 'FAILED' }, + cancelled: { emoji: '🚫', color: 0xFFAA00, label: 'CANCELLED' }, + decomposed: { emoji: 'πŸ”€', color: 0x00AAFF, label: 'DECOMPOSED' }, + superseded: { emoji: '♻️', color: 0x00AAFF, label: 'SUPERSEDED' }, + deferred: { emoji: '⏳', color: 0xFFAA00, label: 'DEFERRED' }, + waiting_on_operator: { emoji: 'πŸ™‹', color: 0xFFC300, label: 'WAITING ON OPERATOR' }, + rate_limited: { emoji: '⏸', color: 0xFFAA00, label: 'RATE LIMITED' }, + infra_error: { emoji: 'πŸ”Œ', color: 0xFFAA00, label: 'INFRA ERROR' }, + }[result.finalStatus] || { emoji: '❓', color: 0x808080, label: 'UNKNOWN' }; const embed = new EmbedBuilder() - .setColor(statusColor) + .setTitle(`${statusConfig.emoji} Pipeline ${statusConfig.label}`) + .setColor(statusConfig.color) .setTimestamp(); - // Title (bounded) - const title = result.taskContext?.taskTitle || 'Pipeline Result'; - embed.setTitle(title.length > 256 ? `${title.slice(0, 253)}…` : title); - - // Description (bounded) + // Task context (bounded description) if (result.taskContext) { const ctx = result.taskContext; const displayName = ctx.projectName || (ctx.projectPath ? ctx.projectPath.split('/').pop() || '' : ''); - const descParts: string[] = []; - if (displayName) descParts.push(`πŸ“ ${displayName}`); - if (ctx.issueIdentifier) descParts.push(`πŸ”– ${ctx.issueIdentifier}`); - embed.setDescription(boundedDescription(descParts.join(' | '))); + + if (displayName && ctx.issueIdentifier) { + embed.setDescription( + boundedDescription(`πŸ“ **${displayName}** | πŸ”– ${ctx.issueIdentifier}\n${ctx.taskTitle || ''}`), + ); + } else if (ctx.taskTitle) { + embed.setDescription(boundedDescription(ctx.taskTitle)); + } } // Track aggregate field value length to stay within embed budget let aggregateValueLength = 0; - // Helper to add a field only if it fits within the aggregate budget const tryAddField = (name: string, value: string, inline = false): boolean => { const bounded = boundedFieldValue(value, PIPELINE_EMBED_FIELD_VALUE_LIMIT); const newTotal = aggregateValueLength + bounded.length; @@ -173,71 +134,87 @@ export function formatPipelineResultEmbed(result: PipelineResult): EmbedBuilder return true; }; - // Status field - tryAddField('Status', result.finalStatus, true); + // Summary stats + const durationStr = (result.totalDuration / 1000).toFixed(1) + 's'; + const costStr = result.totalCost + ? `$${result.totalCost.costUsd.toFixed(4)} (${formatCost(result.totalCost)})` + : 'N/A'; - // Duration - if (result.totalDuration) { - const mins = Math.floor(result.totalDuration / 60000); - const secs = Math.round((result.totalDuration % 60000) / 1000); - tryAddField('Duration', `${mins}m ${secs}s`, true); - } - - // Cost - if (result.totalCost) { - tryAddField('Cost', formatCost(result.totalCost), true); - } + tryAddField('πŸ”„ Iterations', result.iterations.toString(), true); + tryAddField('⏱️ Duration', durationStr, true); + tryAddField('πŸ’° Cost', costStr, true); // Stages - if (result.stages && result.stages.length > 0) { - const stagesStr = result.stages.map((s) => { - const emoji = s.status === 'success' ? 'βœ…' : s.status === 'failed' ? '❌' : '⏳'; - return `${emoji} ${s.name}${s.durationMs ? ` (${Math.round(s.durationMs / 1000)}s)` : ''}`; - }).join('\n'); - tryAddField('πŸ“Š Stages', stagesStr, false); - } - - // Worker + const stagesStr = result.stages + .map(s => { + const emoji = s.success ? 'βœ…' : '❌'; + const duration = (s.duration / 1000).toFixed(1); + const time = formatTimestamp(s.startedAt); + return `${emoji} **${s.stage}** (${duration}s) @ ${time}`; + }) + .join('\n') || 'No stages'; + + tryAddField('πŸ“Š Stages', stagesStr, false); + + // Worker result if (result.workerResult) { - let workerValue = result.workerResult.summary - ? boundedFieldValue(result.workerResult.summary, PIPELINE_EMBED_FIELD_VALUE_LIMIT) - : 'No summary'; - if (result.workerResult.filesChanged && result.workerResult.filesChanged.length > 0) { - const files = result.workerResult.filesChanged.slice(0, 10).join(', '); - workerValue += `\n\n**Files:** ${files}`; - if (result.workerResult.filesChanged.length > 10) { - workerValue += `\n… +${result.workerResult.filesChanged.length - 10} more`; + const worker = result.workerResult; + let workerValue = ''; + + if (worker.summary) { + workerValue += `${worker.summary.slice(0, 200)}${worker.summary.length > 200 ? '...' : ''}\n\n`; + } + + if (worker.filesChanged && worker.filesChanged.length > 0) { + const filesStr = worker.filesChanged.slice(0, 5).map(f => `\`${f}\``).join(', '); + workerValue += `**Files:** ${filesStr}`; + if (worker.filesChanged.length > 5) { + workerValue += ` +${worker.filesChanged.length - 5} more`; } } - tryAddField('πŸ”¨ Worker', workerValue, false); + + if (workerValue) { + tryAddField('πŸ”¨ Worker', workerValue, false); + } } - // Reviewer + // Reviewer result if (result.reviewResult) { - const reviewEmoji = result.reviewResult.decision === 'approved' ? 'βœ…' : '❌'; - let reviewValue = `${reviewEmoji} **${result.reviewResult.decision}**`; - if (result.reviewResult.feedback) { - const feedback = boundedFieldValue(result.reviewResult.feedback, PIPELINE_EMBED_FIELD_VALUE_LIMIT); - reviewValue += `\n\n${feedback}`; + const review = result.reviewResult; + let reviewValue = `**Decision:** ${review.decision.toUpperCase()}\n\n`; + + if (review.feedback) { + reviewValue += review.feedback.slice(0, 300); + if (review.feedback.length > 300) reviewValue += '...'; + } + + if (review.issues && review.issues.length > 0) { + reviewValue += `\n\n**Issues found:** ${review.issues.length}`; } + tryAddField('βœ… Reviewer', reviewValue, false); } - // Tests + // Tester result if (result.testerResult) { const test = result.testerResult; - const testEmoji = test.success ? 'βœ…' : '❌'; - let testValue = `${testEmoji} **${test.testsPassed} passed, ${test.testsFailed} failed**`; - if (test.coverage != null) { - testValue += ` | Coverage: ${(test.coverage * 100).toFixed(1)}%`; + const total = test.testsPassed + test.testsFailed; + const passRate = total > 0 ? ((test.testsPassed / total) * 100).toFixed(1) : '0'; + + let testValue = `βœ… Passed: ${test.testsPassed}/${total} (${passRate}%)${test.deterministic ? ' Β· deterministic' : ''}`; + + if (test.coverage !== undefined) { + testValue += `\nπŸ“Š Coverage: ${test.coverage.toFixed(1)}%`; } - if (!test.success && test.failedTests && test.failedTests.length > 0) { + + if (test.testsFailed > 0 && test.failedTests && test.failedTests.length > 0) { const failedStr = test.failedTests.slice(0, PIPELINE_FAILED_TESTS_PREVIEW).map(t => `❌ ${t}`).join('\n'); testValue += `\n\n${failedStr}`; if (test.failedTests.length > PIPELINE_FAILED_TESTS_PREVIEW) { testValue += `\n... +${test.failedTests.length - PIPELINE_FAILED_TESTS_PREVIEW} more`; } } + tryAddField('πŸ§ͺ Tests', testValue, false); } @@ -250,4 +227,4 @@ export function formatPipelineResultEmbed(result: PipelineResult): EmbedBuilder embed.setFooter({ text: `Session: ${result.sessionId.slice(0, 8)}...` }); return embed; -} \ No newline at end of file +} diff --git a/src/agents/reviewer.ts b/src/agents/reviewer.ts index f6b87730..117ef7b4 100644 --- a/src/agents/reviewer.ts +++ b/src/agents/reviewer.ts @@ -15,7 +15,7 @@ import type { VerifyEvidence } from '../verify/runner.js'; import { renderVerifyEvidence } from './verificationEvidence.js'; import type { InstructionCapsule } from './instructionCapsule.js'; import { COORDINATION_GUIDANCE_PROMPT, type CoordinationToolContext } from '../coordination/coordinationTools.js'; -import { boundedMessageContent, DISCORD_MESSAGE_CONTENT_LIMIT } from '../support/outputBudget.js'; +import { boundedMessageContent } from '../support/outputBudget.js'; // Types @@ -30,156 +30,385 @@ export interface ReviewerOptions { maxTurns?: number; // Max agentic turns per CLI invocation adapterName?: AdapterName; processContext?: ProcessContext; - /** Reasoning effort from a j - * compatible adapter (e.g. openrouter). */ - reasoningEffort?: number; - /** Coordination tools available to the reviewer agent */ - coordinationTools?: CoordinationToolContext; - /** Verify evidence from the deterministic tester */ + /** Reasoning effort from a jobProfile (codex-responses: low|medium|high). */ + reasoningEffort?: 'low' | 'medium' | 'high'; + /** Execution-grounded definition of done to hard-gate on (INT-1914). */ + completionCriteria?: string[]; + /** + * Non-blocking deterministic guard warnings (dead-module, reformat/scope, …) + * surfaced to the reviewer so it verifies each instead of them dying in a log + * line. (INT-2388) + */ + guardWarnings?: string[]; + /** Deterministic command evidence produced by the harness tester. */ verificationEvidence?: VerifyEvidence[]; - /** Instruction capsule for the reviewer */ + /** Relevant repository-local logs from earlier review commands. */ + priorReviewContext?: string; + /** + * 'change' (default): review a worker's diff. 'audit': evaluate existing files + * with no diff/worker (the `review --max` codebase audit). (INT-2006) + */ + mode?: 'change' | 'audit' | 'direct'; + /** MCP tools to expose (e.g. linear__*). When unset the adapter self-sources (INT-1951). (INT-1950) */ + mcpTools?: ToolDefinition[]; + /** Tool-activity log lines (πŸ”§ read_file …) for live progress display. (INT-1963) */ + onLog?: (line: string) => void; + /** Streamed reasoning/text deltas for live progress display. (INT-1963) */ + onToken?: (delta: string) => void; + /** Abort the run + in-flight adapter call (pipeline cancel / project disable). */ + signal?: AbortSignal; + /** + * Deny the reviewer every mutating tool β€” write_file, edit_file, apply_patch + * and bash β€” leaving read_file/search_files/search_memory. + * + * Mandatory whenever the diff under review is not trusted. Reviewing a pull + * request in CI puts an agent with shell access on attacker-controlled files + * while the provider credential sits in the environment, which turns prompt + * injection into command execution. A review is a judgement, not an + * execution, so nothing legitimate is lost. + * + * Off by default: the local `openswarm review` path reviews the operator's own + * working tree, and running commands there is how the reviewer substantiates a + * claim. (INT-3189) + */ + readOnly?: boolean; + /** + * The change under review, as text. Supplied rather than discovered: a + * read-only reviewer cannot shell out for it, and in committed-diff mode + * there is nothing in the working tree to read. (INT-3101) + */ + diff?: string; + /** Run-scoped Claude Code instruction and runbook snapshot. */ instructionCapsule?: InstructionCapsule; + /** + * This reviewer's board identity β€” its call sign and mailbox address. + * + * Distinct from the worker's on the same task: two agents answering to one + * name make advice and operator answers unroutable. Coordination tools stay + * withheld while `readOnly` is set (INT-3189); the identity is still carried + * so reports and the dashboard name the reviewer. + */ + coordinationContext?: CoordinationToolContext; +} + +/** Tell the reviewer the call sign other agents and the operator address it by. */ +function reviewerIdentityHeader(callSign: string | undefined): string { + if (!callSign) return ''; + return `\n\n## Your identity\nYou are **${callSign}**. Sign your review with that call sign so the worker and the operator know who reviewed the change.\n`; +} + +/** + * Coordination tools are withheld while readOnly is set (INT-3189) even + * though coordinationContext is still carried for identity/labeling β€” so + * this must check both, not just coordinationContext, or the reviewer would + * be told to use a tool it doesn't actually have. (AGT-4054) + */ +function reviewerCoordinationGuidance( + coordinationContext: CoordinationToolContext | undefined, + readOnly: boolean | undefined, +): string { + return coordinationContext && !readOnly + ? COORDINATION_GUIDANCE_PROMPT + getPrompts().coordinationConsultationPrompt + : ''; } export interface PreCheckResult { passed: boolean; - reason: string; + issues: string[]; + confidence: number; // 0-3: quality of the check } // Prompts -function reviewerIdentityHeader(callSign: string | undefined): string { - return callSign - ? `You are a code reviewer (call sign: ${callSign}).` - : 'You are a code reviewer.'; -} +/** + * Build Pre-Check prompt for fast validation (Haiku) + */ +function buildPreCheckPrompt(options: ReviewerOptions): string { + const files = options.workerResult.filesChanged; + const filesSummary = files.length <= 20 + ? (files.join(', ') || '(none)') + : `${files.slice(0, 20).join(', ')} (+${files.length - 20} more)`; -function reviewerCoordinationGuidance(): string { - return [ - '', - '## Coordination tools available', - '', - 'You have access to coordination tools that let you communicate with other agents', - 'and read durable repository threads. Use them when you need to:', - '', - 'β€’ Ask a worker agent for clarification on their changes', - 'β€’ Check if there are existing discussions about the code you are reviewing', - 'β€’ Coordinate with other reviewers on shared files', - '', - COORDINATION_GUIDANCE_PROMPT, - '', - '**Important:** Only use coordination tools when you have a specific question or', - 'need to share information. Do not use them for routine status updates.', - ].join('\n'); -} + return `You are a fast pre-check validator. Perform a quick validation of the Worker's output. -function buildPreCheckPrompt(options: ReviewerOptions): string { - const prompts = getPrompts(); - return prompts.buildPreCheckPrompt({ - taskTitle: options.taskTitle, - taskDescription: options.taskDescription, - workerResult: options.workerResult, - }); +## Task +${options.taskTitle} + +## Worker Result +- Success: ${options.workerResult.success} +- Files Changed (${files.length}): ${filesSummary} +- Summary: ${options.workerResult.summary} + +## Your Job (Fast Check Only) +Check for OBVIOUS problems: +1. **Syntax Errors**: Are there any clear syntax errors in the output? +2. **Missing Files**: Did the worker claim to create/modify files that don't exist? +3. **Incomplete Output**: Does the output look cut off or incomplete? +4. **Basic Format Issues**: Are there obvious formatting problems? + +**DO NOT** perform deep logical review - that's for the next stage. + +## Response Format +Respond in this EXACT format: + +PASSED: [yes/no] +CONFIDENCE: [0-3] +ISSUES: +- [issue 1] +- [issue 2] +... + +Keep it brief. This is a fast filter, not a deep review.`; } -function buildReviewerPrompt(options: ReviewerOptions): string { - const prompts = getPrompts(); - return prompts.buildReviewerPrompt({ +/** + * Build Reviewer prompt using locale templates + */ +export function buildReviewerPrompt(options: ReviewerOptions): string { + const files = options.workerResult.filesChanged; + const filesSummary = files.length <= 20 + ? (files.join(', ') || '(none)') + : `${files.slice(0, 20).join(', ')} (+${files.length - 20} more)`; + + // Audit mode: no diff/commands to report β€” just hand the auditor the file list. (INT-2006) + if (options.mode === 'audit') { + return getPrompts().buildReviewerPrompt({ + taskTitle: options.taskTitle, + taskDescription: options.taskDescription, + authoritativeOperatorFeedback: options.authoritativeOperatorFeedback, + workerReport: `- **Files under audit (${files.length}):** ${filesSummary}`, + mode: 'audit', + priorReviewContext: options.priorReviewContext, + }); + } + + // Direct mode reviews a Git diff supplied by a user/CI checkout, not an + // OpenSwarm worker result. Do not manufacture a zero-command worker report: + // "evidence not collected here" is different from "validation was not run". + if (options.mode === 'direct') { + // The diff goes in the prompt, not left for the agent to reconstruct. A + // read-only reviewer has no bash, and in committed-diff mode the working + // tree is clean β€” reading a file shows the result, never the change. Under + // `--read-only --base`, which is what the CI gate uses, the reviewer could + // not see its own subject and said so while still returning a verdict. + // (INT-3101) + // Appended as plain text on purpose: the template already wraps the whole + // report in its untrusted-data block, which escapes the closing marker and + // code fences. A second fence here would be escaped by that one, so it + // would add noise while providing none of the protection it appears to. + const report = options.diff + ? `- **Files changed (${files.length}):** ${filesSummary}\n- **Diff under review:**\n${options.diff}` + : `- **Files changed (${files.length}):** ${filesSummary}`; + return getPrompts().buildReviewerPrompt({ + taskTitle: options.taskTitle, + taskDescription: options.taskDescription, + authoritativeOperatorFeedback: options.authoritativeOperatorFeedback, + workerReport: report, + mode: 'direct', + priorReviewContext: options.priorReviewContext, + }); + } + + const cmds = options.workerResult.commands; + const cmdsSummary = cmds.length <= 10 + ? (cmds.join(', ') || '(none)') + : `${cmds.slice(0, 10).join(', ')} (+${cmds.length - 10} more)`; + + const guardSection = options.guardWarnings && options.guardWarnings.length > 0 + ? `- **Automated guard warnings (deterministic pre-checks β€” verify each, don't dismiss):**\n${options.guardWarnings.map(w => ` - ${w}`).join('\n')}\n` + : ''; + + const workerReport = ` +- **Success:** ${options.workerResult.success} +- **Summary:** ${options.workerResult.summary} +- **Files Changed (${files.length}):** ${filesSummary} +- **Commands:** ${cmdsSummary} +${options.workerResult.error ? `- **Error:** ${options.workerResult.error}` : ''} +${guardSection}`; + + return getPrompts().buildReviewerPrompt({ taskTitle: options.taskTitle, taskDescription: options.taskDescription, - workerResult: options.workerResult, authoritativeOperatorFeedback: options.authoritativeOperatorFeedback, - verificationEvidence: options.verificationEvidence, - instructionCapsule: options.instructionCapsule, + workerReport, + completionCriteria: options.completionCriteria, + verificationEvidence: renderVerifyEvidence(options.verificationEvidence ?? []), + priorReviewContext: options.priorReviewContext, }); } -// Execution +// Pre-Check Execution (Fast Validation with Haiku) -async function runPreCheck(options: ReviewerOptions): Promise { +/** + * Run fast pre-check validation with Haiku model + * This is a cheap filter before expensive Sonnet review + * Expected to catch 30-40% of obvious issues, saving ~35% on review costs + */ +export async function runPreCheck(options: ReviewerOptions): Promise { const prompt = buildPreCheckPrompt(options); - const adapter = getAdapter(options.adapterName || 'cli'); - const result = await spawnCli(adapter, prompt, { - processContext: options.processContext, - timeoutMs: options.timeoutMs ?? 120_000, - maxTurns: options.maxTurns ?? 5, - model: options.model, - reasoningEffort: options.reasoningEffort, - }); + const cwd = expandPath(options.projectPath); + const adapter = getAdapter(options.adapterName); + + try { + // Use Haiku for fast validation + const raw = await spawnCli(adapter, { + prompt, + cwd, + timeoutMs: 30000, // 30 seconds max for pre-check + model: options.model, + maxTurns: options.maxTurns, + processContext: options.processContext, + onLog: options.onLog, + signal: options.signal, + readOnly: options.readOnly, + }); + + // DEBUG: Log raw Haiku output for troubleshooting + console.log('[Reviewer] Pre-check raw output (first 500 chars):', raw.stdout.slice(0, 500)); + + // Parse pre-check output + const lines = raw.stdout.split('\n'); + const passedLine = lines.find((l: string) => l.startsWith('PASSED:')); + const confidenceLine = lines.find((l: string) => l.startsWith('CONFIDENCE:')); + + const passed = passedLine?.includes('yes') ?? false; + const confidence = parseInt(confidenceLine?.split(':')[1]?.trim() || '1', 10); + + const issueStart = lines.findIndex((l: string) => l.startsWith('ISSUES:')); + const issues = issueStart >= 0 + ? lines.slice(issueStart + 1) + .filter((l: string) => l.trim().startsWith('-')) + .map((l: string) => l.replace(/^-\s*/, '').trim()) + .filter(Boolean) + : []; + + // DEBUG: Log parsing results + if (!passed && issues.length === 0) { + console.warn('[Reviewer] Pre-check failed but no issues found. Haiku may not be following format.'); + console.warn('[Reviewer] PASSED line:', passedLine || '(not found)'); + console.warn('[Reviewer] ISSUES section:', issueStart >= 0 ? 'found' : 'not found'); + + // Provide better default error message + if (!passedLine) { + issues.push('Haiku did not provide PASSED: line in expected format'); + } + if (issueStart < 0) { + issues.push('Haiku did not provide ISSUES: section in expected format'); + } + } - const output = result.output.trim().toLowerCase(); - const passed = output.includes('yes') || output.includes('pass') || output.includes('approve'); - return { passed, reason: result.output.trim() }; + return { + passed, + issues: issues.length > 0 ? issues : ['Pre-check failed with no specific issues (format parsing error)'], + confidence: Math.min(3, Math.max(0, confidence)), + }; + } catch (error) { + // Rate limit errors must propagate so the scheduler can pause β€” pre-check + // failure otherwise just passes through to the full review. + if (error instanceof RateLimitError) throw error; + // If pre-check fails, allow proceeding to full review + console.warn('[Reviewer] Pre-check failed, proceeding to full review:', error); + return { + passed: true, // Don't block on pre-check failure + issues: ['Pre-check timed out or failed'], + confidence: 0, + }; + } } +// Reviewer Execution + +/** + * Run Reviewer agent (full review with Sonnet) + */ export async function runReviewer(options: ReviewerOptions): Promise { const prompt = buildReviewerPrompt(options); - const adapter = getAdapter(options.adapterName || 'cli'); - const result = await spawnCli(adapter, prompt, { - processContext: options.processContext, - timeoutMs: options.timeoutMs ?? 120_000, - maxTurns: options.maxTurns ?? 10, - model: options.model, - reasoningEffort: options.reasoningEffort, - coordinationTools: options.coordinationTools, - }); - - const output = result.output.trim(); + const cwd = expandPath(options.projectPath); + const adapter = getAdapter(options.adapterName); - // Try to parse structured output try { - const parsed = JSON.parse(output); - if (parsed.decision && parsed.feedback !== undefined) { - return { - decision: parsed.decision, - feedback: parsed.feedback, - issues: parsed.issues || [], - suggestions: parsed.suggestions || [], - costInfo: result.costInfo, - }; + // Run CLI via adapter + const raw = await spawnCli(adapter, { + prompt, + cwd, + timeoutMs: options.timeoutMs ?? 300000, // 5 min default + model: options.model, + maxTurns: options.maxTurns, + processContext: options.processContext, + systemPrompt: getPrompts().systemPrompt + + reviewerIdentityHeader(options.coordinationContext?.actorName) + + reviewerCoordinationGuidance(options.coordinationContext, options.readOnly) + + (options.instructionCapsule?.text ?? ''), + reasoningEffort: options.reasoningEffort, + mcpTools: options.mcpTools, + onLog: options.onLog, + onToken: options.onToken, + signal: options.signal, + readOnly: options.readOnly, + coordinationContext: options.coordinationContext, + }); + + // Parse result via adapter + const parsedResult = adapter.parseReviewerOutput(raw); + // Backfill loop-measured usage for adapters that don't extract their own. (INT-2508) + if (raw.costInfo && !parsedResult.costInfo) { + parsedResult.costInfo = raw.costInfo; } - } catch { - // Not JSON, use raw output + return parsedResult; + } catch (error) { + // Rate limit errors must propagate so the scheduler can pause. + if (error instanceof RateLimitError) throw error; + // An infra failure (CLI exit, auth, spawn, timeout) means the REVIEWER never + // ran β€” it is NOT a quality verdict. Propagate so the pipeline classifies it + // as 'infra_error' instead of letting it masquerade as a 'reject' that + // increments the rejection-limit STUCK counter. (INT-2010) + if (isInfraError(error)) throw error; + // Whatever is left ran the reviewer but produced NO usable verdict β€” most often + // adapter.parseReviewerOutput throwing on malformed output. This is NOT a quality + // 'reject' (a 'reject' discards the worker's work AND counts toward the + // rejectionβ†’STUCK limit, turning a reviewer-side parse bug into a false STUCK), + // and NOT a 'revise' (which the CLI reads as an exit-0 success and which spends + // the worker's revision budget on a reviewer-side problem). Throw an infra-marked + // error β†’ the pipeline classifies infra_error (backoff retry, no STUCK) and the + // CLI exits non-zero. (INT-2521) + const msg = error instanceof Error ? error.message : String(error); + throw new Error(`reviewer-stage: produced no parseable verdict: ${msg}`, { cause: error }); } - - return { - decision: output.includes('approve') ? 'approved' : 'rejected', - feedback: output, - issues: [], - suggestions: [], - costInfo: result.costInfo, - }; } +// Formatting + /** - * Format review feedback as a Discord message. + * Format Reviewer result as a Discord message. * Enforces Discord message content limits to prevent payload rejection. */ export function formatReviewFeedback(result: ReviewResult): string { - const decisionEmoji = result.decision === 'approved' ? 'βœ…' : '❌'; + const decisionEmoji = { + approve: 'βœ…', + revise: 'πŸ”„', + reject: '❌', + }[result.decision]; + + const decisionText = { + approve: 'APPROVED', + revise: 'REVISION NEEDED', + reject: 'REJECTED', + }[result.decision]; + const lines: string[] = []; - lines.push(`${decisionEmoji} **Review Decision: ${result.decision}**`); + lines.push(`${decisionEmoji} ${t('agents.reviewer.report.decision', { text: decisionText })}`); lines.push(''); + lines.push(t('agents.reviewer.report.feedback', { text: result.feedback })); - // Feedback (bounded per Discord message limit) - if (result.feedback) { - lines.push(boundedMessageContent(result.feedback)); - } - - // Issues if (result.issues && result.issues.length > 0) { lines.push(''); lines.push(t('agents.reviewer.report.issues')); - for (const issue of result.issues.slice(0, 10)) { - lines.push(` ⚠️ ${issue}`); - } - if (result.issues.length > 10) { - lines.push(` … +${result.issues.length - 10} more`); + for (const issue of result.issues.slice(0, 5)) { + lines.push(` β€’ ${issue}`); } } - // Suggestions if (result.suggestions && result.suggestions.length > 0) { lines.push(''); lines.push(t('agents.reviewer.report.suggestions')); @@ -188,11 +417,7 @@ export function formatReviewFeedback(result: ReviewResult): string { } } - const full = lines.join('\n'); - // Ensure the entire message fits within Discord limits - return full.length > DISCORD_MESSAGE_CONTENT_LIMIT - ? full.slice(0, DISCORD_MESSAGE_CONTENT_LIMIT - 3) + '…' - : full; + return boundedMessageContent(lines.join('\n')); } /** @@ -205,4 +430,4 @@ export function buildRevisionPrompt(result: ReviewResult): string { issues: result.issues || [], suggestions: result.suggestions || [], }); -} \ No newline at end of file +} diff --git a/src/agents/skillDocumenter.ts b/src/agents/skillDocumenter.ts index 54a69ed1..343635e4 100644 --- a/src/agents/skillDocumenter.ts +++ b/src/agents/skillDocumenter.ts @@ -9,7 +9,7 @@ import { getAdapter, spawnCli } from '../adapters/index.js'; import { type CostInfo, extractCostFromStreamJson, formatCost } from '../support/costTracker.js'; import { expandPath } from '../core/config.js'; import { RateLimitError } from '../adapters/rateLimitError.js'; -import { boundedMessageContent, DISCORD_MESSAGE_CONTENT_LIMIT } from '../support/outputBudget.js'; +import { boundedMessageContent } from '../support/outputBudget.js'; // Types @@ -44,73 +44,142 @@ function buildSkillDocumenterPrompt(options: SkillDocumenterOptions): string { return `/documents -## Task - -${options.taskTitle} - -${options.taskDescription} - -## Worker Report +## Task Context +- **Task:** ${options.taskTitle} +- **Description:** ${options.taskDescription.slice(0, 200)}${options.taskDescription.length > 200 ? '...' : ''} +## Worker's Changes ${workerReport} -## Instructions +Update the project documentation to reflect the changes from the above task. +After the documentation update is complete, output the result in the following JSON format: -Review the worker's changes and update the project's documentation accordingly. - -1. Check if any documentation files need updating based on the changes made. -2. Update relevant documentation files. -3. If no documentation changes are needed, report that. +\`\`\`json +{ + "success": true, + "updatedFiles": ["CLAUDE.md", "docs/architecture.md"], + "summary": "Added new module description to architecture docs" +} +\`\`\` -## Output Format +When there is nothing to update: +\`\`\`json +{ + "success": true, + "updatedFiles": [], + "summary": "No documentation update needed (minor change)" +} +\`\`\` -Return a JSON object with the following structure: +On failure: \`\`\`json { - "success": true/false, - "updatedFiles": ["path/to/file1.md", ...], - "summary": "Brief summary of documentation changes" + "success": false, + "updatedFiles": [], + "summary": "Documentation update failed", + "error": "Detailed error message" } -\`\`\``; +\`\`\` +`; } -// Execution +// Skill Documenter Execution export async function runSkillDocumenter(options: SkillDocumenterOptions): Promise { const prompt = buildSkillDocumenterPrompt(options); - const adapter = getAdapter(options.adapterName || 'cli'); - const result = await spawnCli(adapter, prompt, { - timeoutMs: options.timeoutMs ?? 120_000, - maxTurns: options.maxTurns ?? 5, - model: options.model, - }); - - const output = result.output.trim(); - const parsed = parseSkillDocumenterOutput(output); + const cwd = expandPath(options.projectPath); + const adapter = getAdapter(options.adapterName); - return { - ...parsed, - costInfo: result.costInfo, - }; + try { + const raw = await spawnCli(adapter, { + prompt, + cwd, + timeoutMs: options.timeoutMs, + model: options.model, + maxTurns: options.maxTurns, + }); + return parseSkillDocumenterOutput(raw.stdout); + } catch (error) { + if (error instanceof RateLimitError) throw error; + return { + success: false, + updatedFiles: [], + summary: 'Skill Documenter execution failed', + error: error instanceof Error ? error.message : String(error), + }; + } } -// Parsing +// Output Parsing function parseSkillDocumenterOutput(output: string): SkillDocumenterResult { - // Try JSON extraction first - const jsonResult = extractResultJson(output); - if (jsonResult) return jsonResult; - - // Fallback to text extraction - return extractFromText(output); + try { + const costInfo = extractCostFromStreamJson(output); + if (costInfo) { + console.log(`[SkillDocumenter] Cost: ${formatCost(costInfo)}`); + } + + // Extract result entry from NDJSON + let resultText = ''; + for (const line of output.split('\n')) { + try { + const event = JSON.parse(line.trim()); + if (event.type === 'result' && event.result) { + resultText = event.result; + break; + } + if (event.type === 'item.completed' && event.item?.type === 'agent_message' && event.item.text) { + resultText = event.item.text; + } + } catch { /* skip non-JSON lines */ } + } + + if (!resultText) { + const result = extractFromText(output); + result.costInfo = costInfo; + return result; + } + + const result = extractResultJson(resultText) || extractFromText(resultText); + result.costInfo = costInfo; + return result; + } catch (error) { + console.error('[SkillDocumenter] Parse error:', error); + return extractFromText(output); + } } function extractResultJson(text: string): SkillDocumenterResult | null { - const jsonMatch = text.match(/\{[\s\S]*"success"[\s\S]*\}/); - if (!jsonMatch) return null; + const jsonMatch = text.match(/```json\s*([\s\S]*?)\s*```/); + if (!jsonMatch) { + const objMatch = text.match(/\{\s*"success"\s*:/); + if (!objMatch) return null; + + const startIdx = objMatch.index!; + let depth = 0; + let endIdx = startIdx; + + for (let i = startIdx; i < text.length; i++) { + if (text[i] === '{') depth++; + if (text[i] === '}') { + depth--; + if (depth === 0) { + endIdx = i + 1; + break; + } + } + } + + try { + const parsed = JSON.parse(text.slice(startIdx, endIdx)); + return normalizeResult(parsed); + } catch { + return null; + } + } try { - const parsed = JSON.parse(jsonMatch[0]); + const parsed = JSON.parse(jsonMatch[1]); return normalizeResult(parsed); } catch { return null; @@ -121,28 +190,52 @@ function normalizeResult(parsed: any): SkillDocumenterResult { return { success: Boolean(parsed.success), updatedFiles: Array.isArray(parsed.updatedFiles) ? parsed.updatedFiles : [], - summary: typeof parsed.summary === 'string' ? parsed.summary : '', - error: typeof parsed.error === 'string' ? parsed.error : undefined, + summary: parsed.summary || '(no summary)', + error: parsed.error, }; } function extractFromText(text: string): SkillDocumenterResult { + const hasError = /error|fail|exception/i.test(text); + const hasSuccess = /success|completed|updated|documented/i.test(text); + + const updatedFiles: string[] = []; + const filePatterns = [ + /(?:updated?|modified?|created?|wrote?):\s*(.+\.(?:md|rst|txt))/gi, + /(?:CLAUDE|AGENTS|README|docs?)\.md/gi, + ]; + + for (const pattern of filePatterns) { + const matches = text.matchAll(pattern); + for (const m of matches) { + const file = m[1] || m[0]; + if (!updatedFiles.includes(file)) { + updatedFiles.push(file); + } + } + } + return { - success: text.includes('success') || text.includes('updated'), - updatedFiles: extractSummary(text).split('\n').filter(l => l.includes('.md') || l.includes('.ts')), + success: !hasError || hasSuccess, + updatedFiles: updatedFiles.slice(0, 10), summary: extractSummary(text), - error: extractErrorMessage(text), + error: hasError ? extractErrorMessage(text) : undefined, }; } function extractSummary(text: string): string { - const lines = text.split('\n').filter(l => l.length > 0); - return lines.slice(0, 5).join('\n'); + const lines = text.split('\n').filter((l) => l.trim().length > 10); + if (lines.length === 0) return '(no summary)'; + const summary = lines[0].trim(); + return summary.length > 200 ? summary.slice(0, 200) + '...' : summary; } -function extractErrorMessage(text: string): string | undefined { - const errorMatch = text.match(/error:?\s*(.+)/i); - return errorMatch ? errorMatch[1] : undefined; +function extractErrorMessage(text: string): string { + const errorMatch = text.match(/(?:error|exception|failed?):\s*(.+)/i); + if (errorMatch) return errorMatch[1].slice(0, 200); + const lines = text.split('\n').filter((l) => /error|fail/i.test(l)); + if (lines.length > 0) return lines[0].slice(0, 200); + return 'Unknown error'; } // Formatting @@ -169,9 +262,5 @@ export function formatSkillDocReport(result: SkillDocumenterResult): string { lines.push(`**Error:** ${result.error}`); } - const full = lines.join('\n'); - // Ensure the entire message fits within Discord limits - return full.length > DISCORD_MESSAGE_CONTENT_LIMIT - ? full.slice(0, DISCORD_MESSAGE_CONTENT_LIMIT - 3) + '…' - : full; -} \ No newline at end of file + return boundedMessageContent(lines.join('\n')); +} diff --git a/src/agents/tester.ts b/src/agents/tester.ts index 9142e6b8..10f2c2d3 100644 --- a/src/agents/tester.ts +++ b/src/agents/tester.ts @@ -52,126 +52,306 @@ export interface TesterResult { * Build Tester prompt */ function buildTesterPrompt(options: TesterOptions): string { - const prompts = getPrompts(); - return prompts.buildTesterPrompt({ - taskTitle: options.taskTitle, - taskDescription: options.taskDescription, - workerResult: options.workerResult, - }); + const workerReport = ` +- **Success:** ${options.workerResult.success} +- **Summary:** ${options.workerResult.summary} +- **Files Changed:** ${options.workerResult.filesChanged.join(', ') || '(none)'} +- **Commands:** ${options.workerResult.commands.join(', ') || '(none)'} +`; + + return `# Tester Agent + +## Original Task +- **Title:** ${options.taskTitle} +- **Description:** ${options.taskDescription.slice(0, 200)}${options.taskDescription.length > 200 ? '...' : ''} + +## Worker's Changes +${workerReport} + +## Instructions +1. Run tests for the changed files +2. Verify that all existing tests pass +3. Suggest new tests if needed for new functionality +4. Report test coverage if available + +## Test Execution Steps +1. Check the project's test command (package.json, pytest.ini, etc.) +2. Run relevant test files +3. Analyze any failed tests +4. Determine if additional tests are needed + +## Output Format (IMPORTANT - must output in this format at the end) +After testing is complete, output the result in the following JSON format: + +\`\`\`json +{ + "success": true, + "testsPassed": 10, + "testsFailed": 0, + "coverage": 85.5, + "failedTests": [], + "suggestions": ["Additional test suggestions (if any)"] +} +\`\`\` + +On failure: +\`\`\`json +{ + "success": false, + "testsPassed": 8, + "testsFailed": 2, + "coverage": 75.0, + "failedTests": ["test_feature.py::test_case1", "test_feature.py::test_case2"], + "suggestions": ["Failure cause analysis", "Fix suggestions"], + "error": "Detailed error message" +} +\`\`\` +`; } -// Execution +// Tester Execution +/** + * Run Tester agent + */ export async function runTester(options: TesterOptions): Promise { const prompt = buildTesterPrompt(options); - const adapter = getAdapter(options.adapterName || 'cli'); - const result = await spawnCli(adapter, prompt, { - timeoutMs: options.timeoutMs ?? 120_000, - maxTurns: options.maxTurns ?? 5, - model: options.model, - }); - - const output = result.output.trim(); - const parsed = parseTesterOutput(output); + const cwd = expandPath(options.projectPath); + const adapter = getAdapter(options.adapterName); - return { - ...parsed, - costInfo: result.costInfo, - }; + try { + const raw = await spawnCli(adapter, { + prompt, + cwd, + timeoutMs: options.timeoutMs, + model: options.model, + maxTurns: options.maxTurns, + }); + + return parseTesterOutput(raw.stdout); + } catch (error) { + // Rate-limit AND infra failures (CLI exit, timeout, auth, spawn) mean the + // TESTER never ran β€” they are NOT "tests failed". Propagate so the pipeline + // classifies rate_limited / infra_error instead of feeding a bogus + // "fix the tests" self-repair loop that burns iterations β†’ false STUCK. + // worker.ts:337 / reviewer.ts:264 already do this; the tester was missing it. (INT-2521) + if (error instanceof RateLimitError) throw error; + if (isInfraError(error)) throw error; + return { + success: false, + testsPassed: 0, + testsFailed: 0, + output: '', + error: error instanceof Error ? error.message : String(error), + }; + } } -// Parsing - +/** + * Parse Tester output + */ export function parseTesterOutput(output: string): TesterResult { - // Try JSON extraction first - const jsonResult = extractResultJson(output); - if (jsonResult) return jsonResult; + try { + const costInfo = extractCostFromStreamJson(output); + if (costInfo) { + console.log(`[Tester] Cost: ${formatCost(costInfo)}`); + } + + // Extract result entry from NDJSON + let resultText = ''; + for (const line of output.split('\n')) { + try { + const event = JSON.parse(line.trim()); + if (event.type === 'result' && event.result) { + resultText = event.result; + break; + } + if (event.type === 'item.completed' && event.item?.type === 'agent_message' && event.item.text) { + resultText = event.item.text; + } + } catch { /* skip non-JSON lines */ } + } + + if (!resultText) { + const result = extractFromText(output); + result.costInfo = costInfo; + return result; + } - // Fallback to text extraction - return extractFromText(output); + // Extract JSON block from result + const result = extractResultJson(resultText) || extractFromText(resultText); + result.costInfo = costInfo; + return result; + } catch (error) { + console.error('[Tester] Parse error:', error); + return extractFromText(output); + } } -export function extractResultJson(text: string): TesterResult | null { - const jsonMatch = text.match(/\{[\s\S]*"success"[\s\S]*\}/); - if (!jsonMatch) return null; +/** + * Extract JSON block from result + */ +function extractResultJson(text: string): TesterResult | null { + // Find ```json ... ``` block + const jsonMatch = text.match(/```json\s*([\s\S]*?)\s*```/); + if (!jsonMatch) { + // Find plain JSON object + const objMatch = text.match(/\{\s*"success"\s*:/); + if (!objMatch) return null; + + const startIdx = objMatch.index!; + let depth = 0; + let endIdx = startIdx; + + for (let i = startIdx; i < text.length; i++) { + if (text[i] === '{') depth++; + if (text[i] === '}') { + depth--; + if (depth === 0) { + endIdx = i + 1; + break; + } + } + } + + try { + const parsed = JSON.parse(text.slice(startIdx, endIdx)); + return normalizeResult(parsed, text); + } catch { + return null; + } + } try { - const parsed = JSON.parse(jsonMatch[0]); + const parsed = JSON.parse(jsonMatch[1]); return normalizeResult(parsed, text); } catch { return null; } } +/** + * Normalize result + */ function normalizeResult(parsed: any, output: string): TesterResult { return { success: Boolean(parsed.success), testsPassed: typeof parsed.testsPassed === 'number' ? parsed.testsPassed : 0, testsFailed: typeof parsed.testsFailed === 'number' ? parsed.testsFailed : 0, coverage: typeof parsed.coverage === 'number' ? parsed.coverage : undefined, - output: typeof parsed.output === 'string' ? parsed.output : output, - failedTests: Array.isArray(parsed.failedTests) ? parsed.failedTests : [], - suggestions: Array.isArray(parsed.suggestions) ? parsed.suggestions : [], - error: typeof parsed.error === 'string' ? parsed.error : undefined, + output, + failedTests: Array.isArray(parsed.failedTests) ? parsed.failedTests : undefined, + suggestions: Array.isArray(parsed.suggestions) ? parsed.suggestions : undefined, + error: parsed.error, }; } +/** + * Extract result from text (when JSON parsing fails) + */ function extractFromText(text: string): TesterResult { + // Estimate success + const hasError = /error|fail|exception|cannot/i.test(text); + const hasSuccess = /pass|success|completed|all tests/i.test(text); + + // Extract test statistics + let testsPassed = 0; + let testsFailed = 0; + + // Common test result patterns + const passMatch = text.match(/(\d+)\s*(?:passed|pass|passing)/i); + const failMatch = text.match(/(\d+)\s*(?:failed|fail|failing)/i); + + if (passMatch) testsPassed = parseInt(passMatch[1], 10); + if (failMatch) testsFailed = parseInt(failMatch[1], 10); + + // Extract coverage + let coverage: number | undefined; + const coverageMatch = text.match(/(?:coverage|cov)[:\s]*(\d+(?:\.\d+)?)\s*%/i); + if (coverageMatch) { + coverage = parseFloat(coverageMatch[1]); + } + + // Extract failed tests + const failedTests: string[] = []; + const failedPattern = /(?:FAILED|FAIL)\s+([^\s]+(?:::[\w_]+)?)/gi; + const failedMatches = text.matchAll(failedPattern); + for (const m of failedMatches) { + if (!failedTests.includes(m[1])) { + failedTests.push(m[1]); + } + } + + // A tester that produced NO output verified nothing β€” the "no error keyword β‡’ + // success" default would fake a PASS on an empty/degenerate run and let + // unverified code through the blocking test gate. Only genuinely empty output is + // flagged, so a short-but-real run ("collected 0 items") is unaffected. (INT-2521) + const noOutput = text.trim().length === 0; return { - success: !text.includes('FAIL') && !text.includes('failed'), - testsPassed: 0, - testsFailed: 0, + success: !noOutput && (!hasError || (hasSuccess && testsFailed === 0)), + testsPassed, + testsFailed, + coverage, output: text, - failedTests: [], - suggestions: [], - error: extractErrorMessage(text), + failedTests: failedTests.length > 0 ? failedTests : undefined, + error: hasError ? extractErrorMessage(text) : (noOutput ? 'Tester produced no output β€” result unverified' : undefined), }; } -function extractErrorMessage(text: string): string | undefined { - const errorMatch = text.match(/error:?\s*(.+)/i); - return errorMatch ? errorMatch[1] : undefined; +/** + * Extract error message + */ +function extractErrorMessage(text: string): string { + const errorMatch = text.match(/(?:error|exception|failed?):\s*(.+)/i); + if (errorMatch) { + return errorMatch[1].slice(0, 200); + } + + const lines = text.split('\n').filter((l) => /error|fail/i.test(l)); + if (lines.length > 0) { + return lines[0].slice(0, 200); + } + + return 'Unknown error'; } // Formatting /** - * Format test report as a Discord message + * Format Tester result as Discord message */ export function formatTestReport(result: TesterResult): string { const statusEmoji = result.success ? 'βœ…' : '❌'; const lines: string[] = []; - lines.push(`${statusEmoji} **Test Results: ${result.success ? 'Passed' : 'Failed'}**`); + lines.push(`${statusEmoji} **Tester Result: ${result.success ? 'PASS' : 'FAIL'}**`); lines.push(''); - lines.push(`**Tests Passed:** ${result.testsPassed}`); - lines.push(`**Tests Failed:** ${result.testsFailed}`); + lines.push(`**Passed:** ${result.testsPassed} | **Failed:** ${result.testsFailed}`); - if (result.coverage != null) { - lines.push(`**Coverage:** ${(result.coverage * 100).toFixed(1)}%`); + if (result.coverage !== undefined) { + lines.push(`**Coverage:** ${result.coverage.toFixed(1)}%`); } if (result.failedTests && result.failedTests.length > 0) { lines.push(''); lines.push('**Failed Tests:**'); - for (const test of result.failedTests.slice(0, PROMPT_FAILED_TESTS_LIMIT)) { - lines.push(` ❌ ${test}`); + for (const test of result.failedTests.slice(0, 5)) { + lines.push(` β€’ \`${test}\``); } - if (result.failedTests.length > PROMPT_FAILED_TESTS_LIMIT) { - lines.push(` … +${result.failedTests.length - PROMPT_FAILED_TESTS_LIMIT} more`); + if (result.failedTests.length > 5) { + lines.push(` β€’ ... +${result.failedTests.length - 5} more`); } } if (result.suggestions && result.suggestions.length > 0) { lines.push(''); lines.push('**Suggestions:**'); - for (const suggestion of result.suggestions.slice(0, PROMPT_SUGGESTIONS_LIMIT)) { + for (const suggestion of result.suggestions.slice(0, 3)) { lines.push(` β€’ ${suggestion}`); } } if (result.error) { - lines.push(''); lines.push(`**Error:** ${result.error}`); } @@ -179,39 +359,34 @@ export function formatTestReport(result: TesterResult): string { } /** - * Build test fix prompt for the worker. + * Convert Tester result to Worker feedback. * Enforces aggregate bounds on feedback to prevent prompt bloat. */ export function buildTestFixPrompt(result: TesterResult): string { const lines: string[] = []; - lines.push(`The tests ${result.success ? 'passed' : 'failed'}.`); - lines.push(`Tests passed: ${result.testsPassed}, Tests failed: ${result.testsFailed}`); - - if (result.coverage != null) { - lines.push(`Coverage: ${(result.coverage * 100).toFixed(1)}%`); - } + lines.push('## Test Failures'); + lines.push(''); + lines.push(`**Passed:** ${result.testsPassed} | **Failed:** ${result.testsFailed}`); - // Bound failed tests list to prevent prompt bloat if (result.failedTests && result.failedTests.length > 0) { lines.push(''); lines.push('### Failed Tests:'); const shown = result.failedTests.slice(0, PROMPT_FAILED_TESTS_LIMIT); for (let i = 0; i < shown.length; i++) { - lines.push(`${i + 1}. \`${shown[i]}\``); + lines.push(`${i + 1}. \`${truncate(shown[i], 200)}\``); } if (result.failedTests.length > PROMPT_FAILED_TESTS_LIMIT) { lines.push(`… +${result.failedTests.length - PROMPT_FAILED_TESTS_LIMIT} more`); } } - // Bound suggestions list to prevent prompt bloat if (result.suggestions && result.suggestions.length > 0) { lines.push(''); lines.push('### Fix Suggestions:'); const shown = result.suggestions.slice(0, PROMPT_SUGGESTIONS_LIMIT); for (let i = 0; i < shown.length; i++) { - lines.push(`${i + 1}. ${shown[i]}`); + lines.push(`${i + 1}. ${truncate(shown[i], 300)}`); } if (result.suggestions.length > PROMPT_SUGGESTIONS_LIMIT) { lines.push(`… +${result.suggestions.length - PROMPT_SUGGESTIONS_LIMIT} more`); @@ -222,11 +397,7 @@ export function buildTestFixPrompt(result: TesterResult): string { lines.push('Fix the above test failures.'); const full = lines.join('\n'); - // Enforce aggregate prompt budget return full.length > PROMPT_FEEDBACK_LIMIT - ? full.slice(0, PROMPT_FEEDBACK_LIMIT - 3) + '…' + ? `${full.slice(0, PROMPT_FEEDBACK_LIMIT - 1)}…` : full; } - -// Re-export getPrompts for tester -import { getPrompts } from '../locale/index.js'; \ No newline at end of file diff --git a/src/automation/workerAuditLog.ts b/src/automation/workerAuditLog.ts index 3749ea7c..162a90e1 100644 --- a/src/automation/workerAuditLog.ts +++ b/src/automation/workerAuditLog.ts @@ -14,8 +14,8 @@ import { AUDIT_COMMANDS_MAX, AUDIT_SUMMARY_CAP, AUDIT_GOAL_CAP, - capArray, - codeList, + AUDIT_ENTRY_CAP, + truncate, } from '../support/outputBudget.js'; /** Caps so a chatty agent can't post a multi-MB comment. */ @@ -26,12 +26,12 @@ const GOAL_CAP = AUDIT_GOAL_CAP; function cap(s: string | undefined, n: number): string { if (!s) return ''; - const trimmed = s.trim(); - return trimmed.length > n ? `${trimmed.slice(0, n - 1)}…` : trimmed; + return truncate(s.trim(), n); } function inlineCode(s: string): string { - return `\`${s.replaceAll('`', '\\`')}\``; + // Cap individual entry length before wrapping so a single path/command can't blow the comment. + return `\`${truncate(s, AUDIT_ENTRY_CAP).replaceAll('`', '\\`')}\``; } /** Render a list as inline code, capped, with an "+N more" suffix when truncated. */ @@ -43,74 +43,69 @@ function codeList(items: string[] | undefined, max: number): string { } export interface WorkerStartInfo { + /** 1-based iteration/attempt number. */ + attempt: number; + maxAttempts?: number; taskTitle: string; - taskDescription: string; - projectPath: string; - /** The goal the worker was asked to achieve (from the task). */ - goal?: string; + /** Prompt summary β€” task description or draft intent summary. */ + taskGoal?: string; + /** Files the worker is expected to touch (from draft analysis / impact). */ + targetFiles?: string[]; + /** Resolved model for this worker run. */ + model?: string; + /** Max agentic turns (proxy for effort budget). */ + maxTurns?: number; + /** True when this run follows reviewer/guard feedback (a revision). */ + isRevision?: boolean; } -/** - * Build the "worker started" audit comment. - */ +/** Comment body posted when a worker run starts (the instruction). */ export function buildWorkerStartComment(info: WorkerStartInfo): string { - const sections: CommentSection[] = []; - - if (info.goal) { - sections.push({ label: 'Goal', body: cap(info.goal, GOAL_CAP) }); + const attemptLabel = info.maxAttempts + ? `attempt #${info.attempt}/${info.maxAttempts}` + : `attempt #${info.attempt}`; + const heading = info.isRevision ? 'Worker revision' : 'Worker instruction'; + + const sections: CommentSection[] = [{ label: 'Task', body: cap(info.taskTitle, 200) }]; + if (info.taskGoal) sections.push({ label: 'Goal', body: cap(info.taskGoal, GOAL_CAP) }); + if (info.targetFiles && info.targetFiles.length > 0) { + sections.push({ label: 'Target files', body: codeList(info.targetFiles, MAX_FILES) }); } return formatAutomationComment({ - heading: 'Worker started', - summary: cap(info.taskTitle, SUMMARY_CAP), + heading: `${heading} (${attemptLabel})`, sections, - meta: { - Project: info.projectPath, - }, + meta: { Model: info.model, 'Max turns': info.maxTurns }, attribution: 'Worker audit log', }); } export interface WorkerCompleteInfo { + attempt: number; + maxAttempts?: number; result: WorkerResult; - /** Seconds the worker ran for. */ + /** Worker run duration in seconds. */ durationSec?: number; - /** Which attempt number this was (1-based). */ - attempt?: number; - /** Max attempts allowed. */ - maxAttempts?: number; } -/** - * Build the "worker completed" audit comment. - * Caps individual file and command entries before rendering to prevent oversized comments. - */ +/** Comment body posted when a worker run completes (the actions taken). */ export function buildWorkerCompleteComment(info: WorkerCompleteInfo): string { const { result } = info; - const verdict = result.success ? 'βœ… Complete' : '❌ Failed'; - const attemptLabel = info.attempt != null && info.maxAttempts != null - ? `(attempt ${info.attempt}/${info.maxAttempts})` - : ''; - - const sections: CommentSection[] = []; - - // Files changed β€” capped to prevent oversized comments - if (result.filesChanged && result.filesChanged.length > 0) { - const { shown, omitted } = capArray(result.filesChanged, MAX_FILES); - const filesStr = shown.map(inlineCode).join(', '); - const body = omitted > 0 ? `${filesStr} _+${omitted} more_` : filesStr; - sections.push({ label: 'Files changed', body }); - } - - // Commands run β€” capped to prevent oversized comments - if (result.commands && result.commands.length > 0) { - const { shown, omitted } = capArray(result.commands, MAX_COMMANDS); - const cmdsStr = shown.map(inlineCode).join(', '); - const body = omitted > 0 ? `${cmdsStr} _+${omitted} more_` : cmdsStr; - sections.push({ label: 'Commands', body }); + const attemptLabel = info.maxAttempts + ? `attempt #${info.attempt}/${info.maxAttempts}` + : `attempt #${info.attempt}`; + const verdict = result.haltReason ? 'Halted' : result.success ? 'Done' : 'Failed'; + + const files = result.filesChanged ?? []; + const commands = result.commands ?? []; + + const sections: CommentSection[] = [ + { label: `Files changed (${files.length})`, body: codeList(files, MAX_FILES) }, + ]; + if (commands.length > 0) { + sections.push({ label: `Commands (${commands.length})`, body: codeList(commands, MAX_COMMANDS) }); } - - // Error β€” capped + if (result.haltReason) sections.push({ label: 'Halt reason', body: cap(result.haltReason, GOAL_CAP) }); if (result.error) sections.push({ label: 'Error', body: cap(result.error, GOAL_CAP) }); const duration = info.durationSec != null @@ -127,4 +122,4 @@ export function buildWorkerCompleteComment(info: WorkerCompleteInfo): string { }, attribution: 'Worker audit log', }); -} \ No newline at end of file +} diff --git a/src/discord/discordHandlers.ts b/src/discord/discordHandlers.ts index fa2a8d97..eddd7627 100644 --- a/src/discord/discordHandlers.ts +++ b/src/discord/discordHandlers.ts @@ -26,13 +26,22 @@ import { formatTimeAgo, } from './discordCore.js'; import { t, getDateLocale } from '../locale/index.js'; +import { + boundedFieldValue, + boundedDescription, + genericUserError, + paginateEmbedFields, + truncate, + DISCORD_EMBED_FIELD_VALUE_LIMIT, + DISCORD_EMBED_TITLE_LIMIT, +} from '../support/outputBudget.js'; /** * Helper: Reply with Embed for consistent Discord UI */ async function replyWithEmbed(msg: Message, content: string, color: number = 0x00ff41): Promise { const embed = new EmbedBuilder() - .setDescription(content) + .setDescription(boundedDescription(content)) .setColor(color) .setTimestamp(); await msg.reply({ embeds: [embed] }); @@ -169,58 +178,53 @@ export async function handleIssues(msg: Message, sessionName?: string): Promise< 'Backlog': 0x95a5a6, }; - // Pagination (max 10 per embed) - const ITEMS_PER_PAGE = 10; - const totalPages = Math.ceil(issues.length / ITEMS_PER_PAGE); + // Build all fields, then paginate by Discord embed field + aggregate budgets + const allFields = issues.map((issue) => { + const priority = priorityEmoji[issue.priority as keyof typeof priorityEmoji] ?? 'βšͺ'; + const stateEmoji = { + 'Todo': 'πŸ“', + 'In Progress': 'βš™οΈ', + 'In Review': 'πŸ‘€', + 'Done': 'βœ…', + 'Backlog': 'πŸ“¦', + }[issue.state] ?? 'πŸ“‹'; + + let value = `${priority} **${issue.identifier}**: ${issue.title}\n`; + value += `${stateEmoji} ${issue.state}`; + + if (issue.project) { + value += ` Β· ${issue.project.name}`; + } - const embeds: EmbedBuilder[] = []; + if (issue.labels && issue.labels.length > 0) { + value += `\n🏷️ ${issue.labels.join(', ')}`; + } - for (let page = 0; page < totalPages; page++) { - const startIdx = page * ITEMS_PER_PAGE; - const endIdx = Math.min(startIdx + ITEMS_PER_PAGE, issues.length); - const pageIssues = issues.slice(startIdx, endIdx); + return { + name: `\u200b`, + value: boundedFieldValue(value, DISCORD_EMBED_FIELD_VALUE_LIMIT), + inline: false, + }; + }); + + const fieldPages = paginateEmbedFields(allFields); + const embeds: EmbedBuilder[] = []; + for (let page = 0; page < fieldPages.length; page++) { + const pageFields = fieldPages[page]; const embed = new EmbedBuilder() .setTitle(sessionName ? t('discord.issues.sessionIssues', { session: sessionName }) : t('discord.issues.myIssues') ) - .setColor(stateColor[pageIssues[0]?.state as keyof typeof stateColor] ?? 0x3498db) + .setColor(stateColor[issues[0]?.state as keyof typeof stateColor] ?? 0x3498db) .setTimestamp(); - if (totalPages > 1) { - embed.setFooter({ text: t('discord.issues.page', { current: page + 1, total: totalPages }) }); + if (fieldPages.length > 1) { + embed.setFooter({ text: t('discord.issues.page', { current: page + 1, total: fieldPages.length }) }); } - const fields = pageIssues.map((issue) => { - const priority = priorityEmoji[issue.priority as keyof typeof priorityEmoji] ?? 'βšͺ'; - const stateEmoji = { - 'Todo': 'πŸ“', - 'In Progress': 'βš™οΈ', - 'In Review': 'πŸ‘€', - 'Done': 'βœ…', - 'Backlog': 'πŸ“¦', - }[issue.state] ?? 'πŸ“‹'; - - let value = `${priority} **${issue.identifier}**: ${issue.title}\n`; - value += `${stateEmoji} ${issue.state}`; - - if (issue.project) { - value += ` Β· ${issue.project.name}`; - } - - if (issue.labels && issue.labels.length > 0) { - value += `\n🏷️ ${issue.labels.join(', ')}`; - } - - return { - name: `\u200b`, - value, - inline: false, - }; - }); - - embed.addFields(...fields); + embed.addFields(...pageFields); embeds.push(embed); } @@ -240,8 +244,7 @@ export async function handleIssues(msg: Message, sessionName?: string): Promise< } } } catch (error) { - const errorMsg = error instanceof Error ? error.message : String(error); - await replyWithEmbed(msg, t('discord.issues.fetchError', { error: errorMsg }), 0xff0000); + await replyWithEmbed(msg, t('discord.issues.fetchError', { error: genericUserError(error) }), 0xff0000); } } @@ -281,18 +284,15 @@ export async function handleIssue(msg: Message, issueId: string): Promise }; const embed = new EmbedBuilder() - .setTitle(`${issue.identifier}: ${issue.title}`) + .setTitle(truncate(`${issue.identifier}: ${issue.title}`, DISCORD_EMBED_TITLE_LIMIT)) .setColor(stateColor[issue.state as keyof typeof stateColor] ?? 0x3498db) .setTimestamp(); // Description if (issue.description) { - const desc = issue.description.length > 1024 - ? issue.description.slice(0, 1021) + '...' - : issue.description; embed.addFields({ name: 'πŸ“ Description', - value: desc, + value: boundedFieldValue(issue.description), inline: false, }); } @@ -319,7 +319,7 @@ export async function handleIssue(msg: Message, issueId: string): Promise embed.addFields({ name: 'πŸ“Š Details', - value: infoValue, + value: boundedFieldValue(infoValue), inline: false, }); @@ -339,7 +339,7 @@ export async function handleIssue(msg: Message, issueId: string): Promise embed.addFields({ name: `πŸ’¬ ${t('discord.issues.commentsCount', { count: issue.comments.length })}`, - value: commentValue, + value: boundedFieldValue(commentValue), inline: false, }); } else { @@ -352,8 +352,7 @@ export async function handleIssue(msg: Message, issueId: string): Promise await msg.reply({ embeds: [embed] }); } catch (error) { - const errorMsg = error instanceof Error ? error.message : String(error); - await replyWithEmbed(msg, t('discord.issue.fetchError', { error: errorMsg }), 0xff0000); + await replyWithEmbed(msg, t('discord.issue.fetchError', { error: genericUserError(error) }), 0xff0000); } } @@ -720,7 +719,7 @@ export async function handleSchedule(msg: Message, args: string[]): Promise { await msg.reply(`βœ… ${t('discord.codex.saveSuccess', { path: summaryPath })}`); } catch (err) { - await msg.reply(`❌ ${t('discord.codex.saveFailed', { error: err instanceof Error ? err.message : String(err) })}`); + await msg.reply(`❌ ${t('discord.codex.saveFailed', { error: genericUserError(err) })}`); } return; } @@ -924,7 +923,7 @@ export async function handleAuto(msg: Message, args: string[]): Promise { : `βœ… ${t('discord.auto.startedSolo')}`; await msg.reply(startMsg); } catch (err) { - await msg.reply(`❌ ${t('discord.errors.startFailed', { error: err instanceof Error ? err.message : String(err) })}`); + await msg.reply(`❌ ${t('discord.errors.startFailed', { error: genericUserError(err) })}`); } return; } diff --git a/src/runners/cliRunner.ts b/src/runners/cliRunner.ts index 0f153366..9c5bbfe3 100644 --- a/src/runners/cliRunner.ts +++ b/src/runners/cliRunner.ts @@ -16,6 +16,14 @@ import { startProgressHeartbeat, type ReviewProgress } from '../cli/reviewProgre import { status } from '../support/colors.js'; import { sanitizeTerminalText } from '../tui/sanitize.js'; import { safeConsole as console } from '../support/safeLog.js'; +import { + sanitizeException, + truncate, + flattenToSingleLine, + CLI_FEEDBACK_LINES, + CLI_STDERR_LINE_LIMIT, + PROMPT_FEEDBACK_LIMIT, +} from '../support/outputBudget.js'; // Types @@ -223,7 +231,7 @@ export async function runCli(options: CliRunOptions): Promise { result = await pipeline.run(task, projectPath); } catch (error) { stopHeartbeat(); - console.error('\n Pipeline execution failed:', error instanceof Error ? error.message : error); + console.error('\n Pipeline execution failed:', sanitizeException(error)); process.exitCode = 1; return; } @@ -269,18 +277,22 @@ function printResult(result: PipelineResult): void { console.log(' ======================================'); - // Summary + // Summary β€” sanitize + bound untrusted pipeline content if (result.workerResult?.summary) { - console.log(` Summary: ${sanitizeTerminalText(result.workerResult.summary)}`); + const summary = truncate( + flattenToSingleLine(sanitizeTerminalText(result.workerResult.summary)), + CLI_STDERR_LINE_LIMIT, + ); + console.log(` Summary: ${summary}`); } // Files changed if (result.workerResult?.filesChanged && result.workerResult.filesChanged.length > 0) { const files = result.workerResult.filesChanged; if (files.length <= 5) { - console.log(` Files: ${files.map(sanitizeTerminalText).join(', ')}`); + console.log(` Files: ${files.map((f) => sanitizeTerminalText(f)).join(', ')}`); } else { - console.log(` Files: ${files.slice(0, 5).join(', ')} +${files.length - 5} more`); + console.log(` Files: ${files.slice(0, 5).map((f) => sanitizeTerminalText(f)).join(', ')} +${files.length - 5} more`); } } @@ -292,13 +304,14 @@ function printResult(result: PipelineResult): void { parts.push(`Duration: ${formatDuration(result.totalDuration)}`); console.log(` ${parts.join(' | ')}`); - // Reviewer feedback on failure + // Reviewer feedback on failure β€” per-field line + aggregate budget if (!result.success && result.reviewResult?.feedback) { console.log(''); console.log(' Feedback:'); - const lines = result.reviewResult.feedback.split('\n').slice(0, 5); + const bounded = truncate(sanitizeTerminalText(result.reviewResult.feedback), PROMPT_FEEDBACK_LIMIT); + const lines = bounded.split('\n').slice(0, CLI_FEEDBACK_LINES); for (const line of lines) { - console.log(` ${line}`); + console.log(` ${truncate(flattenToSingleLine(line), CLI_STDERR_LINE_LIMIT)}`); } } diff --git a/src/support/outputBudget.test.ts b/src/support/outputBudget.test.ts new file mode 100644 index 00000000..d13338d5 --- /dev/null +++ b/src/support/outputBudget.test.ts @@ -0,0 +1,91 @@ +import { describe, it, expect } from 'vitest'; +import { + truncate, + truncateWithSuffix, + capArray, + boundedFieldValue, + boundedMessageContent, + flattenToSingleLine, + sanitizeException, + genericUserError, + paginateEmbedFields, + DISCORD_EMBED_FIELD_VALUE_LIMIT, + DISCORD_MESSAGE_CONTENT_LIMIT, + DISCORD_EMBED_AGGREGATE_VALUE_LIMIT, + LINEAR_DESCRIPTION_LIMIT, + boundedLinearText, +} from './outputBudget.js'; + +describe('outputBudget', () => { + it('truncates with ellipsis without exceeding the limit', () => { + expect(truncate('abcdef', 4)).toBe('abc…'); + expect(truncate('abc', 10)).toBe('abc'); + }); + + it('truncateWithSuffix keeps the final length at the limit', () => { + const out = truncateWithSuffix('x'.repeat(100), 40); + expect(out.length).toBe(40); + expect(out.endsWith('(truncated)')).toBe(true); + }); + + it('truncateWithSuffix falls back when suffix is longer than the limit', () => { + const out = truncateWithSuffix('abcdefghij', 4, '… (truncated)'); + expect(out.length).toBe(4); + expect(out).toBe('abc…'); + }); + + it('caps arrays and reports omitted count', () => { + expect(capArray([1, 2, 3, 4], 2)).toEqual({ shown: [1, 2], omitted: 2 }); + expect(capArray([1], 5)).toEqual({ shown: [1], omitted: 0 }); + }); + + it('bounds Discord field values to the per-field limit', () => { + const value = boundedFieldValue('y'.repeat(DISCORD_EMBED_FIELD_VALUE_LIMIT + 500)); + expect(value.length).toBeLessThanOrEqual(DISCORD_EMBED_FIELD_VALUE_LIMIT); + }); + + it('bounds Discord message content', () => { + const value = boundedMessageContent('z'.repeat(DISCORD_MESSAGE_CONTENT_LIMIT + 200)); + expect(value.length).toBeLessThanOrEqual(DISCORD_MESSAGE_CONTENT_LIMIT); + }); + + it('flattens multiline text to a single line', () => { + expect(flattenToSingleLine('a\nb\r\nc')).toBe('a b c'); + }); + + it('sanitizes exceptions to a single bounded line', () => { + const err = new Error('boom\n at foo.ts:1\n at bar.ts:2'); + expect(sanitizeException(err)).toBe('boom'); + expect(sanitizeException(err).length).toBeLessThanOrEqual(200); + }); + + it('generic user errors never include raw exception text', () => { + const msg = genericUserError(new Error('secret stack /tmp/evil.key')); + expect(msg).not.toContain('secret'); + expect(msg).not.toContain('evil'); + expect(msg.length).toBeLessThan(80); + }); + + it('paginates embed fields by aggregate budget', () => { + const fields = Array.from({ length: 20 }, (_, i) => ({ + name: `\u200b`, + value: 'v'.repeat(800), + inline: false, + })); + const pages = paginateEmbedFields(fields); + expect(pages.length).toBeGreaterThan(1); + for (const page of pages) { + const aggregate = page.reduce((sum, f) => sum + f.value.length, 0); + expect(aggregate).toBeLessThanOrEqual(DISCORD_EMBED_AGGREGATE_VALUE_LIMIT); + expect(page.length).toBeLessThanOrEqual(25); + for (const f of page) { + expect(f.value.length).toBeLessThanOrEqual(DISCORD_EMBED_FIELD_VALUE_LIMIT); + } + } + }); + + it('bounds Linear description payloads', () => { + const out = boundedLinearText('L'.repeat(LINEAR_DESCRIPTION_LIMIT + 100)); + expect(out.length).toBeLessThanOrEqual(LINEAR_DESCRIPTION_LIMIT); + }); +}); diff --git a/src/support/outputBudget.ts b/src/support/outputBudget.ts index 83bfa652..d824bb44 100644 --- a/src/support/outputBudget.ts +++ b/src/support/outputBudget.ts @@ -32,6 +32,8 @@ export const AUDIT_FILES_MAX = 20; export const AUDIT_COMMANDS_MAX = 12; export const AUDIT_SUMMARY_CAP = 600; export const AUDIT_GOAL_CAP = 400; +/** Cap length of a single file path or command entry before rendering. */ +export const AUDIT_ENTRY_CAP = 200; // ── TUI display budgets ── export const TUI_LOG_LINE_LIMIT = 200; @@ -55,7 +57,9 @@ export function truncate(value: string, limit: number): string { /** Truncate a string to `limit` chars, appending a suffix when clipped. */ export function truncateWithSuffix(value: string, limit: number, suffix = '\n… (truncated)'): string { + if (limit <= 0) return ''; if (value.length <= limit) return value; + if (suffix.length >= limit) return truncate(value, limit); return `${value.slice(0, limit - suffix.length)}${suffix}`; } @@ -103,11 +107,55 @@ export function boundedLinearTitle(value: string): string { return truncateWithSuffix(value, LINEAR_TITLE_LIMIT); } -/** Sanitize and bound exception text for user-facing output. */ +/** Sanitize and bound exception text for operator-facing stderr/logs (not end-user Discord). */ export function sanitizeException(error: unknown): string { if (error == null) return 'An unknown error occurred.'; const msg = error instanceof Error ? error.message : String(error); - // Strip any content that looks like a stack trace or internal path + // Strip stack traces / multiline dumps; keep a single bounded line. const cleaned = msg.split('\n')[0].trim(); return truncate(cleaned || 'An error occurred.', 200); +} + +/** Generic, bounded user-visible failure β€” never includes raw exception content. */ +export function genericUserError(_error?: unknown): string { + return 'Something went wrong. Please try again.'; +} + +/** + * Pack Discord embed fields while respecting per-field and aggregate value budgets. + * Returns pages of fields suitable for one embed each. + */ +export function paginateEmbedFields( + fields: Array<{ name: string; value: string; inline?: boolean }>, + options?: { + fieldValueLimit?: number; + aggregateLimit?: number; + maxFields?: number; + }, +): Array> { + const fieldValueLimit = options?.fieldValueLimit ?? DISCORD_EMBED_FIELD_VALUE_LIMIT; + const aggregateLimit = options?.aggregateLimit ?? DISCORD_EMBED_AGGREGATE_VALUE_LIMIT; + const maxFields = options?.maxFields ?? DISCORD_EMBED_FIELDS_PER_EMBED; + + const pages: Array> = []; + let current: Array<{ name: string; value: string; inline?: boolean }> = []; + let aggregate = 0; + + for (const field of fields) { + const value = boundedFieldValue(field.value, fieldValueLimit); + const next = { name: truncate(field.name, DISCORD_EMBED_FIELD_NAME_LIMIT), value, inline: field.inline }; + const fits = + current.length < maxFields && + aggregate + value.length <= aggregateLimit; + if (!fits && current.length > 0) { + pages.push(current); + current = []; + aggregate = 0; + } + // A single field larger than the remaining budget still goes on its own page (already bounded). + current.push(next); + aggregate += value.length; + } + if (current.length > 0) pages.push(current); + return pages; } \ No newline at end of file diff --git a/src/support/workflowLinear.ts b/src/support/workflowLinear.ts index c070dbc9..e003bff9 100644 --- a/src/support/workflowLinear.ts +++ b/src/support/workflowLinear.ts @@ -10,10 +10,16 @@ import { StepResult, topologicalSort, } from '../orchestration/workflow.js'; +import { + boundedLinearText, + LINEAR_DESCRIPTION_LIMIT, + LINEAR_COMMENT_LIMIT, +} from './outputBudget.js'; -const LINEAR_BLOCK_LIMIT = 3000; +const LINEAR_BLOCK_LIMIT = LINEAR_COMMENT_LIMIT; const LINEAR_INLINE_LIMIT = 500; +/** Linear comment blocks: keep the historical ASCII truncation marker for callers/tests. */ function truncateForLinear(value: string, limit: number): string { return value.length > limit ? `${value.slice(0, limit)}\n... (truncated)` : value; } @@ -129,7 +135,7 @@ function buildWorkflowDescription(workflow: WorkflowConfig): string { parts.push('---'); parts.push('_Managed by OpenSwarm Workflow Engine_'); - return parts.join('\n'); + return boundedLinearText(parts.join('\n'), LINEAR_DESCRIPTION_LIMIT); } /** @@ -166,7 +172,7 @@ function buildStepDescription(step: WorkflowStep, workflow: WorkflowConfig): str parts.push('---'); parts.push(`_Part of workflow: ${workflow.name}_`); - return parts.join('\n'); + return boundedLinearText(parts.join('\n'), LINEAR_DESCRIPTION_LIMIT); } /** @@ -214,7 +220,7 @@ export function stepResultToComment(result: StepResult): string { parts.push(result.changedFiles.map(f => `- \`${f}\``).join('\n')); } - return parts.join('\n'); + return boundedLinearText(parts.join('\n'), LINEAR_COMMENT_LIMIT); } /** @@ -291,7 +297,7 @@ export function createExecutionSummary(execution: WorkflowExecution): { } } - return { body: parts.join('\n'), health }; + return { body: boundedLinearText(parts.join('\n'), LINEAR_COMMENT_LIMIT), health }; } // Linear MCP Command Templates diff --git a/src/tui/components/AuditBoard.test.tsx b/src/tui/components/AuditBoard.test.tsx index bde075bd..39d2a42f 100644 --- a/src/tui/components/AuditBoard.test.tsx +++ b/src/tui/components/AuditBoard.test.tsx @@ -75,6 +75,24 @@ describe('AuditBoard (INT-2006)', () => { expect(f).toContain('codex timeout after 300000ms'); }); + it('flattens multiline progress logs before truncating', async () => { + const events = new EventEmitter(); + const r = render(); + await act(tick); + await act(async () => { + events.emit('progress', { type: 'start', label: 'src/a', done: 0, total: 2 }); + events.emit('progress', { + type: 'log', + label: 'src/a', + line: 'first line\nsecond line\nthird line that is quite long', + }); + await tick(); + }); + const f = r.lastFrame()!; + expect(f).toContain('first line second line'); + expect(f).not.toMatch(/first line\nsecond line/); + }); + it('renders fix-pass progress with edited file tally', async () => { const events = new EventEmitter(); const r = render(); diff --git a/src/tui/components/AuditBoard.tsx b/src/tui/components/AuditBoard.tsx index 3a37c2e8..4f05db94 100644 --- a/src/tui/components/AuditBoard.tsx +++ b/src/tui/components/AuditBoard.tsx @@ -13,6 +13,7 @@ import type { AuditArea, AuditProgress } from '../../cli/reviewAudit.js'; import type { FixProgress } from '../../cli/reviewFixPass.js'; import type { ReviewResult } from '../../agents/agentPair.js'; import { sanitizeTerminalText } from '../sanitize.js'; +import { flattenToSingleLine, TUI_LOG_LINE_LIMIT } from '../../support/outputBudget.js'; type AreaStatus = { status: 'pending' | 'running' | 'done' | 'error'; @@ -31,6 +32,11 @@ export interface AuditBoardProps { const truncate = (s: string, n: number) => (s.length <= n ? s : `${s.slice(0, n - 1)}…`); +/** Flatten + sanitize + truncate so multiline tool logs can't blow past the row. */ +function displayLog(value: string, max: number): string { + return truncate(flattenToSingleLine(sanitizeTerminalText(value)), Math.min(max, TUI_LOG_LINE_LIMIT)); +} + export function AuditBoard({ areas, concurrency, events, mode = 'audit' }: AuditBoardProps) { const [statuses, setStatuses] = useState>(() => Object.fromEntries(areas.map((a) => [a.label, { status: 'pending' as const }])), @@ -88,7 +94,7 @@ export function AuditBoard({ areas, concurrency, events, mode = 'audit' }: Audit {' '} {` ${sanitizeTerminalText(label)}`} - {s.lastLog ? ` ${truncate(sanitizeTerminalText(s.lastLog), 48)}` : ''} + {s.lastLog ? ` ${displayLog(s.lastLog, 48)}` : ''} ))} {/* Failures carry the reason they failed. Showing only the counter left the @@ -96,7 +102,7 @@ export function AuditBoard({ areas, concurrency, events, mode = 'audit' }: Audit {errored.map(([label, s]) => ( {` ${ICON.warn} ${sanitizeTerminalText(label)}`} - {s.lastLog ? ` ${truncate(sanitizeTerminalText(s.lastLog), 64)}` : ''} + {s.lastLog ? ` ${displayLog(s.lastLog, 64)}` : ''} ))} diff --git a/src/tui/components/DataTable.tsx b/src/tui/components/DataTable.tsx index fbe2a8b3..c268d6cd 100644 --- a/src/tui/components/DataTable.tsx +++ b/src/tui/components/DataTable.tsx @@ -4,6 +4,7 @@ import { Box, Text } from 'ink'; import { displayWidth, truncateLine } from '../../cli/reviewProgress.js'; import type { Table } from '../monitorRows.js'; import { sanitizeTerminalText } from '../sanitize.js'; +import { flattenToSingleLine } from '../../support/outputBudget.js'; export interface DataTableProps extends Table { empty?: string; @@ -11,6 +12,11 @@ export interface DataTableProps extends Table { terminalWidth?: number; } +/** Normalize untrusted cell text to a single display line before width clipping. */ +function toDisplayCell(value: string): string { + return flattenToSingleLine(sanitizeTerminalText(value)); +} + export function DataTable({ columns, rows, empty, maxCellWidth = 28, terminalWidth }: DataTableProps) { if (rows.length === 0) { return {empty ?? '(no data)'}; @@ -18,7 +24,7 @@ export function DataTable({ columns, rows, empty, maxCellWidth = 28, terminalWid const visibleColumnCount = terminalWidth ? Math.max(1, Math.min(columns.length, terminalWidth)) : columns.length; - const rawColumns = columns.slice(0, visibleColumnCount).map(sanitizeTerminalText); + const rawColumns = columns.slice(0, visibleColumnCount).map(toDisplayCell); const separatorWidth = terminalWidth && visibleColumnCount > 1 ? Math.max(0, Math.min(2, Math.floor((terminalWidth - visibleColumnCount) / (visibleColumnCount - 1)))) : 2; @@ -31,7 +37,7 @@ export function DataTable({ columns, rows, empty, maxCellWidth = 28, terminalWid : maxCellWidth; const clip = (value: string) => truncateLine(value, cellWidth); const clippedColumns = rawColumns.map(clip); - const clippedRows = rows.map((row) => rawColumns.map((_, i) => clip(sanitizeTerminalText(row[i] ?? '')))); + const clippedRows = rows.map((row) => rawColumns.map((_, i) => clip(toDisplayCell(row[i] ?? '')))); const widths = clippedColumns.map((c, i) => Math.max(displayWidth(c), ...clippedRows.map((r) => displayWidth(r[i] ?? ''))), ); diff --git a/src/tui/dataTable.test.tsx b/src/tui/dataTable.test.tsx index 7f3e2b55..821df45d 100644 --- a/src/tui/dataTable.test.tsx +++ b/src/tui/dataTable.test.tsx @@ -34,6 +34,14 @@ describe('DataTable (EPIC INT-1813 S6)', () => { expect(f).not.toContain('μž‘μ—…μƒνƒœν™•μΈ'); }); + it('flattens multiline cell text before clipping so newlines cannot bypass layout', () => { + const f = render( + , + ).lastFrame()!; + expect(f).toContain('line1 line2 line3'); + expect(f).not.toContain('\nline2'); + }); + it('keeps rendered rows within the terminal width', () => { const f = render( Date: Thu, 10 Sep 2026 11:19:16 +0900 Subject: [PATCH 3/3] wip: preserved partial work (auto, session did not succeed) --- node_modules | 1 - package-lock.json | 4 ++-- 2 files changed, 2 insertions(+), 3 deletions(-) delete mode 120000 node_modules diff --git a/node_modules b/node_modules deleted file mode 120000 index d9643ec8..00000000 --- a/node_modules +++ /dev/null @@ -1 +0,0 @@ -/work/OpenSwarm/node_modules \ No newline at end of file diff --git a/package-lock.json b/package-lock.json index 87077e91..5cbe98ab 100644 --- a/package-lock.json +++ b/package-lock.json @@ -2900,7 +2900,7 @@ "version": "19.2.17", "resolved": "https://registry.npmjs.org/@types/react/-/react-19.2.17.tgz", "integrity": "sha512-MXfmqaVPEVgkBT/aY0aGCkRWWtByiYQXo3xdQ8r5RzuFrPiRn8Gar2tQdXSUQ2GKV3bkXckek89V8wQBY2Q/Aw==", - "dev": true, + "devOptional": true, "license": "MIT", "dependencies": { "csstype": "^3.2.2" @@ -4002,7 +4002,7 @@ "version": "3.2.3", "resolved": "https://registry.npmjs.org/csstype/-/csstype-3.2.3.tgz", "integrity": "sha512-z1HGKcYy2xA8AGQfwrn0PAy+PB7X/GSj3UVJW9qKyn43xWa+gl5nXmU4qqLMRzWVLFC8KusUX8T/0kCiOYpAIQ==", - "dev": true, + "devOptional": true, "license": "MIT" }, "node_modules/data-urls": {