diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 8510373..cdc69ba 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "jevpromptcoach", "displayName": "Jev (Prompt Coach)", - "version": "0.3.0", + "version": "0.3.5", "description": "Jev (Prompt Coach) scores how well you prompt a coding agent and shows your habits improving over time. Runs on TypeSafe's Jev model.", "license": "MIT", "keywords": [ diff --git a/.gitignore b/.gitignore index cf1f8c6..981bb37 100644 --- a/.gitignore +++ b/.gitignore @@ -5,7 +5,9 @@ node_modules/ # Eval fixtures are real prompts from real work. They stay on the machine that # made them. See test/fixtures/README.md. test/fixtures/prompts.json +test/fixtures/conversations.json test/eval-raw.json +test/eval-conversations-raw.json # No lockfile at the plugin root, deliberately. # diff --git a/CLAUDE.md b/CLAUDE.md index a14d574..18f3e90 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -41,6 +41,18 @@ easy to break with a change that looks reasonable. test asserts it on the wire. - **`metadata_only` means no text leaves the machine.** Any new code path that sends text must check for it. +- **Claude's replies are filtered in `src/conversation.ts`.** Only the closing + visible text of a turn is read; tool calls, tool output, thinking and + subagent records never leave that module, and an exchange whose prompt was + bypassed with `*` is dropped with its reply. Replies are on by default and go + through `applyPrivacy` like prompts; `JEVPROMPTCOACH_SESSION_REPLIES=0` turns + them off. The wire tests cover each filter. +- **Conversation checks are calibrated separately.** Each check's + `conversation` block in `src/checks.ts` has its own threshold and inline + eligibility, set from `npm run eval -- --conversations` on the gitignored + `test/fixtures/conversations.json`. Only constraints and verification are + inline-eligible so far; the labels behind them were set by the agent, not a + person, and the set is 40 follow-ups, so treat them as provisional. - **`dist/` is committed and must match `src/`.** Claude Code installs with `--ignore-scripts`, so nothing is ever built at install time. Run `npm run build` after any source change; CI fails if it drifts. diff --git a/README.es.md b/README.es.md index f259d4e..3c86597 100644 --- a/README.es.md +++ b/README.es.md @@ -235,6 +235,16 @@ misma sesión, y solo se puntúa el nuevo. Activado por defecto; `JEVPROMPTCOACH_SESSION_CONTEXT=0` lo desactiva. Los umbrales se calibraron con prompts puntuados solos. +**También las respuestas de Claude.** Un seguimiento se envía con los dos últimos +intercambios (tus prompts y el texto final de las respuestas de Claude, nunca las +herramientas ni su salida), y cada comprobación se hace en su forma +conversacional. Un intercambio saltado con `*` se descarta junto con su +respuesta. Las respuestas pasan por la misma depuración que tus prompts. +Activado por defecto; `JEVPROMPTCOACH_SESSION_REPLIES=0` envía solo tus prompts +anteriores. Por ahora solo dos +comprobaciones conversacionales se muestran en línea: lo que no debe cambiar y +los pasos de verificación. + ## Privacidad Los prompts contienen código, rutas y a veces secretos. @@ -249,9 +259,16 @@ de tu máquina salvo durante un comando que tú hayas ejecutado. | `raw` | El texto tal cual. Las cadenas con forma de credencial se eliminan **igualmente**. | Se eliminan en todos los niveles, incluido `raw`: `sk-`, `sk-ant-`, `sk-proj-`, -`ghp_` y similares, `AKIA`/`ASIA`, `AIza`, los `xox*` de Slack, JWT, bloques -PEM, secretos de cliente de Azure, tokens `Bearer`, y cualquier cosa asignada a -un nombre que acabe en `KEY`/`TOKEN`/`SECRET`/`PASSWORD`. +`ghp_` y similares, `github_pat_`, `AKIA`/`ASIA`, `AIza`, los `xox*` de Slack, +Stripe `sk_live_`/`rk_live_`, `npm_`, SendGrid `SG.`, URL de webhooks de Slack y +Discord, JWT, bloques PEM, secretos de cliente y firmas SAS de Azure, tokens +`Bearer`, la contraseña de cualquier URL `esquema://usuario:contraseña@host`, +todo lo etiquetado como `password`, y cualquier cosa asignada a un nombre que +acabe en `KEY`/`TOKEN`/`SECRET`/`PASSWORD` o `_PASS`/`_PWD`/`_AUTH`. Sin prefijo +ni etiqueta, dos formas se eliminan igualmente: cualquier secuencia de 16 o más +caracteres hexadecimales pasa a `[HEX]`, y un token de aspecto aleatorio de 20 o +más caracteres pasa a `[KEY]`. Solo un secreto que no es hexadecimal ni aleatorio +y no lleva etiqueta, como una contraseña con forma de palabra, no se elimina. **Qué se envía exactamente, y cuándo:** @@ -261,6 +278,7 @@ un nombre que acabe en `KEY`/`TOKEN`/`SECRET`/`PASSWORD`. | `/jevpromptcoach:report` | Los prompts registrados sin puntuar, depurados, por lotes | | `config backfill` | Tu historial, depurado, por lotes — **después** de una estimación de coste y una confirmación explícita | | modo `always` | Cada prompt al enviarlo, depurado, más hasta dos prompts anteriores de la misma sesión como contexto, también depurados | +| modo `always`, respuestas de Claude | Por defecto, el contexto son los dos últimos intercambios: tus prompts y el texto final de las respuestas de Claude, depurados. `JEVPROMPTCOACH_SESSION_REPLIES=0` quita las respuestas | | En cualquier otro momento | Nada | Sin telemetría. Sin ningún otro destino de red. La clave de API se lee del diff --git a/README.fr.md b/README.fr.md index 04958a7..d48e413 100644 --- a/README.fr.md +++ b/README.fr.md @@ -244,6 +244,15 @@ même session, et seul le nouveau est noté. Activé par défaut ; `JEVPROMPTCOACH_SESSION_CONTEXT=0` le désactive. Les seuils ont été calibrés sur des prompts notés seuls. +**Les réponses de Claude aussi.** Une relance est envoyée avec les deux derniers +échanges (vos prompts et le texte final des réponses de Claude, jamais les outils +ni leurs sorties), et chaque vérification est posée dans sa forme +conversationnelle. Un échange contourné par `*` est retiré avec sa réponse. Les +réponses passent par le même expurgeage que vos prompts. Activé par défaut ; +`JEVPROMPTCOACH_SESSION_REPLIES=0` n'envoie que vos prompts précédents. Pour l'instant, seules deux +vérifications conversationnelles s'affichent en ligne : ce qui ne doit pas changer +et les étapes de vérification. + ## Confidentialité Les prompts contiennent du code, des chemins, et parfois des secrets. @@ -258,9 +267,17 @@ quitte votre machine en dehors d'une commande que vous avez lancée. | `raw` | Le texte tel qu'écrit. Les chaînes en forme d'identifiant sont **quand même** retirées. | Retiré à tous les niveaux, y compris `raw` : `sk-`, `sk-ant-`, `sk-proj-`, -`ghp_` et apparentés, `AKIA`/`ASIA`, `AIza`, les `xox*` de Slack, les JWT, les -blocs PEM, les secrets clients Azure, les jetons `Bearer`, et tout ce qui est -assigné à un nom finissant par `KEY`/`TOKEN`/`SECRET`/`PASSWORD`. +`ghp_` et apparentés, `github_pat_`, `AKIA`/`ASIA`, `AIza`, les `xox*` de Slack, +Stripe `sk_live_`/`rk_live_`, `npm_`, SendGrid `SG.`, les URL de webhook Slack et +Discord, les JWT, les blocs PEM, les secrets clients et signatures SAS Azure, les +jetons `Bearer`, le mot de passe de toute URL `schéma://utilisateur:motdepasse@hôte`, +tout ce qui est étiqueté `password`, et tout ce qui est assigné à un nom +finissant par `KEY`/`TOKEN`/`SECRET`/`PASSWORD` ou `_PASS`/`_PWD`/`_AUTH`. Sans +préfixe ni étiquette, deux formes partent quand même : toute suite de 16 +caractères hexadécimaux ou plus devient `[HEX]`, et un jeton d'apparence +aléatoire de 20 caractères ou plus devient `[KEY]`. Seul un secret qui n'est ni +hexadécimal ni aléatoire et sans étiquette, comme un mot de passe en forme de +mot, n'est pas retiré. **Ce qui est envoyé, et quand :** @@ -270,6 +287,7 @@ assigné à un nom finissant par `KEY`/`TOKEN`/`SECRET`/`PASSWORD`. | `/jevpromptcoach:report` | Les prompts journalisés pas encore notés, expurgés, par lots | | `config backfill` | Votre historique, expurgé, par lots — **après** une estimation de coût et une confirmation explicite | | mode `always` | Chaque prompt au moment où vous l'envoyez, expurgé, plus jusqu'à deux prompts précédents de la même session comme contexte, expurgés eux aussi | +| mode `always`, réponses de Claude | Par défaut, le contexte est les deux derniers échanges : vos prompts et le texte final des réponses de Claude, expurgés. `JEVPROMPTCOACH_SESSION_REPLIES=0` retire les réponses | | Sinon, jamais | Rien | Aucune télémétrie. Aucune autre destination réseau. La clé d'API est lue depuis diff --git a/README.md b/README.md index 4928135..737ac14 100644 --- a/README.md +++ b/README.md @@ -292,15 +292,31 @@ Guarantees: confidence, often three or four of them, so a 0 said less than it looked. When none of those pass, the notice shows what is missing and leaves the number off. -**Follow-ups are read in context.** The first prompt of a session is scored on -its own, because it has to carry everything the agent needs. A later prompt is -sent with the two prompts before it in the same session, and only the new one -is scored: "commit and push all changes" is a fine follow-up once the earlier -prompt said what the change was. This is on by default. Set -`JEVPROMPTCOACH_SESSION_CONTEXT=0`, in your environment or in -`~/.claude/jevpromptcoach/.env`, to score every prompt alone. The thresholds -were tuned on prompts scored alone; follow-up scores have not yet been through -the eval. +**Follow-ups are read in conversation.** "Yes, commit it" can only be judged +against what Claude offered. The first prompt of a session is scored on its +own, because it has to carry everything the agent needs. A later prompt is sent +with the conversation before it, up to the last two exchanges: your prompt and +the closing text of Claude's reply, twice, then the new prompt, with whatever +is available earlier in a session. Only the new prompt is scored, and each +check is asked in its conversation form, which gives credit for what the +conversation already settled, such as accepting a change Claude described in a +named file. + +Only Claude's visible closing text is read, never tool calls, tool output or +subagent work, and an exchange whose prompt was bypassed with `*` is dropped +along with its reply. Replies go through the same redaction as your prompts. +Both are on by default: `JEVPROMPTCOACH_SESSION_REPLIES=0` leaves Claude's +replies out and sends only your earlier prompts, and +`JEVPROMPTCOACH_SESSION_CONTEXT=0` scores every prompt alone. Either goes in +your environment or in `~/.claude/jevpromptcoach/.env`. + +Measured on 40 labelled follow-ups (see +[test/fixtures/README.md](test/fixtures/README.md)), Claude's replies make +"which file or function" rank noticeably better than judging the follow-up +alone, and leave the other checks level; your earlier prompts on their own add +nothing measurable. Only two conversation checks are steady enough to show +inline so far, what must not change and the verification steps; the rest are +recorded for the report and stay off the line. `always` does not use the mechanism the docs suggest. Writing to stderr with a non-zero exit displays nothing on Claude Code 2.1.277; a top-level @@ -324,9 +340,22 @@ your machine except during a command you ran. | `raw` | Prompt text as written. Credential-shaped strings are **still** stripped. | Stripped at every level, including `raw`: `sk-`, `sk-ant-`, `sk-proj-`, `ghp_` -and friends, `AKIA`/`ASIA`, `AIza`, Slack `xox*`, JWTs, PEM blocks, Azure client -secrets, `Bearer` tokens, and anything assigned to a name ending in -`KEY`/`TOKEN`/`SECRET`/`PASSWORD`. +and friends, `github_pat_`, `AKIA`/`ASIA`, `AIza`, Slack `xox*`, Stripe +`sk_live_`/`rk_live_`, `npm_`, SendGrid `SG.`, Slack and Discord webhook URLs, +JWTs, PEM blocks, Azure client secrets and SAS signatures, `Bearer` tokens, the +password in any `scheme://user:password@host` URL, anything labelled `password` +(JSON keys and "password is …" included), and anything assigned to a name +ending in `KEY`/`TOKEN`/`SECRET`/`PASSWORD` or `_PASS`/`_PWD`/`_AUTH`. With no +prefix and no label, two shapes still go: any run of 16 or more hex characters +becomes `[HEX]` (commit SHAs too; the marker keeps the fact that an identifier +was named), and a random-looking token of 20 or more characters becomes +`[KEY]`. + +Redaction works by shape, and that has a limit: a secret that is neither hex +nor random-looking and carries no label, such as a word-like password on its +own, is not removed. That applies to your prompts and to Claude's replies +alike; set `JEVPROMPTCOACH_SESSION_REPLIES=0` if you would rather replies never +leave the machine. **Exactly what is sent, and when:** @@ -335,7 +364,7 @@ secrets, `Bearer` tokens, and anything assigned to a name ending in | `/jevpromptcoach:score` | The one prompt you passed, redacted | | `/jevpromptcoach:report` | Any logged prompts not yet scored, redacted, batched | | `config backfill` | Your history, redacted, batched — **after** a cost estimate and an explicit confirmation | -| `always` mode | Each prompt as you submit it, redacted, plus up to two earlier prompts from the same session as context, redacted again at the current level | +| `always` mode | Each prompt as you submit it, redacted, plus the conversation before it as context: up to the last two exchanges, your prompts and the closing text of Claude's replies, redacted again at the current level. `JEVPROMPTCOACH_SESSION_REPLIES=0` drops the replies; `JEVPROMPTCOACH_SESSION_CONTEXT=0` drops the context | | Ever, otherwise | Nothing | No telemetry. No other network destination. The API key is read from the diff --git a/dist/chunk-HXIO5BL2.js b/dist/chunk-33DTCFCS.js similarity index 61% rename from dist/chunk-HXIO5BL2.js rename to dist/chunk-33DTCFCS.js index e750922..1e012d7 100644 --- a/dist/chunk-HXIO5BL2.js +++ b/dist/chunk-33DTCFCS.js @@ -4,7 +4,19 @@ import { CORRECTIONS_PATH, LOG_PATH, ensureDataDir -} from "./chunk-7PP552KK.js"; +} from "./chunk-KYVNDBDC.js"; + +// src/hash.ts +import { createHash } from "node:crypto"; +function promptHash(text) { + return createHash("sha256").update(text.trim(), "utf8").digest("hex").slice(0, 16); +} +function stripPreamble(text) { + return text.replace(/^\s*[\s\S]*?<\/system_instruction>\s*/g, "").replace(/[\s\S]*?<\/ide_selection>/g, "").trim(); +} +function promptMatchKey(text) { + return promptHash(stripPreamble(text).replace(/\s+/g, " ")); +} // src/log.ts import { @@ -116,7 +128,69 @@ function writeCorrections(records) { writeFileSync(CORRECTIONS_PATH, JSON.stringify([...records]), { mode: 384 }); } +// src/skip.ts +var MIN_CHARS = 15; +var ACKNOWLEDGEMENTS = /* @__PURE__ */ new Set([ + "yes", + "no", + "y", + "n", + "ok", + "okay", + "k", + "sure", + "yep", + "yeah", + "nope", + "continue", + "go", + "go ahead", + "proceed", + "next", + "stop", + "wait", + "done", + "thanks", + "thank you", + "ty", + "please", + "do it", + "try again", + "retry", + "fix it", + "again", + "good", + "nice", + "perfect", + "great", + "cool", + "hmm" +]); +function skipReason(text, bypassPrefix = "*") { + const trimmed = text.trim(); + if (bypassPrefix && trimmed.startsWith(bypassPrefix)) return "bypass_prefix"; + if (trimmed.startsWith("/")) return "slash_command"; + if (/^<(command-name|command-message|command-args|local-command-stdout|bash-input|bash-stdout|user-memory-input)/.test( + trimmed + )) { + return "command_wrapper"; + } + if (trimmed.startsWith("This session is being continued from a previous conversation")) { + return "session_meta"; + } + if (trimmed.startsWith("Caveat: The messages below were generated")) return "session_meta"; + if (trimmed.startsWith("" and matched no labelled rule below. + // Bounded: unbounded runs made a long line of dots or dashes quadratic. + pattern: /(? m.replace(/((?::|=|-\s|\bis\s)\s*['"`]?)([^\s'"`,;/\\]{12,})/i, "$1[REDACTED]") + }, + { + name: "labelled-password", + // Passwords are short more often than keys are, so the length floor is + // lower than for the generic labels above; the label itself is specific. + // The optional quote after the label covers a JSON key: "password": "x". + // Prose counts too: "the password is x" was found in real history. + pattern: /\b(password|passwd|pwd)\b['"]?(?:\s*[:=]|\s+is)\s*(['"`]?)([^\s'"`,;]{6,})\2/gi, + replace: (m) => m.replace(/((?:[:=]|\bis)\s*['"`]?)([^\s'"`,;]{6,})/i, "$1[REDACTED]") + }, + { + name: "assigned-secret", + // KEY=value / "api_key": "value" / TOKEN: value / DB_PASS=value. The short + // suffixes need an underscore before them, so bypass: or oauth: in code is + // not taken for a secret. + pattern: /\b([A-Za-z_][A-Za-z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIAL)S?|(?:[A-Za-z0-9]+_)+(?:PASS|PWD|AUTH))\b(\s*[:=]\s*)(['"]?)([^\s'"`,;]{6,})\3/gi, + replace: (m) => m.replace(/([:=]\s*['"]?)([^\s'"`,;]{6,})/, "$1[REDACTED]") + }, + { + name: "long-hex", + // Hashes, hex tokens and hex-encoded keys: 16 or more hex characters with a + // digit among them. Commit SHAs go too; the marker still tells the scorer a + // specific identifier was named. UUIDs survive: their hex runs are shorter. + // An 0x prefix is how hex private keys are usually written, and a hyphen + // before or after (token-) does not hide one. + pattern: /(? m.split(/[-_+=/]/).some(looksRandom) || looksRandom(m.replace(/[-_+=/]/g, "")) ? "[KEY]" : m + } +]; +function looksRandom(s) { + if (s.length < 16) return false; + const digits = (s.match(/\d/g) ?? []).length; + const lower = (s.match(/[a-z]/g) ?? []).length; + const upper = (s.match(/[A-Z]/g) ?? []).length; + if (digits < 1 || lower < 2 || upper < 2) return false; + const words = (s.match(/[A-Z][a-z]{2,}/g) ?? []).length; + if (words / upper >= 0.5) return false; + const kind = (c) => /\d/.test(c) ? 0 : /[a-z]/.test(c) ? 1 : /[A-Z]/.test(c) ? 2 : 3; + let switches = 0; + for (let i = 1; i < s.length; i += 1) if (kind(s[i]) !== kind(s[i - 1])) switches += 1; + return switches / (s.length - 1) >= 0.4; +} +var EMAIL_RULE = { + name: "email", + // Bounded to the RFC limits: unbounded runs made a long dotted line quadratic. + pattern: /\b[A-Za-z0-9._%+-]{1,64}@[A-Za-z0-9.-]{1,253}\.[A-Za-z]{2,63}\b/g, + replace: "[EMAIL]" +}; +var PATH_RULES = [ + { + name: "home", + pattern: /(?:\/Users\/|\/home\/|C:\\Users\\)[^\s'"`)\]]+/g, + replace: (m) => { + const base = m.split(/[/\\]/).filter(Boolean).pop() ?? ""; + return base && /[.\w]/.test(base) ? `~/\u2026/${base}` : "~/\u2026"; + } + }, + { + name: "absolute", + pattern: /(? { + const base = m.split("/").filter(Boolean).pop() ?? ""; + return base ? `\u2026/${base}` : "\u2026/"; + } + } +]; +function applyRules(text, rules) { + let out = text; + for (const rule of rules) { + out = typeof rule.replace === "function" ? out.replace(rule.pattern, rule.replace) : out.replace(rule.pattern, rule.replace); + } + return out; +} +function redact(text) { + return applyRules(applyRules(text, CREDENTIAL_RULES), [EMAIL_RULE, ...PATH_RULES]); +} +function stripCredentials(text) { + return applyRules(text, CREDENTIAL_RULES); +} +function features(text) { + return { + chars: text.length, + words: text.trim().split(/\s+/).filter(Boolean).length, + lines: text.split("\n").length, + hasCodeFence: /```/.test(text), + // Anchored and bounded: unanchored, a long run of word characters made this + // quadratic, on every prompt the hook sees. + hasFilePath: /(? typeof b === "object" && b !== null).filter((b) => b.type === "text").map((b) => b.text ?? "").join("\n"); + } + return ""; +} +function hasToolResult(content) { + return Array.isArray(content) && content.some((b) => typeof b === "object" && b !== null && b.type === "tool_result"); +} +function isHumanPrompt(record) { + if (record.type !== "user") return false; + if (record.isSidechain || record.isMeta) return false; + if (hasToolResult(record.message?.content)) return false; + const source = record.promptSource; + if (source !== void 0) return source === "typed"; + return true; +} +function listTranscripts(dir = PROJECTS_DIR) { + if (!existsSync(dir)) return []; + const files = []; + for (const project of readdirSync(dir)) { + const projectDir = join(dir, project); + try { + if (!statSync(projectDir).isDirectory()) continue; + for (const file of readdirSync(projectDir)) { + if (file.endsWith(".jsonl")) files.push(join(projectDir, file)); + } + } catch { + } + } + return files; +} +async function readTranscript(path, out) { + const project = path.split("/").slice(-2, -1)[0] ?? "unknown"; + const stream = createReadStream(path, { encoding: "utf8" }); + const lines = createInterface({ input: stream, crlfDelay: Infinity }); + try { + for await (const line of lines) { + if (!line.trim()) continue; + let record; + try { + record = JSON.parse(line); + } catch { + continue; + } + if (!isHumanPrompt(record)) continue; + const text = stripPreamble(textOf(record.message?.content)); + if (!text) continue; + out.push({ + ts: record.timestamp ?? (/* @__PURE__ */ new Date(0)).toISOString(), + session: record.sessionId ?? path, + text, + project + }); + } + } finally { + lines.close(); + stream.destroy(); + } +} +async function readHistory(dir = PROJECTS_DIR) { + const prompts = []; + for (const file of listTranscripts(dir)) { + try { + await readTranscript(file, prompts); + } catch { + } + } + prompts.sort((a, b) => a.ts.localeCompare(b.ts)); + return prompts; +} + +// src/conversation.ts +import { closeSync, createReadStream as createReadStream2, fstatSync, openSync, readSync } from "node:fs"; +import { createInterface as createInterface2 } from "node:readline"; +var MAX_REPLY_CHARS = 1500; +var TAIL_BYTES = 1024 * 1024; +var EXCLUDED = /* @__PURE__ */ new Set([ + "bypass_prefix", + "slash_command", + "command_wrapper", + "session_meta" +]); +function clampReply(text) { + return text.length <= MAX_REPLY_CHARS ? text : `\u2026 +${text.slice(-(MAX_REPLY_CHARS - 2))}`; +} +function blockType(b) { + return typeof b === "object" && b !== null ? b.type : void 0; +} +function isQueuedPrompt(record) { + return record.type === "user" && record.promptSource === "queued" && !record.isMeta && !(Array.isArray(record.message?.content) && record.message.content.some((b) => blockType(b) === "tool_result")); +} +function isOtherSender(record) { + return record.type === "user" && record.promptSource !== void 0 && // A meta record from the system or an SDK is still a message nobody typed. + (!record.isMeta || record.promptSource === "system" || record.promptSource === "sdk") && !(Array.isArray(record.message?.content) && record.message.content.some((b) => blockType(b) === "tool_result")); +} +var ExchangeBuilder = class { + exchanges = []; + /** + * `excluded` says why an exchange is dropped: 'prompt' when its prompt must + * not be shown (bypassed, a slash command, not typed), 'sender' when it is + * only a boundary after a message nobody typed. + */ + current = null; + parts = []; + bypassPrefix; + constructor(bypassPrefix) { + this.bypassPrefix = bypassPrefix; + } + add(record) { + if (record.isSidechain) return; + if (isHumanPrompt(record) || isQueuedPrompt(record)) { + const inheritsExclusion = isQueuedPrompt(record) && this.current?.excluded === "prompt"; + this.close(); + const raw = textOf(record.message?.content); + const prompt = stripPreamble(raw); + const reason = skipReason(prompt || raw, this.bypassPrefix); + this.current = { + prompt, + key: promptMatchKey(raw), + reply: "", + ts: record.timestamp ?? "", + session: record.sessionId ?? "", + excluded: inheritsExclusion || !prompt || reason !== null && EXCLUDED.has(reason) ? "prompt" : null + }; + return; + } + if (isOtherSender(record)) { + this.close(); + this.current = { prompt: "", key: "", reply: "", ts: "", session: "", excluded: "sender" }; + return; + } + if (record.type !== "assistant" || !this.current) return; + let content = record.message?.content; + if (Array.isArray(content)) { + const lastToolUse = content.findLastIndex((b) => blockType(b)?.endsWith("tool_use") === true); + if (lastToolUse >= 0) { + this.parts = []; + content = content.slice(lastToolUse + 1); + } + } + const text = textOf(content).trim(); + if (text) this.parts.push(text); + } + close() { + if (this.current && !this.current.excluded) { + const { prompt, key, ts, session } = this.current; + this.exchanges.push({ prompt, key, ts, session, reply: this.parts.join("\n\n") }); + } + this.current = null; + this.parts = []; + } +}; +function readTail(path) { + const fd = openSync(path, "r"); + try { + const size = fstatSync(fd).size; + const length = Math.min(size, TAIL_BYTES); + const buffer = Buffer.alloc(length); + readSync(fd, buffer, 0, length, size - length); + return buffer.toString("utf8"); + } finally { + closeSync(fd); + } +} +function parseLine(line) { + if (!line.trim()) return null; + try { + return JSON.parse(line); + } catch { + return null; + } +} +function recentTurns(transcriptPath, currentKey, count, bypassPrefix) { + const builder = new ExchangeBuilder(bypassPrefix); + for (const line of readTail(transcriptPath).split("\n")) { + const record = parseLine(line); + if (record) builder.add(record); + } + builder.close(); + const exchanges = builder.exchanges; + const last = exchanges.at(-1); + if (last && !last.reply && last.key === currentKey) exchanges.pop(); + return toTurns(exchanges.slice(-count)); +} +function toTurns(exchanges) { + return exchanges.flatMap( + (e) => e.reply ? [ + { role: "developer", text: e.prompt }, + { role: "agent", text: e.reply } + ] : [{ role: "developer", text: e.prompt }] + ); +} +async function readConversations(bypassPrefix, dir = PROJECTS_DIR) { + const sessions = []; + for (const file of listTranscripts(dir)) { + const builder = new ExchangeBuilder(bypassPrefix); + const stream = createReadStream2(file, { encoding: "utf8" }); + const lines = createInterface2({ input: stream, crlfDelay: Infinity }); + try { + for await (const line of lines) { + const record = parseLine(line); + if (record) builder.add(record); + } + } catch { + continue; + } finally { + lines.close(); + stream.destroy(); + } + builder.close(); + if (builder.exchanges.length > 1) sessions.push(builder.exchanges); + } + return sessions; +} + +export { + readHistory, + clampReply, + recentTurns, + toTurns, + readConversations +}; diff --git a/dist/chunk-F3M3WE3B.js b/dist/chunk-F3M3WE3B.js deleted file mode 100644 index 21805b7..0000000 --- a/dist/chunk-F3M3WE3B.js +++ /dev/null @@ -1,71 +0,0 @@ -// JevPromptCoach — generated by scripts/build.mjs. Do not edit. - -// src/hash.ts -import { createHash } from "node:crypto"; -function promptHash(text) { - return createHash("sha256").update(text.trim(), "utf8").digest("hex").slice(0, 16); -} - -// src/skip.ts -var MIN_CHARS = 15; -var ACKNOWLEDGEMENTS = /* @__PURE__ */ new Set([ - "yes", - "no", - "y", - "n", - "ok", - "okay", - "k", - "sure", - "yep", - "yeah", - "nope", - "continue", - "go", - "go ahead", - "proceed", - "next", - "stop", - "wait", - "done", - "thanks", - "thank you", - "ty", - "please", - "do it", - "try again", - "retry", - "fix it", - "again", - "good", - "nice", - "perfect", - "great", - "cool", - "hmm" -]); -function skipReason(text, bypassPrefix = "*") { - const trimmed = text.trim(); - if (bypassPrefix && trimmed.startsWith(bypassPrefix)) return "bypass_prefix"; - if (trimmed.startsWith("/")) return "slash_command"; - if (/^<(command-name|command-message|command-args|local-command-stdout|bash-input|bash-stdout|user-memory-input)/.test( - trimmed - )) { - return "command_wrapper"; - } - if (trimmed.startsWith("This session is being continued from a previous conversation")) { - return "session_meta"; - } - if (trimmed.startsWith("Caveat: The messages below were generated")) return "session_meta"; - if (trimmed.startsWith("" and matched no labelled rule below. - pattern: /(? m.replace(/((?::|=|-\s)\s*['"`]?)([^\s'"`,;/\\]{12,})/, "$1[REDACTED]") - }, - { - name: "assigned-secret", - // KEY=value / "api_key": "value" / TOKEN: value - pattern: /\b([A-Za-z_][A-Za-z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIAL)S?)\b(\s*[:=]\s*)(['"]?)([^\s'"`,;]{6,})\3/gi, - replace: (m) => m.replace(/([:=]\s*['"]?)([^\s'"`,;]{6,})/, "$1[REDACTED]") - } -]; -var EMAIL_RULE = { - name: "email", - pattern: /\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b/g, - replace: "[EMAIL]" -}; -var PATH_RULES = [ - { - name: "home", - pattern: /(?:\/Users\/|\/home\/|C:\\Users\\)[^\s'"`)\]]+/g, - replace: (m) => { - const base = m.split(/[/\\]/).filter(Boolean).pop() ?? ""; - return base && /[.\w]/.test(base) ? `~/\u2026/${base}` : "~/\u2026"; - } - }, - { - name: "absolute", - pattern: /(? { - const base = m.split("/").filter(Boolean).pop() ?? ""; - return base ? `\u2026/${base}` : "\u2026/"; - } - } -]; -function applyRules(text, rules) { - let out = text; - for (const rule of rules) { - out = typeof rule.replace === "function" ? out.replace(rule.pattern, rule.replace) : out.replace(rule.pattern, rule.replace); - } - return out; -} -function redact(text) { - return applyRules(applyRules(text, CREDENTIAL_RULES), [EMAIL_RULE, ...PATH_RULES]); -} -function stripCredentials(text) { - return applyRules(text, CREDENTIAL_RULES); -} -function features(text) { - return { - chars: text.length, - words: text.trim().split(/\s+/).filter(Boolean).length, - lines: text.split("\n").length, - hasCodeFence: /```/.test(text), - hasFilePath: /[\w\-/]+\.[a-z]{1,5}\b/i.test(text), - hasQuestionMark: text.includes("?"), - hasErrorWord: /\b(error|exception|traceback|failed|stack ?trace)\b/i.test(text) - }; -} -function applyPrivacy(text, privacy) { - const f = features(text); - if (privacy === "metadata_only") return { text: null, features: f }; - if (privacy === "raw") return { text: stripCredentials(text), features: f }; - return { text: redact(text), features: f }; -} - -export { - redact, - stripCredentials, - features, - applyPrivacy -}; diff --git a/dist/chunk-W564PYU5.js b/dist/chunk-X33S6F2H.js similarity index 62% rename from dist/chunk-W564PYU5.js rename to dist/chunk-X33S6F2H.js index d9d496e..e5824ab 100644 --- a/dist/chunk-W564PYU5.js +++ b/dist/chunk-X33S6F2H.js @@ -1,7 +1,7 @@ // JevPromptCoach — generated by scripts/build.mjs. Do not edit. import { apiKey -} from "./chunk-7PP552KK.js"; +} from "./chunk-KYVNDBDC.js"; // src/checks.ts var GATES = [ @@ -36,6 +36,16 @@ var CHECKS = [ inlineMargin: 0.2, // CV fail-precision 0.96 over 28 failing examples. inlineEligible: true, + conversation: { + instructions: "The agent knows exactly where to work: the message names a concrete file, path, function, class, component, endpoint, or symbol, or points unambiguously at one already named in the conversation.", + criteria: { + true: 'Names a concrete target, or refers without ambiguity to one named earlier by either side: "the email one" after the agent listed email-worker.ts among other files, or "yes, go ahead" accepting a change the agent described in a named file.', + false: "The target is unclear even with the conversation: nothing concrete was named earlier, or several candidates were named and the message does not say which, or the message starts new work described only in general terms." + }, + threshold: 0.15, + // CV fail-precision 0.60 over 11 failing examples: ranks better than standalone (AUC 0.82 vs 0.63) but too thin to stand behind inline. + inlineEligible: false + }, cause: 'You wrote "it" or "the code" instead of a name.', consequence: "The agent has to guess which file you meant. It searches, or it edits the wrong one.", fix: "Name the file, function, or symbol you want changed." @@ -53,6 +63,16 @@ var CHECKS = [ inlineMargin: 0.2, // CV fail-precision 1.00 over 14 failing examples. inlineEligible: true, + conversation: { + instructions: "What should be true once the request in the message is done is known: the message states the outcome or the output it wants, or it approves a proposal from the agent that describes the resulting behaviour.", + criteria: { + true: 'States the intended end state or the output wanted ("so the page shows X", "give me five titles"), or approves a specific proposal in which the agent described what the result will be.', + false: "Only names a step to carry out (commit, push, deploy, merge, review, run something) without an outcome, even when the conversation describes the work around it; or starts new work without an end state." + }, + threshold: 0.2, + // CV fail-precision 0.75 over 18 failing examples, below the 0.90 bar. + inlineEligible: false + }, cause: "Nothing in the request says what should be true at the end.", consequence: "The agent picks its own finish line, and stops somewhere you did not want.", fix: "Add one sentence: what should be true when this works." @@ -70,6 +90,16 @@ var CHECKS = [ inlineMargin: 0.2, // CV fail-precision 1.00 over 10 failing examples. inlineEligible: true, + conversation: { + instructions: "The request, read with the conversation, is one contained, well-defined piece of work rather than an open-ended or sweeping change.", + criteria: { + true: "Asks for one change or a small set of clearly enumerated changes, including accepting one specific proposal the agent described.", + false: 'Asks for something open-ended or sweeping, bundles several unrelated requests, or accepts a broad proposal ("do all of it") whose edges the conversation never set.' + }, + threshold: 0.45, + // Not measurable: 2 failing examples in the conversation set. + inlineEligible: false + }, cause: "The request has no edges, so the agent decides how far to go.", consequence: "You get a huge change touching files you never meant to touch, and reviewing it takes longer than the fix would have.", fix: "Cut it to the one change you want first. Ask for the rest after." @@ -85,8 +115,18 @@ var CHECKS = [ }, threshold: 0.3, inlineMargin: 0.2, - // CV fail-precision 0.97 over 30 failing examples. + // CV fail-precision 1.00 over 30 failing examples (0.97 before the eval redacted). inlineEligible: true, + conversation: { + instructions: "A limit on the work is in force: the message states one, or one stated earlier in the conversation still applies to what the message asks for.", + criteria: { + true: "Names something to leave alone or preserve, or forbids an approach, in the message or earlier in the conversation for this same work, and nothing has withdrawn it.", + false: "No limit that applies to the requested work appears in the message or anywhere in the conversation." + }, + threshold: 0.8, + // CV fail-precision 0.97 over 34 failing examples. 34 of 40 fixtures fail it, so precision is flattered by the base rate; AUC 0.99. + inlineEligible: true + }, cause: "Nothing in the request is marked off-limits.", consequence: "Something that was working gets rewritten along the way.", fix: "Say what must stay as it is \u2014 the API, the database schema, the other callers." @@ -105,6 +145,16 @@ var CHECKS = [ inlineMargin: 0.2, // only 5 failing examples in the fixture set — too thin to stand behind inline. inlineEligible: false, + conversation: { + instructions: "For the failure being reported, the actual evidence is available: in the message, or earlier in the conversation, quoted by the developer or reported by the agent from running something.", + criteria: { + true: "Real output (an error message, stack trace, failing assertion, or log) or a specific expected-versus-actual pair appears in the message or the conversation for this failure.", + false: "The failure is described only in general terms, and no actual output or concrete expected-versus-actual pair for it appears anywhere in the conversation." + }, + threshold: 0.4, + // Not measurable: 1 failing example in the conversation set. + inlineEligible: false + }, cause: "The bug is described, but the actual error text is not in the message.", consequence: "The agent guesses the error from your description and fixes a different problem.", fix: "Paste the real error, and say what you expected to happen instead." @@ -121,8 +171,19 @@ var CHECKS = [ appliesWhen: { gate: "is_large_change", minProbability: 0.5 }, threshold: 0.35, inlineMargin: 0.2, - // CV fail-precision 0.86, below the 0.90 bar. + // 8 failing examples, under the 10 the inline bar asks for (CV fail-precision + // 1.00 on the redacted run, 0.86 before it). inlineEligible: false, + conversation: { + instructions: "Before a large or risky change is carried out, a plan has been asked for or seen: the message asks for one, or it approves a plan the agent laid out in the conversation.", + criteria: { + true: "Asks to plan, propose, outline, or investigate first, or approves a specific plan the agent already described for this change.", + false: "Asks for the large change to be carried out directly, and no plan for it appears in the conversation." + }, + threshold: 0.35, + // Not measurable: 4 failing examples in the conversation set. + inlineEligible: false + }, cause: "You asked for a big or risky change without asking to see the approach first.", consequence: "You find out how it was going to be done only after it has been done.", fix: 'Ask for the plan first, then approve it. "Plan this before changing anything."' @@ -140,6 +201,16 @@ var CHECKS = [ inlineMargin: 0.2, // CV fail-precision 1.00 over 39 failing examples. inlineEligible: true, + conversation: { + instructions: "How the work will be checked is known: the message names a test, command, or check, or accepts a proposal from the agent that names one, or continues work whose verification was stated earlier.", + criteria: { + true: "A runnable or checkable step for this work (a test file or name, a command, a script, a page to load) appears in the message, or in an earlier message or agent proposal the message accepts or continues.", + false: 'No test, command, or check for this work appears in the message or the conversation, or the message says only "make sure it works".' + }, + threshold: 0.75, + // CV fail-precision 0.97 over 38 failing examples. 38 of 40 fixtures fail it, so precision is flattered by the base rate; AUC 0.89. + inlineEligible: true + }, cause: "Nothing in the request says how to tell whether it worked.", consequence: "The agent says it worked, and you find out later that it did not.", fix: 'Name the command. "Verify with npm test -- auth.spec.ts."' @@ -239,11 +310,19 @@ function clampPrompt(text) { \u2026 ${text.slice(-1e3)}`; } -function questionsFor(id, contextIds = []) { +function scopeFor(id, contextIds, mode) { + const ids = contextIds.map((c) => `"${c}"`).join(", "); + if (mode === "conversation" && contextIds.length) { + return `Judge only the message with id "${id}" in the state: the developer's latest prompt to a coding agent. The messages ${ids} are the conversation before it, in order, and "role" says whether the developer or the agent wrote each. They are context and are not themselves being judged.`; + } + if (contextIds.length) { + return `Judge only the message with id "${id}" in the state. The messages ${ids} are earlier prompts from the same conversation, given as context: anything they already state counts as known to the reader of "${id}", but they are not themselves being judged.`; + } + return `Consider only the message with id "${id}" in the state.`; +} +function questionsFor(id, contextIds = [], mode = "alone") { const questions = {}; - const scope = contextIds.length ? `Judge only the message with id "${id}" in the state. The messages ${contextIds.map((c) => `"${c}"`).join( - " and " - )} are earlier prompts from the same conversation, given as context: anything they already state counts as known to the reader of "${id}", but they are not themselves being judged.` : `Consider only the message with id "${id}" in the state.`; + const scope = scopeFor(id, contextIds, mode); for (const gate of GATES) { questions[`${id}__${gate.id}`] = { type: "noul", @@ -252,10 +331,11 @@ function questionsFor(id, contextIds = []) { }; } for (const check of CHECKS) { + const def = mode === "conversation" ? check.conversation : check; questions[`${id}__${check.id}`] = { type: "noul", - instructions: `${scope} ${check.instructions}`, - criteria: check.criteria + instructions: `${scope} ${def.instructions}`, + criteria: def.criteria }; } return questions; @@ -266,22 +346,35 @@ function questionTokens(questions) { 0 ); } +function messagesFor(key, input) { + const turns = (input.conversation ?? []).map((t, i) => ({ + id: `${key}c${i + 1}`, + role: t.role, + text: clampPrompt(t.text) + })); + if (turns.length === 0) return [{ id: key, text: clampPrompt(input.text) }]; + return [...turns, { id: key, role: "developer", text: clampPrompt(input.text) }]; +} +function questionsForInput(key, messages) { + const contextIds = messages.slice(0, -1).map((m) => m.id); + return questionsFor(key, contextIds, contextIds.length ? "conversation" : "alone"); +} function planBatches(inputs) { const batches = []; let current = { items: [], tokens: 0 }; let stateTokens = 0; inputs.forEach((input, index) => { const key = `m${index}`; - const text = clampPrompt(input.text); - const itemStateTokens = estimateTokens(text) + 12; - const itemTokens = itemStateTokens + questionTokens(questionsFor(key)); + const messages = messagesFor(key, input); + const itemStateTokens = messages.reduce((sum, m) => sum + estimateTokens(m.text) + 12, 0); + const itemTokens = itemStateTokens + questionTokens(questionsForInput(key, messages)); const wouldOverflow = current.items.length > 0 && (current.items.length >= MAX_PROMPTS_PER_REQUEST || stateTokens + itemStateTokens > STATE_TOKEN_BUDGET || current.tokens + itemTokens > REQUEST_TOKEN_BUDGET); if (wouldOverflow) { batches.push(current); current = { items: [], tokens: 0 }; stateTokens = 0; } - current.items.push({ key, input: { hash: input.hash, text } }); + current.items.push({ key, input, messages }); stateTokens += itemStateTokens; current.tokens += itemTokens; }); @@ -301,14 +394,15 @@ function interpret(hash, probabilities, gates, opts = {}) { return { id: def.id, label: def.label, verdict: "n/a", probability: null, def }; } } + const { threshold, inlineEligible } = opts.mode === "conversation" ? def.conversation : def; const p = probabilities[def.id]; if (p === void 0) { return { id: def.id, label: def.label, verdict: "undecided", probability: null, def }; } - if (opts.inlineSafe && (!def.inlineEligible || Math.abs(p - def.threshold) < def.inlineMargin)) { + if (opts.inlineSafe && (!inlineEligible || Math.abs(p - threshold) < def.inlineMargin)) { return { id: def.id, label: def.label, verdict: "undecided", probability: p, def }; } - return { id: def.id, label: def.label, verdict: p >= def.threshold ? "pass" : "fail", probability: p, def }; + return { id: def.id, label: def.label, verdict: p >= threshold ? "pass" : "fail", probability: p, def }; }); const decided = checks.filter((c) => c.verdict === "pass" || c.verdict === "fail"); const score = decided.length ? Math.round(decided.filter((c) => c.verdict === "pass").length / decided.length * 100) : null; @@ -328,17 +422,20 @@ function unpack(answers, key, hash, ts) { return { hash, ts, probabilities, gates, model: MODEL }; } async function runBatch(batch, options) { - const state = { - messages: batch.items.map((item) => ({ id: item.key, text: item.input.text })) - }; + const state = { messages: batch.items.flatMap((item) => item.messages) }; const questions = {}; - for (const item of batch.items) Object.assign(questions, questionsFor(item.key)); + for (const item of batch.items) Object.assign(questions, questionsForInput(item.key, item.messages)); const answers = await ask(state, questions, { timeoutMs: options.timeoutMs ?? 6e4, onUsage: options.onUsage }); const now = (/* @__PURE__ */ new Date()).toISOString(); - return batch.items.map((item) => unpack(answers, item.key, item.input.hash, now)); + return batch.items.map((item) => { + const record = unpack(answers, item.key, item.input.hash, now); + const turns = item.input.conversation?.length ?? 0; + if (turns) Object.assign(record, { context: turns, conversation: true }); + return record; + }); } async function scoreMany(inputs, options = {}) { return runPool( @@ -349,20 +446,25 @@ async function scoreMany(inputs, options = {}) { ); } async function scoreOne(text, hash, options = {}) { - const context = (options.context ?? []).map((t, i) => ({ id: `c${i + 1}`, text: clampPrompt(t) })); - const state = { messages: [...context, { id: "m0", text: clampPrompt(text) }] }; - const answers = await tryAsk( - state, - questionsFor( - "m0", - context.map((c) => c.id) - ), - { timeoutMs: options.timeoutMs ?? 2e4 } - ); + let messages; + let mode; + if (options.conversation?.length) { + messages = messagesFor("m0", { hash, text, conversation: options.conversation }); + mode = "conversation"; + } else { + const context = (options.context ?? []).map((t, i) => ({ id: `c${i + 1}`, text: clampPrompt(t) })); + messages = [...context, { id: "m0", text: clampPrompt(text) }]; + mode = context.length ? "prompts" : "alone"; + } + const contextIds = messages.slice(0, -1).map((m) => m.id); + const answers = await tryAsk({ messages }, questionsFor("m0", contextIds, mode), { + timeoutMs: options.timeoutMs ?? 2e4 + }); if (!answers) return null; const record = unpack(answers, "m0", hash, (/* @__PURE__ */ new Date()).toISOString()); - if (context.length) record.context = context.length; - return { record, result: interpret(hash, record.probabilities, record.gates, options) }; + if (contextIds.length) record.context = contextIds.length; + if (mode === "conversation") record.conversation = true; + return { record, result: interpret(hash, record.probabilities, record.gates, { ...options, mode }) }; } export { diff --git a/dist/cli.js b/dist/cli.js index f2ed7a1..95071ff 100644 --- a/dist/cli.js +++ b/dist/cli.js @@ -1,19 +1,23 @@ // JevPromptCoach — generated by scripts/build.mjs. Do not edit. import { - promptHash, - skipReason -} from "./chunk-F3M3WE3B.js"; + clampReply, + readConversations, + readHistory, + toTurns +} from "./chunk-EF7K2A5B.js"; import { appendLogMany, appendScores, clearLocalData, compactScores, hasText, + promptHash, readCorrections, readLog, readScores, + skipReason, writeCorrections -} from "./chunk-HXIO5BL2.js"; +} from "./chunk-33DTCFCS.js"; import { CHECKS, CORRECTION_QUESTION, @@ -27,7 +31,7 @@ import { runPool, scoreMany, scoreOne -} from "./chunk-W564PYU5.js"; +} from "./chunk-X33S6F2H.js"; import { ENV_PATH, LOG_PATH, @@ -35,10 +39,10 @@ import { apiKeySource, loadConfig, saveConfig -} from "./chunk-7PP552KK.js"; +} from "./chunk-KYVNDBDC.js"; import { applyPrivacy -} from "./chunk-VM4R2HGT.js"; +} from "./chunk-A7NLXWHN.js"; // src/cli.ts import { readFileSync, writeFileSync } from "node:fs"; @@ -124,88 +128,6 @@ async function detectCorrections(pairs, options = {}) { ); } -// src/history.ts -import { readdirSync, statSync, createReadStream, existsSync } from "node:fs"; -import { join } from "node:path"; -import { homedir } from "node:os"; -import { createInterface } from "node:readline"; -var PROJECTS_DIR = join(homedir(), ".claude", "projects"); -function textOf(content) { - if (typeof content === "string") return content; - if (Array.isArray(content)) { - return content.filter((b) => typeof b === "object" && b !== null).filter((b) => b.type === "text").map((b) => b.text ?? "").join("\n"); - } - return ""; -} -function hasToolResult(content) { - return Array.isArray(content) && content.some((b) => typeof b === "object" && b !== null && b.type === "tool_result"); -} -function isHumanPrompt(record) { - if (record.type !== "user") return false; - if (record.isSidechain || record.isMeta) return false; - if (hasToolResult(record.message?.content)) return false; - const source = record.promptSource; - if (source !== void 0) return source === "typed"; - return true; -} -function listTranscripts(dir = PROJECTS_DIR) { - if (!existsSync(dir)) return []; - const files = []; - for (const project of readdirSync(dir)) { - const projectDir = join(dir, project); - try { - if (!statSync(projectDir).isDirectory()) continue; - for (const file of readdirSync(projectDir)) { - if (file.endsWith(".jsonl")) files.push(join(projectDir, file)); - } - } catch { - } - } - return files; -} -function stripPreamble(text) { - return text.replace(/^\s*[\s\S]*?<\/system_instruction>\s*/g, "").replace(/[\s\S]*?<\/ide_selection>/g, "").trim(); -} -async function readTranscript(path, out2) { - const project = path.split("/").slice(-2, -1)[0] ?? "unknown"; - const stream = createReadStream(path, { encoding: "utf8" }); - const lines = createInterface({ input: stream, crlfDelay: Infinity }); - try { - for await (const line of lines) { - if (!line.trim()) continue; - let record; - try { - record = JSON.parse(line); - } catch { - continue; - } - if (!isHumanPrompt(record)) continue; - const text = stripPreamble(textOf(record.message?.content)); - if (!text) continue; - out2.push({ - ts: record.timestamp ?? (/* @__PURE__ */ new Date(0)).toISOString(), - session: record.sessionId ?? path, - text, - project - }); - } - } finally { - lines.close(); - stream.destroy(); - } -} -async function readHistory(dir = PROJECTS_DIR) { - const prompts = []; - for (const file of listTranscripts(dir)) { - try { - await readTranscript(file, prompts); - } catch { - } - } - prompts.sort((a, b) => a.ts.localeCompare(b.ts)); - return prompts; -} - // src/report.ts function normalCdf(z) { const sign = z < 0 ? -1 : 1; @@ -232,7 +154,9 @@ function buildReport(input) { for (const entry of entries) { const record = scores.get(entry.hash); if (!record) continue; - const result = interpret(entry.hash, record.probabilities, record.gates); + const result = interpret(entry.hash, record.probabilities, record.gates, { + mode: record.conversation ? "conversation" : "alone" + }); const map = /* @__PURE__ */ new Map(); for (const check of result.checks) map.set(check.id, check.verdict); verdicts.set(entry.hash, map); @@ -732,6 +656,7 @@ async function cmdConfig(argv) { out(`Unknown setting: ${key}`); } async function cmdFixturesInit(argv) { + if (argv.includes("--conversations")) return cmdConversationFixturesInit(argv); const count = Number.parseInt(argv.find((a) => a.startsWith("--count="))?.split("=")[1] ?? "40", 10) || 40; const outPath = argv.find((a) => a.startsWith("--out="))?.split("=")[1] ?? "test/fixtures/prompts.json"; const config = loadConfig(); @@ -743,30 +668,82 @@ async function cmdFixturesInit(argv) { out(`Only ${unique.length} usable prompts in your history; need ${count}.`); return; } - unique.sort((a, b) => a.text.length - b.text.length); - const third = Math.floor(unique.length / 3); - const buckets = [unique.slice(0, third), unique.slice(third, 2 * third), unique.slice(2 * third)]; + const picked = sampleByLength(unique, (p) => p.text.length, count); + const fixtures = picked.map((p, i) => ({ + id: `p${String(i).padStart(2, "0")}`, + text: p.text, + labels: Object.fromEntries(CHECKS.map((c) => [c.id, null])), + gates: Object.fromEntries(GATES.map((g) => [g.id, null])) + })); + writeFileSync(outPath, `${JSON.stringify(fixtures, null, 1)} +`); + out(`Wrote ${fixtures.length} unlabelled fixtures to ${outPath}.`); + out(""); + out("Nothing was sent anywhere. Label them by hand from the criteria in"); + out("src/checks.ts before running `npm run eval` \u2014 see test/fixtures/README.md."); + out("Do not commit this file."); +} +function sampleByLength(items, length, count) { + const sorted = items.toSorted((a, b) => length(a) - length(b)); + const third = Math.floor(sorted.length / 3); + const buckets = [sorted.slice(0, third), sorted.slice(third, 2 * third), sorted.slice(2 * third)]; const perBucket = Math.ceil(count / 3); const picked = []; for (const bucket of buckets) { const step = Math.max(1, Math.floor(bucket.length / perBucket)); for (let i = 0; i < bucket.length && picked.length < count; i += step) { - const prompt = bucket[i]; - if (prompt) picked.push(prompt); + const item = bucket[i]; + if (item !== void 0) picked.push(item); } } - const fixtures = picked.slice(0, count).map((p, i) => ({ - id: `p${String(i).padStart(2, "0")}`, - text: p.text, - labels: Object.fromEntries(CHECKS.map((c) => [c.id, null])), + return picked.slice(0, count); +} +async function cmdConversationFixturesInit(argv) { + const count = Number.parseInt(argv.find((a) => a.startsWith("--count="))?.split("=")[1] ?? "40", 10) || 40; + const outPath = argv.find((a) => a.startsWith("--out="))?.split("=")[1] ?? "test/fixtures/conversations.json"; + const config = loadConfig(); + const privacy = config.privacy === "metadata_only" ? "redact" : config.privacy; + const redact = (text) => applyPrivacy(text, privacy).text ?? ""; + process.stderr.write("Reading Claude Code history\u2026\n"); + const sessions = await readConversations(config.bypassPrefix); + const candidates = /* @__PURE__ */ new Map(); + for (const exchanges of sessions) { + for (let i = 1; i < exchanges.length; i += 1) { + const current = exchanges[i]; + const previous = exchanges[i - 1]; + if (!current || !previous?.reply) continue; + if (skipReason(current.prompt, config.bypassPrefix) !== null) continue; + if (candidates.has(current.prompt)) continue; + candidates.set(current.prompt, { + text: current.prompt, + context: toTurns(exchanges.slice(Math.max(0, i - 2), i)) + }); + } + } + if (candidates.size < count) { + out(`Only ${candidates.size} usable follow-ups in your history; need ${count}.`); + return; + } + const picked = sampleByLength([...candidates.values()], (c) => c.text.length, count); + const fixtures = picked.map((c, i) => ({ + id: `c${String(i).padStart(2, "0")}`, + // Redacted, then clamped exactly as the scorer clamps, so the person + // labelling sees what Jev sees and no more. + context: c.context.map((turn) => ({ + role: turn.role, + text: turn.role === "agent" ? clampReply(redact(turn.text)) : clampPrompt(redact(turn.text)) + })), + text: clampPrompt(redact(c.text)), + labels: Object.fromEntries(CHECKS.map((check) => [check.id, null])), gates: Object.fromEntries(GATES.map((g) => [g.id, null])) })); writeFileSync(outPath, `${JSON.stringify(fixtures, null, 1)} `); - out(`Wrote ${fixtures.length} unlabelled fixtures to ${outPath}.`); + out(`Wrote ${fixtures.length} unlabelled conversation fixtures to ${outPath}.`); out(""); - out("Nothing was sent anywhere. Label them by hand from the criteria in"); - out("src/checks.ts before running `npm run eval` \u2014 see test/fixtures/README.md."); + out("Nothing was sent anywhere. Label the last message of each, read together with"); + out("its context, from the `conversation` criteria in src/checks.ts, before running"); + out("`npm run eval -- --conversations`. See test/fixtures/README.md."); out("Do not commit this file."); } function cmdStatus() { @@ -808,7 +785,9 @@ var run = async () => { case "status": return cmdStatus(); default: - out("usage: cli.js score | report [n] | config [...] | backfill [--confirm] | fixtures-init | status"); + out( + "usage: cli.js score | report [n] | config [...] | backfill [--confirm] | fixtures-init [--conversations] | status" + ); } }; run().catch((err) => { diff --git a/dist/eval.js b/dist/eval.js index 428462b..84f238b 100644 --- a/dist/eval.js +++ b/dist/eval.js @@ -5,10 +5,14 @@ import { MODEL, USD_PER_INPUT_TOKEN, scoreMany -} from "./chunk-W564PYU5.js"; +} from "./chunk-X33S6F2H.js"; import { - apiKey -} from "./chunk-7PP552KK.js"; + apiKey, + loadConfig +} from "./chunk-KYVNDBDC.js"; +import { + applyPrivacy +} from "./chunk-A7NLXWHN.js"; // src/eval.ts import { existsSync, readFileSync, writeFileSync } from "node:fs"; @@ -34,10 +38,24 @@ async function main() { process.stderr.write("TYPESAFE_API_KEY is not set. The eval calls Jev and cannot run without it.\n"); process.exit(1); } - const fixtures = JSON.parse(readFileSync("test/fixtures/prompts.json", "utf8")); + const conversations = process.argv.includes("--conversations"); + const fixturePath = conversations ? "test/fixtures/conversations.json" : "test/fixtures/prompts.json"; + const cachePath = conversations ? "test/eval-conversations-raw.json" : "test/eval-raw.json"; + const resultsPath = conversations ? "test/eval-conversations-results" : "test/eval-results"; + const thresholdOf = (def) => conversations ? def.conversation.threshold : def.threshold; + const fixtures = JSON.parse(readFileSync(fixturePath, "utf8")); + const unlabelled = fixtures.filter((f) => Object.values(f.labels).every((v) => v === null)); + if (unlabelled.length > 0) { + process.stderr.write( + `${unlabelled.length} fixtures in ${fixturePath} have no labels. Label them by hand first; see test/fixtures/README.md. +` + ); + process.exit(1); + } + const { privacy } = loadConfig(); + const redact = (text) => applyPrivacy(text, privacy === "metadata_only" ? "redact" : privacy).text ?? ""; let inputTokens = 0; let records; - const cachePath = "test/eval-raw.json"; if (process.argv.includes("--cached") && existsSync(cachePath)) { const cached = JSON.parse(readFileSync(cachePath, "utf8")); records = cached.records; @@ -48,7 +66,11 @@ async function main() { process.stderr.write(`Scoring ${fixtures.length} fixtures\u2026 `); records = await scoreMany( - fixtures.map((f) => ({ hash: f.id, text: f.text })), + fixtures.map((f) => ({ + hash: f.id, + text: redact(f.text), + ...f.context ? { conversation: f.context.map((t) => ({ role: t.role, text: redact(t.text) })) } : {} + })), { onUsage: (u) => { inputTokens += u.input_tokens; @@ -97,7 +119,7 @@ async function main() { const record = byId.get(fixture.id); const p = record?.probabilities[def.id]; if (p === void 0) continue; - rows.push({ truth, predicted: p >= def.threshold }); + rows.push({ truth, predicted: p >= thresholdOf(def) }); raw.push({ truth, p }); } const fail = metricsFor( @@ -105,7 +127,7 @@ async function main() { true ); const pass = metricsFor(rows, true); - let best = def.threshold; + let best = thresholdOf(def); if (tune) { let bestScore = -1; for (let t = 0.05; t <= 0.95; t += 0.05) { @@ -124,12 +146,12 @@ async function main() { const measurable = fail.support >= MIN_SUPPORT && fail.precision !== null; const clears = measurable && fail.precision >= TARGET_PRECISION; if (measurable && !clears) allClear = false; - const verdict = !measurable ? `too few fail cases (n=${fail.support}) \u2014 not measurable` : clears ? "ok" : `BELOW ${TARGET_PRECISION}`; + const verdict = !measurable ? fail.support < MIN_SUPPORT ? `too few fail cases (n=${fail.support}) \u2014 not measurable` : "never predicts fail at this threshold \u2014 not measurable" : clears ? "ok" : `BELOW ${TARGET_PRECISION}`; lines.push( - ` ${def.id.padEnd(20)} ${def.threshold.toFixed(2)} | ${fmt(fail.precision)} ${fmt(fail.recall)} ${String(fail.support).padStart(2)} | ${fmt(pass.precision)} ${fmt(pass.recall)} ${String(pass.support).padStart(2)} | ${verdict}${tune ? ` (best thr ${best})` : ""}` + ` ${def.id.padEnd(20)} ${thresholdOf(def).toFixed(2)} | ${fmt(fail.precision)} ${fmt(fail.recall)} ${String(fail.support).padStart(2)} | ${fmt(pass.precision)} ${fmt(pass.recall)} ${String(pass.support).padStart(2)} | ${verdict}${tune ? ` (best thr ${best})` : ""}` ); results[def.id] = { - threshold: def.threshold, + threshold: thresholdOf(def), fail: { precision: fail.precision, recall: fail.recall, support: fail.support }, pass: { precision: pass.precision, recall: pass.recall, support: pass.support }, measurable, @@ -148,7 +170,7 @@ async function main() { for (let fold = 0; fold < 5; fold += 1) { const test = labelled.filter((_, i) => i % 5 === fold); const train = labelled.filter((_, i) => i % 5 !== fold); - let thr = def.threshold; + let thr = thresholdOf(def); let bestScore = -1; for (let t = 0.05; t <= 0.95; t += 0.05) { const m = metricsFor( @@ -177,12 +199,13 @@ async function main() { const report = lines.join("\n"); process.stdout.write(report + "\n"); writeFileSync( - "test/eval-results.json", + `${resultsPath}.json`, JSON.stringify( { ranAt: (/* @__PURE__ */ new Date()).toISOString(), model: MODEL, fixtures: fixtures.length, + ...conversations ? { set: "conversations" } : {}, inputTokens, targetPrecision: TARGET_PRECISION, checks: results @@ -191,7 +214,7 @@ async function main() { 2 ) + "\n" ); - writeFileSync("test/eval-results.txt", report + "\n"); + writeFileSync(`${resultsPath}.txt`, report + "\n"); process.exit(allClear ? 0 : 1); } main().catch((err) => { diff --git a/dist/hook.js b/dist/hook.js index e98e591..ef2e818 100644 --- a/dist/hook.js +++ b/dist/hook.js @@ -1,17 +1,16 @@ // JevPromptCoach — generated by scripts/build.mjs. Do not edit. import { + appendLog, promptHash, + promptMatchKey, skipReason -} from "./chunk-F3M3WE3B.js"; -import { - appendLog -} from "./chunk-HXIO5BL2.js"; +} from "./chunk-33DTCFCS.js"; import { loadConfig -} from "./chunk-7PP552KK.js"; +} from "./chunk-KYVNDBDC.js"; import { applyPrivacy -} from "./chunk-VM4R2HGT.js"; +} from "./chunk-A7NLXWHN.js"; // src/hook.ts import { readFileSync } from "node:fs"; @@ -52,8 +51,13 @@ async function main() { } if (config.mode !== "always") return; if (stored === null) return; - const { runInline } = await import("./inline-STU2ZBEP.js"); - const line = await runInline(stored, hash, config, { session: entry.session, ts: entry.ts }); + const { runInline } = await import("./inline-EBGOCXX3.js"); + const line = await runInline(stored, hash, config, { + session: entry.session, + ts: entry.ts, + transcriptPath: input.transcript_path, + promptKey: promptMatchKey(text) + }); if (line) emitLine(line); } main().then( diff --git a/dist/inline-EBGOCXX3.js b/dist/inline-EBGOCXX3.js new file mode 100644 index 0000000..648aba2 --- /dev/null +++ b/dist/inline-EBGOCXX3.js @@ -0,0 +1,93 @@ +// JevPromptCoach — generated by scripts/build.mjs. Do not edit. +import { + clampReply, + recentTurns +} from "./chunk-EF7K2A5B.js"; +import { + appendScores, + readScores, + recentSessionPrompts +} from "./chunk-33DTCFCS.js"; +import { + CHECKS, + clampPrompt, + interpret, + scoreOne +} from "./chunk-X33S6F2H.js"; +import { + sessionContextEnabled, + sessionRepliesEnabled +} from "./chunk-KYVNDBDC.js"; +import { + applyPrivacy +} from "./chunk-A7NLXWHN.js"; + +// src/inline.ts +var CONTEXT_PROMPTS = 2; +var CONTEXT_EXCHANGES = 2; +function format(result) { + const failures = result.checks.filter((c) => c.verdict === "fail"); + if (failures.length === 0) return null; + const order = new Map(CHECKS.map((c, i) => [c.id, i])); + const worst = failures.toSorted((a, b) => (a.probability ?? 1) - (b.probability ?? 1)).slice(0, 2).toSorted((a, b) => (order.get(a.id) ?? 0) - (order.get(b.id) ?? 0)); + const missing = worst.map((c) => c.def.shortfall).join(", "); + const head = result.score ? `Jev (Prompt Coach) - ${result.score}/100` : "Jev (Prompt Coach)"; + return `${head} +Missing: ${missing}.`; +} +var NO_CONTEXT = { prompts: [], conversation: [] }; +function sessionContext(config, at) { + if (at.session === "unknown" || !sessionContextEnabled()) return NO_CONTEXT; + const safe = (text, clamp = clampPrompt) => { + const redacted = applyPrivacy(text, config.privacy).text; + return redacted === null ? null : clamp(redacted); + }; + if (sessionRepliesEnabled() && at.transcriptPath) { + try { + const conversation = recentTurns(at.transcriptPath, at.promptKey, CONTEXT_EXCHANGES, config.bypassPrefix).map((turn) => ({ role: turn.role, text: safe(turn.text, turn.role === "agent" ? clampReply : clampPrompt) })).filter((turn) => turn.text !== null); + return { prompts: [], conversation }; + } catch { + } + } + try { + const prompts = recentSessionPrompts(at.session, at.ts, CONTEXT_PROMPTS).map((entry) => entry.text === null ? null : safe(entry.text)).filter((text) => text !== null); + return { prompts, conversation: [] }; + } catch { + return NO_CONTEXT; + } +} +async function runInline(redactedText, hash, config, at) { + const context = sessionContext(config, at); + const hasContext = context.prompts.length > 0 || context.conversation.length > 0; + if (!hasContext) { + try { + const cached = readScores().get(hash); + if (cached && !cached.context) { + return format(interpret(hash, cached.probabilities, cached.gates, { inlineSafe: true })); + } + } catch { + } + } + const deadline = new Promise((resolve) => { + const timer = setTimeout(() => resolve(null), config.alwaysTimeoutMs); + timer.unref?.(); + }); + const scored = await Promise.race([ + scoreOne(redactedText, hash, { + timeoutMs: config.alwaysTimeoutMs, + inlineSafe: true, + context: context.prompts, + conversation: context.conversation + }), + deadline + ]); + if (!scored) return null; + try { + appendScores([scored.record]); + } catch { + } + return format(scored.result); +} +export { + runInline +}; diff --git a/dist/inline-STU2ZBEP.js b/dist/inline-STU2ZBEP.js deleted file mode 100644 index 8935730..0000000 --- a/dist/inline-STU2ZBEP.js +++ /dev/null @@ -1,67 +0,0 @@ -// JevPromptCoach — generated by scripts/build.mjs. Do not edit. -import { - appendScores, - readScores, - recentSessionPrompts -} from "./chunk-HXIO5BL2.js"; -import { - CHECKS, - interpret, - scoreOne -} from "./chunk-W564PYU5.js"; -import { - sessionContextEnabled -} from "./chunk-7PP552KK.js"; -import { - applyPrivacy -} from "./chunk-VM4R2HGT.js"; - -// src/inline.ts -var CONTEXT_PROMPTS = 2; -function format(result) { - const failures = result.checks.filter((c) => c.verdict === "fail"); - if (failures.length === 0) return null; - const order = new Map(CHECKS.map((c, i) => [c.id, i])); - const worst = failures.toSorted((a, b) => (a.probability ?? 1) - (b.probability ?? 1)).slice(0, 2).toSorted((a, b) => (order.get(a.id) ?? 0) - (order.get(b.id) ?? 0)); - const missing = worst.map((c) => c.def.shortfall).join(", "); - const head = result.score ? `Jev (Prompt Coach) - ${result.score}/100` : "Jev (Prompt Coach)"; - return `${head} -Missing: ${missing}.`; -} -function sessionContext(config, session, ts) { - if (session === "unknown" || !sessionContextEnabled()) return []; - try { - return recentSessionPrompts(session, ts, CONTEXT_PROMPTS).map((entry) => entry.text === null ? null : applyPrivacy(entry.text, config.privacy).text).filter((text) => text !== null); - } catch { - return []; - } -} -async function runInline(redactedText, hash, config, at) { - const context = sessionContext(config, at.session, at.ts); - if (context.length === 0) { - try { - const cached = readScores().get(hash); - if (cached && !cached.context) { - return format(interpret(hash, cached.probabilities, cached.gates, { inlineSafe: true })); - } - } catch { - } - } - const deadline = new Promise((resolve) => { - const timer = setTimeout(() => resolve(null), config.alwaysTimeoutMs); - timer.unref?.(); - }); - const scored = await Promise.race([ - scoreOne(redactedText, hash, { timeoutMs: config.alwaysTimeoutMs, inlineSafe: true, context }), - deadline - ]); - if (!scored) return null; - try { - appendScores([scored.record]); - } catch { - } - return format(scored.result); -} -export { - runInline -}; diff --git a/dist/redact.js b/dist/redact.js index 2820220..eee8a78 100644 --- a/dist/redact.js +++ b/dist/redact.js @@ -4,7 +4,7 @@ import { features, redact, stripCredentials -} from "./chunk-VM4R2HGT.js"; +} from "./chunk-A7NLXWHN.js"; export { applyPrivacy, features, diff --git a/package.json b/package.json index 1529f4c..6082617 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "jevpromptcoach", - "version": "0.3.0", + "version": "0.3.5", "description": "Scores how well you prompt a coding agent, and shows your habits improving over time. Runs on TypeSafe Jev.", "license": "MIT", "private": false, diff --git a/scripts/check-leaks.mjs b/scripts/check-leaks.mjs index 94bd3c7..0255cec 100755 --- a/scripts/check-leaks.mjs +++ b/scripts/check-leaks.mjs @@ -26,6 +26,11 @@ const FORBIDDEN_PATHS = [ { re: /(^|\/)\.env(\.|$)/, why: 'environment file — may hold a TypeSafe API key' }, { re: /^test\/fixtures\/prompts\.json$/, why: 'eval fixtures — real prompts from real work' }, { re: /^test\/eval-raw\.json$/, why: 'raw eval output — derived from private fixtures' }, + { + re: /^test\/fixtures\/conversations\.json$/, + why: 'conversation fixtures — real prompts and agent replies from real work', + }, + { re: /^test\/eval-conversations-raw\.json$/, why: 'raw eval output — derived from private fixtures' }, { re: /\.jsonl$/, why: 'JSONL log — the prompt log is exactly this shape' }, { re: /(^|\/)corrections\.json$/, why: 'correction records — derived from prompt pairs' }, { re: /(^|\/)prompts\.jsonl$/, why: 'the local prompt log' }, @@ -42,6 +47,14 @@ const SECRETS = [ { name: 'AWS access key id', re: /\b(?:AKIA|ASIA)[A-Z0-9]{16}\b/ }, { name: 'Google API key', re: /\bAIza[A-Za-z0-9_-]{30,}/ }, { name: 'Slack token', re: /\bxox[abprs]-[A-Za-z0-9-]{16,}/ }, + { + name: 'Slack or Discord webhook', + re: /https:\/\/(?:hooks\.slack\.com\/services|discord(?:app)?\.com\/api\/webhooks)\/\S{16,}/, + }, + { name: 'Stripe key', re: /\b[sr]k_(?:live|test)_[A-Za-z0-9]{16,}/ }, + { name: 'npm token', re: /\bnpm_[A-Za-z0-9]{30,}/ }, + { name: 'GitHub fine-grained token', re: /\bgithub_pat_[A-Za-z0-9_]{22,}/ }, + { name: 'SendGrid key', re: /\bSG\.[A-Za-z0-9_-]{16,}\.[A-Za-z0-9_-]{16,}/ }, { name: 'TypeSafe API key', re: /\bapikey_[A-Za-z0-9]{16,}/ }, { name: 'private key block', re: /-----BEGIN (?:[A-Z ]+ )?PRIVATE KEY-----/ }, { name: 'JSON Web Token', re: /\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}/ }, @@ -147,7 +160,13 @@ for (const path of trackedFiles()) { } // The ignore rules are themselves part of the guarantee, so verify they hold. -for (const mustIgnore of ['test/fixtures/prompts.json', 'test/eval-raw.json', '.env']) { +for (const mustIgnore of [ + 'test/fixtures/prompts.json', + 'test/fixtures/conversations.json', + 'test/eval-raw.json', + 'test/eval-conversations-raw.json', + '.env', +]) { try { git(['check-ignore', '-q', mustIgnore]); } catch { diff --git a/src/checks.ts b/src/checks.ts index 495538b..b8e4fc0 100644 --- a/src/checks.ts +++ b/src/checks.ts @@ -61,6 +61,21 @@ export interface CheckDef { * asked — and nowhere else. */ inlineEligible: boolean; + /** + * The same habit judged on a follow-up, read together with the conversation + * before it, the agent's replies included. Used only when replies are sent + * (JEVPROMPTCOACH_SESSION_REPLIES). Something counts as present if the + * follow-up supplies it, or if the shown conversation already established it + * and the follow-up relies on it, for instance by accepting what the agent + * proposed. The threshold and inline eligibility come from the conversation + * eval (`npm run eval -- --conversations`), separately from the standalone ones. + */ + conversation: { + instructions: string; + criteria: { true: string; false: string }; + threshold: number; + inlineEligible: boolean; + }; /** Why it matters, shown by /jevpromptcoach:score when the check fails. */ cause: string; consequence: string; @@ -112,6 +127,18 @@ export const CHECKS: CheckDef[] = [ inlineMargin: 0.2, // CV fail-precision 0.96 over 28 failing examples. inlineEligible: true, + conversation: { + instructions: + 'The agent knows exactly where to work: the message names a concrete file, path, function, class, component, endpoint, or symbol, or points unambiguously at one already named in the conversation.', + criteria: { + true: 'Names a concrete target, or refers without ambiguity to one named earlier by either side: "the email one" after the agent listed email-worker.ts among other files, or "yes, go ahead" accepting a change the agent described in a named file.', + false: + 'The target is unclear even with the conversation: nothing concrete was named earlier, or several candidates were named and the message does not say which, or the message starts new work described only in general terms.', + }, + threshold: 0.15, + // CV fail-precision 0.60 over 11 failing examples: ranks better than standalone (AUC 0.82 vs 0.63) but too thin to stand behind inline. + inlineEligible: false, + }, cause: 'You wrote "it" or "the code" instead of a name.', consequence: 'The agent has to guess which file you meant. It searches, or it edits the wrong one.', fix: 'Name the file, function, or symbol you want changed.', @@ -130,6 +157,18 @@ export const CHECKS: CheckDef[] = [ inlineMargin: 0.2, // CV fail-precision 1.00 over 14 failing examples. inlineEligible: true, + conversation: { + instructions: + 'What should be true once the request in the message is done is known: the message states the outcome or the output it wants, or it approves a proposal from the agent that describes the resulting behaviour.', + criteria: { + true: 'States the intended end state or the output wanted ("so the page shows X", "give me five titles"), or approves a specific proposal in which the agent described what the result will be.', + false: + 'Only names a step to carry out (commit, push, deploy, merge, review, run something) without an outcome, even when the conversation describes the work around it; or starts new work without an end state.', + }, + threshold: 0.2, + // CV fail-precision 0.75 over 18 failing examples, below the 0.90 bar. + inlineEligible: false, + }, cause: 'Nothing in the request says what should be true at the end.', consequence: 'The agent picks its own finish line, and stops somewhere you did not want.', fix: 'Add one sentence: what should be true when this works.', @@ -149,6 +188,18 @@ export const CHECKS: CheckDef[] = [ inlineMargin: 0.2, // CV fail-precision 1.00 over 10 failing examples. inlineEligible: true, + conversation: { + instructions: + 'The request, read with the conversation, is one contained, well-defined piece of work rather than an open-ended or sweeping change.', + criteria: { + true: 'Asks for one change or a small set of clearly enumerated changes, including accepting one specific proposal the agent described.', + false: + 'Asks for something open-ended or sweeping, bundles several unrelated requests, or accepts a broad proposal ("do all of it") whose edges the conversation never set.', + }, + threshold: 0.45, + // Not measurable: 2 failing examples in the conversation set. + inlineEligible: false, + }, cause: 'The request has no edges, so the agent decides how far to go.', consequence: 'You get a huge change touching files you never meant to touch, and reviewing it takes longer than the fix would have.', @@ -166,8 +217,19 @@ export const CHECKS: CheckDef[] = [ }, threshold: 0.3, inlineMargin: 0.2, - // CV fail-precision 0.97 over 30 failing examples. + // CV fail-precision 1.00 over 30 failing examples (0.97 before the eval redacted). inlineEligible: true, + conversation: { + instructions: + 'A limit on the work is in force: the message states one, or one stated earlier in the conversation still applies to what the message asks for.', + criteria: { + true: 'Names something to leave alone or preserve, or forbids an approach, in the message or earlier in the conversation for this same work, and nothing has withdrawn it.', + false: 'No limit that applies to the requested work appears in the message or anywhere in the conversation.', + }, + threshold: 0.8, + // CV fail-precision 0.97 over 34 failing examples. 34 of 40 fixtures fail it, so precision is flattered by the base rate; AUC 0.99. + inlineEligible: true, + }, cause: 'Nothing in the request is marked off-limits.', consequence: 'Something that was working gets rewritten along the way.', fix: 'Say what must stay as it is — the API, the database schema, the other callers.', @@ -188,6 +250,18 @@ export const CHECKS: CheckDef[] = [ inlineMargin: 0.2, // only 5 failing examples in the fixture set — too thin to stand behind inline. inlineEligible: false, + conversation: { + instructions: + 'For the failure being reported, the actual evidence is available: in the message, or earlier in the conversation, quoted by the developer or reported by the agent from running something.', + criteria: { + true: 'Real output (an error message, stack trace, failing assertion, or log) or a specific expected-versus-actual pair appears in the message or the conversation for this failure.', + false: + 'The failure is described only in general terms, and no actual output or concrete expected-versus-actual pair for it appears anywhere in the conversation.', + }, + threshold: 0.4, + // Not measurable: 1 failing example in the conversation set. + inlineEligible: false, + }, cause: 'The bug is described, but the actual error text is not in the message.', consequence: 'The agent guesses the error from your description and fixes a different problem.', fix: 'Paste the real error, and say what you expected to happen instead.', @@ -205,8 +279,20 @@ export const CHECKS: CheckDef[] = [ appliesWhen: { gate: 'is_large_change', minProbability: 0.5 }, threshold: 0.35, inlineMargin: 0.2, - // CV fail-precision 0.86, below the 0.90 bar. + // 8 failing examples, under the 10 the inline bar asks for (CV fail-precision + // 1.00 on the redacted run, 0.86 before it). inlineEligible: false, + conversation: { + instructions: + 'Before a large or risky change is carried out, a plan has been asked for or seen: the message asks for one, or it approves a plan the agent laid out in the conversation.', + criteria: { + true: 'Asks to plan, propose, outline, or investigate first, or approves a specific plan the agent already described for this change.', + false: 'Asks for the large change to be carried out directly, and no plan for it appears in the conversation.', + }, + threshold: 0.35, + // Not measurable: 4 failing examples in the conversation set. + inlineEligible: false, + }, cause: 'You asked for a big or risky change without asking to see the approach first.', consequence: 'You find out how it was going to be done only after it has been done.', fix: 'Ask for the plan first, then approve it. "Plan this before changing anything."', @@ -225,6 +311,18 @@ export const CHECKS: CheckDef[] = [ inlineMargin: 0.2, // CV fail-precision 1.00 over 39 failing examples. inlineEligible: true, + conversation: { + instructions: + 'How the work will be checked is known: the message names a test, command, or check, or accepts a proposal from the agent that names one, or continues work whose verification was stated earlier.', + criteria: { + true: 'A runnable or checkable step for this work (a test file or name, a command, a script, a page to load) appears in the message, or in an earlier message or agent proposal the message accepts or continues.', + false: + 'No test, command, or check for this work appears in the message or the conversation, or the message says only "make sure it works".', + }, + threshold: 0.75, + // CV fail-precision 0.97 over 38 failing examples. 38 of 40 fixtures fail it, so precision is flattered by the base rate; AUC 0.89. + inlineEligible: true, + }, cause: 'Nothing in the request says how to tell whether it worked.', consequence: 'The agent says it worked, and you find out later that it did not.', fix: 'Name the command. "Verify with npm test -- auth.spec.ts."', diff --git a/src/cli.ts b/src/cli.ts index 9357ae0..9add73d 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -5,6 +5,7 @@ import { readFileSync, writeFileSync } from 'node:fs'; import { CHECKS, GATES } from './checks.js'; import { apiKey, apiKeySource, ENV_PATH, LOG_PATH, loadConfig, saveConfig } from './config.js'; +import { clampReply, readConversations, toTurns } from './conversation.js'; import { buildPairs, detectCorrections, estimateCorrectionTokens } from './correction.js'; import { promptHash } from './hash.js'; import { readHistory } from './history.js'; @@ -24,7 +25,7 @@ import { import { applyPrivacy } from './redact.js'; import { buildReport } from './report.js'; import { renderReport, renderScore } from './render.js'; -import { estimateScoringTokens, interpret, scoreMany, scoreOne } from './score.js'; +import { clampPrompt, estimateScoringTokens, interpret, scoreMany, scoreOne } from './score.js'; import { skipReason } from './skip.js'; const out = (s: string): void => { @@ -375,6 +376,7 @@ async function cmdConfig(argv: string[]): Promise { * they have to be set by hand, from the criteria, before the eval is run. */ async function cmdFixturesInit(argv: string[]): Promise { + if (argv.includes('--conversations')) return cmdConversationFixturesInit(argv); const count = Number.parseInt(argv.find((a) => a.startsWith('--count='))?.split('=')[1] ?? '40', 10) || 40; const outPath = argv.find((a) => a.startsWith('--out='))?.split('=')[1] ?? 'test/fixtures/prompts.json'; const config = loadConfig(); @@ -389,21 +391,9 @@ async function cmdFixturesInit(argv: string[]): Promise { return; } - // Stratify by length so short, medium and long prompts are all represented. - unique.sort((a, b) => a.text.length - b.text.length); - const third = Math.floor(unique.length / 3); - const buckets = [unique.slice(0, third), unique.slice(third, 2 * third), unique.slice(2 * third)]; - const perBucket = Math.ceil(count / 3); - const picked: typeof unique = []; - for (const bucket of buckets) { - const step = Math.max(1, Math.floor(bucket.length / perBucket)); - for (let i = 0; i < bucket.length && picked.length < count; i += step) { - const prompt = bucket[i]; - if (prompt) picked.push(prompt); - } - } + const picked = sampleByLength(unique, (p) => p.text.length, count); - const fixtures = picked.slice(0, count).map((p, i) => ({ + const fixtures = picked.map((p, i) => ({ id: `p${String(i).padStart(2, '0')}`, text: p.text, labels: Object.fromEntries(CHECKS.map((c) => [c.id, null])), @@ -418,6 +408,85 @@ async function cmdFixturesInit(argv: string[]): Promise { out('Do not commit this file.'); } +/** Stratify by length so short, medium and long items are all represented. */ +function sampleByLength(items: T[], length: (item: T) => number, count: number): T[] { + const sorted = items.toSorted((a, b) => length(a) - length(b)); + const third = Math.floor(sorted.length / 3); + const buckets = [sorted.slice(0, third), sorted.slice(third, 2 * third), sorted.slice(2 * third)]; + const perBucket = Math.ceil(count / 3); + const picked: T[] = []; + for (const bucket of buckets) { + const step = Math.max(1, Math.floor(bucket.length / perBucket)); + for (let i = 0; i < bucket.length && picked.length < count; i += step) { + const item = bucket[i]; + if (item !== undefined) picked.push(item); + } + } + return picked.slice(0, count); +} + +/** + * Build unlabelled conversation fixtures: follow-up prompts, each with the two + * exchanges before it, the agent's replies included. Entirely local, like the + * prompt fixtures. Every piece is redacted as the live path would redact it + * before sending, so what the eval later sends is what `always` mode would. + */ +async function cmdConversationFixturesInit(argv: string[]): Promise { + const count = Number.parseInt(argv.find((a) => a.startsWith('--count='))?.split('=')[1] ?? '40', 10) || 40; + const outPath = argv.find((a) => a.startsWith('--out='))?.split('=')[1] ?? 'test/fixtures/conversations.json'; + const config = loadConfig(); + const privacy = config.privacy === 'metadata_only' ? 'redact' : config.privacy; + const redact = (text: string): string => applyPrivacy(text, privacy).text ?? ''; + + process.stderr.write('Reading Claude Code history…\n'); + const sessions = await readConversations(config.bypassPrefix); + + // A follow-up is any scorable prompt after the first in its session whose + // previous turn ended with the agent saying something: that reply is what + // this fixture set exists to measure. + const candidates = new Map }>(); + for (const exchanges of sessions) { + for (let i = 1; i < exchanges.length; i += 1) { + const current = exchanges[i]; + const previous = exchanges[i - 1]; + if (!current || !previous?.reply) continue; + if (skipReason(current.prompt, config.bypassPrefix) !== null) continue; + if (candidates.has(current.prompt)) continue; + candidates.set(current.prompt, { + text: current.prompt, + context: toTurns(exchanges.slice(Math.max(0, i - 2), i)), + }); + } + } + + if (candidates.size < count) { + out(`Only ${candidates.size} usable follow-ups in your history; need ${count}.`); + return; + } + + const picked = sampleByLength([...candidates.values()], (c) => c.text.length, count); + const fixtures = picked.map((c, i) => ({ + id: `c${String(i).padStart(2, '0')}`, + // Redacted, then clamped exactly as the scorer clamps, so the person + // labelling sees what Jev sees and no more. + context: c.context.map((turn) => ({ + role: turn.role, + text: turn.role === 'agent' ? clampReply(redact(turn.text)) : clampPrompt(redact(turn.text)), + })), + text: clampPrompt(redact(c.text)), + labels: Object.fromEntries(CHECKS.map((check) => [check.id, null])), + gates: Object.fromEntries(GATES.map((g) => [g.id, null])), + })); + + writeFileSync(outPath, `${JSON.stringify(fixtures, null, 1)}\n`); + out(`Wrote ${fixtures.length} unlabelled conversation fixtures to ${outPath}.`); + out(''); + out('Nothing was sent anywhere. Label the last message of each, read together with'); + out('its context, from the `conversation` criteria in src/checks.ts, before running'); + out('`npm run eval -- --conversations`. See test/fixtures/README.md.'); + out('Do not commit this file.'); +} + // ---------------------------------------------------------------- status function cmdStatus(): void { @@ -466,7 +535,9 @@ const run = async (): Promise => { case 'status': return cmdStatus(); default: - out('usage: cli.js score | report [n] | config [...] | backfill [--confirm] | fixtures-init | status'); + out( + 'usage: cli.js score | report [n] | config [...] | backfill [--confirm] | fixtures-init [--conversations] | status', + ); } }; diff --git a/src/config.ts b/src/config.ts index 06fe219..36ac7ad 100644 --- a/src/config.ts +++ b/src/config.ts @@ -95,6 +95,18 @@ export function sessionContextEnabled(): boolean { return value === null || !/^(0|false|off|no)$/i.test(value); } +/** + * Whether follow-ups are scored with the agent's replies as well as the + * developer's earlier prompts. On unless JEVPROMPTCOACH_SESSION_REPLIES is 0, + * false, off or no. Replies go through the same redaction as prompts, which is + * the argument for the default: a secret is as likely to be pasted into a + * prompt as quoted back in a reply. Needs session context to be on as well. + */ +export function sessionRepliesEnabled(): boolean { + const value = envValue('JEVPROMPTCOACH_SESSION_REPLIES'); + return value === null || !/^(0|false|off|no)$/i.test(value); +} + /** * Where the key was found, for reporting. Never returns the key itself. * `null` means no key is available and nothing can be scored. diff --git a/src/conversation.ts b/src/conversation.ts new file mode 100644 index 0000000..2a31d1c --- /dev/null +++ b/src/conversation.ts @@ -0,0 +1,239 @@ +/** + * Conversations: the developer's prompts with the agent's replies between + * them, read from Claude Code's own transcript. + * + * Only the agent's visible text survives, and only what it wrote after its + * last tool call in the turn: the answer it gave or the question it ended on. + * Tool calls, tool output, thinking and subagent traffic never leave this + * module. An exchange whose prompt was bypassed, or was not typed by the + * developer at all, is dropped together with its reply, because the reply can + * repeat whatever the bypassed prompt held. + */ +import { closeSync, createReadStream, fstatSync, openSync, readSync } from 'node:fs'; +import { createInterface } from 'node:readline'; +import { promptMatchKey, stripPreamble } from './hash.js'; +import { isHumanPrompt, listTranscripts, PROJECTS_DIR, textOf, type TranscriptRecord } from './history.js'; +import { type SkipReason, skipReason } from './skip.js'; + +export interface Exchange { + /** The developer's prompt, attachment preamble removed. */ + prompt: string; + /** promptMatchKey of the prompt, to recognise the one being scored. */ + key: string; + /** The agent's closing text for the turn; empty if it wrote none. */ + reply: string; + ts: string; + session: string; +} + +export interface Turn { + role: 'developer' | 'agent'; + text: string; +} + +/** The end of a reply is where its answer or its question is. */ +const MAX_REPLY_CHARS = 1_500; + +/** Transcript lines carry whole tool outputs; 1 MB still covers the last few turns. */ +const TAIL_BYTES = 1024 * 1024; + +/** Prompts that are not the developer talking, or that must not be shown to anyone. */ +const EXCLUDED: ReadonlySet = new Set([ + 'bypass_prefix', + 'slash_command', + 'command_wrapper', + 'session_meta', +]); + +/** + * Keep the end of a reply. Only ever applied to text that has already been + * redacted: a cut made first could split a credential into a fragment that no + * redaction rule recognises. + */ +export function clampReply(text: string): string { + return text.length <= MAX_REPLY_CHARS ? text : `…\n${text.slice(-(MAX_REPLY_CHARS - 2))}`; +} + +function blockType(b: unknown): string | undefined { + return typeof b === 'object' && b !== null ? (b as { type?: string }).type : undefined; +} + +/** + * A prompt the developer typed while the agent was still working. Not a + * human prompt for history (src/history.ts), but in a conversation it is one: + * the replies after it answer it, and the bypass rules must apply to it. + */ +function isQueuedPrompt(record: TranscriptRecord): boolean { + return ( + record.type === 'user' && + record.promptSource === 'queued' && + !record.isMeta && + !(Array.isArray(record.message?.content) && record.message.content.some((b) => blockType(b) === 'tool_result')) + ); +} + +/** + * A user-role message that nobody typed: Claude Code's system notices and SDK + * input. What the agent writes after one is not a reply to the developer's + * prompt, so it closes that exchange rather than extending its reply. + */ +function isOtherSender(record: TranscriptRecord): boolean { + return ( + record.type === 'user' && + record.promptSource !== undefined && + // A meta record from the system or an SDK is still a message nobody typed. + (!record.isMeta || record.promptSource === 'system' || record.promptSource === 'sdk') && + !(Array.isArray(record.message?.content) && record.message.content.some((b) => blockType(b) === 'tool_result')) + ); +} + +/** Feed transcript records in order; read `exchanges` after `close()`. */ +class ExchangeBuilder { + readonly exchanges: Exchange[] = []; + /** + * `excluded` says why an exchange is dropped: 'prompt' when its prompt must + * not be shown (bypassed, a slash command, not typed), 'sender' when it is + * only a boundary after a message nobody typed. + */ + private current: (Exchange & { excluded: 'prompt' | 'sender' | null }) | null = null; + private parts: string[] = []; + + private readonly bypassPrefix: string; + + constructor(bypassPrefix: string) { + this.bypassPrefix = bypassPrefix; + } + + add(record: TranscriptRecord): void { + if (record.isSidechain) return; + if (isHumanPrompt(record) || isQueuedPrompt(record)) { + // A prompt queued while the agent works on a prompt that must not be + // shown inherits that exclusion: the text that follows still answers it. + // A boundary left by a system or SDK message hides nothing, so a queued + // prompt after one stays its own exchange. + const inheritsExclusion = isQueuedPrompt(record) && this.current?.excluded === 'prompt'; + this.close(); + const raw = textOf(record.message?.content); + const prompt = stripPreamble(raw); + const reason = skipReason(prompt || raw, this.bypassPrefix); + this.current = { + prompt, + key: promptMatchKey(raw), + reply: '', + ts: record.timestamp ?? '', + session: record.sessionId ?? '', + excluded: inheritsExclusion || !prompt || (reason !== null && EXCLUDED.has(reason)) ? 'prompt' : null, + }; + return; + } + if (isOtherSender(record)) { + this.close(); + this.current = { prompt: '', key: '', reply: '', ts: '', session: '', excluded: 'sender' }; + return; + } + if (record.type !== 'assistant' || !this.current) return; + let content = record.message?.content; + // Text written before a tool call is narration ("let me check"); only what + // follows the last one is the reply the developer answered. That holds + // inside a record too, should one carry text and a tool call together. + if (Array.isArray(content)) { + // tool_use, server_tool_use (web search), mcp_tool_use and the like. + const lastToolUse = content.findLastIndex((b) => blockType(b)?.endsWith('tool_use') === true); + if (lastToolUse >= 0) { + this.parts = []; + content = content.slice(lastToolUse + 1); + } + } + const text = textOf(content).trim(); + if (text) this.parts.push(text); + } + + close(): void { + if (this.current && !this.current.excluded) { + const { prompt, key, ts, session } = this.current; + // Unclamped: the caller redacts first and clamps after (clampReply). + this.exchanges.push({ prompt, key, ts, session, reply: this.parts.join('\n\n') }); + } + this.current = null; + this.parts = []; + } +} + +function readTail(path: string): string { + const fd = openSync(path, 'r'); + try { + const size = fstatSync(fd).size; + const length = Math.min(size, TAIL_BYTES); + const buffer = Buffer.alloc(length); + readSync(fd, buffer, 0, length, size - length); + return buffer.toString('utf8'); + } finally { + closeSync(fd); + } +} + +function parseLine(line: string): TranscriptRecord | null { + if (!line.trim()) return null; + try { + return JSON.parse(line) as TranscriptRecord; + } catch { + return null; + } +} + +/** + * The last `count` exchanges before the prompt being scored, flattened into + * turns, oldest first. Reads only the tail of the transcript. The prompt being + * scored may or may not be in the transcript yet when the hook runs; it is + * recognised by its match key and left out either way. An earlier identical + * prompt that already has a reply is real context and stays. + */ +export function recentTurns(transcriptPath: string, currentKey: string, count: number, bypassPrefix: string): Turn[] { + const builder = new ExchangeBuilder(bypassPrefix); + // The first line may be cut by the tail boundary; it fails to parse and is skipped. + for (const line of readTail(transcriptPath).split('\n')) { + const record = parseLine(line); + if (record) builder.add(record); + } + builder.close(); + + const exchanges = builder.exchanges; + const last = exchanges.at(-1); + if (last && !last.reply && last.key === currentKey) exchanges.pop(); + return toTurns(exchanges.slice(-count)); +} + +export function toTurns(exchanges: Exchange[]): Turn[] { + return exchanges.flatMap((e): Turn[] => + e.reply + ? [ + { role: 'developer', text: e.prompt }, + { role: 'agent', text: e.reply }, + ] + : [{ role: 'developer', text: e.prompt }], + ); +} + +/** Every session in the local history as its list of exchanges. Local only. */ +export async function readConversations(bypassPrefix: string, dir = PROJECTS_DIR): Promise { + const sessions: Exchange[][] = []; + for (const file of listTranscripts(dir)) { + const builder = new ExchangeBuilder(bypassPrefix); + const stream = createReadStream(file, { encoding: 'utf8' }); + const lines = createInterface({ input: stream, crlfDelay: Infinity }); + try { + for await (const line of lines) { + const record = parseLine(line); + if (record) builder.add(record); + } + } catch { + continue; + } finally { + lines.close(); + stream.destroy(); + } + builder.close(); + if (builder.exchanges.length > 1) sessions.push(builder.exchanges); + } + return sessions; +} diff --git a/src/eval.ts b/src/eval.ts index 3b72004..9f7c1bc 100644 --- a/src/eval.ts +++ b/src/eval.ts @@ -2,7 +2,10 @@ * Eval harness. `npm run eval`. * * Runs the real scoring path over test/fixtures/prompts.json and compares - * against hand labels. + * against hand labels. With `--conversations` it runs over + * test/fixtures/conversations.json instead: follow-ups judged together with the + * conversation before them, against each check's conversation criteria and + * thresholds, with results written beside the standalone ones. * * The headline metric is fail-precision: of the prompts where the plugin says * a habit is MISSING, how many really were. That is the number that matters, @@ -11,15 +14,19 @@ * reported too, but it is not what gates the inline line. */ import { existsSync, readFileSync, writeFileSync } from 'node:fs'; -import { CHECKS, type CheckId, GATES } from './checks.js'; -import { apiKey } from './config.js'; +import { type CheckDef, CHECKS, type CheckId, GATES } from './checks.js'; +import { apiKey, loadConfig } from './config.js'; +import type { Turn } from './conversation.js'; import { MODEL, USD_PER_INPUT_TOKEN } from './jev.js'; import type { ScoreRecord } from './log.js'; +import { applyPrivacy } from './redact.js'; import { scoreMany } from './score.js'; interface Fixture { id: string; text: string; + /** Conversation fixtures only: what came before `text`. */ + context?: Turn[]; labels: Record; gates: Record; } @@ -62,11 +69,30 @@ async function main(): Promise { process.exit(1); } - const fixtures = JSON.parse(readFileSync('test/fixtures/prompts.json', 'utf8')) as Fixture[]; + const conversations = process.argv.includes('--conversations'); + const fixturePath = conversations ? 'test/fixtures/conversations.json' : 'test/fixtures/prompts.json'; + const cachePath = conversations ? 'test/eval-conversations-raw.json' : 'test/eval-raw.json'; + const resultsPath = conversations ? 'test/eval-conversations-results' : 'test/eval-results'; + const thresholdOf = (def: CheckDef): number => (conversations ? def.conversation.threshold : def.threshold); + + const fixtures = JSON.parse(readFileSync(fixturePath, 'utf8')) as Fixture[]; + const unlabelled = fixtures.filter((f) => Object.values(f.labels).every((v) => v === null)); + if (unlabelled.length > 0) { + process.stderr.write( + `${unlabelled.length} fixtures in ${fixturePath} have no labels. Label them by hand first; see test/fixtures/README.md.\n`, + ); + process.exit(1); + } + + // The eval sends what the plugin would send: every text through the + // configured privacy level, as `always` mode and the commands do. Under + // metadata_only nothing is ever scored, so the eval uses `redact` instead. + const { privacy } = loadConfig(); + const redact = (text: string): string => + applyPrivacy(text, privacy === 'metadata_only' ? 'redact' : privacy).text ?? ''; let inputTokens = 0; let records: ScoreRecord[]; - const cachePath = 'test/eval-raw.json'; if (process.argv.includes('--cached') && existsSync(cachePath)) { const cached = JSON.parse(readFileSync(cachePath, 'utf8')) as { inputTokens: number; records: ScoreRecord[] }; records = cached.records; @@ -75,7 +101,11 @@ async function main(): Promise { } else { process.stderr.write(`Scoring ${fixtures.length} fixtures…\n`); records = await scoreMany( - fixtures.map((f) => ({ hash: f.id, text: f.text })), + fixtures.map((f) => ({ + hash: f.id, + text: redact(f.text), + ...(f.context ? { conversation: f.context.map((t) => ({ role: t.role, text: redact(t.text) })) } : {}), + })), { onUsage: (u) => { inputTokens += u.input_tokens; @@ -127,7 +157,7 @@ async function main(): Promise { const record = byId.get(fixture.id); const p = record?.probabilities[def.id]; if (p === undefined) continue; - rows.push({ truth, predicted: p >= def.threshold }); + rows.push({ truth, predicted: p >= thresholdOf(def) }); raw.push({ truth, p }); } @@ -137,7 +167,7 @@ async function main(): Promise { ); const pass = metricsFor(rows, true); - let best = def.threshold; + let best = thresholdOf(def); if (tune) { let bestScore = -1; for (let t = 0.05; t <= 0.95; t += 0.05) { @@ -160,17 +190,19 @@ async function main(): Promise { const clears = measurable && fail.precision! >= TARGET_PRECISION; if (measurable && !clears) allClear = false; const verdict = !measurable - ? `too few fail cases (n=${fail.support}) — not measurable` + ? fail.support < MIN_SUPPORT + ? `too few fail cases (n=${fail.support}) — not measurable` + : 'never predicts fail at this threshold — not measurable' : clears ? 'ok' : `BELOW ${TARGET_PRECISION}`; lines.push( - ` ${def.id.padEnd(20)} ${def.threshold.toFixed(2)} | ${fmt(fail.precision)} ${fmt(fail.recall)} ${String(fail.support).padStart(2)} | ${fmt(pass.precision)} ${fmt(pass.recall)} ${String(pass.support).padStart(2)} | ${verdict}${tune ? ` (best thr ${best})` : ''}`, + ` ${def.id.padEnd(20)} ${thresholdOf(def).toFixed(2)} | ${fmt(fail.precision)} ${fmt(fail.recall)} ${String(fail.support).padStart(2)} | ${fmt(pass.precision)} ${fmt(pass.recall)} ${String(pass.support).padStart(2)} | ${verdict}${tune ? ` (best thr ${best})` : ''}`, ); results[def.id] = { - threshold: def.threshold, + threshold: thresholdOf(def), fail: { precision: fail.precision, recall: fail.recall, support: fail.support }, pass: { precision: pass.precision, recall: pass.recall, support: pass.support }, measurable, @@ -197,7 +229,7 @@ async function main(): Promise { for (let fold = 0; fold < 5; fold += 1) { const test = labelled.filter((_, i) => i % 5 === fold); const train = labelled.filter((_, i) => i % 5 !== fold); - let thr = def.threshold; + let thr = thresholdOf(def); let bestScore = -1; for (let t = 0.05; t <= 0.95; t += 0.05) { const m = metricsFor( @@ -231,12 +263,13 @@ async function main(): Promise { process.stdout.write(report + '\n'); writeFileSync( - 'test/eval-results.json', + `${resultsPath}.json`, JSON.stringify( { ranAt: new Date().toISOString(), model: MODEL, fixtures: fixtures.length, + ...(conversations ? { set: 'conversations' } : {}), inputTokens, targetPrecision: TARGET_PRECISION, checks: results, @@ -245,7 +278,7 @@ async function main(): Promise { 2, ) + '\n', ); - writeFileSync('test/eval-results.txt', report + '\n'); + writeFileSync(`${resultsPath}.txt`, report + '\n'); process.exit(allClear ? 0 : 1); } diff --git a/src/hash.ts b/src/hash.ts index 22177e7..e63d1d0 100644 --- a/src/hash.ts +++ b/src/hash.ts @@ -4,3 +4,21 @@ import { createHash } from 'node:crypto'; export function promptHash(text: string): string { return createHash('sha256').update(text.trim(), 'utf8').digest('hex').slice(0, 16); } + +/** Strip the attachment preamble Claude Code prepends to a prompt with files. */ +export function stripPreamble(text: string): string { + return text + .replace(/^\s*[\s\S]*?<\/system_instruction>\s*/g, '') + .replace(/[\s\S]*?<\/ide_selection>/g, '') + .trim(); +} + +/** + * A key for recognising the same prompt in the hook's input and in the + * transcript, which do not hold it byte for byte: the transcript can split it + * into several text blocks, and a preamble may sit on either side. Only the + * hash crosses between the two, never the text. + */ +export function promptMatchKey(text: string): string { + return promptHash(stripPreamble(text).replace(/\s+/g, ' ')); +} diff --git a/src/history.ts b/src/history.ts index 9c80ae8..1606523 100644 --- a/src/history.ts +++ b/src/history.ts @@ -12,6 +12,7 @@ import { readdirSync, statSync, createReadStream, existsSync } from 'node:fs'; import { join } from 'node:path'; import { homedir } from 'node:os'; import { createInterface } from 'node:readline'; +import { stripPreamble } from './hash.js'; export interface HistoryPrompt { ts: string; @@ -22,7 +23,7 @@ export interface HistoryPrompt { export const PROJECTS_DIR = join(homedir(), '.claude', 'projects'); -interface TranscriptRecord { +export interface TranscriptRecord { type?: string; message?: { role?: string; content?: unknown }; timestamp?: string; @@ -34,7 +35,7 @@ interface TranscriptRecord { userType?: string; } -function textOf(content: unknown): string { +export function textOf(content: unknown): string { if (typeof content === 'string') return content; if (Array.isArray(content)) { return content @@ -53,7 +54,7 @@ function hasToolResult(content: unknown): boolean { ); } -function isHumanPrompt(record: TranscriptRecord): boolean { +export function isHumanPrompt(record: TranscriptRecord): boolean { if (record.type !== 'user') return false; if (record.isSidechain || record.isMeta) return false; if (hasToolResult(record.message?.content)) return false; @@ -84,14 +85,6 @@ export function listTranscripts(dir = PROJECTS_DIR): string[] { return files; } -/** Strip the attachment preamble Claude Code prepends to a prompt with files. */ -function stripPreamble(text: string): string { - return text - .replace(/^\s*[\s\S]*?<\/system_instruction>\s*/g, '') - .replace(/[\s\S]*?<\/ide_selection>/g, '') - .trim(); -} - async function readTranscript(path: string, out: HistoryPrompt[]): Promise { const project = path.split('/').slice(-2, -1)[0] ?? 'unknown'; const stream = createReadStream(path, { encoding: 'utf8' }); diff --git a/src/hook.ts b/src/hook.ts index e7b8e99..84d6169 100644 --- a/src/hook.ts +++ b/src/hook.ts @@ -9,7 +9,7 @@ */ import { readFileSync } from 'node:fs'; import { loadConfig } from './config.js'; -import { promptHash } from './hash.js'; +import { promptHash, promptMatchKey } from './hash.js'; import { appendLog, type LogEntry } from './log.js'; import { applyPrivacy } from './redact.js'; import { skipReason } from './skip.js'; @@ -20,6 +20,8 @@ interface HookInput { /** The submitted text. Claude Code 2.1.x sends `prompt`; older docs say `user_input`. */ prompt?: string; user_input?: string; + /** Claude Code's own transcript of this session; read only when replies are enabled. */ + transcript_path?: string; } function readStdin(): HookInput | null { @@ -82,7 +84,12 @@ async function main(): Promise { // inline.ts for the same reason. The budget is hard: whatever has not // answered by then is abandoned and nothing prints. const { runInline } = await import('./inline.js'); - const line = await runInline(stored, hash, config, { session: entry.session, ts: entry.ts }); + const line = await runInline(stored, hash, config, { + session: entry.session, + ts: entry.ts, + transcriptPath: input.transcript_path, + promptKey: promptMatchKey(text), + }); if (line) emitLine(line); } diff --git a/src/inline.ts b/src/inline.ts index 8288081..8022d1c 100644 --- a/src/inline.ts +++ b/src/inline.ts @@ -3,13 +3,16 @@ * never loads it, and with it never loads the Jev client. */ import { CHECKS } from './checks.js'; -import { type Config, sessionContextEnabled } from './config.js'; +import { type Config, sessionContextEnabled, sessionRepliesEnabled } from './config.js'; +import { clampReply, recentTurns, type Turn } from './conversation.js'; import { appendScores, readScores, recentSessionPrompts } from './log.js'; import { applyPrivacy } from './redact.js'; -import { interpret, type PromptScore, scoreOne } from './score.js'; +import { clampPrompt, interpret, type PromptScore, scoreOne } from './score.js'; /** Earlier prompts from the session sent with a follow-up. */ const CONTEXT_PROMPTS = 2; +/** Earlier prompt-and-reply exchanges sent with a follow-up when replies are on. */ +const CONTEXT_EXCHANGES = 2; function format(result: PromptScore): string | null { // Only confident failures reach the line. `inlineSafe` has already demoted @@ -36,43 +39,80 @@ function format(result: PromptScore): string | null { return `${head}\nMissing: ${missing}.`; } +interface SessionContext { + /** Earlier prompts only, judged under the standalone criteria. */ + prompts: string[]; + /** Earlier prompts and the agent's replies, judged under the conversation criteria. */ + conversation: Turn[]; +} + +const NO_CONTEXT: SessionContext = { prompts: [], conversation: [] }; + /** - * Earlier prompts from this session, re-run through the privacy level: a - * prompt logged under `raw` must still be redacted if the level is now - * `redact`. Empty for the first prompt of a session, which is judged alone - * because it has to carry everything the agent needs. + * What came before this prompt in the session, every piece re-run through the + * privacy level: a prompt logged under `raw`, and every agent reply, must + * still be redacted if the level is now `redact`. Empty for the first prompt + * of a session, which is judged alone because it has to carry everything the + * agent needs. */ -function sessionContext(config: Config, session: string, ts: string): string[] { - if (session === 'unknown' || !sessionContextEnabled()) return []; +function sessionContext(config: Config, at: InlineAt): SessionContext { + if (at.session === 'unknown' || !sessionContextEnabled()) return NO_CONTEXT; + // Redacted whole, then clamped: a cut made first can split a credential into + // a fragment no rule recognises. Redaction is linear, so the whole text is + // cheap; src/redact.ts bounds every quantifier for that reason. + const safe = (text: string, clamp: (t: string) => string = clampPrompt): string | null => { + const redacted = applyPrivacy(text, config.privacy).text; + return redacted === null ? null : clamp(redacted); + }; + if (sessionRepliesEnabled() && at.transcriptPath) { + try { + const conversation = recentTurns(at.transcriptPath, at.promptKey, CONTEXT_EXCHANGES, config.bypassPrefix) + .map((turn) => ({ role: turn.role, text: safe(turn.text, turn.role === 'agent' ? clampReply : clampPrompt) })) + .filter((turn): turn is Turn => turn.text !== null); + return { prompts: [], conversation }; + } catch { + /* transcript unreadable: fall back to the earlier prompts in the log */ + } + } try { - return recentSessionPrompts(session, ts, CONTEXT_PROMPTS) - .map((entry) => (entry.text === null ? null : applyPrivacy(entry.text, config.privacy).text)) + const prompts = recentSessionPrompts(at.session, at.ts, CONTEXT_PROMPTS) + .map((entry) => (entry.text === null ? null : safe(entry.text))) .filter((text): text is string => text !== null); + return { prompts, conversation: [] }; } catch { - return []; + return NO_CONTEXT; } } +interface InlineAt { + session: string; + ts: string; + transcriptPath?: string | undefined; + /** promptMatchKey of the original prompt: a hash, so no text crosses here. */ + promptKey: string; +} + /** * @param redactedText The prompt *after* the configured privacy level has been * applied. The caller must never pass the raw prompt: this is the last hop * before the API and it does no redaction of its own. * @param hash Content hash of the ORIGINAL text, so the cache key is stable * across privacy-level changes. - * @param at The session and submission time of this prompt, to find the - * prompts before it. + * @param at The session, submission time and transcript of this prompt, to + * find what came before it. */ export async function runInline( redactedText: string, hash: string, config: Config, - at: { session: string; ts: string }, + at: InlineAt, ): Promise { - const context = sessionContext(config, at.session, at.ts); + const context = sessionContext(config, at); + const hasContext = context.prompts.length > 0 || context.conversation.length > 0; // A prompt whose text has not changed is never scored twice, unless it is a // follow-up: "commit and push" means something different in every session. - if (context.length === 0) { + if (!hasContext) { try { const cached = readScores().get(hash); if (cached && !cached.context) { @@ -89,7 +129,12 @@ export async function runInline( }); const scored = await Promise.race([ - scoreOne(redactedText, hash, { timeoutMs: config.alwaysTimeoutMs, inlineSafe: true, context }), + scoreOne(redactedText, hash, { + timeoutMs: config.alwaysTimeoutMs, + inlineSafe: true, + context: context.prompts, + conversation: context.conversation, + }), deadline, ]); if (!scored) return null; diff --git a/src/log.ts b/src/log.ts index bfa8934..3025715 100644 --- a/src/log.ts +++ b/src/log.ts @@ -55,6 +55,11 @@ export interface ScoreRecord { * served from the cache, because the same text elsewhere had other context. */ context?: number; + /** + * Scored in conversation, with the agent's replies and the conversation + * criteria. Its probabilities are read against the conversation thresholds. + */ + conversation?: true; } /** The correction-rate verdict for the prompt with this hash. */ diff --git a/src/redact.ts b/src/redact.ts index 7e23144..2d26fdd 100644 --- a/src/redact.ts +++ b/src/redact.ts @@ -24,6 +24,32 @@ const CREDENTIAL_RULES: Rule[] = [ { name: 'aws', pattern: /\b(?:AKIA|ASIA)[A-Z0-9]{16}\b/g, replace: '[KEY]' }, { name: 'google', pattern: /\bAIza[A-Za-z0-9_-]{30,}/g, replace: '[KEY]' }, { name: 'slack', pattern: /\bxox[abprs]-[A-Za-z0-9-]{10,}/g, replace: '[KEY]' }, + { name: 'stripe', pattern: /\b[sr]k_(?:live|test)_[A-Za-z0-9]{16,}/g, replace: '[KEY]' }, + { name: 'npm', pattern: /\bnpm_[A-Za-z0-9]{30,}/g, replace: '[KEY]' }, + { name: 'github-fine-grained', pattern: /\bgithub_pat_[A-Za-z0-9_]{22,}/g, replace: '[KEY]' }, + { name: 'sendgrid', pattern: /\bSG\.[A-Za-z0-9_-]{16,}\.[A-Za-z0-9_-]{16,}/g, replace: '[KEY]' }, + { + name: 'webhook-url', + // The URL is the credential: anyone holding it can post to the channel. + pattern: /https:\/\/(?:hooks\.slack\.com\/services|discord(?:app)?\.com\/api\/webhooks)\/[^\s'"`)\]]+/g, + replace: '[WEBHOOK]', + }, + { + name: 'url-credentials', + // scheme://user:password@host. Agent replies quote connection strings + // back from .env files and config; the user and host are kept, the + // password is not. Runs before the email rule, which would otherwise + // swallow "password@host" by accident and leave the next one in place. + // The user may be empty (redis://:pw@host), and the password may hold '@' + // or '/': it runs greedily to the last '@' in the token that a host + // follows. That can take a path with an '@' in it along with the password; + // masking too much is the safe way round. A port followed by a path + // (host:8080/@handle) is not a password. + pattern: + /\b([a-z][a-z0-9+.-]{0,30}:\/\/[^\s:/@'"`]{0,256}:)(?!\d{1,5}(?:\/|$))[^\s'"`]{1,256}@(?=[^\s@/'"`]{1,256}(?:[\s/'"`:?#]|$))/gi, + replace: '$1[REDACTED]@', + }, + { name: 'azure-sas', pattern: /([?&]sig=)[A-Za-z0-9%+/=]{16,}/g, replace: '$1[REDACTED]' }, { name: 'jwt', pattern: /\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}/g, replace: '[JWT]' }, { name: 'bearer', pattern: /\b[Bb]earer\s+[A-Za-z0-9._-]{12,}/g, replace: 'Bearer [KEY]' }, { @@ -31,7 +57,8 @@ const CREDENTIAL_RULES: Rule[] = [ // Azure client secrets carry a '~' mid-token, which is vanishingly rare in // prose, code identifiers and paths. Found in real local history, where it // was written as "value - " and matched no labelled rule below. - pattern: /(? m.replace(/((?::|=|-\s)\s*['"`]?)([^\s'"`,;/\\]{12,})/, '$1[REDACTED]'), + /\b(value|secret|password|passwd|token|api[ _-]?key|client[ _-]?secret)\b['"]?\s*(?::|=|-\s|is\s)\s*(['"`]?)([^\s'"`,;/\\]{12,})\2/gi, + replace: (m: string) => m.replace(/((?::|=|-\s|\bis\s)\s*['"`]?)([^\s'"`,;/\\]{12,})/i, '$1[REDACTED]'), + }, + { + name: 'labelled-password', + // Passwords are short more often than keys are, so the length floor is + // lower than for the generic labels above; the label itself is specific. + // The optional quote after the label covers a JSON key: "password": "x". + // Prose counts too: "the password is x" was found in real history. + pattern: /\b(password|passwd|pwd)\b['"]?(?:\s*[:=]|\s+is)\s*(['"`]?)([^\s'"`,;]{6,})\2/gi, + replace: (m: string) => m.replace(/((?:[:=]|\bis)\s*['"`]?)([^\s'"`,;]{6,})/i, '$1[REDACTED]'), }, { name: 'assigned-secret', - // KEY=value / "api_key": "value" / TOKEN: value + // KEY=value / "api_key": "value" / TOKEN: value / DB_PASS=value. The short + // suffixes need an underscore before them, so bypass: or oauth: in code is + // not taken for a secret. pattern: - /\b([A-Za-z_][A-Za-z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIAL)S?)\b(\s*[:=]\s*)(['"]?)([^\s'"`,;]{6,})\3/gi, + /\b([A-Za-z_][A-Za-z0-9_]*(?:KEY|TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIAL)S?|(?:[A-Za-z0-9]+_)+(?:PASS|PWD|AUTH))\b(\s*[:=]\s*)(['"]?)([^\s'"`,;]{6,})\3/gi, replace: (m: string) => m.replace(/([:=]\s*['"]?)([^\s'"`,;]{6,})/, '$1[REDACTED]'), }, + { + name: 'long-hex', + // Hashes, hex tokens and hex-encoded keys: 16 or more hex characters with a + // digit among them. Commit SHAs go too; the marker still tells the scorer a + // specific identifier was named. UUIDs survive: their hex runs are shorter. + // An 0x prefix is how hex private keys are usually written, and a hyphen + // before or after (token-) does not hide one. + pattern: /(? + m.split(/[-_+=/]/).some(looksRandom) || looksRandom(m.replace(/[-_+=/]/g, '')) ? '[KEY]' : m, + }, ]; +/** + * Whether a string reads as random rather than as words. + * + * Tuned by simulation and against real prompts and agent replies. Random + * base62 switches between digit, lowercase and uppercase at about 60% of + * positions; identifiers switch once per word. Words also give themselves away + * by their capitals: in CamelCase nearly every capital starts a word of three + * or more letters, in random text about one in five does. With a digit + * required as well, this catches about 92% of random 20-character tokens and + * over 97% from 32 up, while leaving migration names, slugs, constants and + * CamelCase identifiers alone. A random token with no digit at all is missed. + */ +function looksRandom(s: string): boolean { + if (s.length < 16) return false; + const digits = (s.match(/\d/g) ?? []).length; + const lower = (s.match(/[a-z]/g) ?? []).length; + const upper = (s.match(/[A-Z]/g) ?? []).length; + if (digits < 1 || lower < 2 || upper < 2) return false; + const words = (s.match(/[A-Z][a-z]{2,}/g) ?? []).length; + if (words / upper >= 0.5) return false; + const kind = (c: string): number => (/\d/.test(c) ? 0 : /[a-z]/.test(c) ? 1 : /[A-Z]/.test(c) ? 2 : 3); + let switches = 0; + for (let i = 1; i < s.length; i += 1) if (kind(s[i]!) !== kind(s[i - 1]!)) switches += 1; + return switches / (s.length - 1) >= 0.4; +} + const EMAIL_RULE: Rule = { name: 'email', - pattern: /\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b/g, + // Bounded to the RFC limits: unbounded runs made a long dotted line quadratic. + pattern: /\b[A-Za-z0-9._%+-]{1,64}@[A-Za-z0-9.-]{1,253}\.[A-Za-z]{2,63}\b/g, replace: '[EMAIL]', }; @@ -125,7 +210,9 @@ export function features(text: string): Features { words: text.trim().split(/\s+/).filter(Boolean).length, lines: text.split('\n').length, hasCodeFence: /```/.test(text), - hasFilePath: /[\w\-/]+\.[a-z]{1,5}\b/i.test(text), + // Anchored and bounded: unanchored, a long run of word characters made this + // quadratic, on every prompt the hook sees. + hasFilePath: /(?(); for (const check of result.checks) map.set(check.id, check.verdict); verdicts.set(entry.hash, map); diff --git a/src/score.ts b/src/score.ts index 0d19f7c..225cbb6 100644 --- a/src/score.ts +++ b/src/score.ts @@ -9,6 +9,7 @@ */ import { CHECKS, type CheckDef, type CheckId, GATES, type GateId } from './checks.js'; import { ask, estimateTokens, MODEL, type NoulQuestion, tryAsk, type Usage } from './jev.js'; +import type { Turn } from './conversation.js'; import type { ScoreRecord } from './log.js'; import { runPool } from './pool.js'; @@ -37,26 +38,38 @@ export interface PromptScore { gates: Partial>; } +/** + * How a prompt is judged. `alone` is the original check set. `prompts` adds + * earlier session prompts as context under the same criteria. `conversation` + * adds the agent's replies too, and switches to each check's conversation + * criteria and thresholds. + */ +export type Mode = 'alone' | 'prompts' | 'conversation'; + /** Keep the head and tail: a verification command is often the last line. */ export function clampPrompt(text: string): string { if (text.length <= MAX_PROMPT_CHARS) return text; return `${text.slice(0, MAX_PROMPT_CHARS - 1000)}\n…\n${text.slice(-1000)}`; } +function scopeFor(id: string, contextIds: string[], mode: Mode): string { + const ids = contextIds.map((c) => `"${c}"`).join(', '); + if (mode === 'conversation' && contextIds.length) { + return `Judge only the message with id "${id}" in the state: the developer's latest prompt to a coding agent. The messages ${ids} are the conversation before it, in order, and "role" says whether the developer or the agent wrote each. They are context and are not themselves being judged.`; + } + if (contextIds.length) { + return `Judge only the message with id "${id}" in the state. The messages ${ids} are earlier prompts from the same conversation, given as context: anything they already state counts as known to the reader of "${id}", but they are not themselves being judged.`; + } + return `Consider only the message with id "${id}" in the state.`; +} + /** - * @param contextIds Earlier messages from the same session that are in the - * state as background. Only `id` is judged; what the context already told - * the agent counts as known. + * @param contextIds Earlier messages that are in the state as background. + * Only `id` is judged. */ -function questionsFor(id: string, contextIds: string[] = []): Record { +function questionsFor(id: string, contextIds: string[] = [], mode: Mode = 'alone'): Record { const questions: Record = {}; - const scope = contextIds.length - ? `Judge only the message with id "${id}" in the state. The messages ${contextIds - .map((c) => `"${c}"`) - .join( - ' and ', - )} are earlier prompts from the same conversation, given as context: anything they already state counts as known to the reader of "${id}", but they are not themselves being judged.` - : `Consider only the message with id "${id}" in the state.`; + const scope = scopeFor(id, contextIds, mode); for (const gate of GATES) { questions[`${id}__${gate.id}`] = { type: 'noul', @@ -65,10 +78,11 @@ function questionsFor(id: string, contextIds: string[] = []): Record): number { export interface ScoreInput { hash: string; text: string; + /** The conversation before this prompt, when it is judged in conversation. */ + conversation?: Turn[]; +} + +/** A type alias rather than an interface, so it satisfies the SDK's JSON state type. */ +type StateMessage = { id: string; text: string } | { id: string; role: Turn['role']; text: string }; + +/** The state messages for one prompt: its conversation, then the prompt as `key`. */ +function messagesFor(key: string, input: ScoreInput): StateMessage[] { + const turns = (input.conversation ?? []).map((t, i) => ({ + id: `${key}c${i + 1}`, + role: t.role, + text: clampPrompt(t.text), + })); + if (turns.length === 0) return [{ id: key, text: clampPrompt(input.text) }]; + return [...turns, { id: key, role: 'developer', text: clampPrompt(input.text) }]; +} + +function questionsForInput(key: string, messages: StateMessage[]): Record { + const contextIds = messages.slice(0, -1).map((m) => m.id); + return questionsFor(key, contextIds, contextIds.length ? 'conversation' : 'alone'); } interface Batch { - items: { key: string; input: ScoreInput }[]; + items: { key: string; input: ScoreInput; messages: StateMessage[] }[]; /** Estimated input tokens for the request: state plus every question. */ tokens: number; } @@ -100,9 +135,9 @@ export function planBatches(inputs: ScoreInput[]): Batch[] { inputs.forEach((input, index) => { const key = `m${index}`; - const text = clampPrompt(input.text); - const itemStateTokens = estimateTokens(text) + 12; - const itemTokens = itemStateTokens + questionTokens(questionsFor(key)); + const messages = messagesFor(key, input); + const itemStateTokens = messages.reduce((sum, m) => sum + estimateTokens(m.text) + 12, 0); + const itemTokens = itemStateTokens + questionTokens(questionsForInput(key, messages)); const wouldOverflow = current.items.length > 0 && @@ -116,7 +151,7 @@ export function planBatches(inputs: ScoreInput[]): Batch[] { stateTokens = 0; } - current.items.push({ key, input: { hash: input.hash, text } }); + current.items.push({ key, input, messages }); stateTokens += itemStateTokens; current.tokens += itemTokens; }); @@ -140,7 +175,7 @@ export function interpret( hash: string, probabilities: Partial>, gates: Partial>, - opts: { inlineSafe?: boolean } = {}, + opts: { inlineSafe?: boolean; mode?: Mode } = {}, ): PromptScore { const checks = CHECKS.map((def): CheckResult => { const gate = def.appliesWhen; @@ -150,6 +185,8 @@ export function interpret( return { id: def.id, label: def.label, verdict: 'n/a', probability: null, def }; } } + // Conversation scores are calibrated separately from the standalone ones. + const { threshold, inlineEligible } = opts.mode === 'conversation' ? def.conversation : def; const p = probabilities[def.id]; if (p === undefined) { return { id: def.id, label: def.label, verdict: 'undecided', probability: null, def }; @@ -157,10 +194,10 @@ export function interpret( // In the inline path a check we cannot stand behind, or a finding sitting // near the threshold, is dropped rather than shown: a false positive there // interrupts every message. - if (opts.inlineSafe && (!def.inlineEligible || Math.abs(p - def.threshold) < def.inlineMargin)) { + if (opts.inlineSafe && (!inlineEligible || Math.abs(p - threshold) < def.inlineMargin)) { return { id: def.id, label: def.label, verdict: 'undecided', probability: p, def }; } - return { id: def.id, label: def.label, verdict: p >= def.threshold ? 'pass' : 'fail', probability: p, def }; + return { id: def.id, label: def.label, verdict: p >= threshold ? 'pass' : 'fail', probability: p, def }; }); const decided = checks.filter((c) => c.verdict === 'pass' || c.verdict === 'fail'); @@ -194,11 +231,9 @@ function unpack(answers: Record, key: string, hash: string, ts: } async function runBatch(batch: Batch, options: ScoreRunOptions): Promise { - const state = { - messages: batch.items.map((item) => ({ id: item.key, text: item.input.text })), - }; + const state = { messages: batch.items.flatMap((item) => item.messages) }; const questions: Record = {}; - for (const item of batch.items) Object.assign(questions, questionsFor(item.key)); + for (const item of batch.items) Object.assign(questions, questionsForInput(item.key, item.messages)); const answers = await ask(state, questions, { timeoutMs: options.timeoutMs ?? 60_000, @@ -206,7 +241,12 @@ async function runBatch(batch: Batch, options: ScoreRunOptions): Promise unpack(answers, item.key, item.input.hash, now)); + return batch.items.map((item) => { + const record = unpack(answers, item.key, item.input.hash, now); + const turns = item.input.conversation?.length ?? 0; + if (turns) Object.assign(record, { context: turns, conversation: true }); + return record; + }); } /** Score many prompts. Batches that fail are dropped, not retried forever. */ @@ -222,27 +262,34 @@ export async function scoreMany(inputs: ScoreInput[], options: ScoreRunOptions = /** * Score a single prompt. Used by /jevpromptcoach:score and by `always` mode. * - * `context` is earlier prompts from the same session, oldest first, already - * through the configured privacy level. They are sent but not scored. + * `context` is earlier prompts from the same session, oldest first; with + * `conversation`, the agent's replies ride along too and the conversation + * criteria apply. Either way the text must already be through the configured + * privacy level. Context is sent but not scored. */ export async function scoreOne( text: string, hash: string, - options: { timeoutMs?: number; inlineSafe?: boolean; context?: string[] } = {}, + options: { timeoutMs?: number; inlineSafe?: boolean; context?: string[]; conversation?: Turn[] } = {}, ): Promise<{ record: ScoreRecord; result: PromptScore } | null> { - const context = (options.context ?? []).map((t, i) => ({ id: `c${i + 1}`, text: clampPrompt(t) })); - const state = { messages: [...context, { id: 'm0', text: clampPrompt(text) }] }; - const answers = await tryAsk( - state, - questionsFor( - 'm0', - context.map((c) => c.id), - ), - { timeoutMs: options.timeoutMs ?? 20_000 }, - ); + let messages: StateMessage[]; + let mode: Mode; + if (options.conversation?.length) { + messages = messagesFor('m0', { hash, text, conversation: options.conversation }); + mode = 'conversation'; + } else { + const context = (options.context ?? []).map((t, i) => ({ id: `c${i + 1}`, text: clampPrompt(t) })); + messages = [...context, { id: 'm0', text: clampPrompt(text) }]; + mode = context.length ? 'prompts' : 'alone'; + } + const contextIds = messages.slice(0, -1).map((m) => m.id); + const answers = await tryAsk({ messages }, questionsFor('m0', contextIds, mode), { + timeoutMs: options.timeoutMs ?? 20_000, + }); if (!answers) return null; const record = unpack(answers, 'm0', hash, new Date().toISOString()); - if (context.length) record.context = context.length; - return { record, result: interpret(hash, record.probabilities, record.gates, options) }; + if (contextIds.length) record.context = contextIds.length; + if (mode === 'conversation') record.conversation = true; + return { record, result: interpret(hash, record.probabilities, record.gates, { ...options, mode }) }; } diff --git a/test/eval-conversations-results.json b/test/eval-conversations-results.json new file mode 100644 index 0000000..d46b760 --- /dev/null +++ b/test/eval-conversations-results.json @@ -0,0 +1,159 @@ +{ + "ranAt": "2026-09-24T22:05:37.645Z", + "model": "jev-latest", + "fixtures": 40, + "set": "conversations", + "inputTokens": 111736, + "targetPrecision": 0.9, + "checks": { + "named_target": { + "threshold": 0.15, + "fail": { + "precision": 1, + "recall": 0.18181818181818182, + "support": 11 + }, + "pass": { + "precision": 0.7631578947368421, + "recall": 1, + "support": 29 + }, + "measurable": true, + "clearsTarget": true, + "suggestedThreshold": 0.15 + }, + "success_condition": { + "threshold": 0.2, + "fail": { + "precision": 1, + "recall": 0.1111111111111111, + "support": 18 + }, + "pass": { + "precision": 0.5789473684210527, + "recall": 1, + "support": 22 + }, + "measurable": true, + "clearsTarget": true, + "suggestedThreshold": 0.25 + }, + "bounded_scope": { + "threshold": 0.45, + "fail": { + "precision": 0.2857142857142857, + "recall": 1, + "support": 2 + }, + "pass": { + "precision": 1, + "recall": 0.868421052631579, + "support": 38 + }, + "measurable": false, + "clearsTarget": false, + "suggestedThreshold": 0.45 + }, + "constraints": { + "threshold": 0.8, + "fail": { + "precision": 0.9714285714285714, + "recall": 1, + "support": 34 + }, + "pass": { + "precision": 1, + "recall": 0.8333333333333334, + "support": 6 + }, + "measurable": true, + "clearsTarget": true, + "suggestedThreshold": 0.75 + }, + "repro_included": { + "threshold": 0.4, + "fail": { + "precision": 0.5, + "recall": 1, + "support": 1 + }, + "pass": { + "precision": 1, + "recall": 0.5, + "support": 2 + }, + "measurable": false, + "clearsTarget": false, + "suggestedThreshold": 0.4 + }, + "plan_first": { + "threshold": 0.35, + "fail": { + "precision": 1, + "recall": 1, + "support": 4 + }, + "pass": { + "precision": 1, + "recall": 1, + "support": 2 + }, + "measurable": false, + "clearsTarget": false, + "suggestedThreshold": 0.35 + }, + "verification": { + "threshold": 0.75, + "fail": { + "precision": 0.972972972972973, + "recall": 0.9473684210526315, + "support": 38 + }, + "pass": { + "precision": 0.3333333333333333, + "recall": 0.5, + "support": 2 + }, + "measurable": true, + "clearsTarget": true, + "suggestedThreshold": 0.75 + }, + "_crossValidated": { + "named_target": { + "precision": 0.6666666666666666, + "recall": 0.36363636363636365, + "support": 11 + }, + "success_condition": { + "precision": 0.75, + "recall": 0.16666666666666666, + "support": 18 + }, + "bounded_scope": { + "precision": 0.3333333333333333, + "recall": 0.5, + "support": 2 + }, + "constraints": { + "precision": 0.9714285714285714, + "recall": 1, + "support": 34 + }, + "repro_included": { + "precision": 1, + "recall": 1, + "support": 1 + }, + "plan_first": { + "precision": 1, + "recall": 0.75, + "support": 4 + }, + "verification": { + "precision": 0.972972972972973, + "recall": 0.9473684210526315, + "support": 38 + } + } + } +} diff --git a/test/eval-conversations-results.txt b/test/eval-conversations-results.txt new file mode 100644 index 0000000..9e0b830 --- /dev/null +++ b/test/eval-conversations-results.txt @@ -0,0 +1,29 @@ +Gates + gate acc n + is_bug_report 0.97 40 + is_large_change 0.88 40 + +Checks — FAIL is the class that gates `always` mode + + check thr | fail-P fail-R n | pass-P pass-R n | verdict + named_target 0.15 | 1.00 0.18 11 | 0.76 1.00 29 | ok (best thr 0.15) + success_condition 0.20 | 1.00 0.11 18 | 0.58 1.00 22 | ok (best thr 0.25) + bounded_scope 0.45 | 0.29 1.00 2 | 1.00 0.87 38 | too few fail cases (n=2) — not measurable (best thr 0.45) + constraints 0.80 | 0.97 1.00 34 | 1.00 0.83 6 | ok (best thr 0.75) + repro_included 0.40 | 0.50 1.00 1 | 1.00 0.50 2 | too few fail cases (n=1) — not measurable (best thr 0.4) + plan_first 0.35 | 1.00 1.00 4 | 1.00 1.00 2 | too few fail cases (n=4) — not measurable (best thr 0.35) + verification 0.75 | 0.97 0.95 38 | 0.33 0.50 2 | ok (best thr 0.75) + +Five-fold cross-validated fail-precision (thresholds re-selected per fold) + + check fail-P fail-R n + named_target 0.67 0.36 11 + success_condition 0.75 0.17 18 + bounded_scope 0.33 0.50 2 + constraints 0.97 1.00 34 + repro_included 1.00 1.00 1 + plan_first 1.00 0.75 4 + verification 0.97 0.95 38 + +Input tokens: 111,736 (~$0.0047) +All measurable checks clear fail-precision 0.9. diff --git a/test/eval-results.json b/test/eval-results.json index 898720a..941ff13 100644 --- a/test/eval-results.json +++ b/test/eval-results.json @@ -1,8 +1,8 @@ { - "ranAt": "2026-09-18T18:49:21.332Z", + "ranAt": "2026-09-24T22:05:39.832Z", "model": "jev-latest", "fixtures": 40, - "inputTokens": 49666, + "inputTokens": 49652, "targetPrecision": 0.9, "checks": { "named_target": { @@ -54,11 +54,11 @@ "threshold": 0.3, "fail": { "precision": 1, - "recall": 0.9, + "recall": 0.9333333333333333, "support": 30 }, "pass": { - "precision": 0.7692307692307693, + "precision": 0.8333333333333334, "recall": 1, "support": 10 }, @@ -84,11 +84,11 @@ "threshold": 0.35, "fail": { "precision": 1, - "recall": 0.875, + "recall": 1, "support": 8 }, "pass": { - "precision": 0.8333333333333334, + "precision": 1, "recall": 1, "support": 5 }, @@ -127,8 +127,8 @@ "support": 10 }, "constraints": { - "precision": 0.9666666666666667, - "recall": 0.9666666666666667, + "precision": 1, + "recall": 0.9333333333333333, "support": 30 }, "repro_included": { @@ -137,8 +137,8 @@ "support": 5 }, "plan_first": { - "precision": 0.8571428571428571, - "recall": 0.75, + "precision": 1, + "recall": 0.875, "support": 8 }, "verification": { diff --git a/test/eval-results.txt b/test/eval-results.txt index cde8d4e..94eb8dc 100644 --- a/test/eval-results.txt +++ b/test/eval-results.txt @@ -9,9 +9,9 @@ Checks — FAIL is the class that gates `always` mode named_target 0.65 | 0.96 0.93 28 | 0.85 0.92 12 | ok success_condition 0.35 | 1.00 0.86 14 | 0.93 1.00 26 | ok bounded_scope 0.45 | 1.00 0.70 10 | 0.91 1.00 30 | ok - constraints 0.30 | 1.00 0.90 30 | 0.77 1.00 10 | ok + constraints 0.30 | 1.00 0.93 30 | 0.83 1.00 10 | ok repro_included 0.40 | 1.00 1.00 5 | 1.00 1.00 2 | ok - plan_first 0.35 | 1.00 0.88 8 | 0.83 1.00 5 | ok + plan_first 0.35 | 1.00 1.00 8 | 1.00 1.00 5 | ok verification 0.60 | 1.00 1.00 39 | 1.00 1.00 1 | ok Five-fold cross-validated fail-precision (thresholds re-selected per fold) @@ -20,10 +20,10 @@ Five-fold cross-validated fail-precision (thresholds re-selected per fold) named_target 0.96 0.93 28 success_condition 1.00 0.86 14 bounded_scope 1.00 0.60 10 - constraints 0.97 0.97 30 + constraints 1.00 0.93 30 repro_included 1.00 0.80 5 - plan_first 0.86 0.75 8 + plan_first 1.00 0.88 8 verification 1.00 0.97 39 -Input tokens: 49,666 (~$0.0021) +Input tokens: 49,652 (~$0.0021) All measurable checks clear fail-precision 0.9. diff --git a/test/fixtures/README.md b/test/fixtures/README.md index 75ea54a..9a86e55 100644 --- a/test/fixtures/README.md +++ b/test/fixtures/README.md @@ -15,7 +15,7 @@ prompt text in them. ``` npm run build node dist/cli.js fixtures-init # 40 prompts from your history, unlabelled -node dist/cli.js fixtures-init --count 60 --out test/fixtures/prompts.json +node dist/cli.js fixtures-init --count=60 --out=test/fixtures/prompts.json ``` It reads `~/.claude/projects/**`, applies the same skip rules the plugin uses, @@ -60,3 +60,54 @@ before merging. way to see the format is to run it and open the result. Each entry carries an id, the prompt verbatim, a label per check, and the two applicability gates. `src/eval.ts` reads it and `src/checks.ts` names every field. + +## Conversation fixtures + +A follow-up such as "yes, commit it" can only be judged against what came +before it. When `always` mode sends the agent's replies, which it does unless +`JEVPROMPTCOACH_SESSION_REPLIES=0`, each check is asked in its conversation +form instead, and those forms have their own thresholds. They are measured on a +second fixture set, built the same way and kept off the repository for the same +reasons, with one more: it also holds the agent's replies, which quote code and +client detail back. + +``` +node dist/cli.js fixtures-init --conversations +``` + +That writes `conversations.json`: 40 follow-ups from your history, each with the +exchanges before it (your prompt and the agent's closing reply, at most two +exchanges), every label `null`. Only follow-ups whose previous turn ended with +the agent saying something are picked, since that reply is what the set exists +to measure. Everything is redacted at your privacy level, so the eval later +sends exactly what `always` mode would. + +**Label the last message only, read together with its context**, from the +`conversation` criteria in `src/checks.ts`, not the standalone ones: + +- A habit is present if the follow-up supplies it, or if the shown context + already established it and the follow-up relies on it: accepting the agent's + proposal, answering its question, or pointing unambiguously at something + named earlier ("the email one"). +- Only what is shown counts. If you remember the session and know more than the + context holds, label from the context; Jev only ever sees that much. +- The gates describe the follow-up's request in its context, and the two + conditional checks follow them exactly as in the standalone set. + +Then: + +``` +npm run eval -- --conversations +npm run eval -- --conversations --tune +``` + +The eval refuses a file with unlabelled entries, so the order cannot be got +wrong by accident. Results go to `test/eval-conversations-results.txt` and +`.json`, beside the standalone ones, and like them hold no prompt text. A +conversation check reaches the inline line only by the same rule as a +standalone one: cross-validated fail-precision of at least 0.90 over at least +ten failing examples. + +Forty is enough to measure the common failures and thin for the rare ones. A +check needs at least five failing examples to be measured at all and ten to be +allowed inline; if one falls short, `--count=60` gives it more to work with. diff --git a/test/hook-privacy.test.mjs b/test/hook-privacy.test.mjs index 6949773..13b463f 100644 --- a/test/hook-privacy.test.mjs +++ b/test/hook-privacy.test.mjs @@ -14,7 +14,7 @@ import assert from 'node:assert/strict'; import { createServer } from 'node:http'; import { spawn } from 'node:child_process'; import { createHash } from 'node:crypto'; -import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs'; +import { mkdtempSync, mkdirSync, readFileSync, writeFileSync, rmSync } from 'node:fs'; import { tmpdir } from 'node:os'; import { join } from 'node:path'; @@ -61,7 +61,11 @@ after(() => server?.close()); * spawn would block the event loop and the server could never accept the * hook's connection. */ -function runHook(privacy, prompt = PROMPT, { log = [], scores = [], env = {} } = {}) { +function runHook( + privacy, + prompt = PROMPT, + { log = [], scores = [], env = {}, transcript = null, transcriptIsDirectory = false } = {}, +) { const home = mkdtempSync(join(tmpdir(), 'jpc-test-')); mkdirSync(join(home, '.claude', 'jevpromptcoach'), { recursive: true }); if (log.length) { @@ -70,6 +74,14 @@ function runHook(privacy, prompt = PROMPT, { log = [], scores = [], env = {} } = log.map((e) => JSON.stringify({ features: {}, source: 'hook', ...e })).join('\n') + '\n', ); } + let transcriptPath; + if (transcriptIsDirectory) { + transcriptPath = join(home, 'transcript-dir'); + mkdirSync(transcriptPath); + } else if (transcript) { + transcriptPath = join(home, 'transcript.jsonl'); + writeFileSync(transcriptPath, transcript.map((r) => JSON.stringify(r)).join('\n') + '\n'); + } if (scores.length) { writeFileSync( join(home, '.claude', 'jevpromptcoach', 'scores.jsonl'), @@ -85,6 +97,7 @@ function runHook(privacy, prompt = PROMPT, { log = [], scores = [], env = {} } = env: { ...process.env, JEVPROMPTCOACH_SESSION_CONTEXT: '', + JEVPROMPTCOACH_SESSION_REPLIES: '', ...env, HOME: home, TYPESAFE_BASE_URL: `http://127.0.0.1:${port}`, @@ -94,7 +107,15 @@ function runHook(privacy, prompt = PROMPT, { log = [], scores = [], env = {} } = child.stdout.on('data', (c) => { stdout += c; }); - child.stdin.end(JSON.stringify({ session_id: 's', cwd: '/tmp/d', hook_event_name: 'UserPromptSubmit', prompt })); + child.stdin.end( + JSON.stringify({ + session_id: 's', + cwd: '/tmp/d', + hook_event_name: 'UserPromptSubmit', + prompt, + ...(transcriptPath ? { transcript_path: transcriptPath } : {}), + }), + ); return new Promise((resolve) => { child.on('close', (status) => { rmSync(home, { recursive: true, force: true }); @@ -129,7 +150,8 @@ const savedScore = (extra = {}) => ({ }); /** - * A follow-up carries the two prompts before it in the same session. They are + * With replies turned off, a follow-up carries the two prompts before it in the + * same session. They are * written to the log as raw text, as if captured under `raw`, so the test also * proves they are redacted again on the way out under `redact`. */ @@ -140,7 +162,7 @@ const EARLIER = [ { ts: '2026-01-01T00:03:00.000Z', session: 's', hash: 'h2', text: 'Open /Users/someone/clients/acme/app/main.ts' }, ]; -test('a follow-up sends the two earlier prompts, redacted, and scores only the last', async () => { +test('without a transcript, a follow-up sends the two earlier prompts, redacted, and scores only the last', async () => { captured.length = 0; const result = await runHook('redact', FOLLOW_UP, { log: EARLIER }); assert.equal(result.status, 0); @@ -227,6 +249,336 @@ test('/jevpromptcoach:score ignores a saved score that depended on context', asy assert.ok(!stdout.includes('(cached'), stdout); }); +/** + * An invented Claude Code transcript, in the record shapes the plugin reads. + * It holds everything that must never reach the wire from a transcript: + * narration before a tool call, tool output, a subagent's text, and a bypassed + * exchange whose reply repeats what the bypassed prompt held. + */ +const say = (role, content, extra = {}) => ({ + type: role, + message: { role, content }, + sessionId: 's', + timestamp: '2026-01-01T00:00:00.000Z', + ...(role === 'user' ? { promptSource: 'typed' } : {}), + ...extra, +}); +const text = (t) => [{ type: 'text', text: t }]; +const TRANSCRIPT = [ + say('user', 'oldest exchange, outside the window'), + say('assistant', text('reply to the oldest exchange')), + say('user', `Refactor the retry logic in src/queue/worker.ts and mail bob@acme.com`), + say('assistant', text('narration before the tool call')), + say('assistant', [{ type: 'tool_use', id: 't1', name: 'Read', input: {} }]), + say('user', [{ type: 'tool_result', tool_use_id: 't1', content: 'TOOL OUTPUT BODY' }], { promptSource: undefined }), + say('assistant', text('subagent chatter'), { isSidechain: true }), + say('assistant', text(`Done, worker.ts backs off now. Deploy key was ${SECRET}. Want me to commit?`)), + say('user', '*a bypassed prompt about the private client'), + say('assistant', text('reply that repeats the private client')), + say('user', 'Yes, and keep the exported constant as it is'), + say('assistant', text('Kept it. Anything else before I commit?')), +]; +const REPLIES_ON = { JEVPROMPTCOACH_SESSION_REPLIES: '1' }; + +test('with replies on, a follow-up sends the last two exchanges, redacted, and scores only the last', async () => { + captured.length = 0; + await runHook('redact', FOLLOW_UP, { transcript: TRANSCRIPT, env: REPLIES_ON }); + assert.equal(captured.length, 1); + const { state, questions } = captured[0]; + + assert.deepEqual( + state.messages.map((m) => m.role), + ['developer', 'agent', 'developer', 'agent', 'developer'], + 'two exchanges, then the prompt being scored', + ); + assert.equal(state.messages.at(-1).id, 'm0'); + assert.equal(state.messages.at(-1).text, FOLLOW_UP); + assert.match(state.messages[1].text, /Want me to commit\?$/); + + const sent = JSON.stringify(captured[0]); + for (const never of [ + 'oldest exchange', + 'narration before the tool call', + 'TOOL OUTPUT BODY', + 'subagent chatter', + 'bypassed prompt', + 'private client', + ]) { + assert.ok(!sent.includes(never), `"${never}" reached the wire`); + } + assert.ok(!sent.includes(SECRET), 'a reply leaked the API key'); + assert.ok(!sent.includes('bob@acme.com'), 'a prompt in the transcript leaked the email'); + + assert.ok( + Object.keys(questions).every((k) => k.startsWith('m0__')), + 'context must not be scored', + ); + assert.match(questions.m0__named_target.instructions, /conversation/); +}); + +test('with replies on, the prompt being scored is not repeated when the transcript already has it', async () => { + captured.length = 0; + const transcript = [...TRANSCRIPT, say('user', FOLLOW_UP)]; + await runHook('redact', FOLLOW_UP, { transcript, env: REPLIES_ON }); + const texts = captured[0].state.messages.map((m) => m.text); + assert.equal(texts.filter((t) => t === FOLLOW_UP).length, 1); + assert.equal(texts.length, 5); +}); + +test('with replies on, only conversation checks the eval cleared reach the inline line', async () => { + answerFor = () => 0.01; + const { systemMessage } = JSON.parse( + (await runHook('redact', FOLLOW_UP, { transcript: TRANSCRIPT, env: REPLIES_ON })).stdout, + ); + const missing = systemMessage.split('\n')[1]; + assert.match(missing, /what must not change/); + assert.match(missing, /the verification steps/); + // Every check failed on the wire; these did not clear the conversation eval. + for (const never of ['which file or function', 'what "done" looks like', 'a single focused requirement']) { + assert.ok(!missing.includes(never), `${never} is not inline-eligible in conversation`); + } +}); + +test('with a transcript, the first prompt of a session is still scored alone', async () => { + captured.length = 0; + await runHook('redact', FOLLOW_UP, { transcript: [say('user', FOLLOW_UP)] }); + assert.equal(captured[0].state.messages.length, 1); + assert.equal(captured[0].state.messages[0].role, undefined, 'judged by the standalone criteria'); +}); + +test('narration in the same record as a tool call is not sent', async () => { + captured.length = 0; + const transcript = [ + say('user', 'Refactor src/queue/worker.ts to back off exponentially'), + say('assistant', [ + { type: 'text', text: 'narration sharing a record with the call' }, + { type: 'tool_use', id: 't9', name: 'Read', input: {} }, + ]), + say('assistant', text('final answer about worker.ts')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + const sent = JSON.stringify(captured[0]); + assert.ok(!sent.includes('narration sharing a record'), 'narration before the call reached the wire'); + assert.ok(sent.includes('final answer about worker.ts')); +}); + +test('a queued prompt is its own exchange', async () => { + captured.length = 0; + const transcript = [ + say('user', 'Refactor src/queue/worker.ts to back off exponentially'), + say('assistant', text('first reply')), + say('user', 'Also cap the delay at thirty seconds', { promptSource: 'queued' }), + say('assistant', text('capped at 30s')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + assert.deepEqual( + captured[0].state.messages.map((m) => m.text), + [ + 'Refactor src/queue/worker.ts to back off exponentially', + 'first reply', + 'Also cap the delay at thirty seconds', + 'capped at 30s', + FOLLOW_UP, + ], + ); +}); + +test('a bypassed queued prompt is dropped with its reply, and so is anything queued behind it', async () => { + captured.length = 0; + const transcript = [ + say('user', 'Refactor src/queue/worker.ts to back off exponentially'), + say('assistant', text('first reply')), + say('user', '*queued note about the private client', { promptSource: 'queued' }), + say('assistant', text('reply that repeats the private client')), + // Queued means the agent is still on the bypassed item; what follows answers it. + say('user', 'Also cap the delay at thirty seconds', { promptSource: 'queued' }), + say('assistant', text('capped, and the private client is set')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + const { state } = captured[0]; + assert.ok(!JSON.stringify(state).includes('private client'), 'a bypassed turn reached the wire'); + assert.deepEqual( + state.messages.map((m) => m.text), + ['Refactor src/queue/worker.ts to back off exponentially', 'first reply', FOLLOW_UP], + ); +}); + +test("a prompt queued during a bypassed turn does not carry that turn's reply", async () => { + captured.length = 0; + const transcript = [ + say('user', '*SECRETPROMPT deploy with the private key'), + say('assistant', [{ type: 'tool_use', id: 'a1', name: 'Bash', input: {} }]), + say('user', 'also run the tests after', { promptSource: 'queued' }), + say('user', [{ type: 'tool_result', tool_use_id: 'a1', content: 'ok' }], { promptSource: undefined }), + say('assistant', text('Done, used SECRETPROMPT')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + assert.ok(!JSON.stringify(captured[0]).includes('SECRETPROMPT'), 'the bypassed turn reached the wire'); +}); + +test('a meta system or SDK message still closes the exchange before it', async () => { + captured.length = 0; + const transcript = [ + say('user', 'Refactor src/queue/worker.ts to back off exponentially'), + say('assistant', text('the real reply')), + say('user', 'a meta notice nobody typed', { promptSource: 'system', isMeta: true }), + say('assistant', text('text answering the meta notice')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + const sent = JSON.stringify(captured[0]); + assert.ok(sent.includes('the real reply')); + assert.ok(!sent.includes('answering the meta notice'), 'text after a meta system message was taken as a reply'); +}); + +test('a queued prompt after a system message is kept as its own exchange', async () => { + captured.length = 0; + const transcript = [ + say('user', 'Refactor src/queue/worker.ts to back off exponentially'), + say('assistant', text('first reply')), + say('user', 'a notice nobody typed', { promptSource: 'system' }), + say('user', 'Also cap the delay at thirty seconds', { promptSource: 'queued' }), + say('assistant', text('capped at 30s')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + const texts = captured[0].state.messages.map((m) => m.text); + assert.ok(texts.includes('Also cap the delay at thirty seconds'), 'the queued prompt was dropped'); + assert.ok(texts.includes('capped at 30s')); + assert.ok(!texts.some((t) => t.includes('notice nobody typed'))); +}); + +/** + * Context is clamped to the scorer's size. If the cut came before redaction, a + * credential straddling it would leave a fragment no rule recognises. Built + * from fragments so no file holds the whole value. + */ +const STRADDLE_SECRET = ['Zq8vN2mK', '7xP4wL9r'].join(''); +const STRADDLE_URL = ['redis://svc:', STRADDLE_SECRET, '@cache.internal:6379'].join(''); + +test('a credential across the clamp point of an earlier prompt is redacted before the cut', async () => { + captured.length = 0; + // The prompt clamp keeps the first 3,000 characters: put the password across that line. + const long = 'Connect with ' + 'x'.repeat(2_980 - 13) + ' ' + STRADDLE_URL + ' ' + 'y'.repeat(3_000); + const transcript = [say('user', long), say('assistant', text('connected'))]; + await runHook('redact', FOLLOW_UP, { transcript }); + const sent = JSON.stringify(captured[0]); + assert.ok(!sent.includes(STRADDLE_SECRET.slice(0, 6)), 'a fragment of the password reached the wire'); +}); + +test('a credential across the clamp point of a reply is redacted before the cut', async () => { + captured.length = 0; + // The reply clamp keeps the last 1,498 characters: start the tail inside the password. + // A dotless host, so the email rule cannot mask the fragment by accident. + const tail = STRADDLE_SECRET.slice(4) + '@cache:6379 is live. ' + 'z'.repeat(1_498 - 33); + const reply = 'Using redis://svc:' + STRADDLE_SECRET.slice(0, 4) + tail; + const transcript = [ + say('user', 'Point src/cache.ts at the new redis'), + say('assistant', text('w'.repeat(500) + ' ' + reply)), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + const sent = JSON.stringify(captured[0]); + assert.ok(!sent.includes(STRADDLE_SECRET.slice(4)), 'a fragment of the password reached the wire'); +}); + +test('narration before any kind of tool call is not sent', async () => { + captured.length = 0; + const transcript = [ + say('user', 'Search the docs for the retry API in src/queue/worker.ts'), + say('assistant', [ + { type: 'text', text: 'narration before the search' }, + { type: 'server_tool_use', id: 's1', name: 'web_search', input: {} }, + ]), + say('assistant', text('final answer')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + const sent = JSON.stringify(captured[0]); + assert.ok(!sent.includes('narration before the search')); + assert.ok(sent.includes('final answer')); +}); + +test('an unreadable transcript falls back to the earlier prompts in the log', async () => { + captured.length = 0; + // A directory where the transcript should be: reading it throws. + await runHook('redact', FOLLOW_UP, { log: EARLIER, transcript: [], transcriptIsDirectory: true }); + const { state } = captured[0]; + assert.equal(state.messages.length, 3, 'two earlier prompts from the log, then the one being scored'); + assert.ok(state.messages.every((m) => m.role === undefined)); +}); + +test('text after a system or SDK message is not taken as a reply to the prompt before it', async () => { + captured.length = 0; + const transcript = [ + say('user', 'Refactor src/queue/worker.ts to back off exponentially'), + say('assistant', text('the real reply')), + say('user', 'a notice nobody typed', { promptSource: 'system' }), + say('assistant', text('text answering the notice')), + say('user', 'input from an SDK caller', { promptSource: 'sdk' }), + say('assistant', text('text answering the SDK caller')), + ]; + await runHook('redact', FOLLOW_UP, { transcript }); + const sent = JSON.stringify(captured[0]); + assert.ok(sent.includes('the real reply')); + for (const never of ['notice nobody typed', 'answering the notice', 'SDK caller']) { + assert.ok(!sent.includes(never), `"${never}" reached the wire`); + } +}); + +test('the prompt being scored is recognised even when the transcript splits it into blocks', async () => { + captured.length = 0; + const prompt = 'Commit the retry change\n\nthen push it to the branch'; + const transcript = [ + ...TRANSCRIPT, + say('user', [ + { type: 'text', text: 'Commit the retry change' }, + { type: 'text', text: 'then push it to the branch' }, + ]), + ]; + await runHook('redact', prompt, { transcript }); + const texts = captured[0].state.messages.map((m) => m.text); + assert.equal(texts.length, 5, 'two exchanges, then the prompt, with no duplicate'); + assert.equal(texts.filter((t) => t.startsWith('Commit the retry change')).length, 1); +}); + +test('an earlier identical prompt that got a reply stays as context', async () => { + captured.length = 0; + const transcript = [say('user', FOLLOW_UP), say('assistant', text('pushed to the branch'))]; + await runHook('redact', FOLLOW_UP, { transcript }); + assert.deepEqual( + captured[0].state.messages.map((m) => m.text), + [FOLLOW_UP, 'pushed to the branch', FOLLOW_UP], + ); +}); + +test('replies are on by default', async () => { + captured.length = 0; + await runHook('redact', FOLLOW_UP, { transcript: TRANSCRIPT }); + assert.deepEqual( + captured[0].state.messages.map((m) => m.role), + ['developer', 'agent', 'developer', 'agent', 'developer'], + ); +}); + +test('JEVPROMPTCOACH_SESSION_REPLIES=0 leaves the replies out and sends earlier prompts only', async () => { + captured.length = 0; + await runHook('redact', FOLLOW_UP, { + transcript: TRANSCRIPT, + log: EARLIER, + env: { JEVPROMPTCOACH_SESSION_REPLIES: '0' }, + }); + const { state } = captured[0]; + assert.ok( + state.messages.every((m) => m.role === undefined), + 'no agent reply when turned off', + ); + assert.ok(!JSON.stringify(state).includes('Want me to commit')); + assert.equal(state.messages.length, 3, 'two earlier prompts from the log, then the one being scored'); +}); + +test('with replies on, metadata_only still sends nothing', async () => { + captured.length = 0; + await runHook('metadata_only', FOLLOW_UP, { transcript: TRANSCRIPT, env: REPLIES_ON }); + assert.equal(captured.length, 0); +}); + test('a score of 0 is not shown, only what is missing', async () => { answerFor = () => 0.01; const { systemMessage } = JSON.parse((await runHook('redact')).stdout); @@ -244,6 +596,43 @@ test('a score above 0 is shown', async () => { } }); +test('conversation fixtures hold what the scorer sees, redacted then clamped', async () => { + const home = mkdtempSync(join(tmpdir(), 'jpc-test-')); + const project = join(home, '.claude', 'projects', 'demo'); + mkdirSync(project, { recursive: true }); + mkdirSync(join(home, '.claude', 'jevpromptcoach'), { recursive: true }); + const long = 'Connect with ' + 'x'.repeat(2_967) + ' ' + STRADDLE_URL + ' ' + 'y'.repeat(6_000); + const transcript = [ + say('user', long), + say('assistant', text('w'.repeat(3_000) + ' connected, want me to commit?')), + say('user', 'Yes please, commit and push it'), + say('assistant', text('pushed')), + ]; + writeFileSync(join(project, 's.jsonl'), transcript.map((r) => JSON.stringify(r)).join('\n') + '\n'); + const out = join(home, 'conversations.json'); + await new Promise((resolve) => { + const child = spawn( + process.execPath, + ['dist/cli.js', 'fixtures-init', '--conversations', '--count=1', `--out=${out}`], + { + env: { ...process.env, HOME: home }, + stdio: 'ignore', + }, + ); + child.on('close', resolve); + }); + const [fixture] = JSON.parse(readFileSync(out, 'utf8')); + rmSync(home, { recursive: true, force: true }); + const [prompt, reply] = fixture.context; + // clampPrompt keeps 3,000 characters, a three-character marker, then 1,000. + assert.ok(prompt.text.length <= 4_003, `prompt turn is ${prompt.text.length} characters, more than the scorer sends`); + assert.ok(reply.text.length <= 1_500, `reply turn is ${reply.text.length} characters`); + assert.ok( + !JSON.stringify(fixture).includes(STRADDLE_SECRET.slice(0, 6)), + 'a fragment of the password is in the fixture', + ); +}); + test('metadata_only sends nothing at all', async () => { captured.length = 0; const result = await runHook('metadata_only'); diff --git a/test/redact.test.mjs b/test/redact.test.mjs index c0737a9..be663eb 100644 --- a/test/redact.test.mjs +++ b/test/redact.test.mjs @@ -5,7 +5,7 @@ */ import { test } from 'node:test'; import assert from 'node:assert/strict'; -import { redact, stripCredentials, features } from '../dist/redact.js'; +import { redact, stripCredentials, features, applyPrivacy } from '../dist/redact.js'; /** * Every value below is invented, and each is assembled from fragments at @@ -38,6 +38,58 @@ const SECRETS = [ j('eyJhbGciOiJIUzI1NiJ9', '.', 'eyJzdWIiOiIxMjM0NTY3ODkwIn0', '.', 'dBjftJeZ4CVPmB92K27uhbUJU1p1r'), 'dBjftJeZ4CVPmB92K27uhbUJU1p1r', ], + // Shapes an agent's reply quotes back from .env files, config and command output. + ['stripe', j('key ', 'sk_', 'live_', '51HxYzAbCdEfGhIjKlMnOp'), '51HxYzAbCdEfGhIjKlMnOp'], + ['stripe-restricted', j('rk_', 'test_', '51HxYzAbCdEfGhIjKlMnOp'), '51HxYzAbCdEfGhIjKlMnOp'], + ['npm', j('npm', '_', 'AbCdEfGhIjKlMnOpQrStUvWxYz0123456789'), 'AbCdEfGhIjKlMnOpQrStUvWxYz0123456789'], + [ + 'github-fine-grained', + j('github_', 'pat_', '11ABCDEFG0123456789_abcdefghijklmnopqrstuvwxyz'), + '11ABCDEFG0123456789_abcdefghijklmnopqrstuvwxyz', + ], + [ + 'sendgrid', + j('SG', '.', 'abcdefghijklmnopqrstuv', '.', 'abcdefghijklmnopqrstuvwxyz0123'), + 'abcdefghijklmnopqrstuvwxyz0123', + ], + ['slack-webhook', j('https://hooks.', 'slack.com/services/', 'T0000/B0000/', 'XXXXXXXXXXXXXXXX'), 'XXXXXXXXXXXXXXXX'], + ['url-credentials', j('connect with mysql://root:', 's3cretPw9', '@10.0.0.4/app'), 's3cretPw9'], + ['url-credentials-in-env', j('DATABASE_URL=postgres://app:', 'Hunter2pass', '@db.internal:5432/prod'), 'Hunter2pass'], + [ + 'azure-sas', + j('https://acct.blob.core.windows.net/c/f?sv=2022&', 'sig=', 'AbCdEfGhIjKlMnOp%2BqRsT%3D'), + 'AbCdEfGhIjKlMnOp', + ], + ['underscored-pass', j('DB_', 'PASS', '=', 'abcDEF123456'), 'abcDEF123456'], + ['json-password', j('{"pass', 'word": "', 'CorrectHorse99', '"}'), 'CorrectHorse99'], + ['password-in-prose', j('the docs pass', 'word is ', 'apps.demo2031'), 'apps.demo2031'], + ['secret-in-prose', j('the client secret is ', 'Zq8vN2mK7xP4wL9r'), 'Zq8vN2mK7xP4wL9r'], + ['url-credentials-empty-user', j('redis://:', 's3cr3tPass', '@localhost:6379'), 's3cr3tPass'], + ['url-credentials-at-in-password', j('postgres://app:', 'p@ss9word', '@db.internal/prod'), 'ss9word'], + ['url-credentials-slash-in-password', j('postgres://app:', 'ab/cd9xyz', '@db.internal/prod'), 'cd9xyz'], + [ + 'url-credentials-at-and-slash-in-password', + j('postgres://app:', 'abc123@MiddlePassphrase123/rest', '@db.internal/prod'), + 'MiddlePassphrase123', + ], + // No prefix and no label: caught by shape alone. + ['hex-32', j('auth token is ', '0123456789abcdef', '0123456789abcdef'), j('0123456789abcdef', '0123456789abcdef')], + ['hex-16', j('trace ', '9f86d081', '884c7d65'), j('9f86d081', '884c7d65')], + [ + 'hex-64', + j('0x', '4c0883a69102937d', '6231471b5dbb6204', 'fe512961708279f0', 'd1e5b7a4c3a2f1e0'), + j('4c0883a69102937d', '6231471b5dbb6204'), + ], + ['random-token', j('use `', 'tok_', '9QwErT7yUiOp3AsDfGh2JkLz', '` for staging'), '9QwErT7yUiOp3AsDfGh2JkLz'], + ['random-token-bare', j('it is ', 'xK9mP2qR7vB4nL8w', 'Zt3Yc6Hd', ' now'), j('xK9mP2qR7vB4nL8w', 'Zt3Yc6Hd')], + ['hex-after-hyphen', j('token-', '0123456789abcdef', '0123'), j('0123456789abcdef', '0123')], + // Base64 with '/' and '+', the shape of an AWS secret access key. + ['base64-with-slash', j('use ', 'q7Zk2Wm9Xr4T/b8Nv3Lp6', 'Hs1Jd5+Gf0Yc9Ea2Ru7Qo4'), 'Hs1Jd5+Gf0Yc9Ea2Ru7Qo4'], + [ + 'base64-labelled-with-slash', + j('secret is ', 'q7Zk2Wm9Xr4T/b8Nv3Lp6', 'Hs1Jd5+Gf0Yc9Ea2Ru7Qo4'), + 'Hs1Jd5+Gf0Yc9Ea2Ru7Qo4', + ], ]; test('every credential shape is removed by redact()', () => { @@ -54,6 +106,11 @@ test('credentials are stripped even at privacy level raw', () => { } }); +test('the file-path feature still sees a path', () => { + assert.equal(features('edit src/users/service.ts please').hasFilePath, true); + assert.equal(features('no path in this sentence at all').hasFilePath, false); +}); + test('emails go, filenames stay', () => { const out = redact('ask alice@example.com about src/auth/session.ts'); assert.ok(!out.includes('alice@example.com'), out); @@ -86,16 +143,59 @@ test('an unlabelled Azure client secret is removed', () => { } }); +test('redaction stays linear on long runs that used to be quadratic', () => { + // A 50 KB line of dots or dashes took three to four seconds before the + // quantifiers were bounded, and redaction runs on every prompt in the hook. + // The feature extraction had the same problem on long runs of word characters. + const runs = [ + 'a.'.repeat(25_000), + '-'.repeat(50_000), + 'a.a'.repeat(16_666), + '~.'.repeat(25_000), + 'abcdef'.repeat(8_333), + ]; + for (const s of runs) { + const started = performance.now(); + applyPrivacy(s, 'redact'); + const ms = performance.now() - started; + assert.ok(ms < 250, `${ms.toFixed(0)} ms on a ${s.length}-character run of ${JSON.stringify(s.slice(0, 3))}`); + } +}); + +test('a commit SHA becomes a marker, so the scorer still sees an identifier was named', () => { + const out = redact('Revert commit 4f3a9c2e1b8d7a6f5e4d3c2b1a0987654321fedc'); + assert.equal(out, 'Revert commit [HEX]'); +}); + test('ordinary prose and code are left alone', () => { const samples = [ 'Refactor getUserById in src/users/service.ts and keep the signature', 'Run npm test -- auth.spec.ts to check it', - 'The commit is 4f3a9c2e1b8d7a6f5e4d3c2b1a0987654321fedc', 'See https://github.com/CrowdLinker/JevPromptCoach for details', + // Code and prose that share words with the credential rules. + 'const bypass = userSettings.bypassPrefix', + 'oauth: googleOauthClient, author: someoneElse', + 'Run pwd to see the directory, then reset the password field on the form', + 'The value: 3 and the token count are both logged', + 'The password is wrong and the token is expired', + 'Open https://example.com/login?next=/dashboard&sig=short', + // Long identifiers that the random-token rule must leave alone. Measured on + // real history: these shapes are what long mixed tokens mostly are. + 'Run the CreateUsersTable1695312345678 migration', + 'Rename 1695312345678-CreateUserTable.ts', + 'handleUserAuthenticationCallbackForProvider2 is too long', + 'Set NEXT_PUBLIC_API_BASE_URL_FOR_STAGING_ENV in the pipeline', + 'Checkout feature/mem-335-implement-application-insights-custom-events', + 'The id is 550e8400-e29b-41d4-a716-446655440000 and the bundle main.4f3a9c2e.js', + 'Edit `src/users/UserProfileSettingsPanel2024.tsx` then run `npm test`', + 'getUserById2FromCacheV3Handler reads the cache', + 'Open https://example.com:8080/@handle for the profile', + 'The branch is feature/mem-335-implement-application-insights-custom-events', ]; for (const s of samples) { const out = redact(s); assert.ok(!out.includes('[KEY]'), `false positive on: ${s} -> ${out}`); assert.ok(!out.includes('[REDACTED]'), `false positive on: ${s} -> ${out}`); + assert.ok(!out.includes('[HEX]'), `false positive on: ${s} -> ${out}`); } });