From a9305a3cc0061cde5a330557e53c4daecff5c107 Mon Sep 17 00:00:00 2001 From: Prateek Kathal Date: Thu, 24 Sep 2026 16:44:39 -0400 Subject: [PATCH 1/2] feat: score follow-ups in session context and drop the inline 0/100 (0.3.0) The inline score counts only the checks decided with confidence, often three or four, so a 0 usually meant "0 of 3" and read as if the prompt never reached Jev. Across Sep 22-24, 15 of 23 inline zeros scored above 0 on the full seven checks. The notice now leaves the number off at 0 and shows only what is missing. Short follow-ups ("commit and push all changes") were judged as if they opened the conversation. In always mode a follow-up is now sent with the two prompts before it in the same session and only the new one is scored; the first prompt of a session is still judged alone. On by default, JEVPROMPTCOACH_SESSION_CONTEXT=0 turns it off, read from the environment or the key file like the API key. - Context prompts are re-run through applyPrivacy at the current level, so a prompt logged under raw is still redacted on the way out; the wire test asserts it and fails if the re-redaction is removed. - Only the tail of the log is read, and only in always mode; the on-demand path is unchanged and the SDK stays behind the dynamic import. - A score that used context is tagged and never served from the cache, since the same text in another session had other context. - Thresholds were tuned on prompts scored alone. Follow-up scores have not been through the eval; the fixtures are single prompts, so the eval cannot measure this until it grows session fixtures. --- .claude-plugin/plugin.json | 2 +- README.es.md | 12 ++- README.fr.md | 12 ++- README.md | 18 +++- dist/{chunk-JBKH57J6.js => chunk-7PP552KK.js} | 16 +++- dist/{chunk-RC2GUYJC.js => chunk-U3OSFJLP.js} | 40 ++++++++- dist/{chunk-BD4OHVLJ.js => chunk-W564PYU5.js} | 22 +++-- dist/cli.js | 12 +-- dist/eval.js | 4 +- dist/hook.js | 14 +-- ...{inline-4KJUBKYU.js => inline-6S7NCB56.js} | 40 ++++++--- package.json | 2 +- src/config.ts | 26 +++++- src/hook.ts | 7 +- src/inline.ts | 60 ++++++++++--- src/log.ts | 54 ++++++++++- src/score.ts | 38 ++++++-- test/hook-privacy.test.mjs | 90 ++++++++++++++++++- 18 files changed, 393 insertions(+), 76 deletions(-) rename dist/{chunk-JBKH57J6.js => chunk-7PP552KK.js} (76%) rename dist/{chunk-RC2GUYJC.js => chunk-U3OSFJLP.js} (70%) rename dist/{chunk-BD4OHVLJ.js => chunk-W564PYU5.js} (94%) rename dist/{inline-4KJUBKYU.js => inline-6S7NCB56.js} (50%) diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index f20e30f..8510373 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "jevpromptcoach", "displayName": "Jev (Prompt Coach)", - "version": "0.2.7", + "version": "0.3.0", "description": "Jev (Prompt Coach) scores how well you prompt a coding agent and shows your habits improving over time. Runs on TypeSafe's Jev model.", "license": "MIT", "keywords": [ diff --git a/README.es.md b/README.es.md index 62f86a6..f259d4e 100644 --- a/README.es.md +++ b/README.es.md @@ -226,6 +226,14 @@ Garantías: - **Solo hallazgos defendibles.** Dos comprobaciones quedan excluidas de la línea en pantalla según la evidencia de la eval, y cualquier cosa cerca de un umbral se descarta en lugar de mostrarse. +- **Sin 0/100.** Cuando no pasa ninguna comprobación decidida, el aviso muestra + lo que falta, sin el número. + +**Los seguimientos se leen en contexto.** El primer prompt de una sesión se +puntúa solo. Un prompt posterior se envía con los dos prompts anteriores de la +misma sesión, y solo se puntúa el nuevo. Activado por defecto; +`JEVPROMPTCOACH_SESSION_CONTEXT=0` lo desactiva. Los umbrales se calibraron con +prompts puntuados solos. ## Privacidad @@ -252,7 +260,7 @@ un nombre que acabe en `KEY`/`TOKEN`/`SECRET`/`PASSWORD`. | `/jevpromptcoach:score` | El único prompt que pasaste, depurado | | `/jevpromptcoach:report` | Los prompts registrados sin puntuar, depurados, por lotes | | `config backfill` | Tu historial, depurado, por lotes — **después** de una estimación de coste y una confirmación explícita | -| modo `always` | Cada prompt al enviarlo, depurado | +| modo `always` | Cada prompt al enviarlo, depurado, más hasta dos prompts anteriores de la misma sesión como contexto, también depurados | | En cualquier otro momento | Nada | Sin telemetría. Sin ningún otro destino de red. La clave de API se lee del @@ -260,7 +268,7 @@ entorno o del archivo de clave, y nunca se registra, se imprime ni se incluye en un mensaje de error. Un prompt se puntúa una sola vez. Los resultados se cachean por hash del -contenido. +contenido, salvo un seguimiento en modo `always`, cuyo contexto cambia. ## Contribuir diff --git a/README.fr.md b/README.fr.md index 2190a3d..04958a7 100644 --- a/README.fr.md +++ b/README.fr.md @@ -235,6 +235,14 @@ Garanties : - **Seulement les constats défendables.** Deux vérifications sont exclues de la ligne en ligne sur la base des preuves de l'eval, et tout ce qui est proche d'un seuil est écarté plutôt qu'affiché. +- **Pas de 0/100.** Quand aucune vérification décidée ne passe, la notice + indique ce qui manque, sans le chiffre. + +**Les relances sont lues en contexte.** Le premier prompt d'une session est noté +seul. Un prompt suivant est envoyé avec les deux prompts qui le précèdent dans la +même session, et seul le nouveau est noté. Activé par défaut ; +`JEVPROMPTCOACH_SESSION_CONTEXT=0` le désactive. Les seuils ont été calibrés sur +des prompts notés seuls. ## Confidentialité @@ -261,7 +269,7 @@ assigné à un nom finissant par `KEY`/`TOKEN`/`SECRET`/`PASSWORD`. | `/jevpromptcoach:score` | Le seul prompt que vous avez passé, expurgé | | `/jevpromptcoach:report` | Les prompts journalisés pas encore notés, expurgés, par lots | | `config backfill` | Votre historique, expurgé, par lots — **après** une estimation de coût et une confirmation explicite | -| mode `always` | Chaque prompt au moment où vous l'envoyez, expurgé | +| mode `always` | Chaque prompt au moment où vous l'envoyez, expurgé, plus jusqu'à deux prompts précédents de la même session comme contexte, expurgés eux aussi | | Sinon, jamais | Rien | Aucune télémétrie. Aucune autre destination réseau. La clé d'API est lue depuis @@ -269,7 +277,7 @@ l'environnement ou le fichier de clé, et n'est jamais journalisée, affichée, incluse dans un message d'erreur. Un prompt n'est noté qu'une fois. Les résultats sont mis en cache par empreinte -du contenu. +du contenu, sauf pour une relance en mode `always`, dont le contexte change. ## Contribuer diff --git a/README.md b/README.md index 58d0bac..4928135 100644 --- a/README.md +++ b/README.md @@ -288,6 +288,19 @@ Guarantees: - **Only findings we can stand behind.** Two checks are barred from the inline line entirely on the eval evidence below, and anything near a threshold is dropped rather than shown. +- **No 0/100.** The inline score counts only the checks decided with + confidence, often three or four of them, so a 0 said less than it looked. When + none of those pass, the notice shows what is missing and leaves the number off. + +**Follow-ups are read in context.** The first prompt of a session is scored on +its own, because it has to carry everything the agent needs. A later prompt is +sent with the two prompts before it in the same session, and only the new one +is scored: "commit and push all changes" is a fine follow-up once the earlier +prompt said what the change was. This is on by default. Set +`JEVPROMPTCOACH_SESSION_CONTEXT=0`, in your environment or in +`~/.claude/jevpromptcoach/.env`, to score every prompt alone. The thresholds +were tuned on prompts scored alone; follow-up scores have not yet been through +the eval. `always` does not use the mechanism the docs suggest. Writing to stderr with a non-zero exit displays nothing on Claude Code 2.1.277; a top-level @@ -322,7 +335,7 @@ secrets, `Bearer` tokens, and anything assigned to a name ending in | `/jevpromptcoach:score` | The one prompt you passed, redacted | | `/jevpromptcoach:report` | Any logged prompts not yet scored, redacted, batched | | `config backfill` | Your history, redacted, batched — **after** a cost estimate and an explicit confirmation | -| `always` mode | Each prompt as you submit it, redacted | +| `always` mode | Each prompt as you submit it, redacted, plus up to two earlier prompts from the same session as context, redacted again at the current level | | Ever, otherwise | Nothing | No telemetry. No other network destination. The API key is read from the @@ -330,7 +343,8 @@ environment or the key file, and never logged, printed, or included in an error message — error text is scrubbed of it on the way out. A prompt is scored once. Results are cached by content hash, so unchanged text is -never re-sent. +never re-sent. The exception is a follow-up in `always` mode, which is scored +fresh each time because its context differs. ## Backfill diff --git a/dist/chunk-JBKH57J6.js b/dist/chunk-7PP552KK.js similarity index 76% rename from dist/chunk-JBKH57J6.js rename to dist/chunk-7PP552KK.js index b6807dd..de370d5 100644 --- a/dist/chunk-JBKH57J6.js +++ b/dist/chunk-7PP552KK.js @@ -37,16 +37,23 @@ function saveConfig(config) { } var ENV_PATH = join(DATA_DIR, ".env"); function apiKey() { - const fromEnv = process.env.TYPESAFE_API_KEY?.trim(); + return envValue("TYPESAFE_API_KEY"); +} +function envValue(name) { + const fromEnv = process.env[name]?.trim(); if (fromEnv) return fromEnv; try { - const match = /^\s*TYPESAFE_API_KEY\s*=\s*(.+?)\s*$/m.exec(readFileSync(ENV_PATH, "utf8")); - const key = match?.[1]?.replace(/^['"]|['"]$/g, "").trim(); - return key ? key : null; + const match = new RegExp(`^\\s*${name}\\s*=\\s*(.+?)\\s*$`, "m").exec(readFileSync(ENV_PATH, "utf8")); + const value = match?.[1]?.replace(/^['"]|['"]$/g, "").trim(); + return value ? value : null; } catch { return null; } } +function sessionContextEnabled() { + const value = envValue("JEVPROMPTCOACH_SESSION_CONTEXT"); + return value === null || !/^(0|false|off|no)$/i.test(value); +} function apiKeySource() { if (process.env.TYPESAFE_API_KEY?.trim()) return "environment"; return apiKey() ? "key file" : null; @@ -61,5 +68,6 @@ export { saveConfig, ENV_PATH, apiKey, + sessionContextEnabled, apiKeySource }; diff --git a/dist/chunk-RC2GUYJC.js b/dist/chunk-U3OSFJLP.js similarity index 70% rename from dist/chunk-RC2GUYJC.js rename to dist/chunk-U3OSFJLP.js index 6f3104f..e5cdacc 100644 --- a/dist/chunk-RC2GUYJC.js +++ b/dist/chunk-U3OSFJLP.js @@ -4,10 +4,20 @@ import { CORRECTIONS_PATH, LOG_PATH, ensureDataDir -} from "./chunk-JBKH57J6.js"; +} from "./chunk-7PP552KK.js"; // src/log.ts -import { appendFileSync, existsSync, readFileSync, renameSync, writeFileSync } from "node:fs"; +import { + appendFileSync, + closeSync, + existsSync, + fstatSync, + openSync, + readFileSync, + readSync, + renameSync, + writeFileSync +} from "node:fs"; function hasText(entry) { return entry.text !== null; } @@ -31,6 +41,31 @@ function appendLog(entry) { function readLog() { return readJsonl(LOG_PATH); } +var TAIL_BYTES = 256 * 1024; +function recentSessionPrompts(session, before, limit) { + if (!existsSync(LOG_PATH)) return []; + const fd = openSync(LOG_PATH, "r"); + let tail; + try { + const size = fstatSync(fd).size; + const length = Math.min(size, TAIL_BYTES); + const buffer = Buffer.alloc(length); + readSync(fd, buffer, 0, length, size - length); + tail = buffer.toString("utf8"); + } finally { + closeSync(fd); + } + const out = []; + for (const line of tail.split("\n")) { + if (!line.trim()) continue; + try { + const entry = JSON.parse(line); + if (entry.source === "hook" && entry.session === session && entry.ts < before) out.push(entry); + } catch { + } + } + return out.slice(-limit); +} function appendLogMany(entries) { if (entries.length === 0) return; ensureDataDir(); @@ -81,6 +116,7 @@ export { hasText, appendLog, readLog, + recentSessionPrompts, appendLogMany, clearLocalData, readScores, diff --git a/dist/chunk-BD4OHVLJ.js b/dist/chunk-W564PYU5.js similarity index 94% rename from dist/chunk-BD4OHVLJ.js rename to dist/chunk-W564PYU5.js index 282c53b..d9d496e 100644 --- a/dist/chunk-BD4OHVLJ.js +++ b/dist/chunk-W564PYU5.js @@ -1,7 +1,7 @@ // JevPromptCoach — generated by scripts/build.mjs. Do not edit. import { apiKey -} from "./chunk-JBKH57J6.js"; +} from "./chunk-7PP552KK.js"; // src/checks.ts var GATES = [ @@ -239,9 +239,11 @@ function clampPrompt(text) { \u2026 ${text.slice(-1e3)}`; } -function questionsFor(id) { +function questionsFor(id, contextIds = []) { const questions = {}; - const scope = `Consider only the message with id "${id}" in the state.`; + const scope = contextIds.length ? `Judge only the message with id "${id}" in the state. The messages ${contextIds.map((c) => `"${c}"`).join( + " and " + )} are earlier prompts from the same conversation, given as context: anything they already state counts as known to the reader of "${id}", but they are not themselves being judged.` : `Consider only the message with id "${id}" in the state.`; for (const gate of GATES) { questions[`${id}__${gate.id}`] = { type: "noul", @@ -347,11 +349,19 @@ async function scoreMany(inputs, options = {}) { ); } async function scoreOne(text, hash, options = {}) { - const answers = await tryAsk({ messages: [{ id: "m0", text: clampPrompt(text) }] }, questionsFor("m0"), { - timeoutMs: options.timeoutMs ?? 2e4 - }); + const context = (options.context ?? []).map((t, i) => ({ id: `c${i + 1}`, text: clampPrompt(t) })); + const state = { messages: [...context, { id: "m0", text: clampPrompt(text) }] }; + const answers = await tryAsk( + state, + questionsFor( + "m0", + context.map((c) => c.id) + ), + { timeoutMs: options.timeoutMs ?? 2e4 } + ); if (!answers) return null; const record = unpack(answers, "m0", hash, (/* @__PURE__ */ new Date()).toISOString()); + if (context.length) record.context = context.length; return { record, result: interpret(hash, record.probabilities, record.gates, options) }; } diff --git a/dist/cli.js b/dist/cli.js index e5c5a50..00ac818 100644 --- a/dist/cli.js +++ b/dist/cli.js @@ -3,9 +3,6 @@ import { promptHash, skipReason } from "./chunk-F3M3WE3B.js"; -import { - applyPrivacy -} from "./chunk-VM4R2HGT.js"; import { appendLogMany, appendScores, @@ -16,7 +13,7 @@ import { readLog, readScores, writeCorrections -} from "./chunk-RC2GUYJC.js"; +} from "./chunk-U3OSFJLP.js"; import { CHECKS, CORRECTION_QUESTION, @@ -30,7 +27,7 @@ import { runPool, scoreMany, scoreOne -} from "./chunk-BD4OHVLJ.js"; +} from "./chunk-W564PYU5.js"; import { ENV_PATH, LOG_PATH, @@ -38,7 +35,10 @@ import { apiKeySource, loadConfig, saveConfig -} from "./chunk-JBKH57J6.js"; +} from "./chunk-7PP552KK.js"; +import { + applyPrivacy +} from "./chunk-VM4R2HGT.js"; // src/cli.ts import { readFileSync, writeFileSync } from "node:fs"; diff --git a/dist/eval.js b/dist/eval.js index 9d1e268..428462b 100644 --- a/dist/eval.js +++ b/dist/eval.js @@ -5,10 +5,10 @@ import { MODEL, USD_PER_INPUT_TOKEN, scoreMany -} from "./chunk-BD4OHVLJ.js"; +} from "./chunk-W564PYU5.js"; import { apiKey -} from "./chunk-JBKH57J6.js"; +} from "./chunk-7PP552KK.js"; // src/eval.ts import { existsSync, readFileSync, writeFileSync } from "node:fs"; diff --git a/dist/hook.js b/dist/hook.js index 4c0cdbb..0473608 100644 --- a/dist/hook.js +++ b/dist/hook.js @@ -3,15 +3,15 @@ import { promptHash, skipReason } from "./chunk-F3M3WE3B.js"; -import { - applyPrivacy -} from "./chunk-VM4R2HGT.js"; import { appendLog -} from "./chunk-RC2GUYJC.js"; +} from "./chunk-U3OSFJLP.js"; import { loadConfig -} from "./chunk-JBKH57J6.js"; +} from "./chunk-7PP552KK.js"; +import { + applyPrivacy +} from "./chunk-VM4R2HGT.js"; // src/hook.ts import { readFileSync } from "node:fs"; @@ -52,8 +52,8 @@ async function main() { } if (config.mode !== "always") return; if (stored === null) return; - const { runInline } = await import("./inline-4KJUBKYU.js"); - const line = await runInline(stored, hash, config); + const { runInline } = await import("./inline-6S7NCB56.js"); + const line = await runInline(stored, hash, config, { session: entry.session, ts: entry.ts }); if (line) emitLine(line); } main().then( diff --git a/dist/inline-4KJUBKYU.js b/dist/inline-6S7NCB56.js similarity index 50% rename from dist/inline-4KJUBKYU.js rename to dist/inline-6S7NCB56.js index 2661964..6d560f9 100644 --- a/dist/inline-4KJUBKYU.js +++ b/dist/inline-6S7NCB56.js @@ -1,40 +1,58 @@ // JevPromptCoach — generated by scripts/build.mjs. Do not edit. import { appendScores, - readScores -} from "./chunk-RC2GUYJC.js"; + readScores, + recentSessionPrompts +} from "./chunk-U3OSFJLP.js"; import { CHECKS, interpret, scoreOne -} from "./chunk-BD4OHVLJ.js"; -import "./chunk-JBKH57J6.js"; +} from "./chunk-W564PYU5.js"; +import { + sessionContextEnabled +} from "./chunk-7PP552KK.js"; +import { + applyPrivacy +} from "./chunk-VM4R2HGT.js"; // src/inline.ts +var CONTEXT_PROMPTS = 2; function format(result) { const failures = result.checks.filter((c) => c.verdict === "fail"); if (failures.length === 0) return null; const order = new Map(CHECKS.map((c, i) => [c.id, i])); const worst = failures.toSorted((a, b) => (a.probability ?? 1) - (b.probability ?? 1)).slice(0, 2).toSorted((a, b) => (order.get(a.id) ?? 0) - (order.get(b.id) ?? 0)); const missing = worst.map((c) => c.def.shortfall).join(", "); - const head = result.score === null ? "Jev (Prompt Coach)" : `Jev (Prompt Coach) - ${result.score}/100`; + const head = result.score ? `Jev (Prompt Coach) - ${result.score}/100` : "Jev (Prompt Coach)"; return `${head} Missing: ${missing}.`; } -async function runInline(redactedText, hash, config) { +function sessionContext(config, session, ts) { + if (session === "unknown" || !sessionContextEnabled()) return []; try { - const cached = readScores().get(hash); - if (cached) { - return format(interpret(hash, cached.probabilities, cached.gates, { inlineSafe: true })); - } + return recentSessionPrompts(session, ts, CONTEXT_PROMPTS).map((entry) => entry.text === null ? null : applyPrivacy(entry.text, config.privacy).text).filter((text) => text !== null); } catch { + return []; + } +} +async function runInline(redactedText, hash, config, at) { + const context = sessionContext(config, at.session, at.ts); + if (context.length === 0) { + try { + const cached = readScores().get(hash); + if (cached && !cached.context) { + return format(interpret(hash, cached.probabilities, cached.gates, { inlineSafe: true })); + } + } catch { + } } const deadline = new Promise((resolve) => { const timer = setTimeout(() => resolve(null), config.alwaysTimeoutMs); timer.unref?.(); }); const scored = await Promise.race([ - scoreOne(redactedText, hash, { timeoutMs: config.alwaysTimeoutMs, inlineSafe: true }), + scoreOne(redactedText, hash, { timeoutMs: config.alwaysTimeoutMs, inlineSafe: true, context }), deadline ]); if (!scored) return null; diff --git a/package.json b/package.json index 0491d8a..1529f4c 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "jevpromptcoach", - "version": "0.2.7", + "version": "0.3.0", "description": "Scores how well you prompt a coding agent, and shows your habits improving over time. Runs on TypeSafe Jev.", "license": "MIT", "private": false, diff --git a/src/config.ts b/src/config.ts index ba8b99d..06fe219 100644 --- a/src/config.ts +++ b/src/config.ts @@ -66,17 +66,35 @@ export const ENV_PATH = join(DATA_DIR, '.env'); * The file is created 0600 and is the developer's to delete. */ export function apiKey(): string | null { - const fromEnv = process.env.TYPESAFE_API_KEY?.trim(); + return envValue('TYPESAFE_API_KEY'); +} + +/** + * A setting from the environment, falling back to the key file for the same + * reason the key does: the hook does not run under the developer's shell + * profile. `name` is always a constant from this codebase, never user input. + */ +export function envValue(name: string): string | null { + const fromEnv = process.env[name]?.trim(); if (fromEnv) return fromEnv; try { - const match = /^\s*TYPESAFE_API_KEY\s*=\s*(.+?)\s*$/m.exec(readFileSync(ENV_PATH, 'utf8')); - const key = match?.[1]?.replace(/^['"]|['"]$/g, '').trim(); - return key ? key : null; + const match = new RegExp(`^\\s*${name}\\s*=\\s*(.+?)\\s*$`, 'm').exec(readFileSync(ENV_PATH, 'utf8')); + const value = match?.[1]?.replace(/^['"]|['"]$/g, '').trim(); + return value ? value : null; } catch { return null; } } +/** + * Whether `always` mode scores a follow-up against the earlier prompts in its + * session. On unless JEVPROMPTCOACH_SESSION_CONTEXT is set to 0, false, off or no. + */ +export function sessionContextEnabled(): boolean { + const value = envValue('JEVPROMPTCOACH_SESSION_CONTEXT'); + return value === null || !/^(0|false|off|no)$/i.test(value); +} + /** * Where the key was found, for reporting. Never returns the key itself. * `null` means no key is available and nothing can be scored. diff --git a/src/hook.ts b/src/hook.ts index 9aeb11f..e7b8e99 100644 --- a/src/hook.ts +++ b/src/hook.ts @@ -78,10 +78,11 @@ async function main(): Promise { // From here on we are in `always` mode and may call the API. What goes out is // `stored` — the text after the configured privacy level has been applied — - // never the raw prompt. The budget is hard: whatever has not answered by then - // is abandoned and nothing prints. + // never the raw prompt; earlier prompts sent as context are re-redacted in + // inline.ts for the same reason. The budget is hard: whatever has not + // answered by then is abandoned and nothing prints. const { runInline } = await import('./inline.js'); - const line = await runInline(stored, hash, config); + const line = await runInline(stored, hash, config, { session: entry.session, ts: entry.ts }); if (line) emitLine(line); } diff --git a/src/inline.ts b/src/inline.ts index 51d5218..8288081 100644 --- a/src/inline.ts +++ b/src/inline.ts @@ -3,10 +3,14 @@ * never loads it, and with it never loads the Jev client. */ import { CHECKS } from './checks.js'; -import type { Config } from './config.js'; -import { appendScores, readScores } from './log.js'; +import { type Config, sessionContextEnabled } from './config.js'; +import { appendScores, readScores, recentSessionPrompts } from './log.js'; +import { applyPrivacy } from './redact.js'; import { interpret, type PromptScore, scoreOne } from './score.js'; +/** Earlier prompts from the session sent with a follow-up. */ +const CONTEXT_PROMPTS = 2; + function format(result: PromptScore): string | null { // Only confident failures reach the line. `inlineSafe` has already demoted // anything near a threshold to `undecided`, so what is left is worth saying. @@ -25,26 +29,58 @@ function format(result: PromptScore): string | null { const missing = worst.map((c) => c.def.shortfall).join(', '); // Claude Code prefixes the first line with "UserPromptSubmit says:" and keeps // line breaks, so the score shares that line and the detail sits under it. - const head = result.score === null ? 'Jev (Prompt Coach)' : `Jev (Prompt Coach) - ${result.score}/100`; + // A 0 is left off: the inline score counts only the few checks decided with + // confidence, so 0 usually means "0 of 3" and reads as if nothing arrived. + // What is missing is the useful part either way. + const head = result.score ? `Jev (Prompt Coach) - ${result.score}/100` : 'Jev (Prompt Coach)'; return `${head}\nMissing: ${missing}.`; } +/** + * Earlier prompts from this session, re-run through the privacy level: a + * prompt logged under `raw` must still be redacted if the level is now + * `redact`. Empty for the first prompt of a session, which is judged alone + * because it has to carry everything the agent needs. + */ +function sessionContext(config: Config, session: string, ts: string): string[] { + if (session === 'unknown' || !sessionContextEnabled()) return []; + try { + return recentSessionPrompts(session, ts, CONTEXT_PROMPTS) + .map((entry) => (entry.text === null ? null : applyPrivacy(entry.text, config.privacy).text)) + .filter((text): text is string => text !== null); + } catch { + return []; + } +} + /** * @param redactedText The prompt *after* the configured privacy level has been * applied. The caller must never pass the raw prompt: this is the last hop * before the API and it does no redaction of its own. * @param hash Content hash of the ORIGINAL text, so the cache key is stable * across privacy-level changes. + * @param at The session and submission time of this prompt, to find the + * prompts before it. */ -export async function runInline(redactedText: string, hash: string, config: Config): Promise { - // A prompt whose text has not changed is never scored twice. - try { - const cached = readScores().get(hash); - if (cached) { - return format(interpret(hash, cached.probabilities, cached.gates, { inlineSafe: true })); +export async function runInline( + redactedText: string, + hash: string, + config: Config, + at: { session: string; ts: string }, +): Promise { + const context = sessionContext(config, at.session, at.ts); + + // A prompt whose text has not changed is never scored twice, unless it is a + // follow-up: "commit and push" means something different in every session. + if (context.length === 0) { + try { + const cached = readScores().get(hash); + if (cached && !cached.context) { + return format(interpret(hash, cached.probabilities, cached.gates, { inlineSafe: true })); + } + } catch { + /* cache unreadable; score it fresh */ } - } catch { - /* cache unreadable; score it fresh */ } const deadline = new Promise((resolve) => { @@ -53,7 +89,7 @@ export async function runInline(redactedText: string, hash: string, config: Conf }); const scored = await Promise.race([ - scoreOne(redactedText, hash, { timeoutMs: config.alwaysTimeoutMs, inlineSafe: true }), + scoreOne(redactedText, hash, { timeoutMs: config.alwaysTimeoutMs, inlineSafe: true, context }), deadline, ]); if (!scored) return null; diff --git a/src/log.ts b/src/log.ts index 66d053a..8f03736 100644 --- a/src/log.ts +++ b/src/log.ts @@ -3,7 +3,17 @@ * live under ~/.claude/jevpromptcoach/. Nothing here ever leaves the machine; * the commands read it, and only a command sends anything to the API. */ -import { appendFileSync, existsSync, readFileSync, renameSync, writeFileSync } from 'node:fs'; +import { + appendFileSync, + closeSync, + existsSync, + fstatSync, + openSync, + readFileSync, + readSync, + renameSync, + writeFileSync, +} from 'node:fs'; import type { CheckId, GateId } from './checks.js'; import { CACHE_PATH, CORRECTIONS_PATH, ensureDataDir, LOG_PATH } from './config.js'; import type { Features } from './redact.js'; @@ -39,6 +49,12 @@ export interface ScoreRecord { /** Raw probability per applicability gate. */ gates: Partial>; model: string; + /** + * How many earlier prompts from the session rode along as context. Absent for + * a prompt scored on its own. A score that depended on its context is not + * served from the cache, because the same text elsewhere had other context. + */ + context?: number; } /** The correction-rate verdict for the prompt with this hash. */ @@ -75,6 +91,42 @@ export function readLog(): LogEntry[] { return readJsonl(LOG_PATH); } +/** Enough for dozens of recent prompts; the log is never read whole on the prompt path. */ +const TAIL_BYTES = 256 * 1024; + +/** + * The last `limit` hook prompts from `session` logged before `before`, oldest + * first. Reads only the tail of the log, so a session whose earlier prompts + * have scrolled out of it is treated as having none. + */ +export function recentSessionPrompts(session: string, before: string, limit: number): LogEntry[] { + if (!existsSync(LOG_PATH)) return []; + const fd = openSync(LOG_PATH, 'r'); + let tail: string; + try { + const size = fstatSync(fd).size; + const length = Math.min(size, TAIL_BYTES); + const buffer = Buffer.alloc(length); + readSync(fd, buffer, 0, length, size - length); + tail = buffer.toString('utf8'); + } finally { + closeSync(fd); + } + + const out: LogEntry[] = []; + // The first line may be cut mid-way by the tail boundary; it fails to parse and is skipped. + for (const line of tail.split('\n')) { + if (!line.trim()) continue; + try { + const entry = JSON.parse(line) as LogEntry; + if (entry.source === 'hook' && entry.session === session && entry.ts < before) out.push(entry); + } catch { + /* torn line */ + } + } + return out.slice(-limit); +} + export function appendLogMany(entries: LogEntry[]): void { if (entries.length === 0) return; ensureDataDir(); diff --git a/src/score.ts b/src/score.ts index 2938a4c..0d19f7c 100644 --- a/src/score.ts +++ b/src/score.ts @@ -43,9 +43,20 @@ export function clampPrompt(text: string): string { return `${text.slice(0, MAX_PROMPT_CHARS - 1000)}\n…\n${text.slice(-1000)}`; } -function questionsFor(id: string): Record { +/** + * @param contextIds Earlier messages from the same session that are in the + * state as background. Only `id` is judged; what the context already told + * the agent counts as known. + */ +function questionsFor(id: string, contextIds: string[] = []): Record { const questions: Record = {}; - const scope = `Consider only the message with id "${id}" in the state.`; + const scope = contextIds.length + ? `Judge only the message with id "${id}" in the state. The messages ${contextIds + .map((c) => `"${c}"`) + .join( + ' and ', + )} are earlier prompts from the same conversation, given as context: anything they already state counts as known to the reader of "${id}", but they are not themselves being judged.` + : `Consider only the message with id "${id}" in the state.`; for (const gate of GATES) { questions[`${id}__${gate.id}`] = { type: 'noul', @@ -208,17 +219,30 @@ export async function scoreMany(inputs: ScoreInput[], options: ScoreRunOptions = ); } -/** Score a single prompt. Used by /jevpromptcoach:score and by `always` mode. */ +/** + * Score a single prompt. Used by /jevpromptcoach:score and by `always` mode. + * + * `context` is earlier prompts from the same session, oldest first, already + * through the configured privacy level. They are sent but not scored. + */ export async function scoreOne( text: string, hash: string, - options: { timeoutMs?: number; inlineSafe?: boolean } = {}, + options: { timeoutMs?: number; inlineSafe?: boolean; context?: string[] } = {}, ): Promise<{ record: ScoreRecord; result: PromptScore } | null> { - const answers = await tryAsk({ messages: [{ id: 'm0', text: clampPrompt(text) }] }, questionsFor('m0'), { - timeoutMs: options.timeoutMs ?? 20_000, - }); + const context = (options.context ?? []).map((t, i) => ({ id: `c${i + 1}`, text: clampPrompt(t) })); + const state = { messages: [...context, { id: 'm0', text: clampPrompt(text) }] }; + const answers = await tryAsk( + state, + questionsFor( + 'm0', + context.map((c) => c.id), + ), + { timeoutMs: options.timeoutMs ?? 20_000 }, + ); if (!answers) return null; const record = unpack(answers, 'm0', hash, new Date().toISOString()); + if (context.length) record.context = context.length; return { record, result: interpret(hash, record.probabilities, record.gates, options) }; } diff --git a/test/hook-privacy.test.mjs b/test/hook-privacy.test.mjs index 4a4ee6d..6981f37 100644 --- a/test/hook-privacy.test.mjs +++ b/test/hook-privacy.test.mjs @@ -26,6 +26,8 @@ const PROMPT = `Deploy with ${SECRET} from /Users/someone/clients/acme/app/main. let server; let captured; let port; +/** What the capture server answers for each question id. */ +let answerFor = () => 0.01; before(async () => { captured = []; @@ -42,7 +44,7 @@ before(async () => { } const parsed = JSON.parse(body); const answers = {}; - for (const k of Object.keys(parsed.questions ?? {})) answers[k] = { type: 'noul', noul: 0.01 }; + for (const k of Object.keys(parsed.questions ?? {})) answers[k] = { type: 'noul', noul: answerFor(k) }; res.writeHead(200, { 'content-type': 'application/json' }); res.end(JSON.stringify({ model: 'test', answers, usage: { input_tokens: 1, output_tokens: 0 } })); }); @@ -58,16 +60,28 @@ after(() => server?.close()); * spawn would block the event loop and the server could never accept the * hook's connection. */ -function runHook(privacy, prompt = PROMPT) { +function runHook(privacy, prompt = PROMPT, { log = [], env = {} } = {}) { const home = mkdtempSync(join(tmpdir(), 'jpc-test-')); mkdirSync(join(home, '.claude', 'jevpromptcoach'), { recursive: true }); + if (log.length) { + writeFileSync( + join(home, '.claude', 'jevpromptcoach', 'prompts.jsonl'), + log.map((e) => JSON.stringify({ features: {}, source: 'hook', ...e })).join('\n') + '\n', + ); + } writeFileSync(join(home, '.claude', 'jevpromptcoach', '.env'), 'TYPESAFE_API_KEY=not-a-real-key\n'); writeFileSync( join(home, '.claude', 'jevpromptcoach', 'config.json'), JSON.stringify({ mode: 'always', privacy, setupComplete: true, alwaysTimeoutMs: 8000, bypassPrefix: '*' }), ); const child = spawn(process.execPath, ['dist/hook.js'], { - env: { ...process.env, HOME: home, TYPESAFE_BASE_URL: `http://127.0.0.1:${port}` }, + env: { + ...process.env, + JEVPROMPTCOACH_SESSION_CONTEXT: '', + ...env, + HOME: home, + TYPESAFE_BASE_URL: `http://127.0.0.1:${port}`, + }, }); let stdout = ''; child.stdout.on('data', (c) => { @@ -95,6 +109,76 @@ test('always mode sends the redacted prompt, never the raw one', async () => { assert.ok(sent.includes('main.ts'), 'the filename should survive redaction'); }); +/** + * A follow-up carries the two prompts before it in the same session. They are + * written to the log as raw text, as if captured under `raw`, so the test also + * proves they are redacted again on the way out under `redact`. + */ +const EARLIER = [ + { ts: '2026-01-01T00:00:00.000Z', session: 's', hash: 'h0', text: 'oldest prompt, beyond the window' }, + { ts: '2026-01-01T00:01:00.000Z', session: 'other', hash: 'hx', text: 'a prompt from another session' }, + { ts: '2026-01-01T00:02:00.000Z', session: 's', hash: 'h1', text: `Mail bob@acme.com about ${SECRET}` }, + { ts: '2026-01-01T00:03:00.000Z', session: 's', hash: 'h2', text: 'Open /Users/someone/clients/acme/app/main.ts' }, +]; + +test('a follow-up sends the two earlier prompts, redacted, and scores only the last', async () => { + captured.length = 0; + const result = await runHook('redact', 'Now commit and push all of it', { log: EARLIER }); + assert.equal(result.status, 0); + assert.equal(captured.length, 1); + + const { state, questions } = captured[0]; + const texts = state.messages.map((m) => m.text); + assert.equal(texts.length, 3, 'two context prompts plus the one being scored'); + assert.equal(state.messages.at(-1).id, 'm0'); + assert.equal(texts.at(-1), 'Now commit and push all of it'); + assert.ok(!texts.some((t) => t.includes('oldest') || t.includes('another session'))); + + const sent = JSON.stringify(captured[0]); + assert.ok(!sent.includes(SECRET), 'a context prompt leaked the API key'); + assert.ok(!sent.includes('bob@acme.com'), 'a context prompt leaked the email'); + assert.ok(!sent.includes('clients/acme'), 'a context prompt leaked the client path'); + assert.ok(sent.includes('main.ts')); + + assert.ok( + Object.keys(questions).every((k) => k.startsWith('m0__')), + 'context prompts must not be scored', + ); +}); + +test('the first prompt of a session is scored on its own', async () => { + captured.length = 0; + const log = EARLIER.filter((e) => e.session === 'other'); + await runHook('redact', 'Now commit and push all of it', { log }); + assert.equal(captured[0].state.messages.length, 1); +}); + +test('JEVPROMPTCOACH_SESSION_CONTEXT=0 turns the context off', async () => { + captured.length = 0; + await runHook('redact', 'Now commit and push all of it', { + log: EARLIER, + env: { JEVPROMPTCOACH_SESSION_CONTEXT: '0' }, + }); + assert.equal(captured[0].state.messages.length, 1); +}); + +test('a score of 0 is not shown, only what is missing', async () => { + answerFor = () => 0.01; + const { systemMessage } = JSON.parse((await runHook('redact')).stdout); + assert.ok(!systemMessage.includes('/100'), systemMessage); + assert.match(systemMessage, /^Jev \(Prompt Coach\)\nMissing: /); +}); + +test('a score above 0 is shown', async () => { + answerFor = (k) => (k.endsWith('__named_target') ? 0.99 : 0.01); + try { + const { systemMessage } = JSON.parse((await runHook('redact')).stdout); + assert.match(systemMessage, /^Jev \(Prompt Coach\) - \d+\/100\nMissing: /); + } finally { + answerFor = () => 0.01; + } +}); + test('metadata_only sends nothing at all', async () => { captured.length = 0; const result = await runHook('metadata_only'); From fc773d3de8f5fa74cb01e14f923c1b0c63c0d25c Mon Sep 17 00:00:00 2001 From: Prateek Kathal Date: Thu, 24 Sep 2026 16:50:29 -0400 Subject: [PATCH 2/2] fix: keep standalone cache entries when a follow-up is scored in context - readScores() no longer lets a context-based record displace a standalone one for the same hash. Before, the latest record won, so a follow-up scored in context evicted the standalone score: the next first prompt with that text paid for a fresh call, and compactScores after a backfill dropped the standalone record for good. Raised by CodeRabbit on #3. - /jevpromptcoach:score skipped the context guard the inline path had, so it could print a session's context-based score as "cached" for the text judged alone. - Tests for the saved-score rules: a context-based record is never served to a first prompt or to the score command, a standalone record is, a follow-up always scores fresh, and a later context record does not displace a standalone one. Each fails with its guard removed. --- dist/{chunk-U3OSFJLP.js => chunk-HXIO5BL2.js} | 6 +- dist/cli.js | 4 +- dist/hook.js | 4 +- ...{inline-6S7NCB56.js => inline-STU2ZBEP.js} | 2 +- src/cli.ts | 3 +- src/log.ts | 12 ++- test/hook-privacy.test.mjs | 75 +++++++++++++++++-- 7 files changed, 93 insertions(+), 13 deletions(-) rename dist/{chunk-U3OSFJLP.js => chunk-HXIO5BL2.js} (94%) rename dist/{inline-6S7NCB56.js => inline-STU2ZBEP.js} (98%) diff --git a/dist/chunk-U3OSFJLP.js b/dist/chunk-HXIO5BL2.js similarity index 94% rename from dist/chunk-U3OSFJLP.js rename to dist/chunk-HXIO5BL2.js index e5cdacc..e750922 100644 --- a/dist/chunk-U3OSFJLP.js +++ b/dist/chunk-HXIO5BL2.js @@ -80,7 +80,11 @@ function clearLocalData() { } function readScores() { const map = /* @__PURE__ */ new Map(); - for (const record of readJsonl(CACHE_PATH)) map.set(record.hash, record); + for (const record of readJsonl(CACHE_PATH)) { + const existing = map.get(record.hash); + if (record.context && existing && !existing.context) continue; + map.set(record.hash, record); + } return map; } function appendScores(records) { diff --git a/dist/cli.js b/dist/cli.js index 00ac818..f2ed7a1 100644 --- a/dist/cli.js +++ b/dist/cli.js @@ -13,7 +13,7 @@ import { readLog, readScores, writeCorrections -} from "./chunk-U3OSFJLP.js"; +} from "./chunk-HXIO5BL2.js"; import { CHECKS, CORRECTION_QUESTION, @@ -501,7 +501,7 @@ async function cmdScore(argv, stdinText) { const sendable = safe ?? text; const hash = promptHash(text); const cached = readScores().get(hash); - if (cached) { + if (cached && !cached.context) { out(renderScore(text, interpret(hash, cached.probabilities, cached.gates))); out(""); out("(cached \u2014 this exact text was scored before)"); diff --git a/dist/hook.js b/dist/hook.js index 0473608..e98e591 100644 --- a/dist/hook.js +++ b/dist/hook.js @@ -5,7 +5,7 @@ import { } from "./chunk-F3M3WE3B.js"; import { appendLog -} from "./chunk-U3OSFJLP.js"; +} from "./chunk-HXIO5BL2.js"; import { loadConfig } from "./chunk-7PP552KK.js"; @@ -52,7 +52,7 @@ async function main() { } if (config.mode !== "always") return; if (stored === null) return; - const { runInline } = await import("./inline-6S7NCB56.js"); + const { runInline } = await import("./inline-STU2ZBEP.js"); const line = await runInline(stored, hash, config, { session: entry.session, ts: entry.ts }); if (line) emitLine(line); } diff --git a/dist/inline-6S7NCB56.js b/dist/inline-STU2ZBEP.js similarity index 98% rename from dist/inline-6S7NCB56.js rename to dist/inline-STU2ZBEP.js index 6d560f9..8935730 100644 --- a/dist/inline-6S7NCB56.js +++ b/dist/inline-STU2ZBEP.js @@ -3,7 +3,7 @@ import { appendScores, readScores, recentSessionPrompts -} from "./chunk-U3OSFJLP.js"; +} from "./chunk-HXIO5BL2.js"; import { CHECKS, interpret, diff --git a/src/cli.ts b/src/cli.ts index 6e7fde1..9357ae0 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -88,8 +88,9 @@ async function cmdScore(argv: string[], stdinText?: string): Promise { const sendable = safe ?? text; const hash = promptHash(text); + // A score that depended on session context is not this text judged alone. const cached = readScores().get(hash); - if (cached) { + if (cached && !cached.context) { out(renderScore(text, interpret(hash, cached.probabilities, cached.gates))); out(''); out('(cached — this exact text was scored before)'); diff --git a/src/log.ts b/src/log.ts index 8f03736..bfa8934 100644 --- a/src/log.ts +++ b/src/log.ts @@ -141,9 +141,19 @@ export function clearLocalData(): void { writeFileSync(CORRECTIONS_PATH, '[]', { mode: 0o600 }); } +/** + * One record per hash, the latest winning, except that a score which depended + * on session context never displaces one of the text judged alone: the + * standalone record is the one a cache hit may serve, and compactScores keeps + * only what this returns. + */ export function readScores(): Map { const map = new Map(); - for (const record of readJsonl(CACHE_PATH)) map.set(record.hash, record); + for (const record of readJsonl(CACHE_PATH)) { + const existing = map.get(record.hash); + if (record.context && existing && !existing.context) continue; + map.set(record.hash, record); + } return map; } diff --git a/test/hook-privacy.test.mjs b/test/hook-privacy.test.mjs index 6981f37..6949773 100644 --- a/test/hook-privacy.test.mjs +++ b/test/hook-privacy.test.mjs @@ -13,6 +13,7 @@ import { test, before, after } from 'node:test'; import assert from 'node:assert/strict'; import { createServer } from 'node:http'; import { spawn } from 'node:child_process'; +import { createHash } from 'node:crypto'; import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs'; import { tmpdir } from 'node:os'; import { join } from 'node:path'; @@ -60,7 +61,7 @@ after(() => server?.close()); * spawn would block the event loop and the server could never accept the * hook's connection. */ -function runHook(privacy, prompt = PROMPT, { log = [], env = {} } = {}) { +function runHook(privacy, prompt = PROMPT, { log = [], scores = [], env = {} } = {}) { const home = mkdtempSync(join(tmpdir(), 'jpc-test-')); mkdirSync(join(home, '.claude', 'jevpromptcoach'), { recursive: true }); if (log.length) { @@ -69,6 +70,12 @@ function runHook(privacy, prompt = PROMPT, { log = [], env = {} } = {}) { log.map((e) => JSON.stringify({ features: {}, source: 'hook', ...e })).join('\n') + '\n', ); } + if (scores.length) { + writeFileSync( + join(home, '.claude', 'jevpromptcoach', 'scores.jsonl'), + scores.map((r) => JSON.stringify(r)).join('\n') + '\n', + ); + } writeFileSync(join(home, '.claude', 'jevpromptcoach', '.env'), 'TYPESAFE_API_KEY=not-a-real-key\n'); writeFileSync( join(home, '.claude', 'jevpromptcoach', 'config.json'), @@ -109,6 +116,18 @@ test('always mode sends the redacted prompt, never the raw one', async () => { assert.ok(sent.includes('main.ts'), 'the filename should survive redaction'); }); +/** The cache key the hook computes, mirrored from src/hash.ts. */ +const hashOf = (text) => createHash('sha256').update(text.trim(), 'utf8').digest('hex').slice(0, 16); +const FOLLOW_UP = 'Now commit and push all of it'; +const savedScore = (extra = {}) => ({ + hash: hashOf(FOLLOW_UP), + ts: '2026-01-01T00:00:00.000Z', + probabilities: { named_target: 0.99 }, + gates: {}, + model: 'test', + ...extra, +}); + /** * A follow-up carries the two prompts before it in the same session. They are * written to the log as raw text, as if captured under `raw`, so the test also @@ -123,7 +142,7 @@ const EARLIER = [ test('a follow-up sends the two earlier prompts, redacted, and scores only the last', async () => { captured.length = 0; - const result = await runHook('redact', 'Now commit and push all of it', { log: EARLIER }); + const result = await runHook('redact', FOLLOW_UP, { log: EARLIER }); assert.equal(result.status, 0); assert.equal(captured.length, 1); @@ -131,7 +150,7 @@ test('a follow-up sends the two earlier prompts, redacted, and scores only the l const texts = state.messages.map((m) => m.text); assert.equal(texts.length, 3, 'two context prompts plus the one being scored'); assert.equal(state.messages.at(-1).id, 'm0'); - assert.equal(texts.at(-1), 'Now commit and push all of it'); + assert.equal(texts.at(-1), FOLLOW_UP); assert.ok(!texts.some((t) => t.includes('oldest') || t.includes('another session'))); const sent = JSON.stringify(captured[0]); @@ -149,19 +168,65 @@ test('a follow-up sends the two earlier prompts, redacted, and scores only the l test('the first prompt of a session is scored on its own', async () => { captured.length = 0; const log = EARLIER.filter((e) => e.session === 'other'); - await runHook('redact', 'Now commit and push all of it', { log }); + await runHook('redact', FOLLOW_UP, { log }); assert.equal(captured[0].state.messages.length, 1); }); test('JEVPROMPTCOACH_SESSION_CONTEXT=0 turns the context off', async () => { captured.length = 0; - await runHook('redact', 'Now commit and push all of it', { + await runHook('redact', FOLLOW_UP, { log: EARLIER, env: { JEVPROMPTCOACH_SESSION_CONTEXT: '0' }, }); assert.equal(captured[0].state.messages.length, 1); }); +test('a first prompt is not served a score that depended on context', async () => { + captured.length = 0; + await runHook('redact', FOLLOW_UP, { scores: [savedScore({ context: 2 })] }); + assert.equal(captured.length, 1, 'expected a fresh request, not the context-based saved score'); +}); + +test('a first prompt is served a saved standalone score', async () => { + captured.length = 0; + await runHook('redact', FOLLOW_UP, { scores: [savedScore()] }); + assert.equal(captured.length, 0, 'unchanged text scored alone should come from the cache'); +}); + +test('a later score with context does not displace a saved standalone score', async () => { + captured.length = 0; + const later = savedScore({ ts: '2026-01-02T00:00:00.000Z', context: 2 }); + await runHook('redact', FOLLOW_UP, { scores: [savedScore(), later] }); + assert.equal(captured.length, 0, 'the standalone score should still be served from the cache'); +}); + +test('a follow-up is scored fresh even when the same text was saved alone', async () => { + captured.length = 0; + await runHook('redact', FOLLOW_UP, { log: EARLIER, scores: [savedScore()] }); + assert.equal(captured.length, 1, 'expected a fresh request with context'); + assert.equal(captured[0].state.messages.length, 3); +}); + +test('/jevpromptcoach:score ignores a saved score that depended on context', async () => { + captured.length = 0; + const home = mkdtempSync(join(tmpdir(), 'jpc-test-')); + const dir = join(home, '.claude', 'jevpromptcoach'); + mkdirSync(dir, { recursive: true }); + writeFileSync(join(dir, '.env'), 'TYPESAFE_API_KEY=not-a-real-key\n'); + writeFileSync(join(dir, 'scores.jsonl'), JSON.stringify(savedScore({ context: 2 })) + '\n'); + const child = spawn(process.execPath, ['dist/cli.js', 'score', FOLLOW_UP], { + env: { ...process.env, HOME: home, TYPESAFE_BASE_URL: `http://127.0.0.1:${port}` }, + }); + let stdout = ''; + child.stdout.on('data', (c) => { + stdout += c; + }); + await new Promise((resolve) => child.on('close', resolve)); + rmSync(home, { recursive: true, force: true }); + assert.equal(captured.length, 1, 'expected a fresh request, not the context-based saved score'); + assert.ok(!stdout.includes('(cached'), stdout); +}); + test('a score of 0 is not shown, only what is missing', async () => { answerFor = () => 0.01; const { systemMessage } = JSON.parse((await runHook('redact')).stdout);