From 006c556fa833be1923f2548be6c72b339408aff9 Mon Sep 17 00:00:00 2001 From: Ali Ibrahim Jr <48456829+IBJunior@users.noreply.github.com> Date: Sat, 12 Sep 2026 22:33:58 +0200 Subject: [PATCH 1/3] feat(skills): write the expense-reporter skill and its eval cases MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replaces the placeholder body with real guidance: whether to chart at all, which of the two types, how to shape the query, and how to read the result back. The restraint half matters most — nothing in code stops the agent charting a single figure, and a skill that encourages charts without saying when to stop would make answers worse. Adds eval/cases/skills.cases.mts, a defect class the suite did not cover: a capability the agent must choose to consult, and choose not to overuse. Four cases — activation, order, chart-type selection, and the over-triggering guard, which is the one that keeps the rest honest. The fixture already supports this: the seeding formula spreads expenses across nine months with varying totals, so the trend case needed no new seed data and no second source of truth for expected figures. Co-Authored-By: Claude Opus 5 (1M context) --- eval/cases/index.mts | 2 + eval/cases/skills.cases.mts | 81 ++++++++++++++++++++++++++++ skills/expense-reporter/SKILL.md | 93 ++++++++++++++++++++++++++++---- 3 files changed, 165 insertions(+), 11 deletions(-) create mode 100644 eval/cases/skills.cases.mts diff --git a/eval/cases/index.mts b/eval/cases/index.mts index 6c0c5f2..d221ea2 100644 --- a/eval/cases/index.mts +++ b/eval/cases/index.mts @@ -3,6 +3,7 @@ import { cases as analysis } from "./analysis.cases.mts"; import { cases as approval } from "./approval.cases.mts"; import { cases as config } from "./config.cases.mts"; import { cases as csvImport } from "./csvImport.cases.mts"; +import { cases as skills } from "./skills.cases.mts"; import { cases as discipline } from "./discipline.cases.mts"; import { cases as truncation } from "./truncation.cases.mts"; @@ -19,4 +20,5 @@ export const cases: EvalCase[] = [ ...discipline, ...csvImport, ...config, + ...skills, ]; diff --git a/eval/cases/skills.cases.mts b/eval/cases/skills.cases.mts new file mode 100644 index 0000000..b35211f --- /dev/null +++ b/eval/cases/skills.cases.mts @@ -0,0 +1,81 @@ +import { RUN_POLICY } from "../config.mts"; +import { + statesAmount, + toolCalled, + toolCalledBefore, + toolCalledWith, + toolNotCalled, +} from "../graders.mts"; +import { FIXTURE } from "../seed.mts"; +import type { EvalCase } from "../types.mts"; + +/** + * Progressive disclosure: a capability the agent must choose to consult, and choose not to overuse. + * + * Nothing here is checkable without a model. A unit test proves `load_skill` returns the body and + * `render_chart` refuses a bad spec, but not that the agent reaches for the skill at all, reads it + * BEFORE acting, or leaves it alone when a sentence would do. Those are the ways a skill fails in + * practice while every test stays green. + */ + +const { dining } = FIXTURE; + +export const cases: EvalCase[] = [ + { + id: "chart-request-loads-skill", + description: "An explicit chart request should consult the skill that covers charting.", + prompt: "Show me a chart of my spending by category.", + graders: [toolCalled("load_skill"), toolCalled("render_chart")], + runs: RUN_POLICY.single, + tags: ["skills", "charts"], + }, + { + id: "skill-read-before-charting", + description: + "The skill must be read BEFORE the chart is drawn. Loading it afterwards leaves an " + + "identical set of tool calls but means the instructions never informed the work.", + prompt: "Chart my spending by category for me.", + graders: [ + toolCalledBefore("load_skill", "render_chart"), + // Without this the order check passes on a run that charted nothing. + toolCalled("render_chart"), + ], + // Repeats because activation is ~6/8 on claude-haiku-4-5 (3/3 on Sonnet): the model + // sometimes charts without consulting the skill. Whether it loads varies; the ORDER never + // has. A drop to 0-1 of 3 is the regression worth catching. + runs: RUN_POLICY.majority, + tags: ["skills", "charts", "ordering"], + }, + { + id: "trend-question-picks-line", + description: + "A measure over time is a line, not a bar — the one chart-selection judgment with a single " + + "spelling, so it can be graded without a judge.", + prompt: "Show me how my monthly spending has changed over the year.", + graders: [ + toolCalledWith("render_chart", (a) => a.chartType === "line", "chartType=line"), + toolNotCalled("query_transactions"), + ], + // Repeats because chart choice is model judgment, not mechanism — a wrong type here is a + // plausible answer rather than a crash, so one green run would not show it is reliable. + runs: RUN_POLICY.majority, + tags: ["skills", "charts", "chart-selection"], + }, + { + id: "single-figure-needs-no-chart", + description: + "The over-triggering guard. Everything else in this file pushes toward charting; without " + + "this, a prompt or skill edit that makes the agent chart single figures stays green.", + prompt: `How much did I spend on ${dining.category} in total?`, + graders: [ + toolNotCalled("render_chart"), + // The positive half: a run that answered nothing must not pass the negative grader. + toolCalled("run_sql"), + statesAmount(dining.totalMajor), + ], + // Repeats because over-triggering is intermittent by nature: charting a single figure once in + // three is still a regression, and a single run would miss it two times out of three. + runs: RUN_POLICY.majority, + tags: ["skills", "charts", "over-triggering-guard"], + }, +]; diff --git a/skills/expense-reporter/SKILL.md b/skills/expense-reporter/SKILL.md index 459d674..45f8709 100644 --- a/skills/expense-reporter/SKILL.md +++ b/skills/expense-reporter/SKILL.md @@ -5,18 +5,89 @@ description: Turn spending questions into visual reports with charts. Use when t # Expense Reporter -> **Placeholder.** This skill exists so the loader, the prompt section and `load_skill` are -> exercised end to end. Its real instructions — chart selection, the query-to-chart workflow and -> the reporting voice — arrive with the reporting work in the next phase. +## Step 1 — decide whether this needs a chart -## Workflow +Most money questions do not. A chart earns its place only when the **shape** of the data is the +point — a ranking, a trend, a comparison across several things. -1. Establish what the owner is actually asking: a comparison between categories, a trend over - time, a breakdown of a whole, or a single figure. -2. Get the numbers with `run_sql`, aggregating in SQL rather than summing rows by hand. -3. Report the figures plainly, leading with the number that answers the question. +The owner's words decide first. If they asked to **see, visualise, chart, plot or graph** it — or +asked for a report — draw the chart. Otherwise: -## Notes +| The answer is… | Give them | +| ---------------------------------- | ------------------------------------------------------------------------------------- | +| One number | A sentence. `run_sql`, then say the figure. | +| Up to ~6 rows, exact values wanted | The `run_sql` table. Numbers you can read beat bars you have to estimate. | +| ~7 or more categories | A **bar** chart — past a handful, the ranking is easier seen than read. | +| A measure over time (2+ periods) | A **line** chart. A trend is a shape, so chart it even when the rows are few. | +| More than ~12 categories | Top N with `ORDER BY … LIMIT`, plus "and N others". A crowded chart hides the answer. | -- A single figure does not need a chart. Say the number. -- Amounts are stored as positive minor units; divide by 100 for display. +These bands do not overlap: pick the first row that matches. When it is still a close call, answer +in prose — an unasked-for chart is noise, while a missing one is a follow-up question, which is +cheaper. + +## Step 2 — pick the type + +Only two exist, and the x-axis decides: + +- **`line`** — x is time (a month, a week, a date). The question is "is this changing?" +- **`bar`** — x is a category, merchant, or account. The question is "which is biggest?" + +Never use a line for categories: a line between "Groceries" and "Transport" implies a progression +that isn't there. + +Use `series` only when comparing a handful of things over time (2–4). Past that the chart becomes +a thicket — facet it into separate charts or narrow the question instead. + +## Step 3 — write the query + +One row per point, aggregated in SQL. `render_chart` charts what the query returns, so the query +does the shaping. + +```sql +-- bar: spending by category +SELECT c.name AS category, SUM(t.amount_minor) / 100.0 AS total +FROM transaction t JOIN category c ON c.id = t.category_id +WHERE t.type = 'expense' +GROUP BY c.name +ORDER BY total DESC; + +-- line: spending per month +SELECT to_char(t.occurred_at, 'YYYY-MM') AS month, SUM(t.amount_minor) / 100.0 AS total +FROM transaction t +WHERE t.type = 'expense' +GROUP BY month +ORDER BY month; +``` + +Three things that make a chart wrong rather than ugly: + +- **Divide by 100 in the query.** Amounts are minor units; a chart axis reading `14672` when the + owner spent $146.72 is simply incorrect. +- **`ORDER BY` matters.** A bar chart is read as a ranking, so sort by the measure. A line chart is + read left-to-right as time, so sort by the time column. +- **Alias every computed column**, because `x` and `y` must name columns the query actually + returns. `SUM(...)` without an alias is not addressable. + +## Step 4 — render and read it back + +Call `render_chart` with the query, `chartType`, `x`, `y`, and a `title`. It returns a +confirmation, **not the rows** — so any number you quote must come from a `run_sql` you ran. +Never describe values you have not read. + +Then say what the chart shows. The chart is evidence; the sentence is the answer: + +> Groceries is your largest category at **$146.72**, about a third of the $437 total. Utilities +> follows at $120. + +Lead with the figure that answers the question. One or two observations is enough — the owner can +see the rest. Never moralize about the spending. + +## When render_chart refuses + +It returns an error instead of drawing when the result would mislead. Each one is actionable: + +- **A column that isn't in the result** — it lists the real columns; fix `x`/`y` or alias the + column in the SELECT. +- **Too many rows** — aggregate further, narrow the period, or take a top N. +- **No rows** — this does _not_ confirm the category is empty. A misspelled name returns nothing + too. Check with `list_categories` before telling the owner they have no such spending. From b75a0633249cf238277fa059c8c794a822f49959 Mon Sep 17 00:00:00 2001 From: Ali Ibrahim Jr <48456829+IBJunior@users.noreply.github.com> Date: Sat, 12 Sep 2026 22:34:07 +0200 Subject: [PATCH 2/3] fix(prompt): judge a skill by the task, not by the user's wording MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The skills section said "when a task matches a skill" without defining what matching means, so a weaker model fell back to lexical overlap: claude-haiku-4-5 loaded the expense-reporter skill for "show me a chart" but never for "chart my spending" — 0/5, deterministic, same trajectory every time. Sonnet loaded it either way, so the instruction was carrying the difference. Now states the principle — judge the KIND of work, not the vocabulary — and blocks the easy-task escape, since the agent charted correctly without the skill and had no reason to consult it. Deliberately no example phrasings: listing the synonyms would have made the eval pass by feeding it the answer, which tests the hint rather than the reframe. Activation went 0/5 -> ~6/8 on haiku with no vocabulary shared between the prompt and the case. Co-Authored-By: Claude Opus 5 (1M context) --- src/lib/agent/prompt.ts | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/src/lib/agent/prompt.ts b/src/lib/agent/prompt.ts index d95a9fc..a9ba574 100644 --- a/src/lib/agent/prompt.ts +++ b/src/lib/agent/prompt.ts @@ -118,9 +118,15 @@ export function buildSystemPrompt(skills: { name: string; description: string }[ return `${SYSTEM_PROMPT} **Available skills:** These are instruction sets you can load on demand. The descriptions below are all you have until -you load one — when a task matches a skill, call \`load_skill\` with its name and follow what it -returns before starting the work. Loading a skill is read-only and needs no approval; anything it -tells you to do is still governed by the usual approval gate. +you load one. + +Judge a skill by WHAT THE TASK IS, not by the words the user used — a description and a request +describe the same job in different vocabulary. If a skill covers the KIND of work you are about to +do, call \`load_skill\` with its name and follow what it returns BEFORE you start. Do not skip it +because the task looks easy: the skill holds conventions you cannot infer from the tools alone. + +Loading a skill is read-only and needs no approval; anything it tells you to do is still governed +by the usual approval gate. ${lines} `; } From 7dfe5d96afd6af93e7e6c5067c5749c7b7de07e7 Mon Sep 17 00:00:00 2001 From: Ali Ibrahim Jr <48456829+IBJunior@users.noreply.github.com> Date: Sat, 12 Sep 2026 22:34:18 +0200 Subject: [PATCH 3/3] fix(eval): reset the fixture before every run, not just mutating cases MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Case order was load-bearing and silent. Only `approval` and `config` cases reset, so an approved log_expense left a $4.20 row behind and the next case asserting a Dining TOTAL read fixture + leftovers — $2,737.20 against an expected $2,733.00, identically across all three runs. Three existing cases assert that same total and passed only because they run before the approval cases in the barrel. The bug predates them being noticed; a new case simply landed downstream of the mutation. Also fixes no-double-prompt-on-mutation, whose premise went stale when the currency feature shipped: it claims "nothing legitimate to ask for", but with no currency established the agent correctly asks before writing, so the case failed for the inverse of its own reason. Pre-seeding the currency makes the premise true again. Co-Authored-By: Claude Opus 5 (1M context) --- eval/cases/discipline.cases.mts | 5 ++++- eval/run.mts | 9 ++++----- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/eval/cases/discipline.cases.mts b/eval/cases/discipline.cases.mts index a5d7201..060ff70 100644 --- a/eval/cases/discipline.cases.mts +++ b/eval/cases/discipline.cases.mts @@ -17,8 +17,11 @@ export const cases: EvalCase[] = [ "The prompt forbids asking for confirmation in prose, because the SYSTEM already surfaces " + "an approval UI — asking on top of it double-prompts the user. Given every detail it needs, " + "the agent must call the tool (which IS the proposal) rather than stalling on a question.", - // Every field log_expense requires is present, so there is nothing legitimate to ask for. prompt: "Log a $12.50 coffee on my checking account, category Dining.", + // Every field log_expense needs is in the prompt, and the currency is pre-established, so + // there is nothing legitimate left to ask for. Without the currency the agent SHOULD ask + // (see config-log-expense-establishes-currency-first) and this case would fail inverted. + config: { currency: "USD" }, approval: "allow", graders: [ // The gate firing proves it called the tool instead of replying with a question. diff --git a/eval/run.mts b/eval/run.mts index a2167ff..796e80e 100644 --- a/eval/run.mts +++ b/eval/run.mts @@ -133,11 +133,10 @@ async function main() { const perRun: CaseOutcome["perRun"] = []; for (let i = 0; i < n; i++) { - // An approval case mutates, so each run must start from the same ledger — otherwise run 2 - // inherits run 1's writes and every row-count assertion drifts. A `config` case resets too: - // its seeded settings must not survive into the next case, whose premise may be an - // unestablished one. - if (testCase.approval || testCase.config) await resetToFixture(); + // Every run starts from the fixture, including read-only ones: a mutating case upstream + // leaves rows behind, and the next case asserting a TOTAL silently reads fixture + + // leftovers. Resetting only mutating cases made case ORDER decide the result. + await resetToFixture(); // After the reset, or the seeded settings would be wiped by it. if (testCase.config) await seedConfig(testCase.config);