diff --git a/eval/cases/discipline.cases.mts b/eval/cases/discipline.cases.mts index a5d7201..060ff70 100644 --- a/eval/cases/discipline.cases.mts +++ b/eval/cases/discipline.cases.mts @@ -17,8 +17,11 @@ export const cases: EvalCase[] = [ "The prompt forbids asking for confirmation in prose, because the SYSTEM already surfaces " + "an approval UI — asking on top of it double-prompts the user. Given every detail it needs, " + "the agent must call the tool (which IS the proposal) rather than stalling on a question.", - // Every field log_expense requires is present, so there is nothing legitimate to ask for. prompt: "Log a $12.50 coffee on my checking account, category Dining.", + // Every field log_expense needs is in the prompt, and the currency is pre-established, so + // there is nothing legitimate left to ask for. Without the currency the agent SHOULD ask + // (see config-log-expense-establishes-currency-first) and this case would fail inverted. + config: { currency: "USD" }, approval: "allow", graders: [ // The gate firing proves it called the tool instead of replying with a question. diff --git a/eval/cases/index.mts b/eval/cases/index.mts index 6c0c5f2..d221ea2 100644 --- a/eval/cases/index.mts +++ b/eval/cases/index.mts @@ -3,6 +3,7 @@ import { cases as analysis } from "./analysis.cases.mts"; import { cases as approval } from "./approval.cases.mts"; import { cases as config } from "./config.cases.mts"; import { cases as csvImport } from "./csvImport.cases.mts"; +import { cases as skills } from "./skills.cases.mts"; import { cases as discipline } from "./discipline.cases.mts"; import { cases as truncation } from "./truncation.cases.mts"; @@ -19,4 +20,5 @@ export const cases: EvalCase[] = [ ...discipline, ...csvImport, ...config, + ...skills, ]; diff --git a/eval/cases/skills.cases.mts b/eval/cases/skills.cases.mts new file mode 100644 index 0000000..b35211f --- /dev/null +++ b/eval/cases/skills.cases.mts @@ -0,0 +1,81 @@ +import { RUN_POLICY } from "../config.mts"; +import { + statesAmount, + toolCalled, + toolCalledBefore, + toolCalledWith, + toolNotCalled, +} from "../graders.mts"; +import { FIXTURE } from "../seed.mts"; +import type { EvalCase } from "../types.mts"; + +/** + * Progressive disclosure: a capability the agent must choose to consult, and choose not to overuse. + * + * Nothing here is checkable without a model. A unit test proves `load_skill` returns the body and + * `render_chart` refuses a bad spec, but not that the agent reaches for the skill at all, reads it + * BEFORE acting, or leaves it alone when a sentence would do. Those are the ways a skill fails in + * practice while every test stays green. + */ + +const { dining } = FIXTURE; + +export const cases: EvalCase[] = [ + { + id: "chart-request-loads-skill", + description: "An explicit chart request should consult the skill that covers charting.", + prompt: "Show me a chart of my spending by category.", + graders: [toolCalled("load_skill"), toolCalled("render_chart")], + runs: RUN_POLICY.single, + tags: ["skills", "charts"], + }, + { + id: "skill-read-before-charting", + description: + "The skill must be read BEFORE the chart is drawn. Loading it afterwards leaves an " + + "identical set of tool calls but means the instructions never informed the work.", + prompt: "Chart my spending by category for me.", + graders: [ + toolCalledBefore("load_skill", "render_chart"), + // Without this the order check passes on a run that charted nothing. + toolCalled("render_chart"), + ], + // Repeats because activation is ~6/8 on claude-haiku-4-5 (3/3 on Sonnet): the model + // sometimes charts without consulting the skill. Whether it loads varies; the ORDER never + // has. A drop to 0-1 of 3 is the regression worth catching. + runs: RUN_POLICY.majority, + tags: ["skills", "charts", "ordering"], + }, + { + id: "trend-question-picks-line", + description: + "A measure over time is a line, not a bar — the one chart-selection judgment with a single " + + "spelling, so it can be graded without a judge.", + prompt: "Show me how my monthly spending has changed over the year.", + graders: [ + toolCalledWith("render_chart", (a) => a.chartType === "line", "chartType=line"), + toolNotCalled("query_transactions"), + ], + // Repeats because chart choice is model judgment, not mechanism — a wrong type here is a + // plausible answer rather than a crash, so one green run would not show it is reliable. + runs: RUN_POLICY.majority, + tags: ["skills", "charts", "chart-selection"], + }, + { + id: "single-figure-needs-no-chart", + description: + "The over-triggering guard. Everything else in this file pushes toward charting; without " + + "this, a prompt or skill edit that makes the agent chart single figures stays green.", + prompt: `How much did I spend on ${dining.category} in total?`, + graders: [ + toolNotCalled("render_chart"), + // The positive half: a run that answered nothing must not pass the negative grader. + toolCalled("run_sql"), + statesAmount(dining.totalMajor), + ], + // Repeats because over-triggering is intermittent by nature: charting a single figure once in + // three is still a regression, and a single run would miss it two times out of three. + runs: RUN_POLICY.majority, + tags: ["skills", "charts", "over-triggering-guard"], + }, +]; diff --git a/eval/run.mts b/eval/run.mts index a2167ff..796e80e 100644 --- a/eval/run.mts +++ b/eval/run.mts @@ -133,11 +133,10 @@ async function main() { const perRun: CaseOutcome["perRun"] = []; for (let i = 0; i < n; i++) { - // An approval case mutates, so each run must start from the same ledger — otherwise run 2 - // inherits run 1's writes and every row-count assertion drifts. A `config` case resets too: - // its seeded settings must not survive into the next case, whose premise may be an - // unestablished one. - if (testCase.approval || testCase.config) await resetToFixture(); + // Every run starts from the fixture, including read-only ones: a mutating case upstream + // leaves rows behind, and the next case asserting a TOTAL silently reads fixture + + // leftovers. Resetting only mutating cases made case ORDER decide the result. + await resetToFixture(); // After the reset, or the seeded settings would be wiped by it. if (testCase.config) await seedConfig(testCase.config); diff --git a/skills/expense-reporter/SKILL.md b/skills/expense-reporter/SKILL.md index 459d674..45f8709 100644 --- a/skills/expense-reporter/SKILL.md +++ b/skills/expense-reporter/SKILL.md @@ -5,18 +5,89 @@ description: Turn spending questions into visual reports with charts. Use when t # Expense Reporter -> **Placeholder.** This skill exists so the loader, the prompt section and `load_skill` are -> exercised end to end. Its real instructions — chart selection, the query-to-chart workflow and -> the reporting voice — arrive with the reporting work in the next phase. +## Step 1 — decide whether this needs a chart -## Workflow +Most money questions do not. A chart earns its place only when the **shape** of the data is the +point — a ranking, a trend, a comparison across several things. -1. Establish what the owner is actually asking: a comparison between categories, a trend over - time, a breakdown of a whole, or a single figure. -2. Get the numbers with `run_sql`, aggregating in SQL rather than summing rows by hand. -3. Report the figures plainly, leading with the number that answers the question. +The owner's words decide first. If they asked to **see, visualise, chart, plot or graph** it — or +asked for a report — draw the chart. Otherwise: -## Notes +| The answer is… | Give them | +| ---------------------------------- | ------------------------------------------------------------------------------------- | +| One number | A sentence. `run_sql`, then say the figure. | +| Up to ~6 rows, exact values wanted | The `run_sql` table. Numbers you can read beat bars you have to estimate. | +| ~7 or more categories | A **bar** chart — past a handful, the ranking is easier seen than read. | +| A measure over time (2+ periods) | A **line** chart. A trend is a shape, so chart it even when the rows are few. | +| More than ~12 categories | Top N with `ORDER BY … LIMIT`, plus "and N others". A crowded chart hides the answer. | -- A single figure does not need a chart. Say the number. -- Amounts are stored as positive minor units; divide by 100 for display. +These bands do not overlap: pick the first row that matches. When it is still a close call, answer +in prose — an unasked-for chart is noise, while a missing one is a follow-up question, which is +cheaper. + +## Step 2 — pick the type + +Only two exist, and the x-axis decides: + +- **`line`** — x is time (a month, a week, a date). The question is "is this changing?" +- **`bar`** — x is a category, merchant, or account. The question is "which is biggest?" + +Never use a line for categories: a line between "Groceries" and "Transport" implies a progression +that isn't there. + +Use `series` only when comparing a handful of things over time (2–4). Past that the chart becomes +a thicket — facet it into separate charts or narrow the question instead. + +## Step 3 — write the query + +One row per point, aggregated in SQL. `render_chart` charts what the query returns, so the query +does the shaping. + +```sql +-- bar: spending by category +SELECT c.name AS category, SUM(t.amount_minor) / 100.0 AS total +FROM transaction t JOIN category c ON c.id = t.category_id +WHERE t.type = 'expense' +GROUP BY c.name +ORDER BY total DESC; + +-- line: spending per month +SELECT to_char(t.occurred_at, 'YYYY-MM') AS month, SUM(t.amount_minor) / 100.0 AS total +FROM transaction t +WHERE t.type = 'expense' +GROUP BY month +ORDER BY month; +``` + +Three things that make a chart wrong rather than ugly: + +- **Divide by 100 in the query.** Amounts are minor units; a chart axis reading `14672` when the + owner spent $146.72 is simply incorrect. +- **`ORDER BY` matters.** A bar chart is read as a ranking, so sort by the measure. A line chart is + read left-to-right as time, so sort by the time column. +- **Alias every computed column**, because `x` and `y` must name columns the query actually + returns. `SUM(...)` without an alias is not addressable. + +## Step 4 — render and read it back + +Call `render_chart` with the query, `chartType`, `x`, `y`, and a `title`. It returns a +confirmation, **not the rows** — so any number you quote must come from a `run_sql` you ran. +Never describe values you have not read. + +Then say what the chart shows. The chart is evidence; the sentence is the answer: + +> Groceries is your largest category at **$146.72**, about a third of the $437 total. Utilities +> follows at $120. + +Lead with the figure that answers the question. One or two observations is enough — the owner can +see the rest. Never moralize about the spending. + +## When render_chart refuses + +It returns an error instead of drawing when the result would mislead. Each one is actionable: + +- **A column that isn't in the result** — it lists the real columns; fix `x`/`y` or alias the + column in the SELECT. +- **Too many rows** — aggregate further, narrow the period, or take a top N. +- **No rows** — this does _not_ confirm the category is empty. A misspelled name returns nothing + too. Check with `list_categories` before telling the owner they have no such spending. diff --git a/src/lib/agent/prompt.ts b/src/lib/agent/prompt.ts index d95a9fc..a9ba574 100644 --- a/src/lib/agent/prompt.ts +++ b/src/lib/agent/prompt.ts @@ -118,9 +118,15 @@ export function buildSystemPrompt(skills: { name: string; description: string }[ return `${SYSTEM_PROMPT} **Available skills:** These are instruction sets you can load on demand. The descriptions below are all you have until -you load one — when a task matches a skill, call \`load_skill\` with its name and follow what it -returns before starting the work. Loading a skill is read-only and needs no approval; anything it -tells you to do is still governed by the usual approval gate. +you load one. + +Judge a skill by WHAT THE TASK IS, not by the words the user used — a description and a request +describe the same job in different vocabulary. If a skill covers the KIND of work you are about to +do, call \`load_skill\` with its name and follow what it returns BEFORE you start. Do not skip it +because the task looks easy: the skill holds conventions you cannot infer from the tools alone. + +Loading a skill is read-only and needs no approval; anything it tells you to do is still governed +by the usual approval gate. ${lines} `; }