Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion eval/cases/discipline.cases.mts
Original file line number Diff line number Diff line change
Expand Up @@ -17,8 +17,11 @@ export const cases: EvalCase[] = [
"The prompt forbids asking for confirmation in prose, because the SYSTEM already surfaces " +
"an approval UI — asking on top of it double-prompts the user. Given every detail it needs, " +
"the agent must call the tool (which IS the proposal) rather than stalling on a question.",
// Every field log_expense requires is present, so there is nothing legitimate to ask for.
prompt: "Log a $12.50 coffee on my checking account, category Dining.",
// Every field log_expense needs is in the prompt, and the currency is pre-established, so
// there is nothing legitimate left to ask for. Without the currency the agent SHOULD ask
// (see config-log-expense-establishes-currency-first) and this case would fail inverted.
config: { currency: "USD" },
approval: "allow",
graders: [
// The gate firing proves it called the tool instead of replying with a question.
Expand Down
2 changes: 2 additions & 0 deletions eval/cases/index.mts
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ import { cases as analysis } from "./analysis.cases.mts";
import { cases as approval } from "./approval.cases.mts";
import { cases as config } from "./config.cases.mts";
import { cases as csvImport } from "./csvImport.cases.mts";
import { cases as skills } from "./skills.cases.mts";
import { cases as discipline } from "./discipline.cases.mts";
import { cases as truncation } from "./truncation.cases.mts";

Expand All @@ -19,4 +20,5 @@ export const cases: EvalCase[] = [
...discipline,
...csvImport,
...config,
...skills,
];
81 changes: 81 additions & 0 deletions eval/cases/skills.cases.mts
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
import { RUN_POLICY } from "../config.mts";
import {
statesAmount,
toolCalled,
toolCalledBefore,
toolCalledWith,
toolNotCalled,
} from "../graders.mts";
import { FIXTURE } from "../seed.mts";
import type { EvalCase } from "../types.mts";

/**
* Progressive disclosure: a capability the agent must choose to consult, and choose not to overuse.
*
* Nothing here is checkable without a model. A unit test proves `load_skill` returns the body and
* `render_chart` refuses a bad spec, but not that the agent reaches for the skill at all, reads it
* BEFORE acting, or leaves it alone when a sentence would do. Those are the ways a skill fails in
* practice while every test stays green.
*/

const { dining } = FIXTURE;

export const cases: EvalCase[] = [
{
id: "chart-request-loads-skill",
description: "An explicit chart request should consult the skill that covers charting.",
prompt: "Show me a chart of my spending by category.",
graders: [toolCalled("load_skill"), toolCalled("render_chart")],
runs: RUN_POLICY.single,
tags: ["skills", "charts"],
},
{
id: "skill-read-before-charting",
description:
"The skill must be read BEFORE the chart is drawn. Loading it afterwards leaves an " +
"identical set of tool calls but means the instructions never informed the work.",
prompt: "Chart my spending by category for me.",
graders: [
toolCalledBefore("load_skill", "render_chart"),
// Without this the order check passes on a run that charted nothing.
toolCalled("render_chart"),
],
// Repeats because activation is ~6/8 on claude-haiku-4-5 (3/3 on Sonnet): the model
// sometimes charts without consulting the skill. Whether it loads varies; the ORDER never
// has. A drop to 0-1 of 3 is the regression worth catching.
runs: RUN_POLICY.majority,
tags: ["skills", "charts", "ordering"],
},
{
id: "trend-question-picks-line",
description:
"A measure over time is a line, not a bar — the one chart-selection judgment with a single " +
"spelling, so it can be graded without a judge.",
prompt: "Show me how my monthly spending has changed over the year.",
graders: [
toolCalledWith("render_chart", (a) => a.chartType === "line", "chartType=line"),
toolNotCalled("query_transactions"),
],
// Repeats because chart choice is model judgment, not mechanism — a wrong type here is a
// plausible answer rather than a crash, so one green run would not show it is reliable.
runs: RUN_POLICY.majority,
tags: ["skills", "charts", "chart-selection"],
},
{
id: "single-figure-needs-no-chart",
description:
"The over-triggering guard. Everything else in this file pushes toward charting; without " +
"this, a prompt or skill edit that makes the agent chart single figures stays green.",
prompt: `How much did I spend on ${dining.category} in total?`,
graders: [
toolNotCalled("render_chart"),
// The positive half: a run that answered nothing must not pass the negative grader.
toolCalled("run_sql"),
statesAmount(dining.totalMajor),
],
// Repeats because over-triggering is intermittent by nature: charting a single figure once in
// three is still a regression, and a single run would miss it two times out of three.
runs: RUN_POLICY.majority,
tags: ["skills", "charts", "over-triggering-guard"],
},
];
9 changes: 4 additions & 5 deletions eval/run.mts
Original file line number Diff line number Diff line change
Expand Up @@ -133,11 +133,10 @@ async function main() {

const perRun: CaseOutcome["perRun"] = [];
for (let i = 0; i < n; i++) {
// An approval case mutates, so each run must start from the same ledger — otherwise run 2
// inherits run 1's writes and every row-count assertion drifts. A `config` case resets too:
// its seeded settings must not survive into the next case, whose premise may be an
// unestablished one.
if (testCase.approval || testCase.config) await resetToFixture();
// Every run starts from the fixture, including read-only ones: a mutating case upstream
// leaves rows behind, and the next case asserting a TOTAL silently reads fixture +
// leftovers. Resetting only mutating cases made case ORDER decide the result.
await resetToFixture();
// After the reset, or the seeded settings would be wiped by it.
if (testCase.config) await seedConfig(testCase.config);

Expand Down
93 changes: 82 additions & 11 deletions skills/expense-reporter/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,18 +5,89 @@ description: Turn spending questions into visual reports with charts. Use when t

# Expense Reporter

> **Placeholder.** This skill exists so the loader, the prompt section and `load_skill` are
> exercised end to end. Its real instructions — chart selection, the query-to-chart workflow and
> the reporting voice — arrive with the reporting work in the next phase.
## Step 1 — decide whether this needs a chart

## Workflow
Most money questions do not. A chart earns its place only when the **shape** of the data is the
point — a ranking, a trend, a comparison across several things.

1. Establish what the owner is actually asking: a comparison between categories, a trend over
time, a breakdown of a whole, or a single figure.
2. Get the numbers with `run_sql`, aggregating in SQL rather than summing rows by hand.
3. Report the figures plainly, leading with the number that answers the question.
The owner's words decide first. If they asked to **see, visualise, chart, plot or graph** it — or
asked for a report — draw the chart. Otherwise:

## Notes
| The answer is… | Give them |
| ---------------------------------- | ------------------------------------------------------------------------------------- |
| One number | A sentence. `run_sql`, then say the figure. |
| Up to ~6 rows, exact values wanted | The `run_sql` table. Numbers you can read beat bars you have to estimate. |
| ~7 or more categories | A **bar** chart — past a handful, the ranking is easier seen than read. |
| A measure over time (2+ periods) | A **line** chart. A trend is a shape, so chart it even when the rows are few. |
| More than ~12 categories | Top N with `ORDER BY … LIMIT`, plus "and N others". A crowded chart hides the answer. |

- A single figure does not need a chart. Say the number.
- Amounts are stored as positive minor units; divide by 100 for display.
These bands do not overlap: pick the first row that matches. When it is still a close call, answer
in prose — an unasked-for chart is noise, while a missing one is a follow-up question, which is
cheaper.

## Step 2 — pick the type

Only two exist, and the x-axis decides:

- **`line`** — x is time (a month, a week, a date). The question is "is this changing?"
- **`bar`** — x is a category, merchant, or account. The question is "which is biggest?"

Never use a line for categories: a line between "Groceries" and "Transport" implies a progression
that isn't there.

Use `series` only when comparing a handful of things over time (2–4). Past that the chart becomes
a thicket — facet it into separate charts or narrow the question instead.

## Step 3 — write the query

One row per point, aggregated in SQL. `render_chart` charts what the query returns, so the query
does the shaping.

```sql
-- bar: spending by category
SELECT c.name AS category, SUM(t.amount_minor) / 100.0 AS total
FROM transaction t JOIN category c ON c.id = t.category_id
WHERE t.type = 'expense'
GROUP BY c.name
ORDER BY total DESC;

-- line: spending per month
SELECT to_char(t.occurred_at, 'YYYY-MM') AS month, SUM(t.amount_minor) / 100.0 AS total
FROM transaction t
WHERE t.type = 'expense'
GROUP BY month
ORDER BY month;
```

Three things that make a chart wrong rather than ugly:

- **Divide by 100 in the query.** Amounts are minor units; a chart axis reading `14672` when the
owner spent $146.72 is simply incorrect.
- **`ORDER BY` matters.** A bar chart is read as a ranking, so sort by the measure. A line chart is
read left-to-right as time, so sort by the time column.
- **Alias every computed column**, because `x` and `y` must name columns the query actually
returns. `SUM(...)` without an alias is not addressable.

## Step 4 — render and read it back

Call `render_chart` with the query, `chartType`, `x`, `y`, and a `title`. It returns a
confirmation, **not the rows** — so any number you quote must come from a `run_sql` you ran.
Never describe values you have not read.

Then say what the chart shows. The chart is evidence; the sentence is the answer:

> Groceries is your largest category at **$146.72**, about a third of the $437 total. Utilities
> follows at $120.

Lead with the figure that answers the question. One or two observations is enough — the owner can
see the rest. Never moralize about the spending.

## When render_chart refuses

It returns an error instead of drawing when the result would mislead. Each one is actionable:

- **A column that isn't in the result** — it lists the real columns; fix `x`/`y` or alias the
column in the SELECT.
- **Too many rows** — aggregate further, narrow the period, or take a top N.
- **No rows** — this does _not_ confirm the category is empty. A misspelled name returns nothing
too. Check with `list_categories` before telling the owner they have no such spending.
12 changes: 9 additions & 3 deletions src/lib/agent/prompt.ts
Original file line number Diff line number Diff line change
Expand Up @@ -118,9 +118,15 @@ export function buildSystemPrompt(skills: { name: string; description: string }[
return `${SYSTEM_PROMPT}
**Available skills:**
These are instruction sets you can load on demand. The descriptions below are all you have until
you load one — when a task matches a skill, call \`load_skill\` with its name and follow what it
returns before starting the work. Loading a skill is read-only and needs no approval; anything it
tells you to do is still governed by the usual approval gate.
you load one.

Judge a skill by WHAT THE TASK IS, not by the words the user used — a description and a request
describe the same job in different vocabulary. If a skill covers the KIND of work you are about to
do, call \`load_skill\` with its name and follow what it returns BEFORE you start. Do not skip it
because the task looks easy: the skill holds conventions you cannot infer from the tools alone.

Loading a skill is read-only and needs no approval; anything it tells you to do is still governed
by the usual approval gate.
${lines}
`;
}