From 55fb6983136842b1432e6cb631e5cfefeaa19052 Mon Sep 17 00:00:00 2001 From: Coder8124 Date: Sat, 3 Oct 2026 23:52:10 -0700 Subject: [PATCH 1/2] A ruled-out approach whose named file changed since it was recorded is flagged as possibly stale at resume and before_you_try --- cmd/logos/tried.go | 4 + internal/contextpack/pack.go | 38 +++++ internal/contextpack/render.go | 23 ++- internal/contextpack/rulingdrift_test.go | 92 +++++++++++ internal/deadend/deadend.go | 24 ++- internal/deadend/drift_test.go | 75 +++++++++ internal/deadend/render.go | 6 + internal/eval/adapters/logos.go | 69 +++++++- internal/eval/eval.go | 9 +- internal/eval/scenarios.go | 30 ++++ internal/gitstate/anchor.go | 194 +++++++++++++++++++++++ internal/gitstate/anchor_test.go | 111 +++++++++++++ internal/mcpserver/continuity.go | 5 +- internal/mcpserver/server.go | 2 +- 14 files changed, 671 insertions(+), 11 deletions(-) create mode 100644 internal/contextpack/rulingdrift_test.go create mode 100644 internal/deadend/drift_test.go create mode 100644 internal/gitstate/anchor.go create mode 100644 internal/gitstate/anchor_test.go diff --git a/cmd/logos/tried.go b/cmd/logos/tried.go index 29c491e..86b0475 100644 --- a/cmd/logos/tried.go +++ b/cmd/logos/tried.go @@ -2,6 +2,7 @@ package main import ( "fmt" + "os" "strings" "github.com/Coder8124/logos/internal/deadend" @@ -58,6 +59,9 @@ func runTried(args []string) error { if err != nil { return err } + if dir, err := os.Getwd(); err == nil { + deadend.MarkDrift(hits, dir) + } fmt.Print(deadend.Render(proposed, hits)) fmt.Print(deadend.SemanticSkipped(semanticErr)) if len(hits) > 0 { diff --git a/internal/contextpack/pack.go b/internal/contextpack/pack.go index 729f81a..f1976ae 100644 --- a/internal/contextpack/pack.go +++ b/internal/contextpack/pack.go @@ -115,6 +115,12 @@ type Pack struct { // until this was rendered the pack printed it with nothing to compare it // against — though the sha was in hand and the tree was one rev-parse away. Drifted *gitstate.State `json:"drifted,omitempty"` + // RulingDrift holds, by normalizeKey of the ruling's text, the dead ends + // whose named files changed after the commit they were recorded at. Kept + // in the pack rather than dropped: a ruling about a rewritten file may + // still hold, and the reader is the one who can tell — but only if it is + // told the file moved. + RulingDrift map[string]*gitstate.Drift `json:"ruling_drift,omitempty"` // Reader is who asked, when the caller said. See Request.Agent. Reader string `json:"reader,omitempty"` @@ -278,6 +284,7 @@ func Build(ix *index.Index, embed *provider.Provider, embedModel string, req Req } p.Reader = req.Agent p.noteDrift(req.Dir) + p.markRulings(req.Dir) // Uncommitted notes get no such fallback. A checkpoint is a finished record // of where the codebase was; an open note is another agent's live work in // another tree, and presenting that as this session's own progress is @@ -854,6 +861,37 @@ func (p *Pack) noteDrift(dir string) { p.Drifted = &gitstate.State{Branch: branch, Commit: commit} } +// markRulings measures each handed-over dead end against the commit its +// checkpoint recorded. Newest first, so a ruling restated in a later +// checkpoint is measured from the later commit — the one it was last +// confirmed at. +func (p *Pack) markRulings(dir string) { + if p.Checkpoint == nil { + return + } + a := gitstate.NewAnchors(dir) + if a == nil { + return + } + cps := append([]session.Checkpoint{*p.Checkpoint}, p.History...) + for _, c := range cps { + for _, f := range c.Failed { + k := normalizeKey(f) + if _, done := p.RulingDrift[k]; done { + continue + } + d, ok := a.Since(c.Git.Commit, f) + if !ok { + continue + } + if p.RulingDrift == nil { + p.RulingDrift = map[string]*gitstate.Drift{} + } + p.RulingDrift[k] = &d + } + } +} + // IntentFrom names the earlier checkpoint a reason was inherited from, so the // render can say whose reason it is and how old. type IntentFrom struct { diff --git a/internal/contextpack/render.go b/internal/contextpack/render.go index 0240ed2..fa47883 100644 --- a/internal/contextpack/render.go +++ b/internal/contextpack/render.go @@ -7,6 +7,7 @@ import ( "unicode/utf8" "github.com/Coder8124/logos/internal/deadend" + "github.com/Coder8124/logos/internal/gitstate" "github.com/Coder8124/logos/internal/ingest" "github.com/Coder8124/logos/internal/memory" "github.com/Coder8124/logos/internal/project" @@ -210,7 +211,7 @@ func (p *Pack) spendCheckpoint(sp *spender) string { inferredFromTranscript(&head, c) list(&tail, "Decided", c.Decisions) // The most valuable lines in the pack: what has already been ruled out. - failures(&tail, "Already tried, didn't work", c.Failed) + failures(&tail, "Already tried, didn't work", c.Failed, p.RulingDrift) list(&tail, "Still open", c.Questions) list(&tail, "Files touched", c.Files) // Predecessors' dead ends, attributed. Cheap — a line each — and the only @@ -576,7 +577,11 @@ func (p *Pack) priorFailures() []string { if who == "" { who = "an earlier agent" } - out = append(out, fmt.Sprintf("(%s) %s", who, flatten(f))) + line := fmt.Sprintf("(%s) %s", who, flatten(f)) + if d := p.RulingDrift[k]; d != nil { + line += " — ⚠ " + d.Note() + } + out = append(out, line) } } return out @@ -1081,7 +1086,10 @@ func list(b *strings.Builder, heading string, items []string) { // now that vocabulary only ever reached before_you_try, so a handoff to another // tool delivered either the raw pipe-separated source text or a ruling nobody // could place. A free-prose entry still reads exactly as it did. -func failures(b *strings.Builder, heading string, items []string) { +// +// A ruling whose named files changed since it was recorded keeps its line and +// gains a second one saying so (see Pack.RulingDrift). +func failures(b *strings.Builder, heading string, items []string, drift map[string]*gitstate.Drift) { if len(items) == 0 { return } @@ -1090,6 +1098,7 @@ func failures(b *strings.Builder, heading string, items []string) { r := deadend.ParseRecord(it) if !r.Typed() { fmt.Fprintf(b, "- %s\n", inline(it)) + driftLine(b, drift[normalizeKey(it)]) continue } fmt.Fprintf(b, "- %s", inline(r.Route)) @@ -1103,10 +1112,18 @@ func failures(b *strings.Builder, heading string, items []string) { fmt.Fprintf(b, " — try instead: %s", inline(r.Alternative)) } b.WriteString("\n") + driftLine(b, drift[normalizeKey(it)]) } b.WriteString("\n") } +func driftLine(b *strings.Builder, d *gitstate.Drift) { + if d != nil { + // The paths in the note were read out of vault text. + fmt.Fprintf(b, " ⚠ %s\n", inline(d.Note())) + } +} + func dedup(in []string) []string { seen := map[string]bool{} out := make([]string, 0, len(in)) diff --git a/internal/contextpack/rulingdrift_test.go b/internal/contextpack/rulingdrift_test.go new file mode 100644 index 0000000..43584a3 --- /dev/null +++ b/internal/contextpack/rulingdrift_test.go @@ -0,0 +1,92 @@ +package contextpack + +import ( + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/Coder8124/logos/internal/gitstate" + "github.com/Coder8124/logos/internal/session" +) + +func commitFile(t *testing.T, dir, name, body string) { + t.Helper() + path := filepath.Join(dir, name) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte(body), 0o644); err != nil { + t.Fatal(err) + } + for _, args := range [][]string{ + {"add", "--", name}, + {"-c", "user.email=t@example.com", "-c", "user.name=t", "-c", "commit.gpgsign=false", "commit", "-q", "-m", "change " + name}, + } { + cmd := exec.Command("git", args...) + cmd.Dir = dir + if out, err := cmd.CombinedOutput(); err != nil { + t.Skipf("git %s: %v %s", args[0], err, out) + } + } +} + +// "Streaming fails because reader.go loads the whole file" was handed to the +// next agent as settled three weeks after reader.go was rewritten to read in +// chunks — the fix the ruling ruled out was now the one that worked. The +// ruling is kept, and says the file it blames has moved. +func TestARuledOutApproachWhoseFileChangedSinceIsFlaggedAtResume(t *testing.T) { + repo := gitRepo(t) + commitFile(t, repo, "internal/parse/reader.go", "load all\n") + commitFile(t, repo, "internal/sink/writer.go", "write\n") + _, anchor := gitstate.Head(repo) + remote, root := gitstate.Identity(repo) + ix := seedVault(t) + now := time.Now() + git := gitstate.State{Branch: "main", Commit: anchor, Remote: remote, Root: root} + for _, c := range []session.Checkpoint{{ + Project: "api", Agent: "claude", Task: "import the export", TS: now.Add(-2 * time.Hour).Unix(), Git: git, + Failed: []string{"buffering the sink — internal/sink/writer.go flushes per row, so it is no faster"}, + }, { + Project: "api", Agent: "claude", Task: "import the export", TS: now.Add(-time.Hour).Unix(), Git: git, + Failed: []string{"streaming the export — internal/parse/reader.go loads the whole file, out of memory at 2GB"}, + Next: "split the export", + }} { + if err := session.Commit(ix.DB, ix.Vault, &c); err != nil { + t.Fatal(err) + } + } + commitFile(t, repo, "internal/parse/reader.go", "read in chunks\n") + + p, err := Build(ix, nil, "", Request{Task: "continue", Hint: "api", Dir: repo, Now: now.Unix()}) + if err != nil { + t.Fatal(err) + } + out := p.Render() + if !strings.Contains(out, "loads the whole file") { + t.Fatalf("the ruling was dropped rather than flagged:\n%s", out) + } + if !strings.Contains(out, "⚠ recorded at "+anchor+"; internal/parse/reader.go changed in 1 commit since") { + t.Errorf("the ruling about a rewritten file is handed over as settled:\n%s", out) + } + if strings.Count(out, "may no longer hold") != 1 { + t.Errorf("a ruling about a file nobody touched since was flagged too:\n%s", out) + } + + // The same ruling, now in history behind a newer checkpoint, keeps its flag. + if err := session.Commit(ix.DB, ix.Vault, &session.Checkpoint{ + Project: "api", Agent: "claude", Task: "something else", TS: now.Add(-time.Minute).Unix(), Git: git, + }); err != nil { + t.Fatal(err) + } + p, err = Build(ix, nil, "", Request{Task: "continue", Hint: "api", Dir: repo, Now: now.Unix()}) + if err != nil { + t.Fatal(err) + } + out = p.Render() + if !strings.Contains(out, "out of memory at 2GB — ⚠ recorded at "+anchor) { + t.Errorf("an earlier ruling about a rewritten file lost its flag:\n%s", out) + } +} diff --git a/internal/deadend/deadend.go b/internal/deadend/deadend.go index 026fa7b..8b7768e 100644 --- a/internal/deadend/deadend.go +++ b/internal/deadend/deadend.go @@ -36,6 +36,7 @@ import ( "strings" "time" + "github.com/Coder8124/logos/internal/gitstate" "github.com/Coder8124/logos/internal/provider" "github.com/Coder8124/logos/internal/session" "github.com/Coder8124/logos/internal/textmatch" @@ -75,6 +76,27 @@ type Ruling struct { // names may have moved since — see PossiblySuperseded. Never grounds for // dropping the ruling, only for saying so. Stale bool `json:"stale,omitempty"` + // Commit is the repository's HEAD when the checkpoint holding this was + // written. Empty for a working note, which records no repository. + Commit string `json:"commit,omitempty"` + // Drift is set when the files this ruling names have changed since + // Commit — see MarkDrift. Like Stale, a reason to say so, never to drop it. + Drift *gitstate.Drift `json:"drift,omitempty"` +} + +// MarkDrift notes, on each ruling, whether the files it names have changed in +// the repository at dir since it was recorded. Only the rulings about to be +// shown are measured: it costs a git call per named file. +func MarkDrift(hits []Ruling, dir string) { + a := gitstate.NewAnchors(dir) + if a == nil { + return + } + for i := range hits { + if d, ok := a.Since(hits[i].Commit, hits[i].Record.Raw); ok { + hits[i].Drift = &d + } + } } // failureMarkers are how people write down that something did not work. @@ -147,7 +169,7 @@ func Collect(vaultDir string, db *sql.DB, project string) ([]Ruling, error) { rec := ParseRecord(textmatch.Flatten(f)) out = append(out, Ruling{ Text: rec.Route, Project: proj, Agent: c.Agent, - When: c.TS, Slug: c.Slug, Source: FromCheckpoint, + When: c.TS, Slug: c.Slug, Source: FromCheckpoint, Commit: c.Git.Commit, Record: rec, Stale: PossiblySuperseded(rec.Scope, c.TS, time.Now()), }) } diff --git a/internal/deadend/drift_test.go b/internal/deadend/drift_test.go new file mode 100644 index 0000000..61bd63a --- /dev/null +++ b/internal/deadend/drift_test.go @@ -0,0 +1,75 @@ +package deadend + +import ( + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + + "github.com/Coder8124/logos/internal/gitstate" + "github.com/Coder8124/logos/internal/session" +) + +func commitFile(t *testing.T, dir, name, body string) { + t.Helper() + path := filepath.Join(dir, name) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte(body), 0o644); err != nil { + t.Fatal(err) + } + for _, args := range [][]string{ + {"add", "--", name}, + {"-c", "user.email=t@example.com", "-c", "user.name=t", "-c", "commit.gpgsign=false", "commit", "-q", "-m", "change " + name}, + } { + cmd := exec.Command("git", args...) + cmd.Dir = dir + if out, err := cmd.CombinedOutput(); err != nil { + t.Skipf("git %s: %v %s", args[0], err, out) + } + } +} + +// before_you_try is asked at the moment an agent is about to act, and a ruling +// about a file rewritten since is the one most likely to stop it doing the +// thing that now works. Found, kept, and said to have moved. +func TestBeforeYouTryFlagsARulingWhoseFileChangedSince(t *testing.T) { + repo := t.TempDir() + if out, err := exec.Command("git", "init", "-q", repo).CombinedOutput(); err != nil { + t.Skipf("git init: %v %s", err, out) + } + commitFile(t, repo, "internal/parse/reader.go", "load all\n") + _, anchor := gitstate.Head(repo) + + dir, db := seed(t) + if err := session.Commit(db, dir, &session.Checkpoint{ + Project: "kestrel-one", Agent: "claude", Task: "work", Next: "carry on", + Git: gitstate.State{Commit: anchor}, + Failed: []string{"streaming the vendor export — internal/parse/reader.go loads the whole file into memory"}, + }); err != nil { + t.Fatal(err) + } + commitFile(t, repo, "internal/parse/reader.go", "read in chunks\n") + + hits, err := Check(dir, db, nil, "", "streaming the vendor export", "kestrel-one", 5) + if err != nil { + t.Fatal(err) + } + if len(hits) == 0 { + t.Fatal("the ruling should still be found") + } + if hits[0].Commit != anchor { + t.Fatalf("the ruling lost the commit its checkpoint recorded: %q, want %q", hits[0].Commit, anchor) + } + MarkDrift(hits, repo) + out := Render("streaming the vendor export", hits) + if !strings.Contains(out, "⚠ recorded at "+anchor+"; internal/parse/reader.go changed in 1 commit since") { + t.Errorf("a ruling about a rewritten file is presented as settled:\n%s", out) + } + // The seeded rulings carry no commit and name no file in this repository. + if strings.Count(out, "may no longer hold") != 1 { + t.Errorf("rulings with nothing to measure were flagged:\n%s", out) + } +} diff --git a/internal/deadend/render.go b/internal/deadend/render.go index 297f491..cde0167 100644 --- a/internal/deadend/render.go +++ b/internal/deadend/render.go @@ -68,6 +68,12 @@ func Render(proposed string, hits []Ruling) string { if h.Stale { b.WriteString(" ⚠ possibly superseded — version-bound and old enough that the dependency it names may have moved since\n") } + // After Stale and on its own line: the age says the world may have + // moved, this says the code it blamed did. Commit and paths are + // git's, but the paths are as the ruling spelled them, so inlined. + if h.Drift != nil { + fmt.Fprintf(&b, " ⚠ %s\n", untrusted.Inline(h.Drift.Note())) + } } b.WriteString("\nBefore proposing this, say that it has been tried and what happened. ") diff --git a/internal/eval/adapters/logos.go b/internal/eval/adapters/logos.go index ab43fd8..1587631 100644 --- a/internal/eval/adapters/logos.go +++ b/internal/eval/adapters/logos.go @@ -11,12 +11,14 @@ package adapters import ( "fmt" "os" + "os/exec" "path/filepath" "strings" "time" "github.com/Coder8124/logos/internal/contextpack" "github.com/Coder8124/logos/internal/eval" + "github.com/Coder8124/logos/internal/gitstate" "github.com/Coder8124/logos/internal/index" "github.com/Coder8124/logos/internal/memory" "github.com/Coder8124/logos/internal/provider" @@ -30,7 +32,10 @@ import ( // benchmark-only retrieval path — if the suite scores well, the shipped product // scores well. type Logos struct { - root string // a scratch vault, never the user's + root string // a scratch vault, never the user's + // repo is the scenario's code, created by its first commit event. Kept + // outside the vault, as a project's repository is. + repo string ix *index.Index embed *provider.Provider model string @@ -60,6 +65,10 @@ func (b *Logos) Reset() error { if b.root != "" { os.RemoveAll(b.root) } + if b.repo != "" { + os.RemoveAll(b.repo) + b.repo = "" + } root, err := os.MkdirTemp("", "eval-logos-") if err != nil { return err @@ -92,6 +101,9 @@ func (b *Logos) Close() error { if b.ix != nil { b.ix.Close() } + if b.repo != "" { + os.RemoveAll(b.repo) + } if b.root != "" { return os.RemoveAll(b.root) } @@ -118,15 +130,64 @@ func (b *Logos) Write(ev eval.Event) error { return err case eval.KindCheckpoint: - return session.Commit(b.ix.DB, b.root, &session.Checkpoint{ + c := &session.Checkpoint{ Project: ev.Project, Agent: ev.Actor, Task: ev.Task, State: ev.Text, Decisions: ev.Decisions, Failed: ev.Failed, Questions: ev.Questions, Next: ev.Next, TS: ev.TS, - }) + } + // The repository the agent stood in, as checkpoint reads it in use. + // Left to Commit, it would read whatever directory the benchmark runs + // from. + if b.repo != "" { + c.Git = gitstate.Read(b.repo) + } + return session.Commit(b.ix.DB, b.root, c) + + case eval.KindCommit: + return b.commit(ev) } return fmt.Errorf("unknown event kind %q", ev.Kind) } +// commit changes one file in the scenario's repository, dated when the event +// happened so the history reads in the scenario's order. +func (b *Logos) commit(ev eval.Event) error { + if b.repo == "" { + repo, err := os.MkdirTemp("", "eval-repo-") + if err != nil { + return err + } + b.repo = repo + if err := b.git(ev.TS, "init", "-q"); err != nil { + return err + } + } + path := filepath.Join(b.repo, filepath.FromSlash(ev.Title)) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + return err + } + // The message as the content, so every commit to a file changes it. + if err := os.WriteFile(path, []byte(ev.Text+"\n"), 0o644); err != nil { + return err + } + if err := b.git(ev.TS, "add", "--", ev.Title); err != nil { + return err + } + return b.git(ev.TS, "commit", "-q", "-m", ev.Text) +} + +func (b *Logos) git(ts int64, args ...string) error { + args = append([]string{"-c", "user.name=eval", "-c", "user.email=eval@localhost", "-c", "commit.gpgsign=false"}, args...) + cmd := exec.Command("git", args...) + cmd.Dir = b.repo + date := fmt.Sprintf("@%d +0000", ts) + cmd.Env = append(os.Environ(), "GIT_AUTHOR_DATE="+date, "GIT_COMMITTER_DATE="+date) + if out, err := cmd.CombinedOutput(); err != nil { + return fmt.Errorf("git %s: %v: %s", strings.Join(args, " "), err, out) + } + return nil +} + // writeDoc lays a document into the vault the way a person would: the project's // own page under projects/, everything else under topics/, named by its title // so that [[wikilinks]] in other notes resolve. @@ -161,7 +222,7 @@ func (b *Logos) Read(q eval.Query) (eval.Response, error) { b.dirty = false } pack, err := contextpack.Build(b.ix, b.embed, b.model, contextpack.Request{ - Task: q.Task, Hint: q.Project, Budget: q.Budget, Now: q.Now, + Task: q.Task, Hint: q.Project, Budget: q.Budget, Now: q.Now, Dir: b.repo, }) if err != nil { return eval.Response{}, err diff --git a/internal/eval/eval.go b/internal/eval/eval.go index 7e1fbc6..ed81aa5 100644 --- a/internal/eval/eval.go +++ b/internal/eval/eval.go @@ -56,6 +56,10 @@ const ( // KindCheckpoint is where an agent stopped: state, decisions, what failed, // what is next. KindCheckpoint Kind = "checkpoint" + // KindCommit is a change to the project's code: Title is the file, Text the + // commit message. A ruling is only as current as the code it was about, + // and without this a scenario had no way to move the code underneath one. + KindCommit Kind = "commit" ) // An Event is one thing that happened, in the order it happened. Scenarios are @@ -65,7 +69,7 @@ type Event struct { Actor string // "user", "claude", "cursor" — who produced this Kind Kind Project string - Title string // for KindDoc + Title string // for KindDoc; the file changed, for KindCommit Text string // Checkpoint fields. Adapters with a checkpoint primitive should map these @@ -82,6 +86,9 @@ type Event struct { // present, nothing is abbreviated — so that a system scoring badly on this // suite cannot blame the harness for withholding information. func (e Event) Flatten() string { + if e.Kind == KindCommit { + return "Commit to " + e.Title + ": " + e.Text + } if e.Kind != KindCheckpoint { if e.Title != "" { return e.Title + "\n" + e.Text diff --git a/internal/eval/scenarios.go b/internal/eval/scenarios.go index 79aa5f8..5b26a16 100644 --- a/internal/eval/scenarios.go +++ b/internal/eval/scenarios.go @@ -54,6 +54,10 @@ func said(days int, text string) Event { return Event{TS: ago(days), Actor: "user", Kind: KindFact, Text: text} } +func commit(days int, project, path, message string) Event { + return Event{TS: ago(days), Actor: "user", Kind: KindCommit, Project: project, Title: path, Text: message} +} + func msg(days int, actor, text string) Event { return Event{TS: ago(days), Actor: actor, Kind: KindMessage, Text: text} } @@ -416,6 +420,32 @@ func continuity() []Scenario { }, }, }, + { + ID: "handoff-dead-end-code-moved", Family: "continuity", Skill: "staleness", + Why: "A dead end blamed on one file, and that file was rewritten since. Obeyed as settled, it stops the fix that now works.", + Known: KnownStrength, + Setup: []Event{ + doc(40, "ingest-svc", "Ingest service", "Loads vendor CSV exports into the warehouse."), + commit(30, "ingest-svc", "internal/parse/reader.go", "Parse CSV exports with a buffered reader"), + { + TS: ago(14), Actor: "claude", Kind: KindCheckpoint, Project: "ingest-svc", + Task: "import the 3GB vendor export", + Failed: []string{"streaming the 3GB export through the parser — internal/parse/reader.go loads the whole file into memory first, so it runs out of memory at about 2GB"}, + Next: "Split the export into 500MB chunks before importing", + }, + commit(4, "ingest-svc", "internal/parse/reader.go", "Read CSV input in 64KB chunks instead of loading the whole file"), + }, + Query: Query{Task: "import the 3GB vendor export", Project: "ingest-svc", Agent: "cursor", Budget: 4000, Now: benchNow}, + Wordings: []string{"load the big vendor CSV", "get the large export into the warehouse", "pick up the vendor import"}, + Gold: Gold{ + Carry: []Fact{ + {Label: "the dead end itself", Any: []string{"whole file into memory", "out of memory"}}, + }, + Signal: []Fact{ + {Label: "the dead end is flagged as recorded before its file changed", Any: []string{"changed since", "may no longer hold"}}, + }, + }, + }, } } diff --git a/internal/gitstate/anchor.go b/internal/gitstate/anchor.go new file mode 100644 index 0000000..8602d2a --- /dev/null +++ b/internal/gitstate/anchor.go @@ -0,0 +1,194 @@ +package gitstate + +import ( + "fmt" + "path" + "regexp" + "strconv" + "strings" +) + +// A ruling is only as current as the code it was about. "Streaming fails +// because reader.go loads the whole file" was true at the commit it was +// written against; three commits to reader.go later it may be the one thing +// standing in the way of the fix that now works. Wall-clock age cannot tell +// those apart — a year-old ruling about a file nobody touched still holds, and +// a day-old one about a file rewritten this morning may not. The commit a +// checkpoint recorded can. + +// Drift is how far the files a ruling named have moved since it was recorded. +type Drift struct { + // Commit is where the ruling was recorded. + Commit string `json:"commit"` + // Paths are the files it named that changed since, as it named them. + Paths []string `json:"paths"` + // Commits is how many commits since touched any of them. + Commits int `json:"commits"` +} + +// Note is the drift as one clause for a reader. It says "may", because a +// change to the file is evidence the cause moved, not proof: the commit could +// have been a rename of something else in it. +func (d Drift) Note() string { + return fmt.Sprintf("recorded at %s; %s changed in %d commit%s since, so it may no longer hold", + d.Commit, strings.Join(d.Paths, ", "), d.Commits, plural(d.Commits)) +} + +func plural(n int) string { + if n == 1 { + return "" + } + return "s" +} + +// Anchors answers Since for one repository, remembering what it already asked +// git: a resume reads every ruling in a project's history, and many of them +// name the same file at the same commit. +type Anchors struct { + dir string + memo map[string]int +} + +// NewAnchors reads drift against the repository at dir. Nil when dir is not a +// repository, and a nil Anchors finds nothing, so callers need not check. +func NewAnchors(dir string) *Anchors { + if dir == "" || !isRepo(dir) { + return nil + } + return &Anchors{dir: dir, memo: map[string]int{}} +} + +// maxNamed bounds the files read out of one ruling. Each costs a git call, +// and a ruling naming more than this is a list, not a cause. +const maxNamed = 5 + +var ( + // A recorded sha is vault text, which anyone sharing the vault wrote. Hex + // only, so it can never reach git as an option or a revision expression. + shaPattern = regexp.MustCompile(`^[0-9a-f]{7,40}$`) + // A bare file name: a name, a dot, an extension that starts with a letter, + // so "v1.2" and "3.5GB" are not files. + fileName = regexp.MustCompile(`^[A-Za-z0-9_][A-Za-z0-9_.-]*\.[A-Za-z][A-Za-z0-9]{0,7}$`) + // A line suffix, as in reader.go:42, is not part of the name. + lineSuffix = regexp.MustCompile(`(:\d+)+$`) +) + +// Since reports whether files named in text changed after commit. False when +// the text names no file, when nothing it names changed, or when commit is not +// in this repository — a ruling from another clone, or one whose commit was +// rebased away, has no history here to measure against, and saying nothing is +// truer than saying it moved. +// +// Only files the text names are measured, never the checkpoint's whole file +// list. A session touches a dozen files and rules out one thing about one of +// them; flagging every ruling whose session touched a busy file would mark +// nearly all of them, and a warning on everything is read as a warning on +// nothing. +func (a *Anchors) Since(commit, text string) (Drift, bool) { + if a == nil || !shaPattern.MatchString(commit) { + return Drift{}, false + } + var d Drift + var specs []string + for _, name := range namedFiles(text) { + spec := pathspec(name) + n, ok := a.count(commit, spec) + if !ok { + return Drift{}, false + } + if n > 0 { + d.Paths = append(d.Paths, name) + specs = append(specs, spec) + } + } + if len(specs) == 0 { + return Drift{}, false + } + // Counted again across all of them: one commit touching two named files + // is one change, not two. + if len(specs) == 1 { + d.Commits, _ = a.count(commit, specs[0]) + } else { + d.Commits, _ = a.count(commit, specs...) + } + d.Commit = commit + return d, true +} + +func (a *Anchors) count(commit string, specs ...string) (int, bool) { + key := commit + "\x00" + strings.Join(specs, "\x00") + if n, ok := a.memo[key]; ok { + return n, n >= 0 + } + args := append([]string{"rev-list", "--count", commit + "..HEAD", "--"}, specs...) + out, err := SafeGit(a.dir, gitTimeout, args...) + n, convErr := strconv.Atoi(strings.TrimSpace(out)) + if err != nil || convErr != nil { + // An unknown commit fails here, and so does a timeout. Remembered as + // a refusal so a history full of one foreign sha asks git once. + a.memo[key] = -1 + return 0, false + } + a.memo[key] = n + return n, true +} + +// pathspec matches the name literally where it has a directory, and anywhere +// in the tree where it is a bare file name. Literal because the name is vault +// text: a leading colon would otherwise be read as pathspec magic. +func pathspec(name string) string { + if strings.Contains(name, "/") { + return ":(literal)" + name + } + return ":(glob)**/" + name +} + +// namedFiles picks out what in text reads as a repository file: a relative +// path with a directory, or a bare name with an extension. Paths come first, +// because "internal/parse/reader.go" means one file and "reader.go" may mean +// several. +func namedFiles(text string) []string { + var paths, bare []string + seen := map[string]bool{} + for _, tok := range strings.Fields(text) { + tok = strings.Trim(tok, "`'\"()[]{}<>,;:!?*") + tok = strings.TrimRight(tok, ".") + tok = lineSuffix.ReplaceAllString(tok, "") + tok = strings.TrimPrefix(tok, "./") + if tok == "" || seen[tok] || strings.ContainsAny(tok, "*?[]\\") { + continue + } + switch { + case strings.Contains(tok, "/"): + // Not a URL, not absolute, not climbing out of the repository. + if strings.Contains(tok, "://") || strings.HasPrefix(tok, "/") || + strings.Contains(tok, "..") || !fileName.MatchString(path.Base(tok)) { + continue + } + paths = append(paths, tok) + case fileName.MatchString(tok): + bare = append(bare, tok) + default: + continue + } + seen[tok] = true + } + // A bare name already spelled out as a path is the same file. + for _, b := range bare { + dup := false + for _, p := range paths { + if path.Base(p) == b { + dup = true + break + } + } + if !dup { + paths = append(paths, b) + } + } + out := paths + if len(out) > maxNamed { + out = out[:maxNamed] + } + return out +} diff --git a/internal/gitstate/anchor_test.go b/internal/gitstate/anchor_test.go new file mode 100644 index 0000000..27022e6 --- /dev/null +++ b/internal/gitstate/anchor_test.go @@ -0,0 +1,111 @@ +package gitstate + +import ( + "os" + "path/filepath" + "reflect" + "strings" + "testing" +) + +func commitAt(t *testing.T, dir, name, body, message string) { + t.Helper() + if err := os.MkdirAll(filepath.Join(dir, filepath.Dir(name)), 0o755); err != nil { + t.Fatal(err) + } + commit(t, dir, name, body, message) +} + +func TestARulingAboutAFileRewrittenSinceIsReportedAsDrifted(t *testing.T) { + dir := repo(t) + commitAt(t, dir, "internal/parse/reader.go", "load all\n", "buffered reader") + _, anchor := Head(dir) + commitAt(t, dir, "internal/parse/reader.go", "chunks\n", "read in chunks") + commitAt(t, dir, "internal/parse/reader.go", "chunks, tuned\n", "tune the chunk size") + + d, ok := NewAnchors(dir).Since(anchor, "streaming fails — internal/parse/reader.go loads the whole file") + if !ok { + t.Fatal("two commits to the named file since the ruling were not reported") + } + if d.Commits != 2 || !reflect.DeepEqual(d.Paths, []string{"internal/parse/reader.go"}) || d.Commit != anchor { + t.Fatalf("drift = %+v, want 2 commits to internal/parse/reader.go since %s", d, anchor) + } + if note := d.Note(); !strings.Contains(note, "changed in 2 commits since") || !strings.Contains(note, "may no longer hold") { + t.Fatalf("note = %q", note) + } +} + +func TestABareFileNameIsFoundWhereverItLives(t *testing.T) { + dir := repo(t) + commitAt(t, dir, "internal/parse/reader.go", "load all\n", "buffered reader") + _, anchor := Head(dir) + commitAt(t, dir, "internal/parse/reader.go", "chunks\n", "read in chunks") + + d, ok := NewAnchors(dir).Since(anchor, "reader.go loads the whole file, so streaming fails") + if !ok || d.Commits != 1 || d.Paths[0] != "reader.go" { + t.Fatalf("drift = %+v, %v; want one commit to reader.go", d, ok) + } +} + +func TestARulingAboutAnUntouchedFileStillHolds(t *testing.T) { + dir := repo(t) + commitAt(t, dir, "internal/parse/reader.go", "load all\n", "buffered reader") + _, anchor := Head(dir) + // The repository moved, just not the file the ruling is about. Flagging + // this would put the warning on every ruling in an active repository. + commitAt(t, dir, "internal/other.go", "x\n", "unrelated") + + if d, ok := NewAnchors(dir).Since(anchor, "internal/parse/reader.go loads the whole file"); ok { + t.Fatalf("an unchanged file was reported as drifted: %+v", d) + } +} + +func TestARulingThatNamesNoFileIsNeverFlagged(t *testing.T) { + dir := repo(t) + commitAt(t, dir, "a.go", "1\n", "one") + _, anchor := Head(dir) + commitAt(t, dir, "a.go", "2\n", "two") + + if d, ok := NewAnchors(dir).Since(anchor, "tried raising the timeout to 3.5s in v1.2; no good"); ok { + t.Fatalf("prose with no file name was reported as drifted: %+v", d) + } +} + +func TestACommitFromAnotherRepositoryIsNotMeasured(t *testing.T) { + dir := repo(t) + commitAt(t, dir, "a.go", "1\n", "one") + commitAt(t, dir, "a.go", "2\n", "two") + + // A sha this repository never had: another clone's, or one rebased away. + // There is no history here to say the file moved since it. + if d, ok := NewAnchors(dir).Since("deadbeefcafe", "a.go is wrong"); ok { + t.Fatalf("a foreign commit was measured against: %+v", d) + } +} + +func TestARecordedShaThatIsNotHexNeverReachesGit(t *testing.T) { + dir := repo(t) + commitAt(t, dir, "a.go", "1\n", "one") + commitAt(t, dir, "a.go", "2\n", "two") + + // Vault text. "HEAD~1" would be a working revision, "--all" an option. + for _, sha := range []string{"HEAD~1", "--all", "main", ""} { + if d, ok := NewAnchors(dir).Since(sha, "a.go is wrong"); ok { + t.Fatalf("Since(%q) measured from a non-sha: %+v", sha, d) + } + } +} + +func TestOutsideARepositoryNothingDrifts(t *testing.T) { + if d, ok := NewAnchors(t.TempDir()).Since("abc1234", "a.go"); ok { + t.Fatalf("drift outside a repository: %+v", d) + } +} + +func TestNamedFilesPicksPathsAndFileNamesOutOfProse(t *testing.T) { + got := namedFiles("`internal/parse/reader.go:42` loads it all (see reader.go, go.mod and https://x.io/a.go); v1.2 at 3.5GB, ../etc/passwd, /abs/b.go, internal/parse") + want := []string{"internal/parse/reader.go", "go.mod"} + if !reflect.DeepEqual(got, want) { + t.Fatalf("namedFiles = %q, want %q", got, want) + } +} diff --git a/internal/mcpserver/continuity.go b/internal/mcpserver/continuity.go index 592d13f..40e94f3 100644 --- a/internal/mcpserver/continuity.go +++ b/internal/mcpserver/continuity.go @@ -62,7 +62,9 @@ func (s *Server) lead(pack contextpack.Pack) string { // // here is the project the check is counted under in the usage ledger, which is // not project: the search stays unscoped, the count belongs to the work here. -func (s *Server) beforeYouTry(approach, project, here string) (string, error) { +// dir is the repository the agent stands in, which the rulings found are +// measured against: one blaming a file changed since is marked so. +func (s *Server) beforeYouTry(approach, project, here, dir string) (string, error) { if strings.TrimSpace(approach) == "" { return "", fmt.Errorf("before_you_try needs the approach you are considering") } @@ -73,6 +75,7 @@ func (s *Server) beforeYouTry(approach, project, here string) (string, error) { if err != nil { return "", err } + deadend.MarkDrift(hits, dir) // The corpus is gathered unranked and unfiltered (p=nil, so RecallProcedures // takes its All()-backed fallback with no reinforcement side effect) — Check // does its own lexical-plus-semantic scoring below, and a candidate the diff --git a/internal/mcpserver/server.go b/internal/mcpserver/server.go index f7c55aa..7246863 100644 --- a/internal/mcpserver/server.go +++ b/internal/mcpserver/server.go @@ -748,7 +748,7 @@ func (s *Session) dispatch(name string, args map[string]any) (string, error) { // the vault on purpose, and the project only labels which rulings came // from elsewhere. Scoping it to the current folder would suppress the // cross-project warnings that are the whole reason it exists. - out, err := s.beforeYouTry(argStr(args, "approach"), argStr(args, "project"), s.resolveScope(argStr(args, "project"))) + out, err := s.beforeYouTry(argStr(args, "approach"), argStr(args, "project"), s.resolveScope(argStr(args, "project")), scopeDir(s.roots)) // Counted under the current folder even though the search above is // deliberately unscoped: what is being counted is that this agent is // about to change something here, which is the point a session starts From d788ccf48ac64088f6f24899a23467a6d5f9e313 Mon Sep 17 00:00:00 2001 From: Coder8124 Date: Sun, 4 Oct 2026 15:17:11 -0700 Subject: [PATCH 2/2] Move the continuity eval suite into the sibling logos-bench repo, driven over its own CLI --- bench/README.md | 243 --------- bench/adapters/letta_adapter.py | 290 ---------- bench/adapters/mem0_adapter.py | 152 ------ bench/adapters/mempalace_adapter.py | 159 ------ cmd/logos/bench.go | 38 ++ cmd/logos/evalbench.go | 155 ------ cmd/logos/help.go | 7 +- cmd/logos/main.go | 6 +- internal/eval/adapters/baselines.go | 236 -------- internal/eval/adapters/bridge.go | 235 -------- internal/eval/adapters/logos.go | 273 ---------- internal/eval/eval.go | 180 ------- internal/eval/eval_test.go | 217 -------- internal/eval/report.go | 245 --------- internal/eval/report_text_test.go | 17 - internal/eval/run.go | 109 ---- internal/eval/scenarios.go | 802 ---------------------------- internal/eval/score.go | 340 ------------ internal/eval/variants.go | 259 --------- internal/eval/variants_test.go | 127 ----- internal/memory/bench.go | 222 +++++++- 21 files changed, 252 insertions(+), 4060 deletions(-) delete mode 100644 bench/README.md delete mode 100644 bench/adapters/letta_adapter.py delete mode 100644 bench/adapters/mem0_adapter.py delete mode 100644 bench/adapters/mempalace_adapter.py delete mode 100644 cmd/logos/evalbench.go delete mode 100644 internal/eval/adapters/baselines.go delete mode 100644 internal/eval/adapters/bridge.go delete mode 100644 internal/eval/adapters/logos.go delete mode 100644 internal/eval/eval.go delete mode 100644 internal/eval/eval_test.go delete mode 100644 internal/eval/report.go delete mode 100644 internal/eval/report_text_test.go delete mode 100644 internal/eval/run.go delete mode 100644 internal/eval/scenarios.go delete mode 100644 internal/eval/score.go delete mode 100644 internal/eval/variants.go delete mode 100644 internal/eval/variants_test.go diff --git a/bench/README.md b/bench/README.md deleted file mode 100644 index 9d9507d..0000000 --- a/bench/README.md +++ /dev/null @@ -1,243 +0,0 @@ -# The continuity benchmark - -A benchmark for what survives when one agent stops and another starts. - -Recall benchmarks — LongMemEval and its relatives — measure whether a store can -surface the session holding an answer. That is the easy half. It says nothing -about the case this project exists for: an agent stops mid-task, a *different* -agent arrives, and nobody re-explains anything. Nothing scores that, so nothing -optimises for it. - -The suite lives in [`internal/eval`](../internal/eval); this directory holds the -bridges to memory systems that are not written in Go. - -## Running it - -```sh -go build -o bin/logos ./cmd/logos -./bin/logos bench continuity list # what is measured, and what we expect to fail -./bin/logos bench continuity # every system installed on this machine -./bin/logos bench continuity --verbose # per scenario, with what was missed -./bin/logos bench continuity --dump # the raw retrieved context, for auditing labels -./bin/logos bench continuity --logos-only # skip the comparison, ~12s -``` - -Useful flags: `--only ` to narrow, `--no-embed` to score -lexical and graph retrieval with no model loaded. - -## What it measures - -| Metric | Meaning | -|---|---| -| **pass** | met every bar the scenario set, signal labels included | -| **fidelity** | recall × (1 − leakage) — carrying what is needed while keeping out what is wrong | -| **recall** | share of required facts that arrived | -| **leak** | share of facts that should have been suppressed but were not | -| **signal** | share of meta-properties exhibited: that context is stale, that the store does not know, who found a thing | -| **dens/1k** | required facts per 1000 tokens — the price of that recall | - -Fidelity is multiplicative on purpose. A response carrying every required fact -alongside a superseded price is not half right: the agent acts on the stale -number, and the correct facts beside it do not undo that. - -Scoring is mechanical — substring matching over normalised text against -hand-written gold labels. Every headline number is reproducible with no model -running. The one place the harness asserts rather than measures is the -durability family, and it says so in the report. - -Two things the matcher does that are worth knowing: - -- **The question is stripped from the answer before matching.** Systems that - head their output with the task they were given would otherwise satisfy any - label whose wording overlaps the question. One case scored a clean 100% this - way before the fix, matching `"after signing"` in an echoed header while - returning two undated facts in arbitrary order. -- **Surface variants are the scenario author's job.** Write - `Any: {"71 percent", "71%"}` rather than hoping a cleverer matcher guesses. A - fuzzy matcher that silently accepts near-misses is how a benchmark starts - flattering everyone equally. - -## The honest half - -Roughly a third of the suite is marked `KnownWeakness` — cases logos fails -today. Several were found by an agent picking apart logos's own handoff output -during a live test: stale context presented without its age, circular sourcing -where prose restates a claim the data contradicts, and abstention, which the -LongMemEval harness in `internal/memory/bench.go` filters out before scoring. - -The report ends with a **predictions that were wrong** section comparing each -label against what actually happened. A case marked a weakness that starts -passing is progress worth noticing; one marked a strength that starts failing is -a regression the averages would otherwise absorb. Either way the label is now -wrong and somebody has to look. - -## The systems compared - -Baselines are built in and always run: - -- **none** — the floor. Whatever it scores, the suite gives away for free. -- **static-file** — a hand-maintained `CLAUDE.md`. Receives documents only, - because nobody appends every dead end to a rules file. This is the real - incumbent: a memory system that cannot beat it is not worth installing. -- **recency-window** — the newest events that fit. What conversation compaction - approximates: no notion of relevance, only of lateness. -- **full-dump** — the entire history, newest first, to the budget. The ceiling, - and the reason density is a headline column. -- **vector-rag** — top-k cosine, no lexical arm, no graph, no checkpoint. What - most "memory layers" reduce to once the marketing is removed. - -Third-party systems run as subprocesses over a JSON-lines bridge -(`internal/eval/adapters/bridge.go`) and are skipped with a printed reason if -not installed — a comparison table with a quietly missing column flatters -whoever is left. - -```sh -cd bench/adapters -uv venv .venv-mem0 && VIRTUAL_ENV=.venv-mem0 uv pip install "mem0ai[extras]" ollama -uv venv .venv-mempalace && VIRTUAL_ENV=.venv-mempalace uv pip install mempalace -uv venv .venv-letta && VIRTUAL_ENV=.venv-letta uv pip install letta asyncpg pgvector psycopg2-binary -``` - -All are pointed at Ollama, so no API keys are needed and the comparison is -like-for-like: same models, same hardware, no network. - -### Letta needs a little more - -Letta 0.16 is a server, and since it dropped SQLite it needs PostgreSQL with -pgvector. Nothing here touches an existing database — the cluster is scratch and -can be deleted afterwards. - -```sh -brew install pgvector # ships for postgresql@17 and @18 - -PGD=~/.logos-bench-pg -initdb -D "$PGD" -U letta --auth=trust # use the @17 or @18 initdb -pg_ctl -D "$PGD" -o "-p 5433 -k $PGD" -l "$PGD/server.log" start -psql -h "$PGD" -p 5433 -U letta -d postgres -c "CREATE DATABASE letta OWNER letta;" -psql -h "$PGD" -p 5433 -U letta -d letta -c "CREATE EXTENSION vector;" - -export LETTA_PG_URI="postgresql://letta@127.0.0.1:5433/letta" -export LETTA_DIR=~/.logos-bench-letta -export OLLAMA_BASE_URL="http://localhost:11434" # without this the model list is empty -.venv-letta/bin/letta server --port 8289 -``` - -`OLLAMA_BASE_URL` is the one that fails quietly. Without it the server starts, -answers `/v1/health/`, and still serves a model list from rows written at a -previous start — so `/v1/models/` looks right while `agents.create` rejects -every handle with `must be one of []`. Set it on every start, not just the -first. - -The published wheel carries no Alembic migrations, so the schema has to be -created once from the ORM metadata, and one column needs a default the ORM does -not declare: - -```sh -.venv-letta/bin/python -c " -from sqlalchemy import create_engine -import letta.orm -from letta.orm.base import Base -Base.metadata.create_all(create_engine('postgresql+psycopg2://letta@127.0.0.1:5433/letta'))" - -psql -h "$PGD" -p 5433 -U letta -d letta \ - -c "CREATE SEQUENCE IF NOT EXISTS messages_sequence_id_seq OWNED BY messages.sequence_id;" \ - -c "ALTER TABLE messages ALTER COLUMN sequence_id SET DEFAULT nextval('messages_sequence_id_seq');" -``` - -Tear down with `pg_ctl -D ~/.logos-bench-pg stop && rm -rf ~/.logos-bench-pg -~/.logos-bench-letta`. - -### Fairness notes - -Each adapter is written to make its system look as good as that system can look. -logos uses checkpoints because it has them; a store with only `add()` and -`search()` receives the same information flattened into complete prose rather -than withheld. Where a system loses, it should lose for what it is, not for how -it was driven here. - -- **mem0** is run with `infer=False`. Its `add(infer=True)` path runs an LLM - over every write to extract and reconcile facts — its real design, and its - real advantage on contradictions — but it is one model call per event, and a - single scenario writes over two hundred. `infer=False` stores text verbatim - and embeds it, which for a retrieval benchmark is if anything favourable: - nothing is lost to extraction. `MEM0_INFER=1` runs it the authentic way. - `fastembed` is installed so its BM25 arm is available, since logos is scored - with a lexical arm too. -- **mem0 telemetry is disabled** in the shim. It ships analytics on by default - and opens a PostHog client at import; a benchmark claiming every system runs - locally has to mean it. -- **Letta** is scored by default on its **archival memory** - (`passages.create` / `passages.search`), which is the part of its three-tier - memory that corresponds to what the other systems do. As with mem0, that - shortcut is favourable on retrieval and unfavourable on reconciliation. - `LETTA_AGENT_LOOP=1` runs it the authentic way: every write becomes a real - agent turn with the base memory tools available, and the model decides what - belongs in core memory and what gets filed in archival. - - Reads are identical in both modes — archival search, plus core memory when - the loop has filled it. The benchmark scores *the context a system hands the - next agent*, so asking the agent a question and grading its reply would - measure something no other row measures; but discarding core memory would run - the loop and then throw away its main output. - - In agent-loop mode the model actually runs, so it must be local. The default - is `ollama/glm-4.7-flash:latest`; override with `LETTA_MODEL`, using a handle - `client.models.list()` reports, since Letta filters Ollama models by - tool-calling support and small ones may not appear. - -### What authentic mode costs - -The suite is **596 write events** across 32 scenarios, 201 of them in -`scale-haystack` alone. Measured on one machine: - -| Mode | Per write | Whole suite | -|---|---|---| -| mem0 `infer=False`, Letta archival | milliseconds | ~3 minutes | -| `MEM0_INFER=1` | ~10 s | ~1.7 hours | -| `LETTA_AGENT_LOOP=1` | ~33 s | ~5.4 hours | - -Timed on `supersession-current-value`: Letta's loop took 12m29s for 23 events -and issued **108 chat completions for those 23 writes — about 4.7 model calls -each**, since a turn is reason → call a tool → read the result → often step -again. An earlier note here guessed "roughly a turn each", which was low. - -Treat ~7 hours for the pair as a **floor, not an estimate**. Both systems grow -their context as memory accumulates — Letta's context estimate climbed 2,379 → -8,917 tokens across those 23 events — so `scale-haystack`'s 201 events cost -more than 8.7× a 23-event scenario, not the same per write. - -```sh -MEM0_INFER=1 LETTA_AGENT_LOOP=1 go run ./cmd/logos bench continuity -``` - -What it buys, measured on the scenario the caveat pointed at: Letta goes from -0% to 50% fidelity, correctly dropping the first superseded price and still -leaking the second. mem0 stays at 0% and fails in a new way — it synthesises -"currently $199, but will increase to $229 … then to $249", reading a history -of revisions as a schedule of increases. Full write-up in -[docs/continuity-benchmark.md §6.1](../docs/continuity-benchmark.md). - -Note that `infer=True` also wants spaCy (`mem0ai[nlp]`), which the install line -above does not pull; without it that path logs a fallback warning. -- **Letta embeddings must be given a `/v1` endpoint.** It routes every - embedding through its OpenAI-compatible client and appends `/embeddings` to - whatever base it is handed, regardless of `embedding_endpoint_type`, and - Ollama serves that path only under `/v1`. Without it every write returns a - 500 that the server logs as "an unknown error occurred". -- **MemPalace** is the only third-party system here with a handoff story of its - own — it ships an `artifact` command for agent handoffs and a `wake-up` - context command. The benchmark does not use them: `artifact` is an exact-file - exchange rather than a retrieval surface, and wiring it would score a - different feature than the one under test. - -## Adding a system - -Implement `eval.Adapter` in Go, or drop a `_adapter.py` in -`bench/adapters/` speaking the bridge protocol: one JSON object per line in, one -per line out, ops `reset` / `write` / `read` / `close`, plus `--probe` exiting -non-zero with a readable reason when the system cannot run. `Discover()` picks -it up automatically. - -Systems that distinguish a source of truth from a rebuildable cache should also -implement `eval.Durable`. Not implementing it is itself an answer, and the suite -reports it as one. diff --git a/bench/adapters/letta_adapter.py b/bench/adapters/letta_adapter.py deleted file mode 100644 index babb941..0000000 --- a/bench/adapters/letta_adapter.py +++ /dev/null @@ -1,290 +0,0 @@ -#!/usr/bin/env python3 -"""Letta (formerly MemGPT) under the continuity benchmark. - -Speaks the bridge protocol on stdin/stdout: one JSON object per line in, one -per line out. See internal/eval/adapters/bridge.go. - -What is measured, and why -------------------------- -Letta is an agent framework, not a memory library. Its memory has three parts: -core memory blocks that live in the context window, recall memory over the -message history, and *archival* memory — an embedded passage store the agent -searches on demand. Archival is the part that corresponds to what every other -system in this table does, so archival is what is scored: passages in, -semantic search out. - -Two modes, and the difference is the whole caveat -------------------------------------------------- -**Default (`LETTA_AGENT_LOOP` unset).** Passages go straight into archival and -come back by semantic search. No completion is ever requested, so the run is -fast and deterministic. On a retrieval benchmark this is *favourable* to Letta — -nothing is lost to a small model's extraction — and unfavourable on the -reconciliation cases, where deciding that a new number replaces an old one is -exactly the work the loop does and this mode skips. - -**`LETTA_AGENT_LOOP=1`.** Every write becomes a real agent turn: the event is -sent as a message, and the model decides what to put in core memory and what to -file in archival, with the base memory tools available to it. That is Letta as -its authors intend it, and it is the mode that answers the caveat above. - -Reads are deliberately identical in both modes: archival search, plus whatever -the loop wrote into core memory. The benchmark scores *the context a system -hands the next agent*, not a generated answer, so asking the agent a question -and grading its reply would be measuring something no other row measures. Core -memory is included because in agent-loop mode that is where the loop puts what -it considers settled — leaving it out would run the loop and then discard its -main output. - -Cost, measured rather than guessed: the suite is 596 write events across 32 -scenarios (201 of them in `scale-haystack` alone). A write is not one model -call — timing `supersession-current-value` gave 108 chat completions for 23 -writes, about 4.7 each, at ~33 s per event, because a turn is reason → call a -tool → read the result → often step again. That puts the suite near 5.4 hours, -and superlinear rather than flat: context grows as memory accumulates, so long -scenarios cost more per write than short ones. - -What the loop buys, on that scenario: fidelity 0% → 50%. Archival mode returns -all three superseded prices, exactly as plain vector search does; the loop -drops the first entirely and leaks only the second. The scenario still fails, -because `pass` is conjunctive and one leak scores like ten — but the difference -is real, and it is only visible in `fidelity`. - -Running it requires more than an import: Letta 0.16 needs a PostgreSQL server -with pgvector and a running `letta server`. See bench/README.md. The probe below -fails cleanly when that is not up, so the row is dropped rather than reported -as zeros. -""" - -import json -import os -import sys -import urllib.request - -BASE = os.environ.get("LETTA_BASE_URL", "http://localhost:8289") -# The /v1 suffix is required: Letta routes every embedding through its -# OpenAI-compatible client and appends "/embeddings" to whatever endpoint it -# is given, regardless of embedding_endpoint_type. Ollama serves that path -# only under /v1. -OLLAMA = os.environ.get("LETTA_OLLAMA", "http://localhost:11434/v1") -EMBED_MODEL = os.environ.get("LETTA_EMBED", "nomic-embed-text") -EMBED_DIMS = int(os.environ.get("LETTA_EMBED_DIMS", "768")) -AGENT_LOOP = os.environ.get("LETTA_AGENT_LOOP", "") == "1" - -# In archival-only mode the agent still needs a model handle to be created at -# all, even though no completion is ever requested; letta-free is the built-in -# default and costs nothing because the loop never runs. -# -# In agent-loop mode the model actually runs, so it has to be local — a -# benchmark that claims every system stays on the machine cannot quietly send -# 596 events to a hosted endpoint. The handle must be one `letta server` has -# discovered; ask it with `client.models.list()`. Note that Letta filters -# Ollama models by tool-calling support, so small models may not appear. -DEFAULT_MODEL = "ollama/glm-4.7-flash:latest" if AGENT_LOOP else "letta/letta-free" -MODEL = os.environ.get("LETTA_MODEL", DEFAULT_MODEL) - - -def probe(): - """Exit non-zero with a readable reason if this system cannot run here.""" - try: - from letta_client import Letta # noqa: F401 - except Exception as e: - print(f"letta_client not importable: {e}", file=sys.stderr) - return 1 - try: - with urllib.request.urlopen(f"{BASE}/v1/health/", timeout=5) as r: - json.loads(r.read()) - except Exception: - print( - f"letta server unreachable at {BASE} — start it with " - "`letta server --port 8289` (needs PostgreSQL + pgvector)", - file=sys.stderr, - ) - return 1 - return 0 - - -class Adapter: - def __init__(self): - self.client = None - self.agent_id = None - - def reset(self): - from letta_client import Letta - - self.close() - self.client = Letta(base_url=BASE) - agent = self.client.agents.create( - name=f"bench-{os.getpid()}-{id(self)}", - model=MODEL, - # Explicit config rather than the `ollama/...` handle: the handle - # resolves to a provider row written at first server start, which - # routes embeddings through the OpenAI-compatible path. Naming the - # endpoint type here uses Ollama's native embeddings API instead. - embedding_config={ - "embedding_endpoint_type": "ollama", - "embedding_endpoint": OLLAMA, - "embedding_model": EMBED_MODEL, - "embedding_dim": EMBED_DIMS, - "batch_size": 32, - }, - # Archival-only: core memory is filled by the agent loop, which is - # not being run, so there is nothing to give it. Agent-loop mode - # needs a block to write into and the base tools to write with. - memory_blocks=( - [ - { - "label": "project", - "value": "", - "description": ( - "What is currently true about the work: decisions and " - "the reasons for them, what has been ruled out and why, " - "open questions, and the next step. Replace a fact when " - "it is superseded rather than appending beside it." - ), - "limit": 4000, - } - ] - if AGENT_LOOP - else [] - ), - include_base_tools=AGENT_LOOP, - ) - self.agent_id = agent.id - - def write(self, ev): - # `flat` is the harness's prose rendering of an event — for a checkpoint - # that means task, decisions, what failed and what is next, spelled out. - # Letta has no checkpoint primitive, so this is the fullest form it can - # take. - if AGENT_LOOP: - return self._write_through_loop(ev) - - # created_at carries the event's real time, which the suite backdates, - # so Letta's temporal filters see the same history logos does. - kwargs = {"text": ev["flat"]} - if ev.get("ts"): - kwargs["created_at"] = self._iso(ev["ts"]) - self.client.agents.passages.create(self.agent_id, **kwargs) - - def _write_through_loop(self, ev): - """One agent turn per event: the model decides what to keep and where. - - The message says when the event happened because the loop cannot see - `created_at` on a passage it has not written yet, and half the suite - turns on which of two facts is later. It does not tell the agent *how* - to store anything beyond that — choosing between core and archival, and - deciding that a new number replaces an old one, is the work under test. - """ - when = f" (recorded {self._iso(ev['ts'])})" if ev.get("ts") else "" - self.client.agents.messages.create( - self.agent_id, - messages=[ - { - "role": "user", - "content": f"Record this{when}:\n\n{ev['flat']}", - } - ], - ) - - @staticmethod - def _iso(ts): - import datetime - - return datetime.datetime.fromtimestamp(ts, datetime.timezone.utc).isoformat() - - def read(self, q): - query = q["task"] - if q.get("project"): - query = f"{q['project']}: {query}" - hits = self.client.agents.passages.search(self.agent_id, query=query, top_k=20) - - results = getattr(hits, "results", None) - if results is None: - results = hits if isinstance(hits, list) else [] - - budget = q.get("budget") or 2000 - out, used = [], 0 - - # Core memory first, and only in agent-loop mode, where it holds what - # the loop decided was settled. It is what Letta would actually put in - # the next agent's context window, so leaving it out would run the loop - # and then throw away its main output. It goes first because it is what - # Letta itself ranked as most important, and the budget is tight. - if AGENT_LOOP: - for text in self._core_memory(): - cost = len(text) // 4 + 1 - if used + cost > budget and out: - break - out.append(text) - used += cost - - for h in results: - text = getattr(h, "content", None) or getattr(h, "text", None) or "" - if not text and isinstance(h, dict): - text = h.get("content") or h.get("text") or "" - if not text: - continue - cost = len(text) // 4 + 1 - if used + cost > budget and out: - break - out.append(text) - used += cost - return "\n\n".join(out) - - def _core_memory(self): - """The agent's core memory blocks, as plain text. Empty on any failure — - a read that raises would score as a crash rather than as a miss.""" - try: - blocks = self.client.agents.blocks.list(self.agent_id) - except Exception: - return [] - texts = [] - for b in blocks: - value = getattr(b, "value", None) or "" - if value.strip(): - texts.append(value.strip()) - return texts - - def close(self): - if self.client and self.agent_id: - try: - self.client.agents.delete(self.agent_id) - except Exception: - pass - self.agent_id = None - self.client = None - - -def main(): - if "--probe" in sys.argv: - sys.exit(probe()) - - adapter = Adapter() - for line in sys.stdin: - line = line.strip() - if not line: - continue - try: - req = json.loads(line) - op = req.get("op") - if op == "reset": - adapter.reset() - reply = {"ok": True} - elif op == "write": - adapter.write(req["event"]) - reply = {"ok": True} - elif op == "read": - reply = {"ok": True, "text": adapter.read(req["query"])} - elif op == "close": - adapter.close() - reply = {"ok": True} - else: - reply = {"ok": False, "error": f"unknown op {op!r}"} - except Exception as e: - reply = {"ok": False, "error": f"{type(e).__name__}: {e}"} - sys.stdout.write(json.dumps(reply) + "\n") - sys.stdout.flush() - - -if __name__ == "__main__": - main() diff --git a/bench/adapters/mem0_adapter.py b/bench/adapters/mem0_adapter.py deleted file mode 100644 index b3c6a10..0000000 --- a/bench/adapters/mem0_adapter.py +++ /dev/null @@ -1,152 +0,0 @@ -#!/usr/bin/env python3 -"""mem0 under the continuity benchmark. - -Speaks the bridge protocol on stdin/stdout: one JSON object per line in, one -per line out. See internal/eval/adapters/bridge.go. - -Everything runs against Ollama, so no API key is needed and nothing leaves the -machine — the same local models logos is scored with. - -One deviation worth stating plainly. mem0's `add(infer=True)` runs an LLM over -every write to extract and reconcile facts, which is its real design and its -real advantage on contradictions. It is also one model call per event, and the -suite writes over two hundred events in a single scenario. The default here is -infer=False, which stores text verbatim and embeds it. That is faster and, for -a retrieval benchmark, mostly *favourable* to mem0: nothing is lost to -extraction. Set MEM0_INFER=1 to run it the slow, authentic way; the write-up -reports both where it was affordable to. -""" - -import json -import os -import shutil -import sys -import tempfile - -# mem0 ships analytics on by default and opens a PostHog client at import. A -# benchmark that claims every system runs locally has to actually mean it, so -# this is set before mem0 is imported anywhere below. -os.environ.setdefault("MEM0_TELEMETRY", "False") -os.environ.setdefault("ANONYMIZED_TELEMETRY", "False") - -OLLAMA = os.environ.get("OLLAMA_HOST", "http://localhost:11434") -LLM_MODEL = os.environ.get("MEM0_LLM", "gemma3:4b") -EMBED_MODEL = os.environ.get("MEM0_EMBED", "nomic-embed-text") -EMBED_DIMS = int(os.environ.get("MEM0_EMBED_DIMS", "768")) -INFER = os.environ.get("MEM0_INFER", "") == "1" -USER = "bench" - - -def probe(): - """Exit non-zero with a readable reason if this system cannot run here.""" - try: - import ollama # noqa: F401 - from mem0 import Memory # noqa: F401 - except Exception as e: # pragma: no cover - diagnostic path - print(f"mem0 not importable: {e}", file=sys.stderr) - return 1 - try: - import urllib.request - - urllib.request.urlopen(f"{OLLAMA}/api/tags", timeout=5).read() - except Exception: - print(f"ollama unreachable at {OLLAMA}", file=sys.stderr) - return 1 - return 0 - - -class Adapter: - def __init__(self): - self.mem = None - self.dir = None - - def reset(self): - from mem0 import Memory - - self.close() - self.dir = tempfile.mkdtemp(prefix="mem0-bench-") - self.mem = Memory.from_config( - { - "llm": { - "provider": "ollama", - "config": {"model": LLM_MODEL, "ollama_base_url": OLLAMA}, - }, - "embedder": { - "provider": "ollama", - "config": {"model": EMBED_MODEL, "ollama_base_url": OLLAMA}, - }, - "vector_store": { - "provider": "qdrant", - "config": { - "path": self.dir, - "collection_name": "bench", - "embedding_model_dims": EMBED_DIMS, - }, - }, - } - ) - - def write(self, ev): - # `flat` is the harness's prose rendering, which spells a checkpoint out - # in full — task, decisions, what failed, what is next. mem0 has no - # checkpoint primitive, so this is the most complete form it can accept. - self.mem.add(ev["flat"], user_id=USER, infer=INFER) - - def read(self, q): - query = q["task"] - if q.get("project"): - query = f"{q['project']}: {query}" - hits = self.mem.search(query, filters={"user_id": USER}, top_k=20) - results = hits.get("results", hits) if isinstance(hits, dict) else hits - - budget = q.get("budget") or 2000 - out, used = [], 0 - for h in results: - text = h.get("memory") or h.get("text") or "" - cost = len(text) // 4 + 1 - if used + cost > budget and out: - break - out.append(text) - used += cost - return "\n\n".join(out) - - def close(self): - self.mem = None - if self.dir: - shutil.rmtree(self.dir, ignore_errors=True) - self.dir = None - - -def main(): - if "--probe" in sys.argv: - sys.exit(probe()) - - adapter = Adapter() - for line in sys.stdin: - line = line.strip() - if not line: - continue - try: - req = json.loads(line) - op = req.get("op") - if op == "reset": - adapter.reset() - reply = {"ok": True} - elif op == "write": - adapter.write(req["event"]) - reply = {"ok": True} - elif op == "read": - reply = {"ok": True, "text": adapter.read(req["query"])} - elif op == "close": - adapter.close() - reply = {"ok": True} - else: - reply = {"ok": False, "error": f"unknown op {op}"} - except Exception as e: - reply = {"ok": False, "error": f"{type(e).__name__}: {e}"} - sys.stdout.write(json.dumps(reply) + "\n") - sys.stdout.flush() - - -if __name__ == "__main__": - main() diff --git a/bench/adapters/mempalace_adapter.py b/bench/adapters/mempalace_adapter.py deleted file mode 100644 index 1d5f067..0000000 --- a/bench/adapters/mempalace_adapter.py +++ /dev/null @@ -1,159 +0,0 @@ -#!/usr/bin/env python3 -"""MemPalace under the continuity benchmark. - -Speaks the bridge protocol on stdin/stdout. See internal/eval/adapters/bridge.go. - -MemPalace is file-oriented: you put markdown in a directory, `mine` files it -into rooms and drawers, and `search` retrieves with cosine plus BM25. That maps -cleanly onto the suite — every event becomes a file, which is the shape the -system was designed for and its strongest form. - -Isolation is per scenario via MEMPALACE_PALACE_PATH, so no run inherits another -run's drawers. Mining is run with --no-llm and the local embedding model, for -the same reason mem0 is run against Ollama: every system in the table gets the -same hardware and no network. - -It is worth saying that MemPalace is the only third-party system here with a -handoff story of its own — it ships an `artifact` command for agent handoffs and -a `wake-up` context command. This benchmark does not use them: `artifact` is an -exact-file exchange rather than a retrieval surface, and wiring it would be -scoring a different feature than the one under test. The search path is what -answers "what do we know", and that is what is measured. -""" - -import json -import os -import shutil -import subprocess -import sys -import tempfile - -HERE = os.path.dirname(os.path.abspath(__file__)) -MP = os.path.join(HERE, ".venv-mempalace", "bin", "mempalace") -EMBED_MODEL = os.environ.get("MEMPALACE_EMBEDDING_MODEL", "") - - -def probe(): - if not os.path.exists(MP): - print(f"mempalace CLI not found at {MP}", file=sys.stderr) - return 1 - try: - import mempalace # noqa: F401 - except Exception as e: - print(f"mempalace not importable: {e}", file=sys.stderr) - return 1 - return 0 - - -class Adapter: - def __init__(self): - self.root = None - self.palace = None - self.n = 0 - self.mined = False - - def _env(self): - env = dict(os.environ) - env["MEMPALACE_PALACE_PATH"] = self.palace - if EMBED_MODEL: - env["MEMPALACE_EMBEDDING_MODEL"] = EMBED_MODEL - return env - - def _run(self, args, timeout=900): - # stdin must be closed, not inherited. `init` asks "Mine this directory - # now? [Y/n]" even under --yes, and with the bridge's stdin attached it - # blocks forever waiting for an answer that will never come. - return subprocess.run( - [MP] + args, - env=self._env(), - stdin=subprocess.DEVNULL, - capture_output=True, - text=True, - timeout=timeout, - ) - - def reset(self): - self.close() - self.root = tempfile.mkdtemp(prefix="mempalace-src-") - self.palace = tempfile.mkdtemp(prefix="mempalace-db-") - os.makedirs(os.path.join(self.root, "notes"), exist_ok=True) - self.n = 0 - self.mined = False - - def write(self, ev): - # One file per event. Titled so the miner has something to key on, and - # carrying the flattened prose so a checkpoint arrives complete. - self.n += 1 - title = ev.get("title") or ev.get("task") or f"{ev['kind']} {self.n}" - body = f"# {title}\n\n{ev['flat']}\n" - name = f"{self.n:04d}-{ev['kind']}.md" - with open(os.path.join(self.root, "notes", name), "w") as fh: - fh.write(body) - self.mined = False - - def _mine(self): - init = self._run(["init", self.root, "--yes", "--no-llm"]) - if init.returncode != 0: - raise RuntimeError(f"init failed: {init.stderr.strip()[-300:]}") - mine = self._run(["mine", self.root]) - if mine.returncode != 0: - raise RuntimeError(f"mine failed: {mine.stderr.strip()[-300:]}") - self.mined = True - - def read(self, q): - if not self.mined: - self._mine() - query = q["task"] - if q.get("project"): - query = f"{q['project']}: {query}" - res = self._run(["search", query]) - if res.returncode != 0: - raise RuntimeError(f"search failed: {res.stderr.strip()[-300:]}") - - # The CLI prints a banner, then indented result blocks. Keep the blocks. - text = res.stdout - budget = q.get("budget") or 2000 - if len(text) // 4 + 1 > budget: - text = text[: budget * 4] - return text - - def close(self): - for path in (self.root, self.palace): - if path: - shutil.rmtree(path, ignore_errors=True) - self.root = self.palace = None - - -def main(): - if "--probe" in sys.argv: - sys.exit(probe()) - - adapter = Adapter() - for line in sys.stdin: - line = line.strip() - if not line: - continue - try: - req = json.loads(line) - op = req.get("op") - if op == "reset": - adapter.reset() - reply = {"ok": True} - elif op == "write": - adapter.write(req["event"]) - reply = {"ok": True} - elif op == "read": - reply = {"ok": True, "text": adapter.read(req["query"])} - elif op == "close": - adapter.close() - reply = {"ok": True} - else: - reply = {"ok": False, "error": f"unknown op {op}"} - except Exception as e: - reply = {"ok": False, "error": f"{type(e).__name__}: {e}"} - sys.stdout.write(json.dumps(reply) + "\n") - sys.stdout.flush() - - -if __name__ == "__main__": - main() diff --git a/cmd/logos/bench.go b/cmd/logos/bench.go index 9eb5d30..3cdf9d3 100644 --- a/cmd/logos/bench.go +++ b/cmd/logos/bench.go @@ -44,3 +44,41 @@ func runBench(path string, n int, hybrid bool) error { } return nil } + +// runBenchQA runs the end-to-end LongMemEval pipeline — retrieve, answer, +// grade — the metric actually comparable to another memory system's +// published LongMemEval score, unlike runBench's recall@k. +func runBenchQA(path string, n int, hybrid bool, depth int) error { + rt, err := openRouter() + if err != nil { + return err + } + embed, _ := rt.Model(router.T0) + genModel, err := rt.ModelFor(router.T1, false) + if err != nil { + return err + } + judgeModel, err := rt.ModelFor(router.T1, true) + if err != nil { + return err + } + + mode := "hybrid (vector+BM25)" + if !hybrid { + mode = "vector-only" + } + fmt.Printf("· LongMemEval QA accuracy over %d instances · %s · gen %s · judge %s\n", n, mode, genModel, judgeModel) + results, err := memory.RunLongMemEvalQA(rt.Local(), embed, genModel, judgeModel, path, n, hybrid, depth, func(done, total int) { + fmt.Printf("\r %d/%d …", done, total) + }) + if err != nil { + return err + } + fmt.Printf("\r%40s\r", "") + + fmt.Printf("\n%-26s %8s %6s\n", "category", "accuracy", "n") + for _, r := range results { + fmt.Printf("%-26s %7.1f%% %6d\n", r.Category, r.Accuracy()*100, r.N) + } + return nil +} diff --git a/cmd/logos/evalbench.go b/cmd/logos/evalbench.go deleted file mode 100644 index 2247d25..0000000 --- a/cmd/logos/evalbench.go +++ /dev/null @@ -1,155 +0,0 @@ -package main - -import ( - "fmt" - "os" - "strings" - - "github.com/Coder8124/logos/internal/eval" - "github.com/Coder8124/logos/internal/eval/adapters" - "github.com/Coder8124/logos/internal/provider" - "github.com/Coder8124/logos/internal/router" -) - -// runContinuityBench runs the handoff-and-memory suite over every system that -// can be reached on this machine. -// -// Systems that are not installed are skipped with a line saying so rather than -// silently omitted, because a comparison table with a quietly missing column is -// a comparison table that flatters whoever is left. -func runContinuityBench(args []string) error { - var ( - only = flagStr(args, "--only", "") - verbose = hasFlag(args, "--verbose") - noEmbed = hasFlag(args, "--no-embed") - bare = hasFlag(args, "--logos-only") - dump = hasFlag(args, "--dump") - varied = hasFlag(args, "--variants") - ) - - suite := eval.Select(eval.Suite(), only) - if len(suite) == 0 { - return fmt.Errorf("no scenarios match %q", only) - } - - var embed *provider.Provider - var model string - if !noEmbed { - rt, err := openRouter() - if err != nil { - return fmt.Errorf("%w\n(run with --no-embed to score lexical and graph retrieval only)", err) - } - model, err = rt.Model(router.T0) - if err != nil { - return err - } - embed = rt.Local() - } - - fmt.Printf("· %s\n", eval.Composition(suite)) - if only != "" { - fmt.Printf("· filtered to %q\n", only) - } - if varied { - // Selected before expanding, so --only names a scenario and gets - // every variant of it. - suite = eval.Expand(suite, variantSeeds) - fmt.Printf("· %d cases with variants: each scenario as written, reworded, and among %d distractor sets\n", - len(suite), variantSeeds) - } - if embed == nil { - fmt.Println("· no embeddings: lexical and graph retrieval only") - } else { - fmt.Printf("· embeddings: %s\n", model) - } - - logos, err := adapters.NewLogos(embed, model) - if err != nil { - return err - } - defer logos.Close() - - systems := []eval.Adapter{logos} - if !bare { - systems = append(systems, &adapters.StaticFile{}, &adapters.Recency{}, &adapters.Dump{}) - if embed != nil { - systems = append(systems, adapters.NewVectorRAG(embed, model, 8)) - } - systems = append(systems, adapters.None{}) - - // Third-party systems, if they are installed. Each runs in a subprocess - // speaking the bridge protocol; see bench/adapters/. - for _, ext := range adapters.Discover() { - systems = append(systems, ext) - } - } - - var results []eval.Result - for _, sys := range systems { - opts := eval.Options{ - Progress: func(adapter, scenario string, done, total int) { - fmt.Printf("\r %-18s %3d/%d %-36s", adapter, done, total, scenario) - }, - } - if dump { - opts.Trace = func(sc eval.Scenario, resp eval.Response, score eval.Score) { - fmt.Printf("\r%80s\r", "") - fmt.Printf("\n╭─ %s · %s · fidelity %.0f%%\n", sys.Name(), sc.ID, score.Fidelity()*100) - for _, line := range splitLines(resp.Text) { - fmt.Printf("│ %s\n", line) - } - if resp.Err != nil { - fmt.Printf("│ ERROR: %v\n", resp.Err) - } - fmt.Printf("╰─ missed: %v · leaked: %v · unsignalled: %v\n", - score.Missed, score.Leaked, score.Unsignalled) - } - } - scores, err := eval.Run(sys, suite, opts) - fmt.Printf("\r%80s\r", "") - if err != nil { - fmt.Fprintf(os.Stderr, " %s failed: %v\n", sys.Name(), err) - continue - } - results = append(results, eval.Result{Adapter: sys.Name(), Scores: scores}) - sys.Close() - } - - fmt.Print(eval.Report(results, verbose)) - return nil -} - -// variantSeeds is how many distractor sets each scenario runs with. Two keeps a -// logos-only run of the suite to a few minutes. -const variantSeeds = 2 - -// listBenchScenarios prints the suite without running it — what is being -// measured, and what logos is expected to do on each case. -func listBenchScenarios(only string) error { - suite := eval.Select(eval.Suite(), only) - fmt.Printf("· %s\n\n", eval.Composition(suite)) - for _, s := range suite { - mark := "+" - if s.Known == eval.KnownWeakness { - mark = "−" - } - fmt.Printf("%s %-34s %-12s %-18s %s\n", mark, s.ID, s.Family, clipStr(s.Skill, 18), s.Why) - } - fmt.Println("\n+ expected strength − known weakness") - return nil -} - -func clipStr(s string, n int) string { - if len(s) <= n { - return s - } - return s[:n-1] + "…" -} - -// splitLines keeps the dump readable when a response is one long block. -func splitLines(s string) []string { - if s == "" { - return []string{"(empty)"} - } - return strings.Split(strings.TrimRight(s, "\n"), "\n") -} diff --git a/cmd/logos/help.go b/cmd/logos/help.go index 29a0c1a..4c1cd87 100644 --- a/cmd/logos/help.go +++ b/cmd/logos/help.go @@ -170,10 +170,9 @@ SETUP AND DIAGNOSTICS logos help [all] the three core journeys, or this list BENCHMARKS - logos bench continuity [list] [--only X] [--verbose] [--logos-only] [--variants] - the handoff + memory suite, against every system installed - logos bench memory | bench pipeline - LongMemEval retrieval recall; the extract→recall loop + logos bench memory [--qa] | bench pipeline + LongMemEval retrieval recall / QA accuracy; the extract→recall loop + (the continuity suite now lives in the sibling logos-bench repo) ENV LOGOS_VAULT path to the vault (default ~/logos) diff --git a/cmd/logos/main.go b/cmd/logos/main.go index 0e6de69..77fa883 100644 --- a/cmd/logos/main.go +++ b/cmd/logos/main.go @@ -192,14 +192,12 @@ func main() { case cmd == "mcp" && len(args) >= 1 && args[0] == "serve": serveTools = toolSetFrom(args) err = runMCPServe() + case cmd == "bench" && len(args) >= 2 && args[0] == "memory" && hasFlag(args, "--qa"): + err = runBenchQA(args[1], flagInt(args, "--n", 100), !hasFlag(args, "--vector"), flagInt(args, "--depth", 5)) case cmd == "bench" && len(args) >= 2 && args[0] == "memory": err = runBench(args[1], flagInt(args, "--n", 100), !hasFlag(args, "--vector")) case cmd == "bench" && len(args) >= 1 && args[0] == "pipeline": err = runPipelineBench() - case cmd == "bench" && len(args) >= 2 && args[0] == "continuity" && args[1] == "list": - err = listBenchScenarios(flagStr(args, "--only", "")) - case cmd == "bench" && len(args) >= 1 && args[0] == "continuity": - err = runContinuityBench(args[1:]) case cmd == "graph": hops := flagInt(args, "--hops", 0) if hops == 0 { diff --git a/internal/eval/adapters/baselines.go b/internal/eval/adapters/baselines.go deleted file mode 100644 index b624bd6..0000000 --- a/internal/eval/adapters/baselines.go +++ /dev/null @@ -1,236 +0,0 @@ -package adapters - -import ( - "math" - "sort" - "strings" - - "github.com/Coder8124/logos/internal/eval" - "github.com/Coder8124/logos/internal/provider" -) - -// The baselines exist to keep the headline numbers honest. -// -// Without a floor, a suite cannot tell a good memory system from an easy suite: -// if None scores well on a scenario, that scenario is measuring nothing. Without -// a ceiling, recall looks like an achievement when it is really just a budget -// being ignored — Dump wins recall on almost everything by pasting the entire -// history into the window, which is exactly why Density is reported next to it. -// -// StaticFile is the one that matters commercially. It is what people actually -// do today: a CLAUDE.md or .cursorrules that a human remembers to update. Any -// memory system that cannot beat a hand-maintained text file is not worth -// installing. - -// --------------------------------------------------------------------------- - -// None answers nothing. The floor: whatever it scores, the suite is giving away -// for free. -type None struct{} - -func (None) Name() string { return "none" } -func (None) Reset() error { return nil } -func (None) Write(eval.Event) error { return nil } -func (None) Read(eval.Query) (eval.Response, error) { return eval.Response{}, nil } -func (None) Close() error { return nil } - -// --------------------------------------------------------------------------- - -// Dump concatenates the entire history, newest first, until the budget runs -// out. The "just give the model everything" ceiling — and the reason Density is -// a headline metric rather than a footnote. -type Dump struct{ events []eval.Event } - -func (d *Dump) Name() string { return "full-dump" } -func (d *Dump) Reset() error { d.events = nil; return nil } -func (d *Dump) Write(ev eval.Event) error { - d.events = append(d.events, ev) - return nil -} -func (d *Dump) Close() error { return nil } - -func (d *Dump) Read(q eval.Query) (eval.Response, error) { - // Newest first: if something has to be cut, cut the oldest. This is the most - // favourable ordering for a dump, so the comparison is against the strong - // form of the idea. - ordered := make([]eval.Event, len(d.events)) - copy(ordered, d.events) - sort.SliceStable(ordered, func(i, j int) bool { return ordered[i].TS > ordered[j].TS }) - - var b strings.Builder - for _, ev := range ordered { - line := ev.Flatten() - if q.Budget > 0 && eval.Tokens(b.String())+eval.Tokens(line) > q.Budget { - break - } - b.WriteString(line) - b.WriteString("\n\n") - } - return eval.Response{Text: b.String()}, nil -} - -// --------------------------------------------------------------------------- - -// Recency keeps the newest events that fit. This is what conversation -// compaction approximates: no notion of relevance, only of lateness. -type Recency struct{ events []eval.Event } - -func (r *Recency) Name() string { return "recency-window" } -func (r *Recency) Reset() error { r.events = nil; return nil } -func (r *Recency) Write(ev eval.Event) error { - r.events = append(r.events, ev) - return nil -} -func (r *Recency) Close() error { return nil } - -func (r *Recency) Read(q eval.Query) (eval.Response, error) { - ordered := make([]eval.Event, len(r.events)) - copy(ordered, r.events) - sort.SliceStable(ordered, func(i, j int) bool { return ordered[i].TS > ordered[j].TS }) - - budget := q.Budget - if budget == 0 { - budget = 2000 - } - // Half the ceiling, because compaction leaves room for the conversation - // itself. A window that fills the context with history is not a window. - budget /= 2 - - var kept []string - used := 0 - for _, ev := range ordered { - line := ev.Flatten() - cost := eval.Tokens(line) - if used+cost > budget && len(kept) > 0 { - break - } - kept = append(kept, line) - used += cost - } - return eval.Response{Text: strings.Join(kept, "\n\n")}, nil -} - -// --------------------------------------------------------------------------- - -// StaticFile is the CLAUDE.md baseline: a hand-maintained document that a human -// updates when they remember to. -// -// It receives only documents. That is not a handicap the harness invented — it -// is the defining property of the approach. Nobody appends every working note -// and every dead end to a rules file, which is why the file is always current -// on architecture and always silent on what happened yesterday. -// -// It is Durable: a file survives anything, which is precisely its appeal. -type StaticFile struct{ docs []eval.Event } - -func (s *StaticFile) Name() string { return "static-file" } -func (s *StaticFile) Reset() error { s.docs = nil; return nil } -func (s *StaticFile) Close() error { return nil } -func (s *StaticFile) Write(ev eval.Event) error { - if ev.Kind == eval.KindDoc { - s.docs = append(s.docs, ev) - } - return nil -} -func (s *StaticFile) DropDerived() error { return nil } // a file has nothing derived to lose - -func (s *StaticFile) Read(q eval.Query) (eval.Response, error) { - var b strings.Builder - for _, d := range s.docs { - line := d.Flatten() - if q.Budget > 0 && eval.Tokens(b.String())+eval.Tokens(line) > q.Budget { - break - } - b.WriteString(line) - b.WriteString("\n\n") - } - return eval.Response{Text: b.String()}, nil -} - -// --------------------------------------------------------------------------- - -// VectorRAG is top-k cosine over embedded events, with no lexical arm, no -// graph, and no notion of a checkpoint. It is what most "memory layers" reduce -// to once the marketing is removed, and it is the honest thing to compare -// against — beating None proves nothing; beating this proves something. -type VectorRAG struct { - p *provider.Provider - model string - k int - events []eval.Event - vecs [][]float32 -} - -func NewVectorRAG(p *provider.Provider, model string, k int) *VectorRAG { - if k <= 0 { - k = 8 - } - return &VectorRAG{p: p, model: model, k: k} -} - -func (v *VectorRAG) Name() string { return "vector-rag" } -func (v *VectorRAG) Reset() error { v.events, v.vecs = nil, nil; return nil } -func (v *VectorRAG) Close() error { return nil } - -func (v *VectorRAG) Write(ev eval.Event) error { - text := ev.Flatten() - vecs, err := v.p.Embed(v.model, []string{text}) - if err != nil { - return err - } - v.events = append(v.events, ev) - if len(vecs) == 1 { - v.vecs = append(v.vecs, vecs[0]) - } else { - v.vecs = append(v.vecs, nil) - } - return nil -} - -func (v *VectorRAG) Read(q eval.Query) (eval.Response, error) { - query := q.Task - if q.Project != "" { - query = q.Project + ": " + query - } - qv, err := v.p.Embed(v.model, []string{query}) - if err != nil || len(qv) == 0 { - return eval.Response{}, err - } - - type scored struct { - i int - sim float64 - } - ranked := make([]scored, 0, len(v.events)) - for i, vec := range v.vecs { - ranked = append(ranked, scored{i, cosine(qv[0], vec)}) - } - sort.Slice(ranked, func(a, b int) bool { return ranked[a].sim > ranked[b].sim }) - - var b strings.Builder - for i := 0; i < v.k && i < len(ranked); i++ { - line := v.events[ranked[i].i].Flatten() - if q.Budget > 0 && eval.Tokens(b.String())+eval.Tokens(line) > q.Budget { - break - } - b.WriteString(line) - b.WriteString("\n\n") - } - return eval.Response{Text: b.String()}, nil -} - -func cosine(a, b []float32) float64 { - if len(a) == 0 || len(b) == 0 || len(a) != len(b) { - return 0 - } - var dot, na, nb float64 - for i := range a { - dot += float64(a[i]) * float64(b[i]) - na += float64(a[i]) * float64(a[i]) - nb += float64(b[i]) * float64(b[i]) - } - if na == 0 || nb == 0 { - return 0 - } - return dot / (math.Sqrt(na) * math.Sqrt(nb)) -} diff --git a/internal/eval/adapters/bridge.go b/internal/eval/adapters/bridge.go deleted file mode 100644 index 1007e11..0000000 --- a/internal/eval/adapters/bridge.go +++ /dev/null @@ -1,235 +0,0 @@ -package adapters - -import ( - "bufio" - "encoding/json" - "fmt" - "github.com/Coder8124/logos/internal/eval" - "os" - "os/exec" - "path/filepath" - "sort" - "strings" - "time" -) - -// Bridge runs a memory system that is not written in Go. -// -// The systems worth comparing against — mem0, mempalace — are Python. Rather -// than reimplementing them (which would compare this project against my reading -// of their README) each runs as a subprocess in its own virtualenv, speaking -// newline-delimited JSON on stdin and stdout. The shim on the other side is -// thin on purpose: it translates events into that system's own API and gets out -// of the way, so what is measured is their retrieval, not my wrapper. -// -// Both are pointed at Ollama, so no API keys are needed and nothing leaves the -// machine. That also means the comparison is like-for-like: every system in the -// table is running against the same local models on the same hardware. -type Bridge struct { - label string - path string - - cmd *exec.Cmd - in *bufio.Writer - out *bufio.Scanner -} - -type bridgeReply struct { - OK bool `json:"ok"` - Text string `json:"text"` - Error string `json:"error"` -} - -// BridgeDir is where the Python shims live, relative to the repo root. -const BridgeDir = "bench/adapters" - -// Discover returns a Bridge for every shim that reports itself runnable. -// -// A shim that cannot import its own package is skipped, not failed: the common -// case is that the user has not installed that system, and a missing row is -// honest where a row of zeros would be a lie about the system's quality. -func Discover() []*Bridge { - dir := findBridgeDir() - if dir == "" { - return nil - } - entries, err := os.ReadDir(dir) - if err != nil { - return nil - } - names := make([]string, 0, len(entries)) - for _, e := range entries { - if strings.HasSuffix(e.Name(), "_adapter.py") { - names = append(names, e.Name()) - } - } - sort.Strings(names) - - var out []*Bridge - for _, name := range names { - path := filepath.Join(dir, name) - label := strings.TrimSuffix(name, "_adapter.py") - if reason := probe(path); reason != "" { - fmt.Fprintf(os.Stderr, "· skipping %s: %s\n", label, reason) - continue - } - out = append(out, &Bridge{label: label, path: path}) - } - return out -} - -// probe asks a shim whether it can run, with a short timeout so a system that -// hangs on import cannot hang the benchmark. -func probe(path string) string { - cmd := exec.Command(pythonFor(path), path, "--probe") - done := make(chan error, 1) - var out strings.Builder - cmd.Stdout = &out - cmd.Stderr = &out - if err := cmd.Start(); err != nil { - return err.Error() - } - go func() { done <- cmd.Wait() }() - select { - case err := <-done: - if err != nil { - line := strings.TrimSpace(out.String()) - if i := strings.LastIndex(line, "\n"); i >= 0 { - line = line[i+1:] - } - if line == "" { - line = err.Error() - } - return line - } - return "" - case <-time.After(90 * time.Second): - cmd.Process.Kill() - return "probe timed out" - } -} - -// pythonFor prefers a virtualenv sitting next to the shim, so each system can -// pin its own dependency tree without any of them colliding. -func pythonFor(shim string) string { - venv := filepath.Join(filepath.Dir(shim), ".venv-"+strings.TrimSuffix(filepath.Base(shim), "_adapter.py"), "bin", "python") - if _, err := os.Stat(venv); err == nil { - return venv - } - return "python3" -} - -func findBridgeDir() string { - dir, err := os.Getwd() - if err != nil { - return "" - } - for i := 0; i < 6; i++ { - candidate := filepath.Join(dir, BridgeDir) - if info, err := os.Stat(candidate); err == nil && info.IsDir() { - return candidate - } - parent := filepath.Dir(dir) - if parent == dir { - break - } - dir = parent - } - return "" -} - -func (b *Bridge) Name() string { return b.label } - -func (b *Bridge) start() error { - cmd := exec.Command(pythonFor(b.path), b.path) - stdin, err := cmd.StdinPipe() - if err != nil { - return err - } - stdout, err := cmd.StdoutPipe() - if err != nil { - return err - } - cmd.Stderr = os.Stderr - if err := cmd.Start(); err != nil { - return err - } - b.cmd = cmd - b.in = bufio.NewWriter(stdin) - b.out = bufio.NewScanner(stdout) - // Responses carry whole retrieved contexts, which comfortably exceed the - // scanner's default 64K line limit. - b.out.Buffer(make([]byte, 0, 1<<20), 16<<20) - return nil -} - -func (b *Bridge) call(req map[string]any) (bridgeReply, error) { - if b.cmd == nil { - if err := b.start(); err != nil { - return bridgeReply{}, err - } - } - line, err := json.Marshal(req) - if err != nil { - return bridgeReply{}, err - } - if _, err := b.in.Write(append(line, '\n')); err != nil { - return bridgeReply{}, err - } - if err := b.in.Flush(); err != nil { - return bridgeReply{}, err - } - if !b.out.Scan() { - if err := b.out.Err(); err != nil { - return bridgeReply{}, err - } - return bridgeReply{}, fmt.Errorf("%s exited mid-run", b.label) - } - var reply bridgeReply - if err := json.Unmarshal(b.out.Bytes(), &reply); err != nil { - return bridgeReply{}, fmt.Errorf("%s sent unparseable output: %w", b.label, err) - } - if !reply.OK { - return reply, fmt.Errorf("%s: %s", b.label, reply.Error) - } - return reply, nil -} - -func (b *Bridge) Reset() error { - _, err := b.call(map[string]any{"op": "reset"}) - return err -} - -// Write hands the event over in both forms: the structured fields, for a system -// that can use them, and Flatten's prose, for one that cannot. Nothing is -// withheld from a system because its API is simpler than logos's. -func (b *Bridge) Write(ev eval.Event) error { - _, err := b.call(map[string]any{"op": "write", "event": map[string]any{ - "ts": ev.TS, "actor": ev.Actor, "kind": string(ev.Kind), "project": ev.Project, - "title": ev.Title, "text": ev.Text, "task": ev.Task, - "decisions": ev.Decisions, "failed": ev.Failed, "questions": ev.Questions, - "next": ev.Next, "flat": ev.Flatten(), - }}) - return err -} - -func (b *Bridge) Read(q eval.Query) (eval.Response, error) { - reply, err := b.call(map[string]any{"op": "read", "query": map[string]any{ - "task": q.Task, "project": q.Project, "agent": q.Agent, "budget": q.Budget, "now": q.Now, - }}) - if err != nil { - return eval.Response{}, err - } - return eval.Response{Text: reply.Text}, nil -} - -func (b *Bridge) Close() error { - if b.cmd == nil { - return nil - } - b.call(map[string]any{"op": "close"}) - b.cmd.Process.Kill() - b.cmd.Wait() - b.cmd = nil - return nil -} diff --git a/internal/eval/adapters/logos.go b/internal/eval/adapters/logos.go deleted file mode 100644 index 1587631..0000000 --- a/internal/eval/adapters/logos.go +++ /dev/null @@ -1,273 +0,0 @@ -// Package adapters holds one implementation of eval.Adapter per memory system -// under test. -// -// The adapters are the fair-comparison surface. Each one is written to make its -// system look as good as that system can look: logos gets to use checkpoints -// because it has them, and a store with only add() and search() gets the same -// information flattened into prose rather than withheld. Where a system loses, -// it should lose because of what it is, not because of how it was driven here. -package adapters - -import ( - "fmt" - "os" - "os/exec" - "path/filepath" - "strings" - "time" - - "github.com/Coder8124/logos/internal/contextpack" - "github.com/Coder8124/logos/internal/eval" - "github.com/Coder8124/logos/internal/gitstate" - "github.com/Coder8124/logos/internal/index" - "github.com/Coder8124/logos/internal/memory" - "github.com/Coder8124/logos/internal/provider" - "github.com/Coder8124/logos/internal/secretary" - "github.com/Coder8124/logos/internal/session" - "github.com/Coder8124/logos/internal/vault" -) - -// Logos drives this project through the same entry points the MCP server and -// the CLI use: notes and checkpoints go in, contextpack.Build comes out. No -// benchmark-only retrieval path — if the suite scores well, the shipped product -// scores well. -type Logos struct { - root string // a scratch vault, never the user's - // repo is the scenario's code, created by its first commit event. Kept - // outside the vault, as a project's repository is. - repo string - ix *index.Index - embed *provider.Provider - model string - - dirty bool // vault has unsynced files - paths map[string]int // title collisions get a suffix rather than an overwrite -} - -// NewLogos builds an adapter over a scratch vault. Pass a nil provider to run -// lexical-and-graph only, which is fast and needs no model loaded. -func NewLogos(embed *provider.Provider, model string) (*Logos, error) { - b := &Logos{embed: embed, model: model} - return b, b.Reset() -} - -func (b *Logos) Name() string { - if b.embed == nil { - return "logos (no embed)" - } - return "logos" -} - -func (b *Logos) Reset() error { - if b.ix != nil { - b.ix.Close() - } - if b.root != "" { - os.RemoveAll(b.root) - } - if b.repo != "" { - os.RemoveAll(b.repo) - b.repo = "" - } - root, err := os.MkdirTemp("", "eval-logos-") - if err != nil { - return err - } - b.root = root - b.paths = map[string]int{} - b.dirty = false - return b.open() -} - -func (b *Logos) open() error { - ix, err := index.Open(b.root) - if err != nil { - return err - } - b.ix = ix - for _, init := range []func() error{ - func() error { return memory.Init(ix.DB) }, - func() error { return session.Init(ix.DB) }, - func() error { return secretary.Init(ix.DB) }, - } { - if err := init(); err != nil { - return err - } - } - return nil -} - -func (b *Logos) Close() error { - if b.ix != nil { - b.ix.Close() - } - if b.repo != "" { - os.RemoveAll(b.repo) - } - if b.root != "" { - return os.RemoveAll(b.root) - } - return nil -} - -func (b *Logos) Write(ev eval.Event) error { - switch ev.Kind { - case eval.KindDoc: - return b.writeDoc(ev) - - case eval.KindNote: - _, err := session.AddNoteAt(b.ix.DB, ev.Project, ev.Actor, ev.Text, ev.TS) - return err - - case eval.KindFact, eval.KindMessage: - // A stated fact and a user turn both become memories. Created is set - // from the event so anything reasoning about age has a real clock to - // read. - _, err := memory.Store(b.ix.DB, b.embed, b.model, &memory.Memory{ - Text: ev.Text, Kind: memory.Fact, Project: ev.Project, - Source: "manual", Created: ev.TS, - }) - return err - - case eval.KindCheckpoint: - c := &session.Checkpoint{ - Project: ev.Project, Agent: ev.Actor, Task: ev.Task, - State: ev.Text, Decisions: ev.Decisions, Failed: ev.Failed, - Questions: ev.Questions, Next: ev.Next, TS: ev.TS, - } - // The repository the agent stood in, as checkpoint reads it in use. - // Left to Commit, it would read whatever directory the benchmark runs - // from. - if b.repo != "" { - c.Git = gitstate.Read(b.repo) - } - return session.Commit(b.ix.DB, b.root, c) - - case eval.KindCommit: - return b.commit(ev) - } - return fmt.Errorf("unknown event kind %q", ev.Kind) -} - -// commit changes one file in the scenario's repository, dated when the event -// happened so the history reads in the scenario's order. -func (b *Logos) commit(ev eval.Event) error { - if b.repo == "" { - repo, err := os.MkdirTemp("", "eval-repo-") - if err != nil { - return err - } - b.repo = repo - if err := b.git(ev.TS, "init", "-q"); err != nil { - return err - } - } - path := filepath.Join(b.repo, filepath.FromSlash(ev.Title)) - if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { - return err - } - // The message as the content, so every commit to a file changes it. - if err := os.WriteFile(path, []byte(ev.Text+"\n"), 0o644); err != nil { - return err - } - if err := b.git(ev.TS, "add", "--", ev.Title); err != nil { - return err - } - return b.git(ev.TS, "commit", "-q", "-m", ev.Text) -} - -func (b *Logos) git(ts int64, args ...string) error { - args = append([]string{"-c", "user.name=eval", "-c", "user.email=eval@localhost", "-c", "commit.gpgsign=false"}, args...) - cmd := exec.Command("git", args...) - cmd.Dir = b.repo - date := fmt.Sprintf("@%d +0000", ts) - cmd.Env = append(os.Environ(), "GIT_AUTHOR_DATE="+date, "GIT_COMMITTER_DATE="+date) - if out, err := cmd.CombinedOutput(); err != nil { - return fmt.Errorf("git %s: %v: %s", strings.Join(args, " "), err, out) - } - return nil -} - -// writeDoc lays a document into the vault the way a person would: the project's -// own page under projects/, everything else under topics/, named by its title -// so that [[wikilinks]] in other notes resolve. -func (b *Logos) writeDoc(ev eval.Event) error { - slug := slugify(ev.Title) - dir := "topics" - if ev.Project != "" && slug == slugify(ev.Project) { - dir = "projects" - } - rel := filepath.Join(dir, slug) - if n := b.paths[rel]; n > 0 { - rel = fmt.Sprintf("%s-%d", rel, n+1) - } - b.paths[filepath.Join(dir, slug)]++ - - kind := "topic" - if dir == "projects" { - kind = "project" - } - body := fmt.Sprintf("---\ntype: %s\ntitle: %s\nfirst_seen: %s\n---\n%s\n", - kind, ev.Title, time.Unix(ev.TS, 0).UTC().Format("2006-01-02"), ev.Text) - - b.dirty = true - return vault.WriteAtomic(filepath.Join(b.root, rel+".md"), []byte(body)) -} - -func (b *Logos) Read(q eval.Query) (eval.Response, error) { - if b.dirty { - if _, err := b.ix.Sync(); err != nil { - return eval.Response{}, err - } - b.dirty = false - } - pack, err := contextpack.Build(b.ix, b.embed, b.model, contextpack.Request{ - Task: q.Task, Hint: q.Project, Budget: q.Budget, Now: q.Now, Dir: b.repo, - }) - if err != nil { - return eval.Response{}, err - } - return eval.Response{Text: pack.Render()}, nil -} - -// DropDerived is `rm -rf .logos` followed by `logos index` — the claim that the -// database is a cache, executed. Whatever comes back afterwards is what the -// vault really held. -func (b *Logos) DropDerived() error { - if b.ix != nil { - b.ix.Close() - b.ix = nil - } - if err := os.RemoveAll(filepath.Join(b.root, ".logos")); err != nil { - return err - } - if err := b.open(); err != nil { - return err - } - if _, err := b.ix.Sync(); err != nil { - return err - } - // Both halves, exactly as `logos index` runs them. Rebuilding notes but not - // memories would measure a reindex nobody performs. - _, _, err := b.ix.SyncMemories(b.embed, b.model) - return err -} - -func slugify(s string) string { - s = strings.ToLower(strings.TrimSpace(s)) - var out []rune - dash := false - for _, r := range s { - switch { - case r >= 'a' && r <= 'z', r >= '0' && r <= '9': - out = append(out, r) - dash = false - default: - if !dash && len(out) > 0 { - out = append(out, '-') - dash = true - } - } - } - return strings.Trim(string(out), "-") -} diff --git a/internal/eval/eval.go b/internal/eval/eval.go deleted file mode 100644 index ed81aa5..0000000 --- a/internal/eval/eval.go +++ /dev/null @@ -1,180 +0,0 @@ -// Package eval is a benchmark for agentic memory and handoff. -// -// # Why another benchmark -// -// LongMemEval and its relatives measure one thing well: given a question, can -// the store surface the session holding the answer. That is recall, and recall -// is the easy half. It says nothing about the question this project exists to -// answer — an agent stopped mid-task, a *different* agent arrives, and the user -// re-explains nothing. Nobody scores that, so nobody optimises for it. -// -// So this suite measures continuity: what survives the boundary between two -// agents. The metric that matters most is the one no recall benchmark has a -// slot for — whether the approaches already ruled out make it across. Anyone -// can restate a goal. The expensive knowledge is the three things that didn't -// work, and it is exactly what dies when a session ends. -// -// # The honest half -// -// A benchmark whose author picks the categories is a benchmark its author wins. -// The counterweight is deliberate: this suite includes families logos is known -// to be bad at, several of them found by an agent criticising logos's own -// output during a live handoff — stale context presented without its age, -// circular sourcing where prose restates a claim the data contradicts, and -// abstention, which the LongMemEval harness in memory/bench.go explicitly -// filters out before scoring. Those cases are marked Expect: Fail below. If a -// change makes them pass, that is progress; if a change makes a passing case -// fail, the suite says so. -// -// # What is compared -// -// Every system under test implements Adapter — write events, read back context -// for a task. That common denominator is the point. logos has checkpoint and -// resume as primitives; a store with only add() and search() must flatten a -// checkpoint into prose. The asymmetry is not a handicap imposed by the -// harness, it is the finding. -package eval - -import "strings" - -// Kind classifies an event, because systems with richer primitives are allowed -// to use them. An adapter that has no notion of a checkpoint is expected to -// flatten one into ordinary text — see Event.Flatten. -type Kind string - -const ( - // KindMessage is a conversational turn. - KindMessage Kind = "message" - // KindFact is a durable statement about the user, the sort a memory tool - // stores verbatim. - KindFact Kind = "fact" - // KindDoc is a document that exists in the world — a spec, a spreadsheet - // export, a project page. - KindDoc Kind = "doc" - // KindNote is a line of working progress, written while the work happens. - KindNote Kind = "note" - // KindCheckpoint is where an agent stopped: state, decisions, what failed, - // what is next. - KindCheckpoint Kind = "checkpoint" - // KindCommit is a change to the project's code: Title is the file, Text the - // commit message. A ruling is only as current as the code it was about, - // and without this a scenario had no way to move the code underneath one. - KindCommit Kind = "commit" -) - -// An Event is one thing that happened, in the order it happened. Scenarios are -// built from these, and every adapter receives exactly the same sequence. -type Event struct { - TS int64 // unix seconds; scenarios use real offsets so staleness is testable - Actor string // "user", "claude", "cursor" — who produced this - Kind Kind - Project string - Title string // for KindDoc; the file changed, for KindCommit - Text string - - // Checkpoint fields. Adapters with a checkpoint primitive should map these - // onto it; the rest get them via Flatten. - Task string - Decisions []string - Failed []string // the approaches already ruled out - Questions []string - Next string -} - -// Flatten renders an event as the plain prose a store without structured -// primitives would have to swallow. It is deliberately generous — everything is -// present, nothing is abbreviated — so that a system scoring badly on this -// suite cannot blame the harness for withholding information. -func (e Event) Flatten() string { - if e.Kind == KindCommit { - return "Commit to " + e.Title + ": " + e.Text - } - if e.Kind != KindCheckpoint { - if e.Title != "" { - return e.Title + "\n" + e.Text - } - return e.Text - } - - var b strings.Builder - b.WriteString("Checkpoint by " + e.Actor) - if e.Project != "" { - b.WriteString(" on " + e.Project) - } - b.WriteString(".\n") - section := func(label, body string) { - if strings.TrimSpace(body) != "" { - b.WriteString(label + ": " + body + "\n") - } - } - list := func(label string, items []string) { - if len(items) > 0 { - b.WriteString(label + ": " + strings.Join(items, "; ") + "\n") - } - } - section("Task", e.Task) - section("State", e.Text) - list("Decisions", e.Decisions) - list("Already tried and it did not work", e.Failed) - list("Open questions", e.Questions) - section("Next", e.Next) - return b.String() -} - -// A Query is what the arriving agent asks for. Budget is a token ceiling: a -// system may return less, but returning more is measured and counted against -// it, because context that overflows the window is context nobody reads. -type Query struct { - Task string - Project string - Agent string // who is asking now — for handoff cases, not whoever wrote the events - Budget int - Now int64 // the clock at read time, so staleness has a reference point -} - -// A Response is what the system would put into the arriving agent's context -// window. Text is scored; Tokens is measured by the harness, not by the -// adapter, so nobody can under-report their own cost. -type Response struct { - Text string - // Err is set when the system declined or failed to answer. A failed read is - // scored as an empty response rather than aborting the run — a memory layer - // that errors on an unfamiliar question is one an agent learns to avoid, and - // that should show up in the numbers rather than in a stack trace. - Err error -} - -// An Adapter is a memory system under test. Reset must leave it as empty as a -// fresh install: the suite runs every scenario against a clean store, so a -// system cannot score by accumulating context across cases. -type Adapter interface { - Name() string - Reset() error - Write(Event) error - Read(Query) (Response, error) - Close() error -} - -// Durable is implemented by systems that distinguish a source of truth from a -// rebuildable cache. DropDerived deletes everything the system considers -// disposable; whatever survives is what the system really owns. -// -// Not implementing this is itself an answer, and the suite reports it as one: -// a store whose only copy lives in its own index has no source of truth, and -// scores zero on the durability family rather than being excused from it. -type Durable interface { - DropDerived() error -} - -// Tokens estimates cost at four characters per token. -// -// A real tokenizer would be a model-specific dependency answering a question -// that only needs to be roughly right, and — more to the point — every adapter -// is measured with this same function. Consistency across systems matters more -// here than absolute accuracy against any one vocabulary. -func Tokens(s string) int { - if s == "" { - return 0 - } - return len(s)/4 + 1 -} diff --git a/internal/eval/eval_test.go b/internal/eval/eval_test.go deleted file mode 100644 index 14856eb..0000000 --- a/internal/eval/eval_test.go +++ /dev/null @@ -1,217 +0,0 @@ -package eval - -import ( - "strings" - "testing" -) - -// The harness scores other people's systems, so its own failure modes are -// expensive: a matcher that is too generous inflates everyone equally and the -// benchmark stops distinguishing anything. These tests are mostly about the -// ways matching can be wrong in a system's favour. - -// The bug that made this test necessary: logos's render opens with "Context -// for: ", so a gold label whose wording overlapped the question was -// satisfied by the echo. temporal-ordering scored 100% while returning two -// undated facts in arbitrary order. -func TestTheQuestionCannotAnswerItself(t *testing.T) { - q := Query{Task: "did we lock the industrial design before or after signing with Pegatron?"} - echoed := "# Context for: did we lock the industrial design before or after signing with Pegatron?\n" + - "- Signed the Pegatron agreement.\n- Locked the industrial design.\n" - - f := Fact{Label: "ordering", Any: []string{"after signing", "signed first"}} - if f.In(stripEcho(echoed, q)) { - t.Error("a label matched text the system copied from the question") - } - - // The same words in real content still count — only the verbatim query goes. - real := echoed + "The design was locked after signing with the manufacturer.\n" - if !f.In(stripEcho(real, q)) { - t.Error("stripping the echo also removed a genuine answer") - } -} - -// Abstention cases have no facts to fetch: the right answer is an admission of -// ignorance. Scoring them as a total recall miss marked every system down for -// declining to invent something. -func TestRequiringNothingIsNotAMiss(t *testing.T) { - sc := Scenario{ - ID: "t", Query: Query{Task: "what is the warranty period?"}, - Gold: Gold{Signal: []Fact{{Label: "admits ignorance", Any: []string{"no record"}}}}, - } - - honest := grade(sc, "x", Response{Text: "No record of a warranty period."}) - if honest.Recall() != 1 { - t.Errorf("nothing required, nothing missed: want recall 1, got %v", honest.Recall()) - } - if !honest.Pass() { - t.Error("admitting ignorance is the correct answer here and must pass") - } - - silent := grade(sc, "x", Response{Text: "The Suzhou plant runs two shifts."}) - if silent.Pass() { - t.Error("changing the subject is not an admission of ignorance") - } -} - -// Fidelity is multiplicative: carrying a superseded price next to the right one -// is not half credit, because the agent will act on the stale number. -func TestALeakedFactCancelsTheCorrectOne(t *testing.T) { - sc := Scenario{ - ID: "t", Query: Query{Task: "price?"}, - Gold: Gold{ - Carry: []Fact{{Label: "current", Any: []string{"$249"}}}, - Avoid: []Fact{{Label: "superseded", Any: []string{"$199"}}}, - }, - } - clean := grade(sc, "x", Response{Text: "Retail is $249."}) - leaky := grade(sc, "x", Response{Text: "Retail is $249. Earlier we said $199."}) - - if clean.Fidelity() != 1 { - t.Errorf("want 1, got %v", clean.Fidelity()) - } - if leaky.Fidelity() != 0 { - t.Errorf("a leaked superseded value must not score: got %v", leaky.Fidelity()) - } - if leaky.Recall() != 1 { - t.Error("recall should still record that the right fact was present") - } -} - -// A scenario passes only when it meets every bar it set. Fidelity alone let the -// staleness case score 100% while doing exactly what it was written to catch. -func TestSignalCountsTowardPassing(t *testing.T) { - sc := Scenario{ - ID: "t", Query: Query{Task: "plan the next spin"}, - Gold: Gold{ - Carry: []Fact{{Label: "the note", Any: []string{"tooling freeze"}}}, - Signal: []Fact{{Label: "its age", Any: []string{"days ago"}}}, - }, - } - undated := grade(sc, "x", Response{Text: "Sixteen days before the tooling freeze."}) - if undated.Fidelity() != 1 { - t.Errorf("fidelity should be clean here: got %v", undated.Fidelity()) - } - if undated.Pass() { - t.Error("the case asked for the age of the note and did not get it") - } - - dated := grade(sc, "x", Response{Text: "Sixteen days before the tooling freeze. (13 days ago)"}) - if !dated.Pass() { - t.Error("both bars were met") - } -} - -// All means all; Any means at least one. Compound facts exist because some -// claims are only right together. -func TestFactMatching(t *testing.T) { - both := Fact{All: []string{"claude", "cursor"}} - if both.In("only claude was here") { - t.Error("All must require every term") - } - if !both.In("claude found it, cursor confirmed it") { - t.Error("All should match when every term is present") - } - - either := Fact{Any: []string{"71 percent", "71%"}} - if !either.In("yield runs at 71% first-pass") { - t.Error("Any should accept a surface variant the author listed") - } - if either.In("yield runs at seventy-one percent") { - t.Error("Any must not match a variant the author did not list — that is the author's job") - } -} - -// Whitespace and case are noise; a system should not be punished for wrapping. -func TestMatchingIgnoresLayout(t *testing.T) { - f := Fact{Any: []string{"no movement under 10k units"}} - if !f.In("Re-quoting the waveguide —\n No Movement Under\n 10k Units.") { - t.Error("matching should survive wrapping and capitalisation") - } -} - -// Leakage and signal are averaged only over the cases that test them, so a -// suite cannot dilute either metric by adding unrelated scenarios. -func TestUntestedMetricsDoNotDilute(t *testing.T) { - scores := []Score{ - {CarryHit: 1, CarryTotal: 1, LeakHit: 1, LeakTotal: 1}, // leaks everything - {CarryHit: 1, CarryTotal: 1}, // tests no leakage at all - } - got := Roll(scores, func(Score) string { return "all" })[0] - if got.Leakage != 1 { - t.Errorf("want leakage 1 over the single case that tests it, got %v", got.Leakage) - } -} - -// silent answers nothing. Defined here rather than imported from the adapters -// package, which depends on this one. -type silent struct{} - -func (silent) Name() string { return "silent" } -func (silent) Reset() error { return nil } -func (silent) Write(Event) error { return nil } -func (silent) Read(Query) (Response, error) { return Response{}, nil } -func (silent) Close() error { return nil } - -// The floor has to be a floor. If doing nothing scores, the suite is measuring -// nothing — this is the guard that keeps scenario authors honest. -func TestAnEmptyAnswerScoresNothing(t *testing.T) { - suite := Suite() - scores, err := Run(silent{}, suite, Options{}) - if err != nil { - t.Fatal(err) - } - var passed []string - for _, s := range scores { - if s.Pass() { - passed = append(passed, s.Scenario) - } - } - if len(passed) > 0 { - t.Errorf("returning nothing passed %d scenarios: %s", len(passed), strings.Join(passed, ", ")) - } -} - -// Every scenario must be able to fail and must say what it is for, or it is -// decoration rather than measurement. -func TestSuiteIsWellFormed(t *testing.T) { - seen := map[string]bool{} - for _, s := range Suite() { - if seen[s.ID] { - t.Errorf("duplicate scenario id %q", s.ID) - } - seen[s.ID] = true - - if s.Why == "" || s.Family == "" || s.Skill == "" { - t.Errorf("%s: needs a family, a skill, and a line saying what it tests", s.ID) - } - if len(s.Gold.Carry)+len(s.Gold.Avoid)+len(s.Gold.Signal) == 0 { - t.Errorf("%s: no gold labels, so nothing can fail", s.ID) - } - if len(s.Setup) == 0 { - t.Errorf("%s: no history to retrieve from", s.ID) - } - if s.Query.Task == "" { - t.Errorf("%s: no task", s.ID) - } - } -} - -// A checkpoint has to survive being flattened, because that is the only form -// systems without a checkpoint primitive can receive. If Flatten dropped the -// ruled-out approaches, every comparison in the suite would be rigged. -func TestFlatteningAChveckpointKeepsWhatFailed(t *testing.T) { - got := Event{ - Kind: KindCheckpoint, Actor: "claude", Project: "kestrel-one", - Task: "cut the BOM", - Failed: []string{"re-quoting the waveguide", "dropping the second mic"}, - Next: "quote the display driver", - }.Flatten() - - for _, want := range []string{"cut the BOM", "re-quoting the waveguide", - "dropping the second mic", "quote the display driver", "claude"} { - if !strings.Contains(got, want) { - t.Errorf("flattening lost %q:\n%s", want, got) - } - } -} diff --git a/internal/eval/report.go b/internal/eval/report.go deleted file mode 100644 index 1c8f89a..0000000 --- a/internal/eval/report.go +++ /dev/null @@ -1,245 +0,0 @@ -package eval - -import ( - "fmt" - "sort" - "strings" - - "github.com/Coder8124/logos/internal/text" -) - -// A Result is one adapter's full run. -type Result struct { - Adapter string - Scores []Score -} - -// Report renders the comparison. -// -// Fidelity leads because it is the number that cannot be gamed by returning -// more: it rises only when the required facts arrive and the wrong ones stay -// out. Density sits beside it for the opposite reason — a system can buy recall -// with tokens, and the column shows the price. -func Report(results []Result, verbose bool) string { - var b strings.Builder - - // With variants in the run, every table down to the predictions is still - // the suite as written, so its numbers stay comparable with runs that had - // none; the variants get their own section after it. - all, varied := results, hasVariants(results) - if varied { - results = asWritten(results) - b.WriteString("\n(the tables down to the predictions are the scenarios as written; variants follow)\n") - } - - b.WriteString("\n── overall ───────────────────────────────────────────────────────────────\n\n") - b.WriteString(fmt.Sprintf("%-18s %7s %8s %8s %8s %8s %8s %8s\n", - "system", "pass", "fidelity", "recall", "leak", "signal", "tokens", "dens/1k")) - for _, r := range results { - a := Roll(r.Scores, func(Score) string { return "all" }) - if len(a) == 0 { - continue - } - o := a[0] - b.WriteString(fmt.Sprintf("%-18s %6.1f%% %7.1f%% %7.1f%% %7.1f%% %7.1f%% %8d %8.1f\n", - r.Adapter, o.PassRate*100, o.Fidelity*100, o.Recall*100, o.Leakage*100, o.Signal*100, o.Tokens, o.Density)) - } - b.WriteString("\npass = met every bar the scenario set, including its signal labels.\n") - b.WriteString("fidelity = recall × (1 − leakage). leak and signal are averaged only over\n") - b.WriteString("the cases that test them. tokens is the mean response size; dens/1k is\n") - b.WriteString("required facts carried per 1000 tokens — the cost of that recall.\n") - - // ---- by family ------------------------------------------------------- - families := distinct(results, func(s Score) string { return s.Family }) - b.WriteString("\n── pass rate by family ───────────────────────────────────────────────────\n\n") - b.WriteString(fmt.Sprintf("%-18s", "system")) - for _, f := range families { - b.WriteString(fmt.Sprintf(" %14s", f)) - } - b.WriteString("\n") - for _, r := range results { - byFam := map[string]Aggregate{} - for _, a := range Roll(r.Scores, func(s Score) string { return s.Family }) { - byFam[a.Group] = a - } - b.WriteString(fmt.Sprintf("%-18s", r.Adapter)) - for _, f := range families { - if a, ok := byFam[f]; ok { - b.WriteString(fmt.Sprintf(" %13.0f%%", a.PassRate*100)) - } else { - b.WriteString(fmt.Sprintf(" %14s", "—")) - } - } - b.WriteString("\n") - } - - // ---- by skill -------------------------------------------------------- - b.WriteString("\n── pass rate by skill ────────────────────────────────────────────────────\n\n") - skills := distinct(results, func(s Score) string { return s.Skill }) - b.WriteString(fmt.Sprintf("%-22s", "skill")) - for _, r := range results { - b.WriteString(fmt.Sprintf(" %11s", clip(r.Adapter, 11))) - } - b.WriteString("\n") - for _, sk := range skills { - b.WriteString(fmt.Sprintf("%-22s", clip(sk, 22))) - for _, r := range results { - var hit, n float64 - for _, s := range r.Scores { - if s.Skill == sk { - if s.Pass() { - hit++ - } - n++ - } - } - if n == 0 { - b.WriteString(fmt.Sprintf(" %11s", "—")) - continue - } - b.WriteString(fmt.Sprintf(" %10.0f%%", hit/n*100)) - } - b.WriteString("\n") - } - - // ---- the predictions we got wrong ------------------------------------ - b.WriteString(surprises(results)) - - if varied { - b.WriteString(stability(all)) - } - if verbose { - b.WriteString(detail(all)) - } - return b.String() -} - -// surprises compares each scenario's recorded expectation against what the -// first-listed system actually did. -// -// This is the part of the report that keeps the suite honest over time. A case -// labelled a weakness that starts passing is progress; a case labelled a -// strength that starts failing is a regression the averages would otherwise -// absorb. Either way the label is now wrong and somebody has to look. -func surprises(results []Result) string { - if len(results) == 0 { - return "" - } - subject := results[0] - - var better, worse []string - for _, s := range subject.Scores { - switch { - case s.Known == KnownWeakness && s.Pass(): - better = append(better, fmt.Sprintf(" %-34s marked a weakness, %s", s.Scenario, why(s))) - case s.Known == KnownStrength && !s.Pass(): - worse = append(worse, fmt.Sprintf(" %-34s marked a strength, %s", s.Scenario, why(s))) - } - } - if len(better) == 0 && len(worse) == 0 { - return "\n── predictions ───────────────────────────────────────────────────────────\n\n" + - fmt.Sprintf(" every case in the suite behaved as %s's labels predicted.\n", subject.Adapter) - } - - var b strings.Builder - b.WriteString("\n── predictions that were wrong ───────────────────────────────────────────\n\n") - if len(worse) > 0 { - b.WriteString("expected to pass, did not:\n") - b.WriteString(strings.Join(worse, "\n") + "\n") - } - if len(better) > 0 { - if len(worse) > 0 { - b.WriteString("\n") - } - b.WriteString("expected to fail, passed:\n") - b.WriteString(strings.Join(better, "\n") + "\n") - } - return b.String() -} - -// why says which bar a scenario cleared or missed, so the line is actionable -// without reopening the suite. -func why(s Score) string { - parts := []string{fmt.Sprintf("fidelity %.0f%%", s.Fidelity()*100)} - if s.SignalTotal > 0 { - parts = append(parts, fmt.Sprintf("signal %.0f%%", s.Signal()*100)) - } - if s.Err != nil { - parts = append(parts, "error") - } - return strings.Join(parts, ", ") -} - -func detail(results []Result) string { - var b strings.Builder - b.WriteString("\n── per scenario ──────────────────────────────────────────────────────────\n") - for _, r := range results { - b.WriteString("\n" + r.Adapter + "\n") - for _, s := range r.Scores { - flag := " " - if s.Over() { - flag = "!" - } - b.WriteString(fmt.Sprintf(" %s %-34s fid %3.0f%% %d/%d carried %5d tok\n", - flag, s.Scenario, s.Fidelity()*100, s.CarryHit, s.CarryTotal, s.Tokens)) - if s.Axis != "" { - b.WriteString(" " + s.Axis + ": " + s.Variant + "\n") - } - if len(s.Missed) > 0 { - b.WriteString(" missed: " + strings.Join(s.Missed, "; ") + "\n") - } - if len(s.Leaked) > 0 { - b.WriteString(" leaked: " + strings.Join(s.Leaked, "; ") + "\n") - } - if len(s.Unsignalled) > 0 { - b.WriteString(" no signal: " + strings.Join(s.Unsignalled, "; ") + "\n") - } - if s.Err != nil { - b.WriteString(" error: " + s.Err.Error() + "\n") - } - } - } - return b.String() -} - -func distinct(results []Result, key func(Score) string) []string { - seen := map[string]bool{} - var out []string - for _, r := range results { - for _, s := range r.Scores { - if k := key(s); !seen[k] { - seen[k] = true - out = append(out, k) - } - } - } - sort.Strings(out) - return out -} - -func clip(s string, n int) string { - return text.Ellipsize(s, n) -} - -// Composition summarises what the suite contains, so a reader can see the -// balance of strengths to weaknesses before reading any score. -func Composition(suite []Scenario) string { - byFamily := map[string]int{} - byKnown := map[Known]int{} - for _, s := range suite { - byFamily[s.Family]++ - byKnown[s.Known]++ - } - fams := make([]string, 0, len(byFamily)) - for f := range byFamily { - fams = append(fams, f) - } - sort.Strings(fams) - - var parts []string - for _, f := range fams { - parts = append(parts, fmt.Sprintf("%s %d", f, byFamily[f])) - } - return fmt.Sprintf("%d scenarios (%s) · %d expected strengths, %d known weaknesses", - len(suite), strings.Join(parts, ", "), byKnown[KnownStrength], byKnown[KnownWeakness]) -} diff --git a/internal/eval/report_text_test.go b/internal/eval/report_text_test.go deleted file mode 100644 index 1f49695..0000000 --- a/internal/eval/report_text_test.go +++ /dev/null @@ -1,17 +0,0 @@ -package eval - -import ( - "strings" - "testing" - "unicode/utf8" -) - -// clip byte-sliced at a fixed offset (s[:n-1]), which cuts a multibyte rune in -// half when a scenario name or score note contains emoji or CJK text. -func TestClipNeverSplitsAMultibyteRune(t *testing.T) { - in := strings.Repeat("🌍", 50) - got := clip(in, 30) - if !utf8.ValidString(got) { - t.Fatalf("clip produced invalid UTF-8: %q", got) - } -} diff --git a/internal/eval/run.go b/internal/eval/run.go deleted file mode 100644 index 18cdf29..0000000 --- a/internal/eval/run.go +++ /dev/null @@ -1,109 +0,0 @@ -package eval - -import ( - "fmt" - "strings" -) - -// Options controls one benchmark run. -type Options struct { - // Only runs the scenarios whose id, family or skill contains this string. - Only string - // Progress is called before each scenario. - Progress func(adapter, scenario string, done, total int) - // Trace, if set, receives the raw response alongside its score. The suite is - // only as good as its gold labels, and the only way to tell a real pass from - // a substring that happened to appear is to read what the system actually - // returned. - Trace func(sc Scenario, resp Response, score Score) -} - -// Select filters a suite by the Only string. -func Select(suite []Scenario, only string) []Scenario { - if strings.TrimSpace(only) == "" { - return suite - } - only = strings.ToLower(only) - var out []Scenario - for _, s := range suite { - if strings.Contains(strings.ToLower(s.ID), only) || - strings.Contains(strings.ToLower(s.Family), only) || - strings.Contains(strings.ToLower(s.Skill), only) { - out = append(out, s) - } - } - return out -} - -// Run puts one adapter through a suite and returns a score per scenario. -// -// Every scenario starts from Reset, so no system can score by carrying state -// between cases — a store that answered the last question well should get no -// help on the next one. Setup events are delivered in timestamp order, exactly -// as written, to every adapter alike. -func Run(ad Adapter, suite []Scenario, opt Options) ([]Score, error) { - scores := make([]Score, 0, len(suite)) - - for i, sc := range suite { - if opt.Progress != nil { - opt.Progress(ad.Name(), sc.ID, i+1, len(suite)) - } - if err := ad.Reset(); err != nil { - return nil, fmt.Errorf("%s: reset before %s: %w", ad.Name(), sc.ID, err) - } - - // A write failure is the adapter's problem, not the harness's: record it - // as a failed case and keep going, so one brittle system cannot abort a - // comparison run that the others would have completed. - var writeErr error - for _, ev := range sc.Setup { - if err := ad.Write(ev); err != nil { - writeErr = fmt.Errorf("write %s: %w", ev.Kind, err) - break - } - } - if writeErr != nil { - scores = append(scores, grade(sc, ad.Name(), Response{Err: writeErr})) - continue - } - - resp := readFor(ad, sc) - score := grade(sc, ad.Name(), resp) - if opt.Trace != nil { - opt.Trace(sc, resp, score) - } - scores = append(scores, score) - } - return scores, nil -} - -// readFor performs the scenario's read, including the durability drop. -// -// A system that does not implement Durable has no source of truth outside its -// own index, so wiping the derived state wipes everything. The harness records -// that as an empty response rather than skipping the case. This is the one -// place the suite asserts an outcome instead of measuring it, and it is stated -// plainly in the report for that reason — the alternative, excusing those -// systems from the family, would hide the very property being tested. -func readFor(ad Adapter, sc Scenario) Response { - if sc.DropDerived { - d, ok := ad.(Durable) - if !ok { - return Response{Err: errNoSourceOfTruth} - } - if err := d.DropDerived(); err != nil { - return Response{Err: fmt.Errorf("dropping derived state: %w", err)} - } - } - resp, err := ad.Read(sc.Query) - if err != nil { - resp.Err = err - } - return resp -} - -var errNoSourceOfTruth = fmt.Errorf("no source of truth outside the cache: dropping derived state loses everything") - -// ErrNoSourceOfTruth reports whether a score failed because the system had -// nothing left after its cache was dropped. -func ErrNoSourceOfTruth(err error) bool { return err == errNoSourceOfTruth } diff --git a/internal/eval/scenarios.go b/internal/eval/scenarios.go deleted file mode 100644 index 5b26a16..0000000 --- a/internal/eval/scenarios.go +++ /dev/null @@ -1,802 +0,0 @@ -package eval - -import ( - "fmt" - "time" -) - -// The suite. -// -// Every case is hand-written, and every case is trying to break something -// specific. Generated scenarios were considered and rejected: a model asked for -// three hundred handoffs produces three hundred variations on the same easy -// one, and gold labels from a model are labels you cannot audit. -// -// The world is shared across cases on purpose — a smart-glasses company with a -// bill of materials that will not close, a factory missing yield, and a -// schedule with a critical path. Reusing one setting means a system cannot do -// well by pattern-matching a case's vocabulary, and it lets the harder cases -// depend on facts established elsewhere in the same fiction. -// -// Roughly a third of the suite is marked KnownWeakness. Those are not -// aspirational; they are cases logos fails today, several found by an agent -// picking apart logos's own handoff output during a live test. A benchmark -// whose author chooses the categories is a benchmark its author wins, and the -// only defence is to write the losses down first. - -// The clock is anchored to now, and the offsets are what is fixed. -// -// It was a constant at first, which was wrong in a way that hid a whole family -// of failures: pinned to 2026-01-01 and run in August, a note written "thirteen -// days ago" is really eight months old, and any system reporting the age of its -// own context reports the true one. The staleness cases were unwinnable and the -// suite was measuring the wrong thing. What has to stay stable between runs is -// the *interval* — thirteen days before the question — not the wall date. -var benchNow = time.Now().Unix() - -const day = 86400 - -func ago(days int) int64 { return benchNow - int64(days)*day } - -// --------------------------------------------------------------------------- -// Event constructors, kept terse so the scenarios below read as prose. -// --------------------------------------------------------------------------- - -func doc(days int, project, title, text string) Event { - return Event{TS: ago(days), Actor: "user", Kind: KindDoc, Project: project, Title: title, Text: text} -} - -func note(days int, actor, project, text string) Event { - return Event{TS: ago(days), Actor: actor, Kind: KindNote, Project: project, Text: text} -} - -func said(days int, text string) Event { - return Event{TS: ago(days), Actor: "user", Kind: KindFact, Text: text} -} - -func commit(days int, project, path, message string) Event { - return Event{TS: ago(days), Actor: "user", Kind: KindCommit, Project: project, Title: path, Text: message} -} - -func msg(days int, actor, text string) Event { - return Event{TS: ago(days), Actor: actor, Kind: KindMessage, Text: text} -} - -// Suite returns the whole benchmark. -func Suite() []Scenario { - var s []Scenario - s = append(s, continuity()...) - s = append(s, memoryCases()...) - s = append(s, durability()...) - return s -} - -// --------------------------------------------------------------------------- -// Continuity — what survives the boundary between two agents. -// --------------------------------------------------------------------------- - -func continuity() []Scenario { - return []Scenario{ - { - ID: "handoff-failed-approaches", Family: "continuity", Skill: "negative-knowledge", - Why: "The three things already ruled out are the expensive knowledge and the first thing lost.", - Known: KnownStrength, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses, $249 retail. Target BOM $118, actual $141.20."), - note(6, "claude", "kestrel-one", "Re-quoted the waveguide with Lumus — no movement under 10k units."), - note(6, "claude", "kestrel-one", "Tried dropping the second microphone; audio team vetoed, beamforming needs two."), - { - TS: ago(5), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "cut the BOM from $141.20 to the $118 target", - Text: "Down to $137.40. The remaining gap is concentrated in the optics stack.", - Decisions: []string{"Keep the dual-mic array; audio quality is a reviewable feature"}, - Failed: []string{ - "Re-quoting the waveguide with Lumus — no movement under 10k units", - "Dropping the second microphone — vetoed by audio, beamforming needs two", - "Switching to a plastic frame — fails the drop test at 1.2m", - }, - Questions: []string{"Can we move the display driver to the cheaper Himax part without a recert?"}, - Next: "Quote the single-source display driver alternatives", - }, - }, - Query: Query{Task: "keep cutting the BOM toward target", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"where are we on getting the bill of materials down?", "resume the cost-down work", "what's left to try on unit cost?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "waveguide re-quote failed", Any: []string{"lumus", "waveguide"}}, - {Label: "mic removal vetoed", Any: []string{"second microphone", "beamforming", "dual-mic"}}, - {Label: "plastic frame failed drop test", Any: []string{"plastic frame", "drop test", "1.2m"}}, - {Label: "current BOM figure", Any: []string{"137.40", "$137"}}, - {Label: "next step", Any: []string{"display driver"}}, - }, - }, - }, - { - ID: "handoff-uncommitted-notes", Family: "continuity", Skill: "died-mid-task", - Why: "An agent killed before it could check in leaves only working notes; those are the whole record.", - Known: KnownStrength, - Setup: []Event{ - doc(30, "ota-firmware", "OTA firmware", "Over-the-air update path for Kestrel One. Blocked: no spare flash budgeted for an A/B partition scheme."), - note(2, "claude", "ota-firmware", "Certification objection is wrong — a firmware-only OTA does not re-trigger the RF filing, confirmed against the FCC scope."), - note(2, "claude", "ota-firmware", "RMA math: at 2.1% return rate and $41 handling, one field-fixable bug pays for the whole OTA effort."), - note(2, "claude", "ota-firmware", "Vault has NO flash size number for the SoC — only that it was costed at 'the smaller part' on the $27.80 SoC+memory line. Cannot size an A/B scheme without the part number."), - }, - Query: Query{Task: "pick up the OTA firmware work", Project: "ota-firmware", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"where did the over-the-air update work get to?", "resume OTA", "what was the last agent doing on remote updates?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "certification objection resolved", Any: []string{"does not re-trigger", "rf filing", "fcc"}}, - {Label: "RMA math", Any: []string{"2.1%", "$41", "return rate"}}, - {Label: "the open blocker, in full", Any: []string{"part number", "cannot size"}}, - {Label: "the $27.80 line it was costed on", Any: []string{"27.80"}}, - }, - Signal: []Fact{ - {Label: "marked as never checkpointed", Any: []string{"not yet checkpointed", "uncommitted", "working note"}}, - }, - }, - }, - { - ID: "handoff-attribution", Family: "continuity", Skill: "attribution", - Why: "When two agents contributed, which of them found a thing is part of the finding.", - Known: KnownStrength, - Setup: []Event{ - doc(30, "yield", "Bonding yield", "Display bonding runs at 71 percent first-pass."), - note(4, "claude", "yield", "Traced the yield loss to the ACF bonding temperature ramp, not the alignment stage."), - note(3, "cursor", "yield", "Vendor confirmed the ramp is fixed in firmware and cannot be tuned on our units."), - }, - Query: Query{Task: "continue the yield investigation", Project: "yield", Agent: "codex", Budget: 4000, Now: benchNow}, - Wordings: []string{"what have we learned about the yield losses so far?", "pick up the yield problem", "where did the others leave the scrap investigation?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the ACF ramp finding", Any: []string{"acf", "temperature ramp"}}, - {Label: "the vendor answer", Any: []string{"fixed in firmware", "cannot be tuned"}}, - }, - Signal: []Fact{ - {Label: "both agents named", All: []string{"claude", "cursor"}}, - }, - }, - }, - { - ID: "handoff-cold-start", Family: "continuity", Skill: "cold-start", - Why: "The arriving agent knows only the project name. Everything else has to come from the store.", - Known: KnownStrength, - Setup: []Event{ - doc(40, "kestrel-one", "Kestrel One", "Smart glasses, $249 retail, ship date November 12."), - { - TS: ago(3), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "decide whether to hold the November ship date", - Text: "Tooling freeze is the binding constraint, not the BOM.", - Decisions: []string{"Hold the date; slip the second colourway to a spring refresh"}, - Failed: []string{"Compressing the certification window — the lab has no slot before October"}, - Next: "Get Tomas to confirm the tooling freeze date in writing", - }, - }, - Query: Query{Task: "continue", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"where did we leave off?", "resume work", "what was I doing?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the actual task", Any: []string{"november", "ship date"}}, - {Label: "the decision", Any: []string{"hold the date", "colourway", "spring refresh"}}, - {Label: "the ruled-out approach", Any: []string{"certification window", "no slot before october"}}, - {Label: "the next action", Any: []string{"tomas", "tooling freeze"}}, - }, - }, - }, - { - ID: "handoff-scope-isolation", Family: "continuity", Skill: "scope", - Why: "Two projects in one store. Resuming one must not import the other's ruled-out approaches.", - Known: KnownStrength, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses hardware programme."), - doc(30, "website", "Website rebuild", "Marketing site rebuild ahead of launch."), - { - TS: ago(4), Actor: "claude", Kind: KindCheckpoint, Project: "website", - Task: "pick a CMS", - Failed: []string{"Contentful — pricing tier jumps at exactly our seat count"}, - Next: "Trial Sanity for a week", - }, - { - TS: ago(3), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "close the BOM gap", - Failed: []string{"Re-quoting the waveguide — no movement under 10k units"}, - Next: "Quote the display driver alternatives", - }, - }, - Query: Query{Task: "keep working the BOM", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"back to cost reduction on the glasses", "what's next on the bill of materials?", "resume the parts-cost work"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "this project's failed approach", Any: []string{"waveguide"}}, - {Label: "this project's next step", Any: []string{"display driver"}}, - }, - Avoid: []Fact{ - {Label: "the other project's CMS decision", Any: []string{"contentful", "sanity"}}, - }, - }, - }, - { - ID: "handoff-budget-discipline", Family: "continuity", Skill: "budget", - Why: "At a tight ceiling, the ruled-out approaches must outrank background prose.", - Known: KnownStrength, - Setup: append( - noise(20, "kestrel-one", "Routine standup note"), - doc(30, "kestrel-one", "Kestrel One", "Smart glasses, $249 retail."), - Event{ - TS: ago(2), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "close the BOM gap", - Failed: []string{"Re-quoting the waveguide — no movement under 10k units"}, - Next: "Quote the display driver alternatives", - }, - ), - Query: Query{Task: "continue the BOM work", Project: "kestrel-one", Agent: "cursor", Budget: 700, Now: benchNow}, - Wordings: []string{"carry on with the bill of materials", "where's the BOM at?", "pick up the cost-down"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "failed approach survives the squeeze", Any: []string{"waveguide"}}, - {Label: "next step survives the squeeze", Any: []string{"display driver"}}, - }, - }, - }, - { - ID: "handoff-noise-resistance", Family: "continuity", Skill: "distractors", - Why: "Forty routine notes around one checkpoint. Volume must not bury the handoff.", - Known: KnownStrength, - Setup: append( - noise(40, "kestrel-one", "Standup: no blockers, continuing as planned"), - Event{ - TS: ago(1), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "resolve the thermal throttling on the left temple", - Failed: []string{"Adding a copper spreader — no room once the battery moved forward"}, - Next: "Test the 400MHz cap against the review benchmark suite", - }, - ), - Query: Query{Task: "continue the thermal work", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"what's the state of the overheating issue?", "resume the left-temple heat problem", "pick up where the throttling investigation stopped"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the failed approach", Any: []string{"copper spreader"}}, - {Label: "the next step", Any: []string{"400mhz", "benchmark"}}, - }, - }, - }, - { - ID: "handoff-decision-rationale", Family: "continuity", Skill: "rationale", - Why: "A decision without its reason gets re-litigated by the next agent.", - Known: KnownStrength, - Setup: []Event{ - doc(20, "kestrel-one", "Kestrel One", "Smart glasses programme."), - { - TS: ago(2), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "choose the battery chemistry", - Decisions: []string{"Stay with the pouch cell rather than moving to cylindrical, because the temple geometry cannot take a 14mm diameter without widening the hinge"}, - Next: "Confirm the pouch supplier's second source", - }, - }, - Query: Query{Task: "continue the battery decision", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"where did we land on the battery?", "resume the cell chemistry choice", "what's decided about power?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the decision", Any: []string{"pouch cell"}}, - {Label: "the reason, not just the decision", Any: []string{"14mm", "hinge", "temple geometry"}}, - }, - }, - }, - { - ID: "handoff-chain-three-agents", Family: "continuity", Skill: "multi-hop-handoff", - Why: "A ruled out something in the first session. C, two handoffs later, must still not retry it.", - Known: KnownStrength, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses programme."), - { - TS: ago(9), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "reduce weight below 48g", - Failed: []string{"Thinning the magnesium frame — fails the torsion spec at 0.8mm"}, - Next: "Look at the battery pack carrier", - }, - { - TS: ago(5), Actor: "cursor", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "reduce weight below 48g", - Text: "Carrier redesign saves 2.1g.", - Failed: []string{"Removing the carrier entirely — the cell needs the constraint under vibration"}, - Next: "Look at the hinge assembly", - }, - }, - Query: Query{Task: "continue reducing weight", Project: "kestrel-one", Agent: "codex", Budget: 4000, Now: benchNow}, - Wordings: []string{"how far are we from the weight target?", "keep making the glasses lighter", "what's been tried to cut grams?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the most recent failure", Any: []string{"removing the carrier", "vibration"}}, - {Label: "the earlier failure two handoffs back", Any: []string{"magnesium", "torsion", "0.8mm"}}, - {Label: "progress so far", Any: []string{"2.1g"}}, - }, - }, - }, - { - ID: "handoff-open-question", Family: "continuity", Skill: "open-questions", - Why: "An unresolved question is work in progress; losing it means somebody answers it twice.", - Known: KnownStrength, - Setup: []Event{ - doc(20, "ota-firmware", "OTA firmware", "Over-the-air update path."), - { - TS: ago(2), Actor: "claude", Kind: KindCheckpoint, Project: "ota-firmware", - Task: "decide whether OTA is viable", - Questions: []string{"Does the eMMC part have enough spare capacity for a second slot, or do we need a bigger part?"}, - Next: "Ask Tomas for the eMMC part number", - }, - }, - Query: Query{Task: "continue the OTA assessment", Project: "ota-firmware", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"is over-the-air updating going to work for us?", "resume the OTA feasibility question", "what's unresolved on remote updates?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the open question", Any: []string{"emmc", "spare capacity", "second slot"}}, - }, - }, - }, - - // ---- the ones we lose ------------------------------------------------ - - { - ID: "handoff-stale-context", Family: "continuity", Skill: "staleness", - Why: "Notes are thirteen days old against a moving schedule. Nothing marks them as old.", - Known: KnownStrength, - Setup: []Event{ - doc(60, "kestrel-one", "Kestrel One", "Ship date November 12. Tooling freeze six weeks prior."), - note(13, "claude", "kestrel-one", "We have about sixteen days before the tooling freeze, so there is room for one more DVT spin."), - }, - Query: Query{Task: "plan the next DVT spin", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"plan the next engineering build", "what should the next validation build include?", "schedule the upcoming hardware spin"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the note itself", Any: []string{"dvt", "tooling freeze"}}, - }, - Signal: []Fact{ - {Label: "the age of the estimate is surfaced", Any: []string{"13 days", "thirteen days", "days ago", "as of", "may be out of date", "stale"}}, - }, - }, - }, - { - ID: "handoff-superseded-plan", Family: "continuity", Skill: "supersession", - Why: "The plan in the checkpoint was abandoned by the user afterwards. Replaying it as current is worse than silence.", - Known: KnownStrength, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses programme."), - { - TS: ago(6), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "close the BOM gap", - Next: "Move the display driver to the Himax part to save $4.10", - }, - msg(2, "user", "We are not moving to the Himax display driver — legal flagged the licensing terms and we are staying with Solomon. Drop that thread entirely."), - }, - Query: Query{Task: "continue the BOM work", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"where are we on the parts cost?", "resume the BOM", "what's the next move on cost?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the current position", Any: []string{"solomon", "legal", "licensing"}}, - }, - Avoid: []Fact{ - {Label: "the abandoned plan presented as next", Any: []string{"move the display driver to the himax", "next: move the display driver"}}, - }, - }, - }, - { - ID: "handoff-conflicting-sources", Family: "continuity", Skill: "conflict", - Why: "Prose says one thing, the exported data says another. Surfacing only the prose confirms the wrong answer with a citation attached.", - Known: KnownWeakness, - Setup: []Event{ - doc(30, "ota-firmware", "OTA firmware", "Blocked: there is no spare flash budgeted for an A/B partition scheme, so OTA is not viable on this hardware."), - doc(28, "ota-firmware", "BOM export (rev 14)", "Line 22: SoC+memory module, 64GB eMMC, $27.80. Line 23: no discrete boot flash — boot from eMMC."), - }, - Query: Query{Task: "is an A/B partition scheme affordable on this hardware?", Project: "ota-firmware", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"do we have the storage for dual firmware slots?", "can the device hold two firmware images for safe updates?", "is there room for A/B updates?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the contradicting data line", Any: []string{"64gb", "emmc"}}, - }, - Signal: []Fact{ - {Label: "the conflict is flagged rather than resolved silently", Any: []string{"conflict", "contradict", "disagree", "however", "but the"}}, - }, - }, - }, - { - ID: "handoff-abandoned-thread", Family: "continuity", Skill: "recency-conflict", - Why: "The freshest note is not the most important one; the older blocker still blocks.", - Known: KnownWeakness, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses programme."), - note(8, "claude", "kestrel-one", "HARD BLOCKER: the certification lab has no slot before October 14; nothing ships before that date regardless of tooling."), - note(7, "claude", "kestrel-one", "Renamed the CAD files to the new convention."), - note(6, "claude", "kestrel-one", "Updated the colour swatches in the spec sheet."), - note(5, "claude", "kestrel-one", "Fixed a typo in the packaging copy."), - note(4, "claude", "kestrel-one", "Archived the old renders."), - }, - Query: Query{Task: "what is standing between us and shipping?", Project: "kestrel-one", Agent: "cursor", Budget: 700, Now: benchNow}, - Wordings: []string{"what's blocking launch?", "what do we still need before we can ship?", "biggest risk to the release right now?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the blocker, not the housekeeping", Any: []string{"october 14", "certification lab"}}, - }, - Avoid: []Fact{ - {Label: "housekeeping crowding out the blocker", Any: []string{"colour swatches", "typo in the packaging"}}, - }, - }, - }, - { - ID: "handoff-dead-end-code-moved", Family: "continuity", Skill: "staleness", - Why: "A dead end blamed on one file, and that file was rewritten since. Obeyed as settled, it stops the fix that now works.", - Known: KnownStrength, - Setup: []Event{ - doc(40, "ingest-svc", "Ingest service", "Loads vendor CSV exports into the warehouse."), - commit(30, "ingest-svc", "internal/parse/reader.go", "Parse CSV exports with a buffered reader"), - { - TS: ago(14), Actor: "claude", Kind: KindCheckpoint, Project: "ingest-svc", - Task: "import the 3GB vendor export", - Failed: []string{"streaming the 3GB export through the parser — internal/parse/reader.go loads the whole file into memory first, so it runs out of memory at about 2GB"}, - Next: "Split the export into 500MB chunks before importing", - }, - commit(4, "ingest-svc", "internal/parse/reader.go", "Read CSV input in 64KB chunks instead of loading the whole file"), - }, - Query: Query{Task: "import the 3GB vendor export", Project: "ingest-svc", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"load the big vendor CSV", "get the large export into the warehouse", "pick up the vendor import"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "the dead end itself", Any: []string{"whole file into memory", "out of memory"}}, - }, - Signal: []Fact{ - {Label: "the dead end is flagged as recorded before its file changed", Any: []string{"changed since", "may no longer hold"}}, - }, - }, - }, - } -} - -// --------------------------------------------------------------------------- -// Memory — what the store knows about the user and their world. -// --------------------------------------------------------------------------- - -func memoryCases() []Scenario { - return []Scenario{ - { - ID: "recall-direct", Family: "memory", Skill: "recall", - Why: "The floor. A stated fact, asked for directly.", - Known: KnownStrength, - Setup: append(noiseFacts(30), - said(20, "Our contract manufacturer is Pegatron, in the Suzhou plant."), - ), - Query: Query{Task: "who manufactures our hardware?", Budget: 2000, Now: benchNow}, - Wordings: []string{"who builds our devices?", "which contract manufacturer do we use?", "who makes the hardware for us?"}, - Gold: Gold{Carry: []Fact{{Label: "the manufacturer", Any: []string{"pegatron"}}}}, - }, - { - ID: "recall-preference", Family: "memory", Skill: "preference", - Why: "A standing preference stated once among distractors.", - Known: KnownStrength, - Setup: append(noiseFacts(30), - said(25, "I prefer written proposals over meetings — send me a doc and I will comment on it."), - ), - Query: Query{Task: "how should I bring a proposal to you?", Budget: 2000, Now: benchNow}, - Wordings: []string{"what's the best way to pitch you an idea?", "how do you like to receive proposals?", "should I just call you with a suggestion?"}, - Gold: Gold{Carry: []Fact{{Label: "the preference", Any: []string{"written proposal", "send me a doc"}}}}, - }, - { - ID: "recall-lexical-needle", Family: "memory", Skill: "lexical", - Why: "A one-off with a distinctive term the embedding blurs but exact match finds.", - Known: KnownStrength, - Setup: append(noiseFacts(30), - said(18, "The anodising line we use is called Fuyao Line 3 and it has a four week lead time."), - ), - Query: Query{Task: "what is the lead time on Fuyao Line 3?", Budget: 2000, Now: benchNow}, - Wordings: []string{"Fuyao Line 3 lead time?", "how long does Fuyao Line 3 take to deliver?", "when would an order from Fuyao Line 3 arrive?"}, - Gold: Gold{Carry: []Fact{{Label: "the lead time", Any: []string{"four week", "4 week"}}}}, - }, - { - ID: "recall-graph-reach", Family: "memory", Skill: "graph-reach", - Why: "The relevant note shares no words with the question and is reachable only through a link the user wrote.", - Known: KnownStrength, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses. Target BOM $118, actual $141.20. Margin depends on [[bonding-yield]]."), - doc(30, "", "Bonding yield", "Display bonding runs at 71 percent first-pass. Every scrapped unit is absorbed by the units that ship."), - doc(29, "", "Packaging", "Recycled moulded pulp tray, $1.90 per unit."), - }, - Query: Query{Task: "why is the per-unit cost higher than the parts add up to?", Project: "kestrel-one", Budget: 2000, Now: benchNow}, - Wordings: []string{"why does each unit cost more than its components?", "where is the extra per-unit cost coming from?", "what explains the gap between parts cost and unit cost?"}, - Gold: Gold{Carry: []Fact{ - {Label: "the yield note, reached by link not by wording", Any: []string{"71 percent", "71%", "scrapped"}}, - }}, - }, - { - ID: "recall-multi-hop", Family: "memory", Skill: "multi-hop", - Why: "The answer needs two facts from two places; either alone is not an answer.", - Known: KnownWeakness, - Setup: append(noiseFacts(20), - said(30, "Tomas runs manufacturing operations and owns every supplier relationship."), - said(22, "The tooling freeze needs sign-off from whoever owns supplier relationships."), - ), - Query: Query{Task: "who has to sign off the tooling freeze?", Budget: 2000, Now: benchNow}, - Wordings: []string{"whose approval does freezing the tooling need?", "who approves locking the tooling?", "who owns the sign-off on the tooling lock?"}, - Gold: Gold{Carry: []Fact{ - {Label: "the person", Any: []string{"tomas"}}, - {Label: "the link that identifies them", Any: []string{"supplier relationship"}}, - }}, - }, - { - ID: "supersession-current-value", Family: "memory", Skill: "supersession", - Why: "A number that changed twice. The current one must come back and the old ones must not.", - Known: KnownStrength, - Setup: append(noiseFacts(20), - said(40, "We are targeting a $199 retail price."), - said(25, "Retail is moving to $229 after the optics quote came back."), - said(6, "Final call: retail price is $249. That is locked for launch."), - ), - Query: Query{Task: "what is the retail price?", Budget: 2000, Now: benchNow}, - Wordings: []string{"how much will the glasses sell for?", "what's our price point?", "what are we charging customers?"}, - Gold: Gold{ - Carry: []Fact{{Label: "the current price", Any: []string{"$249", "249"}}}, - Avoid: []Fact{ - {Label: "the first superseded price", Any: []string{"$199", "199"}}, - {Label: "the second superseded price", Any: []string{"$229", "229"}}, - }, - }, - }, - - // ---- the ones we lose ------------------------------------------------ - - { - ID: "negation-decided-against", Family: "memory", Skill: "negation", - Why: "'We decided not to use X' and 'we use X' sit next to each other in embedding space.", - Known: KnownStrength, - Setup: append(noiseFacts(20), - said(15, "We decided against Rust for the firmware — the vendor SDK is C only and the bindings were a maintenance sink."), - ), - Query: Query{Task: "what language is the firmware written in?", Budget: 2000, Now: benchNow}, - Wordings: []string{"which programming language does the firmware use?", "are we writing the firmware in Rust?", "what's the device code written in?"}, - Gold: Gold{ - Carry: []Fact{{Label: "the actual answer", Any: []string{" c only", "vendor sdk is c"}}}, - Signal: []Fact{ - {Label: "the negation is preserved, not dropped", Any: []string{"decided against", "not use rust", "rejected rust"}}, - }, - }, - }, - { - ID: "negation-preference", Family: "memory", Skill: "negation", - Why: "A negative preference retrieved for a positively-worded question reads as an endorsement.", - Known: KnownStrength, - Setup: append(noiseFacts(20), - said(20, "Do not schedule me anything before 10am — I am useless in the mornings."), - ), - Query: Query{Task: "when should I schedule the supplier call?", Budget: 2000, Now: benchNow}, - Wordings: []string{"what time works for the call with the supplier?", "can I book the vendor call for 9am?", "when's a good slot for the supplier meeting?"}, - Gold: Gold{ - Carry: []Fact{{Label: "the constraint", Any: []string{"before 10am", "10am"}}}, - Signal: []Fact{ - {Label: "carried as a prohibition, not a preference for mornings", Any: []string{"do not", "don't", "avoid", "never"}}, - }, - }, - }, - { - ID: "abstention-never-recorded", Family: "memory", Skill: "abstention", - Why: "Asked about something never said. Saying so is the correct answer; the LongMemEval harness filters these out before scoring.", - Known: KnownStrength, - Setup: append(noiseFacts(30), - said(20, "Our contract manufacturer is Pegatron, in the Suzhou plant."), - ), - Query: Query{Task: "what did we agree the warranty period would be?", Budget: 2000, Now: benchNow}, - Wordings: []string{"how long is the warranty?", "what warranty did we settle on?", "what's the warranty coverage?"}, - Gold: Gold{ - Signal: []Fact{ - {Label: "admits it does not know", Any: []string{"nothing recorded", "no record", "not know", "nothing on", "no memories", "nothing found", "no matching", "not recorded"}}, - }, - Avoid: []Fact{ - {Label: "confabulating a neighbour as the answer", Any: []string{"warranty period is", "warranty is 1", "warranty is 2", "12 months", "24 months"}}, - }, - }, - }, - { - ID: "abstention-adjacent", Family: "memory", Skill: "abstention", - Why: "The store knows the neighbouring fact. Returning it unlabelled reads as an answer to a question nobody answered.", - Known: KnownStrength, - Setup: append(noiseFacts(20), - said(20, "The Suzhou plant handles final assembly."), - said(18, "The Suzhou plant runs two shifts."), - ), - Query: Query{Task: "which plant does the optical bonding?", Budget: 2000, Now: benchNow}, - Wordings: []string{"who handles optical bonding for us?", "where is the display bonding done?", "which factory bonds the optics?"}, - Gold: Gold{ - Signal: []Fact{ - {Label: "flags that bonding specifically is unrecorded", Any: []string{"nothing recorded", "no record", "not know", "nothing on", "no matching", "not recorded", "unclear"}}, - }, - }, - }, - { - ID: "numeric-aggregation", Family: "memory", Skill: "arithmetic", - Why: "The answer is a sum across five notes. Retrieval can surface the parts; it cannot add them.", - Known: KnownWeakness, - Setup: []Event{ - doc(30, "tooling", "Tooling — injection mould A", "Injection mould set A: $42,000."), - doc(29, "tooling", "Tooling — injection mould B", "Injection mould set B: $38,500."), - doc(28, "tooling", "Tooling — CNC fixtures", "CNC fixtures: $11,200."), - doc(27, "tooling", "Tooling — test jigs", "Test jigs: $7,800."), - doc(26, "tooling", "Tooling — line retooling", "Assembly line retooling: $16,500."), - }, - Query: Query{Task: "what have we spent on tooling in total?", Project: "tooling", Budget: 2000, Now: benchNow}, - Wordings: []string{"how much has tooling cost us so far?", "total tooling spend?", "add up everything we've paid for tooling"}, - Gold: Gold{ - Carry: []Fact{{Label: "the computed total", Any: []string{"116,000", "116000", "$116"}}}, - }, - }, - { - ID: "temporal-ordering", Family: "memory", Skill: "temporal", - // Every system measured here scores zero on this, including the two - // that run a model over their writes. The label asks the store to - // state a comparison, and a store retrieves — this is the same bucket - // as numeric-aggregation, and it stays a weakness because it is one. - // - // What did change is that the material is now answerable. Three facts - // each saying "today", recorded weeks apart, used to arrive with no - // dates at all; they now carry the day they were written, so a reader - // can order them. Carrying the evidence is retrieval's job. Drawing - // the conclusion is not. - Why: "Which came first. The store holds timestamps but the answer needs them compared, not listed.", - Known: KnownWeakness, - Setup: append(noiseFacts(15), - said(35, "Signed the Pegatron manufacturing agreement today."), - said(28, "Locked the industrial design today — no more changes to the shell."), - said(12, "Kicked off certification prep today."), - ), - Query: Query{Task: "did we lock the industrial design before or after signing with Pegatron?", Budget: 2000, Now: benchNow}, - Wordings: []string{"which came first, signing Pegatron or the industrial design lock?", "was the ID frozen before we signed our manufacturer?", "did the Pegatron contract come before the design freeze?"}, - Gold: Gold{ - Carry: []Fact{{Label: "the ordering, stated", Any: []string{"after signing", "after the pegatron", "pegatron first", "signed first"}}}, - }, - }, - { - ID: "temporal-window", Family: "memory", Skill: "temporal", - // Was a weakness until internal/when: the phrase is now resolved - // against the read clock and the out-of-period notes are dropped, - // with a line saying how many and what window ran. Kept as a strength - // rather than retired, because the failure it caught — returning - // everything to a question that named a period — is the default - // behaviour of every other system here and would return the moment - // the parser stopped being consulted. - Why: "'Last month' has to be resolved against the clock, not matched as a string.", - Known: KnownStrength, - Setup: []Event{ - note(45, "user", "kestrel-one", "Spent the week on the optics quote comparison."), - note(35, "user", "kestrel-one", "Ran the drop test series on the magnesium frame."), - note(5, "user", "kestrel-one", "Started the packaging design review."), - }, - Query: Query{Task: "what was I working on about five weeks ago?", Project: "kestrel-one", Budget: 2000, Now: benchNow}, - Wordings: []string{"what was on my plate a month and a bit ago?", "what was I doing roughly five weeks back?", "what was I busy with around 35 days ago?"}, - Gold: Gold{ - Carry: []Fact{{Label: "the item in the window", Any: []string{"drop test", "magnesium"}}}, - Avoid: []Fact{{Label: "the recent item, which is out of the window", Any: []string{"packaging design review"}}}, - }, - }, - { - ID: "contradiction-unflagged", Family: "memory", Skill: "conflict", - Why: "Two sources disagree and neither supersedes the other. Picking one silently is the failure.", - Known: KnownStrength, - Setup: append(noiseFacts(15), - doc(20, "", "Ops summary", "First-pass bonding yield is 71 percent."), - doc(19, "", "Factory report week 34", "Bonding first-pass yield measured at 63 percent across the week."), - ), - Query: Query{Task: "what is the bonding yield?", Budget: 2000, Now: benchNow}, - Wordings: []string{"what yield are we getting on bonding?", "how good is the bonding yield?", "bonding yield number?"}, - Gold: Gold{ - Carry: []Fact{ - {Label: "both figures present", All: []string{"71", "63"}}, - }, - Signal: []Fact{ - {Label: "the disagreement is named", Any: []string{"conflict", "contradict", "disagree", "two figures", "differs", "however"}}, - }, - }, - }, - { - ID: "scale-haystack", Family: "memory", Skill: "scale", - Why: "Two hundred facts, one of them the answer. Precision under volume.", - Known: KnownStrength, - Setup: append(noiseFacts(200), - said(9, "The hinge supplier is Sugatsune and they quoted 11 weeks for the custom detent."), - ), - Query: Query{Task: "how long is the hinge lead time?", Budget: 2000, Now: benchNow}, - Wordings: []string{"hinge lead time?", "how many weeks to get hinges?", "when would hinges arrive if we ordered today?"}, - Gold: Gold{Carry: []Fact{{Label: "the lead time", Any: []string{"11 week", "eleven week"}}}}, - }, - } -} - -// --------------------------------------------------------------------------- -// Durability — what is left when the cache is gone. -// --------------------------------------------------------------------------- - -func durability() []Scenario { - return []Scenario{ - { - ID: "durability-handoff-survives-wipe", Family: "durability", Skill: "source-of-truth", - Why: "Delete every derived artifact. A handoff recorded as a file survives; one recorded in an index does not.", - Known: KnownStrength, - DropDerived: true, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses programme."), - { - TS: ago(2), Actor: "claude", Kind: KindCheckpoint, Project: "kestrel-one", - Task: "close the BOM gap", - Failed: []string{"Re-quoting the waveguide — no movement under 10k units"}, - Next: "Quote the display driver alternatives", - }, - }, - Query: Query{Task: "continue the BOM work", Project: "kestrel-one", Agent: "cursor", Budget: 4000, Now: benchNow}, - Wordings: []string{"resume the BOM work", "where's the bill of materials at?", "carry on cutting costs"}, - Gold: Gold{Carry: []Fact{ - {Label: "the failed approach survived", Any: []string{"waveguide"}}, - {Label: "the next step survived", Any: []string{"display driver"}}, - }}, - }, - { - ID: "durability-memories-survive-wipe", Family: "durability", Skill: "source-of-truth", - Why: "The principle says the database is a cache. Memories are files now, so losing the cache costs nothing.", - Known: KnownStrength, - DropDerived: true, - Setup: append(noiseFacts(10), - said(20, "Our contract manufacturer is Pegatron, in the Suzhou plant."), - said(18, "I prefer written proposals over meetings."), - ), - Query: Query{Task: "who manufactures our hardware?", Budget: 2000, Now: benchNow}, - Wordings: []string{"who builds our devices?", "which contract manufacturer do we use?", "who makes our hardware?"}, - Gold: Gold{Carry: []Fact{ - {Label: "the fact survived the wipe", Any: []string{"pegatron"}}, - }}, - }, - { - ID: "durability-notes-survive-wipe", Family: "durability", Skill: "source-of-truth", - Why: "Documents written as files should come back byte-identical after a rebuild.", - Known: KnownStrength, - DropDerived: true, - Setup: []Event{ - doc(30, "kestrel-one", "Kestrel One", "Smart glasses. Target BOM $118, actual $141.20."), - doc(30, "", "Bonding yield", "Display bonding runs at 71 percent first-pass."), - }, - Query: Query{Task: "what is the BOM gap?", Project: "kestrel-one", Budget: 2000, Now: benchNow}, - Wordings: []string{"how far off target is the BOM?", "how big is the cost gap?", "where does the bill of materials stand against target?"}, - Gold: Gold{Carry: []Fact{ - {Label: "the note survived", Any: []string{"141.20", "$141"}}, - }}, - }, - } -} - -// --------------------------------------------------------------------------- -// Filler. Distractors are part of the measurement: a store that returns the -// right answer from a haystack of three is not doing the job it will be asked -// to do live. -// --------------------------------------------------------------------------- - -var fillerTopics = []string{ - "The staging server restarts nightly at 3am", - "Design review notes are filed under the sprint folder", - "Invoices go to accounts payable on the first of the month", - "The office wifi password rotates quarterly", - "Standup is at 9:15 in the small room", - "Build artifacts are retained for thirty days", - "The shared calendar is owned by operations", - "Expense reports need a receipt over twenty dollars", - "The parking permit renews in March", - "Slack retention is set to one year", -} - -func noise(n int, project, prefix string) []Event { - out := make([]Event, 0, n) - for i := 0; i < n; i++ { - out = append(out, note(60-i%50, "claude", project, - fmt.Sprintf("%s (%s)", prefix, fillerTopics[i%len(fillerTopics)]))) - } - return out -} - -func noiseFacts(n int) []Event { - out := make([]Event, 0, n) - for i := 0; i < n; i++ { - out = append(out, said(60-i%50, - fmt.Sprintf("%s, item %d.", fillerTopics[i%len(fillerTopics)], i))) - } - return out -} diff --git a/internal/eval/score.go b/internal/eval/score.go deleted file mode 100644 index ae849fd..0000000 --- a/internal/eval/score.go +++ /dev/null @@ -1,340 +0,0 @@ -package eval - -import ( - "sort" - "strings" -) - -// Scoring is deliberately mechanical. -// -// Every headline number here can be computed with no model running, which is -// what makes the suite reproducible on a plane and comparable between two -// people who never talk to each other. A model-judged layer exists (judge.go) -// and answers a question matching cannot — whether an agent handed this context -// actually avoids repeating the ruled-out work — but it sits on top and is -// reported separately. If the two ever disagree, that disagreement is a finding -// about the metric, not a tie to be broken silently. - -// A Fact is something the harness looks for in a response. Matching is -// substring, case-insensitive, over whitespace-collapsed text. -// -// Surface variants are the scenario author's job, not the matcher's: write -// Any: {"71 percent", "71%"} rather than hoping a cleverer matcher guesses. A -// fuzzy matcher that silently accepts near-misses is how a benchmark starts -// flattering everyone equally. -type Fact struct { - Label string // what to call this in the report - Any []string // at least one must appear - All []string // every one must appear (for facts that are only right together) -} - -// In reports whether the fact is present in the given text. -func (f Fact) In(hay string) bool { - hay = normalize(hay) - if len(f.All) > 0 { - for _, want := range f.All { - if !strings.Contains(hay, normalize(want)) { - return false - } - } - } - if len(f.Any) > 0 { - for _, want := range f.Any { - if strings.Contains(hay, normalize(want)) { - return true - } - } - return false - } - return len(f.All) > 0 -} - -func normalize(s string) string { - return strings.ToLower(strings.Join(strings.Fields(s), " ")) -} - -// Gold is what a correct answer looks like. -type Gold struct { - // Carry must appear. These are the facts the arriving agent cannot do the - // job without. - Carry []Fact - // Avoid must NOT appear — a superseded price, a cancelled plan, a value the - // user corrected. Carrying a stale fact forward is worse than carrying - // nothing, because the agent will act on it. - Avoid []Fact - // Signal is meta-knowledge about the answer rather than the answer: that the - // information is old, that the store does not know, that a finding came from - // a particular agent. Scored separately because almost every system alive - // scores zero here, and folding that into the main number would say more - // about the state of the art than about any one system. - Signal []Fact -} - -// Known is what logos is expected to do on a case today. Recording the -// prediction in the suite is what turns a weakness into a tracked one: a -// "weakness" that starts passing is progress worth noticing, and a "strength" -// that starts failing is a regression the aggregate numbers would otherwise -// average away. -type Known string - -const ( - KnownStrength Known = "strength" - KnownWeakness Known = "weakness" -) - -// A Scenario is one case: a history, a task, and what a correct answer carries. -type Scenario struct { - ID string - Family string // continuity | memory | durability - Skill string // the specific capability under test - Why string // one line, for the report — what this case is really asking - Known Known // logos's expected outcome, so regressions are visible - - Setup []Event - Query Query - Gold Gold - - // DropDerived wipes every rebuildable artifact between writing and reading. - // Systems that keep no source of truth outside their own index return - // nothing after this, which is the intended result. - DropDerived bool - - // Wordings are other ways a user would ask the same thing, run by Expand - // as separate cases. None may contain a gold term, or a reworded case - // would pass on its own echo. - Wordings []string - - // Base and Axis are set by Expand: the scenario a variant came from, and - // what it varies ("" for the case as written). Variant describes it for - // the report. - Base, Axis, Variant string -} - -// A Score is one adapter's result on one scenario. -type Score struct { - Scenario string - Family string - Skill string - Known Known - Adapter string - - Base, Axis, Variant string - - CarryHit, CarryTotal int - LeakHit, LeakTotal int - SignalHit, SignalTotal int - - Tokens int - Budget int - - Missed []string - Leaked []string - Unsignalled []string - Err error -} - -func ratio(hit, total int) float64 { - if total == 0 { - return 0 - } - return float64(hit) / float64(total) -} - -// Recall is the share of required facts that survived. -// -// A scenario that requires nothing scores 1, not 0. The abstention cases have -// no Carry labels by design — the correct answer is an admission of ignorance, -// and there is no fact to fetch. Dividing zero by zero and calling it a miss -// marked every system down for declining to invent one. -func (s Score) Recall() float64 { - if s.CarryTotal == 0 { - return 1 - } - return ratio(s.CarryHit, s.CarryTotal) -} - -// Leakage is the share of facts that should have been suppressed but were not. -func (s Score) Leakage() float64 { return ratio(s.LeakHit, s.LeakTotal) } - -// Signal is the share of meta-properties the answer exhibited. -func (s Score) Signal() float64 { return ratio(s.SignalHit, s.SignalTotal) } - -// Fidelity is the headline: carrying what is needed while not carrying what is -// wrong. Multiplicative rather than averaged, because a response that carries -// every required fact alongside a superseded price is not "half right" — the -// agent will act on the stale number, and the correct facts sitting next to it -// do not undo that. -func (s Score) Fidelity() float64 { return s.Recall() * (1 - s.Leakage()) } - -// Pass is whether the scenario was satisfied on every axis it declares. -// -// Fidelity alone is not enough, and the staleness case is why: its gold labels -// ask for the note (carried) and for its age (a signal). Judged on fidelity it -// scored 100% while doing precisely the thing the case was written to catch. A -// scenario passes when it meets every bar it set, not the most flattering one. -func (s Score) Pass() bool { - const bar = 0.75 - if s.Err != nil || s.Fidelity() < bar { - return false - } - return s.SignalTotal == 0 || s.Signal() >= bar -} - -// Density is required facts carried per 1000 tokens spent. The axis on which -// "put the whole history in the window" loses: it scores well on recall by -// brute force and terribly here, and the window it burns is the window the -// actual conversation needed. -func (s Score) Density() float64 { - if s.Tokens == 0 { - return 0 - } - return float64(s.CarryHit) * 1000 / float64(s.Tokens) -} - -// Over reports whether the response blew the token ceiling it was given. -func (s Score) Over() bool { return s.Budget > 0 && s.Tokens > s.Budget } - -// stripEcho removes the question from the answer before matching. -// -// Systems that head their output with the task they were given ("Context for: -// did we lock the design before or after signing with Pegatron?") would -// otherwise satisfy any gold label whose wording overlaps the question. The -// temporal-ordering case scored a clean 100% this way while returning two -// undated facts in arbitrary order — it matched "after signing" in the echoed -// header. Only the verbatim query string is removed, so a real note that -// happens to share words with the question still counts. -func stripEcho(text string, q Query) string { - hay := normalize(text) - for _, echo := range []string{q.Task, q.Project} { - if e := normalize(echo); len(e) > 8 { - hay = strings.ReplaceAll(hay, e, " ") - } - } - return hay -} - -// grade scores one response against a scenario's gold labels. -func grade(sc Scenario, ad string, r Response) Score { - out := Score{ - Scenario: sc.ID, Family: sc.Family, Skill: sc.Skill, Known: sc.Known, - Adapter: ad, Tokens: Tokens(r.Text), Budget: sc.Query.Budget, Err: r.Err, - Base: sc.Base, Axis: sc.Axis, Variant: sc.Variant, - } - if out.Base == "" { - out.Base = sc.ID - } - hay := stripEcho(r.Text, sc.Query) - - for _, f := range sc.Gold.Carry { - out.CarryTotal++ - if f.In(hay) { - out.CarryHit++ - } else { - out.Missed = append(out.Missed, f.Label) - } - } - for _, f := range sc.Gold.Avoid { - out.LeakTotal++ - if f.In(hay) { - out.LeakHit++ - out.Leaked = append(out.Leaked, f.Label) - } - } - for _, f := range sc.Gold.Signal { - out.SignalTotal++ - if f.In(hay) { - out.SignalHit++ - } else { - out.Unsignalled = append(out.Unsignalled, f.Label) - } - } - return out -} - -// An Aggregate rolls scores up for one adapter over one slice of the suite. -type Aggregate struct { - Adapter string - Group string - N int - - Recall float64 - Leakage float64 - Signal float64 - Fidelity float64 - Density float64 - // PassRate is the share of scenarios that met every bar they set. It is the - // honest summary for families where fidelity does not apply — an abstention - // case has no facts to carry, so it scores full recall by definition and - // only the signal says whether the system did the right thing. - PassRate float64 - Tokens int // mean - OverRuns int // responses that exceeded their budget - Errors int -} - -// Roll aggregates scores by a grouping function. Means are unweighted across -// scenarios: every case counts once, so a family with more facts in its gold -// labels does not quietly dominate the headline. -func Roll(scores []Score, group func(Score) string) []Aggregate { - byGroup := map[string][]Score{} - var order []string - for _, s := range scores { - g := group(s) - if _, seen := byGroup[g]; !seen { - order = append(order, g) - } - byGroup[g] = append(byGroup[g], s) - } - sort.Strings(order) - - out := make([]Aggregate, 0, len(order)) - for _, g := range order { - ss := byGroup[g] - a := Aggregate{Group: g, N: len(ss)} - if len(ss) > 0 { - a.Adapter = ss[0].Adapter - } - var tokens int - // Leakage and Signal are averaged only over the scenarios that actually - // test them. Counting a case with no Avoid labels as perfect leakage - // would let a suite dilute the metric just by adding unrelated cases. - var leakN, sigN int - for _, s := range ss { - a.Recall += s.Recall() - a.Fidelity += s.Fidelity() - a.Density += s.Density() - if s.Pass() { - a.PassRate++ - } - tokens += s.Tokens - if s.LeakTotal > 0 { - a.Leakage += s.Leakage() - leakN++ - } - if s.SignalTotal > 0 { - a.Signal += s.Signal() - sigN++ - } - if s.Over() { - a.OverRuns++ - } - if s.Err != nil { - a.Errors++ - } - } - n := float64(len(ss)) - a.Recall /= n - a.Fidelity /= n - a.Density /= n - a.PassRate /= n - a.Tokens = tokens / len(ss) - if leakN > 0 { - a.Leakage /= float64(leakN) - } - if sigN > 0 { - a.Signal /= float64(sigN) - } - out = append(out, a) - } - return out -} diff --git a/internal/eval/variants.go b/internal/eval/variants.go deleted file mode 100644 index 9ab41bd..0000000 --- a/internal/eval/variants.go +++ /dev/null @@ -1,259 +0,0 @@ -package eval - -import ( - "fmt" - "hash/fnv" - "math/rand" - "strings" -) - -// Variants. -// -// Thirty-two cases asked one way each cannot tell a real improvement from a -// lucky phrasing: one case flipping moves the pass rate by three points. The -// recall floor made the cost concrete. It broke handoff-superseded-plan only -// when the task was vague, and the suite caught it because that case happened -// to be worded vaguely — a sharper wording would have hidden it. -// -// So each case can also run with the question asked differently and with -// unrelated history around it. Both are still hand-written or drawn from a -// hand-written pool, for the reason the suite gives against generated cases: -// gold labels must stay auditable, and the labels here never change — only -// what surrounds them does. -// -// Each axis is varied on its own rather than crossed, so a failing variant -// names its cause: a case that fails reworded and passes with distractors is -// sensitive to wording, not to noise. - -// Axes a variant can differ on. The empty axis is the case as written. -const ( - AxisWording = "reworded" - AxisDistractors = "distractors" -) - -// Expand returns every scenario as written, followed by one case per extra -// wording and one per distractor seed. seeds <= 0 adds no distractor cases. -func Expand(suite []Scenario, seeds int) []Scenario { - var out []Scenario - for _, sc := range suite { - sc.Base = sc.ID - out = append(out, sc) - for i, w := range sc.Wordings { - v := sc - v.ID = fmt.Sprintf("%s~w%d", sc.ID, i+1) - v.Axis, v.Variant = AxisWording, fmt.Sprintf("%q", w) - v.Query.Task = w - out = append(out, v) - } - for seed := 1; seed <= seeds; seed++ { - v := sc - v.ID = fmt.Sprintf("%s~d%d", sc.ID, seed) - v.Axis, v.Variant = AxisDistractors, fmt.Sprintf("distractor set %d", seed) - v.Setup = append(append([]Event(nil), sc.Setup...), distractors(sc, seed)...) - out = append(out, v) - } - } - return out -} - -// goldTerms is every string a gold label matches on. A wording or distractor -// that contains one could pass or fail a case by itself, and the variant would -// then be measuring the harness rather than the system. -func goldTerms(g Gold) []string { - var out []string - for _, fs := range [][]Fact{g.Carry, g.Avoid, g.Signal} { - for _, f := range fs { - out = append(out, f.Any...) - out = append(out, f.All...) - } - } - return out -} - -func containsAny(text string, terms []string) bool { - hay := normalize(text) - for _, t := range terms { - if strings.Contains(hay, normalize(t)) { - return true - } - } - return false -} - -// distractorPool is ordinary traffic from the same programme the cases are set -// in. The filler topics are office admin an embedding separates from a BOM -// question easily; these share the case's world and vocabulary, which is what -// a real project's history around one checkpoint looks like. -var distractorPool = []string{ - "Marketing wants the launch video cut to ninety seconds.", - "The charging case lid hinge squeaks on the EVT units; logged for mechanical.", - "Colour team picked graphite and sand for the first two colourways.", - "The companion app pairing flow needs a second pass from design.", - "Customer support asked for a spare nose-pad SKU.", - "The Bluetooth antenna passed pre-scan with 3dB of margin.", - "Photography for the retail box is booked for the fourteenth.", - "The speaker grille mesh supplier sent revised samples.", - "Regulatory asked for updated battery label artwork.", - "Firmware build 0.9.3 fixes the wake-word false triggers.", - "The prescription insert partner wants an NDA before sharing lens data.", - "EVT units shipped to the three pilot customers.", - "The retail demo stand needs a locking USB-C port.", - "Sales wants a volume forecast by region before the board meeting.", - "Legal reviewed the privacy notice for the camera indicator LED.", - "The design team is moving its files to the new shared drive.", - "The touch strip on the right temple misreads wet fingers.", - "Industrial design wants a matte finish on the temple tips.", - "Packaging copy is waiting on the final product name.", - "The Reno warehouse can hold two thousand units.", - "Supplier audit of the frame vendor is scheduled for next quarter.", - "The camera module passed its focus calibration on the pilot line.", - "Pilot customers asked for a quieter charging chime.", - "The accessibility review flagged low contrast on the setup screens.", -} - -// distractors picks a seeded handful from the pool for one case. The seed is -// mixed with the case id so two cases do not get the same set, and the same -// case gets the same set on every run — a variant that changed between runs -// would reintroduce the noise it exists to measure. Anything that contains -// one of the case's gold terms is skipped. -func distractors(sc Scenario, seed int) []Event { - h := fnv.New64a() - h.Write([]byte(sc.ID)) - rng := rand.New(rand.NewSource(int64(h.Sum64()) + int64(seed))) - - terms := goldTerms(sc.Gold) - const n = 10 - var out []Event - for i, idx := range rng.Perm(len(distractorPool)) { - if len(out) == n { - break - } - text := distractorPool[idx] - if containsAny(text, terms) { - continue - } - days := 1 + rng.Intn(40) - // Where the case is a project's history, the noise is that project's - // working notes, from more than one agent; otherwise it is things the - // user said, alongside the facts the case asks about. - if sc.Query.Project != "" { - actor := []string{"claude", "cursor"}[i%2] - out = append(out, note(days, actor, sc.Query.Project, text)) - } else { - out = append(out, said(days, text)) - } - } - return out -} - -// hasVariants reports whether any score came from a variant. -func hasVariants(results []Result) bool { - for _, r := range results { - for _, s := range r.Scores { - if s.Axis != "" { - return true - } - } - } - return false -} - -// asWritten keeps only the scores for scenarios as written. The headline stays -// on those so a number from a variant run can be set beside every earlier one. -func asWritten(results []Result) []Result { - out := make([]Result, 0, len(results)) - for _, r := range results { - kept := Result{Adapter: r.Adapter} - for _, s := range r.Scores { - if s.Axis == "" { - kept.Scores = append(kept.Scores, s) - } - } - out = append(out, kept) - } - return out -} - -// stability reports how each system holds up away from the wording and -// history each case was written with, and names every case whose outcome -// depends on the variant — for the first-listed system, the one under test. -func stability(results []Result) string { - var b strings.Builder - b.WriteString("\n── across variants ───────────────────────────────────────────────────────\n\n") - b.WriteString(fmt.Sprintf("%-18s %11s %10s %12s %11s\n", "system", "as written", AxisWording, AxisDistractors, "consistent")) - for _, r := range results { - pass := map[string]float64{} - n := map[string]float64{} - outcomes := map[string]map[bool]bool{} - var bases []string - for _, s := range r.Scores { - n[s.Axis]++ - if s.Pass() { - pass[s.Axis]++ - } - if outcomes[s.Base] == nil { - outcomes[s.Base] = map[bool]bool{} - bases = append(bases, s.Base) - } - outcomes[s.Base][s.Pass()] = true - } - consistent := 0 - for _, base := range bases { - if len(outcomes[base]) == 1 { - consistent++ - } - } - rate := func(axis string) string { - if n[axis] == 0 { - return "—" - } - return fmt.Sprintf("%.1f%%", pass[axis]/n[axis]*100) - } - b.WriteString(fmt.Sprintf("%-18s %11s %10s %12s %11s\n", r.Adapter, - rate(""), rate(AxisWording), rate(AxisDistractors), fmt.Sprintf("%d/%d", consistent, len(bases)))) - } - b.WriteString("\nconsistent = the case passed on every variant, or failed on every one.\n") - - if len(results) == 0 { - return b.String() - } - subject := results[0] - type group struct { - pass, n int - failed []string - } - groups := map[string]*group{} - var order []string - for _, s := range subject.Scores { - g := groups[s.Base] - if g == nil { - g = &group{} - groups[s.Base] = g - order = append(order, s.Base) - } - g.n++ - if s.Pass() { - g.pass++ - continue - } - label := "as written" - if s.Axis != "" { - label = s.Axis + " " + s.Variant - } - g.failed = append(g.failed, label) - } - var mixed []string - for _, base := range order { - g := groups[base] - if g.pass > 0 && g.pass < g.n { - mixed = append(mixed, fmt.Sprintf(" %-34s %d/%d failed: %s", base, g.pass, g.n, strings.Join(g.failed, "; "))) - } - } - if len(mixed) == 0 { - b.WriteString(fmt.Sprintf("\nevery case gave %s the same outcome on every variant.\n", subject.Adapter)) - return b.String() - } - b.WriteString(fmt.Sprintf("\ncases whose outcome depends on the variant (%s):\n", subject.Adapter)) - b.WriteString(strings.Join(mixed, "\n") + "\n") - return b.String() -} diff --git a/internal/eval/variants_test.go b/internal/eval/variants_test.go deleted file mode 100644 index 8ca98a9..0000000 --- a/internal/eval/variants_test.go +++ /dev/null @@ -1,127 +0,0 @@ -package eval - -import ( - "reflect" - "strings" - "testing" -) - -// A rewording that contains a gold term passes on its own echo whenever the -// response paraphrases the question, and stripEcho only removes the verbatim -// task — so the variant would measure the wording, not the system. -func TestEveryScenarioHasThreeRewordingsThatDoNotGiveAwayTheAnswer(t *testing.T) { - for _, sc := range Suite() { - if len(sc.Wordings) < 3 { - t.Errorf("%s has %d rewordings, want at least 3", sc.ID, len(sc.Wordings)) - } - terms := goldTerms(sc.Gold) - seen := map[string]bool{normalize(sc.Query.Task): true} - for _, w := range sc.Wordings { - if seen[normalize(w)] { - t.Errorf("%s: rewording %q repeats the task or another rewording", sc.ID, w) - } - seen[normalize(w)] = true - for _, term := range terms { - if strings.Contains(normalize(w), normalize(term)) { - t.Errorf("%s: rewording %q contains the gold term %q", sc.ID, w, term) - } - } - } - } -} - -// A distractor that carried a gold term would make a case pass or leak by -// noise alone. The pool is filtered per case; this holds the filter to it. -func TestInjectedDistractorsNeverCarryTheAnswer(t *testing.T) { - base := map[string]Scenario{} - for _, sc := range Suite() { - base[sc.ID] = sc - } - for _, v := range Expand(Suite(), 2) { - if v.Axis != AxisDistractors { - continue - } - added := v.Setup[len(base[v.Base].Setup):] - if len(added) == 0 { - t.Errorf("%s added no history, so it tests nothing the case as written does not", v.ID) - } - for _, e := range added { - if containsAny(e.Text, goldTerms(v.Gold)) { - t.Errorf("%s: distractor %q contains a gold term", v.ID, e.Text) - } - } - } -} - -// A variant that changed between runs would put back the run-to-run noise the -// variants exist to measure, and two runs could no longer be compared. -func TestDistractorVariantsAreTheSameOnEveryRun(t *testing.T) { - a, b := Expand(Suite(), 2), Expand(Suite(), 2) - for i := range a { - if !reflect.DeepEqual(a[i].Setup, b[i].Setup) { - t.Fatalf("%s drew different distractors on a second expansion", a[i].ID) - } - } - d1, d2 := distractors(Suite()[0], 1), distractors(Suite()[0], 2) - if reflect.DeepEqual(d1, d2) { - t.Error("two seeds drew the same distractor set, so the second adds nothing") - } -} - -func TestAnEmptyAnswerScoresNothingOnAnyVariant(t *testing.T) { - scores, err := Run(silent{}, Expand(Suite(), 2), Options{}) - if err != nil { - t.Fatal(err) - } - for _, s := range scores { - if s.Pass() { - t.Errorf("returning nothing passed %s", s.Scenario) - } - } -} - -func scored(base, axis, variant string, pass bool) Score { - s := Score{Scenario: base, Base: base, Axis: axis, Variant: variant, Family: "memory", Skill: "recall", CarryTotal: 1} - if axis != "" { - s.Scenario = base + "~" + axis - } - if pass { - s.CarryHit = 1 - } - return s -} - -// The headline has to stay comparable with every run made before variants -// existed; averaging in the variants would move it without the system changing. -func TestTheHeadlineStaysOnTheScenariosAsWritten(t *testing.T) { - r := []Result{{Adapter: "logos", Scores: []Score{ - scored("a", "", "", true), - scored("a", AxisWording, `"reworded"`, false), - scored("a", AxisDistractors, "distractor set 1", false), - }}} - out := Report(r, false) - overall := out[strings.Index(out, "── overall"):strings.Index(out, "pass = ")] - if !strings.Contains(overall, "100.0%") { - t.Errorf("the overall table did not report the case as written:\n%s", overall) - } -} - -// The point of the variants is the scenario that passes one way and fails -// another; a rate alone would hide which one and how it was asked. -func TestAFlakyScenarioIsNamedWithTheVariantsItFailed(t *testing.T) { - r := []Result{{Adapter: "logos", Scores: []Score{ - scored("steady", "", "", true), - scored("steady", AxisWording, `"x"`, true), - scored("flaky", "", "", true), - scored("flaky", AxisWording, `"where did we leave off?"`, false), - scored("flaky", AxisDistractors, "distractor set 2", true), - }}} - out := Report(r, false) - if !strings.Contains(out, "flaky") || !strings.Contains(out, "2/3") || - !strings.Contains(out, `reworded "where did we leave off?"`) { - t.Errorf("the flaky case was not named with its failing variant:\n%s", out) - } - if strings.Contains(out[strings.Index(out, "── across variants"):], "steady ") { - t.Errorf("a case with one outcome on every variant was listed as flaky:\n%s", out) - } -} diff --git a/internal/memory/bench.go b/internal/memory/bench.go index 387d669..0dd5463 100644 --- a/internal/memory/bench.go +++ b/internal/memory/bench.go @@ -144,6 +144,28 @@ func RunLongMemEval(p *provider.Provider, embedModel, path string, limit int, hy // evidenceRank returns the best (lowest) rank at which any evidence session // appears in the retrieval ordering, or -1 if none within the top `depth`. func evidenceRank(p *provider.Provider, embedModel string, in lmeInstance, hybrid bool, depth int) (int, error) { + ordered, _, err := retrieveTopK(p, embedModel, in, hybrid, depth) + if err != nil { + return -1, err + } + + evidence := map[string]bool{} + for _, e := range in.AnswerSessionIDs { + evidence[normalizeSessionID(e)] = true + } + for rank, id := range ordered { + if evidence[normalizeSessionID(id)] { + return rank, nil + } + } + return -1, nil +} + +// retrieveTopK runs the same ranking evidenceRank scores against, returning +// the ordered session ids alongside their full text so a QA pass can build an +// answer prompt from exactly what retrieval surfaced — not from the evidence +// sessions LongMemEval labels as correct, which the memory backend never sees. +func retrieveTopK(p *provider.Provider, embedModel string, in lmeInstance, hybrid bool, depth int) ([]string, map[string]string, error) { texts := make([]string, 0, len(in.HaystackSessions)) ids := make([]string, 0, len(in.HaystackSessions)) for si, sess := range in.HaystackSessions { @@ -164,11 +186,16 @@ func evidenceRank(p *provider.Provider, embedModel string, in lmeInstance, hybri sessVecs, err := p.Embed(embedModel, truncateAll(texts, 2000)) if err != nil { - return -1, err + return nil, nil, err } qVec, err := p.Embed(embedModel, []string{in.Question}) if err != nil || len(qVec) == 0 { - return -1, err + return nil, nil, err + } + + byID := make(map[string]string, len(ids)) + for i, id := range ids { + byID[id] = texts[i] } var ordered []string @@ -192,17 +219,7 @@ func evidenceRank(p *provider.Provider, embedModel string, in lmeInstance, hybri ordered = append(ordered, r[i].id) } } - - evidence := map[string]bool{} - for _, e := range in.AnswerSessionIDs { - evidence[normalizeSessionID(e)] = true - } - for rank, id := range ordered { - if evidence[normalizeSessionID(id)] { - return rank, nil - } - } - return -1, nil + return ordered, byID, nil } // normalizeSessionID reduces the two id conventions LongMemEval uses (a plain @@ -212,6 +229,185 @@ func normalizeSessionID(id string) string { return id } +// QAResult is end-to-end accuracy for one LongMemEval category: a generated +// answer, graded against the gold answer, after retrieval — as opposed to +// BenchResult, which only asks whether the right session was surfaced. A +// system can have perfect recall and a wrong answer; this is the number that +// is actually comparable to another memory system's published LongMemEval +// score, because every competitor reports QA accuracy, not recall@k. +type QAResult struct { + Category string + N int + Correct int +} + +// Accuracy returns the fraction of this category's questions answered +// correctly. +func (r QAResult) Accuracy() float64 { + if r.N == 0 { + return 0 + } + return float64(r.Correct) / float64(r.N) +} + +// RunLongMemEvalQA runs the full pipeline LongMemEval actually measures: +// retrieve, generate an answer from what was retrieved, grade it against the +// gold answer. genModel answers the question from retrieved context alone; +// judgeModel (often the same model) grades the generated answer against the +// gold one. Both run through the local router, per this project's no-cloud-call +// rule — there is no hosted grading step hiding in here. +func RunLongMemEvalQA(p *provider.Provider, embedModel, genModel, judgeModel, path string, limit int, hybrid bool, depth int, progress func(done, total int)) ([]QAResult, error) { + raw, err := os.ReadFile(path) + if err != nil { + return nil, err + } + var instances []lmeInstance + if err := json.Unmarshal(raw, &instances); err != nil { + return nil, fmt.Errorf("parsing %s: %w", path, err) + } + + // Abstention questions are scored on their own: a correct answer to one is + // refusing to answer, which judgeAnswer treats as a distinct outcome rather + // than grading it against a gold string that was never meant to be found. + instances = sampleInstances(instances, limit) + + byCat := map[string]*QAResult{} + overall := &QAResult{Category: "OVERALL"} + + for i, in := range instances { + ordered, byID, err := retrieveTopK(p, embedModel, in, hybrid, depth) + if err != nil { + return nil, err + } + var ctx strings.Builder + for _, id := range ordered { + ctx.WriteString(byID[id]) + ctx.WriteString("\n---\n") + } + + answer, err := p.Chat(genModel, + "Answer the question using only the context given. If the context does not contain the answer, say you don't know — never guess.", + fmt.Sprintf("Context:\n%s\n\nQuestion: %s", ctx.String(), in.Question), + nil, + ) + if err != nil { + return nil, err + } + + correct, err := judgeAnswer(p, judgeModel, in, answer) + if err != nil { + return nil, err + } + + cat := byCat[in.QuestionType] + if cat == nil { + cat = &QAResult{Category: in.QuestionType} + byCat[in.QuestionType] = cat + } + cat.N++ + overall.N++ + if correct { + cat.Correct++ + overall.Correct++ + } + if progress != nil { + progress(i+1, len(instances)) + } + } + + out := []QAResult{*overall} + var cats []string + for c := range byCat { + cats = append(cats, c) + } + sort.Strings(cats) + for _, c := range cats { + out = append(out, *byCat[c]) + } + return out, nil +} + +// judgeSchema constrains the grader to a single boolean field — the same +// trick Chat's doc comment explains: small local models follow a forced JSON +// shape far more reliably than they follow "answer yes or no" in prose. +var judgeSchema = map[string]any{ + "type": "object", + "properties": map[string]any{"correct": map[string]any{"type": "boolean"}}, + "required": []string{"correct"}, + "additionalProperties": false, +} + +// judgeAnswer asks judgeModel whether the generated answer matches the gold +// answer closely enough to count, the same way LongMemEval's own reference +// grader works: semantic match, not exact string equality — "around 3pm" and +// "15:00" are both correct for a question with gold answer "3:00 PM". +// +// An abstention question's gold answer is empty or absent; there, "correct" +// means the model declined rather than fabricated one, so the grading prompt +// is different for that case. +func judgeAnswer(p *provider.Provider, judgeModel string, in lmeInstance, generated string) (bool, error) { + gold := goldAnswerText(in.Answer) + abstention := strings.HasSuffix(in.QuestionType, "_abs") || len(in.AnswerSessionIDs) == 0 + + var user string + if abstention { + user = fmt.Sprintf("Question: %s\n\nThe correct behavior is to say the information is not available. Model's answer: %q\n\nDid the model correctly decline to answer, rather than inventing one?", in.Question, generated) + } else { + user = fmt.Sprintf("Question: %s\nGold answer: %s\nModel's answer: %q\n\nDoes the model's answer match the gold answer in substance? Minor wording or formatting differences do not count against it.", in.Question, gold, generated) + } + + out, err := p.Chat(judgeModel, + "You are grading an answer against a known-correct answer. Respond only with the JSON schema given.", + user, + judgeSchema, + ) + if err != nil { + return false, err + } + var parsed struct { + Correct bool `json:"correct"` + } + if err := json.Unmarshal([]byte(out), &parsed); err != nil { + return false, fmt.Errorf("judge returned non-conforming JSON: %w", err) + } + return parsed.Correct, nil +} + +// goldAnswerText renders LongMemEval's answer field as text regardless of +// whether a given instance's gold answer is a string or a list of strings. +func goldAnswerText(a any) string { + switch v := a.(type) { + case string: + return v + case []any: + parts := make([]string, 0, len(v)) + for _, p := range v { + parts = append(parts, fmt.Sprintf("%v", p)) + } + return strings.Join(parts, "; ") + default: + return fmt.Sprintf("%v", v) + } +} + +// sampleInstances applies RunLongMemEval's stride-sampling (proportional +// category coverage instead of a prefix) without excluding abstention +// questions, since RunLongMemEvalQA — unlike recall scoring — grades them too. +func sampleInstances(instances []lmeInstance, limit int) []lmeInstance { + if limit <= 0 || limit >= len(instances) { + return instances + } + step := len(instances) / limit + if step < 1 { + step = 1 + } + sampled := make([]lmeInstance, 0, limit) + for i := 0; i < len(instances) && len(sampled) < limit; i += step { + sampled = append(sampled, instances[i]) + } + return sampled +} + func truncateAll(s []string, n int) []string { out := make([]string, len(s)) for i, v := range s {