Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
62 changes: 57 additions & 5 deletions cmd/agentloop/main.go
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@ import (
"github.com/FreePeak/agentloop/internal/onegw"
"github.com/FreePeak/agentloop/internal/planner"
"github.com/FreePeak/agentloop/internal/tools"
"github.com/FreePeak/agentloop/internal/xdev"
)

// Server holds the in-memory run store, the runner/gate registry,
Expand All @@ -36,6 +37,9 @@ type Server struct {
tools tools.ToolRegistry
evalRunner *eval.Runner
model *onegw.Client
// sandboxDir is the workspace sandboxed turns run in; empty when no
// executor is configured.
sandboxDir string
}

// NewServer creates a Server with the v1 tool set and empty stores.
Expand All @@ -56,12 +60,14 @@ type Server struct {
// and no model: the M6 deploy gate must stay deterministic and offline, so it
// must not depend on LeanKG or onegw being up (PRD §11.4).
func NewServer() *Server {
reg := tools.NewRegistryWithKnowledge(leankgFromEnv())
sandboxDir := os.Getenv("AGENTLOOP_XDEV_DIR")
reg := tools.NewRegistryWithKnowledge(leankgFromEnv()).WithSandbox(xdevFromEnv(context.Background(), sandboxDir), sandboxDir)
return &Server{
runs: make(map[string]loop.RunResult),
runners: make(map[string]*loop.LoopRunner),
gates: make(map[string]*loop.ApprovalGate),
tools: reg,
runs: make(map[string]loop.RunResult),
runners: make(map[string]*loop.LoopRunner),
gates: make(map[string]*loop.ApprovalGate),
tools: reg,
sandboxDir: sandboxDir,
model: onegw.New(
envOr("AGENTLOOP_ONEGW_URL", "http://127.0.0.1:8080"),
os.Getenv("AGENTLOOP_ONEGW_KEY"),
Expand All @@ -73,6 +79,9 @@ func NewServer() *Server {
// bare runner here made the adversarial case unscoreable — no
// gate means no pause, and a pause is its whole premise.
g := budget.New(cfg.CostBudget, float64(loop.DailyCeilingMult)*cfg.CostBudget)
// No sandbox and no knowledge client: the M6 deploy gate must
// stay deterministic and offline (PRD §11.4), so it must not
// depend on xdev or LeanKG being up.
reg := tools.NewRegistry()
gate := loop.NewApprovalGate()
cfg.Gate = gate
Expand Down Expand Up @@ -103,6 +112,48 @@ func leankgFromEnv() *leankg.Client {
return leankg.New(envOr("AGENTLOOP_LEANKG_URL", "http://127.0.0.1:8090"))
}

// xdevFromEnv starts ONE sandbox child and returns it, or nil when no
// executor is wanted.
//
// AGENTLOOP_XDEV_BIN default "xdev"
// AGENTLOOP_XDEV_DIR the workspace turns run in; default is a fresh
// temp dir, because a sandbox pointed at the
// server's own cwd can edit agentloop itself
// AGENTLOOP_XDEV_OFF any value disables it
//
// One child is shared by every run on purpose: `xdev rpc` keeps a session
// and a model conversation, so a per-run child would pay startup on every
// step and lose the thread between them. It is still one turn at a time —
// the client serialises prompts.
//
// A child that will not start is not fatal: the registry is built without
// a sandbox and `run_tests`/`write_file` report that no executor is
// configured. That is the same shape as LeanKG being down, and it keeps a
// deploy without xdev able to run the read-only paths.
func xdevFromEnv(ctx context.Context, dir string) *xdev.Client {
if os.Getenv("AGENTLOOP_XDEV_OFF") != "" {
return nil
}
if dir == "" {
d, err := os.MkdirTemp("", "agentloop-sandbox-")
if err != nil {
log.Printf("sandbox: no workspace: %v", err)
return nil
}
dir = d
}
c, err := xdev.Start(ctx, xdev.Config{
Command: envOr("AGENTLOOP_XDEV_BIN", "xdev"),
Dir: dir,
})
if err != nil {
log.Printf("sandbox: xdev unavailable, run_tests and write_file will report no executor: %v", err)
return nil
}
log.Printf("sandbox: xdev rpc started in %s", dir)
return c
}

type runRequest struct {
Goal string `json:"goal"`
Context string `json:"context"`
Expand Down Expand Up @@ -139,6 +190,7 @@ func (s *Server) submitRun(w http.ResponseWriter, r *http.Request) {
Goal: body.Goal,
Context: body.Context,
Gate: gate,
SandboxDir: s.sandboxDir,
// M8: the loop can call the portfolio's gateway for synthesis.
// The eval runner deliberately gets no model — the deploy gate
// must stay deterministic and offline.
Expand Down
3 changes: 2 additions & 1 deletion docs/PRD.md
Original file line number Diff line number Diff line change
Expand Up @@ -136,7 +136,7 @@ Rules (Ch.6, `design.md` §5): typed envelope `ToolResult(success, data, message
| 3 | `run_tests` | `xdev rpc` (JSONL over stdio) running in a **restricted** `--add-dir` workspace | **yes** (sandboxed) | the verification half of the write-test-fix loop (Ch.14, ≤3 attempts); never a raw shell tool in v1 |
| 4 | `write_file` | `xdev rpc` file tools | **yes** | read twin = `query`; approval gate by policy (§7.3); idempotency key on every call (§4.2) |

**Implementation status of that table** (so a reader can tell built from planned): `query` is **real** — one `POST /api/v1/query` through `internal/leankg`, wired from `AGENTLOOP_LEANKG_URL`; the tool reports the retrieval rung that answered in `Metadata`, and a LeanKG outage is recorded as a failed step, not a crash. `web_search`, `run_tests` and `write_file` are still stubs and say `stub` in their result message. The M6 eval runner deliberately gets a registry with **no** knowledge client, so the deploy gate stays deterministic and offline.
**Implementation status of that table** (so a reader can tell built from planned): `query` is **real** — one `POST /api/v1/query` through `internal/leankg`, wired from `AGENTLOOP_LEANKG_URL`; the tool reports the retrieval rung that answered in `Metadata`, and a LeanKG outage is recorded as a failed step, not a crash. `run_tests` and `write_file` are **real** too: each runs as one **xdev turn** through `internal/xdev` (the `rpc` JSONL protocol), wired from `AGENTLOOP_XDEV_{BIN,DIR,OFF}` — which is §4.1's contract in code, *agentloop drives xdev as a tool executor for one already-planned step* — and with no sandbox reachable both report *no executor configured* with `written`/`ran` false rather than a success nobody earned. `web_search` is the last stub. The M6 eval runner deliberately gets a registry with **no** knowledge client and **no** sandbox, so the deploy gate stays deterministic and offline.

**Divergence from `design.md` §18, stated on purpose:** `design.md`'s starting set is *"3 reads + 1 search + 1 ticket/incident writer"*. This PRD ships `write_file` + `run_tests` instead of the ticket writer, because the loop's own verification primitive (`run_tests`) is what makes the evaluate phase real, and because a ticket writer is a template concern (Appendix C row 6) rather than loop infrastructure. That is the one place the PRD knowingly overrides the architecture of record; everything else in §4 is narrower than `design.md`, not different from it.

Expand Down Expand Up @@ -730,6 +730,7 @@ Written the way an unfriendly reviewer would write it, then answered. Every find

**Read next.** §13.1 (scope → milestones), §17 (defaults), §18 (where to discount the source), §22 (this document's own weaknesses).

* Last updated: 2026-09-21 (The loop can finally **write and verify**. `internal/xdev` speaks xdev's `rpc` JSONL protocol (ready-frame version gate, event-before-response interleaving, one turn at a time, child killed when the step's budget expires) and `run_tests`/`write_file` run as one xdev turn each, in a sandboxed workspace from `AGENTLOOP_XDEV_DIR`. With no sandbox the two report *no executor configured* and `written`/`ran` stay false — "no sandbox" can never read as "the tests passed". `nextToolDefault` now gives each tool the arguments it needs to be a real call, so a write has a target instead of failing closed on a missing path. `web_search` is the last stub.)
* Last updated: 2026-09-21 (The M6 eval gate was reporting **0.5 — deploy blocked — for days** while `go test ./...` was green. Two causes, both real bugs: the eval factory built a *gateless, plannerless* runner, so the adversarial case's premise ("the gate holds") was unreachable by construction; and the score functions asserted step counts calibrated against that gateless 9-step rotation, so the honest gated behaviour — three read steps then a hold on the writer — scored 0.3. The factory now builds the service's runner (minus the model, as §11.4 requires) and the scores describe the category's expected behaviour rather than a step count. Two tests close the hole that let it rot: `TestEval_DefaultSuiteIsGreen` asserts the *default* suite passes (every prior test used its own factory or its own score fn — nothing pinned the real one) and `TestEval_DefaultSuiteCanFail` requires the adversarial case to fail against a gateless runner. Verified by breaking `Categorize` and watching the endpoint block at 0.75.)
* Last updated: 2026-09-21 (CI exists: `.github/workflows/ci.yml` runs gofmt, `go vet`, `go test`, `golangci-lint` (config pinned in `.golangci.yml`) and `docs/check-prd.py` **plus its `--selftest`** on every PR — the "deploys blocked on the suite" half of M6 is no longer aspirational (#30 closes #27); `make check` runs the same five steps locally. **Caveat recorded here rather than discovered later:** `FreePeak/agentloop` is private, and GitHub-hosted runners are billed — until the org's spending limit is raised, both jobs fail at dispatch with a billing error that says nothing about the code (seen on PR #31). `make check` is the fallback that keeps the gate honest in the meantime. Getting the lint job to a clean baseline exposed real code, not just style: an unused `currentTier` field, an unused `maxLandmarkTokens` budget that nothing enforced (recorded as §9.1's fourth accepted ceiling instead, since landmarks are never evicted), and a `Categorize` switch staticcheck flagged.)
* Last updated: 2026-09-21 (Docs synced to `b492cc2`. §13's M5 row recorded the `Categorize` fix (#25) and its stale test count; the M6 row now says out loud that the "deploys blocked on the full-suite gate" half is **not** enforced — there is no CI in this repo (#27); the §13 task-record line stopped claiming the repo has no commits/remote and now records the tracker state: priority bands P0–P3 defined and every open issue labelled (#28, half done), with issue closure still not PR-linked. `docs/USAGE.md` lost a duplicated line and its "does nothing" summary was split into what reads today vs what still does not. `docs/check-prd.py` stops false-failing on §-refs into other docs — the cause of the long-standing dual-§-ref FAIL, whose refs pointed at `docs/JEV-INTEGRATION.md`, not this PRD.)
Expand Down
22 changes: 16 additions & 6 deletions docs/USAGE.md
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,7 @@ Read this before you plan work around it. As of this writing:
| HITL approval gate wired into the runner | **implemented and working**: the gate holds a write, and approving it **resumes** the run. The read-only three tools never interrupt; `write_file` always holds — §3, §9 |
| HTTP API, admin console, eval harness | **implemented** |
| Model calls to onegw | **partial** — one outbound call, at synthesis (§2, §9). The loop's *planning* still makes none |
| The four built-in tools | **one real, three stubs** — `query` reaches LeanKG and returns its envelope; `web_search`, `run_tests`, `write_file` return canned results whose message says `stub` (`internal/tools/registry_impl.go`) |
| The four built-in tools | **three real, one stub** — `query` reaches LeanKG; `run_tests` and `write_file` each run as one **xdev turn** in a sandboxed workspace (so the loop can actually write and verify); `web_search` still returns a canned result whose message says `stub` |
| The planner | **deterministic**, no model calls; model-driven planning is the documented production path |
| M7 multi-agent (`internal/supervisor`) | **gated shut** by design — refused unless one of [PRD §10](PRD.md#10-multi-agent-stance)'s four conditions is met |

Expand Down Expand Up @@ -102,6 +102,9 @@ Deployment facts, not compiled defaults:
| `AGENTLOOP_ONEGW_COMBO` | `dev` | combo name, used as the wire `model` |
| `AGENTLOOP_LEANKG_URL` | `http://127.0.0.1:8090` | LeanKG REST root behind the `query` tool |
| `AGENTLOOP_LEANKG_OFF` | *(unset)* | any value disables the knowledge client; `query` then says no service is configured |
| `AGENTLOOP_XDEV_BIN` | `xdev` | the sandbox binary; agentloop speaks its `rpc` JSONL protocol |
| `AGENTLOOP_XDEV_DIR` | a fresh temp dir | the workspace `write_file`/`run_tests` turns run in |
| `AGENTLOOP_XDEV_OFF` | *(unset)* | any value disables the sandbox; those two tools then report no executor |

Take the onegw key from `onegw.toml`'s `[auth] [[auth.keys]]`; the combo must
exist there too, since the client sends whatever name you give it and onegw
Expand Down Expand Up @@ -293,11 +296,18 @@ Stated plainly, so nobody discovers it the hard way:
stored state *and* signals the runner's kill channel, which `Run()` checks at
every step boundary. The stop is bounded by one step, not instant — a step
already in flight finishes first.
- **Three of the four tools are stubs.** `query` is **real**: it reaches LeanKG
`POST /api/v1/query` and reports the retrieval rung that answered.
`web_search` has no client, and `run_tests`/`write_file` should go through xdev
rpc in a restricted workspace. Each stub says so in its result message, so a
step that did nothing cannot be read as one that worked.
- **`web_search` is the last stub.** It has no client, and says so in its result
message, so a step that did nothing cannot be read as one that worked.
- **`write_file` and `run_tests` are real, and they need a sandbox.** Each runs
as **one xdev turn** (`internal/xdev` speaks xdev's `rpc` JSONL protocol).
agentloop owns the loop and the bound; xdev owns the turn — the file mutation,
the test run, the model call inside it (PRD §4.3). With no xdev binary
reachable, both tools report *no executor configured* and `Data[written]` /
`Data[ran]` stay false, so "no sandbox" can never be read as "the write
succeeded" or "the tests passed".
- **The turn is bounded by the step, not by xdev.** If a step's budget expires
mid-turn the client kills the child, rather than leaving a half-read stream
and a sandbox still mutating a workspace.
- **Retrieval needs a LeanKG to point at.** `AGENTLOOP_LEANKG_URL` (default
`http://127.0.0.1:8090`); `AGENTLOOP_LEANKG_OFF=1` disables the client, and the
tool then says no knowledge service is configured. A LeanKG that is *down* is
Expand Down
27 changes: 26 additions & 1 deletion internal/loop/runner.go
Original file line number Diff line number Diff line change
Expand Up @@ -77,6 +77,7 @@ type RunnerConfig struct {
ConfidenceFloor float64
EscalationThreshold float64
Gate *ApprovalGate // M5: fail-closed HITL gate (nil = bypass, tests)
SandboxDir string // workspace a sandboxed turn runs in (empty = tool's own default)
Model ModelClient // M8: outbound model transport (nil = deterministic synthesis only)
}

Expand Down Expand Up @@ -638,10 +639,34 @@ func synthesizePartial(steps []StepRecord) string {
// nextToolDefault returns the tool and args for a step. M1 uses a
// deterministic rotation over the 4 v1 tools for testing;
// production replaces this with the planner's output.
// nextToolDefault is the deterministic stand-in for model-driven tool
// selection (PRD §4.3: "agentloop owns the loop, xdev owns the turn").
// It rotates the v1 surface and gives each tool the arguments it needs to
// be a real call rather than a shape:
//
// - the readers get the goal as their query text;
// - run_tests gets the workspace to verify;
// - write_file gets a target path, because a write with no target now
// fails closed — the previous rotation passed only step/goal, so every
// write was rejected before the model inside the sandbox ever saw it.
//
// ponytail: the path is derived from cfg.RunID, not from what the run
// learned. A real planner names the file; this exists so the write path is
// exercisable end to end until that lands.
func nextToolDefault(step int, cfg RunnerConfig) (string, map[string]any) {
tools := []string{"query", "web_search", "run_tests", "write_file"}
name := tools[step%len(tools)]
return name, map[string]any{"step": step, "goal": cfg.Goal}
args := map[string]any{"step": step, "goal": cfg.Goal}
switch name {
case "run_tests":
if cfg.SandboxDir != "" {
args["cwd"] = cfg.SandboxDir
}
case "write_file":
args["path"] = fmt.Sprintf("agentloop-%s-step-%d.txt", cfg.RunID, step)
args["content"] = fmt.Sprintf("step %d of run %s: %s\n", step, cfg.RunID, cfg.Goal)
}
return name, args
}

// dedupKey is the agentloop idempotency fingerprint:
Expand Down
Loading
Loading