From 9f4689d3cd61626df8ee78006784c434cba329fd Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:05:02 +0000 Subject: [PATCH 1/2] feat(PA-8): judge whether this machine can run a local model, and say what is short The two halves that decide whether this feature is honest. The surface that shows a running download is deliberately not here. Measuring first shrank the item: the RUN half already exists. A local runtime is already an allowed provider -- the one trusted OpenAI-compatible package, local URL only -- and the model listing already asks it what it serves. And the probe is not native: the engine is a process on the user's own machine, so memory, cores and arch come straight from the OS. Only the GPU needs a shell out. `local-model-fit.ts` is built on three rules. An unseen GPU is reported as NOT MEASURED, never as absent. Telling a user with a discrete card that their machine has none is a false statement about their own computer, and it is the one they would push back on hardest and be right about. So `undefined` means we did not look, which is a third answer. It never invents a model's size. The figure comes from whoever is offering the model. A fit check against a size this module guessed would be a guess wearing a verdict's clothes. A refusal carries the numbers that produced it, and distinguishes the two cases that matter: short by an amount this machine has if something is closed, versus short by more than it has in total. "Your machine cannot run this" is not actionable and cannot be checked by the person it is about. Memory is checked against AVAILABLE rather than total, and `os.freemem()` understates on Linux because it counts reclaimable page cache as used. That direction is deliberate: it can refuse a model that would in fact have squeezed in, and it will not promise one that gets killed mid-generation. The opposite error loses the user's work. The runtime's own listing and fetch shapes are in `local-model-pull.ts`, because the OpenAI-compatible `GET /models` the provider listing uses reports an id and nothing else -- no size -- so it cannot answer the question this feature turns on. We do not fetch weights ourselves: no mirror choice, no quantisation choice, no digest we chose to trust. The runtime already resumes interrupted downloads and shares progress between concurrent callers; re-implementing that would mean owning model provenance, which is a much larger promise than running a model locally. Shapes come from the runtime's published API reference rather than from reading our own code back, and two traps it states explicitly have a test each: `completed` may be absent while a layer is downloading, and `total` is per-LAYER not per-model -- a caller reading it as whole-model progress shows a bar that resets on every layer. An in-band `error` frame arrives after a 200 is already sent and is terminal, not progress; treated as progress the pull looks like it is still running and never finishes. There is also a test asserting that a whole-body JSON parse FAILS on the real stream, so nobody simplifies the line-by-line reader back into one parse. core 1034 tests 0 fail, 36 new. The fit judgement is pure and needs no machine. The GPU parsers run against the reference's own recorded output, so they are verified on a host with no GPU. No cloud fallback exists here and none should be added. A user who asked for a local model asked for the data to stay on the machine; answering from a hosted model would defeat the only reason to want this, and would do it invisibly. --- .../core/src/config/plugin/local-gpu-probe.ts | 98 +++++++++++ .../core/src/config/plugin/local-model-fit.ts | 143 ++++++++++++++++ .../src/config/plugin/local-model-pull.ts | 156 ++++++++++++++++++ .../core/test/config/local-gpu-probe.test.ts | 88 ++++++++++ .../core/test/config/local-model-fit.test.ts | 129 +++++++++++++++ .../core/test/config/local-model-pull.test.ts | 153 +++++++++++++++++ 6 files changed, 767 insertions(+) create mode 100644 packages/core/src/config/plugin/local-gpu-probe.ts create mode 100644 packages/core/src/config/plugin/local-model-fit.ts create mode 100644 packages/core/src/config/plugin/local-model-pull.ts create mode 100644 packages/core/test/config/local-gpu-probe.test.ts create mode 100644 packages/core/test/config/local-model-fit.test.ts create mode 100644 packages/core/test/config/local-model-pull.test.ts diff --git a/packages/core/src/config/plugin/local-gpu-probe.ts b/packages/core/src/config/plugin/local-gpu-probe.ts new file mode 100644 index 0000000000..024c919080 --- /dev/null +++ b/packages/core/src/config/plugin/local-gpu-probe.ts @@ -0,0 +1,98 @@ +export * as LocalGpuProbe from "./local-gpu-probe" + +/** + * Reading the GPU, where the platform lets us. + * + * Split from `local-model-fit` on purpose. The fit judgement is pure and testable with no + * machine at all; this part shells out, and what it can learn differs per platform and per + * driver. Keeping them apart means the verdict logic is never untestable because the probe + * is. + * + * THE ONLY RULE THAT MATTERS HERE: a probe that cannot see a GPU returns `undefined`, which + * the fit module reads as NOT MEASURED. It never returns "no GPU". Telling a user with a + * discrete card that their machine has none is a false statement about their own computer, + * and it is the statement they would push back on hardest and be right about. + */ + +import { Effect } from "effect" +import { ChildProcess } from "effect/unstable/process" +import { AppProcess, requireSuccess } from "../../process" + +/** What a probe can learn. `vramBytes` is absent when the platform reports a name but no size. */ +export type Gpu = { readonly name: string; readonly vramBytes?: number | undefined } + +/** Short, because this runs on a path a user is waiting on and a wedged driver tool must not hold it. */ +const PROBE_TIMEOUT = "3 seconds" + +const MiB = 1024 * 1024 + +/** + * The per-platform command and how to read it. Exported so the parsers are testable against + * recorded tool output without a GPU, a driver, or that platform. + * + * Each is a listing command with machine-readable output, chosen over the prettier + * human-facing variants so the parse does not depend on column widths. + */ +export const PROBES: ReadonlyArray<{ + readonly platforms: ReadonlyArray + readonly command: string + readonly args: ReadonlyArray + readonly parse: (stdout: string) => Gpu | undefined +}> = [ + { + // NVIDIA, any platform it is installed on. CSV without units keeps the parse trivial. + platforms: ["linux", "win32", "darwin"], + command: "nvidia-smi", + args: ["--query-gpu=name,memory.total", "--format=csv,noheader,nounits"], + parse: (stdout) => { + const line = stdout.split("\n").find((candidate) => candidate.trim().length > 0) + if (line === undefined) return undefined + const [name, memory] = line.split(",").map((field) => field.trim()) + if (name === undefined || name.length === 0) return undefined + const megabytes = Number(memory) + return { name, vramBytes: Number.isFinite(megabytes) && megabytes > 0 ? megabytes * MiB : undefined } + }, + }, + { + // Apple silicon: the GPU shares system memory, so there is a name and no separate VRAM. + // Reporting a number here would be wrong in the direction that matters -- it would be + // counted twice against the same bytes the memory check already counted. + platforms: ["darwin"], + command: "sysctl", + args: ["-n", "machdep.cpu.brand_string"], + parse: (stdout) => { + const brand = stdout.trim() + if (!brand.startsWith("Apple ")) return undefined + return { name: `${brand} (unified memory)`, vramBytes: undefined } + }, + }, +] + +/** + * Try each probe for this platform, first answer wins, every failure is silent. + * + * Silent is right: on a machine with no discrete GPU, `nvidia-smi` not existing is the + * normal state of the world, not a condition to report. The caller gets `undefined` and + * says "not measured", which is the true statement. + * + * Runs through `AppProcess.run`, the service the rest of the engine shells out with, rather + * than driving the spawner directly -- output collection, truncation and timeout handling + * already live there and a second copy of them would drift. + */ +export const probe = Effect.fn("LocalGpuProbe.probe")(function* (platform: string) { + const process = yield* AppProcess.Service + for (const candidate of PROBES) { + if (!candidate.platforms.includes(platform)) continue + const result = yield* Effect.gen(function* () { + const run = yield* process + .run(ChildProcess.make(candidate.command, [...candidate.args], { stdin: "ignore", extendEnv: true })) + .pipe(Effect.flatMap(requireSuccess)) + return candidate.parse(run.stdout.toString("utf8")) + }).pipe( + Effect.timeout(PROBE_TIMEOUT), + Effect.catchCause(() => Effect.succeed(undefined)), + ) + if (result !== undefined) return result + } + return undefined +}) diff --git a/packages/core/src/config/plugin/local-model-fit.ts b/packages/core/src/config/plugin/local-model-fit.ts new file mode 100644 index 0000000000..3b8f40cae6 --- /dev/null +++ b/packages/core/src/config/plugin/local-model-fit.ts @@ -0,0 +1,143 @@ +export * as LocalModelFit from "./local-model-fit" + +/** + * Whether this machine can run a local model, and if not, exactly what is short. + * + * PA-8's decisive half. Running a model locally is not a preference, it is a hardware + * question, and the only dishonest answer is a confident one. So this module is built around + * three rules: + * + * 1. It reports MEASURED facts, never estimates dressed as facts. Total and available + * memory and the CPU come from the OS. A GPU is read where the platform exposes it and + * reported as UNKNOWN where it does not -- `unknown` is a third answer, distinct from + * "no GPU", because a machine with a GPU we cannot see must not be told it has none. + * 2. It never invents a model's size. The required figure comes from whoever is offering + * the model -- the runtime's own registry -- and is passed in. A fit check against a + * size this module guessed would be a guess wearing a verdict's clothes. + * 3. A refusal names the constraint and the shortfall. "Your machine cannot run this" is + * not actionable; "needs 8.0 GiB of available memory, this machine has 3.2 GiB free of + * 15.5 GiB total" tells the user whether closing something would fix it. + * + * There is deliberately NO cloud fallback here, silently or otherwise. A user who asked for + * a local model asked for the data to stay on the machine; quietly answering from a hosted + * model would defeat the only reason to want this, and would do it invisibly. + */ + +import * as os from "node:os" + +/** What the machine is, as measured. Every field is read, none is inferred. */ +export type Hardware = { + readonly platform: string + readonly arch: string + readonly cores: number + readonly totalBytes: number + readonly availableBytes: number + /** + * The GPU, when this platform lets us see it. + * + * `undefined` means NOT MEASURED, not absent. The difference matters: a refusal that + * claims a machine has no GPU when we simply could not look is a false statement about + * the user's own computer. + */ + readonly gpu?: { readonly name: string; readonly vramBytes?: number | undefined } | undefined +} + +/** The verdict. A refusal always carries the numbers that produced it. */ +export type Fit = + | { readonly kind: "fits"; readonly hardware: Hardware } + | { + readonly kind: "short" + readonly hardware: Hardware + /** Which constraint failed, for a caller that wants to branch rather than print. */ + readonly constraint: "memory" | "arch" + readonly requiredBytes: number + readonly shortfallBytes: number + readonly reason: string + } + +/** + * Headroom beyond the weights themselves. + * + * A model's file size is not its running size: the runtime holds the KV cache, the context + * window and its own working set alongside the weights. 20% with a 1 GiB floor is a + * deliberately rough allowance, and it is named here rather than buried so that a caller + * reading a refusal can see it is part of the requirement. + * + * Rough is correct for this. The alternative is a precise-looking formula over quantisation, + * context length and runtime, which would be wrong in a way that reads as authoritative. + */ +export const HEADROOM_FRACTION = 0.2 +export const HEADROOM_FLOOR_BYTES = 1024 ** 3 + +export const requirementFor = (modelBytes: number): number => + modelBytes + Math.max(Math.round(modelBytes * HEADROOM_FRACTION), HEADROOM_FLOOR_BYTES) + +const GiB = 1024 ** 3 +const gib = (bytes: number): string => `${(bytes / GiB).toFixed(1)} GiB` + +/** + * Architectures a local runtime has builds for. + * + * Checked because the failure is otherwise a confusing download: the weights arrive, the + * runtime will not start, and nothing in that sequence mentions the CPU. + */ +const SUPPORTED_ARCH = new Set(["x64", "arm64"]) + +/** + * Measure the machine. + * + * `availableBytes` uses `os.freemem()`, which on Linux counts reclaimable page cache as used + * and therefore UNDERSTATES what a process could actually get. That direction is the safe + * one: it can refuse a model that would in fact have squeezed in, and it will not promise + * one that would have been killed mid-generation. The opposite error loses the user's work. + */ +export const measure = (gpu?: Hardware["gpu"]): Hardware => ({ + platform: os.platform(), + arch: os.arch(), + cores: os.cpus().length, + totalBytes: os.totalmem(), + availableBytes: os.freemem(), + gpu, +}) + +/** + * Can this machine run a model of this size? + * + * Memory is checked against AVAILABLE rather than total, because a model that needs more + * than is free today does not run today, however large the machine is. The refusal reports + * both numbers so the user can tell "buy a bigger machine" from "close your other windows". + */ +export const fit = (input: { readonly modelBytes: number; readonly hardware: Hardware }): Fit => { + const { hardware } = input + if (!SUPPORTED_ARCH.has(hardware.arch)) { + return { + kind: "short", + hardware, + constraint: "arch", + requiredBytes: 0, + shortfallBytes: 0, + reason: `local models need an x64 or arm64 CPU; this machine reports ${hardware.arch}`, + } + } + + const requiredBytes = requirementFor(input.modelBytes) + if (hardware.availableBytes >= requiredBytes) return { kind: "fits", hardware } + + const shortfallBytes = requiredBytes - hardware.availableBytes + const fitsIfFreed = hardware.totalBytes >= requiredBytes + return { + kind: "short", + hardware, + constraint: "memory", + requiredBytes, + shortfallBytes, + reason: [ + `this model needs about ${gib(requiredBytes)} of memory`, + `(${gib(input.modelBytes)} of weights plus runtime headroom)`, + `and ${gib(hardware.availableBytes)} is free of ${gib(hardware.totalBytes)} total`, + fitsIfFreed + ? `-- short by ${gib(shortfallBytes)}, which this machine has if you close something` + : `-- short by ${gib(shortfallBytes)}, more than this machine has in total`, + ].join(" "), + } +} diff --git a/packages/core/src/config/plugin/local-model-pull.ts b/packages/core/src/config/plugin/local-model-pull.ts new file mode 100644 index 0000000000..b19883be0a --- /dev/null +++ b/packages/core/src/config/plugin/local-model-pull.ts @@ -0,0 +1,156 @@ +export * as LocalModelPull from "./local-model-pull" + +/** + * Listing and fetching models from a local runtime. + * + * PA-8's other half. The engine already runs a local model once one is present -- a local + * provider speaks the OpenAI wire protocol and `local-models.ts` lists what it serves. What + * was missing is everything before that: what is on disk, how big, and getting one. + * + * WHY THIS IS A SECOND SET OF ROUTES. The OpenAI-compatible `GET /models` the provider + * listing uses reports an id and nothing else -- no size. So it cannot answer the one + * question PA-8 turns on, which is whether this machine can run the thing. Sizes and the + * fetch itself live on the runtime's own API, so that is what this module speaks, and it is + * gated by the same local-endpoint rule. + * + * WE DO NOT FETCH WEIGHTS OURSELVES. No picking a mirror, no choosing a quantisation, no + * verifying a digest we chose to trust. The runtime already does all of it, resumes an + * interrupted download, and shares progress between concurrent callers -- its own docs say + * so. Re-implementing that would mean owning model provenance, which is a far larger + * promise than "run a model locally". + * + * Shapes here come from the runtime's published API reference, not from reading our own + * code back: `GET /api/tags` returns `models[].{name,size,digest,details}`, and + * `POST /api/pull` streams `{status}` objects that carry `{digest,total,completed}` while + * layers transfer. Two traps are documented there and both are handled below. + */ + +import { Effect, Schema } from "effect" +import { HttpClient, HttpClientRequest } from "effect/unstable/http" + +import { isLocalEndpoint } from "./local-provider" + +/** Long, because a pull is a multi-gigabyte transfer and the caller is watching progress. */ +const PULL_TIMEOUT = "6 hours" + +/** Short, because listing what is already on disk is a local call that either answers or is down. */ +const LIST_TIMEOUT = "2 seconds" + +/** + * One installed model. `size` is the field the whole fit check depends on. + * + * Decoded ONE AT A TIME by the caller, for the reason `local-models.ts` states: one entry + * this build cannot describe should cost that entry, not the list. + */ +export const Installed = Schema.Struct({ + name: Schema.String, + size: Schema.Finite, + digest: Schema.optional(Schema.String), +}) +export type Installed = Schema.Schema.Type + +const TagList = Schema.Struct({ models: Schema.Array(Schema.Unknown) }) + +/** + * One progress frame from a pull. + * + * `total` and `completed` are BOTH optional, and that is not defensive typing -- the + * runtime's own docs say so. Status-only frames (`pulling manifest`, `verifying sha256 + * digest`, `writing manifest`, `success`) carry neither, and a downloading frame may carry + * `total` without `completed` until the first bytes land. + */ +export const PullFrame = Schema.Struct({ + status: Schema.String, + digest: Schema.optional(Schema.String), + total: Schema.optional(Schema.Finite), + completed: Schema.optional(Schema.Finite), + error: Schema.optional(Schema.String), +}) +export type PullFrame = Schema.Schema.Type + +/** Progress as a caller should show it. */ +export type Progress = { + /** The runtime's own status line, passed through rather than re-worded. */ + readonly status: string + /** Bytes transferred and expected FOR THE CURRENT LAYER, when the frame carries them. */ + readonly layerCompletedBytes?: number | undefined + readonly layerTotalBytes?: number | undefined + readonly done: boolean +} + +export const tagsUrl = (base: string): string => `${base.replace(/\/+$/, "")}/api/tags` +export const pullUrl = (base: string): string => `${base.replace(/\/+$/, "")}/api/pull` + +/** + * Read one progress frame. + * + * A frame with an `error` is a failure the runtime is reporting in-band, mid-stream, with a + * 200 already sent. Treated as terminal rather than as progress, because the alternative is + * a pull that looks like it is still going and never finishes. + */ +export const readFrame = (frame: PullFrame): Progress | { readonly error: string } => { + if (frame.error !== undefined && frame.error.length > 0) return { error: frame.error } + return { + status: frame.status, + layerCompletedBytes: frame.completed, + layerTotalBytes: frame.total, + done: frame.status === "success", + } +} + +/** + * Decode a newline-delimited stream body into frames, dropping only what fails. + * + * The runtime emits one JSON object per line and does not wrap them in an array, so a + * whole-body parse never succeeds. A line that does not decode is skipped: these frames are + * progress, and one unreadable frame must not abort a transfer that is working. + */ +export const readFrames = (body: string): ReadonlyArray => { + const frames: PullFrame[] = [] + for (const line of body.split("\n")) { + const trimmed = line.trim() + if (trimmed.length === 0) continue + try { + frames.push(Schema.decodeUnknownSync(PullFrame)(JSON.parse(trimmed))) + } catch { + continue + } + } + return frames +} + +/** + * What is installed on this runtime, with sizes. + * + * Best-effort for the same reason the provider listing is: a local runtime that is not + * running is the normal state of a laptop, not an error. A failed call returns an empty + * list, and the caller distinguishes "nothing installed" from "no runtime" by asking + * whether the runtime answered at all -- which is why `reachable` is returned separately + * rather than inferred from an empty array. + */ +export const installed = Effect.fn("LocalModelPull.installed")(function* (base: string) { + // Re-checked here rather than trusted from the caller, the same way `local-models.ts` + // re-checks it: two independent checks of one rule means a later change that loosens one + // cannot quietly turn this into a way to make the engine call an arbitrary host. + if (!isLocalEndpoint(base)) return { reachable: false, models: [] as ReadonlyArray } + + const client = yield* HttpClient.HttpClient + const result = yield* client + .execute(HttpClientRequest.get(tagsUrl(base))) + .pipe( + Effect.flatMap((response) => response.json), + Effect.timeout(LIST_TIMEOUT), + Effect.catchCause(() => Effect.succeed(undefined)), + ) + if (result === undefined) return { reachable: false, models: [] as ReadonlyArray } + + const list = Schema.decodeUnknownOption(TagList)(result) + if (list._tag === "None") return { reachable: true, models: [] as ReadonlyArray } + + const models: Installed[] = [] + for (const entry of list.value.models) { + const decoded = Schema.decodeUnknownOption(Installed)(entry) + if (decoded._tag === "Some") models.push(decoded.value) + } + return { reachable: true, models } +}) diff --git a/packages/core/test/config/local-gpu-probe.test.ts b/packages/core/test/config/local-gpu-probe.test.ts new file mode 100644 index 0000000000..d02534dd7a --- /dev/null +++ b/packages/core/test/config/local-gpu-probe.test.ts @@ -0,0 +1,88 @@ +/** + * PA-8's GPU parsers, against recorded tool output. + * + * Parsed, not mocked: each fixture is the shape the real command prints, with the flags the + * probe passes. A parser verified only against a string I also wrote would prove the two + * agree and nothing about the tool. + * + * The failure these guard is silent. A parser that returns `undefined` for output it should + * have read makes the machine look like one with no GPU, which is a false claim this module + * exists specifically to avoid. + */ +import { describe, expect, test } from "bun:test" +import { LocalGpuProbe } from "@redrob-code/core/config/plugin/local-gpu-probe" + +const MiB = 1024 * 1024 + +const probeFor = (command: string) => { + const found = LocalGpuProbe.PROBES.find((candidate) => candidate.command === command) + if (found === undefined) throw new Error(`no probe spawns ${command}`) + return found +} + +describe("nvidia-smi", () => { + const nvidia = probeFor("nvidia-smi") + + test("passes the flags that make the output machine-readable", () => { + // `noheader,nounits` is what lets the parse be a comma split instead of a column guess, + // and `nounits` is why the memory field is read as plain MiB. + expect(nvidia.args.join(" ")).toContain("--format=csv,noheader,nounits") + expect(nvidia.args.join(" ")).toContain("memory.total") + }) + + test("reads name and VRAM from one card", () => { + expect(nvidia.parse("NVIDIA GeForce RTX 4090, 24564\n")).toEqual({ + name: "NVIDIA GeForce RTX 4090", + vramBytes: 24564 * MiB, + }) + }) + + test("takes the first card when several are listed", () => { + const output = "NVIDIA A100-SXM4-40GB, 40960\nNVIDIA A100-SXM4-40GB, 40960\n" + expect(nvidia.parse(output)?.name).toBe("NVIDIA A100-SXM4-40GB") + }) + + test("keeps the name when the memory field is unreadable", () => { + // A driver that reports `[N/A]` for memory still tells us the card exists, and a card we + // know about with a size we do not is better than pretending there is no card. + const parsed = nvidia.parse("NVIDIA GeForce GTX 1060, [N/A]\n") + expect(parsed?.name).toBe("NVIDIA GeForce GTX 1060") + expect(parsed?.vramBytes).toBeUndefined() + }) + + test("returns undefined for empty output rather than an empty name", () => { + expect(nvidia.parse("")).toBeUndefined() + expect(nvidia.parse("\n\n")).toBeUndefined() + }) +}) + +describe("apple silicon", () => { + const sysctl = probeFor("sysctl") + + test("names the chip and reports no separate VRAM", () => { + // Unified memory: the GPU uses the same bytes the memory check already counted, so a + // VRAM number here would be those bytes counted twice. + expect(sysctl.parse("Apple M3 Max\n")).toEqual({ name: "Apple M3 Max (unified memory)", vramBytes: undefined }) + }) + + test("declines an Intel Mac rather than calling its CPU a GPU", () => { + expect(sysctl.parse("Intel(R) Core(TM) i9-9980HK CPU @ 2.40GHz\n")).toBeUndefined() + }) + + test("is offered only on darwin", () => { + expect(sysctl.platforms).toEqual(["darwin"]) + }) +}) + +describe("probe ordering", () => { + test("nvidia is tried before the platform-specific fallback on darwin", () => { + // An external or eGPU NVIDIA card is the more specific answer, and the sysctl probe + // would otherwise claim unified memory on a machine that has a discrete card. + const darwin = LocalGpuProbe.PROBES.filter((candidate) => candidate.platforms.includes("darwin")) + expect(darwin[0]?.command).toBe("nvidia-smi") + }) + + test("every probe declares the platforms it applies to", () => { + for (const candidate of LocalGpuProbe.PROBES) expect(candidate.platforms.length).toBeGreaterThan(0) + }) +}) diff --git a/packages/core/test/config/local-model-fit.test.ts b/packages/core/test/config/local-model-fit.test.ts new file mode 100644 index 0000000000..136f9b92f4 --- /dev/null +++ b/packages/core/test/config/local-model-fit.test.ts @@ -0,0 +1,129 @@ +/** + * PA-8's fit judgement. + * + * Every case here is a claim the user could push back on, so each asserts the NUMBERS in the + * refusal rather than that a refusal happened. A verdict with no numbers is unfalsifiable by + * the person it is about, which is the opposite of honest. + */ +import { describe, expect, test } from "bun:test" +import { LocalModelFit } from "@redrob-code/core/config/plugin/local-model-fit" + +const GiB = 1024 ** 3 + +const machine = (overrides: Partial = {}): LocalModelFit.Hardware => ({ + platform: "linux", + arch: "x64", + cores: 8, + totalBytes: 16 * GiB, + availableBytes: 12 * GiB, + ...overrides, +}) + +describe("LocalModelFit.requirementFor", () => { + test("adds proportional headroom above the floor", () => { + // 10 GiB of weights: 20% is 2 GiB, which clears the 1 GiB floor, so the floor does not apply. + expect(LocalModelFit.requirementFor(10 * GiB)).toBe(12 * GiB) + }) + + test("applies the floor when the proportion is smaller than it", () => { + // 2 GiB of weights: 20% is 0.4 GiB, under the floor, so the floor is what is added. + expect(LocalModelFit.requirementFor(2 * GiB)).toBe(3 * GiB) + }) + + test("the headroom is never zero, so a model is never sized at exactly its weights", () => { + expect(LocalModelFit.requirementFor(0)).toBe(LocalModelFit.HEADROOM_FLOOR_BYTES) + }) +}) + +describe("LocalModelFit.fit", () => { + test("fits when available memory covers weights plus headroom", () => { + const verdict = LocalModelFit.fit({ modelBytes: 4 * GiB, hardware: machine() }) + expect(verdict.kind).toBe("fits") + }) + + test("is short by the exact difference, and says closing something would do it", () => { + // Needs 12 GiB (10 + 2 headroom); 8 free of 16 total, so freeing 4 GiB is enough. + const verdict = LocalModelFit.fit({ + modelBytes: 10 * GiB, + hardware: machine({ availableBytes: 8 * GiB, totalBytes: 16 * GiB }), + }) + if (verdict.kind !== "short") throw new Error("expected a refusal") + expect(verdict.constraint).toBe("memory") + expect(verdict.requiredBytes).toBe(12 * GiB) + expect(verdict.shortfallBytes).toBe(4 * GiB) + expect(verdict.reason).toContain("12.0 GiB") + expect(verdict.reason).toContain("8.0 GiB is free of 16.0 GiB total") + expect(verdict.reason).toContain("if you close something") + }) + + test("says the machine is too small when freeing everything would not do it", () => { + // Needs 12 GiB and the machine HAS 8 GiB. No amount of closing windows fixes this, and + // telling the user to try would waste their time. + const verdict = LocalModelFit.fit({ + modelBytes: 10 * GiB, + hardware: machine({ availableBytes: 6 * GiB, totalBytes: 8 * GiB }), + }) + if (verdict.kind !== "short") throw new Error("expected a refusal") + expect(verdict.reason).toContain("more than this machine has in total") + expect(verdict.reason).not.toContain("if you close something") + }) + + test("checks available rather than total, so a busy large machine is refused", () => { + // 64 GiB machine with 2 GiB free. Checking total would promise a model that gets killed + // partway through generating, which loses the user's work and explains nothing. + const verdict = LocalModelFit.fit({ + modelBytes: 10 * GiB, + hardware: machine({ availableBytes: 2 * GiB, totalBytes: 64 * GiB }), + }) + expect(verdict.kind).toBe("short") + }) + + test("refuses an unsupported CPU before talking about memory at all", () => { + // A machine with plenty of memory and no runtime build. Reporting a memory shortfall + // here would send the user to close applications for a problem that is not memory. + const verdict = LocalModelFit.fit({ + modelBytes: 1 * GiB, + hardware: machine({ arch: "ppc64", availableBytes: 60 * GiB, totalBytes: 64 * GiB }), + }) + if (verdict.kind !== "short") throw new Error("expected a refusal") + expect(verdict.constraint).toBe("arch") + expect(verdict.reason).toContain("ppc64") + }) + + test("an unseen GPU is reported as unmeasured, never as absent", () => { + const verdict = LocalModelFit.fit({ modelBytes: 1 * GiB, hardware: machine({ gpu: undefined }) }) + expect(verdict.hardware.gpu).toBeUndefined() + // The refusal text must never make a claim about a GPU we did not look at. + const refused = LocalModelFit.fit({ + modelBytes: 100 * GiB, + hardware: machine({ gpu: undefined }), + }) + if (refused.kind !== "short") throw new Error("expected a refusal") + expect(refused.reason.toLowerCase()).not.toContain("gpu") + expect(refused.reason.toLowerCase()).not.toContain("graphics") + }) + + test("carries a measured GPU through untouched", () => { + const gpu = { name: "Test GPU", vramBytes: 8 * GiB } + const verdict = LocalModelFit.fit({ modelBytes: 1 * GiB, hardware: machine({ gpu }) }) + expect(verdict.hardware.gpu).toEqual(gpu) + }) +}) + +describe("LocalModelFit.measure", () => { + test("reports this machine's real numbers, and no GPU claim without a probe", () => { + const hardware = LocalModelFit.measure() + expect(hardware.totalBytes).toBeGreaterThan(0) + expect(hardware.availableBytes).toBeGreaterThan(0) + expect(hardware.availableBytes).toBeLessThanOrEqual(hardware.totalBytes) + expect(hardware.cores).toBeGreaterThan(0) + expect(hardware.gpu).toBeUndefined() + }) + + test("takes the GPU from its caller rather than probing inside the verdict path", () => { + // The split exists so the pure judgement is testable with no machine. If `measure` ever + // probed on its own, this test would see a GPU it never passed in. + const hardware = LocalModelFit.measure({ name: "Passed In" }) + expect(hardware.gpu?.name).toBe("Passed In") + }) +}) diff --git a/packages/core/test/config/local-model-pull.test.ts b/packages/core/test/config/local-model-pull.test.ts new file mode 100644 index 0000000000..d065c66773 --- /dev/null +++ b/packages/core/test/config/local-model-pull.test.ts @@ -0,0 +1,153 @@ +/** + * PA-8's pull and listing, against the runtime's OWN published examples. + * + * Every fixture below is copied from the runtime's API reference, not written to match this + * code. That distinction is the whole value: a fixture and a parser written from one reading + * of a schema agree with each other and prove nothing about the real producer. + * + * Two traps the reference states explicitly, and both have a test: + * - `completed` MAY BE ABSENT while a layer is downloading ("Until any of the download is + * completed, the `completed` key may not be included"). + * - `total` is per-LAYER, not per-model ("The number of files to be downloaded depends on + * the number of layers specified in the manifest"). + */ +import { describe, expect, test } from "bun:test" +import { Schema } from "effect" +import { LocalModelFit } from "@redrob-code/core/config/plugin/local-model-fit" +import { LocalModelPull } from "@redrob-code/core/config/plugin/local-model-pull" + +describe("urls", () => { + test("join without doubling a slash", () => { + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434")).toBe("http://127.0.0.1:11434/api/tags") + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/")).toBe("http://127.0.0.1:11434/api/tags") + expect(LocalModelPull.pullUrl("http://127.0.0.1:11434///")).toBe("http://127.0.0.1:11434/api/pull") + }) +}) + +describe("readFrames", () => { + // Verbatim from the reference's own pull example, in order. + const stream = [ + '{"status":"pulling manifest"}', + '{"status":"pulling digestname","digest":"digestname","total":2142590208,"completed":241970}', + '{"status":"verifying sha256 digest"}', + '{"status":"writing manifest"}', + '{"status":"removing any unused layers"}', + '{"status":"success"}', + ].join("\n") + + test("reads every frame of the documented stream", () => { + const frames = LocalModelPull.readFrames(stream) + expect(frames.length).toBe(6) + expect(frames[0]?.status).toBe("pulling manifest") + expect(frames[1]?.total).toBe(2142590208) + expect(frames[5]?.status).toBe("success") + }) + + test("status-only frames decode, carrying no byte counts", () => { + const frames = LocalModelPull.readFrames('{"status":"writing manifest"}') + expect(frames[0]?.total).toBeUndefined() + expect(frames[0]?.completed).toBeUndefined() + }) + + test("a downloading frame with no completed key is still a frame", () => { + // The reference says this one happens before the first bytes land. Requiring `completed` + // would drop the frame that announces the size of the download. + const frames = LocalModelPull.readFrames('{"status":"pulling abc","digest":"abc","total":2142590208}') + expect(frames.length).toBe(1) + expect(frames[0]?.total).toBe(2142590208) + expect(frames[0]?.completed).toBeUndefined() + }) + + test("one unreadable line costs that line, not the transfer", () => { + const frames = LocalModelPull.readFrames( + ['{"status":"pulling manifest"}', "{ not json", '{"status":"success"}'].join("\n"), + ) + expect(frames.map((frame) => frame.status)).toEqual(["pulling manifest", "success"]) + }) + + test("a whole-body JSON parse would have failed on this stream", () => { + // Guarding the reason the line-by-line reader exists: the runtime does not wrap the + // frames in an array, so anyone "simplifying" this to one parse breaks every pull. + expect(() => JSON.parse(stream)).toThrow() + }) +}) + +describe("readFrame", () => { + test("marks only the success status as done", () => { + const progress = LocalModelPull.readFrame({ status: "pulling manifest" }) + expect("error" in progress ? undefined : progress.done).toBe(false) + const finished = LocalModelPull.readFrame({ status: "success" }) + expect("error" in finished ? undefined : finished.done).toBe(true) + }) + + test("reports layer bytes as layer bytes", () => { + const progress = LocalModelPull.readFrame({ status: "pulling abc", total: 100, completed: 40 }) + if ("error" in progress) throw new Error("unexpected error frame") + // Named for what they are. A caller that read these as whole-model progress would show a + // bar that jumps back to zero on every layer of a multi-layer model. + expect(progress.layerTotalBytes).toBe(100) + expect(progress.layerCompletedBytes).toBe(40) + }) + + test("an in-band error is terminal, not progress", () => { + // A 200 is already sent by the time this arrives. Treated as progress, the pull would + // look like it is still running and never finish. + const progress = LocalModelPull.readFrame({ status: "error", error: "model not found" }) + expect(progress).toEqual({ error: "model not found" }) + }) + + test("an empty error string is not an error", () => { + const progress = LocalModelPull.readFrame({ status: "success", error: "" }) + expect("error" in progress).toBe(false) + }) + + test("passes the runtime's status through rather than re-wording it", () => { + const progress = LocalModelPull.readFrame({ status: "removing any unused layers" }) + if ("error" in progress) throw new Error("unexpected error frame") + expect(progress.status).toBe("removing any unused layers") + }) +}) + +describe("Installed", () => { + // Verbatim from the reference's `GET /api/tags` example. + const entry = { + name: "deepseek-r1:latest", + model: "deepseek-r1:latest", + modified_at: "2025-05-10T08:06:48.639712648-07:00", + size: 4683075271, + digest: "0a8c266910232fd3291e71e5ba1e058cc5af9d411192cf88b6d30e92b6e73163", + details: { format: "gguf", family: "qwen2", parameter_size: "7.6B", quantization_level: "Q4_K_M" }, + } + + test("decodes through the real schema, keeping the size the fit check needs", () => { + const decoded = Schema.decodeUnknownSync(LocalModelPull.Installed)(entry) + expect(decoded.name).toBe("deepseek-r1:latest") + expect(decoded.size).toBe(4683075271) + expect(decoded.digest).toBe(entry.digest) + }) + + test("the size feeds the fit check as bytes, not as anything else", () => { + // 4.68 GB of weights, so the requirement is that plus headroom -- which is what makes a + // 4 GiB machine the wrong machine for it. The two modules meet exactly here. + const decoded = Schema.decodeUnknownSync(LocalModelPull.Installed)(entry) + const verdict = LocalModelFit.fit({ + modelBytes: decoded.size, + hardware: { + platform: "linux", + arch: "x64", + cores: 8, + totalBytes: 4 * 1024 ** 3, + availableBytes: 3 * 1024 ** 3, + }, + }) + if (verdict.kind !== "short") throw new Error("a 4 GiB machine should not fit a 4.68 GB model") + expect(verdict.constraint).toBe("memory") + expect(verdict.requiredBytes).toBe(LocalModelFit.requirementFor(4683075271)) + }) + + test("an entry with no size is dropped rather than treated as free", () => { + // A model of unknown size that defaulted to zero would pass every fit check and then be + // killed mid-load. Refusing to decode it is the honest outcome. + expect(Schema.decodeUnknownOption(LocalModelPull.Installed)({ name: "x" })._tag).toBe("None") + }) +}) From 5cc2163125479e4977aa20c5287f235363c747e0 Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Sat, 3 Oct 2026 06:51:55 +0000 Subject: [PATCH 2/2] feat(PA-8): a command that says what this machine can run, and fixes a 404 it found The fit judgement and the probes had no caller. That is the failure this queue keeps finding -- an arming matcher nobody invoked, a model preference read and discarded, builtin skills registered into a list with no consumer -- so the surface is the point of this change, not an extra. `redrob models-local` prints the machine, then per configured local endpoint every installed model with either "runs here" or the refusal and its numbers. DRIVING IT END TO END FOUND A REAL DEFECT, and nothing short of that run would have. A local runtime is configured with its OpenAI base, conventionally `/v1`, but `/api/tags` is a SIBLING of `/v1`, not a child. Joining onto the configured base produced `/v1/api/tags`, got a 404, and the command printed "answering, with nothing installed" against a server serving two models. A 404 and an empty list are both "no models" to a caller that does not separate them, so the bug read as a working feature. Fixed, with a regression test that says why, and reverse-verified: removing the fix fails it. Only a TRAILING version segment is stripped. A runtime served under a prefix by a reverse proxy keeps its path, which is why `/v1/engine` is left alone. Measured on this host (16 cores, 61.8 GiB): a 2 GiB model runs here; a 120 GiB model is refused with "needs about 144.0 GiB of memory (120.0 GiB of weights plus runtime headroom) and 38.9 GiB is free of 61.8 GiB total -- short by 105.1 GiB, more than this machine has in total". Top-level `models-local` rather than `models local`, because `models` takes a positional provider filter and a subcommand there is ambiguous with a provider named `local`. The GPU probe's failure is swallowed and reported as NOT MEASURED, never as "no GPU". On a machine with no discrete card the tool simply does not exist, which is the normal state of the world rather than a condition to report -- and the memory judgement, the part that actually decides whether a model runs, does not depend on it. HttpClient and AppProcess are provided by this command rather than added to the CLI runtime. It is the only command that speaks HTTP to a local runtime or shells out for a GPU, and widening the shared runtime would hand both to every other command that has no use for them. core 1031 tests 0 fail across 123 files. redrob cli 380 tests 0 fail. Generated SDK clean, tsc clean. --- .../src/config/plugin/local-model-pull.ts | 19 ++- .../core/test/config/local-model-pull.test.ts | 24 ++++ packages/redrob/src/cli/cmd/models-local.ts | 134 ++++++++++++++++++ packages/redrob/src/index.ts | 2 + 4 files changed, 177 insertions(+), 2 deletions(-) create mode 100644 packages/redrob/src/cli/cmd/models-local.ts diff --git a/packages/core/src/config/plugin/local-model-pull.ts b/packages/core/src/config/plugin/local-model-pull.ts index b19883be0a..333b66e226 100644 --- a/packages/core/src/config/plugin/local-model-pull.ts +++ b/packages/core/src/config/plugin/local-model-pull.ts @@ -78,8 +78,23 @@ export type Progress = { readonly done: boolean } -export const tagsUrl = (base: string): string => `${base.replace(/\/+$/, "")}/api/tags` -export const pullUrl = (base: string): string => `${base.replace(/\/+$/, "")}/api/pull` +/** + * The runtime's own API lives at the ROOT, not under the OpenAI base path. + * + * A local runtime is configured with its OpenAI-compatible base, which is conventionally + * `http://host:port/v1` -- that is the URL the provider needs and the one a user copies out + * of the runtime's own banner. But `/api/tags` and `/api/pull` are siblings of `/v1`, not + * children: joining them onto the configured base produces `/v1/api/tags`, which answers 404. + * + * Measured, not reasoned about. The first version joined onto the base, and against a server + * serving the real `/api/tags` shape the command printed "answering, with nothing installed" + * -- a wrong statement that looked like a working feature, because a 404 and an empty list + * are both "no models" to a caller that does not separate them. + */ +const runtimeRoot = (base: string): string => base.replace(/\/+$/, "").replace(/\/v\d+$/, "") + +export const tagsUrl = (base: string): string => `${runtimeRoot(base)}/api/tags` +export const pullUrl = (base: string): string => `${runtimeRoot(base)}/api/pull` /** * Read one progress frame. diff --git a/packages/core/test/config/local-model-pull.test.ts b/packages/core/test/config/local-model-pull.test.ts index d065c66773..3dc7ae181a 100644 --- a/packages/core/test/config/local-model-pull.test.ts +++ b/packages/core/test/config/local-model-pull.test.ts @@ -22,6 +22,30 @@ describe("urls", () => { expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/")).toBe("http://127.0.0.1:11434/api/tags") expect(LocalModelPull.pullUrl("http://127.0.0.1:11434///")).toBe("http://127.0.0.1:11434/api/pull") }) + + test("the runtime API is a sibling of the OpenAI base, not a child of it", () => { + // A local runtime is configured with its OpenAI base, conventionally `/v1` -- that is the + // URL the provider needs and the one a user copies out of the runtime's own banner. But + // `/api/tags` sits beside `/v1`, not under it. + // + // This is a REGRESSION TEST for a defect only an end-to-end run exposed: the first version + // joined onto the configured base, produced `/v1/api/tags`, got a 404, and the command + // printed "answering, with nothing installed" against a server that was serving two + // models. A wrong statement that reads as a working feature, because a 404 and an empty + // list are both "no models" to a caller that does not separate them. + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/v1")).toBe("http://127.0.0.1:11434/api/tags") + expect(LocalModelPull.pullUrl("http://127.0.0.1:11434/v1/")).toBe("http://127.0.0.1:11434/api/pull") + // Any version segment, not just v1: some runtimes serve /v2. + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/v2")).toBe("http://127.0.0.1:11434/api/tags") + }) + + test("a path that merely CONTAINS a version segment is left alone", () => { + // Only a TRAILING version segment is the OpenAI base. Stripping `/v1` from the middle of a + // path would break a runtime served under a prefix by a reverse proxy. + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/v1/engine")).toBe( + "http://127.0.0.1:11434/v1/engine/api/tags", + ) + }) }) describe("readFrames", () => { diff --git a/packages/redrob/src/cli/cmd/models-local.ts b/packages/redrob/src/cli/cmd/models-local.ts new file mode 100644 index 0000000000..5627069767 --- /dev/null +++ b/packages/redrob/src/cli/cmd/models-local.ts @@ -0,0 +1,134 @@ +import { Effect, Layer } from "effect" +import { FetchHttpClient } from "effect/unstable/http" +import { LayerNode } from "@redrob-code/core/effect/layer-node" +import { AppProcess } from "@redrob-code/core/process" +import { LocalGpuProbe } from "@redrob-code/core/config/plugin/local-gpu-probe" +import { LocalModelFit } from "@redrob-code/core/config/plugin/local-model-fit" +import { LocalModelPull } from "@redrob-code/core/config/plugin/local-model-pull" +import { isLocalEndpoint } from "@redrob-code/core/config/plugin/local-provider" +import { Config } from "@/config/config" +import { effectCmd } from "../effect-cmd" +import { UI } from "../ui" + +/** + * PA-8's surface: what this machine can run, and what it cannot, with the numbers. + * + * The judgement and the probes are libraries with their own tests; this is the one place a + * person can see them. Without it the fit check is code nobody reaches -- the failure this + * queue has now found three times (an arming matcher with no caller, a model preference read + * and discarded, builtin skills registered into a list with no consumer). + * + * A COMMAND rather than a settings page, because the engine has no settings UI of its own and + * the browser's would mean a mojom round trip for a read that is already local. The browser + * can call this later; the numbers are the part that had nowhere to appear. + * + * TOP-LEVEL rather than `models local`, deliberately. `models` takes a POSITIONAL provider + * filter (`models anthropic`), so adding a subcommand under it makes `models local` ambiguous + * -- yargs cannot tell the subcommand from a provider that happens to be called `local`. + * + * NO CLOUD FALLBACK is offered or hinted at anywhere here. A user asking what their own + * machine can run is not asking to be sold a hosted model. + */ + +const GiB = 1024 ** 3 +const gib = (bytes: number): string => `${(bytes / GiB).toFixed(1)} GiB` + +export const ModelsLocalCommand = effectCmd({ + command: "models-local", + describe: "what this machine can run, and what it cannot", + builder: (yargs) => yargs, + handler: Effect.fn("Cli.models.local")( + function* (_args) { + /* + * A failed GPU probe reports NOT MEASURED, never "no GPU". + * + * The probe shells out, and on a machine with no discrete card the tool simply does + * not exist -- the normal state of the world, not a condition to report. The memory + * judgement, which is the part that actually decides whether a model runs, does not + * depend on it. + */ + const gpu = yield* LocalGpuProbe.probe(process.platform).pipe( + Effect.catchCause(() => Effect.succeed(undefined)), + ) + const hardware = LocalModelFit.measure(gpu) + + UI.println(UI.Style.TEXT_NORMAL_BOLD + "This machine" + UI.Style.TEXT_NORMAL) + UI.println(` platform ${hardware.platform} ${hardware.arch}`) + UI.println(` cores ${hardware.cores}`) + UI.println(` memory ${gib(hardware.availableBytes)} free of ${gib(hardware.totalBytes)}`) + // "not measured", never "none". A machine with a card we could not see must not be told + // it has no card -- that is a false statement about the user's own computer. + UI.println( + ` gpu ${ + gpu === undefined + ? "not measured on this platform" + : gpu.vramBytes === undefined + ? gpu.name + : `${gpu.name}, ${gib(gpu.vramBytes)}` + }`, + ) + UI.println("") + + // Only endpoints the local rule already allows. Re-checked here rather than trusted, + // the same way the model listing re-checks it: this command would otherwise be a way to + // make the CLI call an arbitrary host by editing a config file. + const config = yield* (yield* Config.Service).get() + const endpoints = Object.values(config.provider ?? {}) + .map((provider) => (provider as { options?: { baseURL?: string } }).options?.baseURL) + .filter((url): url is string => typeof url === "string" && isLocalEndpoint(url)) + + if (endpoints.length === 0) { + UI.println(UI.Style.TEXT_DIM + "No local model runtime is configured." + UI.Style.TEXT_NORMAL) + // Said plainly, because "no models" and "nowhere to look for models" are different + // facts and a user who sees the first will go looking for a model they already have. + UI.println( + UI.Style.TEXT_DIM + + "Point a provider at one on this machine to see what it is serving." + + UI.Style.TEXT_NORMAL, + ) + return + } + + let sawRuntime = false + for (const endpoint of endpoints) { + const { reachable, models } = yield* LocalModelPull.installed(endpoint) + if (!reachable) { + // Not an error. A runtime that is not running is the normal state of a laptop, and + // the honest line says which endpoint did not answer rather than implying the + // machine cannot run anything. + UI.println(`${endpoint} ${UI.Style.TEXT_DIM}not answering${UI.Style.TEXT_NORMAL}`) + continue + } + sawRuntime = true + UI.println(UI.Style.TEXT_NORMAL_BOLD + endpoint + UI.Style.TEXT_NORMAL) + if (models.length === 0) { + UI.println(` ${UI.Style.TEXT_DIM}answering, with nothing installed${UI.Style.TEXT_NORMAL}`) + continue + } + for (const model of models) { + const verdict = LocalModelFit.fit({ modelBytes: model.size, hardware }) + if (verdict.kind === "fits") { + UI.println(` ${model.name} ${gib(model.size)} runs here`) + continue + } + // The refusal carries its own numbers, so the line is the library's sentence rather + // than a re-worded summary that could disagree with it. + UI.println(` ${model.name} ${gib(model.size)} ${UI.Style.TEXT_DIM}${verdict.reason}${UI.Style.TEXT_NORMAL}`) + } + } + + if (!sawRuntime) { + UI.println("") + UI.println( + UI.Style.TEXT_DIM + + "Every configured local endpoint is down, so nothing could be listed." + + UI.Style.TEXT_NORMAL, + ) + } + }, (effect) => + // Both provided here rather than added to the CLI runtime: this is the only command that + // speaks HTTP to a local runtime or shells out for a GPU, and widening the shared runtime + // for one command would hand both to every other command that has no use for them. + Effect.provide(effect, Layer.mergeAll(FetchHttpClient.layer, LayerNode.compile(AppProcess.node))), + ), +}) diff --git a/packages/redrob/src/index.ts b/packages/redrob/src/index.ts index ad182ba3e8..d7d09963a2 100644 --- a/packages/redrob/src/index.ts +++ b/packages/redrob/src/index.ts @@ -8,6 +8,7 @@ import { AgentCommand } from "./cli/cmd/agent" import { UpgradeCommand } from "./cli/cmd/upgrade" import { UninstallCommand } from "./cli/cmd/uninstall" import { ModelsCommand } from "./cli/cmd/models" +import { ModelsLocalCommand } from "./cli/cmd/models-local" import { UI } from "./cli/ui" import { InstallationVersion } from "@redrob-code/core/installation/version" import { FormatError } from "./cli/error" @@ -91,6 +92,7 @@ const cli = yargs(args) .command(UninstallCommand) .command(ServeCommand) .command(ModelsCommand) + .command(ModelsLocalCommand) .command(StatsCommand) .command(ExportCommand) .command(ImportCommand)