diff --git a/packages/core/src/config/plugin/local-gpu-probe.ts b/packages/core/src/config/plugin/local-gpu-probe.ts new file mode 100644 index 0000000000..024c919080 --- /dev/null +++ b/packages/core/src/config/plugin/local-gpu-probe.ts @@ -0,0 +1,98 @@ +export * as LocalGpuProbe from "./local-gpu-probe" + +/** + * Reading the GPU, where the platform lets us. + * + * Split from `local-model-fit` on purpose. The fit judgement is pure and testable with no + * machine at all; this part shells out, and what it can learn differs per platform and per + * driver. Keeping them apart means the verdict logic is never untestable because the probe + * is. + * + * THE ONLY RULE THAT MATTERS HERE: a probe that cannot see a GPU returns `undefined`, which + * the fit module reads as NOT MEASURED. It never returns "no GPU". Telling a user with a + * discrete card that their machine has none is a false statement about their own computer, + * and it is the statement they would push back on hardest and be right about. + */ + +import { Effect } from "effect" +import { ChildProcess } from "effect/unstable/process" +import { AppProcess, requireSuccess } from "../../process" + +/** What a probe can learn. `vramBytes` is absent when the platform reports a name but no size. */ +export type Gpu = { readonly name: string; readonly vramBytes?: number | undefined } + +/** Short, because this runs on a path a user is waiting on and a wedged driver tool must not hold it. */ +const PROBE_TIMEOUT = "3 seconds" + +const MiB = 1024 * 1024 + +/** + * The per-platform command and how to read it. Exported so the parsers are testable against + * recorded tool output without a GPU, a driver, or that platform. + * + * Each is a listing command with machine-readable output, chosen over the prettier + * human-facing variants so the parse does not depend on column widths. + */ +export const PROBES: ReadonlyArray<{ + readonly platforms: ReadonlyArray + readonly command: string + readonly args: ReadonlyArray + readonly parse: (stdout: string) => Gpu | undefined +}> = [ + { + // NVIDIA, any platform it is installed on. CSV without units keeps the parse trivial. + platforms: ["linux", "win32", "darwin"], + command: "nvidia-smi", + args: ["--query-gpu=name,memory.total", "--format=csv,noheader,nounits"], + parse: (stdout) => { + const line = stdout.split("\n").find((candidate) => candidate.trim().length > 0) + if (line === undefined) return undefined + const [name, memory] = line.split(",").map((field) => field.trim()) + if (name === undefined || name.length === 0) return undefined + const megabytes = Number(memory) + return { name, vramBytes: Number.isFinite(megabytes) && megabytes > 0 ? megabytes * MiB : undefined } + }, + }, + { + // Apple silicon: the GPU shares system memory, so there is a name and no separate VRAM. + // Reporting a number here would be wrong in the direction that matters -- it would be + // counted twice against the same bytes the memory check already counted. + platforms: ["darwin"], + command: "sysctl", + args: ["-n", "machdep.cpu.brand_string"], + parse: (stdout) => { + const brand = stdout.trim() + if (!brand.startsWith("Apple ")) return undefined + return { name: `${brand} (unified memory)`, vramBytes: undefined } + }, + }, +] + +/** + * Try each probe for this platform, first answer wins, every failure is silent. + * + * Silent is right: on a machine with no discrete GPU, `nvidia-smi` not existing is the + * normal state of the world, not a condition to report. The caller gets `undefined` and + * says "not measured", which is the true statement. + * + * Runs through `AppProcess.run`, the service the rest of the engine shells out with, rather + * than driving the spawner directly -- output collection, truncation and timeout handling + * already live there and a second copy of them would drift. + */ +export const probe = Effect.fn("LocalGpuProbe.probe")(function* (platform: string) { + const process = yield* AppProcess.Service + for (const candidate of PROBES) { + if (!candidate.platforms.includes(platform)) continue + const result = yield* Effect.gen(function* () { + const run = yield* process + .run(ChildProcess.make(candidate.command, [...candidate.args], { stdin: "ignore", extendEnv: true })) + .pipe(Effect.flatMap(requireSuccess)) + return candidate.parse(run.stdout.toString("utf8")) + }).pipe( + Effect.timeout(PROBE_TIMEOUT), + Effect.catchCause(() => Effect.succeed(undefined)), + ) + if (result !== undefined) return result + } + return undefined +}) diff --git a/packages/core/src/config/plugin/local-model-fit.ts b/packages/core/src/config/plugin/local-model-fit.ts new file mode 100644 index 0000000000..3b8f40cae6 --- /dev/null +++ b/packages/core/src/config/plugin/local-model-fit.ts @@ -0,0 +1,143 @@ +export * as LocalModelFit from "./local-model-fit" + +/** + * Whether this machine can run a local model, and if not, exactly what is short. + * + * PA-8's decisive half. Running a model locally is not a preference, it is a hardware + * question, and the only dishonest answer is a confident one. So this module is built around + * three rules: + * + * 1. It reports MEASURED facts, never estimates dressed as facts. Total and available + * memory and the CPU come from the OS. A GPU is read where the platform exposes it and + * reported as UNKNOWN where it does not -- `unknown` is a third answer, distinct from + * "no GPU", because a machine with a GPU we cannot see must not be told it has none. + * 2. It never invents a model's size. The required figure comes from whoever is offering + * the model -- the runtime's own registry -- and is passed in. A fit check against a + * size this module guessed would be a guess wearing a verdict's clothes. + * 3. A refusal names the constraint and the shortfall. "Your machine cannot run this" is + * not actionable; "needs 8.0 GiB of available memory, this machine has 3.2 GiB free of + * 15.5 GiB total" tells the user whether closing something would fix it. + * + * There is deliberately NO cloud fallback here, silently or otherwise. A user who asked for + * a local model asked for the data to stay on the machine; quietly answering from a hosted + * model would defeat the only reason to want this, and would do it invisibly. + */ + +import * as os from "node:os" + +/** What the machine is, as measured. Every field is read, none is inferred. */ +export type Hardware = { + readonly platform: string + readonly arch: string + readonly cores: number + readonly totalBytes: number + readonly availableBytes: number + /** + * The GPU, when this platform lets us see it. + * + * `undefined` means NOT MEASURED, not absent. The difference matters: a refusal that + * claims a machine has no GPU when we simply could not look is a false statement about + * the user's own computer. + */ + readonly gpu?: { readonly name: string; readonly vramBytes?: number | undefined } | undefined +} + +/** The verdict. A refusal always carries the numbers that produced it. */ +export type Fit = + | { readonly kind: "fits"; readonly hardware: Hardware } + | { + readonly kind: "short" + readonly hardware: Hardware + /** Which constraint failed, for a caller that wants to branch rather than print. */ + readonly constraint: "memory" | "arch" + readonly requiredBytes: number + readonly shortfallBytes: number + readonly reason: string + } + +/** + * Headroom beyond the weights themselves. + * + * A model's file size is not its running size: the runtime holds the KV cache, the context + * window and its own working set alongside the weights. 20% with a 1 GiB floor is a + * deliberately rough allowance, and it is named here rather than buried so that a caller + * reading a refusal can see it is part of the requirement. + * + * Rough is correct for this. The alternative is a precise-looking formula over quantisation, + * context length and runtime, which would be wrong in a way that reads as authoritative. + */ +export const HEADROOM_FRACTION = 0.2 +export const HEADROOM_FLOOR_BYTES = 1024 ** 3 + +export const requirementFor = (modelBytes: number): number => + modelBytes + Math.max(Math.round(modelBytes * HEADROOM_FRACTION), HEADROOM_FLOOR_BYTES) + +const GiB = 1024 ** 3 +const gib = (bytes: number): string => `${(bytes / GiB).toFixed(1)} GiB` + +/** + * Architectures a local runtime has builds for. + * + * Checked because the failure is otherwise a confusing download: the weights arrive, the + * runtime will not start, and nothing in that sequence mentions the CPU. + */ +const SUPPORTED_ARCH = new Set(["x64", "arm64"]) + +/** + * Measure the machine. + * + * `availableBytes` uses `os.freemem()`, which on Linux counts reclaimable page cache as used + * and therefore UNDERSTATES what a process could actually get. That direction is the safe + * one: it can refuse a model that would in fact have squeezed in, and it will not promise + * one that would have been killed mid-generation. The opposite error loses the user's work. + */ +export const measure = (gpu?: Hardware["gpu"]): Hardware => ({ + platform: os.platform(), + arch: os.arch(), + cores: os.cpus().length, + totalBytes: os.totalmem(), + availableBytes: os.freemem(), + gpu, +}) + +/** + * Can this machine run a model of this size? + * + * Memory is checked against AVAILABLE rather than total, because a model that needs more + * than is free today does not run today, however large the machine is. The refusal reports + * both numbers so the user can tell "buy a bigger machine" from "close your other windows". + */ +export const fit = (input: { readonly modelBytes: number; readonly hardware: Hardware }): Fit => { + const { hardware } = input + if (!SUPPORTED_ARCH.has(hardware.arch)) { + return { + kind: "short", + hardware, + constraint: "arch", + requiredBytes: 0, + shortfallBytes: 0, + reason: `local models need an x64 or arm64 CPU; this machine reports ${hardware.arch}`, + } + } + + const requiredBytes = requirementFor(input.modelBytes) + if (hardware.availableBytes >= requiredBytes) return { kind: "fits", hardware } + + const shortfallBytes = requiredBytes - hardware.availableBytes + const fitsIfFreed = hardware.totalBytes >= requiredBytes + return { + kind: "short", + hardware, + constraint: "memory", + requiredBytes, + shortfallBytes, + reason: [ + `this model needs about ${gib(requiredBytes)} of memory`, + `(${gib(input.modelBytes)} of weights plus runtime headroom)`, + `and ${gib(hardware.availableBytes)} is free of ${gib(hardware.totalBytes)} total`, + fitsIfFreed + ? `-- short by ${gib(shortfallBytes)}, which this machine has if you close something` + : `-- short by ${gib(shortfallBytes)}, more than this machine has in total`, + ].join(" "), + } +} diff --git a/packages/core/src/config/plugin/local-model-pull.ts b/packages/core/src/config/plugin/local-model-pull.ts new file mode 100644 index 0000000000..333b66e226 --- /dev/null +++ b/packages/core/src/config/plugin/local-model-pull.ts @@ -0,0 +1,171 @@ +export * as LocalModelPull from "./local-model-pull" + +/** + * Listing and fetching models from a local runtime. + * + * PA-8's other half. The engine already runs a local model once one is present -- a local + * provider speaks the OpenAI wire protocol and `local-models.ts` lists what it serves. What + * was missing is everything before that: what is on disk, how big, and getting one. + * + * WHY THIS IS A SECOND SET OF ROUTES. The OpenAI-compatible `GET /models` the provider + * listing uses reports an id and nothing else -- no size. So it cannot answer the one + * question PA-8 turns on, which is whether this machine can run the thing. Sizes and the + * fetch itself live on the runtime's own API, so that is what this module speaks, and it is + * gated by the same local-endpoint rule. + * + * WE DO NOT FETCH WEIGHTS OURSELVES. No picking a mirror, no choosing a quantisation, no + * verifying a digest we chose to trust. The runtime already does all of it, resumes an + * interrupted download, and shares progress between concurrent callers -- its own docs say + * so. Re-implementing that would mean owning model provenance, which is a far larger + * promise than "run a model locally". + * + * Shapes here come from the runtime's published API reference, not from reading our own + * code back: `GET /api/tags` returns `models[].{name,size,digest,details}`, and + * `POST /api/pull` streams `{status}` objects that carry `{digest,total,completed}` while + * layers transfer. Two traps are documented there and both are handled below. + */ + +import { Effect, Schema } from "effect" +import { HttpClient, HttpClientRequest } from "effect/unstable/http" + +import { isLocalEndpoint } from "./local-provider" + +/** Long, because a pull is a multi-gigabyte transfer and the caller is watching progress. */ +const PULL_TIMEOUT = "6 hours" + +/** Short, because listing what is already on disk is a local call that either answers or is down. */ +const LIST_TIMEOUT = "2 seconds" + +/** + * One installed model. `size` is the field the whole fit check depends on. + * + * Decoded ONE AT A TIME by the caller, for the reason `local-models.ts` states: one entry + * this build cannot describe should cost that entry, not the list. + */ +export const Installed = Schema.Struct({ + name: Schema.String, + size: Schema.Finite, + digest: Schema.optional(Schema.String), +}) +export type Installed = Schema.Schema.Type + +const TagList = Schema.Struct({ models: Schema.Array(Schema.Unknown) }) + +/** + * One progress frame from a pull. + * + * `total` and `completed` are BOTH optional, and that is not defensive typing -- the + * runtime's own docs say so. Status-only frames (`pulling manifest`, `verifying sha256 + * digest`, `writing manifest`, `success`) carry neither, and a downloading frame may carry + * `total` without `completed` until the first bytes land. + */ +export const PullFrame = Schema.Struct({ + status: Schema.String, + digest: Schema.optional(Schema.String), + total: Schema.optional(Schema.Finite), + completed: Schema.optional(Schema.Finite), + error: Schema.optional(Schema.String), +}) +export type PullFrame = Schema.Schema.Type + +/** Progress as a caller should show it. */ +export type Progress = { + /** The runtime's own status line, passed through rather than re-worded. */ + readonly status: string + /** Bytes transferred and expected FOR THE CURRENT LAYER, when the frame carries them. */ + readonly layerCompletedBytes?: number | undefined + readonly layerTotalBytes?: number | undefined + readonly done: boolean +} + +/** + * The runtime's own API lives at the ROOT, not under the OpenAI base path. + * + * A local runtime is configured with its OpenAI-compatible base, which is conventionally + * `http://host:port/v1` -- that is the URL the provider needs and the one a user copies out + * of the runtime's own banner. But `/api/tags` and `/api/pull` are siblings of `/v1`, not + * children: joining them onto the configured base produces `/v1/api/tags`, which answers 404. + * + * Measured, not reasoned about. The first version joined onto the base, and against a server + * serving the real `/api/tags` shape the command printed "answering, with nothing installed" + * -- a wrong statement that looked like a working feature, because a 404 and an empty list + * are both "no models" to a caller that does not separate them. + */ +const runtimeRoot = (base: string): string => base.replace(/\/+$/, "").replace(/\/v\d+$/, "") + +export const tagsUrl = (base: string): string => `${runtimeRoot(base)}/api/tags` +export const pullUrl = (base: string): string => `${runtimeRoot(base)}/api/pull` + +/** + * Read one progress frame. + * + * A frame with an `error` is a failure the runtime is reporting in-band, mid-stream, with a + * 200 already sent. Treated as terminal rather than as progress, because the alternative is + * a pull that looks like it is still going and never finishes. + */ +export const readFrame = (frame: PullFrame): Progress | { readonly error: string } => { + if (frame.error !== undefined && frame.error.length > 0) return { error: frame.error } + return { + status: frame.status, + layerCompletedBytes: frame.completed, + layerTotalBytes: frame.total, + done: frame.status === "success", + } +} + +/** + * Decode a newline-delimited stream body into frames, dropping only what fails. + * + * The runtime emits one JSON object per line and does not wrap them in an array, so a + * whole-body parse never succeeds. A line that does not decode is skipped: these frames are + * progress, and one unreadable frame must not abort a transfer that is working. + */ +export const readFrames = (body: string): ReadonlyArray => { + const frames: PullFrame[] = [] + for (const line of body.split("\n")) { + const trimmed = line.trim() + if (trimmed.length === 0) continue + try { + frames.push(Schema.decodeUnknownSync(PullFrame)(JSON.parse(trimmed))) + } catch { + continue + } + } + return frames +} + +/** + * What is installed on this runtime, with sizes. + * + * Best-effort for the same reason the provider listing is: a local runtime that is not + * running is the normal state of a laptop, not an error. A failed call returns an empty + * list, and the caller distinguishes "nothing installed" from "no runtime" by asking + * whether the runtime answered at all -- which is why `reachable` is returned separately + * rather than inferred from an empty array. + */ +export const installed = Effect.fn("LocalModelPull.installed")(function* (base: string) { + // Re-checked here rather than trusted from the caller, the same way `local-models.ts` + // re-checks it: two independent checks of one rule means a later change that loosens one + // cannot quietly turn this into a way to make the engine call an arbitrary host. + if (!isLocalEndpoint(base)) return { reachable: false, models: [] as ReadonlyArray } + + const client = yield* HttpClient.HttpClient + const result = yield* client + .execute(HttpClientRequest.get(tagsUrl(base))) + .pipe( + Effect.flatMap((response) => response.json), + Effect.timeout(LIST_TIMEOUT), + Effect.catchCause(() => Effect.succeed(undefined)), + ) + if (result === undefined) return { reachable: false, models: [] as ReadonlyArray } + + const list = Schema.decodeUnknownOption(TagList)(result) + if (list._tag === "None") return { reachable: true, models: [] as ReadonlyArray } + + const models: Installed[] = [] + for (const entry of list.value.models) { + const decoded = Schema.decodeUnknownOption(Installed)(entry) + if (decoded._tag === "Some") models.push(decoded.value) + } + return { reachable: true, models } +}) diff --git a/packages/core/test/config/local-gpu-probe.test.ts b/packages/core/test/config/local-gpu-probe.test.ts new file mode 100644 index 0000000000..d02534dd7a --- /dev/null +++ b/packages/core/test/config/local-gpu-probe.test.ts @@ -0,0 +1,88 @@ +/** + * PA-8's GPU parsers, against recorded tool output. + * + * Parsed, not mocked: each fixture is the shape the real command prints, with the flags the + * probe passes. A parser verified only against a string I also wrote would prove the two + * agree and nothing about the tool. + * + * The failure these guard is silent. A parser that returns `undefined` for output it should + * have read makes the machine look like one with no GPU, which is a false claim this module + * exists specifically to avoid. + */ +import { describe, expect, test } from "bun:test" +import { LocalGpuProbe } from "@redrob-code/core/config/plugin/local-gpu-probe" + +const MiB = 1024 * 1024 + +const probeFor = (command: string) => { + const found = LocalGpuProbe.PROBES.find((candidate) => candidate.command === command) + if (found === undefined) throw new Error(`no probe spawns ${command}`) + return found +} + +describe("nvidia-smi", () => { + const nvidia = probeFor("nvidia-smi") + + test("passes the flags that make the output machine-readable", () => { + // `noheader,nounits` is what lets the parse be a comma split instead of a column guess, + // and `nounits` is why the memory field is read as plain MiB. + expect(nvidia.args.join(" ")).toContain("--format=csv,noheader,nounits") + expect(nvidia.args.join(" ")).toContain("memory.total") + }) + + test("reads name and VRAM from one card", () => { + expect(nvidia.parse("NVIDIA GeForce RTX 4090, 24564\n")).toEqual({ + name: "NVIDIA GeForce RTX 4090", + vramBytes: 24564 * MiB, + }) + }) + + test("takes the first card when several are listed", () => { + const output = "NVIDIA A100-SXM4-40GB, 40960\nNVIDIA A100-SXM4-40GB, 40960\n" + expect(nvidia.parse(output)?.name).toBe("NVIDIA A100-SXM4-40GB") + }) + + test("keeps the name when the memory field is unreadable", () => { + // A driver that reports `[N/A]` for memory still tells us the card exists, and a card we + // know about with a size we do not is better than pretending there is no card. + const parsed = nvidia.parse("NVIDIA GeForce GTX 1060, [N/A]\n") + expect(parsed?.name).toBe("NVIDIA GeForce GTX 1060") + expect(parsed?.vramBytes).toBeUndefined() + }) + + test("returns undefined for empty output rather than an empty name", () => { + expect(nvidia.parse("")).toBeUndefined() + expect(nvidia.parse("\n\n")).toBeUndefined() + }) +}) + +describe("apple silicon", () => { + const sysctl = probeFor("sysctl") + + test("names the chip and reports no separate VRAM", () => { + // Unified memory: the GPU uses the same bytes the memory check already counted, so a + // VRAM number here would be those bytes counted twice. + expect(sysctl.parse("Apple M3 Max\n")).toEqual({ name: "Apple M3 Max (unified memory)", vramBytes: undefined }) + }) + + test("declines an Intel Mac rather than calling its CPU a GPU", () => { + expect(sysctl.parse("Intel(R) Core(TM) i9-9980HK CPU @ 2.40GHz\n")).toBeUndefined() + }) + + test("is offered only on darwin", () => { + expect(sysctl.platforms).toEqual(["darwin"]) + }) +}) + +describe("probe ordering", () => { + test("nvidia is tried before the platform-specific fallback on darwin", () => { + // An external or eGPU NVIDIA card is the more specific answer, and the sysctl probe + // would otherwise claim unified memory on a machine that has a discrete card. + const darwin = LocalGpuProbe.PROBES.filter((candidate) => candidate.platforms.includes("darwin")) + expect(darwin[0]?.command).toBe("nvidia-smi") + }) + + test("every probe declares the platforms it applies to", () => { + for (const candidate of LocalGpuProbe.PROBES) expect(candidate.platforms.length).toBeGreaterThan(0) + }) +}) diff --git a/packages/core/test/config/local-model-fit.test.ts b/packages/core/test/config/local-model-fit.test.ts new file mode 100644 index 0000000000..136f9b92f4 --- /dev/null +++ b/packages/core/test/config/local-model-fit.test.ts @@ -0,0 +1,129 @@ +/** + * PA-8's fit judgement. + * + * Every case here is a claim the user could push back on, so each asserts the NUMBERS in the + * refusal rather than that a refusal happened. A verdict with no numbers is unfalsifiable by + * the person it is about, which is the opposite of honest. + */ +import { describe, expect, test } from "bun:test" +import { LocalModelFit } from "@redrob-code/core/config/plugin/local-model-fit" + +const GiB = 1024 ** 3 + +const machine = (overrides: Partial = {}): LocalModelFit.Hardware => ({ + platform: "linux", + arch: "x64", + cores: 8, + totalBytes: 16 * GiB, + availableBytes: 12 * GiB, + ...overrides, +}) + +describe("LocalModelFit.requirementFor", () => { + test("adds proportional headroom above the floor", () => { + // 10 GiB of weights: 20% is 2 GiB, which clears the 1 GiB floor, so the floor does not apply. + expect(LocalModelFit.requirementFor(10 * GiB)).toBe(12 * GiB) + }) + + test("applies the floor when the proportion is smaller than it", () => { + // 2 GiB of weights: 20% is 0.4 GiB, under the floor, so the floor is what is added. + expect(LocalModelFit.requirementFor(2 * GiB)).toBe(3 * GiB) + }) + + test("the headroom is never zero, so a model is never sized at exactly its weights", () => { + expect(LocalModelFit.requirementFor(0)).toBe(LocalModelFit.HEADROOM_FLOOR_BYTES) + }) +}) + +describe("LocalModelFit.fit", () => { + test("fits when available memory covers weights plus headroom", () => { + const verdict = LocalModelFit.fit({ modelBytes: 4 * GiB, hardware: machine() }) + expect(verdict.kind).toBe("fits") + }) + + test("is short by the exact difference, and says closing something would do it", () => { + // Needs 12 GiB (10 + 2 headroom); 8 free of 16 total, so freeing 4 GiB is enough. + const verdict = LocalModelFit.fit({ + modelBytes: 10 * GiB, + hardware: machine({ availableBytes: 8 * GiB, totalBytes: 16 * GiB }), + }) + if (verdict.kind !== "short") throw new Error("expected a refusal") + expect(verdict.constraint).toBe("memory") + expect(verdict.requiredBytes).toBe(12 * GiB) + expect(verdict.shortfallBytes).toBe(4 * GiB) + expect(verdict.reason).toContain("12.0 GiB") + expect(verdict.reason).toContain("8.0 GiB is free of 16.0 GiB total") + expect(verdict.reason).toContain("if you close something") + }) + + test("says the machine is too small when freeing everything would not do it", () => { + // Needs 12 GiB and the machine HAS 8 GiB. No amount of closing windows fixes this, and + // telling the user to try would waste their time. + const verdict = LocalModelFit.fit({ + modelBytes: 10 * GiB, + hardware: machine({ availableBytes: 6 * GiB, totalBytes: 8 * GiB }), + }) + if (verdict.kind !== "short") throw new Error("expected a refusal") + expect(verdict.reason).toContain("more than this machine has in total") + expect(verdict.reason).not.toContain("if you close something") + }) + + test("checks available rather than total, so a busy large machine is refused", () => { + // 64 GiB machine with 2 GiB free. Checking total would promise a model that gets killed + // partway through generating, which loses the user's work and explains nothing. + const verdict = LocalModelFit.fit({ + modelBytes: 10 * GiB, + hardware: machine({ availableBytes: 2 * GiB, totalBytes: 64 * GiB }), + }) + expect(verdict.kind).toBe("short") + }) + + test("refuses an unsupported CPU before talking about memory at all", () => { + // A machine with plenty of memory and no runtime build. Reporting a memory shortfall + // here would send the user to close applications for a problem that is not memory. + const verdict = LocalModelFit.fit({ + modelBytes: 1 * GiB, + hardware: machine({ arch: "ppc64", availableBytes: 60 * GiB, totalBytes: 64 * GiB }), + }) + if (verdict.kind !== "short") throw new Error("expected a refusal") + expect(verdict.constraint).toBe("arch") + expect(verdict.reason).toContain("ppc64") + }) + + test("an unseen GPU is reported as unmeasured, never as absent", () => { + const verdict = LocalModelFit.fit({ modelBytes: 1 * GiB, hardware: machine({ gpu: undefined }) }) + expect(verdict.hardware.gpu).toBeUndefined() + // The refusal text must never make a claim about a GPU we did not look at. + const refused = LocalModelFit.fit({ + modelBytes: 100 * GiB, + hardware: machine({ gpu: undefined }), + }) + if (refused.kind !== "short") throw new Error("expected a refusal") + expect(refused.reason.toLowerCase()).not.toContain("gpu") + expect(refused.reason.toLowerCase()).not.toContain("graphics") + }) + + test("carries a measured GPU through untouched", () => { + const gpu = { name: "Test GPU", vramBytes: 8 * GiB } + const verdict = LocalModelFit.fit({ modelBytes: 1 * GiB, hardware: machine({ gpu }) }) + expect(verdict.hardware.gpu).toEqual(gpu) + }) +}) + +describe("LocalModelFit.measure", () => { + test("reports this machine's real numbers, and no GPU claim without a probe", () => { + const hardware = LocalModelFit.measure() + expect(hardware.totalBytes).toBeGreaterThan(0) + expect(hardware.availableBytes).toBeGreaterThan(0) + expect(hardware.availableBytes).toBeLessThanOrEqual(hardware.totalBytes) + expect(hardware.cores).toBeGreaterThan(0) + expect(hardware.gpu).toBeUndefined() + }) + + test("takes the GPU from its caller rather than probing inside the verdict path", () => { + // The split exists so the pure judgement is testable with no machine. If `measure` ever + // probed on its own, this test would see a GPU it never passed in. + const hardware = LocalModelFit.measure({ name: "Passed In" }) + expect(hardware.gpu?.name).toBe("Passed In") + }) +}) diff --git a/packages/core/test/config/local-model-pull.test.ts b/packages/core/test/config/local-model-pull.test.ts new file mode 100644 index 0000000000..3dc7ae181a --- /dev/null +++ b/packages/core/test/config/local-model-pull.test.ts @@ -0,0 +1,177 @@ +/** + * PA-8's pull and listing, against the runtime's OWN published examples. + * + * Every fixture below is copied from the runtime's API reference, not written to match this + * code. That distinction is the whole value: a fixture and a parser written from one reading + * of a schema agree with each other and prove nothing about the real producer. + * + * Two traps the reference states explicitly, and both have a test: + * - `completed` MAY BE ABSENT while a layer is downloading ("Until any of the download is + * completed, the `completed` key may not be included"). + * - `total` is per-LAYER, not per-model ("The number of files to be downloaded depends on + * the number of layers specified in the manifest"). + */ +import { describe, expect, test } from "bun:test" +import { Schema } from "effect" +import { LocalModelFit } from "@redrob-code/core/config/plugin/local-model-fit" +import { LocalModelPull } from "@redrob-code/core/config/plugin/local-model-pull" + +describe("urls", () => { + test("join without doubling a slash", () => { + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434")).toBe("http://127.0.0.1:11434/api/tags") + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/")).toBe("http://127.0.0.1:11434/api/tags") + expect(LocalModelPull.pullUrl("http://127.0.0.1:11434///")).toBe("http://127.0.0.1:11434/api/pull") + }) + + test("the runtime API is a sibling of the OpenAI base, not a child of it", () => { + // A local runtime is configured with its OpenAI base, conventionally `/v1` -- that is the + // URL the provider needs and the one a user copies out of the runtime's own banner. But + // `/api/tags` sits beside `/v1`, not under it. + // + // This is a REGRESSION TEST for a defect only an end-to-end run exposed: the first version + // joined onto the configured base, produced `/v1/api/tags`, got a 404, and the command + // printed "answering, with nothing installed" against a server that was serving two + // models. A wrong statement that reads as a working feature, because a 404 and an empty + // list are both "no models" to a caller that does not separate them. + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/v1")).toBe("http://127.0.0.1:11434/api/tags") + expect(LocalModelPull.pullUrl("http://127.0.0.1:11434/v1/")).toBe("http://127.0.0.1:11434/api/pull") + // Any version segment, not just v1: some runtimes serve /v2. + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/v2")).toBe("http://127.0.0.1:11434/api/tags") + }) + + test("a path that merely CONTAINS a version segment is left alone", () => { + // Only a TRAILING version segment is the OpenAI base. Stripping `/v1` from the middle of a + // path would break a runtime served under a prefix by a reverse proxy. + expect(LocalModelPull.tagsUrl("http://127.0.0.1:11434/v1/engine")).toBe( + "http://127.0.0.1:11434/v1/engine/api/tags", + ) + }) +}) + +describe("readFrames", () => { + // Verbatim from the reference's own pull example, in order. + const stream = [ + '{"status":"pulling manifest"}', + '{"status":"pulling digestname","digest":"digestname","total":2142590208,"completed":241970}', + '{"status":"verifying sha256 digest"}', + '{"status":"writing manifest"}', + '{"status":"removing any unused layers"}', + '{"status":"success"}', + ].join("\n") + + test("reads every frame of the documented stream", () => { + const frames = LocalModelPull.readFrames(stream) + expect(frames.length).toBe(6) + expect(frames[0]?.status).toBe("pulling manifest") + expect(frames[1]?.total).toBe(2142590208) + expect(frames[5]?.status).toBe("success") + }) + + test("status-only frames decode, carrying no byte counts", () => { + const frames = LocalModelPull.readFrames('{"status":"writing manifest"}') + expect(frames[0]?.total).toBeUndefined() + expect(frames[0]?.completed).toBeUndefined() + }) + + test("a downloading frame with no completed key is still a frame", () => { + // The reference says this one happens before the first bytes land. Requiring `completed` + // would drop the frame that announces the size of the download. + const frames = LocalModelPull.readFrames('{"status":"pulling abc","digest":"abc","total":2142590208}') + expect(frames.length).toBe(1) + expect(frames[0]?.total).toBe(2142590208) + expect(frames[0]?.completed).toBeUndefined() + }) + + test("one unreadable line costs that line, not the transfer", () => { + const frames = LocalModelPull.readFrames( + ['{"status":"pulling manifest"}', "{ not json", '{"status":"success"}'].join("\n"), + ) + expect(frames.map((frame) => frame.status)).toEqual(["pulling manifest", "success"]) + }) + + test("a whole-body JSON parse would have failed on this stream", () => { + // Guarding the reason the line-by-line reader exists: the runtime does not wrap the + // frames in an array, so anyone "simplifying" this to one parse breaks every pull. + expect(() => JSON.parse(stream)).toThrow() + }) +}) + +describe("readFrame", () => { + test("marks only the success status as done", () => { + const progress = LocalModelPull.readFrame({ status: "pulling manifest" }) + expect("error" in progress ? undefined : progress.done).toBe(false) + const finished = LocalModelPull.readFrame({ status: "success" }) + expect("error" in finished ? undefined : finished.done).toBe(true) + }) + + test("reports layer bytes as layer bytes", () => { + const progress = LocalModelPull.readFrame({ status: "pulling abc", total: 100, completed: 40 }) + if ("error" in progress) throw new Error("unexpected error frame") + // Named for what they are. A caller that read these as whole-model progress would show a + // bar that jumps back to zero on every layer of a multi-layer model. + expect(progress.layerTotalBytes).toBe(100) + expect(progress.layerCompletedBytes).toBe(40) + }) + + test("an in-band error is terminal, not progress", () => { + // A 200 is already sent by the time this arrives. Treated as progress, the pull would + // look like it is still running and never finish. + const progress = LocalModelPull.readFrame({ status: "error", error: "model not found" }) + expect(progress).toEqual({ error: "model not found" }) + }) + + test("an empty error string is not an error", () => { + const progress = LocalModelPull.readFrame({ status: "success", error: "" }) + expect("error" in progress).toBe(false) + }) + + test("passes the runtime's status through rather than re-wording it", () => { + const progress = LocalModelPull.readFrame({ status: "removing any unused layers" }) + if ("error" in progress) throw new Error("unexpected error frame") + expect(progress.status).toBe("removing any unused layers") + }) +}) + +describe("Installed", () => { + // Verbatim from the reference's `GET /api/tags` example. + const entry = { + name: "deepseek-r1:latest", + model: "deepseek-r1:latest", + modified_at: "2025-05-10T08:06:48.639712648-07:00", + size: 4683075271, + digest: "0a8c266910232fd3291e71e5ba1e058cc5af9d411192cf88b6d30e92b6e73163", + details: { format: "gguf", family: "qwen2", parameter_size: "7.6B", quantization_level: "Q4_K_M" }, + } + + test("decodes through the real schema, keeping the size the fit check needs", () => { + const decoded = Schema.decodeUnknownSync(LocalModelPull.Installed)(entry) + expect(decoded.name).toBe("deepseek-r1:latest") + expect(decoded.size).toBe(4683075271) + expect(decoded.digest).toBe(entry.digest) + }) + + test("the size feeds the fit check as bytes, not as anything else", () => { + // 4.68 GB of weights, so the requirement is that plus headroom -- which is what makes a + // 4 GiB machine the wrong machine for it. The two modules meet exactly here. + const decoded = Schema.decodeUnknownSync(LocalModelPull.Installed)(entry) + const verdict = LocalModelFit.fit({ + modelBytes: decoded.size, + hardware: { + platform: "linux", + arch: "x64", + cores: 8, + totalBytes: 4 * 1024 ** 3, + availableBytes: 3 * 1024 ** 3, + }, + }) + if (verdict.kind !== "short") throw new Error("a 4 GiB machine should not fit a 4.68 GB model") + expect(verdict.constraint).toBe("memory") + expect(verdict.requiredBytes).toBe(LocalModelFit.requirementFor(4683075271)) + }) + + test("an entry with no size is dropped rather than treated as free", () => { + // A model of unknown size that defaulted to zero would pass every fit check and then be + // killed mid-load. Refusing to decode it is the honest outcome. + expect(Schema.decodeUnknownOption(LocalModelPull.Installed)({ name: "x" })._tag).toBe("None") + }) +}) diff --git a/packages/redrob/src/cli/cmd/models-local.ts b/packages/redrob/src/cli/cmd/models-local.ts new file mode 100644 index 0000000000..5627069767 --- /dev/null +++ b/packages/redrob/src/cli/cmd/models-local.ts @@ -0,0 +1,134 @@ +import { Effect, Layer } from "effect" +import { FetchHttpClient } from "effect/unstable/http" +import { LayerNode } from "@redrob-code/core/effect/layer-node" +import { AppProcess } from "@redrob-code/core/process" +import { LocalGpuProbe } from "@redrob-code/core/config/plugin/local-gpu-probe" +import { LocalModelFit } from "@redrob-code/core/config/plugin/local-model-fit" +import { LocalModelPull } from "@redrob-code/core/config/plugin/local-model-pull" +import { isLocalEndpoint } from "@redrob-code/core/config/plugin/local-provider" +import { Config } from "@/config/config" +import { effectCmd } from "../effect-cmd" +import { UI } from "../ui" + +/** + * PA-8's surface: what this machine can run, and what it cannot, with the numbers. + * + * The judgement and the probes are libraries with their own tests; this is the one place a + * person can see them. Without it the fit check is code nobody reaches -- the failure this + * queue has now found three times (an arming matcher with no caller, a model preference read + * and discarded, builtin skills registered into a list with no consumer). + * + * A COMMAND rather than a settings page, because the engine has no settings UI of its own and + * the browser's would mean a mojom round trip for a read that is already local. The browser + * can call this later; the numbers are the part that had nowhere to appear. + * + * TOP-LEVEL rather than `models local`, deliberately. `models` takes a POSITIONAL provider + * filter (`models anthropic`), so adding a subcommand under it makes `models local` ambiguous + * -- yargs cannot tell the subcommand from a provider that happens to be called `local`. + * + * NO CLOUD FALLBACK is offered or hinted at anywhere here. A user asking what their own + * machine can run is not asking to be sold a hosted model. + */ + +const GiB = 1024 ** 3 +const gib = (bytes: number): string => `${(bytes / GiB).toFixed(1)} GiB` + +export const ModelsLocalCommand = effectCmd({ + command: "models-local", + describe: "what this machine can run, and what it cannot", + builder: (yargs) => yargs, + handler: Effect.fn("Cli.models.local")( + function* (_args) { + /* + * A failed GPU probe reports NOT MEASURED, never "no GPU". + * + * The probe shells out, and on a machine with no discrete card the tool simply does + * not exist -- the normal state of the world, not a condition to report. The memory + * judgement, which is the part that actually decides whether a model runs, does not + * depend on it. + */ + const gpu = yield* LocalGpuProbe.probe(process.platform).pipe( + Effect.catchCause(() => Effect.succeed(undefined)), + ) + const hardware = LocalModelFit.measure(gpu) + + UI.println(UI.Style.TEXT_NORMAL_BOLD + "This machine" + UI.Style.TEXT_NORMAL) + UI.println(` platform ${hardware.platform} ${hardware.arch}`) + UI.println(` cores ${hardware.cores}`) + UI.println(` memory ${gib(hardware.availableBytes)} free of ${gib(hardware.totalBytes)}`) + // "not measured", never "none". A machine with a card we could not see must not be told + // it has no card -- that is a false statement about the user's own computer. + UI.println( + ` gpu ${ + gpu === undefined + ? "not measured on this platform" + : gpu.vramBytes === undefined + ? gpu.name + : `${gpu.name}, ${gib(gpu.vramBytes)}` + }`, + ) + UI.println("") + + // Only endpoints the local rule already allows. Re-checked here rather than trusted, + // the same way the model listing re-checks it: this command would otherwise be a way to + // make the CLI call an arbitrary host by editing a config file. + const config = yield* (yield* Config.Service).get() + const endpoints = Object.values(config.provider ?? {}) + .map((provider) => (provider as { options?: { baseURL?: string } }).options?.baseURL) + .filter((url): url is string => typeof url === "string" && isLocalEndpoint(url)) + + if (endpoints.length === 0) { + UI.println(UI.Style.TEXT_DIM + "No local model runtime is configured." + UI.Style.TEXT_NORMAL) + // Said plainly, because "no models" and "nowhere to look for models" are different + // facts and a user who sees the first will go looking for a model they already have. + UI.println( + UI.Style.TEXT_DIM + + "Point a provider at one on this machine to see what it is serving." + + UI.Style.TEXT_NORMAL, + ) + return + } + + let sawRuntime = false + for (const endpoint of endpoints) { + const { reachable, models } = yield* LocalModelPull.installed(endpoint) + if (!reachable) { + // Not an error. A runtime that is not running is the normal state of a laptop, and + // the honest line says which endpoint did not answer rather than implying the + // machine cannot run anything. + UI.println(`${endpoint} ${UI.Style.TEXT_DIM}not answering${UI.Style.TEXT_NORMAL}`) + continue + } + sawRuntime = true + UI.println(UI.Style.TEXT_NORMAL_BOLD + endpoint + UI.Style.TEXT_NORMAL) + if (models.length === 0) { + UI.println(` ${UI.Style.TEXT_DIM}answering, with nothing installed${UI.Style.TEXT_NORMAL}`) + continue + } + for (const model of models) { + const verdict = LocalModelFit.fit({ modelBytes: model.size, hardware }) + if (verdict.kind === "fits") { + UI.println(` ${model.name} ${gib(model.size)} runs here`) + continue + } + // The refusal carries its own numbers, so the line is the library's sentence rather + // than a re-worded summary that could disagree with it. + UI.println(` ${model.name} ${gib(model.size)} ${UI.Style.TEXT_DIM}${verdict.reason}${UI.Style.TEXT_NORMAL}`) + } + } + + if (!sawRuntime) { + UI.println("") + UI.println( + UI.Style.TEXT_DIM + + "Every configured local endpoint is down, so nothing could be listed." + + UI.Style.TEXT_NORMAL, + ) + } + }, (effect) => + // Both provided here rather than added to the CLI runtime: this is the only command that + // speaks HTTP to a local runtime or shells out for a GPU, and widening the shared runtime + // for one command would hand both to every other command that has no use for them. + Effect.provide(effect, Layer.mergeAll(FetchHttpClient.layer, LayerNode.compile(AppProcess.node))), + ), +}) diff --git a/packages/redrob/src/index.ts b/packages/redrob/src/index.ts index ad182ba3e8..d7d09963a2 100644 --- a/packages/redrob/src/index.ts +++ b/packages/redrob/src/index.ts @@ -8,6 +8,7 @@ import { AgentCommand } from "./cli/cmd/agent" import { UpgradeCommand } from "./cli/cmd/upgrade" import { UninstallCommand } from "./cli/cmd/uninstall" import { ModelsCommand } from "./cli/cmd/models" +import { ModelsLocalCommand } from "./cli/cmd/models-local" import { UI } from "./cli/ui" import { InstallationVersion } from "@redrob-code/core/installation/version" import { FormatError } from "./cli/error" @@ -91,6 +92,7 @@ const cli = yargs(args) .command(UninstallCommand) .command(ServeCommand) .command(ModelsCommand) + .command(ModelsLocalCommand) .command(StatsCommand) .command(ExportCommand) .command(ImportCommand)