From 5320ac85e13977bd02855c7a501152dc4ba65b6f Mon Sep 17 00:00:00 2001 From: thxCode Date: Mon, 28 Sep 2026 19:06:30 +0800 Subject: [PATCH 1/2] docs: move the model store and model deployment pages into domain dirs Both domains sat flat in docs/reference/, a directory the conventions define as lookup tables with provenance; fourteen pages of contracts, field references, views and a runbook had made it the dumping ground the kv-cache split avoided. They now live in docs/model-store/ and docs/model-deployment/ under short file names whose H1s carry the domain prefix, the way kv-cache pages do; docs/reference/ keeps the four true lookup pages that remain. Every pin on a page path rides along: the owned-key test's _ModelDeploymentDocsPath, the api comment that generation copies into the proto, OpenAPI and applyconfiguration artifacts (regenerated), the docs skill's routing table, page map and conventions, the overview and e2e skills, and every cross-link and label. specs/ records stay untouched, as the docs skill requires of historical records. --- .../skills/gpustack-operator-docs/SKILL.md | 26 +-- .../references/conventions.md | 17 +- .../references/page-map.md | 168 ++++++++++++++---- .agents/skills/gpustack-operator-e2e/SKILL.md | 2 +- .../gpustack-operator-e2e/cases/case-48.sh | 8 +- .../gpustack-operator-e2e/cases/case-59.sh | 2 +- .../gpustack-operator-overview/SKILL.md | 2 +- api/worker/v1alpha1/generated.proto | 2 +- api/worker/v1alpha1/model_deployment.go | 2 +- api/worker/zz_generated.openapi.go | 2 +- docs/README.md | 32 ++-- docs/architecture.md | 4 +- docs/architecture/installation-modes.md | 2 +- docs/architecture/internals.md | 2 +- .../architecture/topology-aware-scheduling.md | 13 +- docs/kv-cache/backend.md | 10 +- docs/kv-cache/pool.md | 4 +- docs/kv-cache/walkthrough.md | 4 +- docs/migration/kv-cache-dtype.md | 2 +- .../deployment.md} | 30 ++-- .../engine-versions.md | 16 +- .../metrics.md} | 14 +- .../prefill-decode.md} | 37 ++-- .../routing.md} | 18 +- .../shutdown.md} | 16 +- .../status.md} | 14 +- .../artifact.md} | 32 ++-- .../image-source.md} | 12 +- .../node-store.md} | 28 +-- .../operations.md} | 20 +-- .../peer-sync.md} | 11 +- .../prefetch.md} | 19 +- .../views.md} | 26 +-- docs/operation/rdma.md | 6 +- docs/operation/topology-aware-scheduling.md | 6 +- docs/reference/commands.md | 2 +- docs/reference/kv-cache-injection.md | 4 +- docs/settings.md | 20 +-- docs/walkthrough.md | 2 +- .../worker/v1alpha1/modeldeploymentrole.go | 2 +- .../worker/model_deployment_docs_test.go | 2 +- 41 files changed, 372 insertions(+), 269 deletions(-) rename docs/{reference/model-deployment.md => model-deployment/deployment.md} (97%) rename docs/{reference => model-deployment}/engine-versions.md (95%) rename docs/{reference/model-deployment-metrics.md => model-deployment/metrics.md} (97%) rename docs/{reference/model-deployment-prefill-decode.md => model-deployment/prefill-decode.md} (92%) rename docs/{reference/model-deployment-routing.md => model-deployment/routing.md} (90%) rename docs/{reference/model-deployment-shutdown.md => model-deployment/shutdown.md} (91%) rename docs/{reference/model-deployment-status.md => model-deployment/status.md} (98%) rename docs/{reference/model-artifact.md => model-store/artifact.md} (93%) rename docs/{reference/model-image-source.md => model-store/image-source.md} (94%) rename docs/{reference/node-model-store.md => model-store/node-store.md} (94%) rename docs/{operation/model-store.md => model-store/operations.md} (95%) rename docs/{reference/node-peer-sync.md => model-store/peer-sync.md} (90%) rename docs/{reference/model-prefetch.md => model-store/prefetch.md} (88%) rename docs/{reference/model-artifact-views.md => model-store/views.md} (89%) diff --git a/.agents/skills/gpustack-operator-docs/SKILL.md b/.agents/skills/gpustack-operator-docs/SKILL.md index 7cd5436fb..a70a4cda4 100644 --- a/.agents/skills/gpustack-operator-docs/SKILL.md +++ b/.agents/skills/gpustack-operator-docs/SKILL.md @@ -40,17 +40,19 @@ one, not to widen the overview. | A `KVCachePool` or `KVCachePoolBinding`: the grant, the reuse domain, a quota ceiling or grant, what a full quota does | `docs/kv-cache/pool.md` | | Standing a cache up end to end, or which object comes first: the pasteable four-object sequence | `docs/kv-cache/walkthrough.md` | | How a **Pod** consumes a pool: the inject label and annotations, the injected keys per engine, a refusal, the isolation record | `docs/reference/kv-cache-injection.md` | -| A `ModelArtifact`: its sources, resolution and revalidation, the manifest digest, how a `ModelDeployment` or an `Instance` mounts or downloads it, claim placement, the weight identity in KV keys | `docs/reference/model-artifact.md` | -| The `image` source of a `ModelArtifact`: the digest contract, building weights into an image, image-volume delivery, the version floors, double storage, kubelet image GC, registry mirrors | `docs/reference/model-image-source.md` | -| A `ModelStore`, `ModelStoreBinding` or `ModelPrefetch`: the grant, the budget, pinning, TTL expiry, the warm-up pod and why it is label-free | `docs/reference/model-prefetch.md` | -| The `v1` views of `ModelArtifact` and `NodeModelStore`, the `progress` subresource and who may read it, the GPUStack server capability map | `docs/reference/model-artifact-views.md` | -| A `NodeModelStore` or the `model-manager` plugin: a field and its writer, the status guard, mount authorization, materialization, a failure reason, collection, a metric | `docs/reference/node-model-store.md` | -| Running node delivery: the chart values, where the node's configuration comes from, reading a node, the watermark cap, switching delivery, where replicas land and turning the preference off, upgrading, removing the cache | `docs/operation/model-store.md` | -| A `ModelDeployment` metrics snapshot, which series each field reads per engine, role and router, cache-hit scope or Pod scrape annotation | `docs/reference/model-deployment-metrics.md` | -| How a prefill role and a decode role are paired: the connector each engine and router renders, `spec.router` and its fields, `spec.kvTransfer`, roles on different hardware, a role's own Service | `docs/reference/model-deployment-prefill-decode.md` | -| Which replica a managed router picks, its default routing policy, switching it through `spec.router.extraArgs`, the router's own per-replica series | `docs/reference/model-deployment-routing.md` | -| What a `ModelDeployment` replica does between its Pod's delete and its engine's exit: the drain hook, its timings, what it does not cover | `docs/reference/model-deployment-shutdown.md` | -| The lowest engine release a deployment shape runs on, the Mooncake client its runner image carries, the store line it needs, which transport each engine can use on each leg | `docs/reference/engine-versions.md` | +| A `ModelArtifact`: its sources, resolution and revalidation, the manifest digest, how a `ModelDeployment` or an `Instance` mounts or downloads it, claim placement, the weight identity in KV keys | `docs/model-store/artifact.md` | +| The `image` source of a `ModelArtifact`: the digest contract, building weights into an image, image-volume delivery, the version floors, double storage, kubelet image GC, registry mirrors | `docs/model-store/image-source.md` | +| A `ModelStore`, `ModelStoreBinding` or `ModelPrefetch`: the grant, the budget, pinning, TTL expiry, the warm-up pod and why it is label-free | `docs/model-store/prefetch.md` | +| The `v1` views of `ModelArtifact` and `NodeModelStore`, the `progress` subresource and who may read it, the GPUStack server capability map | `docs/model-store/views.md` | +| A `NodeModelStore` or the `model-manager` plugin: a field and its writer, the status guard, mount authorization, materialization, a failure reason, collection, a metric | `docs/model-store/node-store.md` | +| Running node delivery: the chart values, where the node's configuration comes from, reading a node, the watermark cap, switching delivery, where replicas land and turning the preference off, upgrading, removing the cache | `docs/model-store/operations.md` | +| The `ModelDeployment` contract: the inherited reuse domain, the three override tiers, the owned-key table, the runner-image formula, prefill/decode pairing, the topology-placement field contract | `docs/model-deployment/deployment.md` | +| A `ModelDeployment` metrics snapshot, which series each field reads per engine, role and router, cache-hit scope or Pod scrape annotation | `docs/model-deployment/metrics.md` | +| What a `ModelDeployment` status condition or published field means, and how to read them when a deployment misbehaves | `docs/model-deployment/status.md` | +| How a prefill role and a decode role are paired: the connector each engine and router renders, `spec.router` and its fields, `spec.kvTransfer`, roles on different hardware, a role's own Service | `docs/model-deployment/prefill-decode.md` | +| Which replica a managed router picks, its default routing policy, switching it through `spec.router.extraArgs`, the router's own per-replica series | `docs/model-deployment/routing.md` | +| What a `ModelDeployment` replica does between its Pod's delete and its engine's exit: the drain hook, its timings, what it does not cover | `docs/model-deployment/shutdown.md` | +| The lowest engine release a deployment shape runs on, the Mooncake client its runner image carries, the store line it needs, which transport each engine can use on each leg | `docs/model-deployment/engine-versions.md` | | A resource key, a request rule, a request example | `docs/accelerator-requests.md` | | How many RDMA endpoints a workload asks for, setting or reading the kubelet TopologyManager policy, what to do about RDMA keys no queue meters, the engine image an EFA leg needs | `docs/operation/rdma.md` | | Enabling Topograph, publishing topology snapshots, webhook trust, requesting a level, TAS diagnosis or EKS validation | `docs/operation/topology-aware-scheduling.md` | @@ -89,7 +91,7 @@ reader's trust. | `deploy/gpustack-operator/chart/README.md`, `values.schema.json` | `values.yaml` + `README.md.gotmpl` via `make generate chart` | generated — never hand-edit; a doc path quoted in a `values.yaml` comment needs a regenerate, and `chart.yml` fails on drift | | `README.md` accelerator matrix | `pkg/nodefeature/knowns.go` (resource names, `SharedResourceMaxSize`, `_ManufacturerPartitionKindMap`) and **whether** each `pkg/devicemanager/detector//device.go` sets `LogicalSliced` at all | nothing fails; the table silently lies about what a vendor can do. The matrix is deliberately ✅/— only — per-card slice counts and the per-vendor isolation mechanism live in `docs/architecture/device-discovery.md`, not on the front page | | `README.md` Quick Start's four request shapes | `docs/accelerator-requests.md` — *The resource keys* and *Worked example per family* | nothing fails; the front page and the normative contract drift apart. This copy is the one sanctioned exception to "state a fact once" (the README is the shop window) — change both together | -| `docs/reference/model-deployment.md` owned-key table | `modelDeploymentOwnedKeys` via `TestModelDeploymentOwnedKeysDocs` | the test matches each owned key inside its engine's **table row**, not anywhere on the page. One-way: a key the code owns must be documented; a name the page merely explains is free. It exists because the code side already had an invariant and the page had none, and the page then omitted a security-relevant key while the webhook refused it | +| `docs/model-deployment/deployment.md` owned-key table | `modelDeploymentOwnedKeys` via `TestModelDeploymentOwnedKeysDocs` | the test matches each owned key inside its engine's **table row**, not anywhere on the page. One-way: a key the code owns must be documented; a name the page merely explains is free. It exists because the code side already had an invariant and the page had none, and the page then omitted a security-relevant key while the webhook refused it | | `docs/settings.md` tables | `pkg/worker/settings` and the `GPUSTACK_*` readers | nothing fails; an operator configures something that no longer exists | | `docs/README.md` page table | the set of files under `docs/` | `check-docs.sh` fails | | `docs/README.md` page **labels** | each page's `#` H1, character for character | `check-docs.sh` fails; a page's file name, H1 and index label are one set of words | diff --git a/.agents/skills/gpustack-operator-docs/references/conventions.md b/.agents/skills/gpustack-operator-docs/references/conventions.md index def0e58cc..1c85d5198 100644 --- a/.agents/skills/gpustack-operator-docs/references/conventions.md +++ b/.agents/skills/gpustack-operator-docs/references/conventions.md @@ -65,10 +65,16 @@ shaped by the directory it lives in: | Directory | H1 form | Example | |---|---|---| | root, `architecture/` | `` | `Installation Modes` | +| a domain directory — `kv-cache/`, `model-store/`, `model-deployment/` | ` ` where the domain reads naturally; a subject that names itself (`Engine Versions`, `Node-to-Node Sync`) stands without it; the domain's runbook keeps the `Operations` suffix | `KV Cache Backend`, `Model Artifact`, `Engine Versions`, `Node-to-Node Sync`, `Model Store Operations` | | `operation/` | ` Operations` | `High Availability Operations` | | `migration/` | `Migrating `; a recovery page is ` Troubleshooting` | `Migrating from v0.5.x`, `Migration Troubleshooting` | | `reference/` | ` Reference` | `Instance Metrics Reference` | +A domain directory collects every page orbiting one CR family — the contracts, the field references, +the views, the runbook — under short file names (`backend.md`, `artifact.md`, `deployment.md`) whose +H1s name the domain where it reads naturally. It exists so `reference/` stays true lookup tables +rather than becoming the dumping ground for whichever domain landed last. + `##` and `###` headings are sentence case. GitHub lowercases anchors, so re-casing a heading keeps every inbound link; changing its *words* does not. @@ -105,10 +111,10 @@ it as follows — when a page starts serving two modes at once, that is the mome | Mode | Reader is… | Our pages | |---|---|---| -| Tutorial | learning by doing | `README.md` Quick Start, `docs/walkthrough.md`, the MIG walkthrough | -| How-to | achieving a goal | `docs/operation/*`, `docs/migration/*`, `docs/development.md` | +| Tutorial | learning by doing | `README.md` Quick Start, `docs/walkthrough.md`, the MIG walkthrough, `docs/kv-cache/walkthrough.md` | +| How-to | achieving a goal | `docs/operation/*`, `docs/migration/*`, `docs/development.md`, a domain runbook (`docs/model-store/operations.md`) | | Reference | looking something up | `docs/accelerator-requests.md`, `docs/settings.md`, `docs/reference/*` | -| Explanation | building understanding | `docs/architecture.md` and `docs/architecture/*` | +| Explanation | building understanding | `docs/architecture.md`, `docs/architecture/*`, and the domain pages under `docs/kv-cache/`, `docs/model-store/` and `docs/model-deployment/` (a domain page serves the mode its reader arrives in — contract pages read as reference, mechanism pages as explanation) | Two consequences worth stating: @@ -158,8 +164,9 @@ Two consequences worth stating: ## Adding a page -1. Put it under the directory of the reader it serves (`architecture/`, `operation/`, `migration/`, - `reference/`). +1. Put it under the directory that fits: the reader it serves (`architecture/`, `operation/`, + `migration/`, `reference/`), or the domain directory of the CR family it orbits + (`kv-cache/`, `model-store/`, `model-deployment/`) when the page joins a family that already has one. 2. Copy the template above; fill the header block honestly — an inflated read time is worse than none. 3. Add a row to the `docs/README.md` page table, and a step to any reading path it belongs on. 4. Add it to the routing table in the skill's `SKILL.md` and to `references/page-map.md`, saying what it diff --git a/.agents/skills/gpustack-operator-docs/references/page-map.md b/.agents/skills/gpustack-operator-docs/references/page-map.md index d7fe3a01d..7dc3fabe5 100644 --- a/.agents/skills/gpustack-operator-docs/references/page-map.md +++ b/.agents/skills/gpustack-operator-docs/references/page-map.md @@ -71,7 +71,7 @@ profiles, generated Kueue Topologies, TAS flavor and queue semantics, per-replic 64-flavor limit, and the fresh-queue and live-profile lifecycle boundaries. **Never** — the commands and manifests an operator follows (`operation/topology-aware-scheduling.md`) -or the full `ModelDeployment` field contract (`reference/model-deployment.md`). Link both. +or the full `ModelDeployment` field contract (`model-deployment/deployment.md`). Link both. ## `docs/architecture/admission.md` @@ -176,6 +176,126 @@ owns it; a second account of `quota.ceiling` or of the snapshot's access modes i apart. If a paragraph here grows past a clause, it belongs on `backend.md`, `leader.md` or `pool.md` and this page should link to it instead. +## `docs/model-store/artifact.md` + +**Owns** — the `ModelArtifact` contract: sources, resolution and revalidation, the manifest digest and +its patterns, how a `ModelDeployment` or an `Instance` consumes it under each delivery, claim +placement, and the weight identity in KV keys. + +**Never** — anything about the deployment that mounts the artifact (that is +`model-deployment/deployment.md`, which keeps one pointer here and states nothing about the artifact +itself), or the node plugin's internals (`node-store.md`). + +## `docs/model-store/image-source.md` + +**Owns** — the `image` source of a `ModelArtifact`: the digest contract and what it does and does not +promise, building weights into an image, image-volume delivery to a `ModelDeployment` or an +`Instance`, the version floors, double storage, kubelet image GC, and registry mirrors. + +**Never** — the other sources or the delivery modes they share (`artifact.md` owns those); this page +is the deep dive on one source. + +## `docs/model-store/prefetch.md` + +**Owns** — the prefetch family: `ModelStore` (pool policy, selector overlap), `ModelStoreBinding` +(budget grant, allowPinned, immutability), `ModelPrefetch` (placement, warm-up pod shape, the +queue-name-label fact, budget projection and accounting, pin union, TTL). + +**Never** — the node plugin's field reference (`node-store.md`), the artifact sources +(`artifact.md`), or the pool layer's operational knobs (`operations.md`). + +## `docs/model-store/node-store.md` + +**Owns** — the `NodeModelStore` resource and the `model-manager` plugin behind it: every field and its +writer, the status guard, mount authorization, materialization, failure reasons, collection and +metrics. + +**Never** — the administrator's procedure (values, Settings, watermarks, switching delivery, +upgrading, removing). That is `operations.md`, which links back. + +## `docs/model-store/peer-sync.md` + +**Owns** — the peer port a node's published trees are served on, the token authentication and +NetworkPolicy that bound it, how a cold node pulls with in-stream checkpoints, and the `source` field +and metrics that tell peer bytes from hub bytes. + +**Never** — how a tree is materialized or verified in the first place (`node-store.md`); this page +owns the node-to-node leg only. + +## `docs/model-store/views.md` + +**Owns** — the `v1` views of `ModelArtifact` and `NodeModelStore`, the `progress` subresource +(aggregate, live, authorized, naming no node), the tenant Role, and the GPUStack server capability +map. + +**Never** — what the underlying fields mean. A view's semantics live on `artifact.md` and +`node-store.md`; this page says how they are read and by whom. + +## `docs/model-store/operations.md` + +**Owns** — the administrator's procedure for node delivery: enabling the plugin, its configuration +layers, reading a node, the watermark cap against kubelet's thresholds, switching delivery, where +replicas land and turning the placement preference off, the upgrade notes, and removing the cache. + +**Rule** — a page with a `## Verify` block states the expected output of every command in it. Unlike +`docs/operation/*`, this runbook is not line-cap exempt — split it if it grows past the cap. + +## `docs/model-deployment/deployment.md` + +**Owns** — the `ModelDeployment` contract: the inherited reuse domain, the three override tiers, the +owned-key table, the runner-image formula, prefill/decode pairing, the topology-placement field +contract, and what turns a replica over. + +**Never** — the artifact it mounts (one pointer to `artifact.md`), the drain window +(`shutdown.md`), the router's own behavior (`routing.md`), or the status views +(`status.md`). Link each. + +## `docs/model-deployment/prefill-decode.md` + +**Owns** — what pairs a prefill role with a decode role: the connector each engine and router +renders, the router block and its fields, the direct transfer's transport, roles on different +hardware, and a role's own Service. + +**Never** — the deployment-level contract around the pair (`deployment.md`) or the KV cache leg's +transport (`../kv-cache/backend.md`); this page owns the P/D leg. + +## `docs/model-deployment/routing.md` + +**Owns** — which replica each managed router picks and how to switch its policy. + +**Pinned** — its defaults and `--policy` values are read from each router's source at the version +`pack/llm-router/Dockerfile` pins, so bumping one of those `ARG`s is a re-read of that page. + +## `docs/model-deployment/metrics.md` + +**Owns** — the structured metrics subresource: which series each field reads per engine, role and +router, windowed cache hits, and the Pod scrape annotations. + +**Never** — the exporter's own gauges (`../reference/instance-metrics.md` owns the exporter side). + +## `docs/model-deployment/status.md` + +**Owns** — what each `ModelDeployment` status condition and published field means, and how to read +them when a deployment misbehaves. + +**Never** — the lifecycle rules the conditions report (`deployment.md` owns those). + +## `docs/model-deployment/shutdown.md` + +**Owns** — the drain window and what a replica does between its Pod's delete and its engine's exit, +including what it does not cover. + +**Never** — what turns a replica over in the first place. That stays on `deployment.md`, which links +here. + +## `docs/model-deployment/engine-versions.md` + +**Owns** — the lowest engine release each deployment shape has been run with, the Mooncake client its +runner image carries, the store line it needs, and which transport each engine can use on each leg. + +**Rule** — the one place an engine's minimum is stated. A paragraph elsewhere that explains why a +release below it fails is collapsed into a link to it, not kept beside it. + ## `docs/accelerator-requests.md` **Owns** — the normative contract: the two families, every resource key, a worked example per family, @@ -241,10 +361,8 @@ manage — the NVIDIA MIG runbook, and `thead-mig.md` — MIG is T-Head's own wo partitioning, as `hgml.GetMigMode()` and the `alibabacloud.com/ppu.partitioned.mig-` key both show, so the page is named for it too. And `rdma.md`, which is a how-to on both sides of one workflow: the request a workload writes, and the kubelet policy an administrator sets so that -request aligns — split apart, each half reads as a guarantee the other half withholds. And -`model-store.md`: enabling the model-manager plugin, its configuration layers, reading a node, the -watermark cap against kubelet's thresholds, switching delivery, and the upgrade notes of node -delivery. +request aligns — split apart, each half reads as a guarantee the other half withholds. The model +cache's runbook lives with its domain, at `model-store/operations.md`. **Rule** — a page with a `## Verify` block states the expected output of every command in it. These pages are exempt from the line cap: a runbook is as long as the hardware makes it. @@ -260,48 +378,26 @@ transition, so do not "modernize" its version numbers. ## `docs/reference/*.md` **Owns** — lookup tables with provenance. Today: the per-product unit-resources presets -(`instance-type-unit-resources.md`), every command the binary offers with its flags and exit codes -(`commands.md`), and the KV cache injection contract — opt-in keys, what each engine receives, every -refusal with its fix (`kv-cache-injection.md`), and the lowest engine release each deployment -shape has been run with, with its runner image's Mooncake client and store line -(`engine-versions.md`), what pairs a prefill role with a decode role — the router block, the direct transfer's transport, -roles on different hardware (`model-deployment-prefill-decode.md`), which replica each managed router -picks and how to switch its policy (`model-deployment-routing.md`), what a `ModelDeployment` replica does between its Pod's delete -and its engine's exit (`model-deployment-shutdown.md`), and the `ModelArtifact` contract — sources, -resolution and revalidation, the manifest digest and its patterns, how a deployment or an Instance -consumes it under each delivery, claim placement and the weight identity in KV keys -(`model-artifact.md`), the `image` source's own page — the digest contract, building weights into -an image, image-volume delivery, the version floors, double storage, kubelet image GC and registry -mirrors (`model-image-source.md`), and the `NodeModelStore` resource with the node plugin behind it — its -writers and status guard, mount authorization, materialization, failure reasons, collection and -metrics (`node-model-store.md`). - -**Not** — `model-deployment-shutdown.md` owns the drain window and what it does not cover; what turns -a replica over in the first place stays on `model-deployment.md`, which links to it. -`commands.md` states what a flag does, not when to reach for the command. The procedure a +(`instance-type-unit-resources.md`), the `instances//metrics` subresource and the Device +Manager's Prometheus exporter behind it (`instance-metrics.md`), every command the binary offers with +its flags and exit codes (`commands.md`), and the KV cache injection contract — opt-in keys, what +each engine receives, every refusal with its fix (`kv-cache-injection.md`). The model store and model +deployment domains have their own directories and their own sections above. + +**Not** — `commands.md` states what a flag does, not when to reach for the command. The procedure a one-shot belongs to lives on its operator page (`docs/operation/preflight.md` for `device-manager preflight`), and the reference row links to it rather than restating it. -`model-artifact.md` owns everything about weights delivery; `model-deployment.md` keeps one pointer to -it from its minimal deployment and states nothing about the artifact itself. `node-model-store.md` -owns what the node plugin does and reports; the administrator's procedure for it (values, Settings, -watermarks, switching, upgrading, removing) is on `docs/operation/model-store.md`, which links back. - **Pinned** — `kv-cache-injection.md` carries per-engine facts read from engine source at named versions. Those rows go stale silently when an engine ships a new build, so a change there is a re-read rather than an edit; the same facts are mirrored in `pkg/worker/kvcache/inject/engine.go`, -which carries the line numbers to re-read from. `model-deployment-routing.md` is the same kind of -page for the routers: its defaults and `--policy` values are read from each router's source at the -version `pack/llm-router/Dockerfile` pins, so bumping one of those `ARG`s is a re-read of that page. +which carries the line numbers to re-read from. `instance-type-unit-resources.md` is matched row-by-row by `TestUnitResourcesPresetDocs`, by path. Do not rename it or reshape its tables. `commands.md` has no test behind it: its flag tables are only as true as the last person who ran `--help`, so change a flag and change the row in the same commit. -`engine-versions.md` is the one place an engine's minimum is stated. A paragraph elsewhere that -explains why a release below it fails is collapsed into a link to it, not kept beside it. - **Rule** — these pages are exempt from the ten-`##` cap: a lookup page is meant to be flat. ## `specs/` — not documentation @@ -316,5 +412,3 @@ supersession is recorded that way because it has no new spec to be recorded in. `**Corrected after shipping.**`. A spec naming another spec by file name is the case that forces one: when that file leaves the tree the name resolves to nothing, and leaving it alone preserves a pointer rather than a record. - -- `docs/reference/model-prefetch.md` owns the prefetch family: `ModelStore` (pool policy, selector overlap), `ModelStoreBinding` (budget grant, allowPinned, immutability), `ModelPrefetch` (placement, warm-up pod shape, the queue-name-label fact, budget projection and accounting, pin union, TTL). It must not absorb the node plugin's field reference (node-model-store.md), the artifact sources (model-artifact.md), or the pool layer's operational knobs (operation/model-store.md). diff --git a/.agents/skills/gpustack-operator-e2e/SKILL.md b/.agents/skills/gpustack-operator-e2e/SKILL.md index 6892ba459..d8724e868 100644 --- a/.agents/skills/gpustack-operator-e2e/SKILL.md +++ b/.agents/skills/gpustack-operator-e2e/SKILL.md @@ -155,7 +155,7 @@ Each case is self-contained; its header (see **Case header contract**) states go | 108 | Download progress: a node writes downloadedBytes on 30 s / 5% thresholds and nothing at rest, the artifact's status.nodes moves in steps of 5, the v1 progress subresource reads the download live, only to its namespace's subjects with the subresource rule and naming no node, no NodeModelStore names a tenant, and the v1 views write nothing as the worker | `pkg/modelmanager/{report,materialize,download}/**`, `pkg/modelmanager/downloads.go`, `pkg/worker/controllers/worker/model_artifact{,_nodes}.go`, `pkg/worker/extensionapis/worker/{model_artifact,model_artifact.progress,node_model_store}.go`, `cases/_model-hub{-lib.sh,.py}` | yes (confirm) | a worker, the chart with `modelManager.enabled`, the stock python image; about ten minutes (a three-minute download, then a five-minute quiet window) | | 109 | The NodeModelStore's lifecycle: a plugin restart and a deletion of the object each end with the node's entries rewritten from disk, and deleting a Node collects its object while the node registered again gets a new one | `pkg/modelmanager/{report,store}/**`, `pkg/worker/controllers/worker/node_model_store{,_kubelet}.go` | yes (confirm; deletes a worker's Node object and restarts its kubelet) | kind with two workers and docker reaching the node containers (else the Node row SKIPs), the chart with `modelManager.enabled`, the stock python image | | 110 | ModelPrefetch: the store layer names itself in the node's spec; a warm-up pod pinned to the target node mounts label-free and the digest goes Ready; a projection past the grant is refused at admission; pinning is refused without `allowPinned` and lands under it; deleting the prefetch removes its pod and its pin while the tree stays for the grace | `pkg/worker/controllers/worker/{model_prefetch,model_store,model_store_binding}.go`, `pkg/worker/webhooks/worker/{model_store_binding,model_prefetch}.go`, `pkg/worker/settings/value.go`, `cases/_model-hub{-lib.sh,.py}` | yes (confirm) | one worker labeled `e2e.gpustack.ai/warm-pool=true` by the case (case-110 labels and unlabels it), the chart with `modelManager.enabled`, the stock python image; docker reaching the node container for the tree row | -| 111 | Node-to-node sync: a cold node materializes a digest from a peer's published tree with source=Peer and its hub byte counter at zero; the seed's plugin pod dying mid-pull still ends Ready (checkpoint resume); a tenant pod cannot reach the peer port | `pkg/modelmanager/peer/**`, `pkg/modelmanager/materialize/materialize.go`, `pkg/modelmanager/config.go`, `deploy/gpustack-operator/chart/templates/model-manager/**`, `docs/reference/node-peer-sync.md` | yes (confirm) | two schedulable workers (a seed and a puller), the chart installed with `modelManager.port` non-zero and the peer NetworkPolicy enabled, `model-store-peer-sync` at its `true` default | +| 111 | Node-to-node sync: a cold node materializes a digest from a peer's published tree with source=Peer and its hub byte counter at zero; the seed's plugin pod dying mid-pull still ends Ready (checkpoint resume); a tenant pod cannot reach the peer port | `pkg/modelmanager/peer/**`, `pkg/modelmanager/materialize/materialize.go`, `pkg/modelmanager/config.go`, `deploy/gpustack-operator/chart/templates/model-manager/**`, `docs/model-store/peer-sync.md` | yes (confirm) | two schedulable workers (a seed and a puller), the chart installed with `modelManager.port` non-zero and the peer NetworkPolicy enabled, `model-store-peer-sync` at its `true` default | | 112 | Image source: a tag-only reference is refused naming the digest contract; an image artifact resolves claim-shaped (no manifest digest, no revision, no `status.nodes`); an Instance pinned to the node mounts the pinned image read-only through an image volume and reads the fixture byte for byte; a `ModelPrefetch` naming the artifact is refused | `pkg/worker/webhooks/worker/{model_artifact,model_prefetch}.go`, `pkg/worker/controllers/worker/{model_artifact_placement,model_deployment_artifact,instance,model_placement_preference}.go`, `pkg/kubediscovery/feature.go`, `cases/case-112.sh` | yes (confirm) | one worker whose kubelet/containerd serve image volumes (kubelet 1.35+, containerd 2.1+), docker on the runner, the stock `registry:2`, `python:3.12-slim` and `busybox:1.36` images pullable | Each note below is something the **lead** must act on before or around a run. What a case *does* — its goal, environment, inputs, assertions and cleanup — lives in its own header, which the **Case header contract** below requires to be readable on its own; the index never restates it. diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-48.sh b/.agents/skills/gpustack-operator-e2e/cases/case-48.sh index 0ca943bdb..094233359 100644 --- a/.agents/skills/gpustack-operator-e2e/cases/case-48.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-48.sh @@ -264,10 +264,10 @@ metadata: spec: engine: name: vLLM - # The minimum vLLM docs/reference/engine-versions.md supports. Inert here -- the role names its - # image explicitly, so nothing is synthesized from it -- but a version no runner ships, or one - # below that minimum, would read as the version under test, and this case tests nothing about - # an engine. + # The minimum vLLM docs/model-deployment/engine-versions.md supports. Inert here -- the + # role names its image explicitly, so nothing is synthesized from it -- but a version no + # runner ships, or one below that minimum, would read as the version under test, and + # this case tests nothing about an engine. version: "0.29.0" model: name: Qwen/Qwen2.5-0.5B-Instruct diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-59.sh b/.agents/skills/gpustack-operator-e2e/cases/case-59.sh index 0b427414a..2d9e85479 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-59.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-59.sh @@ -81,7 +81,7 @@ # # E2E_VLLM_IMAGE=gpustack/runner:cuda12.9-vllm0.29.0 # 0.29.0 because that is the version engine.go's facts table was READ at, and the -# lowest vLLM docs/reference/engine-versions.md supports. This case +# lowest vLLM docs/model-deployment/engine-versions.md supports. This case # proves an engine accepts what we render, and what we render was decided from that # version's source; testing a different one silently changes the question. There is # also a hard FLOOR of 0.21.1: vLLM's Mooncake store connector - the module holding diff --git a/.agents/skills/gpustack-operator-overview/SKILL.md b/.agents/skills/gpustack-operator-overview/SKILL.md index 39dd4ead5..c4e8fcfd0 100644 --- a/.agents/skills/gpustack-operator-overview/SKILL.md +++ b/.agents/skills/gpustack-operator-overview/SKILL.md @@ -93,6 +93,6 @@ controller uses via `WithIndex` — see the `*_test.go` beside each reconciler. - Settings & `GPUSTACK_*` configuration knobs → [settings.md](../../../docs/settings.md) - Every command the binary offers, its flags and a runnable invocation → [reference/commands.md](../../../docs/reference/commands.md) - Checking a node can slice before it has to: the procedure → [operation/preflight.md](../../../docs/operation/preflight.md) -- Node delivery of weights: the plugin and `NodeModelStore` → [reference/node-model-store.md](../../../docs/reference/node-model-store.md); running it → [operation/model-store.md](../../../docs/operation/model-store.md); the `v1` views and `progress` → [reference/model-artifact-views.md](../../../docs/reference/model-artifact-views.md) +- Node delivery of weights: the plugin and `NodeModelStore` → [model-store/node-store.md](../../../docs/model-store/node-store.md); running it → [model-store/operations.md](../../../docs/model-store/operations.md); the `v1` views and `progress` → [model-store/views.md](../../../docs/model-store/views.md) - Build / lint / test / codegen / vendored deps → [development.md](../../../docs/development.md) - Writing or updating any of the above → the `gpustack-operator-docs` skill diff --git a/api/worker/v1alpha1/generated.proto b/api/worker/v1alpha1/generated.proto index 580e3a006..657dfbbb3 100644 --- a/api/worker/v1alpha1/generated.proto +++ b/api/worker/v1alpha1/generated.proto @@ -3510,7 +3510,7 @@ message ModelDeploymentPort { // // A DEPARTURE THIS OPERATOR DID NOT INITIATE IS NOT A ROLLOUT. The replica that left is replaced on // its own, under a new name, while its siblings keep serving — see -// docs/reference/model-deployment.md under "One group per replica" and "Rollout is a rolling +// docs/model-deployment/deployment.md under "One group per replica" and "Rollout is a rolling // replacement". message ModelDeploymentRole { // Name identifies the role, and it is also the name of the Kueue PodSet the role becomes. diff --git a/api/worker/v1alpha1/model_deployment.go b/api/worker/v1alpha1/model_deployment.go index a2a08954a..6740b6d16 100644 --- a/api/worker/v1alpha1/model_deployment.go +++ b/api/worker/v1alpha1/model_deployment.go @@ -325,7 +325,7 @@ type ModelDeploymentKVTransfer struct { // // A DEPARTURE THIS OPERATOR DID NOT INITIATE IS NOT A ROLLOUT. The replica that left is replaced on // its own, under a new name, while its siblings keep serving — see -// docs/reference/model-deployment.md under "One group per replica" and "Rollout is a rolling +// docs/model-deployment/deployment.md under "One group per replica" and "Rollout is a rolling // replacement". type ModelDeploymentRole struct { // Name identifies the role, and it is also the name of the Kueue PodSet the role becomes. diff --git a/api/worker/zz_generated.openapi.go b/api/worker/zz_generated.openapi.go index 1ee69526c..83a9b0106 100644 --- a/api/worker/zz_generated.openapi.go +++ b/api/worker/zz_generated.openapi.go @@ -8995,7 +8995,7 @@ func schema_gpustack_api_worker_v1alpha1_ModelDeploymentRole(ref common.Referenc return common.OpenAPIDefinition{ Schema: spec.Schema{ SchemaProps: spec.SchemaProps{ - Description: "ModelDeploymentRole is one engine role and its replicas.\n\nReplicas, InstanceType and Resources are STRUCTURED FIELDS AND MUST STAY SO. They are inputs to admission and scheduling — Kueue PodSet counts, flavor selection and the request the queue accounts — so a container field able to shadow any of them would make the admission feasibility check read a ledger that does not match reality. That is why the container fields below carry no resource request at all: the accelerator half belongs in Resources and the rest is derived from the InstanceType, and neither can be overridden here.\n\nEDITING A CONTAINER FIELD ROLLS THIS ROLE'S REPLICAS, and only this role's -- with one exception. The declared parallel degrees of a prefill or decode role -- the degree flags in ExtraArgs, and vLLM's VLLM_DP_SIZE environment entry -- are the one container field that can render into a document BOTH roles carry: the vLLM-Ascend transfer leg writes the same parallel blocks into both roles' Pods, so on that leg editing one role's degrees rewrites the other role's Pods too, and the edit rolls the pair. Where no shared document renders them, a degree edit stays this role's own like every other container-field edit: every replica is a Kueue pod group of its own, so they are replaced one at a time -- one per role per pass -- and every sibling role keeps serving throughout. A `replicas` change rolls nothing at all: it adds or removes instances, and every instance that stays keeps running, keeps the accelerators it was admitted with and keeps whatever cache it holds.\n\nTHE SET OF ROLES IS FIXED AFTER CREATION. Admission refuses adding, removing or renaming a role. A replica's group is named from the deployment, its role and its ordinal, so edits to one role's running configuration do not rename a sibling role's groups.\n\nA DEPARTURE THIS OPERATOR DID NOT INITIATE IS NOT A ROLLOUT. The replica that left is replaced on its own, under a new name, while its siblings keep serving — see docs/reference/model-deployment.md under \"One group per replica\" and \"Rollout is a rolling replacement\".", + Description: "ModelDeploymentRole is one engine role and its replicas.\n\nReplicas, InstanceType and Resources are STRUCTURED FIELDS AND MUST STAY SO. They are inputs to admission and scheduling — Kueue PodSet counts, flavor selection and the request the queue accounts — so a container field able to shadow any of them would make the admission feasibility check read a ledger that does not match reality. That is why the container fields below carry no resource request at all: the accelerator half belongs in Resources and the rest is derived from the InstanceType, and neither can be overridden here.\n\nEDITING A CONTAINER FIELD ROLLS THIS ROLE'S REPLICAS, and only this role's -- with one exception. The declared parallel degrees of a prefill or decode role -- the degree flags in ExtraArgs, and vLLM's VLLM_DP_SIZE environment entry -- are the one container field that can render into a document BOTH roles carry: the vLLM-Ascend transfer leg writes the same parallel blocks into both roles' Pods, so on that leg editing one role's degrees rewrites the other role's Pods too, and the edit rolls the pair. Where no shared document renders them, a degree edit stays this role's own like every other container-field edit: every replica is a Kueue pod group of its own, so they are replaced one at a time -- one per role per pass -- and every sibling role keeps serving throughout. A `replicas` change rolls nothing at all: it adds or removes instances, and every instance that stays keeps running, keeps the accelerators it was admitted with and keeps whatever cache it holds.\n\nTHE SET OF ROLES IS FIXED AFTER CREATION. Admission refuses adding, removing or renaming a role. A replica's group is named from the deployment, its role and its ordinal, so edits to one role's running configuration do not rename a sibling role's groups.\n\nA DEPARTURE THIS OPERATOR DID NOT INITIATE IS NOT A ROLLOUT. The replica that left is replaced on its own, under a new name, while its siblings keep serving — see docs/model-deployment/deployment.md under \"One group per replica\" and \"Rollout is a rolling replacement\".", Type: []string{"object"}, Properties: map[string]spec.Schema{ "name": { diff --git a/docs/README.md b/docs/README.md index d00ba4817..498093514 100644 --- a/docs/README.md +++ b/docs/README.md @@ -22,8 +22,8 @@ Everything written about GPUStack Operator, and the order to read it in. Start a inventory, publish a hierarchy, and verify Kueue TAS. 4. [High Availability Operations](operation/high-availability.md) — the replica knob per component. 5. [Settings & Environment Variables](settings.md) — online-adjustable settings and every `GPUSTACK_*`. - [Model Store Operations](operation/model-store.md) — the node model cache: its watermarks, which - delivery a Hugging Face model takes, and the upgrade that switches it. + [Model Store Operations](model-store/operations.md) — the node model cache: its watermarks, + which delivery a Hugging Face model takes, and the upgrade that switches it. 6. [Vendor Prerequisites](vendor-prerequisites.md) — what to install per manufacturer, and which vendor GPU Operator components to disable. 7. [Preflight Operations](operation/preflight.md) — one container run that says what a node can @@ -73,12 +73,25 @@ Everything written about GPUStack Operator, and the order to read it in. Start a | [KV Cache on Disk-Heavy Nodes](kv-cache/disk-heavy-nodes.md) | What to configure on a node whose capacity is disk rather than memory: why no member group can be disk alone, and how small its memory segment may be | operators | ~5 min | | [KV Cache Pool](kv-cache/pool.md) | How a namespace is granted a quota on a store, what a quota ceiling buys, and why a full quota discards data instead of refusing writes | operators, contributors | ~12 min | | [KV Cache Walkthrough](kv-cache/walkthrough.md) | The shortest path from nothing to a ModelDeployment on a shared cache: the four objects in order, the pasteable manifests, and the three things that go wrong | operators, users | ~11 min | +| [Model Artifact](model-store/artifact.md) | How a `ModelArtifact` names a model's weights, how it is resolved and revalidated, the manifest digest, file patterns, and how a `ModelDeployment` or an `Instance` mounts or downloads them under each delivery | users, operators | ~16 min | +| [Model Image Source](model-store/image-source.md) | The `ModelArtifact` image source: the digest-pinned reference and what it does and does not promise, building weights into an image, image-volume delivery to a `ModelDeployment` or an `Instance`, the version floors, double storage, kubelet image GC, and registry mirrors | users, operators | ~8 min | +| [Model Prefetch](model-store/prefetch.md) | Warming a model onto nodes before any Pod asks: the `ModelStore` / `ModelStoreBinding` / `ModelPrefetch` objects, placement, the warm-up pod and why it carries no queue-name label, budgets, pinning and expiry, and the status views | users, operators | ~9 min | +| [Node Model Store](model-store/node-store.md) | The `NodeModelStore` resource and the `model-manager` plugin: every field and its writer, the status guard, mount authorization, how content is downloaded, verified and published, failure reasons, collection and metrics | operators, contributors | ~12 min | +| [Node-to-Node Sync](model-store/peer-sync.md) | The peer port a node's published trees are served on, the token authentication and NetworkPolicy that bound it, how a cold node pulls with in-stream checkpoints, and the `source` field and metrics that tell peer bytes from hub bytes | operators, contributors | ~6 min | +| [Model Artifact Views](model-store/views.md) | The `v1` views of `ModelArtifact` and `NodeModelStore`, the `progress` subresource (aggregate, live, authorized, naming no node), the tenant Role, and the capability map for GPUStack server's model files | users, operators, console developers | ~8 min | +| [Model Store Operations](model-store/operations.md) | Running node delivery: enabling the plugin, where its configuration comes from, reading a node, the capacity rule against kubelet's eviction, switching delivery, where replicas land and turning the placement preference off, upgrade notes, and removing the cache | operators | ~11 min | +| [Model Deployment](model-deployment/deployment.md) | The `ModelDeployment` contract: the inherited reuse domain, the three override tiers and the owned-key table, and the runner-image formula | users, operators, contributors | ~9 min | +| [Model Deployment Prefill and Decode](model-deployment/prefill-decode.md) | What pairs a prefill role with a decode role: the connector each engine and router renders, the router block and its fields, the direct transfer's transport, roles on different hardware, and a role's own address | users, operators, contributors | ~4 min | +| [Engine Versions](model-deployment/engine-versions.md) | The lowest vLLM, vLLM-Ascend and SGLang release each deployment shape has been run with, the Mooncake client its runner image carries, the store line it needs, and which transport each engine can use on each leg | users, operators | reference | +| [Model Deployment Routing](model-deployment/routing.md) | Which replica each managed router picks by default, why shared-prefix traffic lands on one replica, switching `vllm-router` or `sglang-gateway` to round robin through `spec.router.extraArgs`, and the router series that show which replica served a request | users, operators | ~5 min | +| [Model Deployment Metrics](model-deployment/metrics.md) | The structured metrics subresource, which series each field reads per engine, role and router, windowed cache hits and Pod scrape annotations | users, operators, console developers | ~7 min | +| [Model Deployment Status](model-deployment/status.md) | What each `ModelDeployment` status condition and published field means, and how to read them when a deployment misbehaves | users, operators | ~8 min | +| [Model Deployment Shutdown](model-deployment/shutdown.md) | What a replica does between its Pod's delete and its engine's exit: the drain window, the gauges it waits on, and the requests it does not save | users, operators | ~4 min | | [Accelerator Requests](accelerator-requests.md) | The resource keys per family and the seven rules admission enforces, with worked examples | users, contributors | ~11 min | | [Walkthrough](walkthrough.md) | A recorded end-to-end run: every object, before/after each operation | everyone | ~12 min | | [Settings & Environment Variables](settings.md) | Online-adjustable settings, every `GPUSTACK_*` env, per-manufacturer overrides, toolkit paths | operators | ~8 min | | [Vendor Prerequisites](vendor-prerequisites.md) | What to install per manufacturer before GPUStack, and which vendor GPU Operator components to keep or disable | operators | ~10 min | | [Development](development.md) | Build, lint, test, code generation, vendored subcharts and dependencies | contributors | ~6 min | -| [Model Store Operations](operation/model-store.md) | Running node delivery: enabling the plugin, where its configuration comes from, reading a node, the capacity rule against kubelet's eviction, switching delivery, where replicas land and turning the placement preference off, upgrade notes, and removing the cache | operators | ~11 min | | [High Availability Operations](operation/high-availability.md) | Which knob to raise per control-plane component, and the one topology that cannot be redundant | operators | ~4 min | | [Topology-Aware Scheduling Operations](operation/topology-aware-scheduling.md) | Enabling Topograph or publishing generic inventory, requesting a level, verifying TAS, and diagnosing Pending groups | operators, users | ~16 min | | [NVIDIA MIG Operations](operation/nvidia-mig.md) | Enabling/disabling MIG, reboot recovery, and a recorded three-configuration walkthrough | operators | ~21 min | @@ -94,19 +107,6 @@ Everything written about GPUStack Operator, and the order to read it in. Start a | [Instance Metrics Reference](reference/instance-metrics.md) | The `instances//metrics` subresource and the Device Manager's Prometheus exporter: the fields, the gauges, their sources and limits | users, operators, console developers | ~9 min | | [Command Reference](reference/commands.md) | Every command the binary offers: what each does, who runs it, its flags, and a runnable invocation | operators, developers | ~10 min | | [KV Cache Injection Reference](reference/kv-cache-injection.md) | How any Pod joins a KV cache pool with one label: the contract, what is injected per engine, every refusal, and what a cache changes about a workload | users, operators | ~14 min | -| [Model Deployment Reference](reference/model-deployment.md) | The `ModelDeployment` contract: the inherited reuse domain, the three override tiers and the owned-key table, and the runner-image formula | users, operators, contributors | ~9 min | -| [Model Artifact Reference](reference/model-artifact.md) | How a `ModelArtifact` names a model's weights, how it is resolved and revalidated, the manifest digest, file patterns, and how a `ModelDeployment` or an `Instance` mounts or downloads them under each delivery | users, operators | ~16 min | -| [Model Artifact Views Reference](reference/model-artifact-views.md) | The `v1` views of `ModelArtifact` and `NodeModelStore`, the `progress` subresource (aggregate, live, authorized, naming no node), the tenant Role, and the capability map for GPUStack server's model files | users, operators, console developers | ~8 min | -| [Model Image Source Reference](reference/model-image-source.md) | The `ModelArtifact` image source: the digest-pinned reference and what it does and does not promise, building weights into an image, image-volume delivery to a `ModelDeployment` or an `Instance`, the version floors, double storage, kubelet image GC, and registry mirrors | users, operators | ~8 min | -| [Model Prefetch Reference](reference/model-prefetch.md) | Warming a model onto nodes before any Pod asks: the `ModelStore` / `ModelStoreBinding` / `ModelPrefetch` objects, placement, the warm-up pod and why it carries no queue-name label, budgets, pinning and expiry, and the status views | users, operators | ~9 min | -| [Node Model Store Reference](reference/node-model-store.md) | The `NodeModelStore` resource and the `model-manager` plugin: every field and its writer, the status guard, mount authorization, how content is downloaded, verified and published, failure reasons, collection and metrics | operators, contributors | ~12 min | -| [Node-to-Node Sync Reference](reference/node-peer-sync.md) | The peer port a node's published trees are served on, the token authentication and NetworkPolicy that bound it, how a cold node pulls with in-stream checkpoints, and the `source` field and metrics that tell peer bytes from hub bytes | operators, contributors | ~6 min | -| [Model Deployment Prefill and Decode Reference](reference/model-deployment-prefill-decode.md) | What pairs a prefill role with a decode role: the connector each engine and router renders, the router block and its fields, the direct transfer's transport, roles on different hardware, and a role's own address | users, operators, contributors | ~4 min | -| [Engine Versions Reference](reference/engine-versions.md) | The lowest vLLM, vLLM-Ascend and SGLang release each deployment shape has been run with, the Mooncake client its runner image carries, the store line it needs, and which transport each engine can use on each leg | users, operators | reference | -| [Model Deployment Routing Reference](reference/model-deployment-routing.md) | Which replica each managed router picks by default, why shared-prefix traffic lands on one replica, switching `vllm-router` or `sglang-gateway` to round robin through `spec.router.extraArgs`, and the router series that show which replica served a request | users, operators | ~5 min | -| [Model Deployment Metrics Reference](reference/model-deployment-metrics.md) | The structured metrics subresource, which series each field reads per engine, role and router, windowed cache hits and Pod scrape annotations | users, operators, console developers | ~7 min | -| [Model Deployment Status Reference](reference/model-deployment-status.md) | What each `ModelDeployment` status condition and published field means, and how to read them when a deployment misbehaves | users, operators | ~8 min | -| [Model Deployment Shutdown Reference](reference/model-deployment-shutdown.md) | What a replica does between its Pod's delete and its engine's exit: the drain window, the gauges it waits on, and the requests it does not save | users, operators | ~4 min | ## Conventions diff --git a/docs/architecture.md b/docs/architecture.md index e2d6018f6..7906029b8 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -31,7 +31,7 @@ chart. Topograph is also vendored, but stays disabled until an administrator sel | `worker` (alias `w`) | `pkg/worker` | this chart, as a control-plane Deployment | aggregated extension API server + the scheduling-chain controllers | | `worker-gateway` | `pkg/workergateway` | not this chart; run it yourself, wherever the fleet view belongs | aggregates InstanceTypes and capacity across upstream clusters | | `device-manager` | `pkg/devicemanager` | this chart, as one DaemonSet per manufacturer | detects accelerators, maintains the `Devices` ledger, serves the device plugin | -| `model-manager` (alias `mm`) | `pkg/modelmanager` | this chart, as one DaemonSet on every node | the CSI node plugin that mounts a Hugging Face `ModelArtifact` from the node's verified cache ([Node Model Store Reference](reference/node-model-store.md)) | +| `model-manager` (alias `mm`) | `pkg/modelmanager` | this chart, as one DaemonSet on every node | the CSI node plugin that mounts a Hugging Face `ModelArtifact` from the node's verified cache ([Node Model Store](model-store/node-store.md)) | Details, and the startup ordering the worker must keep, are in [Internals](architecture/internals.md). @@ -81,7 +81,7 @@ cannot: | Step | What happens | Detail | |---|---|---| -| Submit | a Pod — plain, or rendered by a GPUStack `Instance` or by a [`ModelDeployment`](reference/model-deployment.md) replica — carries the pool's entrance label `kueue.x-k8s.io/queue-name: gpustack-fnv64-…` and requests `nvidia.com/gpu.sliced: 1` + `nvidia.com/gpu.sliced.memory-percentage: 50` | [Accelerator Requests](accelerator-requests.md) | +| Submit | a Pod — plain, or rendered by a GPUStack `Instance` or by a [`ModelDeployment`](model-deployment/deployment.md) replica — carries the pool's entrance label `kueue.x-k8s.io/queue-name: gpustack-fnv64-…` and requests `nvidia.com/gpu.sliced: 1` + `nvidia.com/gpu.sliced.memory-percentage: 50` | [Accelerator Requests](accelerator-requests.md) | | Gate 1 — Pod webhook | validates the request rules and folds the memory budget into `nvidia.com/gpu.sliced.units`, the credit input | [Admission](architecture/admission.md#gate-1--the-pod-webhook) | | Gate 2 — Kueue | reserves against the pool ClusterQueue's `credits.gpustack.ai/nvidia` quota and fits the complete PodSet inside the selected topology domains | [Topology-Aware Scheduling](architecture/topology-aware-scheduling.md) | | Gate 3 — AdmissionCheck | asks the pool's `Devices` ledger whether one accelerator can really host the slice; holds the workload with `Retry` if not | [Admission](architecture/admission.md#gate-3--the-per-accelerator-admissioncheck) | diff --git a/docs/architecture/installation-modes.md b/docs/architecture/installation-modes.md index c14f79c88..a6be396fc 100644 --- a/docs/architecture/installation-modes.md +++ b/docs/architecture/installation-modes.md @@ -73,7 +73,7 @@ Because they change what a mode installs: cluster has no device managers (useful for control-plane-only). Before chart mode covered them, this switch was how the worker came to install them. - **`modelManager.enabled=false`** — no model-manager DaemonSet and no CSIDriver, and the worker - then seeds no `Node` delivery; see [Model Store Operations](../operation/model-store.md#enable-it). + then seeds no `Node` delivery; see [Model Store Operations](../model-store/operations.md#enable-it). - **`worker.enabled=false`** — the chart deploys only the applications, what image mode's overlay sets. ## The chart deploys workloads; the worker applies the custom resources diff --git a/docs/architecture/internals.md b/docs/architecture/internals.md index 1c1c2f090..41f56919d 100644 --- a/docs/architecture/internals.md +++ b/docs/architecture/internals.md @@ -29,7 +29,7 @@ accelerators, reports a `NodeFeature` + `Devices` CR, and runs the device-plugin allocator. - **`model-manager`** (alias `mm`) is a CSI node plugin on every node, not tied to a manufacturer: it materializes and mounts Hugging Face weights and writes only its own node's `NodeModelStore` - status ([Node Model Store Reference](../reference/node-model-store.md)). + status ([Node Model Store](../model-store/node-store.md)). ## Worker startup order matters diff --git a/docs/architecture/topology-aware-scheduling.md b/docs/architecture/topology-aware-scheduling.md index ac76f4d27..319a8e060 100644 --- a/docs/architecture/topology-aware-scheduling.md +++ b/docs/architecture/topology-aware-scheduling.md @@ -129,7 +129,7 @@ A role replica owns one Workload and its `size` Pods form the fate-sharing PodSe `replicas: 3` and `size: 8` produces three independent eight-Pod topology decisions, not one twenty-four-Pod decision. Different roles and replicas are not required to share a domain. -Several roles still admit as one set. The [joint admission check](../reference/model-deployment.md#prefill-and-decode) +Several roles still admit as one set. The [joint admission check](../model-deployment/deployment.md#prefill-and-decode) holds every group until the whole set has reserved quota, and no role Pod binds to a Node before then. While the set waits, a group that fits keeps its quota reservation and topology assignment, so the domain it holds stays idle. When no role fits, no Workload of the set reserves anything. @@ -138,10 +138,10 @@ so the domain it holds stays idle. When no role fits, no Workload of the set res > sibling roles waiting on each other would trade the same quota back and forth. The hold ends when > the set is admitted, when the deployment is deleted, or when the check parks a set that has not > assembled for 30 minutes (`QuotaReserved` reason `Parked` in the -> [status reference](../reference/model-deployment-status.md#status)). +> [status reference](../model-deployment/status.md#status)). Omitting `requiredLevel` adds no explicit topology request. The queue is still topology-aware, and -Kueue may choose any compatible hierarchy. The [field contract](../reference/model-deployment.md#topology-placement) +Kueue may choose any compatible hierarchy. The [field contract](../model-deployment/deployment.md#topology-placement) defines the implicit hostname level. ## A node-delivered model prefers the nodes holding it @@ -149,7 +149,7 @@ defines the implicit hostname level. When the worker creates a Pod whose Hugging Face weights the node delivers, a `ModelDeployment` replica under `Node` delivery or an `Instance`, it adds one preferred node-affinity term per digest. The term names, by `kubernetes.io/hostname`, the nodes whose -[`NodeModelStore`](../reference/node-model-store.md) lists that digest `Ready`. +[`NodeModelStore`](../model-store/node-store.md) lists that digest `Ready`. Kueue copies the Pod's affinity into the PodSet. With `TASRespectNodeAffinityPreferred` on, the chart's default, TAS ranks the nodes with room by that score before its usual packing order: hot @@ -171,7 +171,7 @@ by name. A digest no node holds adds no term. A Pod carrying `kueue.x-k8s.io/podset-preferred-topology` gets no term: with `TASBalancedPlacement` on, as the chart has it, such a PodSet loses its affinity score. No Pod GPUStack renders carries it. `Engine` delivery and claim artifacts get no term either. Reading placements and turning the -preference off are in [Model Store Operations](../operation/model-store.md#where-replicas-land). +preference off are in [Model Store Operations](../model-store/operations.md#where-replicas-land). ## Capacity and lifecycle limits @@ -217,7 +217,6 @@ Kueue prose into a new stable reason. Operational checks and source examples are --- **See also** — [Scheduling Chain](scheduling-chain.md) (the flavor and queue owners) · -[Admission](admission.md) (the gates after queue admission) · [Model Deployment -Reference](../reference/model-deployment.md) (the user-facing field) +[Admission](admission.md) (the gates after queue admission) · [Model Deployment](../model-deployment/deployment.md) (the user-facing field) **Next** → [Admission](admission.md) — the remaining gates after topology-aware quota reservation. diff --git a/docs/kv-cache/backend.md b/docs/kv-cache/backend.md index e9bf2a290..0ec5a343b 100644 --- a/docs/kv-cache/backend.md +++ b/docs/kv-cache/backend.md @@ -47,7 +47,7 @@ spec: ``` ⚠️ The example's `0.3.13` line fits vLLM's clients and not SGLang's — pick `spec.image` from the -[Engine Versions Reference](../reference/engine-versions.md) before copying it. +[Engine Versions](../model-deployment/engine-versions.md) before copying it. `connection.managed` and `connection.external` are both optional pointers and **exactly one** must be set; neither and both are refused at admission with a message naming the two. Several member groups @@ -189,7 +189,7 @@ toolchain version: `-`, such as `0.3.13.po Each variant is built on `0.3.13.post1`, the line vLLM's supported clients are on, and on `0.3.10.post2`, which serves only engines below the [supported -minimum](../reference/engine-versions.md). SGLang's clients are on the 0.3.12 line, which this +minimum](../model-deployment/engine-versions.md). SGLang's clients are on the 0.3.12 line, which this project does not build. Which line a backend needs is that table's question. **A VRAM group needs a build with VRAM segments compiled in (`USE_VRAM_SEGMENT=ON`), and the stock @@ -240,7 +240,7 @@ resolves to the default tenant. **The client's version is a property of the engine image, not of anything on this CR.** Which client each supported engine's runner image carries, and so which line its store runs, is in the -[Engine Versions Reference](../reference/engine-versions.md); an engine below its minimum there is +[Engine Versions](../model-deployment/engine-versions.md); an engine below its minimum there is not supported. For any other image, read the client off the image in hand rather than off an engine version. A @@ -320,7 +320,7 @@ and defaults to `Auto` whether or not the `transport` block is written at all. * transports, not host fabrics, so they take none of the fabric privileges below. Which engine each value serves, and on which images, is [the transport -matrix](../reference/engine-versions.md#which-transport-each-engine-can-use). +matrix](../model-deployment/engine-versions.md#which-transport-each-engine-can-use). **`members[].transport.protocol` overrides that value for one group; left unset, the group inherits the backend's.** The override exists for the one thing two media do not agree on: a VRAM group @@ -881,7 +881,7 @@ backend object**, so two Bindings reaching one leader through two objects are bo other's blocks.** The reuse identity an engine is handed is the domain **name alone** — each Binding hands its own [`dtype`](pool.md#the-dtype-is-handed-to-the-engine) to its own engines, and `blockSize` reaches no engine at all. So two differently-shaped caches land under - one identity, which is [the silent cache pollution](../reference/model-deployment.md#the-reuse-domain-is-inherited) + one identity, which is [the silent cache pollution](../model-deployment/deployment.md#the-reuse-domain-is-inherited) a wrong `blockSize` or `dtype` causes, reached here without either value being wrong. ⇒ If you point two objects at one leader, either keep their pools' Bindings on **different** diff --git a/docs/kv-cache/pool.md b/docs/kv-cache/pool.md index e87aabc70..7064c4bfe 100644 --- a/docs/kv-cache/pool.md +++ b/docs/kv-cache/pool.md @@ -170,7 +170,7 @@ without this two engines on one domain can write two element types under one key load, so it binds nothing. A Binding stored with it before the refusal stays usable and updatable. - **The engine is not free to disagree.** A role naming `--kv-cache-dtype` itself is refused; the rule and its exceptions are under - [What the operator owns](../reference/model-deployment.md#what-the-operator-owns). + [What the operator owns](../model-deployment/deployment.md#what-the-operator-owns). All of it follows the Setting `model-deployment-kv-cache-dtype-owned`, on by default; turning it off renders and refuses nothing, as before. See [Settings](../settings.md). @@ -385,7 +385,7 @@ is the section above. **See also** — [KV Cache Backend](backend.md) (the store this pool publishes, and where eviction is configured) · [KV Cache Injection](../reference/kv-cache-injection.md) (how a Pod consumes the grant -this page describes) · [Model Deployment](../reference/model-deployment.md) (the other half of the +this page describes) · [Model Deployment](../model-deployment/deployment.md) (the other half of the worked pair: a rendered engine names this Binding through `spec.kvCache.poolRef`) · [Admission](../architecture/admission.md) (the gates and the four-view status pattern) · [Settings & Environment Variables](../settings.md) diff --git a/docs/kv-cache/walkthrough.md b/docs/kv-cache/walkthrough.md index d62847e91..525ed32ba 100644 --- a/docs/kv-cache/walkthrough.md +++ b/docs/kv-cache/walkthrough.md @@ -239,8 +239,8 @@ reading on this page most likely to be escalated as an outage. **See also** — [KV Cache Backend](backend.md) (every field of the store, and what status reports) · [KV Cache Leader](leader.md) (the election, why there is no snapshot, and the member addressing choice in full) · [KV Cache Pool](pool.md) (quota, domains and what a full quota does) · -[Model Deployment Reference](../reference/model-deployment.md) (roles, prefill/decode, rollout) · -[Model Deployment Prefill and Decode Reference](../reference/model-deployment-prefill-decode.md) (router, transfer) · +[Model Deployment](../model-deployment/deployment.md) (roles, prefill/decode, rollout) · +[Model Deployment Prefill and Decode](../model-deployment/prefill-decode.md) (router, transfer) · [KV Cache Injection](../reference/kv-cache-injection.md) (what a Pod actually receives) **Next** → [KV Cache Pool](pool.md) — the quota this walkthrough set once and did not explain. diff --git a/docs/migration/kv-cache-dtype.md b/docs/migration/kv-cache-dtype.md index 06817f8dc..a242828c4 100644 --- a/docs/migration/kv-cache-dtype.md +++ b/docs/migration/kv-cache-dtype.md @@ -98,7 +98,7 @@ An injected Pod carries the same two entries at the end of its `args`, where the --- **See also** — [KV Cache Pool](../kv-cache/pool.md) (the dtype rule) · -[ModelDeployment](../reference/model-deployment.md#what-the-operator-owns) (what the operator owns) · +[ModelDeployment](../model-deployment/deployment.md#what-the-operator-owns) (what the operator owns) · [KV Cache Injection](../reference/kv-cache-injection.md) (the Pod path) · [Settings](../settings.md) (the escape switch) diff --git a/docs/reference/model-deployment.md b/docs/model-deployment/deployment.md similarity index 97% rename from docs/reference/model-deployment.md rename to docs/model-deployment/deployment.md index 70f038b0c..598e28b54 100644 --- a/docs/reference/model-deployment.md +++ b/docs/model-deployment/deployment.md @@ -1,8 +1,9 @@ -# Model Deployment Reference +# Model Deployment > **Purpose** — the `ModelDeployment` contract: what you declare, what the operator owns and will > refuse to merge, and how a role's runner image is assembled. -> **Audience** users, operators, contributors · **Prerequisites** [KV Cache Pool](../kv-cache/pool.md) · +> **Audience** users, operators, contributors · **Prerequisites** [KV Cache +> Pool](../kv-cache/pool.md) · > **Read time** ~9 min A `ModelDeployment` is N replicas of one or more inference-engine roles. It can attach to a KV @@ -53,7 +54,7 @@ spec: `model.name` is what the engine serves. The weights come from the engine's own hub client, a role's volumes, or a `ModelArtifact` named by `model.artifactRef` — see the -[Model Artifact Reference](model-artifact.md). +[Model Artifact](../model-store/artifact.md). `poolRef` is a `LocalObjectReference` on purpose: naming another namespace, the cluster-scoped `KVCachePool`, or a bare endpoint URL is unrepresentable rather than merely rejected. The Binding is @@ -117,7 +118,7 @@ container spec and only the label value differs. `--tensor-parallel-size` or equivalent: the degrees do not decompose from `size` alone, and a formula missing an input is worse than no formula. Composing none is not seeing none — a degree the author declares on the role is read and validated, and [the transfer leg renders from -it](model-deployment-prefill-decode.md#how-a-pair-is-wired). +it](prefill-decode.md#how-a-pair-is-wired). A whole instance is the unit of replacement at every size. That is Kueue's constraint rather than a preference: a deleted member of an admitted group is held on the API server until the group's @@ -182,8 +183,7 @@ Two roles may share a `kind` and differ in `name` only where that `kind` is `Ser servers is a set of equals, whereas nothing consuming these roles expresses a second prefiller, so a deployment declaring one would render a role nothing downstream can reach. -What pairs the two roles is on [Model Deployment Prefill and Decode -Reference](model-deployment-prefill-decode.md): the connector each engine and router renders, the +What pairs the two roles is on [Model Deployment Prefill and Decode](prefill-decode.md): the connector each engine and router renders, the router block and its fields, the direct transfer and its transport, roles on different hardware, and a role's own address. @@ -265,7 +265,7 @@ is not knowable at render time. Users who require tenant isolation must select a compatible engine image and verify it themselves; see -[Tenant compatibility is the image owner's responsibility](kv-cache-injection.md#tenant-compatibility-is-the-image-owners-responsibility) +[Tenant compatibility is the image owner's responsibility](../reference/kv-cache-injection.md#tenant-compatibility-is-the-image-owners-responsibility) for what the image must consume. The API states the requested boundary, while the engine enforces it — the same caveat [KV Cache Pool](../kv-cache/pool.md#what-a-binding-does-not-do) states for capacity. @@ -358,8 +358,8 @@ not in the table because it is conditional: which the hit rate this design rests on cannot be measured at all. It is read by the transfer engine rather than by an engine's config class, so it does not depend on which keys that class accepts. -So are `MC_FORCE_TCP` [on a `tcp` leg](model-deployment-prefill-decode.md#the-direct-transfers-transport) and -[SGLang's two cache switches](kv-cache-injection.md#sglangs-host-memory-tier). +So are `MC_FORCE_TCP` [on a `tcp` leg](prefill-decode.md#the-direct-transfers-transport) and +[SGLang's two cache switches](../reference/kv-cache-injection.md#sglangs-host-memory-tier). Two of SGLang's owned keys are owned for what a user entry would **destroy** rather than duplicate, and the operator does not set either of them: @@ -480,7 +480,7 @@ therefore turns a replica over only when every replica the role declares holds a on a full pool a rollout waits for capacity rather than shedding replicas it cannot re-reserve. The cost is real and worth stating, and it rides on the block lease described under -[What a cache changes about a workload](kv-cache-injection.md#what-a-cache-changes-about-a-workload): a lease survives a long queue and does **not** +[What a cache changes about a workload](../reference/kv-cache-injection.md#what-a-cache-changes-about-a-workload): a lease survives a long queue and does **not** survive an interrupted heartbeat, which is what a departing replica is. So a departing replica costs its siblings the blocks it held. The deployment records an event naming @@ -490,7 +490,7 @@ away has the correlation written down rather than inferred. **An upgrade can trigger the same turnover without any spec edit.** The fingerprint covers a replica's labels, annotations and spec, so a release that changes what every replica renders turns each one over -once: the `role-kind` label above did, and so did the [drain](model-deployment-shutdown.md). Nothing +once: the `role-kind` label above did, and so did the [drain](shutdown.md). Nothing is required of you, but on a busy deployment the restart is worth scheduling. ### A replica that leaves is replaced @@ -668,7 +668,7 @@ own message, because a pass that cannot build a replica aborts before writing an ## Operating notes **Two notes apply to every workload on a pool, replicas included, and are stated once under** -[What a cache changes about a workload](kv-cache-injection.md#what-a-cache-changes-about-a-workload): the transfer engine binds ports nobody +[What a cache changes about a workload](../reference/kv-cache-injection.md#what-a-cache-changes-about-a-workload): the transfer engine binds ports nobody configured, so a NetworkPolicy or port reservation has to be a range rather than a list; and the `transfer_metadata.cpp` "Local segment descriptor not found" line at startup is an `ERROR` that is benign on a client mounting no segment of its own — which is what every replica here is. @@ -708,11 +708,11 @@ and therefore neither locates its window. --- **See also** — [KV Cache Pool](../kv-cache/pool.md) for the Binding that grants the quota and declares -the domain · [Model Deployment Prefill and Decode Reference](model-deployment-prefill-decode.md) for +the domain · [Model Deployment Prefill and Decode](prefill-decode.md) for what pairs a prefill role with a decode role · [Accelerator Requests](../accelerator-requests.md) for the request fields `roles[].resources` mirrors · [Admission](../architecture/admission.md) for the gates a replica passes -as an ordinary Pod · [Model Deployment Status](model-deployment-status.md) for what each condition -and published field means · [Model Deployment Metrics](model-deployment-metrics.md) for the +as an ordinary Pod · [Model Deployment Status](status.md) for what each condition +and published field means · [Model Deployment Metrics](metrics.md) for the structured snapshot and Pod scrape endpoints. **Next** → [Accelerator Requests](../accelerator-requests.md) diff --git a/docs/reference/engine-versions.md b/docs/model-deployment/engine-versions.md similarity index 95% rename from docs/reference/engine-versions.md rename to docs/model-deployment/engine-versions.md index 6a127e73a..1b40900a6 100644 --- a/docs/reference/engine-versions.md +++ b/docs/model-deployment/engine-versions.md @@ -1,10 +1,10 @@ -# Engine Versions Reference +# Engine Versions > **Purpose** — the lowest vLLM, vLLM-Ascend and SGLang release each deployment shape has been run > with on this operator, with the Mooncake client it carries and the store it needs, and which > transport each engine can use on each leg. -> **Audience** users, operators · **Prerequisites** [Model Deployment -> Reference](model-deployment.md) · **Read time** reference — look up your engine +> **Audience** users, operators · **Prerequisites** [Model Deployment](deployment.md) +> **Read time** reference — look up your engine **An engine below its minimum here is not supported.** Older releases fail in ways that belong to those releases, and this documentation does not track them; upgrading is the fix for each of them. @@ -38,7 +38,7 @@ vLLM-Ascend rows ran on the runner's `cann9.1-910b-vllm0.23.0-router` tag, named ## Reading the table **`engine.version` has no default.** The operator assembles the runner image from it — see [The -runner image is a formula](model-deployment.md#the-runner-image-is-a-formula) — so the minimum is a +runner image is a formula](deployment.md#the-runner-image-is-a-formula) — so the minimum is a value you write, and nothing refuses a lower one. **The client comes with the image, not with the version.** The column above is read off the @@ -51,7 +51,7 @@ suggestion — see [The store version must match the engine's client](../kv-cache/backend.md#the-store-version-must-match-the-engines-client). **vLLM-Ascend's direct transfer has no `tcp` shape**, and one router renders it — see [How a pair is -wired](model-deployment-prefill-decode.md#how-a-pair-is-wired). +wired](prefill-decode.md#how-a-pair-is-wired). **The shapes above move KV over `tcp`, or over `ascend` on vLLM-Ascend.** Which other transport each engine can use is [the matrix below](#which-transport-each-engine-can-use). @@ -109,7 +109,7 @@ On AWS the `RDMA` rows do not apply — [choose `EFA` or - **SGLang with a store** holds a pinned host pool and needs the node's available memory above a fixed reserve plus that pool — see [SGLang's host-memory - tier](kv-cache-injection.md#sglangs-host-memory-tier). + tier](../reference/kv-cache-injection.md#sglangs-host-memory-tier). - **SGLang prefill/decode over `TCP` runs out of local ports under sustained load, with or without a store.** The prefill half opens a new connection for every transfer, to the decode half and to each store member, all from the same ephemeral ports of its container. Connections left in @@ -142,8 +142,8 @@ On AWS the `RDMA` rows do not apply — [choose `EFA` or --- -**See also** — [Model Deployment Reference](model-deployment.md) (the fields these versions go into) · +**See also** — [Model Deployment](deployment.md) (the fields these versions go into) · [KV Cache Backend](../kv-cache/backend.md) (the store a client needs) · [RDMA Operations](../operation/rdma.md) (fabric devices and the engine image they need) -**Next** → [Model Deployment Status Reference](model-deployment-status.md) +**Next** → [Model Deployment Status](status.md) diff --git a/docs/reference/model-deployment-metrics.md b/docs/model-deployment/metrics.md similarity index 97% rename from docs/reference/model-deployment-metrics.md rename to docs/model-deployment/metrics.md index 9c85699f1..0f486bc91 100644 --- a/docs/reference/model-deployment-metrics.md +++ b/docs/model-deployment/metrics.md @@ -1,9 +1,9 @@ -# Model Deployment Metrics Reference +# Model Deployment Metrics > **Purpose** — the structured `ModelDeployment` metrics snapshot and the managed Pods' raw > Prometheus scrape endpoints. -> **Audience** users, operators, console developers · **Prerequisites** [Model Deployment -> Reference](model-deployment.md) · **Read time** ~7 min +> **Audience** users, operators, console developers · **Prerequisites** [Model +> Deployment](deployment.md) · **Read time** ~7 min The aggregated API combines selected current gauges from a deployment's engine and router Pods. Each managed Pod also exposes its own native `/metrics` output for Prometheus users. @@ -111,7 +111,7 @@ exported, so what each counts has not been checked against a failed request. Why in [Latency, traffic, and transfer](#latency-traffic-and-transfer). Where a router sent each request is not in the snapshot; its own per-replica series say, listed in -[Seeing where requests went](model-deployment-routing.md#seeing-where-requests-went). +[Seeing where requests went](routing.md#seeing-where-requests-went). ## Windowed cache hits @@ -233,7 +233,7 @@ annotation. The operator installs no `PodMonitor` or `ServiceMonitor`. --- -**See also** — [Model Deployment Reference](model-deployment.md) for role configuration · -[Instance Metrics Reference](instance-metrics.md) for the separate Instance utilization API. +**See also** — [Model Deployment](deployment.md) for role configuration · +[Instance Metrics Reference](../reference/instance-metrics.md) for the separate Instance utilization API. -**Next** → [Model Deployment Status Reference](model-deployment-status.md) +**Next** → [Model Deployment Status](status.md) diff --git a/docs/reference/model-deployment-prefill-decode.md b/docs/model-deployment/prefill-decode.md similarity index 92% rename from docs/reference/model-deployment-prefill-decode.md rename to docs/model-deployment/prefill-decode.md index dbd5b8c01..01e327b20 100644 --- a/docs/reference/model-deployment-prefill-decode.md +++ b/docs/model-deployment/prefill-decode.md @@ -1,13 +1,13 @@ -# Model Deployment Prefill and Decode Reference +# Model Deployment Prefill and Decode > **Purpose** — what pairs a `Prefill` role with a `Decode` role: the connector each engine and > router renders, the router block and its fields, the direct transfer and its transport, roles on > different hardware, and a role's own address. -> **Audience** users, operators, contributors · **Prerequisites** [Model Deployment -> Reference](model-deployment.md) · **Read time** ~4 min +> **Audience** users, operators, contributors · **Prerequisites** [Model Deployment](deployment.md) +> · **Read time** ~4 min A deployment declaring a `Prefill` and a `Decode` role is admitted as one set; the role fields and -that admission are under [Prefill and decode](model-deployment.md#prefill-and-decode). This page is +that admission are under [Prefill and decode](deployment.md#prefill-and-decode). This page is what the operator renders between the two halves once both run. ## Contents @@ -78,7 +78,7 @@ leg's quiet failure, one level down. Admission holds the visible side of the contract: a degree the books cannot be read for is refused, and so is a declared per-member width the role's card request cannot hold — see -[What admission refuses](model-deployment.md#what-admission-refuses). +[What admission refuses](deployment.md#what-admission-refuses). SGLang renders its halves through the engine's own disaggregation arguments rather than this connector path; the two engines' legs differ by [their handshake](#the-direct-transfers-transport) @@ -186,9 +186,9 @@ A router is also **engine-matched**, and a pair outside this table is refused na | `spec.router.name` | Engines it fronts | Shape it renders | Routing policy | | --- | --- | --- | --- | -| `llm-d-router` | `vLLM`, `SGLang` | An endpoint picker behind a proxy, configured by a mounted document | [A fixed scoring profile](model-deployment-routing.md#llm-d-router-takes-no-policy-flag) | -| `vllm-router` | `vLLM` | One process, configured entirely by its command line | [`cache_aware` unless `extraArgs` names another](model-deployment-routing.md#switching-to-round-robin) | -| `sglang-gateway` | `SGLang` | One process, configured entirely by its command line | [`cache_aware` unless `extraArgs` names another](model-deployment-routing.md#switching-to-round-robin) | +| `llm-d-router` | `vLLM`, `SGLang` | An endpoint picker behind a proxy, configured by a mounted document | [A fixed scoring profile](routing.md#llm-d-router-takes-no-policy-flag) | +| `vllm-router` | `vLLM` | One process, configured entirely by its command line | [`cache_aware` unless `extraArgs` names another](routing.md#switching-to-round-robin) | +| `sglang-gateway` | `SGLang` | One process, configured entirely by its command line | [`cache_aware` unless `extraArgs` names another](routing.md#switching-to-round-robin) | `llm-d-router` takes both engines because upstream carries a handshake connector and a metrics configuration for each. The other two are each one project's router for that project's own engine, @@ -256,7 +256,7 @@ Which value works on which engine is [the transport matrix](engine-versions.md#w **`tcp` is enforced, not only requested**: the transfer engine picks its transport from the host, and with no RDMA device a build with multi-node NVLink installs NVLink between hosts with no NVLink -path. So native vLLM also gets the [defaulted](model-deployment.md#what-the-operator-owns) `MC_FORCE_TCP=1`. Both pins +path. So native vLLM also gets the [defaulted](deployment.md#what-the-operator-owns) `MC_FORCE_TCP=1`. Both pins are process-wide, so neither renders beside a store on another transport, and every client at its engine's [supported minimum](engine-versions.md) honors them. @@ -274,7 +274,7 @@ It is also **not** the pool's transport. `KVCacheBackend.spec.transport` feeds t client; this leg is engine to engine and never traverses the store, so the two declare separately — a deployment with no `kvCache` block still has this leg to configure. -Editing it [turns over every role](model-deployment.md#rollout-is-a-rolling-replacement): the value renders into both +Editing it [turns over every role](deployment.md#rollout-is-a-rolling-replacement): the value renders into both ends' arguments, so every role's replicas turn over one at a time. A prefiller and a decoder can disagree on the protocol until both converge — the same window an `engine.version` edit opens. @@ -285,7 +285,7 @@ publishes as its `status.entrance`. Every replica is its own group regardless, s going to share a Workload — on two `instanceType`s or on one. The set is admitted together by an admission check rather than by Kueue's intra-group rule. -See [One group per replica](model-deployment.md#one-group-per-replica) for what that costs an edit, and +See [One group per replica](deployment.md#one-group-per-replica) for what that costs an edit, and [Authoring the InstanceType yourself](../settings.md#authoring-the-instancetype-yourself) for which queues carry the check. @@ -301,7 +301,7 @@ between the halves: - **This operator's own refusal is the one an administrator meets.** On a pool left at its default transport, the renderer [refuses the vLLM-Ascend - half](kv-cache-injection.md#transport-compatibility-at-binding) — the rule and its + half](../reference/kv-cache-injection.md#transport-compatibility-at-binding) — the rule and its remediation are stated there. The check is one-sided — it fires for a single-manufacturer Ascend deployment just the same — and it is the only one of the three that produces a message; following its remediation clears only this refusal; the next two apply regardless. @@ -336,14 +336,13 @@ debugging. With `spec.router`, the operator renders six objects named `-router`: a Deployment, ConfigMap, Service, ServiceAccount, Role and RoleBinding. Removing `spec.router` prunes all six. -What `status.endpoint` publishes in each shape is under [Status](model-deployment-status.md#status). +What `status.endpoint` publishes in each shape is under [Status](status.md#status). --- -**See also** — [Model Deployment Reference](model-deployment.md) for the role fields, the owned-key -table and what admission refuses · [Model Deployment Routing -Reference](model-deployment-routing.md) for which replica each router picks · [Engine Versions -Reference](engine-versions.md) for which transport each engine can use on each leg · [Model -Deployment Status Reference](model-deployment-status.md) for what `status.endpoint` publishes. +**See also** — [Model Deployment](deployment.md) for the role fields, the owned-key +table and what admission refuses · [Model Deployment Routing](routing.md) for which replica each +router picks · [Engine Versions](engine-versions.md) for which transport each engine can use on +each leg · [Model Deployment Status](status.md) for what `status.endpoint` publishes. -**Next** → [Model Deployment Routing Reference](model-deployment-routing.md) +**Next** → [Model Deployment Routing](routing.md) diff --git a/docs/reference/model-deployment-routing.md b/docs/model-deployment/routing.md similarity index 90% rename from docs/reference/model-deployment-routing.md rename to docs/model-deployment/routing.md index 7499d688d..dbd9e09a0 100644 --- a/docs/reference/model-deployment-routing.md +++ b/docs/model-deployment/routing.md @@ -1,9 +1,9 @@ -# Model Deployment Routing Reference +# Model Deployment Routing > **Purpose** — which replica each managed router sends a request to by default, how to change > that choice through `spec.router.extraArgs`, and the router series that show where requests went. -> **Audience** users, operators · **Prerequisites** [Model Deployment -> Reference](model-deployment.md) · **Read time** ~5 min +> **Audience** users, operators · **Prerequisites** [Model Deployment](deployment.md) +> **Read time** ~5 min A router only chooses when a role has more than one replica. Every routing choice measured here ran on a server role; on a prefill/decode pair, where each half is chosen separately, none has been run. @@ -69,7 +69,7 @@ spec: - round_robin ``` -`--policy` is in neither router's [refused list](model-deployment-prefill-decode.md#the-router-block), so admission +`--policy` is in neither router's [refused list](prefill-decode.md#the-router-block), so admission passes it and the operator appends it to the command line it renders. The router then logs `Starting router … | policy: RoundRobin`. @@ -78,7 +78,7 @@ under each router, 120 requests with none failing, 60 on each replica. `vllm-rou in `vllm_router_policy_decisions_total{policy="round_robin"}`, 60 per replica, and `sglang-gateway` in `smg_worker_selection_total{policy="round_robin"}`, 120 in all. -`router.extraArgs` is [editable](model-deployment.md#which-fields-are-the-deployments-identity), so a +`router.extraArgs` is [editable](deployment.md#which-fields-are-the-deployments-identity), so a running deployment can switch as well. The router Pod is then replaced, and the new one starts with no record of earlier prefixes. Only a flag set at creation has been run. @@ -102,7 +102,7 @@ router at another one, is refused there. No field sets them either. ## Seeing where requests went -The deployment's [metrics snapshot](model-deployment-metrics.md) does not say which replica served a +The deployment's [metrics snapshot](metrics.md) does not say which replica served a request. Each router's own `/metrics` does, for the series below, all seen exported in runs. A `worker` label is the worker's URL, which carries its Pod address. @@ -121,8 +121,8 @@ the engine Pods' own counts instead, such as each Pod's TTFT histogram `_count`. --- -**See also** — [Model Deployment Prefill and Decode Reference](model-deployment-prefill-decode.md) -for `spec.router` and the flags each router refuses · [Model Deployment Metrics Reference](model-deployment-metrics.md) for the router +**See also** — [Model Deployment Prefill and Decode](prefill-decode.md) +for `spec.router` and the flags each router refuses · [Model Deployment Metrics](metrics.md) for the router counters a deployment's snapshot reads. -**Next** → [Model Deployment Status Reference](model-deployment-status.md) +**Next** → [Model Deployment Status](status.md) diff --git a/docs/reference/model-deployment-shutdown.md b/docs/model-deployment/shutdown.md similarity index 91% rename from docs/reference/model-deployment-shutdown.md rename to docs/model-deployment/shutdown.md index 7a56c25ee..da8c1a7b7 100644 --- a/docs/reference/model-deployment-shutdown.md +++ b/docs/model-deployment/shutdown.md @@ -1,9 +1,9 @@ -# Model Deployment Shutdown Reference +# Model Deployment Shutdown > **Purpose** — what a `ModelDeployment` replica does between its Pod's delete and its engine's exit, > and which requests that window does not save. -> **Audience** users, operators · **Prerequisites** [Model Deployment -> Reference](model-deployment.md) · **Read time** ~4 min +> **Audience** users, operators · **Prerequisites** [Model Deployment](deployment.md) +> **Read time** ~4 min ## Contents @@ -57,7 +57,7 @@ for that scheduler before killing it. > either way. The replica is idle by then, so the kill at 30 s cuts no request. The hook sums these gauges from the listener the Pod's -[scrape annotations](model-deployment-metrics.md#scraping-the-pods) name. On a direct decoder that is +[scrape annotations](metrics.md#scraping-the-pods) name. On a direct decoder that is the engine behind the routing proxy, not the port the Service fronts. | Engine | Gauges summed | @@ -71,7 +71,7 @@ sample the hook cannot parse. Adding the hook and the grace changed every replica's Pod spec, so upgrading to the release that carries them turns each existing replica over once; see -[Rollout is a rolling replacement](model-deployment.md#rollout-is-a-rolling-replacement). +[Rollout is a rolling replacement](deployment.md#rollout-is-a-rolling-replacement). Changing a role's `terminationGracePeriodSeconds` turns its replicas over the same way, and each replica that leaves in that rollout leaves with the grace it was created with. Writing 30 onto a @@ -100,8 +100,8 @@ role that set none renders the same Pods, so it turns nothing over. --- -**See also** — [Model Deployment Reference](model-deployment.md) for what turns a replica over · -[Model Deployment Metrics Reference](model-deployment-metrics.md) for the scrape endpoints the hook +**See also** — [Model Deployment](deployment.md) for what turns a replica over · +[Model Deployment Metrics](metrics.md) for the scrape endpoints the hook reads. -**Next** → [Model Deployment Status Reference](model-deployment-status.md) +**Next** → [Model Deployment Status](status.md) diff --git a/docs/reference/model-deployment-status.md b/docs/model-deployment/status.md similarity index 98% rename from docs/reference/model-deployment-status.md rename to docs/model-deployment/status.md index e47117a7e..e6f2eeec8 100644 --- a/docs/reference/model-deployment-status.md +++ b/docs/model-deployment/status.md @@ -1,9 +1,9 @@ -# Model Deployment Status Reference +# Model Deployment Status > **Purpose** — what each `ModelDeployment` status condition and published field means, and how to > read them when a deployment misbehaves. -> **Audience** users, operators · **Prerequisites** [Model Deployment -> Reference](model-deployment.md) · **Read time** ~8 min +> **Audience** users, operators · **Prerequisites** [Model Deployment](deployment.md) +> **Read time** ~8 min ## Contents @@ -37,7 +37,7 @@ the effective kind, selector, direct endpoint and any dialable KV-event endpoint are absent when the corresponding rendered Pods do not carry publisher configuration. This router contract names metrics sources for routing; it is not a live utilization sample. -The separate [Model Deployment Metrics](model-deployment-metrics.md) subresource reads the current +The separate [Model Deployment Metrics](metrics.md) subresource reads the current router and engine Pod endpoints and reports partial coverage explicitly. > **Why** — either flag alone is enough, and the rest of the `--ssl-*` family is not enough. Both @@ -139,7 +139,7 @@ would be worse than none. Eight conditions carry the axes a single phase cannot. They are independent: "quota reserved but cache not attached" is a real and actionable state. Seven are described below; `WeightsReady` is -described with the artifact it reports on, in [Model Artifact Reference](model-artifact.md#status). +described with the artifact it reports on, in [Model Artifact](../model-store/artifact.md#status). **`DomainRegistered`** — whether the referenced Binding resolved and its domain was read. @@ -347,8 +347,8 @@ cause. --- -**See also** — [Model Deployment Reference](model-deployment.md) for the contract this status +**See also** — [Model Deployment](deployment.md) for the contract this status reports on · [KV Cache Leader](../kv-cache/leader.md) for the leader process the conditions distinguish from the operator's own. -**Next** → [Model Deployment Reference](model-deployment.md) +**Next** → [Model Deployment](deployment.md) diff --git a/docs/reference/model-artifact.md b/docs/model-store/artifact.md similarity index 93% rename from docs/reference/model-artifact.md rename to docs/model-store/artifact.md index c1b6d4bf5..43ffd6822 100644 --- a/docs/reference/model-artifact.md +++ b/docs/model-store/artifact.md @@ -1,9 +1,9 @@ -# Model Artifact Reference +# Model Artifact > **Purpose** — how a `ModelArtifact` names a model's weights, how the operator resolves and > revalidates it, and how a `ModelDeployment` or an `Instance` consumes it. -> **Audience** users, operators · **Prerequisites** [Model Deployment -> Reference](model-deployment.md) · **Read time** ~16 min +> **Audience** users, operators · **Prerequisites** [Model +> Deployment](../model-deployment/deployment.md) · **Read time** ~16 min A `ModelArtifact` is the one object that says where a model's weights come from and which credential reads them. A `ModelDeployment` names it with `spec.model.artifactRef`, an `Instance` with a `model` @@ -69,7 +69,7 @@ status: - **An `image` member delivers weights already in a registry.** A digest-pinned reference is the artifact's whole identity; kubelet pulls and mounts it through an image volume, outside the node cache. The digest contract, the build, the floors and the costs are on the - [Model Image Source Reference](model-image-source.md). + [Model Image Source](image-source.md). - **Patterns select the files.** They follow Python's `fnmatch.fnmatchcase`: case-sensitive, `*` and `?` cross `/`, a trailing `/` means everything under it, an empty allow list keeps every file, and an ignored file is dropped even when allowed. At most 32 per list, 1 to 256 characters each, @@ -79,8 +79,8 @@ status: - **`status.nodes` counts the content, not the artifact.** Nodes whose `NodeModelStore` lists the digest `Ready`, `Downloading` or `Failed`; artifacts with the same digest see the same nodes, and only numbers cross namespaces. The mean covers the downloading nodes only, each a whole copy. It - is written on the [thresholds](node-model-store.md#the-resource) the nodes store progress on. The - `v1` view's [progress](model-artifact-views.md#the-progress-subresource) answers the same at full + is written on the [thresholds](node-store.md#the-resource) the nodes store progress on. The + `v1` view's [progress](views.md#the-progress-subresource) answers the same at full precision. - **Deletion waits for the last reference.** The finalizer `worker.gpustack.ai/model-artifact-protection` holds a referenced artifact in `Terminating` until no `ModelDeployment` or `Instance` in the @@ -163,7 +163,7 @@ creates no Pod until it resolves. A claim source is always mounted directly. A Hugging Face source takes the delivery the `model-artifact-delivery-mode` Setting names, `Engine` by default and `Node` where the chart deploys -the node plugin ([switching it](../operation/model-store.md#switch-delivery) rolls each such +the node plugin ([switching it](operations.md#switch-delivery) rolls each such deployment once): | | Claim source (`Pvc`) | Hugging Face, `Engine` | Hugging Face, `Node` | Image source (`Image`) | @@ -174,7 +174,7 @@ deployment once): | Both | `--served-model-name `, unless the role states it | same | same | same | An image source takes `Image` whatever the Setting says, and kubelet pulls the pinned image on the -node that needs it — [Model Image Source Reference](model-image-source.md). +node that needs it — [Model Image Source](image-source.md). `--revision` pins the weights and the tokenizer together on both engines. A take-over role (one with `command`) gets the claim or node mount and nothing else, and nothing at all under `Engine`. @@ -182,7 +182,7 @@ with `command`) gets the claim or node mount and nothing else, and nothing at al **Node delivery** mounts an inline CSI volume of the driver `model.csi.gpustack.ai`: the node's `model-manager` plugin downloads the digest once per node, verifies every byte against the manifest before anything is mounted, and mounts only for a resolved artifact in the Pod's own namespace. The -plugin and its resource are on the [Node Model Store Reference](node-model-store.md). +plugin and its resource are on the [Node Model Store](node-store.md). No Hub variable and no cache `emptyDir` is rendered under Node delivery, and the ephemeral-storage limit is not raised: the volume's bytes are not the Pod's. @@ -329,7 +329,7 @@ the CSIDriver does not exist the Instance creates no Pod and says so in its phas An image artifact mounts through an image volume, needing no plugin. An Instance pinned to a node waits instead when that node cannot run one, naming the node and the floor -([Model Image Source Reference](model-image-source.md#delivery)). +([Model Image Source](image-source.md#delivery)). ## Requirements and limits @@ -338,7 +338,7 @@ waits instead when that node cannot run one, naming the node and the floor stays `WeightsNotMounted` while its replicas run. - **Image sources need image volumes** and floors above Kubernetes's own; creation is refused below them. The full line is on the - [Model Image Source Reference](model-image-source.md#versions-and-prerequisites). + [Model Image Source](image-source.md#versions-and-prerequisites). - **`sglang-gateway` fetches a tokenizer by the worker's `model_path`.** With a claim that path is local, so it logs one 404 warning and routes by text; with Engine delivery it would fetch `main` without a token (not measured). @@ -353,9 +353,9 @@ waits instead when that node cannot run one, naming the node and the floor --- -**See also** — [Model Deployment Reference](model-deployment.md) for the rest of the deployment -contract · [Node Model Store Reference](node-model-store.md) for Node delivery · -[Model Artifact Views Reference](model-artifact-views.md) for the `v1` view and `progress` · [KV Cache Injection Reference](kv-cache-injection.md) for the store connector this -prefixes · [Model Deployment Status Reference](model-deployment-status.md) for the other conditions. +**See also** — [Model Deployment](../model-deployment/deployment.md) for the rest of the deployment +contract · [Node Model Store](node-store.md) for Node delivery · +[Model Artifact Views](views.md) for the `v1` view and `progress` · [KV Cache Injection Reference](../reference/kv-cache-injection.md) for the store connector this +prefixes · [Model Deployment Status](../model-deployment/status.md) for the other conditions. -**Next** → [Model Deployment Status Reference](model-deployment-status.md) +**Next** → [Model Deployment Status](../model-deployment/status.md) diff --git a/docs/reference/model-image-source.md b/docs/model-store/image-source.md similarity index 94% rename from docs/reference/model-image-source.md rename to docs/model-store/image-source.md index cd330d7db..fcc845b46 100644 --- a/docs/reference/model-image-source.md +++ b/docs/model-store/image-source.md @@ -1,8 +1,8 @@ -# Model Image Source Reference +# Model Image Source > **Purpose** — the `image` source of a `ModelArtifact`: the digest contract, how weights are built > into an image, how it is delivered, and what it costs in disk and re-pulls. -> **Audience** users, operators · **Prerequisites** [Model Artifact Reference](model-artifact.md) · +> **Audience** users, operators · **Prerequisites** [Model Artifact](artifact.md) · > **Read time** ~8 min ## Contents @@ -144,9 +144,9 @@ layers overlapped pulled in 112 ms and added 11.8 MB where a cold node paid 2m10 --- -**See also** — [Model Artifact Reference](model-artifact.md) for the artifact contract the image -source joins · [Node Model Store Reference](node-model-store.md) for the plugin chain an image -source stays out of · [Model Prefetch Reference](model-prefetch.md) for why there is nothing to +**See also** — [Model Artifact](artifact.md) for the artifact contract the image +source joins · [Node Model Store](node-store.md) for the plugin chain an image +source stays out of · [Model Prefetch](prefetch.md) for why there is nothing to warm. -**Next** → [Model Prefetch Reference](model-prefetch.md) +**Next** → [Model Prefetch](prefetch.md) diff --git a/docs/reference/node-model-store.md b/docs/model-store/node-store.md similarity index 94% rename from docs/reference/node-model-store.md rename to docs/model-store/node-store.md index 4c829f92b..24ffab172 100644 --- a/docs/reference/node-model-store.md +++ b/docs/model-store/node-store.md @@ -1,10 +1,10 @@ -# Node Model Store Reference +# Node Model Store > **Purpose** — the `NodeModelStore` resource and the `model-manager` node plugin behind it: what > each field means and who writes it, when a mount is allowed, how content is downloaded, verified > and published, why an attempt fails, and what the plugin measures. -> **Audience** operators, contributors · **Prerequisites** [Model Artifact -> Reference](model-artifact.md) · **Read time** ~12 min +> **Audience** operators, contributors · **Prerequisites** [Model Artifact](artifact.md) +> **Read time** ~12 min Node delivery replaces an engine's own download with a node-local cache. The `model-manager` plugin runs on every node as a CSI node plugin serving inline ephemeral volumes of the driver @@ -12,7 +12,7 @@ runs on every node as a CSI node plugin serving inline ephemeral volumes of the every byte, publishes the tree in one step and bind-mounts it read-only; every later mount of that digest on that node is one `stat` and a bind mount. -Enabling it and operating it are on [Model Store Operations](../operation/model-store.md). +Enabling it and operating it are on [Model Store Operations](operations.md). ## Contents @@ -107,7 +107,7 @@ status: # the plugin on that node: its fact | `status` | the plugin on that node | admitted only through the status webhook below | With `kubectl`, a write to the object names `v1alpha1` (`nodemodelstores.v1alpha1.worker.gpustack.ai`): -the group's preferred version is the [`v1` view](model-artifact-views.md#the-v1-views), which serves +the group's preferred version is the [`v1` view](views.md#the-v1-views), which serves reads and `delete` only. A deleted object is created again by the worker while its node runs the plugin. @@ -250,11 +250,11 @@ It never removes a referenced tree or a partial being written, and it reads refe node's own mounts, never from the API. A digest listed in `spec.pinned` is never a candidate at all — the pins arrive from -[`ModelPrefetch`](model-prefetch.md) retention and survive collection until the pin is withdrawn — +[`ModelPrefetch`](prefetch.md) retention and survive collection until the pin is withdrawn — though it still counts toward the usage the watermarks read. How the watermarks are capped when the cache shares kubelet's filesystem is under [the capacity -rule](../operation/model-store.md#the-capacity-rule). +rule](operations.md#the-capacity-rule). When the references cannot be read, no configuration has been applied yet, or the node's `spec` fails its check, a collection removes nothing, a stale partial included, and the `Ready` message @@ -278,7 +278,7 @@ No label carries a namespace, an artifact, a repository or a Pod. The same port serves `GET /model/downloads`: the running downloads, each a `digest`, `downloadedBytes`, `sizeBytes` and `source`. The ModelArtifact -[progress](model-artifact-views.md#the-progress-subresource) subresource reads it for live bytes +[progress](views.md#the-progress-subresource) subresource reads it for live bytes between the status thresholds. ## Requirements and limits @@ -296,15 +296,15 @@ between the status thresholds. [a node-delivered model prefers the nodes holding it](../architecture/topology-aware-scheduling.md#a-node-delivered-model-prefers-the-nodes-holding-it). - **A download comes from the Hub**, directly or through the proxy — or, when - [node-to-node sync](node-peer-sync.md) is on, from a peer node that already holds + [node-to-node sync](peer-sync.md) is on, from a peer node that already holds the tree. --- -**See also** — [Model Artifact Reference](model-artifact.md) for the artifact and its other -deliveries · [Model Artifact Views Reference](model-artifact-views.md) for the `v1` views and -`progress` · [Node-to-Node Sync Reference](node-peer-sync.md) for where a node's bytes come from · -[Model Store Operations](../operation/model-store.md) for enabling, configuring and upgrading · +**See also** — [Model Artifact](artifact.md) for the artifact and its other +deliveries · [Model Artifact Views](views.md) for the `v1` views and +`progress` · [Node-to-Node Sync](peer-sync.md) for where a node's bytes come from · +[Model Store Operations](operations.md) for enabling, configuring and upgrading · [Settings](../settings.md#online-adjustable-settings) for the Settings `spec` is built from. -**Next** → [Model Store Operations](../operation/model-store.md) +**Next** → [Model Store Operations](operations.md) diff --git a/docs/operation/model-store.md b/docs/model-store/operations.md similarity index 95% rename from docs/operation/model-store.md rename to docs/model-store/operations.md index de3ab8671..bdd941437 100644 --- a/docs/operation/model-store.md +++ b/docs/model-store/operations.md @@ -3,12 +3,12 @@ > **Purpose** — running node delivery: enabling the `model-manager` plugin, where its configuration > comes from, reading what a node applied and holds, keeping the cache away from kubelet's eviction, > switching delivery, where replicas land, upgrading, and removing it. -> **Audience** operators · **Prerequisites** [Node Model Store -> Reference](../reference/node-model-store.md) · **Read time** ~11 min +> **Audience** operators · **Prerequisites** [Node Model Store](node-store.md) · **Read time** ~11 +> min The `model-manager` DaemonSet keeps one model cache per node and mounts a Hugging Face `ModelArtifact` from it, downloading a digest once per node. What it does on each mount is in the -[reference](../reference/node-model-store.md); this page is what an administrator sets and reads. +[reference](node-store.md); this page is what an administrator sets and reads. ## Contents @@ -84,7 +84,7 @@ kubectl -n gpustack-system patch setting model-store-download-bandwidth --type m ### The pool layer -A [`ModelStore`](../reference/model-prefetch.md) is a fourth layer above the cluster Settings: a +A [`ModelStore`](prefetch.md) is a fourth layer above the cluster Settings: a cluster-scoped object whose `nodeSelector` picks a pool and whose watermarks and download limits override the matched nodes field by field. A field the store leaves out keeps the cluster default, and an explicit `bytesPerSecond: 0` means "unlimited on this pool". The winner's name lands in the @@ -108,10 +108,10 @@ kubectl get nodemodelstores.v1.worker.gpustack.ai # the v1 view: Ready, Used, M `generation` means the plugin applies it. The `Ready` message says when the high watermark was capped, and from what. - **`status.models`** lists each digest the node holds, is downloading or failed on, and whether a - Pod mounts it; a download's `downloadedBytes` [moves in steps](../reference/node-model-store.md#the-resource). + Pod mounts it; a download's `downloadedBytes` [moves in steps](node-store.md#the-resource). It names no tenant: find a deployment's digest in its `status.model.manifestDigest`, and an artifact's nodes in its `status.nodes` or its - [progress](../reference/model-artifact-views.md#the-progress-subresource). + [progress](views.md#the-progress-subresource). - **A `Failed` digest** carries its reason and `retryTime`. Mounts of it do not download again before then; fix the cause (the Secret, the proxy, the CA) and the next attempt after `retryTime` picks it up. @@ -245,7 +245,7 @@ kubelet retries it, and it succeeds once the new Pod serves. ## A deployment waiting for its weights `WeightsReady` on the `ModelDeployment` says which side to look at; the full table is in the -[Model Artifact Reference](../reference/model-artifact.md#status). +[Model Artifact](artifact.md#status). | Reason | Look at | | --- | --- | @@ -257,8 +257,8 @@ kubelet retries it, and it succeeds once the new Pod serves. --- -**See also** — [Node Model Store Reference](../reference/node-model-store.md) for every field and -reason · [Model Artifact Reference](../reference/model-artifact.md) for the artifact and its +**See also** — [Node Model Store](node-store.md) for every field and +reason · [Model Artifact](artifact.md) for the artifact and its patterns · [Settings](../settings.md#online-adjustable-settings) for every Setting. -**Next** → [Model Artifact Reference](../reference/model-artifact.md) +**Next** → [Model Artifact](artifact.md) diff --git a/docs/reference/node-peer-sync.md b/docs/model-store/peer-sync.md similarity index 90% rename from docs/reference/node-peer-sync.md rename to docs/model-store/peer-sync.md index 45124b588..8be5ed19f 100644 --- a/docs/reference/node-peer-sync.md +++ b/docs/model-store/peer-sync.md @@ -1,7 +1,8 @@ -# Node-to-Node Sync Reference +# Node-to-Node Sync > **Purpose** — how a node's plugin serves its published trees to the other nodes' plugins, how a cold node pulls from them, and what bounds and protects that path. -> **Audience** users, operators · **Prerequisites** [Node Model Store Reference](node-model-store.md) · **Read time** ~6 min +> **Audience** users, operators · **Prerequisites** [Node Model Store](node-store.md) +> **Read time** ~6 min A fleet's nodes need the same weights far more often than they need different ones. When the hub serves every node separately, egress grows with node count and a hub outage stalls every @@ -62,7 +63,7 @@ One source at a time; scheduling segments across several peers is a deliberate e `NodeModelStore.status.models[].source` names where the bytes came from (`Hub` or `Peer`, the majority kind), and the plugin's `download_bytes_total` metric splits `hub` from `peer`. On a shared filesystem, kubelet's eviction thresholds still cap the cache's high watermark as -[Node Model Store Reference](node-model-store.md) describes; a threshold kubelet can never fire +[Node Model Store](node-store.md) describes; a threshold kubelet can never fire floors that cap, because the cache's own collection is then the only space reclaimer. ## Cost model @@ -72,5 +73,5 @@ node-to-node link is slower than each node's own hub path, all-hub finishes a si fan-out sooner. Serving costs CPU on the seed node (measured ≈ 0.25 vCPU per concurrent puller on 2-vCPU nodes, plus the link's bandwidth) — on GPU nodes this shares headroom with inference. -**See also** — [Node Model Store Reference](node-model-store.md) for the cache the trees live in, and -[Model Artifact Reference](model-artifact.md) for the manifest digest the listings are bound to. +**See also** — [Node Model Store](node-store.md) for the cache the trees live in, and +[Model Artifact](artifact.md) for the manifest digest the listings are bound to. diff --git a/docs/reference/model-prefetch.md b/docs/model-store/prefetch.md similarity index 88% rename from docs/reference/model-prefetch.md rename to docs/model-store/prefetch.md index afcf6a074..9599dad87 100644 --- a/docs/reference/model-prefetch.md +++ b/docs/model-store/prefetch.md @@ -1,7 +1,8 @@ -# Model Prefetch Reference +# Model Prefetch > **Purpose** — the three objects that warm a model onto nodes before any Pod asks, the budgets and pins that bound them, and what runs on the node. -> **Audience** users, operators, contributors · **Prerequisites** [Model Artifact Reference](model-artifact.md) · **Read time** ~9 min +> **Audience** users, operators, contributors · **Prerequisites** [Model Artifact](artifact.md) · +> **Read time** ~9 min Warming is declared, not performed: a tenant names an artifact and a target set, and the operator lands the weights through the node cache's ordinary delivery. Three objects divide the concerns — @@ -13,8 +14,8 @@ lands the weights through the node cache's ordinary delivery. Three objects divi | `ModelPrefetch` | namespaced | tenant | one artifact's residency intent: a target set, how many nodes make it available, and its retention | The node's effective configuration is still `NodeModelStore` — the store layer merges above the -cluster defaults under [Model Store Operations](../operation/model-store.md), and the plugin side of -every field is in the [Node Model Store Reference](node-model-store.md). +cluster defaults under [Model Store Operations](operations.md), and the plugin side of +every field is in the [Node Model Store](node-store.md). ## Contents @@ -28,7 +29,7 @@ every field is in the [Node Model Store Reference](node-model-store.md). ## The three objects A `ModelStore` selects its pool with `spec.nodeSelector` and states watermarks and download -limits as overrides field by field — the [pool layer](../operation/model-store.md#the-pool-layer) +limits as overrides field by field — the [pool layer](operations.md#the-pool-layer) defines the override semantics, the overlap condition and the tie-break. A `ModelStoreBinding` is the provisioning point: creating one in a namespace is what grants that @@ -104,7 +105,7 @@ content reports its saturation exactly as a cache full of references does (`gc.C Pinning requires the grant. The nodes' pin lists are written as one union by the prefetch controller — the field's single writer — so two namespaces pinning one node keep both digests; how a pinned digest behaves under collection is the [node model store -reference](node-model-store.md#references-restart-and-collection)'s fact. +reference](node-store.md#references-restart-and-collection)'s fact. `spec.retention.ttlAfterLastUse` unpins a node's copy once nothing mounted it for that long. It is enforced at the hour granularity the node already reports in `lastUsedTime`. @@ -123,11 +124,11 @@ would close the gap lapsed past `ttlAfterLastUse`, were unpinned and are not re- message says how many. It stays False while the shortfall has any other cause, so a lapse never masks a warm-up still running or a delivery that failed. -The [v1 view](model-artifact-views.md) proxies every verb and adds no subresource: the per-node +The [v1 view](views.md) proxies every verb and adds no subresource: the per-node facts already live on the nodes' reports, and the table prints who warms what and how far. --- -**See also** — [Model Artifact Reference](model-artifact.md) (what a prefetch warms) · [Node Model Store Reference](node-model-store.md) (the node side of every field named here) · [Model Store Operations](../operation/model-store.md) (the pool layer's knobs in operation) +**See also** — [Model Artifact](artifact.md) (what a prefetch warms) · [Node Model Store](node-store.md) (the node side of every field named here) · [Model Store Operations](operations.md) (the pool layer's knobs in operation) -**Next** → [Node Model Store Reference](node-model-store.md) — the node object the plugin serves. +**Next** → [Node Model Store](node-store.md) — the node object the plugin serves. diff --git a/docs/reference/model-artifact-views.md b/docs/model-store/views.md similarity index 89% rename from docs/reference/model-artifact-views.md rename to docs/model-store/views.md index 9d657186c..155dcd194 100644 --- a/docs/reference/model-artifact-views.md +++ b/docs/model-store/views.md @@ -1,14 +1,14 @@ -# Model Artifact Views Reference +# Model Artifact Views > **Purpose** — the `v1` views of `ModelArtifact` and `NodeModelStore`, the `progress` subresource > that says where an artifact's content is, who may read it, and how GPUStack server's model-file > handling maps onto this API. -> **Audience** users, operators, console developers · **Prerequisites** [Model Artifact -> Reference](model-artifact.md) · **Read time** ~8 min +> **Audience** users, operators, console developers · **Prerequisites** [Model +> Artifact](artifact.md) · **Read time** ~8 min The aggregated API serves `worker.gpustack.ai/v1` beside the `v1alpha1` resources: the read surface GPUStack server and consoles use. A node's download progress [is stored on -thresholds](node-model-store.md#the-resource); `progress` answers between them, on request, and +thresholds](node-store.md#the-resource); `progress` answers between them, on request, and stores nothing. ## Contents @@ -47,8 +47,8 @@ is written back: an object read through `v1` carries `apiVersion: worker.gpustac which watches the group's preferred version and collects only what that version can delete; without it no NodeModelStore went with its Node on Kubernetes 1.36. The worker creates a deleted object again while its node runs the plugin. -- Both views carry the same fields as `v1alpha1`: the [artifact](model-artifact.md#the-resource) and - the [node store](node-model-store.md#the-resource). +- Both views carry the same fields as `v1alpha1`: the [artifact](artifact.md#the-resource) and + the [node store](node-store.md#the-resource). ## The progress subresource @@ -122,7 +122,7 @@ By capability, not by field. | `resolved_paths` on the worker | the fixed mount path in the consumer's container; host paths are never exposed | by design | | `local_dir` | none: the plugin owns the cache layout | by design | | Listing a worker's model files | `v1` NodeModelStore get, list, watch | none | -| Download to a worker ahead of use | a [ModelPrefetch](model-prefetch.md) naming nodes | none | +| Download to a worker ahead of use | a [ModelPrefetch](prefetch.md) naming nodes | none | | Delete with `cleanup_on_delete` | delete the prefetch; collection after the last reference and its grace | none | | `reset` (retry now) | automatic backoff to `retryTime` | not planned | | One row per source per worker | one entry per digest per node, shared by artifacts with the same content | none | @@ -141,12 +141,12 @@ By capability, not by field. --- -**See also** — [Model Artifact Reference](model-artifact.md) for the artifact itself · -[Node Model Store Reference](node-model-store.md) for each node's entries and the progress write rule · -[Model Deployment Metrics Reference](model-deployment-metrics.md) for the other subresource built +**See also** — [Model Artifact](artifact.md) for the artifact itself · +[Node Model Store](node-store.md) for each node's entries and the progress write rule · +[Model Deployment Metrics](../model-deployment/metrics.md) for the other subresource built the same way. -**Next** → [Node Model Store Reference](node-model-store.md) +**Next** → [Node Model Store](node-store.md) ## Table columns @@ -156,5 +156,5 @@ not by the CRDPrinterColumn machinery — so the columns the `v1alpha1` CRD list resources are a different, smaller set, and a client asking either version gets the columns documented above rather than the CRD's. -**See also** — [Node-to-Node Sync Reference](node-peer-sync.md) for where a node's bytes come from, -and [Node Model Store Reference](node-model-store.md) for the status the views project. +**See also** — [Node-to-Node Sync](peer-sync.md) for where a node's bytes come from, +and [Node Model Store](node-store.md) for the status the views project. diff --git a/docs/operation/rdma.md b/docs/operation/rdma.md index 01c11c169..97db57233 100644 --- a/docs/operation/rdma.md +++ b/docs/operation/rdma.md @@ -56,7 +56,7 @@ token count's. Taking a whole adapter exclusively removes it from every other te buys nothing the hardware was not already doing. For a managed `ModelDeployment`, set `spec.roles[].resources.interface` instead of naming the key -in a Pod template. [Model Deployment Reference](../reference/model-deployment.md) states which key a +in a Pod template. [Model Deployment](../model-deployment/deployment.md) states which key a count selects, and the mixed-fabric refusal. The field allocates a device to each rendered engine Pod and does not assert that the engine has transferred bytes. @@ -255,7 +255,7 @@ Three things to do instead of a quota: ## What an EFA leg needs from the engine image Which engine can use EFA on which leg, beside every other transport, is [the transport -matrix](../reference/engine-versions.md#which-transport-each-engine-can-use). This section is what +matrix](../model-deployment/engine-versions.md#which-transport-each-engine-can-use). This section is what the cells marked "own image" there ask of you. REQUIRED: **an EFA build of Mooncake in the engine image, and no runner image carries one.** The @@ -318,7 +318,7 @@ The image is yours from then on, and nothing in the operator tracks it: - **The base.** Rebuild whenever the deployment moves to another runner image or engine version; the operator keeps rendering the image you named, not the one it would assemble. - **The Mooncake version.** Install the EFA build of the version the base carries — which client - each runner image carries is [its table](../reference/engine-versions.md#the-minimum-per-shape) — + each runner image carries is [its table](../model-deployment/engine-versions.md#the-minimum-per-shape) — and keep the store on that client's minor line: [the store version must match the engine's client](../kv-cache/backend.md#the-store-version-must-match-the-engines-client). - **libfabric**, from the EFA installer, whose version the recipe pins. diff --git a/docs/operation/topology-aware-scheduling.md b/docs/operation/topology-aware-scheduling.md index e30c12f32..2cccf7f79 100644 --- a/docs/operation/topology-aware-scheduling.md +++ b/docs/operation/topology-aware-scheduling.md @@ -234,7 +234,7 @@ spec: The value is a label key, not `region`, `zone`, or a concrete zone name. Valid examples include `topology.kubernetes.io/zone`, `topology.gpustack.ai/rack`, and a selected Topograph tier. Omit the -field for unconstrained TAS placement; the [field contract](../reference/model-deployment.md#topology-placement) +field for unconstrained TAS placement; the [field contract](../model-deployment/deployment.md#topology-placement) defines the implicit hostname level. Changing or removing the field participates in the existing render hash and replaces only affected @@ -340,8 +340,8 @@ before leaving the cluster running. The test does not prove an upgrade of previo --- **See also** — [Topology-Aware Scheduling](../architecture/topology-aware-scheduling.md) (mechanism -and boundaries) · [Model Deployment Reference](../reference/model-deployment.md) (field contract) · +and boundaries) · [Model Deployment](../model-deployment/deployment.md) (field contract) · [Installation Modes](../architecture/installation-modes.md) (chart ownership) -**Next** → [Model Deployment Reference](../reference/model-deployment.md) — configure the serving +**Next** → [Model Deployment](../model-deployment/deployment.md) — configure the serving roles that consume topology-aware admission. diff --git a/docs/reference/commands.md b/docs/reference/commands.md index ae4fc13d2..2cf6fdb23 100644 --- a/docs/reference/commands.md +++ b/docs/reference/commands.md @@ -151,7 +151,7 @@ gpustack-operator device-manager serve --manufacturer=nvidia --no-partitioned The per-node model cache. A CSI node plugin serving inline ephemeral volumes of the driver `model.csi.gpustack.ai`: it authorizes each mount against the Pod's namespace, downloads and verifies a Hugging Face artifact's files once per node, and writes that node's `NodeModelStore` -status. See [Node Model Store Reference](node-model-store.md). +status. See [Node Model Store](../model-store/node-store.md). Runs as the Model Manager DaemonSet beside the `node-driver-registrar` sidecar. Its configuration comes from its node's `NodeModelStore.spec`, not from flags. diff --git a/docs/reference/kv-cache-injection.md b/docs/reference/kv-cache-injection.md index b32bc52d3..1888f4e87 100644 --- a/docs/reference/kv-cache-injection.md +++ b/docs/reference/kv-cache-injection.md @@ -168,7 +168,7 @@ renders an RDMA or EFA resource limit on its engine Pod according to the effecti The backend's member Pods configure their own host network and device access separately. The one exception is the Ascend prefill/decode leg, whose engine Pods mount a host driver -tree read-only — see [How a pair is wired](model-deployment-prefill-decode.md#how-a-pair-is-wired). +tree read-only — see [How a pair is wired](../model-deployment/prefill-decode.md#how-a-pair-is-wired). Two observability variables, `MC_TE_METRIC` and `MC_STORE_CLIENT_METRIC_BANDWIDTH`, are set to `1` when the container has not spoken about them. A value you set yourself is left alone. @@ -210,7 +210,7 @@ under [Refusals and their fixes](#refusals-and-their-fixes). A flag is refused in every spelling the engine's own parser reads as it — a unique prefix, and on the vLLM family an underscored or dotted form — by the rule under -[What the operator owns](model-deployment.md#what-the-operator-owns). +[What the operator owns](../model-deployment/deployment.md#what-the-operator-owns). This applies only to `env`: a value supplied through `envFrom` is invisible to the check and **will be overwritten with no symptom**, so declare Mooncake variables in `env`. diff --git a/docs/settings.md b/docs/settings.md index df9f39273..a677c5d2e 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -47,15 +47,15 @@ kubectl -n gpustack-system patch setting instance-type-derived-from-node --type | `model-deployment-router-image` | `GPUSTACK_MODEL_DEPLOYMENT_ROUTER_IMAGE` | `gpustack/llm-router:v0.1.0` | Image a managed router runs when the `ModelDeployment` does not name one in `spec.router.image`. **It carries every router this operator supports, and which binary runs is decided by the rendered command rather than by the image** — `spec.router.name` picks the binary, and this setting only says where the binaries come from. Its default is safe for every cluster at once in a way `kv-cache-backend-image`'s is not: it runs no model, so it links no accelerator runtime and cannot be paired with the wrong one. The default is this project's own build rather than an upstream tag, because the three routers are compiled from three separate sources and one of them carries a patch this repository ships, so no upstream image holds them together. An image named on the object is used verbatim and is never redirected to the cluster mirror, because it is the user's own reference. | | `model-deployment-router-proxy-image` | `GPUSTACK_MODEL_DEPLOYMENT_ROUTER_PROXY_IMAGE` | `gpustack/mirrored-envoy:distroless-v1.33.2` | Proxy fronting a managed router's endpoint picker. It has **no field on the API** to override it: this operator renders the proxy's configuration against one proxy's configuration schema, so swapping the binary would mean swapping that configuration too. The setting exists for registry redirection and for pinning a release back, not for running a different proxy. | | `model-deployment-routing-sidecar-image` | `GPUSTACK_MODEL_DEPLOYMENT_ROUTING_SIDECAR_IMAGE` | `gpustack/mirrored-llm-d-router-disagg-sidecar:v0.10.0` | Sidecar a decoder runs to accept a remote prefill handoff. Same terms as the proxy above: this operator renders its arguments, so the setting is for redirection and pinning rather than for a different implementation. | -| `model-deployment-tcp-tw-reuse` | `GPUSTACK_MODEL_DEPLOYMENT_TCP_TW_REUSE` | `false` | Render `net.ipv4.tcp_tw_reuse=1` on the prefill half of every SGLang prefill/decode pair, which otherwise [runs out of local ports](reference/engine-versions.md#known-failures-at-the-minimum) under sustained load. **Allow the sysctl on the kubelet of every node that can run such a Pod before turning it on**: without that the Pod fails with `SysctlForbidden`, and so does every replacement. The steps, which Pods it reaches and how to confirm the refusal are under [Letting SGLang prefill Pods reuse TIME-WAIT ports](#letting-sglang-prefill-pods-reuse-time-wait-ports). | -| `model-prefetch-warmup-image` | `GPUSTACK_MODEL_PREFETCH_WARMUP_IMAGE` | `python:3.12-alpine` | Image a [`ModelPrefetch`](reference/model-prefetch.md)'s warm-up pod runs on each target node. The pod mounts the artifact's volume, verifies every file reads back and exits, so the image needs only a shell, `find` and `sha256sum` — it never runs a model. It is a setting because the one thing it must guarantee is that the node can pull it, which only the cluster's own registry arrangement decides; an air-gapped cluster points it at its mirror. | -| `model-artifact-huggingface-endpoint` | `GPUSTACK_MODEL_ARTIFACT_HUGGINGFACE_ENDPOINT` | `https://huggingface.co` | The Hugging Face Hub a [`ModelArtifact`](reference/model-artifact.md) resolves and revalidates against, and the `HF_ENDPOINT` an engine downloading the weights is given; the node plugin downloads from it too. What changing it does is under [Engine delivery](reference/model-artifact.md#engine-delivery). | +| `model-deployment-tcp-tw-reuse` | `GPUSTACK_MODEL_DEPLOYMENT_TCP_TW_REUSE` | `false` | Render `net.ipv4.tcp_tw_reuse=1` on the prefill half of every SGLang prefill/decode pair, which otherwise [runs out of local ports](model-deployment/engine-versions.md#known-failures-at-the-minimum) under sustained load. **Allow the sysctl on the kubelet of every node that can run such a Pod before turning it on**: without that the Pod fails with `SysctlForbidden`, and so does every replacement. The steps, which Pods it reaches and how to confirm the refusal are under [Letting SGLang prefill Pods reuse TIME-WAIT ports](#letting-sglang-prefill-pods-reuse-time-wait-ports). | +| `model-prefetch-warmup-image` | `GPUSTACK_MODEL_PREFETCH_WARMUP_IMAGE` | `python:3.12-alpine` | Image a [`ModelPrefetch`](model-store/prefetch.md)'s warm-up pod runs on each target node. The pod mounts the artifact's volume, verifies every file reads back and exits, so the image needs only a shell, `find` and `sha256sum` — it never runs a model. It is a setting because the one thing it must guarantee is that the node can pull it, which only the cluster's own registry arrangement decides; an air-gapped cluster points it at its mirror. | +| `model-artifact-huggingface-endpoint` | `GPUSTACK_MODEL_ARTIFACT_HUGGINGFACE_ENDPOINT` | `https://huggingface.co` | The Hugging Face Hub a [`ModelArtifact`](model-store/artifact.md) resolves and revalidates against, and the `HF_ENDPOINT` an engine downloading the weights is given; the node plugin downloads from it too. What changing it does is under [Engine delivery](model-store/artifact.md#engine-delivery). | | `model-artifact-https-proxy` | `GPUSTACK_MODEL_ARTIFACT_HTTPS_PROXY` | *(blank)* | HTTPS proxy for resolution, revalidation and the node plugin's downloads, and the `HTTPS_PROXY` given to an engine downloading the weights, where a role's own value wins. Blank keeps the worker's own environment and renders nothing. Only an `http` or `https` URL without credentials is accepted, because the value reaches tenant Pods; a value from the environment that is not refuses the worker's start. | | `model-artifact-no-proxy` | `GPUSTACK_MODEL_ARTIFACT_NO_PROXY` | *(blank)* | Comma-separated hosts that bypass `model-artifact-https-proxy`, for the node plugin too, given to engines as `NO_PROXY` on the same terms. | | `model-artifact-ca-bundle` | `GPUSTACK_MODEL_ARTIFACT_CA_BUNDLE` | *(blank)* | Name of a ConfigMap in the worker's namespace whose `ca.crt` the resolution and the node plugin trust beside the system pool. **It is not given to engine Pods**: they run in tenant namespaces, which cannot mount it. | -| `model-artifact-revalidate-interval` | `GPUSTACK_MODEL_ARTIFACT_REVALIDATE_INTERVAL` | `24h` | How often a resolved Hugging Face artifact's access is checked again, at least `1m`; a Secret change checks it at once. What a refusal does is under [Resolution and revalidation](reference/model-artifact.md#resolution-and-revalidation). | -| `model-artifact-delivery-mode` | `GPUSTACK_MODEL_ARTIFACT_DELIVERY_MODE` | `Engine` | How a Hugging Face artifact's weights reach a `ModelDeployment`: `Engine`, the engine downloads them, or `Node`, the node's [`model-manager` plugin](reference/node-model-store.md) mounts a verified copy. The chart seeds `Node` when it deploys the plugin, and a seed never overrides a stored value. `Node` is refused while the CSIDriver `model.csi.gpustack.ai` does not exist. **Changing it rolls every Hugging Face deployment once**; see [Switch delivery](operation/model-store.md#switch-delivery). | -| `model-store-high-watermark` | `GPUSTACK_MODEL_STORE_HIGH_WATERMARK` | `80` | The node cache filesystem's usage percent above which the plugin removes content no Pod mounts; above the low watermark, at most `95`. On a filesystem shared with kubelet the plugin caps it below kubelet's thresholds ([the capacity rule](operation/model-store.md#the-capacity-rule)). | +| `model-artifact-revalidate-interval` | `GPUSTACK_MODEL_ARTIFACT_REVALIDATE_INTERVAL` | `24h` | How often a resolved Hugging Face artifact's access is checked again, at least `1m`; a Secret change checks it at once. What a refusal does is under [Resolution and revalidation](model-store/artifact.md#resolution-and-revalidation). | +| `model-artifact-delivery-mode` | `GPUSTACK_MODEL_ARTIFACT_DELIVERY_MODE` | `Engine` | How a Hugging Face artifact's weights reach a `ModelDeployment`: `Engine`, the engine downloads them, or `Node`, the node's [`model-manager` plugin](model-store/node-store.md) mounts a verified copy. The chart seeds `Node` when it deploys the plugin, and a seed never overrides a stored value. `Node` is refused while the CSIDriver `model.csi.gpustack.ai` does not exist. **Changing it rolls every Hugging Face deployment once**; see [Switch delivery](model-store/operations.md#switch-delivery). | +| `model-store-high-watermark` | `GPUSTACK_MODEL_STORE_HIGH_WATERMARK` | `80` | The node cache filesystem's usage percent above which the plugin removes content no Pod mounts; above the low watermark, at most `95`. On a filesystem shared with kubelet the plugin caps it below kubelet's thresholds ([the capacity rule](model-store/operations.md#the-capacity-rule)). | | `model-store-low-watermark` | `GPUSTACK_MODEL_STORE_LOW_WATERMARK` | `70` | The usage percent a collection removes down to; at least `1`, below the high watermark. | | `model-store-download-concurrency` | `GPUSTACK_MODEL_STORE_DOWNLOAD_CONCURRENCY` | `8` | Concurrent HTTP requests the plugin makes per node, across every download; `1` to `64`. | | `model-store-peer-sync` | `GPUSTACK_MODEL_STORE_PEER_SYNC` | `true` | Whether a node's plugin may pull a model's content from the plugins of other nodes that hold it published, instead of the hub. | @@ -64,7 +64,7 @@ kubectl -n gpustack-system patch setting instance-type-derived-from-node --type | `instance-access-wildcard-dns` | `GPUSTACK_INSTANCE_ACCESS_WILDCARD_DNS` | *(blank)* | Wildcard DNS for all Instances (e.g. `traefik.me`), used to build a per-Instance domain `.`. Only effective when `instance-access-static-address` is not set. | | `instance-privileged-allowed` | `GPUSTACK_INSTANCE_PRIVILEGED_ALLOWED` | `false` | Whether an Instance may request privileged mode (`spec.privileged`), which escapes the container boundary and exposes the node's devices and kernel surface. Enforced by the Instance admission webhook whenever an Instance **takes** privileged mode — at creation, or through a later change, including one made while it is stopped. An Instance that already runs privileged keeps it: with this off it stays updatable, editable while stopped, and restartable. Like every setting here, a change takes up to 30s to reach the webhook, so an Instance created in that window is judged against the previous value. | | `instance-host-path-volume-allowed` | `GPUSTACK_INSTANCE_HOST_PATH_VOLUME_ALLOWED` | `false` | Whether an Instance may mount a hostPath volume (`spec.additionalVolumes[*].hostPath`), which reaches the node's filesystem. Kept separate from `instance-privileged-allowed` because it grants strictly less — the filesystem, but not the node's devices or kernel — so an admin can allow node-path mounts without allowing a container escape. Enforced whenever an Instance **takes** a hostPath mount, on the same terms; a mount it already has is matched by value, so reordering the list is not a new grant while repointing an entry at a different node path is. | -| `instance-persistent-volume-placement` | `GPUSTACK_INSTANCE_PERSISTENT_VOLUME_PLACEMENT` | `true` | Whether an Instance's persistent claims (`spec.volume.persistent`, `spec.additionalVolumes[*].persistent`) decide where its Pod runs, by the [claim placement rules](reference/model-artifact.md#claim-delivery-and-placement): a bound claim adds its PV's required node affinity, and a claim nothing binds holds the Pod back with the reason in `status.phaseMessage`. **`false` is the escape path**: the Pod is built from the claim names alone, as before, and Kueue's topology-aware scheduling may then assign a node the volume cannot reach, leaving the Pod Pending while its Workload holds quota. Read when a Pod is created, so a change never moves a running Instance; stop and start one to apply it. | +| `instance-persistent-volume-placement` | `GPUSTACK_INSTANCE_PERSISTENT_VOLUME_PLACEMENT` | `true` | Whether an Instance's persistent claims (`spec.volume.persistent`, `spec.additionalVolumes[*].persistent`) decide where its Pod runs, by the [claim placement rules](model-store/artifact.md#claim-delivery-and-placement): a bound claim adds its PV's required node affinity, and a claim nothing binds holds the Pod back with the reason in `status.phaseMessage`. **`false` is the escape path**: the Pod is built from the claim names alone, as before, and Kueue's topology-aware scheduling may then assign a node the volume cannot reach, leaving the Pod Pending while its Workload holds quota. Read when a Pod is created, so a change never moves a running Instance; stop and start one to apply it. | | `workload-fit-affinity` | `GPUSTACK_WORKLOAD_FIT_AFFINITY` | `true` | Whether the Workload webhook pins a logical slice, or a shared request of two or more accelerators, on an operator queue to nodes whose [fit labels](architecture/scheduling-chain.md#per-card-fit-labels) show that one accelerator can still host it. When `true`, Kueue's topology-aware scheduling skips a fragmented node; when `false`, Workloads pass unchanged while the fit labels are still published, so the pin can be switched off on its own. Read on every webhook call, so a change takes up to 30s to reach it, and it applies to Workloads created or updated after that. | | `node-management-manual` | `GPUSTACK_NODE_MANAGEMENT_MANUAL` | `false` | Skip auto-managing nodes. When `false`, the operator auto-onboards discovered nodes by injecting the `gpustack.ai/managed=true` label; when `true`, an administrator must opt nodes in manually. Read per-reconcile. | | `instance-type-mixed-on-node` | `GPUSTACK_INSTANCE_TYPE_MIXED_ON_NODE` | `true` | Whether one node may surface both a GPU and a CPU-only InstanceType. When `true`, a node is summarized into every type it can serve; when `false`, a node with accelerators yields only a GPU InstanceType and a CPU-only node only a general one. Read per-reconcile. | @@ -107,8 +107,8 @@ whatever this setting says. It answers `Ready` at once for every Workload that is not a replica of a multi-role ModelDeployment, so the only thing it changes in this mode is that such a deployment is admitted as a set — see -[Prefill and decode](reference/model-deployment.md#prefill-and-decode), and -[Status](reference/model-deployment-status.md#status) for a set that stays short of quota. +[Prefill and decode](model-deployment/deployment.md#prefill-and-decode), and +[Status](model-deployment/status.md#status) for a set that stays short of quota. **Write the InstanceType against the `InstanceTypeFlavor` catalog.** It is the read-only, os/arch-agnostic view of the pools that actually exist — one entry per grouping the settings above @@ -133,7 +133,7 @@ connections, in that Pod's own network namespace only. The prefill half opens a new TCP connection for every transfer, so over `TCP`, with or without a store, its `TIME-WAIT` sockets [use up its local -ports](reference/engine-versions.md#known-failures-at-the-minimum) under sustained load. +ports](model-deployment/engine-versions.md#known-failures-at-the-minimum) under sustained load. It does not reach the decode half, which accepts transfers rather than opening them; a vLLM role, which keeps its transfer connections open; or a [KV cache backend](kv-cache/backend.md)'s members. diff --git a/docs/walkthrough.md b/docs/walkthrough.md index 7034f1da2..733edcc8a 100644 --- a/docs/walkthrough.md +++ b/docs/walkthrough.md @@ -670,7 +670,7 @@ spec: - **They mount into the workload container only.** The SSH sidecar needs no change: it enters that container's mount namespace per session, so every mount is visible over SSH too. - **A persistent claim places the Instance where it can attach.** The workspace claim and every - `persistent` entry follow the [claim placement rules](./reference/model-artifact.md#claim-delivery-and-placement): + `persistent` entry follow the [claim placement rules](model-store/artifact.md#claim-delivery-and-placement): a node-local volume pins the Instance to its node and that node's capacity; network or `ReadWriteMany` storage avoids the pin. A claim nothing binds creates no Pod and says why in `status.phaseMessage`; [`instance-persistent-volume-placement`](./settings.md#online-adjustable-settings) turns this off. diff --git a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/modeldeploymentrole.go b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/modeldeploymentrole.go index 4f9aa9a3d..084c394d7 100644 --- a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/modeldeploymentrole.go +++ b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/modeldeploymentrole.go @@ -38,7 +38,7 @@ import ( // // A DEPARTURE THIS OPERATOR DID NOT INITIATE IS NOT A ROLLOUT. The replica that left is replaced on // its own, under a new name, while its siblings keep serving — see -// docs/reference/model-deployment.md under "One group per replica" and "Rollout is a rolling +// docs/model-deployment/deployment.md under "One group per replica" and "Rollout is a rolling // replacement". type ModelDeploymentRoleApplyConfiguration struct { // Name identifies the role, and it is also the name of the Kueue PodSet the role becomes. diff --git a/pkg/worker/controllers/worker/model_deployment_docs_test.go b/pkg/worker/controllers/worker/model_deployment_docs_test.go index beb915d86..13f3304ec 100644 --- a/pkg/worker/controllers/worker/model_deployment_docs_test.go +++ b/pkg/worker/controllers/worker/model_deployment_docs_test.go @@ -16,7 +16,7 @@ import ( // // Relative to this package, and checked as a whole line rather than by heading, so moving the table // inside the page is free and moving the page is not. -const _ModelDeploymentDocsPath = "../../../../docs/reference/model-deployment.md" +const _ModelDeploymentDocsPath = "../../../../docs/model-deployment/deployment.md" // TestModelDeploymentOwnedKeysDocs pins the owned-key table on the reference page to the catalog the // code actually enforces. From 7dcadeadfae592580e9e253dcecf9607a0721dc2 Mon Sep 17 00:00:00 2001 From: thxCode Date: Mon, 28 Sep 2026 20:00:42 +0800 Subject: [PATCH 2/2] fixup! docs: move the model store and model deployment pages into domain dirs --- docs/model-store/prefetch.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/model-store/prefetch.md b/docs/model-store/prefetch.md index 9599dad87..6e6c20e99 100644 --- a/docs/model-store/prefetch.md +++ b/docs/model-store/prefetch.md @@ -104,8 +104,8 @@ content reports its saturation exactly as a cache full of references does (`gc.C Pinning requires the grant. The nodes' pin lists are written as one union by the prefetch controller — the field's single writer — so two namespaces pinning one node keep both digests; how -a pinned digest behaves under collection is the [node model store -reference](node-store.md#references-restart-and-collection)'s fact. +a pinned digest behaves under collection is the [Node Model +Store](node-store.md#references-restart-and-collection)'s fact. `spec.retention.ttlAfterLastUse` unpins a node's copy once nothing mounted it for that long. It is enforced at the hour granularity the node already reports in `lastUsedTime`.