From 20a93cb058ad4dd9fbe1a7cad35e0238df52d0c4 Mon Sep 17 00:00:00 2001 From: thxCode Date: Tue, 29 Sep 2026 12:12:18 +0800 Subject: [PATCH 1/2] feat(kv-cache): elect Mooncake leaders with Kubernetes Lease by default --- .agents/skills/gpustack-operator-e2e/SKILL.md | 2 +- .../cases/_kvcache-inject-lib.sh | 1 + .../gpustack-operator-e2e/cases/case-43.sh | 3 + .../gpustack-operator-e2e/cases/case-44.sh | 1 + .../gpustack-operator-e2e/cases/case-65.sh | 6 +- .../gpustack-operator-e2e/cases/case-73.sh | 1 + .../gpustack-operator-e2e/cases/case-75.sh | 55 +++++++++++-- .../gpustack-operator-e2e/cases/case-77.sh | 13 ++- api/worker/v1alpha1/generated.pb.go | 40 ++++++++++ api/worker/v1alpha1/generated.proto | 61 ++++++-------- api/worker/v1alpha1/kv_cache_backend.go | 66 +++++++--------- api/worker/v1alpha1/kv_cache_backend_test.go | 18 +++++ api/worker/v1alpha1/zz_generated.crds.go | 19 ++++- api/worker/zz_generated.openapi.go | 13 ++- docs/kv-cache/backend.md | 26 +++--- docs/kv-cache/leader.md | 79 ++++++++++--------- docs/kv-cache/walkthrough.md | 15 ++-- .../worker/v1alpha1/kvcachebackendleader.go | 51 ++++++------ .../kvcachebackendleaderhighavailability.go | 14 +--- .../controllers/worker/kv_cache_backend.go | 12 +-- .../worker/kv_cache_backend_handover.go | 21 ++--- .../worker/kv_cache_backend_handover_test.go | 14 ++-- .../worker/kv_cache_backend_rollout.go | 2 +- .../worker/kv_cache_backend_test.go | 50 ++++++++---- pkg/worker/kvcache/mooncake/ha_rbac.go | 23 +----- pkg/worker/kvcache/mooncake/ha_rbac_test.go | 72 +++++++++++------ pkg/worker/kvcache/mooncake/keys.go | 9 +-- pkg/worker/kvcache/mooncake/leader_flags.go | 15 ++-- .../kvcache/mooncake/leader_flags_test.go | 17 ++-- .../kvcache/mooncake/leader_workload.go | 25 +++--- .../kvcache/mooncake/leader_workload_test.go | 5 +- .../kvcache/mooncake/member_workload.go | 2 +- .../webhooks/worker/kv_cache_backend.go | 17 ++-- .../webhooks/worker/kv_cache_backend_test.go | 32 +++++--- 34 files changed, 465 insertions(+), 335 deletions(-) diff --git a/.agents/skills/gpustack-operator-e2e/SKILL.md b/.agents/skills/gpustack-operator-e2e/SKILL.md index d8724e868..4b63df96e 100644 --- a/.agents/skills/gpustack-operator-e2e/SKILL.md +++ b/.agents/skills/gpustack-operator-e2e/SKILL.md @@ -120,7 +120,7 @@ Each case is self-contained; its header (see **Case header contract**) states go | 71 | A Ready router becomes `status.endpoint`; inability to pull the upstream router image is a stated SKIP rather than loss of CASE 70's lifecycle coverage | `pkg/worker/controllers/worker/model_deployment_status.go`, the default llm-d-router image contract | yes (confirm) | As CASE 70 plus pull access to the default llm-d-router and Envoy images; AUTO-SKIP only for an image-pull reason | | 73 | An engine under KV turnover writes the shared store, and the other replica's replay of the same prefixes is the reuse the chain exists for — the write half is a guard, the read half a KNOWN-FAILURE DETECTOR pair (case-67 polarity) that FAILS the day cross-replica reuse starts working, and must then be inverted into positive guards | `pkg/worker/controllers/worker/model_deployment_connector.go`, `pkg/worker/controllers/worker/model_deployment_binding.go`, `pkg/worker/kvcache/inject/**`, the engine image pin | yes (confirm) | A real accelerator pool with at least TWO free exclusive cards, model weights hostPath-staged on the accelerator nodes (`E2E_VB_WEIGHTS`, default `/mnt/kvcache-weights`), `E2E_VB_INSTANCE_TYPE` naming the accelerated InstanceType (exit 2 without it), and a registry the cluster can pull the Mooncake image from — a CUDA-only tag crashes members on CPU-only nodes, so the image must be CPU-capable. Where no two cards are free, `E2E_VB_EXISTING_BACKEND` + `E2E_VB_EXISTING_DOMAIN` + `E2E_VB_EXISTING_SERVICES` together run every verdict row against an existing engine pair and create nothing. The first case in the suite to run real vLLM engines; every verdict rides on counters (`master_key_count`, `vllm:external_prefix_cache_hits_total`, `mem_cache_hit_nums_`), never on TTFT | | 74 | `leader.highAvailability.snapshot` is not in the installed schema — a strict create is refused naming it as an unknown field, a lenient one is accepted with the block pruned — and the store's snapshot flags are refused in leader.extraArgs on create and on an update to a running backend, each refusal on the entry with its reason (a restore can serve another key's bytes; the other keys are read only under a refused switch); the manifest without them, and an image-only update, are accepted as the positive baseline | `api/worker/v1alpha1/kv_cache_backend.go`, `pkg/worker/kvcache/mooncake/keys.go`, `pkg/worker/webhooks/worker/kv_cache_backend.go` | yes (confirm) | any (no GPU, no RDMA, no storage class); the creates and the updates are server-side dry runs, and the one backend the case persists selects no node, so it renders a leader and no member Pod | -| 75 | An image bump on a replicated leader emits the `KVCacheLeaderHandover` event — read in namespace `default`, where a cluster-scoped backend's events land — with its cumulative count equal to the lease's `leaseTransitions`, every leader pod a replacement on the new image, exactly one ready, and the backend settling Ready with every health condition True (`PoolWrites` reports write activity, not health, and reads Unknown on this idle backend); the only mutation is the spec patch, no pod is deleted by hand | `pkg/worker/controllers/worker/kv_cache_backend_handover.go` | yes (confirm) | any (no GPU, no RDMA) + a registry the cluster can pull BOTH pinned Mooncake tags from (`E2E_MOONCAKE_IMAGE` start / `E2E_MOONCAKE_ROLLOUT_IMAGE` target; the pair must each parse this operator's argv, carry the lease backend, and run on CPU-only nodes) | +| 75 | The default single leader holds a Lease; `electionBackend: None` with three replicas is refused; scaling from one to three preserves the first leader Pod; an image bump then emits `KVCacheLeaderHandover` with its count equal to the Lease transitions, replaces all leader Pods, and settles Ready with one serving replica | `pkg/worker/controllers/worker/kv_cache_backend_handover.go`, `pkg/worker/kvcache/mooncake/leader_workload.go` | yes (confirm) | any (no GPU, no RDMA) + a registry the cluster can pull BOTH pinned Mooncake tags from (`E2E_MOONCAKE_IMAGE` start / `E2E_MOONCAKE_ROLLOUT_IMAGE` target; both must parse this operator's argv, carry the lease backend, and run on CPU-only nodes) | | 76 | `RolloutComplete` stays truthful through a second mid-update (every sample is `Unknown/UpdateNotObserved`, `False/Progressing` or `True/Complete`, never a deadline-ish stall), the update converges with the election gate intact, and the member-re-registration dip clears within its window; the case never gates on `kubectl rollout status`, which times out on every multi-replica leader rollout by construction | `pkg/worker/controllers/worker/kv_cache_backend_rollout.go` | yes (confirm) | as CASE 75 (the same image pair and clauses) | | 77 | The multi-tenant ledger gate: an unregistered tenant's put is refused `-1701` while the client itself stays healthy, and a Pool+Binding whose `domain.name` is the tenant id admits the identical put; the tenant rides the keyword `tenant_id=` (the next positional slot is a TransferEngine pointer and raises), and the teardown drains the domain because a held domain blocks pool deletion open-ended | `pkg/worker/controllers/worker/kv_cache_pool.go` (the domain registration pass), `pkg/worker/kvcache/mooncake/**` (the master argv and lease render) | yes (confirm) | any (no GPU, no RDMA) + a registry the cluster can pull the Mooncake image from (`E2E_MOONCAKE_IMAGE`, CPU-capable, carrying the python client); the backend must be a replicated HA leader — the k8s:// master address and the probe's member Role exist only above one replica | | 78 | Scaling a role moves its queue's admitted quota by exactly one replica in each direction, and touches no replica it did not add or remove | `pkg/worker/controllers/worker/model_deployment{,_pod_group,_rollout}.go`, `api/worker/v1alpha1/model_deployment.go` | yes (confirm) | any (no GPU) + an InstanceType, the pool's entrance LocalQueue in ``, and room in the pool for three replicas of one role. Optionally `E2E_MD_INSTANCE_TYPE` / `E2E_MD_IMAGE` / `E2E_MD_SETTLE` | diff --git a/.agents/skills/gpustack-operator-e2e/cases/_kvcache-inject-lib.sh b/.agents/skills/gpustack-operator-e2e/cases/_kvcache-inject-lib.sh index fbfaf1649..da051d10d 100644 --- a/.agents/skills/gpustack-operator-e2e/cases/_kvcache-inject-lib.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/_kvcache-inject-lib.sh @@ -163,6 +163,7 @@ spec: connection: managed: leader: + electionBackend: None multiTenancy: true members: - nodeSelector: {kubernetes.io/os: linux} diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-43.sh b/.agents/skills/gpustack-operator-e2e/cases/case-43.sh index d378e4c2a..987ad5668 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-43.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-43.sh @@ -197,6 +197,7 @@ spec: connection: managed: leader: + electionBackend: None multiTenancy: true members: - nodeSelector: {kubernetes.io/os: linux} @@ -585,6 +586,7 @@ spec: connection: managed: leader: + electionBackend: None multiTenancy: true members: - nodeSelector: {gpustack.ai/kvc-e2e-absent: "true"} @@ -808,6 +810,7 @@ spec: connection: managed: leader: + electionBackend: None multiTenancy: true members: - nodeSelector: {gpustack.ai/kvc-e2e-absent: "true"} diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-44.sh b/.agents/skills/gpustack-operator-e2e/cases/case-44.sh index 0080b951d..f9d20f4c1 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-44.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-44.sh @@ -184,6 +184,7 @@ spec: connection: managed: leader: + electionBackend: None multiTenancy: true # This case times a lease lapse, so it pins the lease rather than inheriting whatever default # the operator renders. At the operator's own five minutes the lapse below would have to sleep diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-65.sh b/.agents/skills/gpustack-operator-e2e/cases/case-65.sh index e6fbbfdb9..f07d9733e 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-65.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-65.sh @@ -271,11 +271,11 @@ spec: protocol: TCP connection: managed: - # Required by the schema, and empty is the shape this case wants: one leader process, no - # election. Omitting the key is refused at apply, which the host-directory gate above used to + # Required by the schema; this case uses one leader process without election. + # Omitting the key is refused at apply, which the host-directory gate above used to # hide -- that gate exits 0, so on any cluster without the directory this case reported # nothing rather than reporting that it could not build its own fixture. - leader: {} + leader: {electionBackend: None} members: - nodeSelector: {kubernetes.io/os: linux} medium: DRAM diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-73.sh b/.agents/skills/gpustack-operator-e2e/cases/case-73.sh index dde8b2977..e8e1c4c14 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-73.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-73.sh @@ -234,6 +234,7 @@ spec: connection: managed: leader: + electionBackend: None multiTenancy: true members: - nodeSelector: {kubernetes.io/os: linux} diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-75.sh b/.agents/skills/gpustack-operator-e2e/cases/case-75.sh index eb0f4fb30..26c2307dc 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-75.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-75.sh @@ -26,14 +26,16 @@ # the operator's own namespace yields nothing, because events for the cluster-scoped # backend land in namespace default -- every event read below is against default. # -# Inputs: All real, nothing mocked. One KVCacheBackend (3-replica HA leader, multi-tenancy on, -# no snapshot) created on the start image; after it is Ready and steady the ONLY -# mutation is `kubectl patch` of spec.image to the rollout target. A clean baseline is +# Inputs: All real, nothing mocked. One KVCacheBackend starts with the default single-replica +# Kubernetes election, then scales to three without replacing the first leader Pod. +# After it is Ready and steady the rollout's ONLY mutation is an image patch. A clean baseline is # asserted first: zero KVCacheLeaderHandover events for this backend anywhere. The # pod-deletion trigger of the failover cases is deliberately NOT used here, so the # event observed can only be the rollout's. # -# Expected: - zero KVCacheLeaderHandover events for the backend at steady state (clean baseline); +# Expected: - the default single leader holds a Lease; None with three replicas is refused; +# - scaling to three preserves the original leader Pod UID; +# - zero KVCacheLeaderHandover events for the backend at steady state (clean baseline); # - after the patch, the event exists, read from namespace default; # - the event's cumulative count equals the lease's leaseTransitions after the rollout # (on a fresh backend the baseline is 0, so the count also equals the delta); @@ -140,9 +142,8 @@ spec: connection: managed: leader: - replicas: 3 + replicas: 1 multiTenancy: true - highAvailability: {} members: - nodeSelector: {kubernetes.io/os: linux} medium: DRAM @@ -160,6 +161,48 @@ if ! wait_for deployment "$LEADER" '{.status.readyReplicas}' 1 300 >/dev/null; t results; exit 1 fi +FIRST_UID="$(kubectl -n "$NS" get pod -l "$LEADER_SEL" -o jsonpath='{.items[0].metadata.uid}' 2>/dev/null)" +HOLDER="$(kubectl -n "$NS" get leases.coordination.k8s.io "$LEADER" -o jsonpath='{.spec.holderIdentity}' 2>/dev/null)" +ARGS="$(kubectl -n "$NS" get deployment "$LEADER" -o jsonpath='{.spec.template.spec.containers[0].args}' 2>/dev/null)" +if [ -n "$FIRST_UID" ] && [ -n "$HOLDER" ] && [[ "$ARGS" == *-enable_ha=true* ]]; then + record PASS "the default single leader already elects" "leader Pod UID=${FIRST_UID}; Lease holder=${HOLDER}" +else + record FAIL "the default single leader already elects" "UID=${FIRST_UID:-}; holder=${HOLDER:-}; args=${ARGS:-}" + results; exit 1 +fi + +if REFUSAL="$(kubectl patch kvcachebackends.worker.gpustack.ai "$BACKEND" --dry-run=server --type merge \ + -p '{"spec":{"connection":{"managed":{"leader":{"replicas":3,"electionBackend":"None"}}}}}' 2>&1)"; then + record FAIL "None with three replicas is refused" "server accepted the invalid patch" + results; exit 1 +elif [[ "$REFUSAL" == *"leader.electionBackend"* ]]; then + record PASS "None with three replicas is refused" "$REFUSAL" +else + record FAIL "None with three replicas is refused" "$REFUSAL" + results; exit 1 +fi + +if ! kubectl patch kvcachebackends.worker.gpustack.ai "$BACKEND" --type merge \ + -p '{"spec":{"connection":{"managed":{"leader":{"replicas":3}}}}}' >/dev/null 2>&1; then + record FAIL "scaling to three is accepted" "replica patch failed" + results; exit 1 +fi +if ! wait_for deployment "$LEADER" '{.status.replicas}' 3 300 >/dev/null; then + record FAIL "scaling to three converges" "the Deployment never reached three replicas" + results; exit 1 +fi +if ! wait_for kvcachebackends.worker.gpustack.ai "$BACKEND" '{.status.phase}' Ready 300 >/dev/null; then + record FAIL "scaling to three converges" "the backend did not return to Ready" + results; exit 1 +fi +SCALE_UIDS="$(kubectl -n "$NS" get pod -l "$LEADER_SEL" -o jsonpath='{range .items[*]}{.metadata.uid}{" "}{end}' 2>/dev/null)" +if [ -n "$FIRST_UID" ] && [[ " $SCALE_UIDS " == *" $FIRST_UID "* ]]; then + record PASS "scaling to three preserves the first leader Pod" "first UID=${FIRST_UID}; current UIDs=${SCALE_UIDS}" +else + record FAIL "scaling to three preserves the first leader Pod" "first UID=${FIRST_UID}; current UIDs=${SCALE_UIDS}" + results; exit 1 +fi + # Steady means the election has stopped moving: give a settled lease one more read after a short # quiet window, so a provisioning-time flap is counted into the baseline instead of misread as the # rollout's own handover. diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-77.sh b/.agents/skills/gpustack-operator-e2e/cases/case-77.sh index 67a897359..b8980fa9c 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-77.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-77.sh @@ -21,12 +21,11 @@ # operator's system namespace. The store image (E2E_MOONCAKE_IMAGE, default pinned to # a CPU-capable tag carrying both the lease backend and the mooncake python client) is # also the probe pod's image, so the client library always matches the master. The -# backend MUST be a replicated HA leader: the client reaches the master through +# backend uses a replicated HA leader. The client reaches the master through # k8s:///-leader, which resolves through the election's Lease, and the -# probe's own authorization rides the member Role the operator renders only above one -# replica -- a single-replica leader renders neither the Lease nor the Role, and the -# client fails at setup instead of exercising the gate (measured; it is a shape -# constraint, not a defect). +# probe's authorization rides the member Role. The default single-replica leader now +# also renders that Lease and Role; replication is retained here to exercise the +# multi-replica shape alongside the tenant gate. # # Inputs: All real, nothing mocked. One KVCacheBackend (3-replica HA leader, multi-tenancy # on, no snapshot, no Pool, no Binding at first, the leader's read-lease TTL passed @@ -228,7 +227,7 @@ fi # The k8s:// prerequisites, asserted rather than assumed: the member Role exists (the probe's # authorization) and the Lease has a holder (the master address resolves through it). A -# single-replica leader renders neither, and the failure mode is a client that cannot even setup. +# a missing grant or holder prevents the client from reaching the elected master. HOLDER="" for ((i = 0; i < 120; i += 3)); do HOLDER="$(kubectl -n "$NS" get leases.coordination.k8s.io "${BACKEND}-leader" -o jsonpath='{.spec.holderIdentity}' 2>/dev/null)" @@ -241,7 +240,7 @@ if wait_for roles.rbac.authorization.k8s.io "${BACKEND}-member" '{.metadata.name "role ${BACKEND}-member exists, lease ${BACKEND}-leader holder='${HOLDER}'" else record FAIL "the election is rendered (member Role + Lease with a holder)" \ - "role or lease holder missing; a single-replica leader renders neither and the k8s:// client cannot run" + "role or lease holder missing; the k8s:// client cannot run" results; exit 1 fi diff --git a/api/worker/v1alpha1/generated.pb.go b/api/worker/v1alpha1/generated.pb.go index ffd0be073..92e112c41 100644 --- a/api/worker/v1alpha1/generated.pb.go +++ b/api/worker/v1alpha1/generated.pb.go @@ -3433,6 +3433,11 @@ func (m *KVCacheBackendLeader) MarshalToSizedBuffer(dAtA []byte) (int, error) { _ = i var l int _ = l + i -= len(m.ElectionBackend) + copy(dAtA[i:], m.ElectionBackend) + i = encodeVarintGenerated(dAtA, i, uint64(len(m.ElectionBackend))) + i-- + dAtA[i] = 0x3a if len(m.ExtraEnv) > 0 { for iNdEx := len(m.ExtraEnv) - 1; iNdEx >= 0; iNdEx-- { { @@ -9888,6 +9893,8 @@ func (m *KVCacheBackendLeader) Size() (n int) { n += 1 + l + sovGenerated(uint64(l)) } } + l = len(m.ElectionBackend) + n += 1 + l + sovGenerated(uint64(l)) return n } @@ -12691,6 +12698,7 @@ func (this *KVCacheBackendLeader) String() string { `MultiTenancy:` + valueToStringGenerated(this.MultiTenancy) + `,`, `ExtraArgs:` + fmt.Sprintf("%v", this.ExtraArgs) + `,`, `ExtraEnv:` + repeatedStringForExtraEnv + `,`, + `ElectionBackend:` + fmt.Sprintf("%v", this.ElectionBackend) + `,`, `}`, }, "") return s @@ -24472,6 +24480,38 @@ func (m *KVCacheBackendLeader) Unmarshal(dAtA []byte) error { return err } iNdEx = postIndex + case 7: + if wireType != 2 { + return fmt.Errorf("proto: wrong wireType = %d for field ElectionBackend", wireType) + } + var stringLen uint64 + for shift := uint(0); ; shift += 7 { + if shift >= 64 { + return ErrIntOverflowGenerated + } + if iNdEx >= l { + return io.ErrUnexpectedEOF + } + b := dAtA[iNdEx] + iNdEx++ + stringLen |= uint64(b&0x7F) << shift + if b < 0x80 { + break + } + } + intStringLen := int(stringLen) + if intStringLen < 0 { + return ErrInvalidLengthGenerated + } + postIndex := iNdEx + intStringLen + if postIndex < 0 { + return ErrInvalidLengthGenerated + } + if postIndex > l { + return io.ErrUnexpectedEOF + } + m.ElectionBackend = string(dAtA[iNdEx:postIndex]) + iNdEx = postIndex default: iNdEx = preIndex skippy, err := skipGenerated(dAtA[iNdEx:]) diff --git a/api/worker/v1alpha1/generated.proto b/api/worker/v1alpha1/generated.proto index 822520953..48ba725c6 100644 --- a/api/worker/v1alpha1/generated.proto +++ b/api/worker/v1alpha1/generated.proto @@ -1487,13 +1487,10 @@ message KVCacheBackendLeader { // Replicas is how many leader processes run, of which exactly one serves at a time. The rest are // standbys: they hold no data, answer no request, and exist to take over. // - // - More than one REQUIRES HighAvailability. Electing a leader among several needs a leadership - // record, and the webhook refuses the pair without one rather than silently running two - // leaders against the same members. - // - Raising this past one TURNS THE ELECTION ON, and the flip is re-evaluated on every - // reconcile rather than decided at create. It restarts the leader and rolls every member — - // the member's master entry changes shape with it — so the store's cached contents do not - // survive the crossing. The same holds on the way back down to one. + // - More than one REQUIRES ElectionBackend to be Kubernetes. The webhook refuses None with several + // replicas, and the renderer clamps it to one if admission is unavailable. + // - Changing the replica count does not change the election mode. The default Kubernetes mode + // runs the election even at one replica, so scaling it up does not restart the first leader. // - Raising this adds no capacity, which members do. The ceiling is here to catch the reading // that it does, and it is duplicated in the webhook on purpose: this one still holds when // the webhook is not installed, which is when a second leader would be rendered rather than @@ -1504,27 +1501,25 @@ message KVCacheBackendLeader { // +k8s:validation:maximum=5 optional int32 replicas = 1; - // HighAvailability elects the leader through a Kubernetes Lease, and it is what allows Replicas - // above 1. The election itself needs no settings: the Lease is named after this backend, so - // there is no connection target to supply, and the API access it needs is rendered beside the - // workload. What the block does carry is how members find the leader it elects. - // - // - Unset, the leader runs as a single process exactly as before — no election flag, no extra - // object, the command line it ran before this field existed. - // - Set with Replicas at 1, the ELECTION is INERT: one process has nothing to elect between, - // so no election flag, Lease or API token is rendered until Replicas rises above 1. That - // makes an empty block safe to set up front on a store image built without the k8s-lease - // backend, whose master fails at startup the moment the election flags appear — those flags - // arrive only when there is something for them to elect. Snapshot is the exception and says - // so on itself. - // - With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile - // interval. Each replica seeds its tenant quota policy at its own start, so a standby that - // took over after a quota was raised applies the older, lower ceiling, and an over-quota - // write in this store is not refused — it evicts that tenant's own older objects, - // irreversibly and without moving any counter. The quota itself is not lost: the pool - // reconciler is the authority and writes the difference back on its next pass. + // HighAvailability configures how members find the elected leader. Election itself is selected + // by ElectionBackend, so this block is optional even when Replicas exceeds one. optional KVCacheBackendLeaderHighAvailability highAvailability = 2; + // ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica, + // so scaling up does not change the first leader's startup flags. None is for a single leader + // whose image cannot use Kubernetes election. More than one replica with None is refused. + // The enum can add other backends when they are supported. + // + // With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile interval. + // Each replica seeds its tenant quota policy at its own start, so a standby that took over after + // a quota was raised applies the older, lower ceiling. An over-quota write evicts that tenant's + // own older objects, irreversibly and without moving any counter. The quota itself is not lost: + // the pool reconciler writes the difference back on its next pass. + // + // +k8s:validation:default="Kubernetes" + // +k8s:validation:enum=["None","Kubernetes"] + optional string electionBackend = 7; + // AllocationStrategy is how the leader picks which member takes a new write. Random spreads // them; FreeRatioFirst biases toward the emptier member. // @@ -1595,17 +1590,11 @@ message KVCacheBackendLeader { repeated InstanceEnvVar extraEnv = 6; } -// KVCacheBackendLeaderHighAvailability turns leader election on, and carries how members find the -// leader it elects. -// -// DECLARING THE BLOCK IS THE SWITCH, and there is no key inside it to turn the feature back off. -// An `enabled: false` beside `replicas: 3` would be a third state that admission would have to -// adjudicate and every reader would have to remember, while presence has no such state. Lease -// tuning — duration, renew deadline — can also be added here later without a breaking change. +// KVCacheBackendLeaderHighAvailability configures how members find the elected leader. // -// A standby REPLICATES NOTHING. The store's operation log is the only way to feed one, and it runs -// on a leadership backend this operator's image cannot carry, so a failover or a restart starts -// from an empty cache. The store's snapshot is not offered either: restoring one can make the cache +// A standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM +// key index. A member's local disk tier can re-register keys it has fully offloaded after the +// election. The store's snapshot is not offered: restoring one can make the cache // serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs. message KVCacheBackendLeaderHighAvailability { // MemberAddressing selects how a member is told to find the master once an election runs. Both diff --git a/api/worker/v1alpha1/kv_cache_backend.go b/api/worker/v1alpha1/kv_cache_backend.go index ebd0b8cfe..bbd6b3291 100644 --- a/api/worker/v1alpha1/kv_cache_backend.go +++ b/api/worker/v1alpha1/kv_cache_backend.go @@ -278,13 +278,10 @@ type KVCacheBackendLeader struct { // Replicas is how many leader processes run, of which exactly one serves at a time. The rest are // standbys: they hold no data, answer no request, and exist to take over. // - // - More than one REQUIRES HighAvailability. Electing a leader among several needs a leadership - // record, and the webhook refuses the pair without one rather than silently running two - // leaders against the same members. - // - Raising this past one TURNS THE ELECTION ON, and the flip is re-evaluated on every - // reconcile rather than decided at create. It restarts the leader and rolls every member — - // the member's master entry changes shape with it — so the store's cached contents do not - // survive the crossing. The same holds on the way back down to one. + // - More than one REQUIRES ElectionBackend to be Kubernetes. The webhook refuses None with several + // replicas, and the renderer clamps it to one if admission is unavailable. + // - Changing the replica count does not change the election mode. The default Kubernetes mode + // runs the election even at one replica, so scaling it up does not restart the first leader. // - Raising this adds no capacity, which members do. The ceiling is here to catch the reading // that it does, and it is duplicated in the webhook on purpose: this one still holds when // the webhook is not installed, which is when a second leader would be rendered rather than @@ -295,27 +292,25 @@ type KVCacheBackendLeader struct { // +k8s:validation:maximum=5 Replicas *int32 `json:"replicas,omitempty" protobuf:"varint,1,opt,name=replicas"` - // HighAvailability elects the leader through a Kubernetes Lease, and it is what allows Replicas - // above 1. The election itself needs no settings: the Lease is named after this backend, so - // there is no connection target to supply, and the API access it needs is rendered beside the - // workload. What the block does carry is how members find the leader it elects. - // - // - Unset, the leader runs as a single process exactly as before — no election flag, no extra - // object, the command line it ran before this field existed. - // - Set with Replicas at 1, the ELECTION is INERT: one process has nothing to elect between, - // so no election flag, Lease or API token is rendered until Replicas rises above 1. That - // makes an empty block safe to set up front on a store image built without the k8s-lease - // backend, whose master fails at startup the moment the election flags appear — those flags - // arrive only when there is something for them to elect. Snapshot is the exception and says - // so on itself. - // - With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile - // interval. Each replica seeds its tenant quota policy at its own start, so a standby that - // took over after a quota was raised applies the older, lower ceiling, and an over-quota - // write in this store is not refused — it evicts that tenant's own older objects, - // irreversibly and without moving any counter. The quota itself is not lost: the pool - // reconciler is the authority and writes the difference back on its next pass. + // HighAvailability configures how members find the elected leader. Election itself is selected + // by ElectionBackend, so this block is optional even when Replicas exceeds one. HighAvailability *KVCacheBackendLeaderHighAvailability `json:"highAvailability,omitempty" protobuf:"bytes,2,opt,name=highAvailability"` + // ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica, + // so scaling up does not change the first leader's startup flags. None is for a single leader + // whose image cannot use Kubernetes election. More than one replica with None is refused. + // The enum can add other backends when they are supported. + // + // With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile interval. + // Each replica seeds its tenant quota policy at its own start, so a standby that took over after + // a quota was raised applies the older, lower ceiling. An over-quota write evicts that tenant's + // own older objects, irreversibly and without moving any counter. The quota itself is not lost: + // the pool reconciler writes the difference back on its next pass. + // + // +k8s:validation:default="Kubernetes" + // +k8s:validation:enum=["None","Kubernetes"] + ElectionBackend string `json:"electionBackend,omitempty" protobuf:"bytes,7,opt,name=electionBackend"` + // AllocationStrategy is how the leader picks which member takes a new write. Random spreads // them; FreeRatioFirst biases toward the emptier member. // @@ -393,17 +388,16 @@ func (in KVCacheBackendLeader) MultiTenancyEnabled() bool { return in.MultiTenancy == nil || *in.MultiTenancy } -// KVCacheBackendLeaderHighAvailability turns leader election on, and carries how members find the -// leader it elects. -// -// DECLARING THE BLOCK IS THE SWITCH, and there is no key inside it to turn the feature back off. -// An `enabled: false` beside `replicas: 3` would be a third state that admission would have to -// adjudicate and every reader would have to remember, while presence has no such state. Lease -// tuning — duration, renew deadline — can also be added here later without a breaking change. +// KubernetesElectionEnabled applies the schema default to objects built without API server defaulting. +func (in KVCacheBackendLeader) KubernetesElectionEnabled() bool { + return in.ElectionBackend == "" || in.ElectionBackend == "Kubernetes" +} + +// KVCacheBackendLeaderHighAvailability configures how members find the elected leader. // -// A standby REPLICATES NOTHING. The store's operation log is the only way to feed one, and it runs -// on a leadership backend this operator's image cannot carry, so a failover or a restart starts -// from an empty cache. The store's snapshot is not offered either: restoring one can make the cache +// A standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM +// key index. A member's local disk tier can re-register keys it has fully offloaded after the +// election. The store's snapshot is not offered: restoring one can make the cache // serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs. type KVCacheBackendLeaderHighAvailability struct { // MemberAddressing selects how a member is told to find the master once an election runs. Both diff --git a/api/worker/v1alpha1/kv_cache_backend_test.go b/api/worker/v1alpha1/kv_cache_backend_test.go index 48f122757..3a1dfb975 100644 --- a/api/worker/v1alpha1/kv_cache_backend_test.go +++ b/api/worker/v1alpha1/kv_cache_backend_test.go @@ -34,6 +34,24 @@ func memberSchema(t *testing.T) extension.JSONSchemaProps { return *schema.Items.Schema } +func TestKVCacheBackendElectionBackendSchema(t *testing.T) { + crd := GetCustomResourceDefinitions()["KVCacheBackend"] + require.NotNil(t, crd) + leader := *crd.Spec.Versions[0].Schema.OpenAPIV3Schema + for _, level := range []string{"spec", "connection", "managed", "leader"} { + next, ok := leader.Properties[level] + require.True(t, ok, "missing %s", level) + leader = next + } + backend, ok := leader.Properties["electionBackend"] + require.True(t, ok) + require.NotNil(t, backend.Default) + assert.Equal(t, `"Kubernetes"`, string(backend.Default.Raw)) + require.Len(t, backend.Enum, 2) + assert.Equal(t, []string{`"None"`, `"Kubernetes"`}, + []string{string(backend.Enum[0].Raw), string(backend.Enum[1].Raw)}) +} + // memberStatusListSchema returns status.members as the generated CRD carries it. func memberStatusListSchema(t *testing.T) extension.JSONSchemaProps { t.Helper() diff --git a/api/worker/v1alpha1/zz_generated.crds.go b/api/worker/v1alpha1/zz_generated.crds.go index 009a7ac3f..d06f6fd17 100644 --- a/api/worker/v1alpha1/zz_generated.crds.go +++ b/api/worker/v1alpha1/zz_generated.crds.go @@ -2459,6 +2459,21 @@ func crd_gpustack_api_worker_v1alpha1_KVCacheBackend() *v1.CustomResourceDefinit }, }, }, + "electionBackend": { + Description: "ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica,\nso scaling up does not change the first leader's startup flags. None is for a single leader\nwhose image cannot use Kubernetes election. More than one replica with None is refused.\nThe enum can add other backends when they are supported.\nWith MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile interval.\nEach replica seeds its tenant quota policy at its own start, so a standby that took over after\na quota was raised applies the older, lower ceiling. An over-quota write evicts that tenant's\nown older objects, irreversibly and without moving any counter. The quota itself is not lost:\nthe pool reconciler writes the difference back on its next pass.", + Type: "string", + Default: &v1.JSON{ + Raw: []byte(`"Kubernetes"`), + }, + Enum: []v1.JSON{ + { + Raw: []byte(`"None"`), + }, + { + Raw: []byte(`"Kubernetes"`), + }, + }, + }, "extraArgs": { Description: "ExtraArgs passes flags this API does not enumerate straight through to the leader, after\nthe derived ones. Each entry is one flag token of its own, \"-flag\" or \"-flag=value\", and the\nentries render verbatim in the order written. An entry whose key — what precedes the first\n\"=\" once the leading dashes are off — collides with a flag rendered from a field above is\nrefused at admission, because two sources for one flag make the rendered command ambiguous.\nEVERY VALUE HERE IS WORLD-READABLE: stored verbatim on this cluster-scoped object, then\nrendered into the leader container's argv, readable by anyone who can reach the Pod or the\nDeployment, for the life of the object. A credential does not belong here, and since this\noperator renders no flag that carries one, this field is the only way one arrives.", Type: "array", @@ -2499,7 +2514,7 @@ func crd_gpustack_api_worker_v1alpha1_KVCacheBackend() *v1.CustomResourceDefinit XListType: ptr.To[string]("map"), }, "highAvailability": { - Description: "HighAvailability elects the leader through a Kubernetes Lease, and it is what allows Replicas\nabove 1. The election itself needs no settings: the Lease is named after this backend, so\nthere is no connection target to supply, and the API access it needs is rendered beside the\nworkload. What the block does carry is how members find the leader it elects.\n- Unset, the leader runs as a single process exactly as before — no election flag, no extra\nobject, the command line it ran before this field existed.\n- Set with Replicas at 1, the ELECTION is INERT: one process has nothing to elect between,\nso no election flag, Lease or API token is rendered until Replicas rises above 1. That\nmakes an empty block safe to set up front on a store image built without the k8s-lease\nbackend, whose master fails at startup the moment the election flags appear — those flags\narrive only when there is something for them to elect. Snapshot is the exception and says\nso on itself.\n- With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile\ninterval. Each replica seeds its tenant quota policy at its own start, so a standby that\ntook over after a quota was raised applies the older, lower ceiling, and an over-quota\nwrite in this store is not refused — it evicts that tenant's own older objects,\nirreversibly and without moving any counter. The quota itself is not lost: the pool\nreconciler is the authority and writes the difference back on its next pass.", + Description: "HighAvailability configures how members find the elected leader. Election itself is selected\nby ElectionBackend, so this block is optional even when Replicas exceeds one.", Type: "object", Properties: map[string]v1.JSONSchemaProps{ "memberAddressing": { @@ -2529,7 +2544,7 @@ func crd_gpustack_api_worker_v1alpha1_KVCacheBackend() *v1.CustomResourceDefinit Nullable: true, }, "replicas": { - Description: "Replicas is how many leader processes run, of which exactly one serves at a time. The rest are\nstandbys: they hold no data, answer no request, and exist to take over.\n- More than one REQUIRES HighAvailability. Electing a leader among several needs a leadership\nrecord, and the webhook refuses the pair without one rather than silently running two\nleaders against the same members.\n- Raising this past one TURNS THE ELECTION ON, and the flip is re-evaluated on every\nreconcile rather than decided at create. It restarts the leader and rolls every member —\nthe member's master entry changes shape with it — so the store's cached contents do not\nsurvive the crossing. The same holds on the way back down to one.\n- Raising this adds no capacity, which members do. The ceiling is here to catch the reading\nthat it does, and it is duplicated in the webhook on purpose: this one still holds when\nthe webhook is not installed, which is when a second leader would be rendered rather than\nrefused. Raise both together; widening a maximum is not a breaking change.", + Description: "Replicas is how many leader processes run, of which exactly one serves at a time. The rest are\nstandbys: they hold no data, answer no request, and exist to take over.\n- More than one REQUIRES ElectionBackend to be Kubernetes. The webhook refuses None with several\nreplicas, and the renderer clamps it to one if admission is unavailable.\n- Changing the replica count does not change the election mode. The default Kubernetes mode\nruns the election even at one replica, so scaling it up does not restart the first leader.\n- Raising this adds no capacity, which members do. The ceiling is here to catch the reading\nthat it does, and it is duplicated in the webhook on purpose: this one still holds when\nthe webhook is not installed, which is when a second leader would be rendered rather than\nrefused. Raise both together; widening a maximum is not a breaking change.", Type: "integer", Format: "int32", Default: &v1.JSON{ diff --git a/api/worker/zz_generated.openapi.go b/api/worker/zz_generated.openapi.go index f41918f71..a73246ca7 100644 --- a/api/worker/zz_generated.openapi.go +++ b/api/worker/zz_generated.openapi.go @@ -6460,7 +6460,7 @@ func schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeader(ref common.Referen Properties: map[string]spec.Schema{ "replicas": { SchemaProps: spec.SchemaProps{ - Description: "Replicas is how many leader processes run, of which exactly one serves at a time. The rest are standbys: they hold no data, answer no request, and exist to take over.\n\n - More than one REQUIRES HighAvailability. Electing a leader among several needs a leadership\n record, and the webhook refuses the pair without one rather than silently running two\n leaders against the same members.\n - Raising this past one TURNS THE ELECTION ON, and the flip is re-evaluated on every\n reconcile rather than decided at create. It restarts the leader and rolls every member —\n the member's master entry changes shape with it — so the store's cached contents do not\n survive the crossing. The same holds on the way back down to one.\n - Raising this adds no capacity, which members do. The ceiling is here to catch the reading\n that it does, and it is duplicated in the webhook on purpose: this one still holds when\n the webhook is not installed, which is when a second leader would be rendered rather than\n refused. Raise both together; widening a maximum is not a breaking change.", + Description: "Replicas is how many leader processes run, of which exactly one serves at a time. The rest are standbys: they hold no data, answer no request, and exist to take over.\n\n - More than one REQUIRES ElectionBackend to be Kubernetes. The webhook refuses None with several\n replicas, and the renderer clamps it to one if admission is unavailable.\n - Changing the replica count does not change the election mode. The default Kubernetes mode\n runs the election even at one replica, so scaling it up does not restart the first leader.\n - Raising this adds no capacity, which members do. The ceiling is here to catch the reading\n that it does, and it is duplicated in the webhook on purpose: this one still holds when\n the webhook is not installed, which is when a second leader would be rendered rather than\n refused. Raise both together; widening a maximum is not a breaking change.", Minimum: ptr.To[float64](1), Maximum: ptr.To[float64](5), Type: []string{"integer"}, @@ -6469,10 +6469,17 @@ func schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeader(ref common.Referen }, "highAvailability": { SchemaProps: spec.SchemaProps{ - Description: "HighAvailability elects the leader through a Kubernetes Lease, and it is what allows Replicas above 1. The election itself needs no settings: the Lease is named after this backend, so there is no connection target to supply, and the API access it needs is rendered beside the workload. What the block does carry is how members find the leader it elects.\n\n - Unset, the leader runs as a single process exactly as before — no election flag, no extra\n object, the command line it ran before this field existed.\n - Set with Replicas at 1, the ELECTION is INERT: one process has nothing to elect between,\n so no election flag, Lease or API token is rendered until Replicas rises above 1. That\n makes an empty block safe to set up front on a store image built without the k8s-lease\n backend, whose master fails at startup the moment the election flags appear — those flags\n arrive only when there is something for them to elect. Snapshot is the exception and says\n so on itself.\n - With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile\n interval. Each replica seeds its tenant quota policy at its own start, so a standby that\n took over after a quota was raised applies the older, lower ceiling, and an over-quota\n write in this store is not refused — it evicts that tenant's own older objects,\n irreversibly and without moving any counter. The quota itself is not lost: the pool\n reconciler is the authority and writes the difference back on its next pass.", + Description: "HighAvailability configures how members find the elected leader. Election itself is selected by ElectionBackend, so this block is optional even when Replicas exceeds one.", Ref: ref(v1alpha1.KVCacheBackendLeaderHighAvailability{}.OpenAPIModelName()), }, }, + "electionBackend": { + SchemaProps: spec.SchemaProps{ + Description: "ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica, so scaling up does not change the first leader's startup flags. None is for a single leader whose image cannot use Kubernetes election. More than one replica with None is refused. The enum can add other backends when they are supported.\n\nWith MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile interval. Each replica seeds its tenant quota policy at its own start, so a standby that took over after a quota was raised applies the older, lower ceiling. An over-quota write evicts that tenant's own older objects, irreversibly and without moving any counter. The quota itself is not lost: the pool reconciler writes the difference back on its next pass.", + Type: []string{"string"}, + Format: "", + }, + }, "allocationStrategy": { SchemaProps: spec.SchemaProps{ Description: "AllocationStrategy is how the leader picks which member takes a new write. Random spreads them; FreeRatioFirst biases toward the emptier member.\n\nThe enum is deliberately the two any pooled store would have, not every value the current artifact's flag accepts: the rest are specific to one medium or one locality model, are reachable through ExtraArgs, and would fix this API to one implementation's vocabulary. Widening the enum later is not a breaking change.", @@ -6541,7 +6548,7 @@ func schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeaderHighAvailability(re return common.OpenAPIDefinition{ Schema: spec.Schema{ SchemaProps: spec.SchemaProps{ - Description: "KVCacheBackendLeaderHighAvailability turns leader election on, and carries how members find the leader it elects.\n\nDECLARING THE BLOCK IS THE SWITCH, and there is no key inside it to turn the feature back off. An `enabled: false` beside `replicas: 3` would be a third state that admission would have to adjudicate and every reader would have to remember, while presence has no such state. Lease tuning — duration, renew deadline — can also be added here later without a breaking change.\n\nA standby REPLICATES NOTHING. The store's operation log is the only way to feed one, and it runs on a leadership backend this operator's image cannot carry, so a failover or a restart starts from an empty cache. The store's snapshot is not offered either: restoring one can make the cache serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs.", + Description: "KVCacheBackendLeaderHighAvailability configures how members find the elected leader.\n\nA standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM key index. A member's local disk tier can re-register keys it has fully offloaded after the election. The store's snapshot is not offered: restoring one can make the cache serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs.", Type: []string{"object"}, Properties: map[string]spec.Schema{ "memberAddressing": { diff --git a/docs/kv-cache/backend.md b/docs/kv-cache/backend.md index a55060cfe..64b7a2221 100644 --- a/docs/kv-cache/backend.md +++ b/docs/kv-cache/backend.md @@ -192,11 +192,13 @@ Each variant is built on `0.3.13.post1`, the line vLLM's supported clients are o minimum](../model-deployment/engine-versions.md). SGLang's clients are on the 0.3.12 line, which this project does not build. Which line a backend needs is that table's question. -**A `0.3.10.post2` variant also needs `leader.multiTenancy: false` written out.** The tenant ledger -hangs on the master's `-enable_multi_tenants` switch, which Mooncake took in 0.3.12, and the field -defaults on — so on an older image the default renders a flag the master does not recognize, and it -exits at startup. The explicit false renders no flag, which is the command line such an image has -always run. +**A `0.3.10.post2` variant needs `leader.electionBackend: None` and +`leader.multiTenancy: false` written out.** It cannot run the default Kubernetes election. + +The tenant ledger hangs on the master's `-enable_multi_tenants` switch, which Mooncake took in 0.3.12, +and the field defaults on. On an older image the default renders a flag the master does not recognize, +and it exits at startup. The explicit false renders no flag, which is the command line such an image +has always run. **A VRAM group needs a build with VRAM segments compiled in (`USE_VRAM_SEGMENT=ON`), and the stock `-cpu` default is not one.** VRAM segments exist only on the `0.3.13` line — the `0.3.10.post2` @@ -263,13 +265,13 @@ its own per-version rule — it needs a master from 0.3.12 on — see ## The metadata plane **The metadata plane is peer-to-peer and has no API field.** The member's `metadata_server` renders as -the literal `P2PHANDSHAKE`, unconditionally. A single-leader backend therefore has **zero external -dependencies beyond its image** — no etcd, no Redis, nothing to deploy alongside it. +the literal `P2PHANDSHAKE`, unconditionally. It needs no etcd or Redis. The default leader election +does use a Kubernetes Lease and its API access, even with one leader replica. Two axes get confused here, so both are stated. The metadata plane is how clients find one another. -The **HA backend store** — `-enable_ha` with `-ha_backend_type` — is how leader replicas elect one -among them, and that is where the Kubernetes Lease lives. It is -[`highAvailability`](leader.md#high-availability), and it moves nothing on this plane. +The **HA backend store** — `-enable_ha` with `-ha_backend_type` — is how the leader elects, even at +one replica by default. The Kubernetes Lease is selected by +[`leader.electionBackend`](leader.md#high-availability), and it moves nothing on the metadata plane. ⛔ **A manifest that tries to configure the metadata plane is not refused with a helpful message.** There is no field, so there is nothing for a webhook to see: @@ -570,8 +572,8 @@ Five phases — `Provisioning`, `Ready`, `Degraded`, `Error`, `Deleting`. `Ready Conditions report the axes: `LeaderAvailable`, `MembersMounted`, `CapacityObserved`, `PoolWrites`, `Deletable` and `RolloutComplete`. Two more appear only where they have something to judge — `ElectionObserved` -above one leader replica, and `TierWasEmpty` when a member group carries a -[local disk tier](local-disk-tier.md). +when `leader.electionBackend` is `Kubernetes` (including at one replica), and +`TierWasEmpty` when a member group carries a [local disk tier](local-disk-tier.md). **Those last three do not move the phase, and that is deliberate.** A rollout in flight, an election that has not happened, and a disk tier found holding data are all states in which the backend serves diff --git a/docs/kv-cache/leader.md b/docs/kv-cache/leader.md index 8023fc8dc..a594d4879 100644 --- a/docs/kv-cache/leader.md +++ b/docs/kv-cache/leader.md @@ -57,24 +57,23 @@ same message a member gets. The schema keys the list by `name`, so one name cann `replicas` defaults to `1`, and `5` is the ceiling in the **webhook** and in the schema alike: only one leader ever serves, so further replicas are spare processes rather than capacity. More than one -requires [`highAvailability`](#high-availability) and is refused by the webhook without it, naming -the field that is missing. An enum would answer `Unsupported value: 2` and teach nothing. +requires `electionBackend: Kubernetes`; the webhook refuses `None` at that count. -⛔ **Without `highAvailability` the Deployment runs one replica whatever `replicas` says.** The -webhook refuses that combination, but a schema cannot express a cross-field rule — so where the -webhook is not installed this clamp is what keeps unelected masters off one pool. +⛔ **With `electionBackend: None` the Deployment runs one replica whatever `replicas` says.** The +webhook refuses a larger count, and this clamp also protects clusters without the webhook. -**`highAvailability` with one replica is inert: the election exists only above one replica.** A -single process has nothing to elect between, so no election flag, Lease or API token is rendered -until `replicas` rises past 1 — set the field up front and a later scale-up is a one-field change. +**`electionBackend` defaults to `Kubernetes`, even at one replica.** The first leader campaigns for +a Lease and starts with its election flags and API token. Raising `replicas` from one to three adds +standbys without changing that leader's Pod template or rolling the members. -The gate is re-evaluated on every reconcile, not decided at create: crossing `replicas: 1` in -either direction flips the election on or off, and the flip restarts the leader and rolls every -member, so the store's cached contents do not survive the crossing. +Set `electionBackend: None` only for a single leader whose image cannot run the Kubernetes Lease +backend. Switching between `None` and `Kubernetes` changes the leader and member Pod templates, so +it restarts them and loses their DRAM cache contents. -**Rising past one replica takes two updates.** The first recreates the leader at one replica with the -election on; the standbys follow once no Pod of the old template is left, and `RolloutComplete` reads -`False/ReplicasPending` in between. Falling back to one is a single `Recreate`. +**Rising past one replica normally takes one update.** If a live Deployment still runs an unelected +template, the controller first replaces it with an elected template at one replica, then adds +standbys once that template has rolled out. `RolloutComplete` reads `False/ReplicasPending` while +the second step waits. Falling back to one keeps the election on unless the field is set to `None`. > **Why** — the Deployment controller applies a replica change to the only active ReplicaSet before it > applies any strategy. In the update that also turns the election on, that is the old one, so a single @@ -97,10 +96,10 @@ would take every standby down together with the leader and leave nothing to elec > here, so `replicas-1` makes that `1 > 1`: the old leader is never removed, the new replicas cannot > become ready until it releases the Lease, and the rollout stalls for good. -⛔ **A rollout still has a window with no serving master**, and high availability shortens it rather -than removing it. A floor of zero available replicas is what lets the old leader go, so it can go -before a replacement has taken the Lease. The window is bounded by the lease expiry plus activation, -not by a Pod start — the replacements are already running as standbys, contending for it. +⛔ **A rollout still has a window with no serving master**, and standby replicas can shorten it. +A floor of zero available replicas is what lets the old leader go, so it can go before a replacement +has taken the Lease. The window depends on election and activation; replacements already running as +standbys do not have to wait for scheduling or image pulls. **The two probes deliberately take different paths**, and this is the one configuration detail on this page that must not be "simplified": @@ -122,16 +121,14 @@ The health document has four fields that matter: ⛔ **`status` is a hard-coded constant.** It reads `"ok"` on a leader that is serving nothing. **`service_ready` is the only verdict in the document**, and every readiness decision rests on it. -A single leader reports `service_ready: true` from its first answer, because the non-HA path sets it -unconditionally three lines after the admin server starts. Under high availability it is the standby -marker, and the readiness gate above is what turns it into an endpoint decision. +With `electionBackend: None`, a single leader reports `service_ready: true` from its first answer. +Under the default Kubernetes election, readiness waits for the process to win the Lease. ## High availability -Set `leader.highAvailability` and the leader elects through a **Kubernetes Lease** — once `replicas` -exceeds one; below that the election is inert (see above). The election itself needs no settings — -the Lease carries the leader's own object name, `-leader`, in this operator's namespace — -so an empty block is the switch: +The default `leader.electionBackend: Kubernetes` elects through a **Kubernetes Lease** at every +replica count. The Lease is named `-leader` in the operator's namespace. Scale an already +elected leader by changing only its count: ```yaml spec: @@ -139,9 +136,11 @@ spec: managed: leader: replicas: 3 - highAvailability: {} ``` +`leader.highAvailability.memberAddressing` is optional and only selects how members find the +winner. It does not turn the election on. + ⛔ **A published `kvcacheai/mooncake` image cannot do this, on either side.** Leadership backend availability is a compile-time switch and every option ships **off**: @@ -154,9 +153,9 @@ Use an image built from [`pack/mirrored-mooncake`](../../pack/mirrored-mooncake/ `spec.image` **and for every `members[].image`**, on Mooncake 0.3.12 or later: an electing leader is also rendered `-pod_name` and `-pod_namespace` to label the winner, and a 0.3.11 master exits on both. -A lease-less image is not refused outright: at one replica the election flags are never rendered, so -such an image runs a single-leader backend even with `highAvailability` set — the flags arrive only -when `replicas` rises past 1, which is where the missing backend would fail the leader at startup. +A lease-less image can run one leader with `electionBackend: None`. It cannot serve a backend with +multiple leader replicas. Set `None` when creating a backend with such an image; an omitted field +selects `Kubernetes` and renders election flags that the image cannot use. A member group on `RDMA`, `ROCM` or `CANN` runs under high availability on the build that carries its transport: every `mirrored-mooncake` target — the default build and the `cuda`, `cann` and @@ -236,20 +235,24 @@ Both first failed at 31.41 seconds and converged around 60.6 seconds, so electio result; retest if election timing changes. The Service endpoint transition was inferred from the result, not observed directly. Changing the value rolls every member group when HA is active. -**Several leaders shorten the outage; they do not keep the cache.** A standby holds no data. The +**Standby leaders can shorten the outage; they do not keep the cache.** A standby holds no data. The replica that takes over, like a single leader that restarts, learns the members' segments from their -remounts and none of the keys in them, so every object held in member memory misses until it is -written again. +remounts and none of the keys in them, so a DRAM-only key misses until it is written again. -The exception is what a member's [local disk tier](local-disk-tier.md) already holds: when the new -leader does not know the disk segment, the member registers it again together with the objects on it. +Keys fully written to a member's [local disk tier](local-disk-tier.md) may recover after a disk +segment and its objects are registered again. The directory must still be available and the scan +must complete. A member restart also loses its DRAM bytes; in the tested Mooncake version, an old +disk replica record can block the restarted member's new client ID from registering its disk keys. +Disk recovery is therefore conditional, even when the files survive. What a second replica buys is time. On a single-node test cluster a failover left the store -unusable for about 16 seconds and a single-leader restart for about 30; on a real cluster a single -leader also waits for its replacement to be scheduled and its image pulled. +unusable for about 16 seconds and a single-leader restart for about 30. Those measured service +gaps are not a bound on when each disk key first becomes a hit; a disk failover test eventually +returned 1024/1024 keys but did not time their first hits. DRAM keys continue to miss until +rewritten. On a real cluster a single leader also waits for scheduling and image pulls. -**Run one leader by default.** Add replicas when that gap costs more than what an election needs: the -store image and the two accounts described above. +**Run one leader by default.** Add replicas when the shorter service gap justifies the standby +processes. The default single leader already uses the Lease and the two accounts described above. ⛔ **The store's snapshot is not offered, and its flags are refused in `leader.extraArgs`.** A snapshot records where each key sits in member memory, and restoring one does not check that the diff --git a/docs/kv-cache/walkthrough.md b/docs/kv-cache/walkthrough.md index 1905a4461..a520c3d37 100644 --- a/docs/kv-cache/walkthrough.md +++ b/docs/kv-cache/walkthrough.md @@ -180,8 +180,8 @@ $ kubectl -n team-a get md qwen-chat -o jsonpath='{.status.conditions[?(@.type== ## Step 4: high availability -Everything so far runs one leader process. An update or a node failure takes the store's metadata with -it, and every member re-registers into an empty one. Electing between several replicas is one edit: +Everything so far runs one leader process that already holds a Kubernetes Lease. An update or a node +failure takes the store's metadata with it. Add standbys with one edit: ```yaml spec: @@ -189,7 +189,6 @@ spec: managed: leader: replicas: 3 - highAvailability: {} ``` ⛔ **The healthy steady state now reads `3 desired / 1 ready`, and that is not a broken Deployment.** @@ -197,16 +196,16 @@ Exactly one leader serves; the other two are standbys, deliberately not ready so endpoints never include a process that cannot serve. During a healthy failover `2` are briefly ready as the old leader steps down. Both readings are normal. -⛔ **Crossing `replicas: 1` in either direction restarts the leader and rolls every member**, because -the HA accounts and token mounts change. With an explicit `Lease` address, the member's master entry -also changes shape. The store's cached contents do not survive the crossing, so make this edit before -the cache is worth keeping — or accept a cold start. +**Scaling from one to three keeps the first leader Pod and the member templates** because the Lease +election and accounts were already present at one replica. See +[the leader Deployment](leader.md#the-deployment-and-the-two-probes) for the +cost of changing `leader.electionBackend`. **`leader.highAvailability.memberAddressing` chooses how members find the master**, and defaults to `Service`. An explicit `Lease` value uses the member's API access to read the current holder. See [High availability](leader.md#high-availability) for the measured failover limits. -Two conditions appear at this point that the phase deliberately does not summarize: +These conditions apply at one replica too; the phase does not summarize them: | Condition | True means | False means | |---|---|---| diff --git a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go index 2a4b633ea..057c21e9a 100644 --- a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go +++ b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go @@ -11,38 +11,29 @@ type KVCacheBackendLeaderApplyConfiguration struct { // Replicas is how many leader processes run, of which exactly one serves at a time. The rest are // standbys: they hold no data, answer no request, and exist to take over. // - // - More than one REQUIRES HighAvailability. Electing a leader among several needs a leadership - // record, and the webhook refuses the pair without one rather than silently running two - // leaders against the same members. - // - Raising this past one TURNS THE ELECTION ON, and the flip is re-evaluated on every - // reconcile rather than decided at create. It restarts the leader and rolls every member — - // the member's master entry changes shape with it — so the store's cached contents do not - // survive the crossing. The same holds on the way back down to one. + // - More than one REQUIRES ElectionBackend to be Kubernetes. The webhook refuses None with several + // replicas, and the renderer clamps it to one if admission is unavailable. + // - Changing the replica count does not change the election mode. The default Kubernetes mode + // runs the election even at one replica, so scaling it up does not restart the first leader. // - Raising this adds no capacity, which members do. The ceiling is here to catch the reading // that it does, and it is duplicated in the webhook on purpose: this one still holds when // the webhook is not installed, which is when a second leader would be rendered rather than // refused. Raise both together; widening a maximum is not a breaking change. Replicas *int32 `json:"replicas,omitempty"` - // HighAvailability elects the leader through a Kubernetes Lease, and it is what allows Replicas - // above 1. The election itself needs no settings: the Lease is named after this backend, so - // there is no connection target to supply, and the API access it needs is rendered beside the - // workload. What the block does carry is how members find the leader it elects. - // - // - Unset, the leader runs as a single process exactly as before — no election flag, no extra - // object, the command line it ran before this field existed. - // - Set with Replicas at 1, the ELECTION is INERT: one process has nothing to elect between, - // so no election flag, Lease or API token is rendered until Replicas rises above 1. That - // makes an empty block safe to set up front on a store image built without the k8s-lease - // backend, whose master fails at startup the moment the election flags appear — those flags - // arrive only when there is something for them to elect. Snapshot is the exception and says - // so on itself. - // - With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile - // interval. Each replica seeds its tenant quota policy at its own start, so a standby that - // took over after a quota was raised applies the older, lower ceiling, and an over-quota - // write in this store is not refused — it evicts that tenant's own older objects, - // irreversibly and without moving any counter. The quota itself is not lost: the pool - // reconciler is the authority and writes the difference back on its next pass. + // HighAvailability configures how members find the elected leader. Election itself is selected + // by ElectionBackend, so this block is optional even when Replicas exceeds one. HighAvailability *KVCacheBackendLeaderHighAvailabilityApplyConfiguration `json:"highAvailability,omitempty"` + // ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica, + // so scaling up does not change the first leader's startup flags. None is for a single leader + // whose image cannot use Kubernetes election. More than one replica with None is refused. + // The enum can add other backends when they are supported. + // + // With MultiTenancy on, a failover costs HIT RATE for up to one KVCachePool reconcile interval. + // Each replica seeds its tenant quota policy at its own start, so a standby that took over after + // a quota was raised applies the older, lower ceiling. An over-quota write evicts that tenant's + // own older objects, irreversibly and without moving any counter. The quota itself is not lost: + // the pool reconciler writes the difference back on its next pass. + ElectionBackend *string `json:"electionBackend,omitempty"` // AllocationStrategy is how the leader picks which member takes a new write. Random spreads // them; FreeRatioFirst biases toward the emptier member. // @@ -122,6 +113,14 @@ func (b *KVCacheBackendLeaderApplyConfiguration) WithHighAvailability(value *KVC return b } +// WithElectionBackend sets the ElectionBackend field in the declarative configuration to the given value +// and returns the receiver, so that objects can be built by chaining "With" function invocations. +// If called multiple times, the ElectionBackend field is set to the value of the last call. +func (b *KVCacheBackendLeaderApplyConfiguration) WithElectionBackend(value string) *KVCacheBackendLeaderApplyConfiguration { + b.ElectionBackend = &value + return b +} + // WithAllocationStrategy sets the AllocationStrategy field in the declarative configuration to the given value // and returns the receiver, so that objects can be built by chaining "With" function invocations. // If called multiple times, the AllocationStrategy field is set to the value of the last call. diff --git a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go index 5a13f89e1..dc52448a1 100644 --- a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go +++ b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go @@ -5,17 +5,11 @@ package v1alpha1 // KVCacheBackendLeaderHighAvailabilityApplyConfiguration represents a declarative configuration of the KVCacheBackendLeaderHighAvailability type for use // with apply. // -// KVCacheBackendLeaderHighAvailability turns leader election on, and carries how members find the -// leader it elects. +// KVCacheBackendLeaderHighAvailability configures how members find the elected leader. // -// DECLARING THE BLOCK IS THE SWITCH, and there is no key inside it to turn the feature back off. -// An `enabled: false` beside `replicas: 3` would be a third state that admission would have to -// adjudicate and every reader would have to remember, while presence has no such state. Lease -// tuning — duration, renew deadline — can also be added here later without a breaking change. -// -// A standby REPLICATES NOTHING. The store's operation log is the only way to feed one, and it runs -// on a leadership backend this operator's image cannot carry, so a failover or a restart starts -// from an empty cache. The store's snapshot is not offered either: restoring one can make the cache +// A standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM +// key index. A member's local disk tier can re-register keys it has fully offloaded after the +// election. The store's snapshot is not offered: restoring one can make the cache // serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs. type KVCacheBackendLeaderHighAvailabilityApplyConfiguration struct { // MemberAddressing selects how a member is told to find the master once an election runs. Both diff --git a/pkg/worker/controllers/worker/kv_cache_backend.go b/pkg/worker/controllers/worker/kv_cache_backend.go index f5471edf8..b78c30f72 100644 --- a/pkg/worker/controllers/worker/kv_cache_backend.go +++ b/pkg/worker/controllers/worker/kv_cache_backend.go @@ -1661,7 +1661,7 @@ func (r *KVCacheBackendReconciler) ensureHARBAC( mooncake.MemberRBACObjectName(kvcb), mooncake.RenderMemberRBAC(kvcb)) } -// pruneHARBAC removes the access again when high availability is turned off, and runs AFTER every +// pruneHARBAC removes the access again when Kubernetes election is turned off, and runs AFTER every // workload has been re-rendered without it. // // REQUIRED: the ordering is the reverse of ensureHARBAC's, and it is not symmetry for its own sake. @@ -1671,7 +1671,7 @@ func (r *KVCacheBackendReconciler) ensureHARBAC( // a grant that outlives its use by one reconcile. // // It is also why the removal exists at all: the owner reference collects these when the BACKEND is -// deleted, but a backend that merely drops `highAvailability` is not deleted, so without this it +// deleted, but a backend that changes `electionBackend` to None is not deleted, so without this it // keeps an account that can still take a Lease. func (r *KVCacheBackendReconciler) pruneHARBAC( ctx context.Context, kvcb *workercore.KVCacheBackend, @@ -1872,8 +1872,8 @@ func resolveKVCacheBackendImage(ctx context.Context, kvcb *workercore.KVCacheBac // counts still describe the rollout BEFORE it -- one replica, updated and total, on the unelected // template -- and reading them as this rollout's is exactly the write this gate exists to hold back. // -// Lowering to one never waits. The scaling event there removes elected replicas only, and Recreate -// then stops the last of them before the unelected master starts. +// Lowering to one never waits. The scaling event there removes excess replicas; if election is +// disabled too, Recreate stops the last elected process before the unelected one starts. func leaderCountRiseMustWait(aDeploy, rendered *apps.Deployment) bool { replicas := ptr.Deref(aDeploy.Spec.Replicas, 1) if ptr.Deref(rendered.Spec.Replicas, 1) <= 1 || replicas > 1 { @@ -1902,8 +1902,8 @@ func alignLeaderDeploymentFn( return func(aDeploy *apps.Deployment) (*apps.Deployment, bool, error) { skip := true - // Raising a live Deployment past one replica is two writes, and the first holds the count, - // strategy and deadline at one replica while the template already elects -- see + // Raising an unelected live Deployment past one replica takes two writes. The first holds + // the count, strategy and deadline at one replica while the template already elects -- see // LeaderDeploymentAtOneReplica for why one write would run unelected masters beside the new // one. Decided before anything below changes aDeploy, because the gate reads the live object. eDeploy := rendered diff --git a/pkg/worker/controllers/worker/kv_cache_backend_handover.go b/pkg/worker/controllers/worker/kv_cache_backend_handover.go index b350b0213..a6a2be9c7 100644 --- a/pkg/worker/controllers/worker/kv_cache_backend_handover.go +++ b/pkg/worker/controllers/worker/kv_cache_backend_handover.go @@ -88,8 +88,7 @@ func (h *leaseHolders) forget(backend string) { // that took it. // // IT READS ONLY WHEN A REPORTER WILL LOOK. Each of them returns early on a backend that elects -// nothing, and the condition here is the union of those two: a backend with no high availability and -// at most one leader replica pays for no read, because neither reporter would reach the answer. +// nothing. A backend with electionBackend None pays for no read. // // A NotFound is returned as the error rather than flattened to a nil lease, because the two readers // treat it differently: one forgets its baseline, the other carries on to ask whether a leader is @@ -98,8 +97,7 @@ func (r *KVCacheBackendReconciler) leaderLease( ctx context.Context, kvcb *workercore.KVCacheBackend, ) (*coordination.Lease, error) { managed := kvcb.Spec.Connection.Managed - if managed == nil || - (managed.Leader.HighAvailability == nil && mooncake.LeaderReplicas(managed.Leader) <= 1) { + if managed == nil || !managed.Leader.KubernetesElectionEnabled() { return nil, nil } @@ -130,13 +128,12 @@ func (r *KVCacheBackendReconciler) reportLeaderHandover( kvcb *workercore.KVCacheBackend, lease *coordination.Lease, leaseErr error, ) { managed := kvcb.Spec.Connection.Managed - if managed == nil || managed.Leader.HighAvailability == nil { + if managed == nil || !managed.Leader.KubernetesElectionEnabled() { return } if leaseErr != nil || lease == nil { - // A missing lease is the ordinary state of a backend whose election has not started -- one - // replica, or a leader still coming up. Forgetting rather than keeping the last holder is + // A missing lease is the ordinary state of a leader still coming up. Forgetting the last holder is // what stops the first campaign after a gap from reading as a handover. if leaseErr == nil || kerrors.IsNotFound(leaseErr) { r.leaseHolders.forget(kvcb.Name) @@ -180,11 +177,9 @@ func (r *KVCacheBackendReconciler) reportElectionObserved( lease *coordination.Lease, leaseErr error, ) { managed := kvcb.Spec.Connection.Managed - if managed == nil || mooncake.LeaderReplicas(managed.Leader) <= 1 { - // Below two replicas there is nothing to elect between, so there is no lease to be the - // artifact of anything. Dropped rather than left, because the status this pass builds starts - // as a copy of the observed one: a backend scaled back to one leader would otherwise go on - // publishing the last verdict, with a transition time that makes it look current. + if managed == nil || !managed.Leader.KubernetesElectionEnabled() { + // With election disabled there is no Lease to observe. Drop the previous verdict because + // status starts as a copy of the last observation, which would otherwise look current. holder.Status.Conditions = slices.DeleteFunc(holder.Status.Conditions, func(c gpustack.Condition) bool { return c.Type == string(KVCacheBackendConditionElectionObserved) @@ -238,7 +233,7 @@ func (r *KVCacheBackendReconciler) enqueueKVCacheBackendWhenLeaseChanged( return nil } if kvcb.Spec.Connection.Managed == nil || - kvcb.Spec.Connection.Managed.Leader.HighAvailability == nil { + !kvcb.Spec.Connection.Managed.Leader.KubernetesElectionEnabled() { return nil } diff --git a/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go b/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go index f1913d290..d794da130 100644 --- a/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go +++ b/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go @@ -51,6 +51,7 @@ func electingBackend(name string) *workercore.KVCacheBackend { Managed: &workercore.KVCacheBackendManaged{ Leader: workercore.KVCacheBackendLeader{ Replicas: ptr.To[int32](3), + ElectionBackend: "Kubernetes", HighAvailability: &workercore.KVCacheBackendLeaderHighAvailability{}, }, }, @@ -240,10 +241,10 @@ func TestLeaderHandoverIgnoresBackendsWithNoElection(t *testing.T) { kvcb *workercore.KVCacheBackend }{ { - name: "no high availability", + name: "no election", kvcb: func() *workercore.KVCacheBackend { k := electingBackend("store") - k.Spec.Connection.Managed.Leader.HighAvailability = nil + k.Spec.Connection.Managed.Leader.ElectionBackend = "None" return k }(), }, @@ -409,10 +410,11 @@ func TestElectionObservedDiscriminates(t *testing.T) { wantReason: "NoHolder", }, { - name: "one replica, which elects nothing by design", + name: "one replica still elects", objects: []ctrlcli.Object{readyLeaderDeployment("store")}, replicas: 1, - wantAbsent: true, + wantStatus: meta.ConditionFalse, + wantReason: "NoHolder", }, } { t.Run(tc.name, func(t *testing.T) { @@ -465,14 +467,14 @@ func TestElectionObservedNamesNoCause(t *testing.T) { } // TestElectionObservedIsDroppedWhenTheElectionGoesAway pins the removal half, which the status -// carrying forward makes necessary: a backend scaled back to one replica would otherwise go on +// carrying forward makes necessary: a backend switched to electionBackend None would otherwise go on // publishing a verdict about a lease nothing takes. func TestElectionObservedIsDroppedWhenTheElectionGoesAway(t *testing.T) { r := &KVCacheBackendReconciler{ Client: ctrlfake.NewClientBuilder().WithScheme(scheme.Scheme).Build(), } kvcb := electingBackend("store") - kvcb.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) + kvcb.Spec.Connection.Managed.Leader.ElectionBackend = "None" holder := kvcb.DeepCopy() holder.Status.Conditions = append(holder.Status.Conditions, gpustack.Condition{ diff --git a/pkg/worker/controllers/worker/kv_cache_backend_rollout.go b/pkg/worker/controllers/worker/kv_cache_backend_rollout.go index 0501d0c16..cf244bf87 100644 --- a/pkg/worker/controllers/worker/kv_cache_backend_rollout.go +++ b/pkg/worker/controllers/worker/kv_cache_backend_rollout.go @@ -76,7 +76,7 @@ func (r *KVCacheBackendReconciler) reportRolloutComplete( "counts still describe the update before it", deploy.Generation, deployName)) case ptr.Deref(deploy.Spec.Replicas, 1) != mooncake.LeaderReplicas(kvcb.Spec.Connection.Managed.Leader): - // Raising the count past one is two writes, and between them the Deployment runs one + // Raising an unelected leader past one is two writes, and between them the Deployment runs one // replica of the elected template while the backend asks for more. Every count below agrees // with the Deployment's own spec then, so without this the pause between the writes would // read as a finished rollout. diff --git a/pkg/worker/controllers/worker/kv_cache_backend_test.go b/pkg/worker/controllers/worker/kv_cache_backend_test.go index 2e258afad..296ae992c 100644 --- a/pkg/worker/controllers/worker/kv_cache_backend_test.go +++ b/pkg/worker/controllers/worker/kv_cache_backend_test.go @@ -119,7 +119,7 @@ func newKVCacheBackendObject(usedBy ...workercore.KVCacheObjectReference) *worke Image: "example.com/mooncake:v0", Connection: workercore.KVCacheBackendConnection{ Managed: &workercore.KVCacheBackendManaged{ - Leader: workercore.KVCacheBackendLeader{Replicas: ptr.To[int32](1)}, + Leader: workercore.KVCacheBackendLeader{Replicas: ptr.To[int32](1), ElectionBackend: "None"}, Members: []workercore.KVCacheBackendMember{{ NodeSelector: map[string]string{"kvcache-dram": "true"}, Medium: "DRAM", @@ -1300,10 +1300,11 @@ func TestKVCacheBackendReconciler_ConvergesAHighAvailabilitySwitch(t *testing.T) got := new(workercore.KVCacheBackend) require.NoError(t, cli.Get(ctx, ctrlcli.ObjectKey{Name: kvcb.Name}, got)) if on { + got.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" got.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} - // The election exists only above one replica, so asking for it means standbys. got.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) } else { + got.Spec.Connection.Managed.Leader.ElectionBackend = "None" got.Spec.Connection.Managed.Leader.HighAvailability = nil got.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) } @@ -1440,12 +1441,10 @@ func backfillDeprecatedServiceAccount(obj ctrlcli.Object) { // either one alone empties the name. func TestKVCacheBackendReconciler_LeavingHighAvailabilityReleasesTheAccount(t *testing.T) { for name, leave := range map[string]func(*workercore.KVCacheBackendLeader){ - "ScaledToOneReplica": func(l *workercore.KVCacheBackendLeader) { + "ElectionDisabled": func(l *workercore.KVCacheBackendLeader) { + l.ElectionBackend = "None" l.Replicas = ptr.To[int32](1) }, - "HighAvailabilityRemoved": func(l *workercore.KVCacheBackendLeader) { - l.HighAvailability = nil - }, } { t.Run(name, func(t *testing.T) { kvcb := newKVCacheBackendObject() @@ -1523,8 +1522,10 @@ func TestKVCacheBackendReconciler_ConvergesTheRolloutShapeOnALiveDeployment(t *t require.NoError(t, cli.Get(ctx, ctrlcli.ObjectKey{Name: kvcb.Name}, got)) got.Spec.Connection.Managed.Leader.Replicas = ptr.To(n) if n > 1 { + got.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" got.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} } else { + got.Spec.Connection.Managed.Leader.ElectionBackend = "None" got.Spec.Connection.Managed.Leader.HighAvailability = nil } require.NoError(t, cli.Update(ctx, got)) @@ -1577,15 +1578,14 @@ func setLeaderRollout(t *testing.T, cli ctrlcli.Client, kvcb *workercore.KVCache } // TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters pins how an EXISTING leader -// Deployment crosses one replica, which is where the election turns on or off. +// Deployment crosses one replica, with and without a concurrent election mode change. // // The Deployment controller handles a replica change as a scaling event BEFORE it looks at the // strategy, and scales the only active ReplicaSet to the new count -- which, in the update that also // changes the template, is the OLD one. So one write carrying both the election and three replicas // starts two more masters that do not elect, whatever the strategy says. Rising is therefore two -// writes: the elected template at one replica under Recreate, then the count once the live -// Deployment reports that template rolled out. Falling needs no such split, because the scaling -// event only removes elected replicas before Recreate replaces the last one. +// writes for an unelected live template: elect at one replica, then raise the count after that +// template rolls out. An already elected single leader can scale without a Pod template change. func TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters(t *testing.T) { ctx := context.Background() @@ -1613,6 +1613,7 @@ func TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters(t *testi scaleTo := func(n int32) func(*workercore.KVCacheBackend) { return func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To(n) + k.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" k.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} } } @@ -1678,10 +1679,29 @@ func TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters(t *testi assertRendered(t, live(t, cli, kvcb), kvcb) }) - t.Run("falling to one is one write", func(t *testing.T) { + t.Run("an elected single leader scales without changing its pod template", func(t *testing.T) { + kvcb := newKVCacheBackendObject() + kvcb.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" + cli := newKVCacheBackendClient(kvcb) + require.NotNil(t, reconcileKVCacheBackend(t, cli, kvcb.Name)) + before := live(t, cli, kvcb) + require.True(t, elects(before)) + setLeaderRollout(t, cli, kvcb, true) + want := edit(t, cli, kvcb, func(k *workercore.KVCacheBackend) { + k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) + }) + after := live(t, cli, kvcb) + assertRendered(t, after, want) + assert.Equal(t, ptr.To[int32](3), after.Spec.Replicas) + assert.Equal(t, before.Spec.Template, after.Spec.Template, + "scaling an already elected leader must not roll its first pod") + }) + + t.Run("falling to one and disabling election is one write", func(t *testing.T) { cli, kvcb := newElecting(t, 3) want := edit(t, cli, kvcb, func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) + k.Spec.Connection.Managed.Leader.ElectionBackend = "None" }) deploy := live(t, cli, kvcb) assert.False(t, elects(deploy)) @@ -1689,10 +1709,11 @@ func TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters(t *testi assertRendered(t, deploy, want) }) - t.Run("dropping highAvailability above one replica is one write", func(t *testing.T) { + t.Run("disabling election above one replica clamps to one", func(t *testing.T) { cli, kvcb := newElecting(t, 3) want := edit(t, cli, kvcb, func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) + k.Spec.Connection.Managed.Leader.ElectionBackend = "None" k.Spec.Connection.Managed.Leader.HighAvailability = nil }) deploy := live(t, cli, kvcb) @@ -1744,8 +1765,7 @@ func turnOnHighAvailability(t *testing.T, cli ctrlcli.Client, name string) error got := new(workercore.KVCacheBackend) require.NoError(t, cli.Get(ctx, ctrlcli.ObjectKey{Name: name}, got)) got.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} - // The field alone renders no election: one replica has nothing to elect between, so an election - // takes standbys. + got.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" got.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) require.NoError(t, cli.Update(ctx, got)) @@ -1941,7 +1961,7 @@ func TestKVCacheBackendReconciler_KeepsTheGrantWhenTheWorkloadUpdateFails(t *tes refuseLeaderUpdate = true got := new(workercore.KVCacheBackend) require.NoError(t, cli.Get(ctx, ctrlcli.ObjectKey{Name: kvcb.Name}, got)) - got.Spec.Connection.Managed.Leader.HighAvailability = nil + got.Spec.Connection.Managed.Leader.ElectionBackend = "None" require.NoError(t, cli.Update(ctx, got)) r := &KVCacheBackendReconciler{ diff --git a/pkg/worker/kvcache/mooncake/ha_rbac.go b/pkg/worker/kvcache/mooncake/ha_rbac.go index fb00f595b..05a2d4255 100644 --- a/pkg/worker/kvcache/mooncake/ha_rbac.go +++ b/pkg/worker/kvcache/mooncake/ha_rbac.go @@ -20,31 +20,14 @@ func MemberRBACObjectName(kvcb *workercore.KVCacheBackend) string { return kvcb.Name + "-member" } -// leaderNeedsAPIAccess reports whether an election runs. The leader then needs API access, and the -// member account stays available for an explicit Lease address. -// -// The election exists only ABOVE ONE REPLICA. A single process has nothing to elect between, so -// highAvailability with one replica renders no election flag, no Lease and no API token -- and the -// leader runs exactly the command line it would run with the field unset. That reading is also the -// only one a store image built without the k8s-lease backend survives: the backend's absence fails -// the master at startup the moment the election flags appear, so rendering them for one replica -// would buy nothing and cost such an image the whole backend. -// -// The pairing is re-evaluated on every render, not decided at create: raising replicas past one -// turns the election on, and lowering back to one turns it off. Both flips restart the leader and -// roll every member because their token and account settings change. An explicit Lease address also -// changes shape with the election -- see MemberMasterEntry. +// leaderNeedsAPIAccess reports whether Kubernetes election runs. It is independent of the replica +// count: one replica also campaigns, so scaling it up does not change the first leader's template. // // Without an election neither role has a reason to hold a token. Under HA, the leader elects through // a Lease and an explicitly Lease-addressed member reads it. The member account is rendered for both // address values. The predicate is named because five renderings ask it. func leaderNeedsAPIAccess(leader workercore.KVCacheBackendLeader) bool { - if leader.HighAvailability == nil { - return false - } - // Nil is the schema's default of one, which is below the election's floor the same as an - // explicit 1. - return leader.Replicas != nil && *leader.Replicas > 1 + return leader.KubernetesElectionEnabled() } // leaderServiceAccountName is the account the leader Pod runs as, and it is EMPTY without HA. diff --git a/pkg/worker/kvcache/mooncake/ha_rbac_test.go b/pkg/worker/kvcache/mooncake/ha_rbac_test.go index 1e9b84e4d..d1a6fb965 100644 --- a/pkg/worker/kvcache/mooncake/ha_rbac_test.go +++ b/pkg/worker/kvcache/mooncake/ha_rbac_test.go @@ -12,11 +12,11 @@ import ( "gpustack.ai/gpustack/pkg/worker/kuberess" ) -// haBackend is the shared fixture with the election turned on, which takes standbys: one replica -// has nothing to elect between, so the field alone renders no election. +// haBackend is the shared fixture with Kubernetes election and three replicas. func haBackend(mutate ...func(*workercore.KVCacheBackend)) *workercore.KVCacheBackend { all := append([]func(*workercore.KVCacheBackend){ func(kvcb *workercore.KVCacheBackend) { + kvcb.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" kvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} kvcb.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) }, @@ -24,27 +24,46 @@ func haBackend(mutate ...func(*workercore.KVCacheBackend)) *workercore.KVCacheBa return testBackend(all...) } -// TestTheElectionExistsOnlyAboveOneReplica pins the gate itself, and across the WHOLE surface the -// predicate feeds rather than only the RBAC half: a single-replica backend that sets -// highAvailability must render exactly the non-election shape everywhere, because each rendering -// that disagreed would strand one side -- a member following a Lease no leader takes, or a leader -// campaigning for one on an image that has no lease backend. -// -// The boundary is asserted at two rather than at one: the flip ON is the behavior a scale-up -// depends on, and a gate that never opened would pass every assertion about staying shut. -func TestTheElectionExistsOnlyAboveOneReplica(t *testing.T) { +func TestLeaderElectionChoice(t *testing.T) { + cases := []struct { + name string + backend string + replicas int32 + want bool + count int32 + }{ + {name: "default at one replica", replicas: 1, want: true, count: 1}, + {name: "explicit Kubernetes at one replica", backend: "Kubernetes", replicas: 1, want: true, count: 1}, + {name: "None at one replica", backend: "None", replicas: 1, count: 1}, + {name: "default at three replicas", replicas: 3, want: true, count: 3}, + {name: "None is clamped without admission", backend: "None", replicas: 3, count: 1}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + kvcb := testBackend(func(k *workercore.KVCacheBackend) { + k.Spec.Connection.Managed.Leader.ElectionBackend = tc.backend + k.Spec.Connection.Managed.Leader.Replicas = ptr.To(tc.replicas) + }) + deploy := RenderLeaderDeployment(kvcb, "mooncake:v0.3.13") + assert.Equal(t, tc.count, *deploy.Spec.Replicas) + assert.Equal(t, tc.want, LeaderTemplateElects(deploy.Spec.Template)) + assert.Equal(t, tc.want, RenderLeaderRBAC(kvcb).Wanted()) + assert.Equal(t, tc.want, RenderMemberRBAC(kvcb).Wanted()) + }) + } +} + +// TestTheElectionPersistsAtOneReplica pins the startup shape needed for a later scale-up. +func TestTheElectionPersistsAtOneReplica(t *testing.T) { one := testBackend(func(kvcb *workercore.KVCacheBackend) { - kvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + kvcb.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" }) - for _, flag := range RenderLeaderFlags(one) { - assert.NotContains(t, flag, "enable_ha", - "one process has nothing to elect between, so the field stays inert") - } - assert.False(t, RenderLeaderRBAC(one).Wanted()) - assert.False(t, RenderMemberRBAC(one).Wanted()) - assert.Empty(t, leaderServiceAccountName(one)) - assert.Empty(t, memberServiceAccountName(one)) + assert.Contains(t, RenderLeaderFlags(one), "-enable_ha=true") + assert.True(t, RenderLeaderRBAC(one).Wanted()) + assert.True(t, RenderMemberRBAC(one).Wanted()) + assert.NotEmpty(t, leaderServiceAccountName(one)) + assert.NotEmpty(t, memberServiceAccountName(one)) assert.Equal(t, LeaderServiceHost(one)+":50051", MemberMasterEntry(one), "and the member is pointed at the Service, not at a Lease nobody takes") @@ -52,12 +71,13 @@ func TestTheElectionExistsOnlyAboveOneReplica(t *testing.T) { require.NotNil(t, deploy.Spec.Replicas) assert.Equal(t, int32(1), *deploy.Spec.Replicas) for _, e := range deploy.Spec.Template.Spec.Containers[0].Env { - assert.NotEqual(t, LeaderPodIPEnv, e.Name, - "the advertised address is read only by an election, so one replica does not define it") + if e.Name == LeaderPodIPEnv { + assert.NotEmpty(t, e.ValueFrom) + } } two := testBackend(func(kvcb *workercore.KVCacheBackend) { - kvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + kvcb.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" kvcb.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](2) }) assert.Contains(t, RenderLeaderFlags(two), "-enable_ha=true", @@ -66,7 +86,7 @@ func TestTheElectionExistsOnlyAboveOneReplica(t *testing.T) { assert.Equal(t, LeaderServiceHost(two)+":50051", MemberMasterEntry(two)) // The reader the aligner judges a live template with agrees with the renderer both ways. - assert.False(t, LeaderTemplateElects(deploy.Spec.Template)) + assert.True(t, LeaderTemplateElects(deploy.Spec.Template)) assert.True(t, LeaderTemplateElects(RenderLeaderDeployment(two, "mooncake:v0.3.13").Spec.Template)) } @@ -78,7 +98,7 @@ func TestTheElectionExistsOnlyAboveOneReplica(t *testing.T) { // granted. func TestRenderLeaderRBAC_FollowsTheField(t *testing.T) { assert.False(t, RenderLeaderRBAC(testBackend()).Wanted(), - "a backend without highAvailability renders no API access") + "a backend with electionBackend None renders no API access") got := RenderLeaderRBAC(haBackend()) require.True(t, got.Wanted()) @@ -96,7 +116,7 @@ func TestRenderLeaderRBAC_FollowsTheField(t *testing.T) { } assert.Equal(t, "mooncake-dram-leader", leaderServiceAccountName(haBackend())) assert.Empty(t, leaderServiceAccountName(testBackend()), - "without HA the PodSpec names no account, which is what it did before this existed") + "without election the PodSpec names no account") for _, ns := range []string{ got.ServiceAccount.Namespace, got.Role.Namespace, got.RoleBinding.Namespace, diff --git a/pkg/worker/kvcache/mooncake/keys.go b/pkg/worker/kvcache/mooncake/keys.go index de68042e4..202653826 100644 --- a/pkg/worker/kvcache/mooncake/keys.go +++ b/pkg/worker/kvcache/mooncake/keys.go @@ -82,7 +82,7 @@ var LeaderExtraArgsRules = ExtraArgsRules{ "allocation_strategy", // The four the election renders, reserved as one group because that is how they are // rendered: together or not at all. Reserved UNCONDITIONALLY, like the offload pair below, - // even though nothing renders them for a backend without leader.highAvailability -- a + // even though nothing renders them with electionBackend None -- a // passthrough enable_ha with no connection string beside it is a leader that exits at // startup, and cluster_id reached this way would key the store's namespace off something // the object does not say. @@ -372,10 +372,9 @@ var MemberDerivedEnvs = []string{ // It is a PLAIN LIST for the same reason MemberDerivedEnvs is: the exclusive and forbidden kinds // would both be empty here, and nothing in the leader's namespace voids another setting. // -// None is rendered unconditionally: all three only under high availability. They are reserved -// UNCONDITIONALLY for the same reason the election flags are: an object must be creatable with the -// variable already in place and the field turned on afterwards, and a passthrough value would -// silently win over the reference the rendered argv arrives with. +// All three render with Kubernetes election, including at one replica. They are reserved +// UNCONDITIONALLY because an existing passthrough could otherwise override the Pod identity when +// electionBackend changes from None to Kubernetes. var LeaderDerivedEnvs = []string{ LeaderPodIPEnv, LeaderPodNameEnv, diff --git a/pkg/worker/kvcache/mooncake/leader_flags.go b/pkg/worker/kvcache/mooncake/leader_flags.go index fae7c7094..117045bea 100644 --- a/pkg/worker/kvcache/mooncake/leader_flags.go +++ b/pkg/worker/kvcache/mooncake/leader_flags.go @@ -19,7 +19,7 @@ const ( LeaderMetricsPort = 9003 // LeaderPodNameEnv and LeaderPodNamespaceEnv are the environment variables the rendered argv - // refers to under high availability. The workload that runs this argv has to define both from + // refers to under Kubernetes election. The workload that runs this argv has to define both from // the downward API, which is what makes the reference resolve. // // They carry this repository's own names rather than the bare POD_NAME / POD_NAMESPACE the @@ -28,7 +28,7 @@ const ( // component here already spells them this way. LeaderPodNameEnv = "KUBERNETES_POD_NAME" LeaderPodNamespaceEnv = "KUBERNETES_POD_NAMESPACE" - // LeaderPodIPEnv is the third, and like the other two it is defined only under high availability + // LeaderPodIPEnv is the third, and like the other two it is defined only under Kubernetes election // because only the election reads it. See the -rpc_address flag for what it decides. LeaderPodIPEnv = "KUBERNETES_POD_IP" @@ -104,15 +104,12 @@ func RenderLeaderFlags(kvcb *workercore.KVCacheBackend) []string { // no inspection of extraArgs needed here. flags = append(flags, "-default_kv_lease_ttl="+LeaderKVLeaseTTL) - // The election, rendered as one group or not at all, and only when one runs: highAvailability - // with a single replica has nothing to elect -- see leaderNeedsAPIAccess. Splitting the group - // is what a partial render would - // do, and each half alone is a specific failure: -enable_ha without a connection string exits at + // The election is rendered as one group whenever electionBackend is Kubernetes, including at + // one replica. Splitting the group is a partial render: -enable_ha without a connection string exits at // startup, and a connection string without -enable_ha is accepted and ignored with a warning. // - // -ha_backend_type is not a field. The image carries two leadership backends -- the Lease and - // Redis -- and only the Lease exists inside Kubernetes, so this operator has one value to render - // and a single-value enum in an API is a name, not a choice. + // ElectionBackend currently accepts Kubernetes or None. The Kubernetes value renders this + // backend type; future values need their own flag rendering. // // -ha_backend_connstring is "namespace/lease-name", and both halves are derived: a backend is // cluster-scoped, its objects live in one shared namespace, and LeaderObjectName is already what diff --git a/pkg/worker/kvcache/mooncake/leader_flags_test.go b/pkg/worker/kvcache/mooncake/leader_flags_test.go index 7bf412069..c54829472 100644 --- a/pkg/worker/kvcache/mooncake/leader_flags_test.go +++ b/pkg/worker/kvcache/mooncake/leader_flags_test.go @@ -184,6 +184,9 @@ func TestRenderLeaderFlags_DiskTier(t *testing.T) { // election flags are derived from. The fixture rather than a bare object, so a rename of the // backend or a change to LeaderObjectName reaches every case here. func leaderBackend(leader workercore.KVCacheBackendLeader) *workercore.KVCacheBackend { + if leader.ElectionBackend == "" { + leader.ElectionBackend = "None" + } return testBackend(func(kvcb *workercore.KVCacheBackend) { kvcb.Spec.Connection.Managed.Leader = leader }) @@ -203,6 +206,7 @@ func leaderBackend(leader workercore.KVCacheBackendLeader) *workercore.KVCacheBa func TestRenderLeaderFlags_HighAvailability(t *testing.T) { kvcb := leaderBackend(workercore.KVCacheBackendLeader{ Replicas: ptr.To[int32](3), + ElectionBackend: "Kubernetes", AllocationStrategy: "FreeRatioFirst", MultiTenancy: ptr.To(false), HighAvailability: &workercore.KVCacheBackendLeaderHighAvailability{}, @@ -241,14 +245,15 @@ func TestRenderLeaderFlags_PodIdentityOnlyUnderElection(t *testing.T) { kvcb *workercore.KVCacheBackend want bool }{ - {"a single leader", testBackend(), false}, + {"a single leader without election", testBackend(), false}, { - "highAvailability at one replica, which elects nothing", + "Kubernetes election at one replica", leaderBackend(workercore.KVCacheBackendLeader{ Replicas: ptr.To[int32](1), + ElectionBackend: "Kubernetes", HighAvailability: &workercore.KVCacheBackendLeaderHighAvailability{}, }), - false, + true, }, {"an electing backend", haBackend(), true}, } @@ -351,10 +356,8 @@ func TestRenderLeaderFlags_IsDeterministic(t *testing.T) { // the metadata plane is peer-to-peer so there is no store to point at, and -port is deprecated. A // flag appearing here later would be a behavior change nobody asked for, so the test names each one. // -// The four election flags are in this list for a DIFFERENT reason than the rest, and it is the one -// worth keeping: a backend that does not set leader.highAvailability must render the argv it -// rendered before that field existed. Every other entry is absent because nothing renders it at -// all; these four are absent because the object did not ask. +// The election flags are absent because this fixture explicitly selects electionBackend None. +// Every other entry is absent because nothing in this scope renders it. func TestRenderLeaderFlags_OmitsWhatThisScopeDoesNotRun(t *testing.T) { absent := []string{ "-etcd_endpoints", diff --git a/pkg/worker/kvcache/mooncake/leader_workload.go b/pkg/worker/kvcache/mooncake/leader_workload.go index 22f40f8ce..bfcfb64f3 100644 --- a/pkg/worker/kvcache/mooncake/leader_workload.go +++ b/pkg/worker/kvcache/mooncake/leader_workload.go @@ -138,20 +138,15 @@ func LeaderEndpoints(kvcb *workercore.KVCacheBackend) []workercore.KVCacheBacken } } -// LeaderReplicas is how many leader processes the backend RUNS, which is spec.replicas only once -// something elects between them. +// LeaderReplicas is how many leader processes the backend runs. // // REQUIRED: the clamp is the split brain, not defensive tidiness. The schema's `maximum=5` caps the // value but cannot express a cross-field pairing, and the schema is the documented authority exactly -// where the webhook is NOT installed -- so that cluster admits `replicas: 3` with no -// `highAvailability`, and three masters with nothing electing between them each serve, each +// where the webhook is NOT installed -- so that cluster admits `replicas: 3` with +// `electionBackend: None`, and three masters with nothing electing between them each serve, each // allocating against one pool. One process is the only safe reading of that object, and the // divergence from spec.replicas is visible in `kubectl get deploy` rather than in a log line. // -// The same clamp covers highAvailability with one replica, where it is a no-op on the number: the -// election the field asks for does not exist below two processes (see leaderNeedsAPIAccess), so the -// count the field names is the count that renders anyway. -// // Everything else the replica count decides -- the update strategy, the rollout deadline -- reads // this rather than the field, so the three cannot disagree about whether there is an election. func LeaderReplicas(leader workercore.KVCacheBackendLeader) int32 { @@ -187,8 +182,8 @@ func LeaderTemplateElects(template core.PodTemplateSpec) bool { // LeaderDeploymentAtOneReplica is a rendered leader Deployment with the count, strategy and deadline // of one replica, and the template left as rendered. // -// It is the first of the two writes that raise an EXISTING Deployment past one replica, where the -// election turns on. The Deployment controller handles a replica change as a scaling event before it +// It is the first of the two writes that raise an unelected Deployment past one replica. The +// Deployment controller handles a replica change as a scaling event before it // consults the strategy, and scales the only active ReplicaSet to the new count -- in the update that // also changes the template, that is the OLD one. One write carrying both would start more masters // that do not elect, next to the one already serving, whatever the strategy says. Written at one @@ -205,12 +200,10 @@ func LeaderDeploymentAtOneReplica(deploy *apps.Deployment) *apps.Deployment { // leaderUpdateStrategy picks how the leader's Deployment is updated, and the two answers are // opposites rather than variations. // -// SINGLE REPLICA: Recreate. A RollingUpdate's maxSurge defaults to 25%, which ROUNDS UP -- to one, -// against one desired replica -- so the new master starts before the old one stops and the two run -// at once. Without an election that is a split brain, on every image or flag change rather than -// never. The cost is a gap with no master, which is the right trade: a member that loses its master -// keeps its segment and re-registers, while two masters allocating against one pool cannot be -// reconciled after the fact. +// SINGLE REPLICA: Recreate. A RollingUpdate's maxSurge defaults to 25%, which rounds up to one. +// With election disabled, the old and new masters would serve together. With election enabled, +// the new standby cannot become ready until the old leader releases the Lease, so the default +// maxUnavailable of zero can stall the rollout. Recreate stops the old process first. // // SEVERAL REPLICAS: RollingUpdate, and both parameters invert. // - maxSurge may exceed zero, because the lease admits one leader however many processes run. diff --git a/pkg/worker/kvcache/mooncake/leader_workload_test.go b/pkg/worker/kvcache/mooncake/leader_workload_test.go index 8563e346f..c77a8303a 100644 --- a/pkg/worker/kvcache/mooncake/leader_workload_test.go +++ b/pkg/worker/kvcache/mooncake/leader_workload_test.go @@ -122,6 +122,7 @@ func testBackend(mutate ...func(*workercore.KVCacheBackend)) *workercore.KVCache Managed: &workercore.KVCacheBackendManaged{ Leader: workercore.KVCacheBackendLeader{ Replicas: ptr.To[int32](1), + ElectionBackend: "None", AllocationStrategy: "FreeRatioFirst", }, Members: []workercore.KVCacheBackendMember{ @@ -206,8 +207,8 @@ func withLeaderReplicas(replicas int32) func(*workercore.KVCacheBackend) { // TestLeaderWorkload_UpdateStrategyInvertsWithStandbys asserts the two update shapes ARE opposites, // not variations, and each case names the failure the other one produces. // -// One replica has no election, so a surging second master is a split brain. Several replicas have an -// election, so the surge is safe -- and now the danger is the reverse: exactly one replica is ever +// With one replica, Recreate avoids an unelected split brain and an elected rollout deadlock. +// Several replicas have an election, so the surge is safe. Exactly one replica is ever // ready, because the standbys deliberately are not, so maxUnavailable has to be the replica count // itself. `[跑]` replicas-1 was measured to deadlock on a single-node cluster: the controller scales // an old Pod down only while availablePodCount exceeds replicas-maxUnavailable, which at replicas-1 diff --git a/pkg/worker/kvcache/mooncake/member_workload.go b/pkg/worker/kvcache/mooncake/member_workload.go index 6d3c6bb46..a754fd0eb 100644 --- a/pkg/worker/kvcache/mooncake/member_workload.go +++ b/pkg/worker/kvcache/mooncake/member_workload.go @@ -583,7 +583,7 @@ func memberContainerSpec( // standby does not answer as the serving leader. func MemberMasterEntry(kvcb *workercore.KVCacheBackend) string { leader := kvcb.Spec.Connection.Managed.Leader - if leaderNeedsAPIAccess(leader) && + if leaderNeedsAPIAccess(leader) && leader.HighAvailability != nil && leader.HighAvailability.MemberAddressing == MemberAddressingLease { return fmt.Sprintf("k8s://%s/%s", kuberess.SystemNamespaceName, LeaderObjectName(kvcb)) } diff --git a/pkg/worker/webhooks/worker/kv_cache_backend.go b/pkg/worker/webhooks/worker/kv_cache_backend.go index 62b6cc438..6886ba6b4 100644 --- a/pkg/worker/webhooks/worker/kv_cache_backend.go +++ b/pkg/worker/webhooks/worker/kv_cache_backend.go @@ -472,17 +472,17 @@ func validateKVCacheBackendManaged( MaxLeaderReplicas))) } - // The pairing rule, and it names the field that is MISSING rather than the one that is set. + // The pairing rule rejects None when several leader replicas are requested. // Several leaders with nothing electing between them is not a degraded configuration: each // one serves, and the members register with whichever they were told about. - if *replicas > 1 && managed.Leader.HighAvailability == nil { + if *replicas > 1 && !managed.Leader.KubernetesElectionEnabled() { errs = append(errs, field.Invalid(fldPath.Child("leader", "replicas"), *replicas, - "more than one leader requires leader.highAvailability, which elects one of them "+ - "through a Kubernetes Lease; without it every replica would serve")) + "more than one leader requires leader.electionBackend to be Kubernetes; "+ + "with None every replica would serve")) } } - // REQUIRED: an update that switches high availability on or off re-runs these rules even over a + // REQUIRED: an update that switches Kubernetes election on or off re-runs these rules even over a // list it did not touch, and the exemption below is what makes that necessary. Turning the field // on is what turns `enable_ha`, `ha_backend_type`, `ha_backend_connstring` and `cluster_id` into // DERIVED flags; an object admitted before they were derived may carry one, and the renderer @@ -493,8 +493,7 @@ func validateKVCacheBackendManaged( haUnchanged := true if oldManaged != nil { oldLeaderExtraArgs = oldManaged.Leader.ExtraArgs - haUnchanged = (oldManaged.Leader.HighAvailability == nil) == - (managed.Leader.HighAvailability == nil) + haUnchanged = oldManaged.Leader.KubernetesElectionEnabled() == managed.Leader.KubernetesElectionEnabled() } if !haUnchanged || !unchangedPassthrough(oldManaged != nil, oldLeaderExtraArgs, managed.Leader.ExtraArgs) { @@ -508,7 +507,7 @@ func validateKVCacheBackendManaged( // // The reserved list being unconditional is what makes the coupling necessary, not what makes it // unnecessary. What a declaration moves is which of those names the renderer EMITS: the Pod's - // identity and IP are emitted only under high availability. A list carrying one of those names + // identity and IP are emitted only under Kubernetes election. A list carrying one of those names // from before it was reserved is exempted by the passthrough comparison, and the renderer appends // the hatch AFTER the derived variables -- so the update that turns the declaration on is the // moment the stale value starts overriding the reference the argv arrives with, and it is the one @@ -1073,7 +1072,7 @@ func hasParentDirComponent(path string) bool { // // A user who touches the list gets the rule. A user who touches anything else, and the controller // touching nothing, do not. The leader's extraArgs caller adds one condition on top of this: an -// update that moves `highAvailability` moves which keys are derived, so it re-runs the rules +// update that moves `electionBackend` between None and Kubernetes changes the derived keys, so it re-runs the rules // regardless. func unchangedPassthrough[T comparable](isUpdate bool, old, current []T) bool { return isUpdate && slices.Equal(old, current) diff --git a/pkg/worker/webhooks/worker/kv_cache_backend_test.go b/pkg/worker/webhooks/worker/kv_cache_backend_test.go index dd0957c12..4df2ce4f2 100644 --- a/pkg/worker/webhooks/worker/kv_cache_backend_test.go +++ b/pkg/worker/webhooks/worker/kv_cache_backend_test.go @@ -34,6 +34,7 @@ func newKVCacheBackend() *workercore.KVCacheBackend { Managed: &workercore.KVCacheBackendManaged{ Leader: workercore.KVCacheBackendLeader{ Replicas: ptr.To[int32](1), + ElectionBackend: "None", AllocationStrategy: "FreeRatioFirst", }, Members: []workercore.KVCacheBackendMember{{ @@ -194,28 +195,37 @@ func TestKVCacheBackendWebhook_ValidateCreate(t *testing.T) { {"replicas 1", func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) }, ""}, + {"replicas 1 without election", func(k *workercore.KVCacheBackend) { + k.Spec.Connection.Managed.Leader.ElectionBackend = "None" + }, ""}, + {"replicas 3 without election", func(k *workercore.KVCacheBackend) { + k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) + k.Spec.Connection.Managed.Leader.ElectionBackend = "None" + }, "requires leader.electionBackend"}, {"replicas unset, which the schema defaults", func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = nil }, ""}, // The pairing rule, in both directions. Refusing the first without accepting the second // would be a rule nobody can satisfy, and accepting the second without refusing the first // is the configuration it exists to prevent: several leaders, each one serving. - {"replicas 3 without the field that elects", func(k *workercore.KVCacheBackend) { + {"replicas 3 with None", func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) - }, "requires leader.highAvailability"}, - {"replicas 3 with it", func(k *workercore.KVCacheBackend) { + }, "requires leader.electionBackend"}, + {"replicas 3 with the default election", func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) - k.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + k.Spec.Connection.Managed.Leader.ElectionBackend = "" + }, ""}, + {"replicas 3 with Kubernetes", func(k *workercore.KVCacheBackend) { + k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) + k.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" }, ""}, // The ceiling still applies WITH the field: HA lifts the pairing rule, not the bound. {"replicas past the ceiling, even with the field", func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](MaxLeaderReplicas + 1) - k.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + k.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" }, "at most"}, - // One replica with the field is accepted rather than refused as pointless: the field is inert - // below two replicas and turns live the moment replicas rises, so setting it up front is how - // a later scale-up stays a one-field change. - {"replicas 1 with the field", func(k *workercore.KVCacheBackend) { + // The addressing block does not override an explicit None election choice. + {"replicas 1 with addressing but no election", func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) k.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} }, ""}, @@ -1000,7 +1010,7 @@ func TestKVCacheBackendWebhook_AGrandfatheredExtraArgIsNotRefusedOnEveryUpdate(t grandfathered := []string{"-enable_ha=false"} oldKvcb.Spec.Connection.Managed.Leader.ExtraArgs = grandfathered newKvcb.Spec.Connection.Managed.Leader.ExtraArgs = slices.Clone(grandfathered) - newKvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + newKvcb.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" _, err := wh.ValidateUpdate(context.Background(), oldKvcb, newKvcb) require.Error(t, err, "the field moved, so the keys it derives are read again") @@ -1031,7 +1041,7 @@ func TestKVCacheBackendWebhook_AGrandfatheredExtraArgIsNotRefusedOnEveryUpdate(t } oldKvcb, newKvcb := newKVCacheBackend(), newKVCacheBackend() oldKvcb.Spec.Connection.Managed.Leader.ExtraEnv = grandfathered - newKvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + newKvcb.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" newKvcb.Spec.Connection.Managed.Leader.ExtraEnv = slices.Clone(grandfathered) _, err := wh.ValidateUpdate(context.Background(), oldKvcb, newKvcb) From 4a643040209228df07dd1dbee8fdc11d69392c05 Mon Sep 17 00:00:00 2001 From: thxCode Date: Tue, 29 Sep 2026 12:24:46 +0800 Subject: [PATCH 2/2] refactor(kv-cache): flatten Mooncake leader addressing --- .../references/page-map.md | 4 +- .agents/skills/gpustack-operator-e2e/SKILL.md | 2 +- .../gpustack-operator-e2e/cases/case-62.sh | 5 +- .../gpustack-operator-e2e/cases/case-63.sh | 5 +- .../gpustack-operator-e2e/cases/case-64.sh | 3 +- .../gpustack-operator-e2e/cases/case-74.sh | 53 +----- .../gpustack-operator-e2e/cases/case-76.sh | 1 - .../gpustack-operator-e2e/cases/case-77.sh | 1 - api/worker/v1alpha1/generated.pb.go | 176 ++---------------- api/worker/v1alpha1/generated.proto | 48 ++--- .../v1alpha1/generated.protomessage.pb.go | 2 - api/worker/v1alpha1/kv_cache_backend.go | 48 ++--- api/worker/v1alpha1/kv_cache_backend_test.go | 63 +------ api/worker/v1alpha1/zz_generated.crds.go | 31 ++- api/worker/v1alpha1/zz_generated.deepcopy.go | 21 --- .../v1alpha1/zz_generated.model_name.go | 5 - api/worker/zz_generated.openapi.go | 30 +-- docs/kv-cache/leader.md | 5 +- docs/kv-cache/walkthrough.md | 2 +- docs/settings.md | 2 +- pkg/kubeclients/applyconfiguration/utils.go | 2 - .../worker/v1alpha1/kvcachebackendleader.go | 24 ++- .../kvcachebackendleaderhighavailability.go | 46 ----- .../worker/kv_cache_backend_handover_test.go | 5 +- .../worker/kv_cache_backend_test.go | 36 ++-- pkg/worker/kvcache/mooncake/ha_rbac_test.go | 8 +- .../kvcache/mooncake/leader_flags_test.go | 10 +- .../kvcache/mooncake/leader_snapshot_test.go | 8 +- .../kvcache/mooncake/leader_workload.go | 2 +- .../kvcache/mooncake/leader_workload_test.go | 2 +- .../kvcache/mooncake/member_workload.go | 3 +- pkg/worker/settings/value.go | 2 +- .../webhooks/worker/kv_cache_backend_test.go | 9 +- 33 files changed, 149 insertions(+), 515 deletions(-) delete mode 100644 pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go diff --git a/.agents/skills/gpustack-operator-docs/references/page-map.md b/.agents/skills/gpustack-operator-docs/references/page-map.md index 7dc3fabe5..641713022 100644 --- a/.agents/skills/gpustack-operator-docs/references/page-map.md +++ b/.agents/skills/gpustack-operator-docs/references/page-map.md @@ -124,8 +124,8 @@ in one clause and does not describe it. **Owns** — the leader process end to end: the Deployment and ClusterIP Service and their two ports, the replica ceiling and the clamp that survives a missing webhook, the update strategy per replica count, the two probes and why they take different paths, the health document's four fields, and all -of `leader.highAvailability` -- the Lease, the image both roles need, the per-role ServiceAccounts, -and how a missing grant fails on each side. +of leader election -- `leader.electionBackend`, `leader.memberAddressing`, the Lease, the image both +roles need, the per-role ServiceAccounts, and how a missing grant fails on each side. **Never** — the member groups, the transport, the status algebra. Those stay on `backend.md`, which links here. It was split out when that page hit both the line and the `##` cap. diff --git a/.agents/skills/gpustack-operator-e2e/SKILL.md b/.agents/skills/gpustack-operator-e2e/SKILL.md index 4b63df96e..e4211a8d9 100644 --- a/.agents/skills/gpustack-operator-e2e/SKILL.md +++ b/.agents/skills/gpustack-operator-e2e/SKILL.md @@ -119,7 +119,7 @@ Each case is self-contained; its header (see **Case header contract**) states go | 70 | A routed P/D deployment owns and garbage-collects all six router objects, converges both `spec.router` transitions, remains Starting without an accelerator, and reports every cluster-observable `KVEventsPublishing` status/reason pair | `pkg/worker/controllers/worker/model_deployment_router.go`, `model_deployment_kv_events.go`, `model_deployment_status.go` | yes (confirm) | Single Ready node with no accelerator; CASE 1 has materialized one usable general InstanceType, and the cluster can pull the Mooncake fixture image so a real Ready Binding can make `Publishing` reachable. The router image is deliberately unpullable because readiness is isolated in CASE 71 | | 71 | A Ready router becomes `status.endpoint`; inability to pull the upstream router image is a stated SKIP rather than loss of CASE 70's lifecycle coverage | `pkg/worker/controllers/worker/model_deployment_status.go`, the default llm-d-router image contract | yes (confirm) | As CASE 70 plus pull access to the default llm-d-router and Envoy images; AUTO-SKIP only for an image-pull reason | | 73 | An engine under KV turnover writes the shared store, and the other replica's replay of the same prefixes is the reuse the chain exists for — the write half is a guard, the read half a KNOWN-FAILURE DETECTOR pair (case-67 polarity) that FAILS the day cross-replica reuse starts working, and must then be inverted into positive guards | `pkg/worker/controllers/worker/model_deployment_connector.go`, `pkg/worker/controllers/worker/model_deployment_binding.go`, `pkg/worker/kvcache/inject/**`, the engine image pin | yes (confirm) | A real accelerator pool with at least TWO free exclusive cards, model weights hostPath-staged on the accelerator nodes (`E2E_VB_WEIGHTS`, default `/mnt/kvcache-weights`), `E2E_VB_INSTANCE_TYPE` naming the accelerated InstanceType (exit 2 without it), and a registry the cluster can pull the Mooncake image from — a CUDA-only tag crashes members on CPU-only nodes, so the image must be CPU-capable. Where no two cards are free, `E2E_VB_EXISTING_BACKEND` + `E2E_VB_EXISTING_DOMAIN` + `E2E_VB_EXISTING_SERVICES` together run every verdict row against an existing engine pair and create nothing. The first case in the suite to run real vLLM engines; every verdict rides on counters (`master_key_count`, `vllm:external_prefix_cache_hits_total`, `mem_cache_hit_nums_`), never on TTFT | -| 74 | `leader.highAvailability.snapshot` is not in the installed schema — a strict create is refused naming it as an unknown field, a lenient one is accepted with the block pruned — and the store's snapshot flags are refused in leader.extraArgs on create and on an update to a running backend, each refusal on the entry with its reason (a restore can serve another key's bytes; the other keys are read only under a refused switch); the manifest without them, and an image-only update, are accepted as the positive baseline | `api/worker/v1alpha1/kv_cache_backend.go`, `pkg/worker/kvcache/mooncake/keys.go`, `pkg/worker/webhooks/worker/kv_cache_backend.go` | yes (confirm) | any (no GPU, no RDMA, no storage class); the creates and the updates are server-side dry runs, and the one backend the case persists selects no node, so it renders a leader and no member Pod | +| 74 | The store's snapshot flags are refused in leader.extraArgs on create and on an update to a running backend, each refusal on the entry with its reason (a restore can serve another key's bytes; the other keys are read only under a refused switch); the manifest without them, and an image-only update, are accepted as the positive baseline | `pkg/worker/kvcache/mooncake/keys.go`, `pkg/worker/webhooks/worker/kv_cache_backend.go` | yes (confirm) | any (no GPU, no RDMA, no storage class); the creates and the updates are server-side dry runs, and the one backend the case persists selects no node, so it renders a leader and no member Pod | | 75 | The default single leader holds a Lease; `electionBackend: None` with three replicas is refused; scaling from one to three preserves the first leader Pod; an image bump then emits `KVCacheLeaderHandover` with its count equal to the Lease transitions, replaces all leader Pods, and settles Ready with one serving replica | `pkg/worker/controllers/worker/kv_cache_backend_handover.go`, `pkg/worker/kvcache/mooncake/leader_workload.go` | yes (confirm) | any (no GPU, no RDMA) + a registry the cluster can pull BOTH pinned Mooncake tags from (`E2E_MOONCAKE_IMAGE` start / `E2E_MOONCAKE_ROLLOUT_IMAGE` target; both must parse this operator's argv, carry the lease backend, and run on CPU-only nodes) | | 76 | `RolloutComplete` stays truthful through a second mid-update (every sample is `Unknown/UpdateNotObserved`, `False/Progressing` or `True/Complete`, never a deadline-ish stall), the update converges with the election gate intact, and the member-re-registration dip clears within its window; the case never gates on `kubectl rollout status`, which times out on every multi-replica leader rollout by construction | `pkg/worker/controllers/worker/kv_cache_backend_rollout.go` | yes (confirm) | as CASE 75 (the same image pair and clauses) | | 77 | The multi-tenant ledger gate: an unregistered tenant's put is refused `-1701` while the client itself stays healthy, and a Pool+Binding whose `domain.name` is the tenant id admits the identical put; the tenant rides the keyword `tenant_id=` (the next positional slot is a TransferEngine pointer and raises), and the teardown drains the domain because a held domain blocks pool deletion open-ended | `pkg/worker/controllers/worker/kv_cache_pool.go` (the domain registration pass), `pkg/worker/kvcache/mooncake/**` (the master argv and lease render) | yes (confirm) | any (no GPU, no RDMA) + a registry the cluster can pull the Mooncake image from (`E2E_MOONCAKE_IMAGE`, CPU-capable, carrying the python client); the backend must be a replicated HA leader — the k8s:// master address and the probe's member Role exist only above one replica | diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-62.sh b/.agents/skills/gpustack-operator-e2e/cases/case-62.sh index c28caafcf..037698970 100644 --- a/.agents/skills/gpustack-operator-e2e/cases/case-62.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-62.sh @@ -10,7 +10,7 @@ # cluster-scoped; the Deployment, Service, Lease, ServiceAccounts, Roles and RoleBindings it renders # all live in . # -# Goal: With leader.replicas=3 and leader.highAvailability={}, three leader processes run and +# Goal: With leader.replicas=3 and leader.electionBackend=Kubernetes, three leader processes run and # exactly one serves, elected through a Kubernetes Lease. This proves on a real API # server what rendered objects and unit tests cannot: # (1) THE STEADY STATE IS AN EQUALITY: 3 desired / 1 ready. Three ready would mean @@ -40,7 +40,7 @@ # the wrong image, not a flake. Override with E2E_MOONCAKE_IMAGE; the default below # carries the backend. # -# Inputs: All real, nothing mocked. One KVCacheBackend (replicas 3, highAvailability, one DRAM +# Inputs: All real, nothing mocked. One KVCacheBackend (replicas 3, Kubernetes election, one DRAM # member group of 2Gi per node). The failover is induced by deleting the ready leader # Pod -- the one the Lease names. # @@ -249,7 +249,6 @@ spec: managed: leader: replicas: 3 - highAvailability: {} members: - nodeSelector: {kubernetes.io/os: linux} medium: DRAM diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-63.sh b/.agents/skills/gpustack-operator-e2e/cases/case-63.sh index 4e2997595..d4c08264d 100644 --- a/.agents/skills/gpustack-operator-e2e/cases/case-63.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-63.sh @@ -10,7 +10,7 @@ # cluster-scoped; its rendered objects live in , and the two probe Pods live in a namespace this # case creates and removes. # -# Goal: Under leader.highAvailability a member can be handed MOONCAKE_MASTER in one of two +# Goal: Under Kubernetes leader election a member can be handed MOONCAKE_MASTER in one of two # forms. Path A is "k8s:///": the client reads the Lease itself and follows # view changes, which costs the member's image a compiled-in lease backend and so # excludes every vendor image. Path B, the default, is the plain leader Service @@ -44,7 +44,7 @@ # what proves the probe itself can work before any failover is measured, so a # compiled-out backend fails there, loudly, rather than as a zero measurement. # -# Inputs: All real, nothing mocked. One KVCacheBackend (leader replicas 3, highAvailability, +# Inputs: All real, nothing mocked. One KVCacheBackend (leader replicas 3, Kubernetes election, # one DRAM member group of 2Gi). Two probe Pods running the store image's python # client: probe A set up with "k8s:///-leader" (it gets a ServiceAccount # bound to the member Role -- `get leases` only -- because that read is exactly what @@ -276,7 +276,6 @@ spec: managed: leader: replicas: 3 - highAvailability: {} members: - nodeSelector: {kubernetes.io/os: linux} medium: DRAM diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-64.sh b/.agents/skills/gpustack-operator-e2e/cases/case-64.sh index 96e8b9d96..56f999baa 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-64.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-64.sh @@ -21,7 +21,7 @@ # ready" wait times out. That timeout is the signature of the wrong image, not a # flake. Override with E2E_MOONCAKE_IMAGE; the default below carries the backend. # -# Inputs: All real, nothing mocked. One KVCacheBackend (replicas 3, highAvailability, one +# Inputs: All real, nothing mocked. One KVCacheBackend (replicas 3, Kubernetes election, one # DRAM member group of 2Gi per node). The failure is injected with # `kubectl delete pod --force --grace-period=0` on the Lease holder: the API object # vanishes immediately and the container runtime SIGKILLs the process, so the @@ -212,7 +212,6 @@ spec: managed: leader: replicas: 3 - highAvailability: {} members: - nodeSelector: {kubernetes.io/os: linux} medium: DRAM diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-74.sh b/.agents/skills/gpustack-operator-e2e/cases/case-74.sh index 7e3ed14fe..950ac359a 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-74.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-74.sh @@ -1,16 +1,13 @@ #!/usr/bin/env bash # -# CASE 74 — The API has no leader snapshot, and the store's snapshot flags are refused in the leader's -# extraArgs with the reason (MUTATING, self-recovering) +# CASE 74 — The store's snapshot flags are refused in the leader's extraArgs with the reason +# (MUTATING, self-recovering) # # case-74.sh # # Goal: The store's snapshot is not offered, because restoring one can make the cache serve -# another key's bytes instead of a miss. This case proves both halves on a live API -# server: `leader.highAvailability.snapshot` is not a field of the installed schema, so a -# strict client is refused on it as an unknown field and a lenient one has it pruned; and -# the flags that would turn the snapshot on through the escape hatch are refused by the -# webhook with that reason, on create and on an update to a running backend. +# another key's bytes instead of a miss. This case checks that the webhook refuses +# the snapshot flags in extraArgs on create and on an update to a running backend. # # Environment: Any cluster with the operator deployed; no GPU, no RDMA, no storage class. keeps # the suite's calling convention and is read only to clean up the leader Lease: a backend @@ -18,14 +15,11 @@ # # Inputs: All real, nothing mocked. The creates and the updates are sent as server-side dry runs, # which pass through admission and persist nothing. The update needs an object to -# update, so the case creates one backend with high availability, whose member group +# update, so the case creates one backend with Kubernetes election, whose member group # selects no node: it renders a leader and no member Pod. # -# Expected: - the manifest without a snapshot is accepted (the positive baseline: without it a +# Expected: - the manifest without snapshot flags is accepted (the positive baseline: without it a # refusal of every backend would pass); -# - with `leader.highAvailability.snapshot`, a strict create is refused naming that -# path as an unknown field, and a lenient create is accepted with the block pruned from -# the object the server returns; # - `-enable_snapshot=true` in leader.extraArgs is refused on that entry with the # wrong-data reason, and `-snapshot_interval_seconds=60` with the reason that it is # read only under a refused switch; @@ -54,7 +48,6 @@ LEADER="${BACKEND}-leader" # Every refusal is matched on its field AND its reason: extraArgs carries several refusals for other # keys, so the path alone would pass on any of them. -UNKNOWN_RE='unknown field "spec\.connection\.managed\.leader\.highAvailability\.snapshot"' SWITCH_RE='spec\.connection\.managed\.leader\.extraArgs\[0\]: Forbidden: snapshots are not supported: .*can serve another key.s bytes instead of a miss' COMPANION_RE='spec\.connection\.managed\.leader\.extraArgs\[0\]: Forbidden: it is read only when enable_snapshot or enable_snapshot_restore is set' @@ -83,7 +76,7 @@ teardown() { trap teardown EXIT # manifest prints one backend with three elected leaders. The argument lands -# under `leader:`, so it can add a snapshot block or an extraArgs list. The member group selects a +# under `leader:` to add an extraArgs list. The member group selects a # label no node carries, so the one object this case persists renders no member Pod. manifest() { cat < sends a server-side dry-run create and records # whether it was refused for the reason the regex names, rather than for any other. expect_refused() { @@ -129,33 +117,10 @@ expect_refused() { # ------------------------------------------------------------- create, as dry runs if out="$(manifest | kubectl create --dry-run=server -f - 2>&1)"; then - record PASS "create without a snapshot is accepted" "${BACKEND}" -else - record FAIL "create without a snapshot is accepted" "${out}" -fi - -if out="$(manifest "$SNAPSHOT_YAML" | kubectl create --dry-run=server --validate=strict -f - 2>&1)"; then - record FAIL "a strict create with a snapshot is refused as an unknown field" "accepted: ${out}" -elif [[ "$out" =~ $UNKNOWN_RE ]]; then - record PASS "a strict create with a snapshot is refused as an unknown field" "${BACKEND}" -else - record FAIL "a strict create with a snapshot is refused as an unknown field" "refused by another rule: ${out}" -fi - -# The lenient path: the warning goes to stderr and the object the server would store to stdout, so -# the two are read apart. -warn_file="$(mktemp)" -if obj="$(manifest "$SNAPSHOT_YAML" | kubectl create --dry-run=server --validate=warn -o json -f - 2>"$warn_file")"; then - ha="$(printf '%s' "$obj" | jq -c '.spec.connection.managed.leader.highAvailability')" - if [[ "$(cat "$warn_file")" =~ $UNKNOWN_RE ]] && [ "$(printf '%s' "$obj" | jq '.spec.connection.managed.leader.highAvailability | has("snapshot")')" = false ]; then - record PASS "a lenient create with a snapshot is accepted with the block pruned" "highAvailability=${ha}" - else - record FAIL "a lenient create with a snapshot is accepted with the block pruned" "highAvailability=${ha} warnings=$(cat "$warn_file")" - fi + record PASS "create without snapshot flags is accepted" "${BACKEND}" else - record FAIL "a lenient create with a snapshot is accepted with the block pruned" "refused: $(cat "$warn_file")" + record FAIL "create without snapshot flags is accepted" "${out}" fi -rm -f "$warn_file" expect_refused "-enable_snapshot in extraArgs is refused for serving wrong data" "$SWITCH_RE" \ ' extraArgs: ["-enable_snapshot=true"]' diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-76.sh b/.agents/skills/gpustack-operator-e2e/cases/case-76.sh index 55149a3fc..63bc3b1ad 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-76.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-76.sh @@ -161,7 +161,6 @@ spec: leader: replicas: 3 multiTenancy: true - highAvailability: {} members: - nodeSelector: {kubernetes.io/os: linux} medium: DRAM diff --git a/.agents/skills/gpustack-operator-e2e/cases/case-77.sh b/.agents/skills/gpustack-operator-e2e/cases/case-77.sh index b8980fa9c..19dabe63f 100755 --- a/.agents/skills/gpustack-operator-e2e/cases/case-77.sh +++ b/.agents/skills/gpustack-operator-e2e/cases/case-77.sh @@ -205,7 +205,6 @@ spec: leader: replicas: 3 multiTenancy: true - highAvailability: {} # The read-lease TTL, shortened for the teardown drain alone. The read-back after the admitted # put grants the key a lease of the leader's -default_kv_lease_ttl, a remove is refused with # OBJECT_HAS_LEASE until it expires, and the operator renders five minutes -- longer than diff --git a/api/worker/v1alpha1/generated.pb.go b/api/worker/v1alpha1/generated.pb.go index 92e112c41..098aabefe 100644 --- a/api/worker/v1alpha1/generated.pb.go +++ b/api/worker/v1alpha1/generated.pb.go @@ -140,8 +140,6 @@ func (m *KVCacheBackendExternal) Reset() { *m = KVCacheBackendExternal{} } func (m *KVCacheBackendLeader) Reset() { *m = KVCacheBackendLeader{} } -func (m *KVCacheBackendLeaderHighAvailability) Reset() { *m = KVCacheBackendLeaderHighAvailability{} } - func (m *KVCacheBackendList) Reset() { *m = KVCacheBackendList{} } func (m *KVCacheBackendManaged) Reset() { *m = KVCacheBackendManaged{} } @@ -3476,51 +3474,16 @@ func (m *KVCacheBackendLeader) MarshalToSizedBuffer(dAtA []byte) (int, error) { i = encodeVarintGenerated(dAtA, i, uint64(len(m.AllocationStrategy))) i-- dAtA[i] = 0x1a - if m.HighAvailability != nil { - { - size, err := m.HighAvailability.MarshalToSizedBuffer(dAtA[:i]) - if err != nil { - return 0, err - } - i -= size - i = encodeVarintGenerated(dAtA, i, uint64(size)) - } - i-- - dAtA[i] = 0x12 - } - if m.Replicas != nil { - i = encodeVarintGenerated(dAtA, i, uint64(*m.Replicas)) - i-- - dAtA[i] = 0x8 - } - return len(dAtA) - i, nil -} - -func (m *KVCacheBackendLeaderHighAvailability) Marshal() (dAtA []byte, err error) { - size := m.Size() - dAtA = make([]byte, size) - n, err := m.MarshalToSizedBuffer(dAtA[:size]) - if err != nil { - return nil, err - } - return dAtA[:n], nil -} - -func (m *KVCacheBackendLeaderHighAvailability) MarshalTo(dAtA []byte) (int, error) { - size := m.Size() - return m.MarshalToSizedBuffer(dAtA[:size]) -} - -func (m *KVCacheBackendLeaderHighAvailability) MarshalToSizedBuffer(dAtA []byte) (int, error) { - i := len(dAtA) - _ = i - var l int - _ = l i -= len(m.MemberAddressing) copy(dAtA[i:], m.MemberAddressing) i = encodeVarintGenerated(dAtA, i, uint64(len(m.MemberAddressing))) i-- dAtA[i] = 0x12 + if m.Replicas != nil { + i = encodeVarintGenerated(dAtA, i, uint64(*m.Replicas)) + i-- + dAtA[i] = 0x8 + } return len(dAtA) - i, nil } @@ -9872,10 +9835,8 @@ func (m *KVCacheBackendLeader) Size() (n int) { if m.Replicas != nil { n += 1 + sovGenerated(uint64(*m.Replicas)) } - if m.HighAvailability != nil { - l = m.HighAvailability.Size() - n += 1 + l + sovGenerated(uint64(l)) - } + l = len(m.MemberAddressing) + n += 1 + l + sovGenerated(uint64(l)) l = len(m.AllocationStrategy) n += 1 + l + sovGenerated(uint64(l)) if m.MultiTenancy != nil { @@ -9898,17 +9859,6 @@ func (m *KVCacheBackendLeader) Size() (n int) { return n } -func (m *KVCacheBackendLeaderHighAvailability) Size() (n int) { - if m == nil { - return 0 - } - var l int - _ = l - l = len(m.MemberAddressing) - n += 1 + l + sovGenerated(uint64(l)) - return n -} - func (m *KVCacheBackendList) Size() (n int) { if m == nil { return 0 @@ -12693,7 +12643,7 @@ func (this *KVCacheBackendLeader) String() string { repeatedStringForExtraEnv += "}" s := strings.Join([]string{`&KVCacheBackendLeader{`, `Replicas:` + valueToStringGenerated(this.Replicas) + `,`, - `HighAvailability:` + strings.Replace(this.HighAvailability.String(), "KVCacheBackendLeaderHighAvailability", "KVCacheBackendLeaderHighAvailability", 1) + `,`, + `MemberAddressing:` + fmt.Sprintf("%v", this.MemberAddressing) + `,`, `AllocationStrategy:` + fmt.Sprintf("%v", this.AllocationStrategy) + `,`, `MultiTenancy:` + valueToStringGenerated(this.MultiTenancy) + `,`, `ExtraArgs:` + fmt.Sprintf("%v", this.ExtraArgs) + `,`, @@ -12703,16 +12653,6 @@ func (this *KVCacheBackendLeader) String() string { }, "") return s } -func (this *KVCacheBackendLeaderHighAvailability) String() string { - if this == nil { - return "nil" - } - s := strings.Join([]string{`&KVCacheBackendLeaderHighAvailability{`, - `MemberAddressing:` + fmt.Sprintf("%v", this.MemberAddressing) + `,`, - `}`, - }, "") - return s -} func (this *KVCacheBackendList) String() string { if this == nil { return "nil" @@ -24327,9 +24267,9 @@ func (m *KVCacheBackendLeader) Unmarshal(dAtA []byte) error { m.Replicas = &v case 2: if wireType != 2 { - return fmt.Errorf("proto: wrong wireType = %d for field HighAvailability", wireType) + return fmt.Errorf("proto: wrong wireType = %d for field MemberAddressing", wireType) } - var msglen int + var stringLen uint64 for shift := uint(0); ; shift += 7 { if shift >= 64 { return ErrIntOverflowGenerated @@ -24339,27 +24279,23 @@ func (m *KVCacheBackendLeader) Unmarshal(dAtA []byte) error { } b := dAtA[iNdEx] iNdEx++ - msglen |= int(b&0x7F) << shift + stringLen |= uint64(b&0x7F) << shift if b < 0x80 { break } } - if msglen < 0 { + intStringLen := int(stringLen) + if intStringLen < 0 { return ErrInvalidLengthGenerated } - postIndex := iNdEx + msglen + postIndex := iNdEx + intStringLen if postIndex < 0 { return ErrInvalidLengthGenerated } if postIndex > l { return io.ErrUnexpectedEOF } - if m.HighAvailability == nil { - m.HighAvailability = &KVCacheBackendLeaderHighAvailability{} - } - if err := m.HighAvailability.Unmarshal(dAtA[iNdEx:postIndex]); err != nil { - return err - } + m.MemberAddressing = string(dAtA[iNdEx:postIndex]) iNdEx = postIndex case 3: if wireType != 2 { @@ -24533,88 +24469,6 @@ func (m *KVCacheBackendLeader) Unmarshal(dAtA []byte) error { } return nil } -func (m *KVCacheBackendLeaderHighAvailability) Unmarshal(dAtA []byte) error { - l := len(dAtA) - iNdEx := 0 - for iNdEx < l { - preIndex := iNdEx - var wire uint64 - for shift := uint(0); ; shift += 7 { - if shift >= 64 { - return ErrIntOverflowGenerated - } - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := dAtA[iNdEx] - iNdEx++ - wire |= uint64(b&0x7F) << shift - if b < 0x80 { - break - } - } - fieldNum := int32(wire >> 3) - wireType := int(wire & 0x7) - if wireType == 4 { - return fmt.Errorf("proto: KVCacheBackendLeaderHighAvailability: wiretype end group for non-group") - } - if fieldNum <= 0 { - return fmt.Errorf("proto: KVCacheBackendLeaderHighAvailability: illegal tag %d (wire type %d)", fieldNum, wire) - } - switch fieldNum { - case 2: - if wireType != 2 { - return fmt.Errorf("proto: wrong wireType = %d for field MemberAddressing", wireType) - } - var stringLen uint64 - for shift := uint(0); ; shift += 7 { - if shift >= 64 { - return ErrIntOverflowGenerated - } - if iNdEx >= l { - return io.ErrUnexpectedEOF - } - b := dAtA[iNdEx] - iNdEx++ - stringLen |= uint64(b&0x7F) << shift - if b < 0x80 { - break - } - } - intStringLen := int(stringLen) - if intStringLen < 0 { - return ErrInvalidLengthGenerated - } - postIndex := iNdEx + intStringLen - if postIndex < 0 { - return ErrInvalidLengthGenerated - } - if postIndex > l { - return io.ErrUnexpectedEOF - } - m.MemberAddressing = string(dAtA[iNdEx:postIndex]) - iNdEx = postIndex - default: - iNdEx = preIndex - skippy, err := skipGenerated(dAtA[iNdEx:]) - if err != nil { - return err - } - if (skippy < 0) || (iNdEx+skippy) < 0 { - return ErrInvalidLengthGenerated - } - if (iNdEx + skippy) > l { - return io.ErrUnexpectedEOF - } - iNdEx += skippy - } - } - - if iNdEx > l { - return io.ErrUnexpectedEOF - } - return nil -} func (m *KVCacheBackendList) Unmarshal(dAtA []byte) error { l := len(dAtA) iNdEx := 0 diff --git a/api/worker/v1alpha1/generated.proto b/api/worker/v1alpha1/generated.proto index 48ba725c6..86f450b43 100644 --- a/api/worker/v1alpha1/generated.proto +++ b/api/worker/v1alpha1/generated.proto @@ -1501,9 +1501,22 @@ message KVCacheBackendLeader { // +k8s:validation:maximum=5 optional int32 replicas = 1; - // HighAvailability configures how members find the elected leader. Election itself is selected - // by ElectionBackend, so this block is optional even when Replicas exceeds one. - optional KVCacheBackendLeaderHighAvailability highAvailability = 2; + // MemberAddressing selects how a member finds the master once an election runs. Both forms + // reach the serving leader by different routes. Changing this rolls every member group. + // With ElectionBackend None, members use the Service even when this field is Lease. + // + // - Lease: the member reads the Lease's current holder itself. This needs API server access. + // - Service: the member uses the leader Service, which publishes only ready endpoints. A + // standby is not ready; client reconnect behavior across an election needs verification. + // + // Service is the default. In one failover comparison the two forms differed by 0.13 seconds, + // within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 + // seconds, so leader election dominated that comparison. Recheck after changing election timing. + // The Service route's endpoint transition was inferred from the result, not observed directly. + // + // +k8s:validation:default="Service" + // +k8s:validation:enum=["Lease","Service"] + optional string memberAddressing = 2; // ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica, // so scaling up does not change the first leader's startup flags. None is for a single leader @@ -1590,35 +1603,6 @@ message KVCacheBackendLeader { repeated InstanceEnvVar extraEnv = 6; } -// KVCacheBackendLeaderHighAvailability configures how members find the elected leader. -// -// A standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM -// key index. A member's local disk tier can re-register keys it has fully offloaded after the -// election. The store's snapshot is not offered: restoring one can make the cache -// serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs. -message KVCacheBackendLeaderHighAvailability { - // MemberAddressing selects how a member is told to find the master once an election runs. Both - // forms reach the leader that is serving, by different routes, and they are rendered into the - // same one variable — so changing this rolls every member group. - // - // - Lease: the member is handed the Lease's coordinates and reads the current holder itself. - // This needs the member to talk to the API server, which is why the member image has to - // carry the leadership backend at all. - // - Service: the member is handed the leader Service's address, exactly as it is without high - // availability. The Service publishes only READY endpoints and a standby deliberately is not - // ready, so the address resolves to the serving leader — the open part is whether the - // client's reconnect follows that endpoint across an election, and how long it takes. - // - // Service is the default. In one failover comparison the two forms differed by 0.13 seconds, - // within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 - // seconds, so leader election dominated that comparison. Recheck after changing election timing. - // The Service route's endpoint transition was inferred from the result, not observed directly. - // - // +k8s:validation:default="Service" - // +k8s:validation:enum=["Lease","Service"] - optional string memberAddressing = 2; -} - // KVCacheBackendList holds the list of KVCacheBackend. // // +k8s:deepcopy-gen:interfaces=k8s.io/apimachinery/pkg/runtime.Object diff --git a/api/worker/v1alpha1/generated.protomessage.pb.go b/api/worker/v1alpha1/generated.protomessage.pb.go index 1f14717ef..7fa6412e0 100644 --- a/api/worker/v1alpha1/generated.protomessage.pb.go +++ b/api/worker/v1alpha1/generated.protomessage.pb.go @@ -120,8 +120,6 @@ func (*KVCacheBackendExternal) ProtoMessage() {} func (*KVCacheBackendLeader) ProtoMessage() {} -func (*KVCacheBackendLeaderHighAvailability) ProtoMessage() {} - func (*KVCacheBackendList) ProtoMessage() {} func (*KVCacheBackendManaged) ProtoMessage() {} diff --git a/api/worker/v1alpha1/kv_cache_backend.go b/api/worker/v1alpha1/kv_cache_backend.go index bbd6b3291..cbdfd129c 100644 --- a/api/worker/v1alpha1/kv_cache_backend.go +++ b/api/worker/v1alpha1/kv_cache_backend.go @@ -292,9 +292,22 @@ type KVCacheBackendLeader struct { // +k8s:validation:maximum=5 Replicas *int32 `json:"replicas,omitempty" protobuf:"varint,1,opt,name=replicas"` - // HighAvailability configures how members find the elected leader. Election itself is selected - // by ElectionBackend, so this block is optional even when Replicas exceeds one. - HighAvailability *KVCacheBackendLeaderHighAvailability `json:"highAvailability,omitempty" protobuf:"bytes,2,opt,name=highAvailability"` + // MemberAddressing selects how a member finds the master once an election runs. Both forms + // reach the serving leader by different routes. Changing this rolls every member group. + // With ElectionBackend None, members use the Service even when this field is Lease. + // + // - Lease: the member reads the Lease's current holder itself. This needs API server access. + // - Service: the member uses the leader Service, which publishes only ready endpoints. A + // standby is not ready; client reconnect behavior across an election needs verification. + // + // Service is the default. In one failover comparison the two forms differed by 0.13 seconds, + // within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 + // seconds, so leader election dominated that comparison. Recheck after changing election timing. + // The Service route's endpoint transition was inferred from the result, not observed directly. + // + // +k8s:validation:default="Service" + // +k8s:validation:enum=["Lease","Service"] + MemberAddressing string `json:"memberAddressing,omitempty" protobuf:"bytes,2,opt,name=memberAddressing"` // ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica, // so scaling up does not change the first leader's startup flags. None is for a single leader @@ -393,35 +406,6 @@ func (in KVCacheBackendLeader) KubernetesElectionEnabled() bool { return in.ElectionBackend == "" || in.ElectionBackend == "Kubernetes" } -// KVCacheBackendLeaderHighAvailability configures how members find the elected leader. -// -// A standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM -// key index. A member's local disk tier can re-register keys it has fully offloaded after the -// election. The store's snapshot is not offered: restoring one can make the cache -// serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs. -type KVCacheBackendLeaderHighAvailability struct { - // MemberAddressing selects how a member is told to find the master once an election runs. Both - // forms reach the leader that is serving, by different routes, and they are rendered into the - // same one variable — so changing this rolls every member group. - // - // - Lease: the member is handed the Lease's coordinates and reads the current holder itself. - // This needs the member to talk to the API server, which is why the member image has to - // carry the leadership backend at all. - // - Service: the member is handed the leader Service's address, exactly as it is without high - // availability. The Service publishes only READY endpoints and a standby deliberately is not - // ready, so the address resolves to the serving leader — the open part is whether the - // client's reconnect follows that endpoint across an election, and how long it takes. - // - // Service is the default. In one failover comparison the two forms differed by 0.13 seconds, - // within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 - // seconds, so leader election dominated that comparison. Recheck after changing election timing. - // The Service route's endpoint transition was inferred from the result, not observed directly. - // - // +k8s:validation:default="Service" - // +k8s:validation:enum=["Lease","Service"] - MemberAddressing string `json:"memberAddressing,omitempty" protobuf:"bytes,2,opt,name=memberAddressing"` -} - // KVCacheBackendTransport is the data plane the members use. type KVCacheBackendTransport struct { // Protocol is the transport the members are ASKED to use. Auto resolves to TCP. diff --git a/api/worker/v1alpha1/kv_cache_backend_test.go b/api/worker/v1alpha1/kv_cache_backend_test.go index 3a1dfb975..5b2044541 100644 --- a/api/worker/v1alpha1/kv_cache_backend_test.go +++ b/api/worker/v1alpha1/kv_cache_backend_test.go @@ -5,10 +5,7 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" - "k8s.io/apiextensions-apiserver/pkg/apis/apiextensions" extension "k8s.io/apiextensions-apiserver/pkg/apis/apiextensions/v1" - structuralschema "k8s.io/apiextensions-apiserver/pkg/apiserver/schema" - "k8s.io/apiextensions-apiserver/pkg/apiserver/schema/pruning" "k8s.io/utils/ptr" ) @@ -50,6 +47,14 @@ func TestKVCacheBackendElectionBackendSchema(t *testing.T) { require.Len(t, backend.Enum, 2) assert.Equal(t, []string{`"None"`, `"Kubernetes"`}, []string{string(backend.Enum[0].Raw), string(backend.Enum[1].Raw)}) + addressing, ok := leader.Properties["memberAddressing"] + require.True(t, ok) + require.NotNil(t, addressing.Default) + assert.Equal(t, `"Service"`, string(addressing.Default.Raw)) + require.Len(t, addressing.Enum, 2) + assert.Equal(t, []string{`"Lease"`, `"Service"`}, + []string{string(addressing.Enum[0].Raw), string(addressing.Enum[1].Raw)}) + assert.NotContains(t, leader.Properties, "highAvailability") } // memberStatusListSchema returns status.members as the generated CRD carries it. @@ -139,58 +144,6 @@ func TestKVCacheBackendDiskTierRequiresItsPath(t *testing.T) { "ceiling, while an unset path would mean a host directory this operator chose") } -// TestKVCacheBackendLeaderSnapshotIsPrunedBySchema pins what the API server does with a -// `leader.highAvailability.snapshot` block: the schema has no such field, so the block is PRUNED -// and reported as an unknown field rather than refused by any rule of this operator. A client asking -// for strict field validation, which is kubectl's default, turns that report into a refusal; one -// that does not gets a warning and an object stored without the block. -// -// It runs the API server's own pruning over the generated schema, because that is the only layer -// that sees the block at all: the Go type has no field to decode it into, so no webhook can. -func TestKVCacheBackendLeaderSnapshotIsPrunedBySchema(t *testing.T) { - crd := GetCustomResourceDefinitions()["KVCacheBackend"] - require.NotNil(t, crd, "KVCacheBackend is not registered") - require.Len(t, crd.Spec.Versions, 1) - - var internal apiextensions.JSONSchemaProps - require.NoError(t, extension.Convert_v1_JSONSchemaProps_To_apiextensions_JSONSchemaProps( - crd.Spec.Versions[0].Schema.OpenAPIV3Schema, &internal, nil)) - structural, err := structuralschema.NewStructural(&internal) - require.NoError(t, err) - - highAvailability := func(ha map[string]any) map[string]any { - return map[string]any{ - "apiVersion": "worker.gpustack.ai/v1alpha1", - "kind": "KVCacheBackend", - "metadata": map[string]any{"name": "mooncake"}, - "spec": map[string]any{ - "connection": map[string]any{ - "managed": map[string]any{ - "leader": map[string]any{"highAvailability": ha}, - }, - }, - }, - } - } - - // The positive baseline: a field the schema does carry survives, so an empty result below is - // the schema speaking and not a pruner that reports nothing. - kept := highAvailability(map[string]any{"memberAddressing": "Lease"}) - assert.Empty(t, pruning.PruneWithOptions(kept, structural, true, - structuralschema.UnknownFieldPathOptions{TrackUnknownFieldPaths: true})) - assert.Equal(t, highAvailability(map[string]any{"memberAddressing": "Lease"}), kept) - - pruned := highAvailability(map[string]any{ - "memberAddressing": "Lease", - "snapshot": map[string]any{"persistentVolumeClaimName": "mooncake-snapshots"}, - }) - unknown := pruning.PruneWithOptions(pruned, structural, true, - structuralschema.UnknownFieldPathOptions{TrackUnknownFieldPaths: true}) - assert.Equal(t, []string{"spec.connection.managed.leader.highAvailability.snapshot"}, unknown) - assert.Equal(t, highAvailability(map[string]any{"memberAddressing": "Lease"}), pruned, - "the block is dropped and the rest of the object is stored as written") -} - // TestKVCacheBackendFabricInterfaceCountHasNoDefault pins that an unset interface count stays unset // in storage. A schema default would store a count on every group, including a TCP group that // renders nothing from it, and the object would then read as asking for an interface it never gets. diff --git a/api/worker/v1alpha1/zz_generated.crds.go b/api/worker/v1alpha1/zz_generated.crds.go index d06f6fd17..156e91cbb 100644 --- a/api/worker/v1alpha1/zz_generated.crds.go +++ b/api/worker/v1alpha1/zz_generated.crds.go @@ -2513,27 +2513,20 @@ func crd_gpustack_api_worker_v1alpha1_KVCacheBackend() *v1.CustomResourceDefinit }, XListType: ptr.To[string]("map"), }, - "highAvailability": { - Description: "HighAvailability configures how members find the elected leader. Election itself is selected\nby ElectionBackend, so this block is optional even when Replicas exceeds one.", - Type: "object", - Properties: map[string]v1.JSONSchemaProps{ - "memberAddressing": { - Description: "MemberAddressing selects how a member is told to find the master once an election runs. Both\nforms reach the leader that is serving, by different routes, and they are rendered into the\nsame one variable — so changing this rolls every member group.\n- Lease: the member is handed the Lease's coordinates and reads the current holder itself.\nThis needs the member to talk to the API server, which is why the member image has to\ncarry the leadership backend at all.\n- Service: the member is handed the leader Service's address, exactly as it is without high\navailability. The Service publishes only READY endpoints and a standby deliberately is not\nready, so the address resolves to the serving leader — the open part is whether the\nclient's reconnect follows that endpoint across an election, and how long it takes.\nService is the default. In one failover comparison the two forms differed by 0.13 seconds,\nwithin the noise of one run. Both first failed at 31.41 seconds and converged around 60.6\nseconds, so leader election dominated that comparison. Recheck after changing election timing.\nThe Service route's endpoint transition was inferred from the result, not observed directly.", - Type: "string", - Default: &v1.JSON{ - Raw: []byte(`"Service"`), - }, - Enum: []v1.JSON{ - { - Raw: []byte(`"Lease"`), - }, - { - Raw: []byte(`"Service"`), - }, - }, + "memberAddressing": { + Description: "MemberAddressing selects how a member finds the master once an election runs. Both forms\nreach the serving leader by different routes. Changing this rolls every member group.\nWith ElectionBackend None, members use the Service even when this field is Lease.\n- Lease: the member reads the Lease's current holder itself. This needs API server access.\n- Service: the member uses the leader Service, which publishes only ready endpoints. A\nstandby is not ready; client reconnect behavior across an election needs verification.\nService is the default. In one failover comparison the two forms differed by 0.13 seconds,\nwithin the noise of one run. Both first failed at 31.41 seconds and converged around 60.6\nseconds, so leader election dominated that comparison. Recheck after changing election timing.\nThe Service route's endpoint transition was inferred from the result, not observed directly.", + Type: "string", + Default: &v1.JSON{ + Raw: []byte(`"Service"`), + }, + Enum: []v1.JSON{ + { + Raw: []byte(`"Lease"`), + }, + { + Raw: []byte(`"Service"`), }, }, - Nullable: true, }, "multiTenancy": { Description: "MultiTenancy turns on the leader's per-tenant quota ledger and the tenant-scoped shard index\nbehind it. Off, every request falls into one default tenant and the index degrades to a plain\nkey hash, so two callers using different tenant names read each other's cache.\nIt is a FIELD rather than an extraArgs entry because another API validates against it: a\nKVCachePool over a backend with no ledger to write quota into is admitted with a warning that\nno per-tenant quota is in force, withdrawing the flag from a backend a pool already holds is\nrefused, and a webhook reading an unschema'd \"true\", \"1\" or \"True\" would be judging a value\ndomain that belongs to whoever typed it. The store's global -quota_bytes flag stays in\nextraArgs for the converse reason: no other API needs to interpret it.\nIT DEFAULTS TO TRUE, because the ledger is what makes the rest of this API mean what it says:\nwithout it a KVCachePoolBinding's ceiling is recorded but not enforced, and a master serves one\nreuse domain only, so a second Binding on it is refused. Mooncake has taken the switch since\n0.3.12, and the default store image is on 0.3.13.post1.\nOMITTING THIS KEY AND WRITING `multiTenancy: false` ARE DIFFERENT — the first takes the\ndefault, the second declines the ledger: only the explicit false renders no switch. A store\nimage older than Mooncake 0.3.12 does not recognize the switch and its master exits at\nstartup, so a backend on such an image, including the 0.3.10.post2 variants this project\nalso publishes, sets false here. Read it through KVCacheBackendLeader.MultiTenancyEnabled.", diff --git a/api/worker/v1alpha1/zz_generated.deepcopy.go b/api/worker/v1alpha1/zz_generated.deepcopy.go index f8a667a50..365f3da83 100644 --- a/api/worker/v1alpha1/zz_generated.deepcopy.go +++ b/api/worker/v1alpha1/zz_generated.deepcopy.go @@ -1287,11 +1287,6 @@ func (in *KVCacheBackendLeader) DeepCopyInto(out *KVCacheBackendLeader) { *out = new(int32) **out = **in } - if in.HighAvailability != nil { - in, out := &in.HighAvailability, &out.HighAvailability - *out = new(KVCacheBackendLeaderHighAvailability) - **out = **in - } if in.MultiTenancy != nil { in, out := &in.MultiTenancy, &out.MultiTenancy *out = new(bool) @@ -1320,22 +1315,6 @@ func (in *KVCacheBackendLeader) DeepCopy() *KVCacheBackendLeader { return out } -// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. -func (in *KVCacheBackendLeaderHighAvailability) DeepCopyInto(out *KVCacheBackendLeaderHighAvailability) { - *out = *in - return -} - -// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new KVCacheBackendLeaderHighAvailability. -func (in *KVCacheBackendLeaderHighAvailability) DeepCopy() *KVCacheBackendLeaderHighAvailability { - if in == nil { - return nil - } - out := new(KVCacheBackendLeaderHighAvailability) - in.DeepCopyInto(out) - return out -} - // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *KVCacheBackendList) DeepCopyInto(out *KVCacheBackendList) { *out = *in diff --git a/api/worker/v1alpha1/zz_generated.model_name.go b/api/worker/v1alpha1/zz_generated.model_name.go index a5e2798fa..1cef5a7d8 100644 --- a/api/worker/v1alpha1/zz_generated.model_name.go +++ b/api/worker/v1alpha1/zz_generated.model_name.go @@ -280,11 +280,6 @@ func (in KVCacheBackendLeader) OpenAPIModelName() string { return "ai.gpustack.worker.v1alpha1.KVCacheBackendLeader" } -// OpenAPIModelName returns the OpenAPI model name for this type. -func (in KVCacheBackendLeaderHighAvailability) OpenAPIModelName() string { - return "ai.gpustack.worker.v1alpha1.KVCacheBackendLeaderHighAvailability" -} - // OpenAPIModelName returns the OpenAPI model name for this type. func (in KVCacheBackendList) OpenAPIModelName() string { return "ai.gpustack.worker.v1alpha1.KVCacheBackendList" diff --git a/api/worker/zz_generated.openapi.go b/api/worker/zz_generated.openapi.go index a73246ca7..017d729a7 100644 --- a/api/worker/zz_generated.openapi.go +++ b/api/worker/zz_generated.openapi.go @@ -124,7 +124,6 @@ func GetOpenAPIDefinitions(ref common.ReferenceCallback) map[string]common.OpenA v1alpha1.KVCacheBackendEndpoint{}.OpenAPIModelName(): schema_gpustack_api_worker_v1alpha1_KVCacheBackendEndpoint(ref), v1alpha1.KVCacheBackendExternal{}.OpenAPIModelName(): schema_gpustack_api_worker_v1alpha1_KVCacheBackendExternal(ref), v1alpha1.KVCacheBackendLeader{}.OpenAPIModelName(): schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeader(ref), - v1alpha1.KVCacheBackendLeaderHighAvailability{}.OpenAPIModelName(): schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeaderHighAvailability(ref), v1alpha1.KVCacheBackendList{}.OpenAPIModelName(): schema_gpustack_api_worker_v1alpha1_KVCacheBackendList(ref), v1alpha1.KVCacheBackendManaged{}.OpenAPIModelName(): schema_gpustack_api_worker_v1alpha1_KVCacheBackendManaged(ref), v1alpha1.KVCacheBackendMember{}.OpenAPIModelName(): schema_gpustack_api_worker_v1alpha1_KVCacheBackendMember(ref), @@ -6467,10 +6466,11 @@ func schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeader(ref common.Referen Format: "int32", }, }, - "highAvailability": { + "memberAddressing": { SchemaProps: spec.SchemaProps{ - Description: "HighAvailability configures how members find the elected leader. Election itself is selected by ElectionBackend, so this block is optional even when Replicas exceeds one.", - Ref: ref(v1alpha1.KVCacheBackendLeaderHighAvailability{}.OpenAPIModelName()), + Description: "MemberAddressing selects how a member finds the master once an election runs. Both forms reach the serving leader by different routes. Changing this rolls every member group. With ElectionBackend None, members use the Service even when this field is Lease.\n\n - Lease: the member reads the Lease's current holder itself. This needs API server access.\n - Service: the member uses the leader Service, which publishes only ready endpoints. A\n standby is not ready; client reconnect behavior across an election needs verification.\n\nService is the default. In one failover comparison the two forms differed by 0.13 seconds, within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 seconds, so leader election dominated that comparison. Recheck after changing election timing. The Service route's endpoint transition was inferred from the result, not observed directly.", + Type: []string{"string"}, + Format: "", }, }, "electionBackend": { @@ -6540,27 +6540,7 @@ func schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeader(ref common.Referen }, }, Dependencies: []string{ - v1alpha1.InstanceEnvVar{}.OpenAPIModelName(), v1alpha1.KVCacheBackendLeaderHighAvailability{}.OpenAPIModelName()}, - } -} - -func schema_gpustack_api_worker_v1alpha1_KVCacheBackendLeaderHighAvailability(ref common.ReferenceCallback) common.OpenAPIDefinition { - return common.OpenAPIDefinition{ - Schema: spec.Schema{ - SchemaProps: spec.SchemaProps{ - Description: "KVCacheBackendLeaderHighAvailability configures how members find the elected leader.\n\nA standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM key index. A member's local disk tier can re-register keys it has fully offloaded after the election. The store's snapshot is not offered: restoring one can make the cache serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs.", - Type: []string{"object"}, - Properties: map[string]spec.Schema{ - "memberAddressing": { - SchemaProps: spec.SchemaProps{ - Description: "MemberAddressing selects how a member is told to find the master once an election runs. Both forms reach the leader that is serving, by different routes, and they are rendered into the same one variable — so changing this rolls every member group.\n\n - Lease: the member is handed the Lease's coordinates and reads the current holder itself.\n This needs the member to talk to the API server, which is why the member image has to\n carry the leadership backend at all.\n - Service: the member is handed the leader Service's address, exactly as it is without high\n availability. The Service publishes only READY endpoints and a standby deliberately is not\n ready, so the address resolves to the serving leader — the open part is whether the\n client's reconnect follows that endpoint across an election, and how long it takes.\n\nService is the default. In one failover comparison the two forms differed by 0.13 seconds, within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 seconds, so leader election dominated that comparison. Recheck after changing election timing. The Service route's endpoint transition was inferred from the result, not observed directly.", - Type: []string{"string"}, - Format: "", - }, - }, - }, - }, - }, + v1alpha1.InstanceEnvVar{}.OpenAPIModelName()}, } } diff --git a/docs/kv-cache/leader.md b/docs/kv-cache/leader.md index a594d4879..e6f8b67bd 100644 --- a/docs/kv-cache/leader.md +++ b/docs/kv-cache/leader.md @@ -138,8 +138,9 @@ spec: replicas: 3 ``` -`leader.highAvailability.memberAddressing` is optional and only selects how members find the -winner. It does not turn the election on. +`leader.memberAddressing` is optional and only selects how members find the +winner. It does not turn the election on. With `electionBackend: None`, members +use the leader Service even if `memberAddressing: Lease` is set. ⛔ **A published `kvcacheai/mooncake` image cannot do this, on either side.** Leadership backend availability is a compile-time switch and every option ships **off**: diff --git a/docs/kv-cache/walkthrough.md b/docs/kv-cache/walkthrough.md index a520c3d37..0cff77ae4 100644 --- a/docs/kv-cache/walkthrough.md +++ b/docs/kv-cache/walkthrough.md @@ -201,7 +201,7 @@ election and accounts were already present at one replica. See [the leader Deployment](leader.md#the-deployment-and-the-two-probes) for the cost of changing `leader.electionBackend`. -**`leader.highAvailability.memberAddressing` chooses how members find the master**, and defaults to +**`leader.memberAddressing` chooses how members find the master**, and defaults to `Service`. An explicit `Lease` value uses the member's API access to read the current holder. See [High availability](leader.md#high-availability) for the measured failover limits. diff --git a/docs/settings.md b/docs/settings.md index a677c5d2e..562198805 100644 --- a/docs/settings.md +++ b/docs/settings.md @@ -42,7 +42,7 @@ kubectl -n gpustack-system patch setting instance-type-derived-from-node --type | `image-pull-policy` | `GPUSTACK_IMAGE_PULL_POLICY` | `IfNotPresent` | Image pull policy for all built-in applications' deployments, on the same boundary as the secret above: a controller-rendered workload carries its own. | | `instance-general-resources-overcommit` | `GPUSTACK_INSTANCE_GENERAL_RESOURCES_OVERCOMMIT` | `true` | Overcommit an Instance's general resources: when enabled, a general unit requests 800m CPU / 128Mi RAM and one-eighth local storage, an accelerated unit 100m CPU / 128Mi RAM and one-eighth local storage — so e.g. a 1C/4Gi + 128Gi type requesting 2 accelerators and 64Gi storage resolves to 200m/256Mi + 8Gi. A CPU limit whose whole cores cost more than the limit itself (e.g. `100m`, or `1500m` whose half core rounds up) requests the limit instead — a request may never exceed its limit. | | `instance-ssh-server-image` | `GPUSTACK_INSTANCE_SSH_SERVER_IMAGE` | `gpustack/ssh-server:v1.3.0` | Image of the SSH server used when deploying Instances. | -| `kv-cache-backend-image` | `GPUSTACK_KV_CACHE_BACKEND_IMAGE` | `gpustack/mirrored-mooncake:0.3.13.post1-cpu` | Image every role of a [KV cache backend](kv-cache/backend.md) runs when the object does not name one itself. The default is this project's own build — a CPU build carrying TCP and EFA over DRAM, and the only image that can run [`leader.highAvailability`](kv-cache/leader.md#high-availability). The per-vendor variant builds, their tags and which one a VRAM group needs are under [The project's own build variants](kv-cache/backend.md#the-projects-own-build-variants). **It does not fit every backend**: a backend on another transport or runtime names its own `spec.image`, which always wins over this. **Clearing this Setting restores the admission refusal** for a backend naming no image anywhere. Why one default cannot fit every backend, and what having one costs, is under [The image](kv-cache/backend.md#the-image). | +| `kv-cache-backend-image` | `GPUSTACK_KV_CACHE_BACKEND_IMAGE` | `gpustack/mirrored-mooncake:0.3.13.post1-cpu` | Image every role of a [KV cache backend](kv-cache/backend.md) runs when the object does not name one itself. The default is this project's own build — a CPU build carrying TCP and EFA over DRAM, and the only image that can run [`leader.electionBackend`](kv-cache/leader.md#high-availability). The per-vendor variant builds, their tags and which one a VRAM group needs are under [The project's own build variants](kv-cache/backend.md#the-projects-own-build-variants). **It does not fit every backend**: a backend on another transport or runtime names its own `spec.image`, which always wins over this. **Clearing this Setting restores the admission refusal** for a backend naming no image anywhere. Why one default cannot fit every backend, and what having one costs, is under [The image](kv-cache/backend.md#the-image). | | `model-deployment-kv-cache-dtype-owned` | `GPUSTACK_MODEL_DEPLOYMENT_KV_CACHE_DTYPE_OWNED` | `true` | Hand every `KVCachePoolBinding`'s `spec.domain.dtype` to the engines attached through it as `--kv-cache-dtype`, verbatim, and refuse a pool-attached role or injected Pod naming that flag itself, and a new Binding declaring `auto` — see [The dtype is handed to the engine](kv-cache/pool.md#the-dtype-is-handed-to-the-engine). **It is the escape from a Binding whose spelling the engine rejects**, which makes every new Pod fail argument parsing: `false` renders and refuses nothing, as before the flag was owned. Flipping it recreates every pool-attached replica at its next reconcile. The recovery is under [Upgrading to an enforced Binding dtype](migration/kv-cache-dtype.md). | | `model-deployment-router-image` | `GPUSTACK_MODEL_DEPLOYMENT_ROUTER_IMAGE` | `gpustack/llm-router:v0.1.0` | Image a managed router runs when the `ModelDeployment` does not name one in `spec.router.image`. **It carries every router this operator supports, and which binary runs is decided by the rendered command rather than by the image** — `spec.router.name` picks the binary, and this setting only says where the binaries come from. Its default is safe for every cluster at once in a way `kv-cache-backend-image`'s is not: it runs no model, so it links no accelerator runtime and cannot be paired with the wrong one. The default is this project's own build rather than an upstream tag, because the three routers are compiled from three separate sources and one of them carries a patch this repository ships, so no upstream image holds them together. An image named on the object is used verbatim and is never redirected to the cluster mirror, because it is the user's own reference. | | `model-deployment-router-proxy-image` | `GPUSTACK_MODEL_DEPLOYMENT_ROUTER_PROXY_IMAGE` | `gpustack/mirrored-envoy:distroless-v1.33.2` | Proxy fronting a managed router's endpoint picker. It has **no field on the API** to override it: this operator renders the proxy's configuration against one proxy's configuration schema, so swapping the binary would mean swapping that configuration too. The setting exists for registry redirection and for pinning a release back, not for running a different proxy. | diff --git a/pkg/kubeclients/applyconfiguration/utils.go b/pkg/kubeclients/applyconfiguration/utils.go index f2486c6b8..511a9e38e 100644 --- a/pkg/kubeclients/applyconfiguration/utils.go +++ b/pkg/kubeclients/applyconfiguration/utils.go @@ -1392,8 +1392,6 @@ func ForKind(kind schema.GroupVersionKind) interface{} { return &applyconfigurationworkerv1alpha1.KVCacheBackendExternalApplyConfiguration{} case workerv1alpha1.SchemeGroupVersion.WithKind("KVCacheBackendLeader"): return &applyconfigurationworkerv1alpha1.KVCacheBackendLeaderApplyConfiguration{} - case workerv1alpha1.SchemeGroupVersion.WithKind("KVCacheBackendLeaderHighAvailability"): - return &applyconfigurationworkerv1alpha1.KVCacheBackendLeaderHighAvailabilityApplyConfiguration{} case workerv1alpha1.SchemeGroupVersion.WithKind("KVCacheBackendManaged"): return &applyconfigurationworkerv1alpha1.KVCacheBackendManagedApplyConfiguration{} case workerv1alpha1.SchemeGroupVersion.WithKind("KVCacheBackendMember"): diff --git a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go index 057c21e9a..ffdc6a26e 100644 --- a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go +++ b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleader.go @@ -20,9 +20,19 @@ type KVCacheBackendLeaderApplyConfiguration struct { // the webhook is not installed, which is when a second leader would be rendered rather than // refused. Raise both together; widening a maximum is not a breaking change. Replicas *int32 `json:"replicas,omitempty"` - // HighAvailability configures how members find the elected leader. Election itself is selected - // by ElectionBackend, so this block is optional even when Replicas exceeds one. - HighAvailability *KVCacheBackendLeaderHighAvailabilityApplyConfiguration `json:"highAvailability,omitempty"` + // MemberAddressing selects how a member finds the master once an election runs. Both forms + // reach the serving leader by different routes. Changing this rolls every member group. + // With ElectionBackend None, members use the Service even when this field is Lease. + // + // - Lease: the member reads the Lease's current holder itself. This needs API server access. + // - Service: the member uses the leader Service, which publishes only ready endpoints. A + // standby is not ready; client reconnect behavior across an election needs verification. + // + // Service is the default. In one failover comparison the two forms differed by 0.13 seconds, + // within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 + // seconds, so leader election dominated that comparison. Recheck after changing election timing. + // The Service route's endpoint transition was inferred from the result, not observed directly. + MemberAddressing *string `json:"memberAddressing,omitempty"` // ElectionBackend selects the leader election backend. Kubernetes uses a Lease even with one replica, // so scaling up does not change the first leader's startup flags. None is for a single leader // whose image cannot use Kubernetes election. More than one replica with None is refused. @@ -105,11 +115,11 @@ func (b *KVCacheBackendLeaderApplyConfiguration) WithReplicas(value int32) *KVCa return b } -// WithHighAvailability sets the HighAvailability field in the declarative configuration to the given value +// WithMemberAddressing sets the MemberAddressing field in the declarative configuration to the given value // and returns the receiver, so that objects can be built by chaining "With" function invocations. -// If called multiple times, the HighAvailability field is set to the value of the last call. -func (b *KVCacheBackendLeaderApplyConfiguration) WithHighAvailability(value *KVCacheBackendLeaderHighAvailabilityApplyConfiguration) *KVCacheBackendLeaderApplyConfiguration { - b.HighAvailability = value +// If called multiple times, the MemberAddressing field is set to the value of the last call. +func (b *KVCacheBackendLeaderApplyConfiguration) WithMemberAddressing(value string) *KVCacheBackendLeaderApplyConfiguration { + b.MemberAddressing = &value return b } diff --git a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go b/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go deleted file mode 100644 index dc52448a1..000000000 --- a/pkg/kubeclients/applyconfiguration/worker/v1alpha1/kvcachebackendleaderhighavailability.go +++ /dev/null @@ -1,46 +0,0 @@ -// Code generated by api. DO NOT EDIT. - -package v1alpha1 - -// KVCacheBackendLeaderHighAvailabilityApplyConfiguration represents a declarative configuration of the KVCacheBackendLeaderHighAvailability type for use -// with apply. -// -// KVCacheBackendLeaderHighAvailability configures how members find the elected leader. -// -// A standby REPLICATES NOTHING. Without a snapshot or operation log, the new leader loses the DRAM -// key index. A member's local disk tier can re-register keys it has fully offloaded after the -// election. The store's snapshot is not offered: restoring one can make the cache -// serve another key's bytes instead of a miss, which is why its flags are refused in extraArgs. -type KVCacheBackendLeaderHighAvailabilityApplyConfiguration struct { - // MemberAddressing selects how a member is told to find the master once an election runs. Both - // forms reach the leader that is serving, by different routes, and they are rendered into the - // same one variable — so changing this rolls every member group. - // - // - Lease: the member is handed the Lease's coordinates and reads the current holder itself. - // This needs the member to talk to the API server, which is why the member image has to - // carry the leadership backend at all. - // - Service: the member is handed the leader Service's address, exactly as it is without high - // availability. The Service publishes only READY endpoints and a standby deliberately is not - // ready, so the address resolves to the serving leader — the open part is whether the - // client's reconnect follows that endpoint across an election, and how long it takes. - // - // Service is the default. In one failover comparison the two forms differed by 0.13 seconds, - // within the noise of one run. Both first failed at 31.41 seconds and converged around 60.6 - // seconds, so leader election dominated that comparison. Recheck after changing election timing. - // The Service route's endpoint transition was inferred from the result, not observed directly. - MemberAddressing *string `json:"memberAddressing,omitempty"` -} - -// KVCacheBackendLeaderHighAvailabilityApplyConfiguration constructs a declarative configuration of the KVCacheBackendLeaderHighAvailability type for use with -// apply. -func KVCacheBackendLeaderHighAvailability() *KVCacheBackendLeaderHighAvailabilityApplyConfiguration { - return &KVCacheBackendLeaderHighAvailabilityApplyConfiguration{} -} - -// WithMemberAddressing sets the MemberAddressing field in the declarative configuration to the given value -// and returns the receiver, so that objects can be built by chaining "With" function invocations. -// If called multiple times, the MemberAddressing field is set to the value of the last call. -func (b *KVCacheBackendLeaderHighAvailabilityApplyConfiguration) WithMemberAddressing(value string) *KVCacheBackendLeaderHighAvailabilityApplyConfiguration { - b.MemberAddressing = &value - return b -} diff --git a/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go b/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go index d794da130..7870e9fba 100644 --- a/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go +++ b/pkg/worker/controllers/worker/kv_cache_backend_handover_test.go @@ -50,9 +50,8 @@ func electingBackend(name string) *workercore.KVCacheBackend { Connection: workercore.KVCacheBackendConnection{ Managed: &workercore.KVCacheBackendManaged{ Leader: workercore.KVCacheBackendLeader{ - Replicas: ptr.To[int32](3), - ElectionBackend: "Kubernetes", - HighAvailability: &workercore.KVCacheBackendLeaderHighAvailability{}, + Replicas: ptr.To[int32](3), + ElectionBackend: "Kubernetes", }, }, }, diff --git a/pkg/worker/controllers/worker/kv_cache_backend_test.go b/pkg/worker/controllers/worker/kv_cache_backend_test.go index 296ae992c..38bbb578a 100644 --- a/pkg/worker/controllers/worker/kv_cache_backend_test.go +++ b/pkg/worker/controllers/worker/kv_cache_backend_test.go @@ -1281,15 +1281,15 @@ func TestKVCacheBackendReconciler_ConvergesAMultiTenancySwitch(t *testing.T) { assert.NotContains(t, back.Containers[0].Args, "-enable_multi_tenants=true") } -// TestKVCacheBackendReconciler_ConvergesAHighAvailabilitySwitch pins that the API access follows the +// TestKVCacheBackendReconciler_ConvergesAnElectionSwitch pins that the API access follows the // field in BOTH directions. // // The off-to-on half is the obvious one. The on-to-off half is the one with a reason worth writing // down: the owner reference collects these three when the BACKEND is deleted, and a backend that -// merely drops highAvailability is not deleted -- so without an explicit removal it keeps a +// disables election is not deleted -- so without an explicit removal it keeps a // ServiceAccount that can still take a Lease, bound to a leader with no election left. Nothing // reports that, which is why it is asserted rather than left to the garbage collector. -func TestKVCacheBackendReconciler_ConvergesAHighAvailabilitySwitch(t *testing.T) { +func TestKVCacheBackendReconciler_ConvergesAnElectionSwitch(t *testing.T) { kvcb := newKVCacheBackendObject() cli := newKVCacheBackendClient(kvcb) ctx := context.Background() @@ -1301,11 +1301,9 @@ func TestKVCacheBackendReconciler_ConvergesAHighAvailabilitySwitch(t *testing.T) require.NoError(t, cli.Get(ctx, ctrlcli.ObjectKey{Name: kvcb.Name}, got)) if on { got.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" - got.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} got.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) } else { got.Spec.Connection.Managed.Leader.ElectionBackend = "None" - got.Spec.Connection.Managed.Leader.HighAvailability = nil got.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) } require.NoError(t, cli.Update(ctx, got)) @@ -1433,13 +1431,12 @@ func backfillDeprecatedServiceAccount(obj ctrlcli.Object) { pod.DeprecatedServiceAccount = pod.ServiceAccountName } -// TestKVCacheBackendReconciler_LeavingHighAvailabilityReleasesTheAccount pins that both workloads +// TestKVCacheBackendReconciler_LeavingElectionReleasesTheAccount pins that both workloads // stop naming their account once the election is gone, against a client that backfills the // deprecated alias the way the API server does. The accounts are deleted on the same pass, so a // template still naming one leaves every new Pod refused with "serviceaccount not found" and the -// backend without a leader or a member. Both ways out of the election are covered, because -// either one alone empties the name. -func TestKVCacheBackendReconciler_LeavingHighAvailabilityReleasesTheAccount(t *testing.T) { +// backend without a leader or a member. Disabling election must clear both references. +func TestKVCacheBackendReconciler_LeavingElectionReleasesTheAccount(t *testing.T) { for name, leave := range map[string]func(*workercore.KVCacheBackendLeader){ "ElectionDisabled": func(l *workercore.KVCacheBackendLeader) { l.ElectionBackend = "None" @@ -1472,7 +1469,7 @@ func TestKVCacheBackendReconciler_LeavingHighAvailabilityReleasesTheAccount(t *t Build() require.NotNil(t, reconcileKVCacheBackend(t, cli, kvcb.Name)) - require.NoError(t, turnOnHighAvailability(t, cli, kvcb.Name)) + require.NoError(t, turnOnKubernetesElection(t, cli, kvcb.Name)) leader, member := new(apps.Deployment), new(apps.DaemonSet) require.NoError(t, cli.Get(ctx, leaderObjectKey(kvcb), leader)) @@ -1523,10 +1520,8 @@ func TestKVCacheBackendReconciler_ConvergesTheRolloutShapeOnALiveDeployment(t *t got.Spec.Connection.Managed.Leader.Replicas = ptr.To(n) if n > 1 { got.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" - got.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} } else { got.Spec.Connection.Managed.Leader.ElectionBackend = "None" - got.Spec.Connection.Managed.Leader.HighAvailability = nil } require.NoError(t, cli.Update(ctx, got)) require.NotNil(t, reconcileKVCacheBackend(t, cli, kvcb.Name)) @@ -1614,7 +1609,6 @@ func TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters(t *testi return func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To(n) k.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" - k.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} } } assertOneUnderRecreate := func(t *testing.T, deploy *apps.Deployment) { @@ -1714,7 +1708,6 @@ func TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters(t *testi want := edit(t, cli, kvcb, func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) k.Spec.Connection.Managed.Leader.ElectionBackend = "None" - k.Spec.Connection.Managed.Leader.HighAvailability = nil }) deploy := live(t, cli, kvcb) assert.False(t, elects(deploy)) @@ -1755,16 +1748,15 @@ func TestKVCacheBackendReconciler_CrossesOneReplicaWithoutMixingMasters(t *testi } } -// turnOnHighAvailability edits the live object to ask for an election and runs one pass, handing +// turnOnKubernetesElection edits the live object to ask for an election and runs one pass, handing // back whatever that pass returned. It is separate from reconcileKVCacheBackend because the cases // below are about a pass that must FAIL, which that helper asserts against. -func turnOnHighAvailability(t *testing.T, cli ctrlcli.Client, name string) error { +func turnOnKubernetesElection(t *testing.T, cli ctrlcli.Client, name string) error { t.Helper() ctx := context.Background() got := new(workercore.KVCacheBackend) require.NoError(t, cli.Get(ctx, ctrlcli.ObjectKey{Name: name}, got)) - got.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} got.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" got.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) require.NoError(t, cli.Update(ctx, got)) @@ -1812,7 +1804,7 @@ func TestKVCacheBackendReconciler_ARoleBindingPointingElsewhere(t *testing.T) { } require.NoError(t, cli.Create(ctx, foreign(key.Namespace, key.Name))) - err := turnOnHighAvailability(t, cli, kvcb.Name) + err := turnOnKubernetesElection(t, cli, kvcb.Name) require.Error(t, err, "a name collision this operator cannot resolve fails the pass") assert.Contains(t, err.Error(), key.Name, "and names the object, which is the only thing an administrator can act on") @@ -1835,7 +1827,7 @@ func TestKVCacheBackendReconciler_ARoleBindingPointingElsewhere(t *testing.T) { // The positive baseline: without it, a check that refused every mismatch would satisfy the // case above just as well, and the repair this path exists for would be gone. - require.NoError(t, turnOnHighAvailability(t, cli, kvcb.Name)) + require.NoError(t, turnOnKubernetesElection(t, cli, kvcb.Name)) live := new(rbac.RoleBinding) require.NoError(t, cli.Get(ctx, key, live)) require.True(t, renderedForKVCacheBackend(live, kvcb), "this one is ours") @@ -1884,7 +1876,7 @@ func TestKVCacheBackendReconciler_WillNotClaimAnObjectItDidNotCreate(t *testing. }, })) - err := turnOnHighAvailability(t, cli, kvcb.Name) + err := turnOnKubernetesElection(t, cli, kvcb.Name) require.Error(t, err, "a name collision this operator cannot resolve fails the pass") assert.Contains(t, err.Error(), key.Name, "and names the object to act on") @@ -1905,7 +1897,7 @@ func TestKVCacheBackendReconciler_WillNotClaimAnObjectItDidNotCreate(t *testing. Namespace: leaderObjectKey(kvcb).Namespace, Name: mooncake.LeaderObjectName(kvcb), } - require.NoError(t, turnOnHighAvailability(t, cli, kvcb.Name)) + require.NoError(t, turnOnKubernetesElection(t, cli, kvcb.Name)) live := new(core.ServiceAccount) require.NoError(t, cli.Get(ctx, key, live)) require.True(t, renderedForKVCacheBackend(live, kvcb), "this one is ours") @@ -1952,7 +1944,7 @@ func TestKVCacheBackendReconciler_KeepsTheGrantWhenTheWorkloadUpdateFails(t *tes }). Build() - require.NoError(t, turnOnHighAvailability(t, cli, kvcb.Name)) + require.NoError(t, turnOnKubernetesElection(t, cli, kvcb.Name)) key := ctrlcli.ObjectKey{ Namespace: leaderObjectKey(kvcb).Namespace, Name: mooncake.LeaderObjectName(kvcb), } diff --git a/pkg/worker/kvcache/mooncake/ha_rbac_test.go b/pkg/worker/kvcache/mooncake/ha_rbac_test.go index d1a6fb965..b192aa046 100644 --- a/pkg/worker/kvcache/mooncake/ha_rbac_test.go +++ b/pkg/worker/kvcache/mooncake/ha_rbac_test.go @@ -17,7 +17,6 @@ func haBackend(mutate ...func(*workercore.KVCacheBackend)) *workercore.KVCacheBa all := append([]func(*workercore.KVCacheBackend){ func(kvcb *workercore.KVCacheBackend) { kvcb.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" - kvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} kvcb.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](3) }, }, mutate...) @@ -210,7 +209,7 @@ func TestMemberMasterEntry_UsesServiceWithAndWithoutHA(t *testing.T) { func TestMemberMasterEntry_AddressingSelectsExplicitLease(t *testing.T) { addressed := func(value string) *workercore.KVCacheBackend { return haBackend(func(kvcb *workercore.KVCacheBackend) { - kvcb.Spec.Connection.Managed.Leader.HighAvailability.MemberAddressing = value + kvcb.Spec.Connection.Managed.Leader.MemberAddressing = value }) } @@ -221,6 +220,11 @@ func TestMemberMasterEntry_AddressingSelectsExplicitLease(t *testing.T) { assert.Equal(t, lease, MemberMasterEntry(addressed(MemberAddressingLease))) assert.Equal(t, service, MemberMasterEntry(addressed(MemberAddressingService)), "the Service publishes only ready endpoints, and a standby is not ready") + withoutElection := testBackend(func(kvcb *workercore.KVCacheBackend) { + kvcb.Spec.Connection.Managed.Leader.MemberAddressing = MemberAddressingLease + }) + assert.Equal(t, service, MemberMasterEntry(withoutElection), + "Lease addressing takes effect only while Kubernetes election runs") accountFor := func(value string) string { return RenderMemberDaemonSet(addressed(value), 0, "mooncake:v0.3.13"). diff --git a/pkg/worker/kvcache/mooncake/leader_flags_test.go b/pkg/worker/kvcache/mooncake/leader_flags_test.go index c54829472..110a36f1c 100644 --- a/pkg/worker/kvcache/mooncake/leader_flags_test.go +++ b/pkg/worker/kvcache/mooncake/leader_flags_test.go @@ -192,7 +192,7 @@ func leaderBackend(leader workercore.KVCacheBackendLeader) *workercore.KVCacheBa }) } -// TestRenderLeaderFlags_HighAvailability asserts the election group, which is the one part of this +// TestRenderLeaderFlags_KubernetesElection asserts the election group, which is the one part of this // argv derived from the OBJECT rather than from the leader spec. // // The five flags are asserted as a contiguous group in order, not probed for individually: they are @@ -203,13 +203,12 @@ func leaderBackend(leader workercore.KVCacheBackendLeader) *workercore.KVCacheBa // with -rpc_port into the string it campaigns with, so it is the election's identity AND the address // the Lease hands to members that explicitly choose Lease addressing. Its 0.0.0.0 default would // give every replica one identity and send those members to an address that resolves to itself. -func TestRenderLeaderFlags_HighAvailability(t *testing.T) { +func TestRenderLeaderFlags_KubernetesElection(t *testing.T) { kvcb := leaderBackend(workercore.KVCacheBackendLeader{ Replicas: ptr.To[int32](3), ElectionBackend: "Kubernetes", AllocationStrategy: "FreeRatioFirst", MultiTenancy: ptr.To(false), - HighAvailability: &workercore.KVCacheBackendLeaderHighAvailability{}, }) assert.Equal(t, []string{ @@ -249,9 +248,8 @@ func TestRenderLeaderFlags_PodIdentityOnlyUnderElection(t *testing.T) { { "Kubernetes election at one replica", leaderBackend(workercore.KVCacheBackendLeader{ - Replicas: ptr.To[int32](1), - ElectionBackend: "Kubernetes", - HighAvailability: &workercore.KVCacheBackendLeaderHighAvailability{}, + Replicas: ptr.To[int32](1), + ElectionBackend: "Kubernetes", }), true, }, diff --git a/pkg/worker/kvcache/mooncake/leader_snapshot_test.go b/pkg/worker/kvcache/mooncake/leader_snapshot_test.go index abc212a79..ae81dc329 100644 --- a/pkg/worker/kvcache/mooncake/leader_snapshot_test.go +++ b/pkg/worker/kvcache/mooncake/leader_snapshot_test.go @@ -14,8 +14,8 @@ import ( // snapshot: no flag, no volume, no mount and no environment variable. // // The fixtures cover each branch that adds to the command line or the pod spec, because a snapshot -// piece added under one of them is invisible from the others: the election, one leader under a -// high-availability block, and the tenant quota policy, which is the one feature that does mount a +// piece added under one of them is invisible from the others: the election, one leader with Lease +// addressing, and the tenant quota policy, which is the one feature that does mount a // volume. func TestRenderLeader_RendersNoSnapshot(t *testing.T) { cases := []struct { @@ -24,8 +24,8 @@ func TestRenderLeader_RendersNoSnapshot(t *testing.T) { }{ {"a plain backend", testBackend()}, {"a backend electing its leader", haBackend()}, - {"one leader under a high-availability block", testBackend(func(kvcb *workercore.KVCacheBackend) { - kvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + {"one leader with Lease addressing", testBackend(func(kvcb *workercore.KVCacheBackend) { + kvcb.Spec.Connection.Managed.Leader.MemberAddressing = MemberAddressingLease })}, {"an elected leader under multi-tenancy", haBackend(func(kvcb *workercore.KVCacheBackend) { kvcb.Spec.Connection.Managed.Leader.MultiTenancy = ptr.To(true) diff --git a/pkg/worker/kvcache/mooncake/leader_workload.go b/pkg/worker/kvcache/mooncake/leader_workload.go index bfcfb64f3..7cc787d9d 100644 --- a/pkg/worker/kvcache/mooncake/leader_workload.go +++ b/pkg/worker/kvcache/mooncake/leader_workload.go @@ -95,7 +95,7 @@ const ( `printf '%s' "$POLICY_EMPTY" > "$POLICY_FILE"; fi` ) -// The two values leader.highAvailability.memberAddressing takes. An empty value renders Service, +// The two values leader.memberAddressing takes. An empty value renders Service, // which is the schema default for objects that passed admission. const ( MemberAddressingLease = "Lease" diff --git a/pkg/worker/kvcache/mooncake/leader_workload_test.go b/pkg/worker/kvcache/mooncake/leader_workload_test.go index c77a8303a..2235b24f3 100644 --- a/pkg/worker/kvcache/mooncake/leader_workload_test.go +++ b/pkg/worker/kvcache/mooncake/leader_workload_test.go @@ -282,7 +282,7 @@ func TestLeaderWorkload_UpdateStrategyInvertsWithStandbys(t *testing.T) { // TestLeaderWorkload_ReplicasAreClampedWithoutAnElection pins the renderer's own refusal to run // several masters, which is the only one left where the webhook is not installed. // -// The schema caps `replicas` at five but cannot express the pairing with `highAvailability`, and it +// The schema caps `replicas` at five but cannot express its pairing with `electionBackend`, and it // is documented as the authority in exactly that cluster -- so without this clamp such a cluster // admits `replicas: 3`, and three masters with nothing electing between them each serve, each // allocating against one pool. Both directions are asserted: the clamp must not also swallow the diff --git a/pkg/worker/kvcache/mooncake/member_workload.go b/pkg/worker/kvcache/mooncake/member_workload.go index a754fd0eb..b9f3aa5c9 100644 --- a/pkg/worker/kvcache/mooncake/member_workload.go +++ b/pkg/worker/kvcache/mooncake/member_workload.go @@ -583,8 +583,7 @@ func memberContainerSpec( // standby does not answer as the serving leader. func MemberMasterEntry(kvcb *workercore.KVCacheBackend) string { leader := kvcb.Spec.Connection.Managed.Leader - if leaderNeedsAPIAccess(leader) && leader.HighAvailability != nil && - leader.HighAvailability.MemberAddressing == MemberAddressingLease { + if leaderNeedsAPIAccess(leader) && leader.MemberAddressing == MemberAddressingLease { return fmt.Sprintf("k8s://%s/%s", kuberess.SystemNamespaceName, LeaderObjectName(kvcb)) } diff --git a/pkg/worker/settings/value.go b/pkg/worker/settings/value.go index f369e25b7..54526c6ae 100644 --- a/pkg/worker/settings/value.go +++ b/pkg/worker/settings/value.go @@ -83,7 +83,7 @@ var ( // not name one itself. // // The default is this project's own build, pack/mirrored-mooncake. It is the only image that can - // run leader.highAvailability, because no published upstream image carries a leadership backend + // run leader.electionBackend: Kubernetes, because no published upstream image carries a leadership backend // at all, and it is the build every cluster case in this repository exercises. // // WHAT THE DEFAULT DOES NOT FIT, because one value cannot be right for every backend at once: diff --git a/pkg/worker/webhooks/worker/kv_cache_backend_test.go b/pkg/worker/webhooks/worker/kv_cache_backend_test.go index 4df2ce4f2..aa545b6ad 100644 --- a/pkg/worker/webhooks/worker/kv_cache_backend_test.go +++ b/pkg/worker/webhooks/worker/kv_cache_backend_test.go @@ -224,10 +224,11 @@ func TestKVCacheBackendWebhook_ValidateCreate(t *testing.T) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](MaxLeaderReplicas + 1) k.Spec.Connection.Managed.Leader.ElectionBackend = "Kubernetes" }, "at most"}, - // The addressing block does not override an explicit None election choice. + // Addressing does not override an explicit None election choice. {"replicas 1 with addressing but no election", func(k *workercore.KVCacheBackend) { k.Spec.Connection.Managed.Leader.Replicas = ptr.To[int32](1) - k.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} + k.Spec.Connection.Managed.Leader.ElectionBackend = "None" + k.Spec.Connection.Managed.Leader.MemberAddressing = "Lease" }, ""}, // The oplog key: refused because the leader cannot START with it, not because this operator @@ -1045,7 +1046,7 @@ func TestKVCacheBackendWebhook_AGrandfatheredExtraArgIsNotRefusedOnEveryUpdate(t newKvcb.Spec.Connection.Managed.Leader.ExtraEnv = slices.Clone(grandfathered) _, err := wh.ValidateUpdate(context.Background(), oldKvcb, newKvcb) - require.Error(t, err, "high availability moved, so the names it emits are read again") + require.Error(t, err, "election changed, so the names it emits are read again") require.Contains(t, err.Error(), "this variable is rendered from this spec") }) @@ -1056,9 +1057,7 @@ func TestKVCacheBackendWebhook_AGrandfatheredExtraArgIsNotRefusedOnEveryUpdate(t {Name: mooncake.LeaderPodIPEnv, Value: "10.0.0.1"}, } oldKvcb, newKvcb := newKVCacheBackend(), newKVCacheBackend() - oldKvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} oldKvcb.Spec.Connection.Managed.Leader.ExtraEnv = grandfathered - newKvcb.Spec.Connection.Managed.Leader.HighAvailability = &workercore.KVCacheBackendLeaderHighAvailability{} newKvcb.Spec.Connection.Managed.Leader.ExtraEnv = slices.Clone(grandfathered) newKvcb.Spec.Image = "example.com/mooncake:v1"