Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 28 additions & 28 deletions internal/checks/registry.go
Original file line number Diff line number Diff line change
Expand Up @@ -25,11 +25,10 @@ import (
// credentials are supplied and SKIP — never pass — when they are not.
//
// The host inventory is intake.Profile.Registries, not a grep of this
// repository. Four of those hosts (ecr-public.aws.com, reg.kyverno.io,
// registry.cn-hangzhou.aliyuncs.com, sandbox-registry.cn-zhangjiakou…) are
// upstream chart defaults Bud installs unmodified and appear nowhere in the
// tree (§5.4.0) — precisely the hosts a corporate egress policy is most likely
// to block. registry.upstream-defaults is what keeps that inventory honest
// repository. Three of those hosts (ecr-public.aws.com, reg.kyverno.io,
// sandbox-registry.cn-zhangjiakou…) are upstream chart defaults Bud installs
// unmodified and appear nowhere in the tree (§5.4.0) — precisely the hosts a
// corporate egress policy is most likely to block. registry.upstream-defaults is what keeps that inventory honest
// between releases (§5.4.2).
//
// Nothing here requests a blob. Every question is answered from /v2/ and from
Expand All @@ -38,24 +37,25 @@ import (

const (
// HAMi's one non-Docker-Hub image. budcluster installs HAMi on any cluster
// where NVIDIA GPUs are detected and passes no registry override, so this
// Alibaba CN-region mirror is a hard dependency of GPU onboarding (§5.4.1).
regHAMiRegistry = "registry.cn-hangzhou.aliyuncs.com"
regHAMiRepo = "google_containers/kube-scheduler"

// The remedy must override BOTH fields. global.imageRegistry alone is a
// trap: it keeps the google_containers/ prefix, and
// registry.k8s.io/google_containers/kube-scheduler 404s (verified).
regHAMiRemedy = "override BOTH fields when HAMi is installed (playbooks/setup_cluster.yaml):\n" +
// where NVIDIA GPUs are detected, and overrides the chart's Alibaba
// CN-region default (registry.cn-hangzhou.aliyuncs.com/google_containers/
// kube-scheduler) to the upstream image. The tag is still derived from the
// cluster's own version, so that tag has to exist upstream (§5.4.1).
regHAMiRegistry = "registry.k8s.io"
regHAMiRepo = "kube-scheduler"

// A mirror must override BOTH fields. global.imageRegistry alone is a
// trap: it keeps the chart's google_containers/ prefix, and
// <mirror>/google_containers/kube-scheduler does not exist.
regHAMiRemedy = "allow " + regHAMiRegistry + " from the GPU nodes, or mirror " + regHAMiRegistry + "/" + regHAMiRepo +
" and point HAMi at the mirror (playbooks/setup_cluster.yaml), overriding BOTH fields:\n" +
" scheduler:\n" +
" kubeScheduler:\n" +
" image:\n" +
" registry: registry.k8s.io\n" +
" registry: <mirror>\n" +
" repository: kube-scheduler # NOT google_containers/kube-scheduler\n" +
"global.imageRegistry alone is NOT sufficient and is a trap: it yields " +
"registry.k8s.io/google_containers/kube-scheduler, which returns 404, and it also " +
"redirects every Docker Hub image in the chart. The alternative is to mirror " +
regHAMiRegistry + "/" + regHAMiRepo + " into a registry this cluster can reach."
"global.imageRegistry alone is NOT sufficient: it keeps the chart's google_containers/ prefix, " +
"and it also redirects every Docker Hub image in the chart."
)

func init() {
Expand Down Expand Up @@ -941,10 +941,10 @@ func regKeys[V any](m map[string]V) []string {

// regHAMiScheduler probes the one HAMi image that is not on Docker Hub. Three
// facts make it a blocker rather than a curiosity (§5.4.1): budcluster installs
// HAMi automatically wherever NVIDIA GPUs are detected, it passes no registry
// override so the Alibaba CN-region default stands, and the Helm task runs with
// atomic: true — so this single unpullable image rolls the entire release back
// instead of leaving a diagnosable ImagePullBackOff.
// HAMi automatically wherever NVIDIA GPUs are detected, the image's tag is the
// cluster's own Kubernetes version, and the Helm task runs with atomic: true —
// so this single unpullable image rolls the entire release back instead of
// leaving a diagnosable ImagePullBackOff.
func regHAMiScheduler(ctx context.Context, c *engine.Ctx, ch *engine.Check) engine.Result {
nodes, known := regGPUNodes(ctx, c)
if !known {
Expand Down Expand Up @@ -987,10 +987,10 @@ func regHAMiScheduler(ctx context.Context, c *engine.Ctx, ch *engine.Check) engi
Bounds("resolved from this workstation: the pull happens on the GPU node (registry.from-cluster), and HAMi's install is " +
"atomic, so a node-side failure rolls the whole release back")
case adapters.ManifestNotFound:
// The mirror is well stocked (v1.28.0 through v1.35.7 all resolve), so
// a miss is far more likely to mean this cluster's version is outside
// the mirrored range than that the derivation is wrong — but either way
// the pull fails, and the fix is the same override.
// registry.k8s.io publishes kube-scheduler for every Kubernetes patch
// release, so a miss means this cluster reports a version upstream
// never shipped — but either way the pull fails, and the fix is a
// mirror carrying the tag.
return ch.FailAs(engine.Risk,
fmt.Sprintf("%s answers but has no tag %s — HAMi would try to pull an image that does not exist and its atomic install would roll back", regHAMiRegistry, tag),
regHAMiRemedy,
Expand All @@ -1004,7 +1004,7 @@ func regHAMiScheduler(ctx context.Context, c *engine.Ctx, ch *engine.Check) engi
return ch.Fail(
fmt.Sprintf("%s is unreachable, so HAMi cannot pull %s — GPU cluster onboarding fails with a rolled-back release (atomic: true), not with a visible ImagePullBackOff", regHAMiRegistry, ref),
regHAMiRemedy,
append(detail, "reachability is the whole risk here: the mirror stocks v1.28.0 through v1.35.7, so the tag itself is rarely the problem")...).
append(detail, "reachability is the whole risk here: "+regHAMiRegistry+" publishes kube-scheduler for every Kubernetes release, so the tag itself is rarely the problem")...).
WithEvidence(ev)
}
}
Expand Down
49 changes: 35 additions & 14 deletions internal/checks/registry_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -392,8 +392,7 @@ func TestRegistryReachableRisksWhenOnlyInactiveFeatureHostsFail(t *testing.T) {
})
r := registryRun(t, f, "registry.reachable", func(c *engine.Ctx) {
registryStubNet(c, registryAnswers(http.StatusOK, map[string]int{
"nvcr.io": 0,
"registry.cn-hangzhou.aliyuncs.com": 0,
"nvcr.io": 0,
"sandbox-registry.cn-zhangjiakou.cr.aliyuncs.com": 0,
}))
})
Expand All @@ -419,11 +418,11 @@ func TestRegistryReachableBlocksWhenGPUNodesArePresentEvenIfUnanswered(t *testin
f := vanilla().with("nodes", "", node("n1"), node("g1", withGPU("4")))
r := registryRun(t, f, "registry.reachable", func(c *engine.Ctx) {
registryStubNet(c, registryAnswers(http.StatusOK, map[string]int{
"registry.cn-hangzhou.aliyuncs.com": 0,
"nvcr.io": 0,
}))
})
assertStatus(t, r, "BLOCK")
registryAssertMentions(t, r, "registry.cn-hangzhou.aliyuncs.com")
registryAssertMentions(t, r, "nvcr.io")
}

// mirror.gcr.io is dormant only while Harbor's trivy stays off. The promotion
Expand Down Expand Up @@ -1074,23 +1073,45 @@ func TestRegistryUpstreamDefaultsSkipsWithoutARender(t *testing.T) {
// HAMi's Helm task is atomic, so this one unpullable image does not leave a
// diagnosable ImagePullBackOff — it rolls the whole release back and GPU
// onboarding fails with nothing to look at.
func TestRegistryHAMiSchedulerBlocksWhenTheAlibabaMirrorIsUnreachable(t *testing.T) {
func TestRegistryHAMiSchedulerBlocksWhenRegistryK8sIsUnreachable(t *testing.T) {
f := vanilla().with("nodes", "", node("n1"), node("g1", withGPU("4")))
r := registryRun(t, f, "registry.hami-scheduler", func(c *engine.Ctx) {
registryStubNet(c, registryAnswers(http.StatusOK, map[string]int{regHAMiRegistry: 0}))
})
assertStatus(t, r, "BLOCK")
registryAssertMentions(t, r,
"registry.cn-hangzhou.aliyuncs.com",
// The remedy must override BOTH fields and warn about the trap: the
// obvious one-line fix produces a 404.
"registry: registry.k8s.io",
"registry.k8s.io/kube-scheduler",
// A mirror must override BOTH fields and warn about the trap: the
// obvious one-line fix keeps the chart's google_containers/ prefix.
"repository: kube-scheduler",
"global.imageRegistry alone is NOT sufficient",
"404",
)
}

// budcluster overrides HAMi's Alibaba CN-region default, so the check must
// probe the upstream image a GPU node will actually pull — probing the old
// mirror would block every cluster whose egress policy rightly excludes it.
func TestRegistryHAMiSchedulerNeverProbesTheAlibabaMirror(t *testing.T) {
f := vanilla().with("nodes", "", node("g1", withGPU("2")))
rec := &registryRecorder{next: registryAnswers(http.StatusOK, nil)}
r := registryRun(t, f, "registry.hami-scheduler", func(c *engine.Ctx) {
c.Platform.Version = "v1.36.3+k3s1"
registryStubNet(c, rec.answer)
})
assertStatus(t, r, "PASS")
if !rec.sawContaining("https://registry.k8s.io/v2/kube-scheduler/manifests/v1.36.3") {
t.Fatalf("expected a HEAD of the upstream kube-scheduler image, got:\n%s", strings.Join(rec.seen(), "\n"))
}
if rec.sawContaining("aliyuncs.com") {
t.Fatalf("the check still dialled the Alibaba mirror:\n%s", strings.Join(rec.seen(), "\n"))
}
for _, e := range egressTestProfile(t).Registries {
if e.Host == "registry.cn-hangzhou.aliyuncs.com" {
t.Fatal("the registry inventory still lists registry.cn-hangzhou.aliyuncs.com, so registry.reachable and registry.from-cluster would still probe it")
}
}
}

// A GPU node whose device plugin is not installed yet advertises no
// nvidia.com/* allocatable at all — and it is precisely the cluster budcluster
// is about to onboard HAMi onto.
Expand Down Expand Up @@ -1129,9 +1150,9 @@ func TestRegistryHAMiSchedulerDerivesTheTagFromTheClusterVersion(t *testing.T) {
}
}

// The mirror is well stocked, so a 404 is more likely to mean this cluster is
// outside the mirrored range than that the derivation is wrong — either way the
// pull fails, and the operator needs the same override.
// registry.k8s.io publishes every release, so a 404 means this cluster reports a
// version upstream never shipped — either way the pull fails, and the operator
// needs a mirror carrying the tag.
func TestRegistryHAMiSchedulerRisksWhenTheDerivedTagIsMissing(t *testing.T) {
f := vanilla().with("nodes", "", node("g1", withGPU("2")))
r := registryRun(t, f, "registry.hami-scheduler", func(c *engine.Ctx) {
Expand All @@ -1156,7 +1177,7 @@ func TestRegistryHAMiSchedulerRisksWhenAnonymousAccessIsRefused(t *testing.T) {

// No GPU nodes means budcluster never installs HAMi, so this image is never
// pulled. It must SKIP with that reason stated — not pass, which would read as
// "the Alibaba mirror is reachable".
// "the scheduler image is reachable".
func TestRegistryHAMiSchedulerSkipsWithoutGPUNodes(t *testing.T) {
f := vanilla().with("nodes", "", node("n1"), node("n2"))
r := registryRun(t, f, "registry.hami-scheduler", func(c *engine.Ctx) {
Expand Down
14 changes: 7 additions & 7 deletions internal/intake/defaults.yaml
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
# Floors and inventories budctl embeds at build time (FRD-020 §7, §9.3).
# These are the values an operator cannot reasonably be asked for. Capacity
# requirements do NOT live here — they derive from the intake answers.
catalogVersion: "2026-09-15"
catalogVersion: "2026-09-15.1"

# 80 GiB, not 60. The 14 first-party images at 1.2.8 measure 19.8 GiB compressed
# from registry manifests, inferring to ~40 GiB extracted; novu x4, otel,
Expand Down Expand Up @@ -73,12 +73,16 @@ appsetComponents:
- bud

# Verified by fetching the upstream charts and cross-checking against the images
# running in a live cluster. Four of these appear NOWHERE in the bud-runtime
# running in a live cluster. Three of these appear NOWHERE in the bud-runtime
# repository: they are upstream chart defaults that Bud installs unmodified.
#
# budimages.azurecr.io is retired and deliberately absent. Leftover references
# still name it — budcluster's NODE_INFO_* image env in the bud chart, which no
# code reads — but nothing an install or a deployment runs pulls from it.
#
# registry.cn-hangzhou.aliyuncs.com is absent too. It is HAMi's chart default for
# the kube-scheduler sidecar, and budcluster overrides that to
# registry.k8s.io/kube-scheduler, a host already required above.
registries:
- host: registry.bud.studio
pulledBy: every first-party image, and the OCI charts
Expand All @@ -93,7 +97,7 @@ registries:
pulledBy: Dapr, CloudNativePG, OpenTelemetry operator, valkey-operator, budecosystem scaler
requirement: required
- host: registry.k8s.io
pulledBy: prometheus-adapter, NFD, kube-state-metrics, CSI sidecars
pulledBy: prometheus-adapter, NFD, kube-state-metrics, CSI sidecars, HAMi's kube-scheduler sidecar
requirement: required
- host: ecr-public.aws.com
pulledBy: ArgoCD's bundled Redis (docker/library/redis), HAProxy in HA mode
Expand All @@ -106,10 +110,6 @@ registries:
pulledBy: NVIDIA GPU operator and dcgm-exporter
requirement: conditional
feature: gpu
- host: registry.cn-hangzhou.aliyuncs.com
pulledBy: HAMi's google_containers/kube-scheduler sidecar
requirement: conditional
feature: gpu
- host: sandbox-registry.cn-zhangjiakou.cr.aliyuncs.com
pulledBy: OpenSandbox controller, server, egress, image-committer
requirement: conditional
Expand Down
Loading