From 280662f42e5990666bdad923483f68f891b0e5b5 Mon Sep 17 00:00:00 2001 From: michael Date: Fri, 2 Oct 2026 22:29:38 +0300 Subject: [PATCH 01/16] csi-driver: replication RPCs act on the chain's last LOCAL member, not its end A volume's PV keeps the handle it was created with across fail-overs, and the chain of replication relationships behind that handle alternates between the sites: every fail-over adds a hop to the other side. resolveToLocalReplica walked the chain to its active end for every csi-addons Replication RPC. After an unplanned fail-over A->B, Ramen makes the old primary on A secondary: DemoteVolume (and DisableVolumeReplication on VR deletion) on site A resolved to the chain's end -- the NEW primary on site B -- and demoted it (live 2026-10-02, realbed WordPress: the demote fenced the live primary's paths and took demote snapshots of it; the VRG on A never became secondary, so the fail-back never got PeerReady). The driver needs to know which clusters are its own. The operator marks the clusters it manages `local` in the CSI secret's entries (an entry another site registered for cross-cluster handle resolution is not), and the resolver returns the chain's last member on a local cluster. A secret that marks no cluster local -- an operator predating the flag -- keeps the previous behaviour. Co-Authored-By: Claude Fable 5.1 --- csi-driver/internal/clusters/clusters.go | 22 ++++++++ .../internal/csi/controller/replication.go | 51 +++++++++++++++++-- .../csi/controller/replication_local_test.go | 50 ++++++++++++++++++ .../cluster/storagecluster_controller.go | 9 ++++ 4 files changed, 128 insertions(+), 4 deletions(-) create mode 100644 csi-driver/internal/csi/controller/replication_local_test.go diff --git a/csi-driver/internal/clusters/clusters.go b/csi-driver/internal/clusters/clusters.go index a71616866..daadbacf6 100644 --- a/csi-driver/internal/clusters/clusters.go +++ b/csi-driver/internal/clusters/clusters.go @@ -39,6 +39,10 @@ type Config struct { ClusterID string `json:"cluster_id"` ClusterEndpoint string `json:"cluster_endpoint"` ClusterSecret string `json:"cluster_secret"` + // Local marks a cluster of the site this driver runs on (written by the + // site's operator); an entry without it belongs to another site, kept so + // that a failed-over volume's handle still resolves. + Local bool `json:"local,omitempty"` } // Info is the secret file as a whole. @@ -86,6 +90,24 @@ func Load() (Info, error) { return clusters, nil } +// Local returns the ids of the clusters the secret marks local, and whether +// the secret marks any: a secret written by an operator that predates the +// flag marks none, and callers then fall back to treating every cluster as +// local. +func Local() (map[string]bool, bool, error) { + clusters, err := Load() + if err != nil { + return nil, false, err + } + local := map[string]bool{} + for _, cluster := range clusters.Clusters { + if cluster.Local { + local[cluster.ClusterID] = true + } + } + return local, len(local) > 0, nil +} + // List returns the ID of every cluster in the secret. func List() ([]string, error) { clusters, err := Load() diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index 242763056..f81a55ebc 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -93,19 +93,36 @@ func volumeIDFrom(req volumeIDCarrier) string { // combining active_lvol_id with ANOTHER record's cluster is exactly the bug // this walk exists to avoid. An empty ActiveLvolID (backend predating the // field) stops after the first hop, the old single-step behavior. +// +// The chain behind a handle alternates between the sites: every fail-over +// adds a hop to the other side. The volume this driver must act on is the +// chain's last member on a LOCAL cluster (the secret marks the site's own +// clusters, clusters.Local), not the chain's end: after an unplanned +// fail-over A->B, Ramen makes the old primary on A secondary, and the +// chain's end is the NEW primary on B. Resolving to the end demoted -- and +// on VR deletion detached -- the live production volume on the other site +// (2026-10-02, WordPress: the demote fenced the live primary's paths and +// took demote snapshots of it; the fail-back never got PeerReady). A secret +// that marks no cluster local (an operator predating the flag) keeps the +// previous behaviour, the chain's active end. func resolveToLocalReplica( ctx context.Context, h *lvol.Handle, client *atlascp.Client, ) (*lvol.Handle, *atlascp.Client, error) { + local, flagged, err := clusters.Local() + if err != nil { + return nil, nil, err + } + hops := []chainHop{{h: h, client: client}} for range 8 { // one hop per past fail-over; capped far above any real chain rel, err := client.GetVolumeReplicationRelationship(ctx, h.Handle()) if err != nil { if errors.Is(err, errs.ErrNotFound) { - return h, client, nil + break } return nil, nil, err } if !rel.IsSource { - return h, client, nil + break } target := &lvol.Handle{ClusterID: rel.TargetClusterID, PoolRef: rel.TargetPoolID, VolumeID: rel.TargetLvolID} targetClient, err := clusters.ReplicationClient(ctx, target.ClusterID) @@ -113,11 +130,37 @@ func resolveToLocalReplica( return nil, nil, err } h, client = target, targetClient + hops = append(hops, chainHop{h: h, client: client}) if rel.ActiveLvolID == "" || rel.ActiveLvolID == rel.TargetLvolID { - return h, client, nil + break + } + } + pick := chooseReplica(hops, local, flagged) + return pick.h, pick.client, nil +} + +// chainHop is one member of a replication chain, with the client of its +// cluster. +type chainHop struct { + h *lvol.Handle + client *atlascp.Client +} + +// chooseReplica picks the chain member a Replication RPC acts on: the last +// member on a local cluster when the secret marks local clusters (and the +// chain's end when none of the members is local, e.g. a volume that only +// ever lived elsewhere), else the chain's end. +func chooseReplica(hops []chainHop, local map[string]bool, flagged bool) chainHop { + end := hops[len(hops)-1] + if !flagged { + return end + } + for i := len(hops) - 1; i >= 0; i-- { + if local[hops[i].h.ClusterID] { + return hops[i] } } - return h, client, nil + return end } // EnableVolumeReplication attaches the volume to the policy named by the diff --git a/csi-driver/internal/csi/controller/replication_local_test.go b/csi-driver/internal/csi/controller/replication_local_test.go new file mode 100644 index 000000000..2dab4de1e --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_local_test.go @@ -0,0 +1,50 @@ +package controller + +import ( + "testing" + + "github.com/simplyblock/atlas/lvol" +) + +// The chain of 2026-10-02 (realbed, WordPress): created on A (0aea, gone), +// failed over to B (6e83), relocated back to A (80e3), failed over to B +// (e3d4, the live primary). Ramen then made the old primary on A secondary. +func liveChain() []chainHop { + mk := func(cluster, id string) chainHop { return chainHop{h: &lvol.Handle{ClusterID: cluster, PoolRef: "p", VolumeID: id}} } + return []chainHop{mk("A", "0aea"), mk("B", "6e83"), mk("A", "80e3"), mk("B", "e3d4")} +} + +func TestChooseReplicaOnTheOldPrimarysSiteIsTheOldPrimaryNotTheLivePrimary(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{"A": true}, true) + if got.h.VolumeID != "80e3" { + t.Fatalf("site A acts on %s, want 80e3 (its own, superseded primary); e3d4 is the live primary on B", got.h.VolumeID) + } +} + +func TestChooseReplicaOnTheNewPrimarysSiteIsTheLivePrimary(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{"B": true}, true) + if got.h.VolumeID != "e3d4" { + t.Fatalf("site B acts on %s, want e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaWithoutLocalFlagsKeepsTheChainsEnd(t *testing.T) { + got := chooseReplica(liveChain(), nil, false) + if got.h.VolumeID != "e3d4" { + t.Fatalf("unflagged secret: %s, want the chain's end e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaWithNoLocalMemberKeepsTheChainsEnd(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{"C": true}, true) + if got.h.VolumeID != "e3d4" { + t.Fatalf("no member on C: %s, want the chain's end e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaOfAVolumeWithoutARelationshipIsTheVolume(t *testing.T) { + one := liveChain()[:1] + if got := chooseReplica(one, map[string]bool{"B": true}, true); got.h.VolumeID != "0aea" { + t.Fatalf("got %s", got.h.VolumeID) + } +} diff --git a/operator/internal/controllers/cluster/storagecluster_controller.go b/operator/internal/controllers/cluster/storagecluster_controller.go index 6f30bef45..14a963ce3 100644 --- a/operator/internal/controllers/cluster/storagecluster_controller.go +++ b/operator/internal/controllers/cluster/storagecluster_controller.go @@ -201,6 +201,14 @@ type CSIClusterEntry struct { ClusterID string `json:"cluster_id"` ClusterEndpoint string `json:"cluster_endpoint"` ClusterSecret string `json:"cluster_secret"` + // Local marks a cluster this operator manages, i.e. the storage of the + // Kubernetes cluster the driver runs on, as opposed to an entry another + // site registered so that a failed-over volume's handle still resolves. + // The driver's csi-addons Replication RPCs act on the LOCAL member of a + // replication chain: a volume's PV keeps the handle it was created with + // across fail-overs, and the chain of relationships behind it alternates + // between the sites. + Local bool `json:"local,omitempty"` } // +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storageclusters,verbs=get;list;watch;create;update;patch;delete @@ -1126,6 +1134,7 @@ func (r *StorageClusterReconciler) upsertCSICredentials( ClusterID: clusterID, ClusterEndpoint: r.API.Endpoint(ctx), ClusterSecret: clusterSecret, + Local: true, } for i := range creds.Clusters { if creds.Clusters[i].ClusterID == clusterID { From 5106e180ce770c1a9be38d3e68fd96d43d141d49 Mon Sep 17 00:00:00 2001 From: michael Date: Fri, 2 Oct 2026 22:41:53 +0300 Subject: [PATCH 02/16] csi-driver: lint (gofmt, goconst) for the local-replica resolver Co-Authored-By: Claude Fable 5.1 --- .../csi/controller/replication_local_test.go | 26 ++++++++++++------- 1 file changed, 17 insertions(+), 9 deletions(-) diff --git a/csi-driver/internal/csi/controller/replication_local_test.go b/csi-driver/internal/csi/controller/replication_local_test.go index 2dab4de1e..9dda81232 100644 --- a/csi-driver/internal/csi/controller/replication_local_test.go +++ b/csi-driver/internal/csi/controller/replication_local_test.go @@ -9,42 +9,50 @@ import ( // The chain of 2026-10-02 (realbed, WordPress): created on A (0aea, gone), // failed over to B (6e83), relocated back to A (80e3), failed over to B // (e3d4, the live primary). Ramen then made the old primary on A secondary. +const ( + siteA, siteB = "A", "B" + oldPrimaryOnA = "80e3" + livePrimaryOnB = "e3d4" +) + func liveChain() []chainHop { - mk := func(cluster, id string) chainHop { return chainHop{h: &lvol.Handle{ClusterID: cluster, PoolRef: "p", VolumeID: id}} } - return []chainHop{mk("A", "0aea"), mk("B", "6e83"), mk("A", "80e3"), mk("B", "e3d4")} + mk := func(cluster, id string) chainHop { + return chainHop{h: &lvol.Handle{ClusterID: cluster, PoolRef: "p", VolumeID: id}} + } + return []chainHop{mk(siteA, "0aea"), mk(siteB, "6e83"), mk(siteA, oldPrimaryOnA), mk(siteB, livePrimaryOnB)} } func TestChooseReplicaOnTheOldPrimarysSiteIsTheOldPrimaryNotTheLivePrimary(t *testing.T) { - got := chooseReplica(liveChain(), map[string]bool{"A": true}, true) - if got.h.VolumeID != "80e3" { + got := chooseReplica(liveChain(), map[string]bool{siteA: true}, true) + if got.h.VolumeID != oldPrimaryOnA { t.Fatalf("site A acts on %s, want 80e3 (its own, superseded primary); e3d4 is the live primary on B", got.h.VolumeID) } } func TestChooseReplicaOnTheNewPrimarysSiteIsTheLivePrimary(t *testing.T) { - got := chooseReplica(liveChain(), map[string]bool{"B": true}, true) - if got.h.VolumeID != "e3d4" { + got := chooseReplica(liveChain(), map[string]bool{siteB: true}, true) + if got.h.VolumeID != livePrimaryOnB { t.Fatalf("site B acts on %s, want e3d4", got.h.VolumeID) } } func TestChooseReplicaWithoutLocalFlagsKeepsTheChainsEnd(t *testing.T) { got := chooseReplica(liveChain(), nil, false) - if got.h.VolumeID != "e3d4" { + if got.h.VolumeID != livePrimaryOnB { t.Fatalf("unflagged secret: %s, want the chain's end e3d4", got.h.VolumeID) } } func TestChooseReplicaWithNoLocalMemberKeepsTheChainsEnd(t *testing.T) { got := chooseReplica(liveChain(), map[string]bool{"C": true}, true) - if got.h.VolumeID != "e3d4" { + if got.h.VolumeID != livePrimaryOnB { t.Fatalf("no member on C: %s, want the chain's end e3d4", got.h.VolumeID) } } func TestChooseReplicaOfAVolumeWithoutARelationshipIsTheVolume(t *testing.T) { one := liveChain()[:1] - if got := chooseReplica(one, map[string]bool{"B": true}, true); got.h.VolumeID != "0aea" { + if got := chooseReplica(one, map[string]bool{siteB: true}, true); got.h.VolumeID != "0aea" { t.Fatalf("got %s", got.h.VolumeID) } } From 18c0824117bd224d1352fe576adaa604dd99fe70 Mon Sep 17 00:00:00 2001 From: michael Date: Fri, 2 Oct 2026 23:03:35 +0300 Subject: [PATCH 03/16] csi-driver: demoting or detaching a reaped chain member succeeds The old primary of an unplanned fail-over is reaped by the control plane once its fail-over completed; when the site returns Ramen still demotes it and deletes its VolumeReplication. Resolved to that member, the demote and the detach got a 404 and the VR stayed Degraded (live 2026-10-02, site A, 80e3e748). A 404 on a member of a chain the backend records is nothing left to demote or detach: success. A 404 on a handle without any relationship stays NotFound. Co-Authored-By: Claude Fable 5.1 --- .../internal/csi/controller/replication.go | 45 +++++++++++++++---- .../csi/controller/replication_local_test.go | 30 +++++++++++++ 2 files changed, 67 insertions(+), 8 deletions(-) diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index f81a55ebc..5a18a2710 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -108,10 +108,23 @@ func volumeIDFrom(req volumeIDCarrier) string { func resolveToLocalReplica( ctx context.Context, h *lvol.Handle, client *atlascp.Client, ) (*lvol.Handle, *atlascp.Client, error) { + h, client, _, err := resolveReplica(ctx, h, client) + return h, client, err +} + +// resolveReplica is resolveToLocalReplica reporting also whether h has a +// replication relationship at all (known): the member it resolves to is then +// one of a chain the backend records, and a volume of that chain that no +// longer exists is a superseded, reaped old primary -- nothing left to demote +// or detach -- rather than an unknown handle. +func resolveReplica( + ctx context.Context, h *lvol.Handle, client *atlascp.Client, +) (*lvol.Handle, *atlascp.Client, bool, error) { local, flagged, err := clusters.Local() if err != nil { - return nil, nil, err + return nil, nil, false, err } + known := false hops := []chainHop{{h: h, client: client}} for range 8 { // one hop per past fail-over; capped far above any real chain rel, err := client.GetVolumeReplicationRelationship(ctx, h.Handle()) @@ -119,15 +132,16 @@ func resolveToLocalReplica( if errors.Is(err, errs.ErrNotFound) { break } - return nil, nil, err + return nil, nil, false, err } + known = true if !rel.IsSource { break } target := &lvol.Handle{ClusterID: rel.TargetClusterID, PoolRef: rel.TargetPoolID, VolumeID: rel.TargetLvolID} targetClient, err := clusters.ReplicationClient(ctx, target.ClusterID) if err != nil { - return nil, nil, err + return nil, nil, false, err } h, client = target, targetClient hops = append(hops, chainHop{h: h, client: client}) @@ -136,7 +150,16 @@ func resolveToLocalReplica( } } pick := chooseReplica(hops, local, flagged) - return pick.h, pick.client, nil + return pick.h, pick.client, known, nil +} + +// reapedChainMember is whether a Replication verb on a resolved chain member +// found the volume gone (404): a superseded old primary the control plane +// has reaped after its fail-over completed (deferred removal; live +// 2026-10-02 on site A). Demoting or detaching it is a no-op that succeeds; +// a 404 on a handle with no relationship stays NotFound. +func reapedChainMember(known bool, ce classifiedError) bool { + return known && status.Code(ce) == codes.NotFound } // chainHop is one member of a replication chain, with the client of its @@ -267,12 +290,14 @@ func (cs *Server) DisableVolumeReplication( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + h, client, known, err := resolveReplica(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } if err := client.DisableVolumeReplication(ctx, h.Handle()); err != nil { - return nil, classifyDisableVolumeReplicationError(err) + if ce := classifyDisableVolumeReplicationError(err); !reapedChainMember(known, ce) { + return nil, ce + } } return &replication.DisableVolumeReplicationResponse{}, nil } @@ -418,13 +443,17 @@ func (cs *Server) DemoteVolume( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + h, client, known, err := resolveReplica(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } done, err := client.DemoteVolume(ctx, h.Handle()) if err != nil { - return nil, classifyDemoteVolumeError(err) + ce := classifyDemoteVolumeError(err) + if !reapedChainMember(known, ce) { + return nil, ce + } + done = true } if !done { return nil, status.Error(codes.Aborted, "demote is still converging") diff --git a/csi-driver/internal/csi/controller/replication_local_test.go b/csi-driver/internal/csi/controller/replication_local_test.go index 9dda81232..96fbe9e60 100644 --- a/csi-driver/internal/csi/controller/replication_local_test.go +++ b/csi-driver/internal/csi/controller/replication_local_test.go @@ -1,8 +1,10 @@ package controller import ( + "context" "testing" + "github.com/csi-addons/spec/lib/go/replication" "github.com/simplyblock/atlas/lvol" ) @@ -56,3 +58,31 @@ func TestChooseReplicaOfAVolumeWithoutARelationshipIsTheVolume(t *testing.T) { t.Fatalf("got %s", got.h.VolumeID) } } + +// The old primary of an unplanned fail-over is reaped by the control plane +// once its fail-over completed; Ramen still demotes it (and deletes its VR) +// when the site returns. Nothing is left to demote: success, not NotFound +// (live 2026-10-02: the VR on site A stayed Degraded on a 404). +func TestDemoteAndDisableOfAReapedChainMemberSucceed(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + gone := "99999999-aaaa-bbbb-cccc-dddddddddddd" + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": gone, + "active_lvol_id": gone, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + if _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{VolumeId: testReplVolID}); err != nil { + t.Fatalf("demote of a reaped chain member: %v", err) + } + if _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{VolumeId: testReplVolID}); err != nil { + t.Fatalf("disable of a reaped chain member: %v", err) + } +} From 9ee0e79de87bef6739fe448d3a5303684dafd1d4 Mon Sep 17 00:00:00 2001 From: michael Date: Fri, 2 Oct 2026 23:26:10 +0300 Subject: [PATCH 04/16] csi-driver: Resync and the status read of a reaped local member go to the chain's active end After an unplanned fail-over the recovered site's old primary is reaped once its fail-over completed. Ramen still resyncs that site's secondary VR (and reads its status). Addressed to the reaped member both got a 404 and the VR stayed Degraded (live 2026-10-02, site A). sbcli's replication_failback is addressed to the failed-over clone and, given the original source cluster, re-aims the clone's replication at that site's node so only the delta ships: Resync falls back to the chain's active end with the local cluster as the fail-back source, and the status read follows. Co-Authored-By: Claude Fable 5.1 --- .../internal/csi/controller/replication.go | 88 +++++++++++++++++-- .../csi/controller/replication_local_test.go | 14 ++- 2 files changed, 92 insertions(+), 10 deletions(-) diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index 5a18a2710..335774a6c 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -9,6 +9,7 @@ package controller import ( "context" "errors" + "sort" "github.com/csi-addons/spec/lib/go/replication" "google.golang.org/grpc/codes" @@ -120,10 +121,22 @@ func resolveToLocalReplica( func resolveReplica( ctx context.Context, h *lvol.Handle, client *atlascp.Client, ) (*lvol.Handle, *atlascp.Client, bool, error) { + hops, known, err := resolveChain(ctx, h, client) + if err != nil { + return nil, nil, false, err + } local, flagged, err := clusters.Local() if err != nil { return nil, nil, false, err } + pick := chooseReplica(hops, local, flagged) + return pick.h, pick.client, known, nil +} + +// resolveChain walks the replication chain behind h: h itself, then every +// IsSource->target hop to the active end. known is whether h has a +// relationship at all. +func resolveChain(ctx context.Context, h *lvol.Handle, client *atlascp.Client) ([]chainHop, bool, error) { known := false hops := []chainHop{{h: h, client: client}} for range 8 { // one hop per past fail-over; capped far above any real chain @@ -132,7 +145,7 @@ func resolveReplica( if errors.Is(err, errs.ErrNotFound) { break } - return nil, nil, false, err + return nil, false, err } known = true if !rel.IsSource { @@ -141,7 +154,7 @@ func resolveReplica( target := &lvol.Handle{ClusterID: rel.TargetClusterID, PoolRef: rel.TargetPoolID, VolumeID: rel.TargetLvolID} targetClient, err := clusters.ReplicationClient(ctx, target.ClusterID) if err != nil { - return nil, nil, false, err + return nil, false, err } h, client = target, targetClient hops = append(hops, chainHop{h: h, client: client}) @@ -149,8 +162,36 @@ func resolveReplica( break } } - pick := chooseReplica(hops, local, flagged) - return pick.h, pick.client, known, nil + return hops, known, nil +} + +// activeEndFallback is where a Resync or a status read goes when the local +// member of the chain is reaped: the chain's active end -- the live primary +// on the other site -- and, as the cluster to fail back to, the local one. +// sbcli's replication_failback is addressed to the failed-over clone and +// re-aims its replication at the original site's node (the recovered-source +// case: only the delta ships), which is exactly the fail-back of a site that +// lost its primary (live 2026-10-02, site A after the unplanned fail-over of +// WordPress: the old primary 80e3e748 was reaped, the clone e3d439ca on B +// holds the data). +func activeEndFallback(hops []chainHop, sourceClusterID string) (chainHop, string) { + end := hops[len(hops)-1] + if local, flagged, err := clusters.Local(); err == nil && flagged { + ids := make([]string, 0, len(local)) + for id := range local { + ids = append(ids, id) + } + sort.Strings(ids) + for _, hop := range hops { + if local[hop.h.ClusterID] { + return end, hop.h.ClusterID + } + } + if len(ids) > 0 { + return end, ids[0] + } + } + return end, sourceClusterID } // reapedChainMember is whether a Replication verb on a resolved chain member @@ -339,13 +380,29 @@ func (cs *Server) GetVolumeReplicationInfo( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + hops, known, err := resolveChain(ctx, h, client) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + local, flagged, err := clusters.Local() if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } + pick := chooseReplica(hops, local, flagged) + h, client = pick.h, pick.client info, err := client.GetVolumeReplicationInfo(ctx, h.Handle()) if err != nil { - return nil, classifyGetVolumeReplicationInfoError(err) + ce := classifyGetVolumeReplicationInfoError(err) + if !reapedChainMember(known, ce) { + return nil, ce + } + // The local member is reaped: the status that matters is the active + // end's, replicating back to this site after a Resync. + end, _ := activeEndFallback(hops, "") + h, client = end.h, end.client + if info, err = client.GetVolumeReplicationInfo(ctx, h.Handle()); err != nil { + return nil, classifyGetVolumeReplicationInfoError(err) + } } resp := &replication.GetVolumeReplicationInfoResponse{} if info.LastReplicatedAt != nil { @@ -500,13 +557,28 @@ func (cs *Server) ResyncVolume( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + hops, known, err := resolveChain(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } + local, flagged, err := clusters.Local() + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + pick := chooseReplica(hops, local, flagged) + h, client = pick.h, pick.client sourceClusterID := req.GetParameters()[sourceClusterIDParam] if err := client.ResyncVolume(ctx, h.Handle(), sourceClusterID); err != nil { - return nil, classifyResyncVolumeError(err) + ce := classifyResyncVolumeError(err) + if !reapedChainMember(known, ce) { + return nil, ce + } + var end chainHop + end, sourceClusterID = activeEndFallback(hops, sourceClusterID) + h, client = end.h, end.client + if err := client.ResyncVolume(ctx, h.Handle(), sourceClusterID); err != nil { + return nil, classifyResyncVolumeError(err) + } } info, err := client.GetVolumeReplicationInfo(ctx, h.Handle()) if err != nil { diff --git a/csi-driver/internal/csi/controller/replication_local_test.go b/csi-driver/internal/csi/controller/replication_local_test.go index 96fbe9e60..83f93c855 100644 --- a/csi-driver/internal/csi/controller/replication_local_test.go +++ b/csi-driver/internal/csi/controller/replication_local_test.go @@ -79,10 +79,20 @@ func TestDemoteAndDisableOfAReapedChainMemberSucceed(t *testing.T) { "active_lvol_id": gone, "target_nqn": "nqn.test", "target_ns_id": 1, } - if _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{VolumeId: testReplVolID}); err != nil { + ctx := context.Background() + if _, err := cs.DemoteVolume(ctx, &replication.DemoteVolumeRequest{VolumeId: testReplVolID}); err != nil { t.Fatalf("demote of a reaped chain member: %v", err) } - if _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{VolumeId: testReplVolID}); err != nil { + disable := &replication.DisableVolumeReplicationRequest{VolumeId: testReplVolID} + if _, err := cs.DisableVolumeReplication(ctx, disable); err != nil { t.Fatalf("disable of a reaped chain member: %v", err) } } + +func TestActiveEndFallbackAimsAtTheChainsEndFromTheLocalSite(t *testing.T) { + // Not flagged in the test secret: the class parameter stays. + end, src := activeEndFallback(liveChain(), "param") + if end.h.VolumeID != livePrimaryOnB || src != "param" { + t.Fatalf("end %s source %s", end.h.VolumeID, src) + } +} From bede3ddee6f5f688d8eb10ac321ce1f8cad6e5a3 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 01:05:14 +0300 Subject: [PATCH 05/16] csi-driver: PromoteVolume addresses the chain's active end, not the local member The control plane's failover endpoint takes the volume that holds the data (the pairing's source) and creates the clone on the replication target. The local member on the promoting site is the volume being replaced -- demoted, or already reaped: on 2026-10-02 the fail-back promote on site A hit the reaped 80e3e748 and 404ed while the live primary on B held the data. Co-Authored-By: Claude Fable 5.1 --- csi-driver/internal/csi/controller/replication.go | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index 335774a6c..ec12e7d71 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -446,10 +446,20 @@ func (cs *Server) PromoteVolume( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + // Promote is addressed to the chain's ACTIVE END, never to the local + // member: the control plane's failover endpoint takes the volume that + // currently holds the data (the source of the pairing) and creates the + // clone on the replication target -- this site. The local member here + // is the volume being replaced: a demoted old primary, or one the + // control plane already reaped (2026-10-02: promote on site A hit the + // reaped 80e3e748 and 404ed while the live primary e3d439ca on B held + // the data). + hops, _, err := resolveChain(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } + end := hops[len(hops)-1] + h, client = end.h, end.client if err := client.PromoteVolume(ctx, h.Handle(), req.GetForce()); err != nil { return nil, classifyPromoteVolumeError(err) } From 2b6e4284488c872b18929c305ff1035381e44bd0 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 01:58:58 +0300 Subject: [PATCH 06/16] csi-driver: Resync and the status read address the chain's active end sbcli's replication_failback takes the volume that holds the data (the failed-over clone) and, given the recovered site as source cluster, re-aims its replication at that site's node. Addressed to the local member -- the demoted or reaped old primary -- it configured nothing for the live clone: after the unplanned fail-over of Gitea the clones on A had no replication and lastGroupSyncTime stayed empty (2026-10-02). The pairing's status is the active end's for the same reason. Demote and Disable keep the local member; Promote already uses the active end. Co-Authored-By: Claude Fable 5.1 --- .../internal/csi/controller/replication.go | 55 +++++++------------ 1 file changed, 19 insertions(+), 36 deletions(-) diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index ec12e7d71..e7b483ecf 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -380,29 +380,19 @@ func (cs *Server) GetVolumeReplicationInfo( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - hops, known, err := resolveChain(ctx, h, client) - if err != nil { - return nil, status.Error(codes.Unavailable, err.Error()) - } - local, flagged, err := clusters.Local() + // The pairing's status is the ACTIVE END's: the volume that holds the + // data and replicates. On the primary site that is the local volume; on + // the secondary site the local member is the demoted or reaped old + // primary, whose status says nothing about the pipe back to this site. + hops, _, err := resolveChain(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - pick := chooseReplica(hops, local, flagged) - h, client = pick.h, pick.client + end := hops[len(hops)-1] + h, client = end.h, end.client info, err := client.GetVolumeReplicationInfo(ctx, h.Handle()) if err != nil { - ce := classifyGetVolumeReplicationInfoError(err) - if !reapedChainMember(known, ce) { - return nil, ce - } - // The local member is reaped: the status that matters is the active - // end's, replicating back to this site after a Resync. - end, _ := activeEndFallback(hops, "") - h, client = end.h, end.client - if info, err = client.GetVolumeReplicationInfo(ctx, h.Handle()); err != nil { - return nil, classifyGetVolumeReplicationInfoError(err) - } + return nil, classifyGetVolumeReplicationInfoError(err) } resp := &replication.GetVolumeReplicationInfoResponse{} if info.LastReplicatedAt != nil { @@ -567,28 +557,21 @@ func (cs *Server) ResyncVolume( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - hops, known, err := resolveChain(ctx, h, client) - if err != nil { - return nil, status.Error(codes.Unavailable, err.Error()) - } - local, flagged, err := clusters.Local() + // Resync is addressed to the chain's ACTIVE END with this site as the + // cluster to fail back to: sbcli's replication_failback takes the volume + // that holds the data (the failed-over clone) and re-aims its replication + // at the recovered site's node, shipping only the delta. The local member + // is the demoted or reaped old primary; re-aiming IT configured nothing + // for the live clone (2026-10-02, Gitea after the unplanned fail-over: + // the clones on A had no replication, lastGroupSyncTime stayed empty). + hops, _, err := resolveChain(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - pick := chooseReplica(hops, local, flagged) - h, client = pick.h, pick.client - sourceClusterID := req.GetParameters()[sourceClusterIDParam] + end, sourceClusterID := activeEndFallback(hops, req.GetParameters()[sourceClusterIDParam]) + h, client = end.h, end.client if err := client.ResyncVolume(ctx, h.Handle(), sourceClusterID); err != nil { - ce := classifyResyncVolumeError(err) - if !reapedChainMember(known, ce) { - return nil, ce - } - var end chainHop - end, sourceClusterID = activeEndFallback(hops, sourceClusterID) - h, client = end.h, end.client - if err := client.ResyncVolume(ctx, h.Handle(), sourceClusterID); err != nil { - return nil, classifyResyncVolumeError(err) - } + return nil, classifyResyncVolumeError(err) } info, err := client.GetVolumeReplicationInfo(ctx, h.Handle()) if err != nil { From 8988179c6238edc5a965b5c9abc32608d73f388f Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 04:27:13 +0300 Subject: [PATCH 07/16] volstack/csi-driver: a plan that names no filesystem mounts what the volume carries A PersistentVolume without fsType -- a static PV adopting an existing volume (the storage operator's test fail-over clone), or one Ramen restored without the field -- gave the node plugin an empty fsType, which it turned into "ext4" before the device was looked at; the filesystem layer then refused every such XFS volume ("the volume carries xfs and this plan asks for ext4"), and the bubble VM never started (realbed 2026-10-03). An empty FsType now expresses no opinion: the layer mounts the filesystem the device carries, formats a blank device as DefaultFsType (ext4), records what it found as its params, and the node plugin remembers that type for the volume. A named filesystem keeps refusing a device carrying another. Co-Authored-By: Claude Fable 5.1 --- atlas-lib/volstack/layers/filesystem.go | 64 ++++++++++++++++++-- atlas-lib/volstack/layers/filesystem_test.go | 38 ++++++++++++ atlas-lib/volstack/plans/node.go | 1 + atlas-lib/volstack/plans/plans.go | 7 ++- csi-driver/internal/csi/node/plan.go | 7 ++- csi-driver/internal/csi/node/stage.go | 34 +++++++++-- 6 files changed, 139 insertions(+), 12 deletions(-) diff --git a/atlas-lib/volstack/layers/filesystem.go b/atlas-lib/volstack/layers/filesystem.go index 7d799c903..10d8a472f 100644 --- a/atlas-lib/volstack/layers/filesystem.go +++ b/atlas-lib/volstack/layers/filesystem.go @@ -63,8 +63,19 @@ type FilesystemConfig struct { // formatted as, and it is also the only filesystem the layer will mount: a // device carrying another is refused, because neither reformatting it nor // serving what is on it is safe. + // + // Empty, the plan expresses no opinion: a device carrying a filesystem is + // mounted as what it carries, and a blank one is formatted as DefaultFsType. + // That is what a PersistentVolume without fsType means -- a static PV that + // adopts an existing volume (a test fail-over's clone, 2026-10-03), or one + // Ramen restored without the field -- and turning it into "ext4" before the + // device was looked at made the layer refuse every such XFS volume. FsType string + // DefaultFsType is what a blank device is formatted as when FsType names + // nothing. Empty means ext4. + DefaultFsType string + // StagingPath is where the filesystem is mounted. StagingPath string @@ -107,6 +118,43 @@ type FilesystemConfig struct { // the volume is, and refuses every other device. type Filesystem struct { cfg FilesystemConfig + // detected is the filesystem found on the device when the plan named none. + detected string +} + +// effective is the filesystem this layer acts with: the one the plan named, +// else the one the device carries, else the default for a blank device. +func (f *Filesystem) effective(reading blockdev.Reading) string { + if f.cfg.FsType != "" { + return f.cfg.FsType + } + if reading.Content == blockdev.ContentFilesystem && reading.Type != "" { + f.detected = reading.Type + return reading.Type + } + if f.detected != "" { + return f.detected + } + return f.defaultFsType() +} + +// known is the filesystem this layer stands for once it has acted: named, +// detected, or the default it formats with. +func (f *Filesystem) known() string { + if f.cfg.FsType != "" { + return f.cfg.FsType + } + if f.detected != "" { + return f.detected + } + return f.defaultFsType() +} + +func (f *Filesystem) defaultFsType() string { + if f.cfg.DefaultFsType != "" { + return f.cfg.DefaultFsType + } + return "ext4" } // NewFilesystem returns the filesystem layer for one volume. @@ -206,11 +254,12 @@ func (f *Filesystem) Ensure(ctx context.Context, below volstack.Artifact) (volst // decide the state and again to act on it. The reading itself is not needed // past that, because the filesystem to act on is the one the plan named and // observe has already refused every device carrying another. - state, _, own, err := f.observe(ctx, below) + state, reading, own, err := f.observe(ctx, below) if err != nil { return volstack.Artifact{}, err } if state == volstack.StateReady { + f.effective(reading) return own, nil } @@ -219,7 +268,7 @@ func (f *Filesystem) Ensure(ctx context.Context, below volstack.Artifact) (volst // disagreement, which is the point: the only two ways to reconcile one are to // reformat, which destroys the volume, and to serve the other filesystem, // which hides the misconfiguration until something else acts on it. - fsType := f.cfg.FsType + fsType := f.effective(reading) if state == volstack.StateAbsent { if err := f.cfg.Ops.Format(ctx, dev.Path, fsType, f.formatOptions(below)); err != nil { return volstack.Artifact{}, fmt.Errorf("filesystem: format %s as %s: %w", dev.Path, fsType, err) @@ -321,7 +370,7 @@ func (f *Filesystem) Heal(ctx context.Context, below, _ volstack.Artifact) error return err } - if err := f.cfg.Ops.Mount(ctx, dev.Path, f.cfg.StagingPath, f.cfg.FsType, f.mountFlags()); err != nil { + if err := f.cfg.Ops.Mount(ctx, dev.Path, f.cfg.StagingPath, f.effective(reading), f.mountFlags()); err != nil { return fmt.Errorf("filesystem: remount %s at %s: %w", dev.Path, f.cfg.StagingPath, err) } return nil @@ -379,7 +428,7 @@ type FilesystemParams struct { // recorded is the one the volume asked for, and a teardown needs no more than // that: what is actually on the device is read from the device. func (f *Filesystem) Params() any { - return FilesystemParams{FsType: f.cfg.FsType} + return FilesystemParams{FsType: f.known()} } // agrees reports whether the filesystem on the device is the one the plan asked @@ -436,13 +485,16 @@ func (f *Filesystem) blank( switch { case prior == "": return volstack.StateAbsent, reading, volstack.Artifact{}, nil - case prior != f.cfg.FsType: + case f.cfg.FsType != "" && prior != f.cfg.FsType: return volstack.StateAbsent, reading, volstack.Artifact{}, fmt.Errorf( "filesystem: refusing to stage %s, which is recorded as carrying %s where the plan "+ "asks for %s: reformatting would destroy the volume, and mounting it as %s would "+ "serve a filesystem the plan does not declare", deviceOf(below), prior, f.cfg.FsType, prior) default: + if f.cfg.FsType == "" { + f.detected = prior + } // Recorded as formatted while nothing was found on it: the reading is a // failed probe rather than an empty device, so the filesystem is treated as // present and unmounted. Mounting it is the honest next step, and a mount @@ -486,7 +538,7 @@ func (f *Filesystem) mountFlags() []string { // asked for. That is also the only one the layer acts on, since a device // carrying another is refused rather than reconciled. func (f *Filesystem) strategy() FilesystemLayerStrategy { - return FilesystemStrategyFor(f.cfg.FsType) + return FilesystemStrategyFor(f.known()) } // deviceOf names the device below for an error message, without asserting there diff --git a/atlas-lib/volstack/layers/filesystem_test.go b/atlas-lib/volstack/layers/filesystem_test.go index aa107980b..742791b16 100644 --- a/atlas-lib/volstack/layers/filesystem_test.go +++ b/atlas-lib/volstack/layers/filesystem_test.go @@ -592,3 +592,41 @@ func TestReleaseForcesWhenAPlainUnmountRefuses(t *testing.T) { t.Fatal("a plain unmount refused and the release did not fall back to its force path") } } + +// A plan that names no filesystem -- a static PV without fsType, such as a +// test fail-over's clone or a PV Ramen restored without the field -- mounts +// what the device carries instead of refusing it for not being ext4 +// (2026-10-03), and formats a blank device as the default. +func TestAPlanNamingNoFilesystemMountsWhatTheDeviceCarries(t *testing.T) { + fs := newFakeFS() + l := newFSAsking(t, fs, "", blockdev.Reading{Content: blockdev.ContentFilesystem, Type: "xfs"}, nil) + if _, err := l.Ensure(context.Background(), belowArtifact()); err != nil { + t.Fatal(err) + } + if len(fs.formatted) != 0 { + t.Fatalf("formatted a device that carries a filesystem: %+v", fs.formatted) + } + if len(fs.mounted) != 1 || fs.mounted[0].fsType != "xfs" { + t.Fatalf("mounted %+v, want once as xfs", fs.mounted) + } + if p, _ := l.Params().(FilesystemParams); p.FsType != "xfs" { + t.Fatalf("params %+v, want the detected xfs recorded", p) + } +} + +func TestAPlanNamingNoFilesystemFormatsABlankDeviceAsTheDefault(t *testing.T) { + fs := newFakeFS() + l := NewFilesystem(FilesystemConfig{ + FsType: "", DefaultFsType: "ext4", StagingPath: stagingPath, Ops: fs, + Content: fakeReader{reading: blockdev.Reading{Content: blockdev.ContentBlank}}, + }) + if _, err := l.Ensure(context.Background(), belowArtifact()); err != nil { + t.Fatal(err) + } + if len(fs.formatted) != 1 || fs.formatted[0].fsType != "ext4" { + t.Fatalf("formatted %+v, want once as ext4", fs.formatted) + } + if len(fs.mounted) != 1 || fs.mounted[0].fsType != "ext4" { + t.Fatalf("mounted %+v, want once as ext4", fs.mounted) + } +} diff --git a/atlas-lib/volstack/plans/node.go b/atlas-lib/volstack/plans/node.go index 5aa5c02d8..f31045d40 100644 --- a/atlas-lib/volstack/plans/node.go +++ b/atlas-lib/volstack/plans/node.go @@ -109,6 +109,7 @@ func (n *Node) fabric(connection lvol.Connection) volstack.Layer { func (n *Node) filesystem(volume Volume) volstack.Layer { return layers.NewFilesystem(layers.FilesystemConfig{ FsType: volume.FsType, + DefaultFsType: volume.DefaultFsType, StagingPath: volume.StagingPath, MountFlags: volume.MountFlags, FormatOptions: volume.FormatOptions, diff --git a/atlas-lib/volstack/plans/plans.go b/atlas-lib/volstack/plans/plans.go index c514dae4a..8395a160f 100644 --- a/atlas-lib/volstack/plans/plans.go +++ b/atlas-lib/volstack/plans/plans.go @@ -52,9 +52,14 @@ type Volume struct { // FsType is the filesystem this volume is. It decides what a blank device is // formatted as, and it is also the only filesystem that will be mounted: a - // device carrying another is refused. + // device carrying another is refused. Empty: whatever the device carries, + // and DefaultFsType for a blank one. FsType string + // DefaultFsType is what a blank device is formatted as when FsType names + // nothing. + DefaultFsType string + // MountFlags are the flags the volume asked for, ahead of the ones the // filesystem layer derives from the filesystem itself. MountFlags []string diff --git a/csi-driver/internal/csi/node/plan.go b/csi-driver/internal/csi/node/plan.go index e0dcaa747..aa86c704d 100644 --- a/csi-driver/internal/csi/node/plan.go +++ b/csi-driver/internal/csi/node/plan.go @@ -159,6 +159,10 @@ func stackVolume( vc map[string]string, volCap *csi.VolumeCapability, ) plans.Volume { + // Named by the record or the capability; empty otherwise, so the layer + // mounts what the device carries (a static PV without fsType: a test + // fail-over's clone, a PV Ramen restored without the field) and formats + // a blank device as ext4. fsType := stagedFsType(vc, volCap) return plans.Volume{ UUID: deviceLvolID(vc), @@ -167,8 +171,9 @@ func stackVolume( PVCName: vc[csicommon.CSIStorageNameKey], StagingPath: stagingPath, FsType: fsType, + DefaultFsType: "ext4", MountFlags: volumeMountFlags(volCap), - FormatOptions: mount.FormatOptions(fsType, vc, wantsVDO(vc)), + FormatOptions: mount.FormatOptions(fsTypeOrDefault(volCap), vc, wantsVDO(vc)), ReservedBlocksPercent: vc["tune2fs_reserved_blocks"], Encrypted: boolFromContext(vc[csicommon.ParamEncryption]), } diff --git a/csi-driver/internal/csi/node/stage.go b/csi-driver/internal/csi/node/stage.go index 2e52d0cd1..004fdb243 100644 --- a/csi-driver/internal/csi/node/stage.go +++ b/csi-driver/internal/csi/node/stage.go @@ -102,7 +102,7 @@ func (ns *Server) NodeStageVolume( return nil, status.Error(codes.Internal, err.Error()) } - ns.rememberStagedVolume(ctx, volumeID, vc, artifact, req.GetVolumeCapability()) + ns.rememberStagedVolume(ctx, volumeID, vc, artifact, plan, req.GetVolumeCapability()) // The CSI spec passes VolumeContext to this RPC and to nothing after it, so // what the later RPCs need is written beside the staging path. @@ -619,6 +619,7 @@ func (ns *Server) rememberStagedVolume( volumeID string, vc map[string]string, artifact volstack.Artifact, + plan volstack.Plan, volCap *csi.VolumeCapability, ) { if device, ok := artifact.Device(); ok { @@ -636,10 +637,33 @@ func (ns *Server) rememberStagedVolume( // The device carries this filesystem, because the layer either put it there // or refused to stage a device carrying another. fsType := stagedFsType(vc, volCap) + if fsType == "" { + // Nobody named it: the layer mounted what the device carries, or + // formatted a blank device as its default, and knows which. + fsType = planFsType(plan) + } + if fsType == "" { + return + } vc[stagedFsTypeKey] = fsType ns.recordOnDiskFilesystem(ctx, volumeID, vc, fsType) } +// planFsType is the filesystem the plan's filesystem layer stands for after +// it acted (layers.FilesystemParams), "" when the plan has none. +func planFsType(plan volstack.Plan) string { + for _, l := range plan { + recorded, ok := l.(interface{ Params() any }) + if !ok { + continue + } + if p, ok := recorded.Params().(layers.FilesystemParams); ok { + return p.FsType + } + } + return "" +} + // priorFormat is what the volume is recorded as carrying, for the layer that // has to decide whether a device reading blank is empty or merely unreadable. // @@ -704,13 +728,15 @@ func (ns *Server) volumeIsBeingDeleted(ctx context.Context, volumeID string) boo } // stagedFsType returns the filesystem a volume was staged with: the one -// recorded at stage time when it is there, and otherwise the one the volume -// capability asks for, which is all a volume staged by an older driver has. +// recorded at stage time when it is there, else the one the volume capability +// asks for, else "" -- no opinion, which lets the filesystem layer mount what +// the device carries instead of refusing an XFS volume for not being the ext4 +// nobody asked for (a static PV without fsType, 2026-10-03). func stagedFsType(volumeContext map[string]string, volCap *csi.VolumeCapability) string { if fsType := strings.TrimSpace(volumeContext[stagedFsTypeKey]); fsType != "" { return fsType } - return fsTypeOrDefault(volCap) + return volCap.GetMount().GetFsType() } // fsTypeOrDefault returns the requested filesystem type, defaulting to ext4. From 599a7415e357895a02b0701b6fc35bcb933c0228 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 05:21:04 +0300 Subject: [PATCH 08/16] operator: a test fail-over's bubble PV and PVC carry the source's volumeMode A VM's disk is a Block claim. The drill's bubble PV and PVC omitted the mode, defaulted to Filesystem, and the kubelet asked the node plugin to mount the raw guest disk (mount: exit status 32); the bubble VM never started (realbed 2026-10-03). The source PV's volumeMode is recorded on the clone and set on both objects. Co-Authored-By: Claude Fable 5.1 --- .../storage.simplyblock.io_testfailovers.yaml | 7 +++++++ operator/api/v1alpha2/testfailover_types.go | 6 ++++++ .../storage.simplyblock.io_testfailovers.yaml | 7 +++++++ .../controller/testfailover_controller.go | 17 +++++++++++++++++ .../storage.simplyblock.io_testfailovers.yaml | 7 +++++++ 5 files changed, 44 insertions(+) diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml index d68cf6ad7..1cca0647d 100644 --- a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml @@ -185,6 +185,13 @@ spec: empty fsType makes the node plugin default to ext4 and refuse to mount an XFS volume. type: string + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string sourceHandle: description: SourceHandle is the source volume's backend handle, read from its PV. diff --git a/operator/api/v1alpha2/testfailover_types.go b/operator/api/v1alpha2/testfailover_types.go index de033a936..cf2e1e852 100644 --- a/operator/api/v1alpha2/testfailover_types.go +++ b/operator/api/v1alpha2/testfailover_types.go @@ -175,6 +175,12 @@ type TestFailoverClone struct { // volume. // +optional SourceFSType string `json:"sourceFSType,omitempty"` + // SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + // carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + // claim that omitted the mode defaulted to Filesystem and the kubelet asked + // the node plugin to mount a raw guest disk (2026-10-03). + // +optional + SourceVolumeMode string `json:"sourceVolumeMode,omitempty"` } // TestFailoverReport is the evidence a drill produces. diff --git a/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml index d68cf6ad7..1cca0647d 100644 --- a/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml +++ b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml @@ -185,6 +185,13 @@ spec: empty fsType makes the node plugin default to ext4 and refuse to mount an XFS volume. type: string + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string sourceHandle: description: SourceHandle is the source volume's backend handle, read from its PV. diff --git a/operator/internal/controller/testfailover_controller.go b/operator/internal/controller/testfailover_controller.go index 53cab8521..3f13340be 100644 --- a/operator/internal/controller/testfailover_controller.go +++ b/operator/internal/controller/testfailover_controller.go @@ -241,6 +241,7 @@ func (r *TestFailoverReconciler) resolveSource(ctx context.Context, tf *simplybl srcAttrs, _, _ := unstructured.NestedStringMap(pv, "spec", "csi", "volumeAttributes") bubbleVC := bubbleVolumeContext(srcAttrs) fsType, _, _ := unstructured.NestedString(pv, "spec", "csi", "fsType") + volumeMode, _, _ := unstructured.NestedString(pv, "spec", "volumeMode") if err := r.transitionTo(ctx, tf, simplyblockv1alpha2.TestFailoverStepResolvingPoint, func(s *simplyblockv1alpha2.TestFailoverStatus) { s.Clones = []simplyblockv1alpha2.TestFailoverClone{{ @@ -248,6 +249,7 @@ func (r *TestFailoverReconciler) resolveSource(ctx context.Context, tf *simplybl SourceHandle: handle, SourceVolumeContext: bubbleVC, SourceFSType: fsType, + SourceVolumeMode: volumeMode, }} s.Message = "resolved the source volume; resolving the recovery point" }); err != nil { @@ -322,6 +324,7 @@ func (r *TestFailoverReconciler) resolveSourceGroup(ctx context.Context, tf *sim srcAttrs, _, _ := unstructured.NestedStringMap(pv, "spec", "csi", "volumeAttributes") bubbleVC := bubbleVolumeContext(srcAttrs) fsType, _, _ := unstructured.NestedString(pv, "spec", "csi", "fsType") + volumeMode, _, _ := unstructured.NestedString(pv, "spec", "volumeMode") clones := make([]simplyblockv1alpha2.TestFailoverClone, 0, len(memberIDs)) for _, id := range memberIDs { @@ -330,6 +333,7 @@ func (r *TestFailoverReconciler) resolveSourceGroup(ctx context.Context, tf *sim SourceRef: v.PVCName, SourceHandle: srcUUID + ":" + v.PoolID + ":" + v.LvolID, SourceFSType: fsType, + SourceVolumeMode: volumeMode, SourceVolumeContext: bubbleVC, SizeBytes: v.Size, }) @@ -998,6 +1002,17 @@ func (r *TestFailoverReconciler) placeBubble(ctx context.Context, tf *simplybloc return ctrl.Result{}, nil } +// volumeModeOf is the bubble PV's and PVC's volumeMode: the source's, so a +// VM's Block disk stays a block device instead of being mounted as a +// filesystem; nil (the default, Filesystem) when the source did not say. +func volumeModeOf(clone simplyblockv1alpha2.TestFailoverClone) *corev1.PersistentVolumeMode { + if clone.SourceVolumeMode == "" { + return nil + } + mode := corev1.PersistentVolumeMode(clone.SourceVolumeMode) + return &mode +} + // bubbleVolumeContextStripKeys are the source PV volumeAttributes that must NOT // be carried onto the bubble PV: they identify the SOURCE volume and its NVMe-oF // target. The node plugin re-resolves the clone's own identity from the clone @@ -1078,6 +1093,7 @@ func (r *TestFailoverReconciler) bubbleManifestWork(tf *simplyblockv1alpha2.Test AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, PersistentVolumeReclaimPolicy: corev1.PersistentVolumeReclaimRetain, StorageClassName: scName, + VolumeMode: volumeModeOf(clone), ClaimRef: &corev1.ObjectReference{ Kind: "PersistentVolumeClaim", APIVersion: "v1", Namespace: ns, Name: pvcName, }, @@ -1099,6 +1115,7 @@ func (r *TestFailoverReconciler) bubbleManifestWork(tf *simplyblockv1alpha2.Test Resources: corev1.VolumeResourceRequirements{Requests: corev1.ResourceList{corev1.ResourceStorage: capacity}}, StorageClassName: &scName, VolumeName: pvName, + VolumeMode: volumeModeOf(clone), }, } for _, obj := range []client.Object{pv, pvc} { diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml index d68cf6ad7..1cca0647d 100644 --- a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml @@ -185,6 +185,13 @@ spec: empty fsType makes the node plugin default to ext4 and refuse to mount an XFS volume. type: string + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string sourceHandle: description: SourceHandle is the source volume's backend handle, read from its PV. From 7d5f38bc45f393bb596f90c25463c6ac54403a40 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 12:27:06 +0300 Subject: [PATCH 09/16] chart: the Control Center on the integrate_csi_addons line The console templates and their values block of feature/control-center-ui (PR #512), copied as they are: self-contained (sbcc.* helpers only), so the chart deploys the UI with the control plane on this line too. One addition: controlCenter.service.nodePort pins the node port of a NodePort Service. Co-Authored-By: Claude Fable 5.1 --- .../templates/_control_center_helpers.tpl | 61 ++++++ .../templates/control-center-mock.yaml | 96 +++++++++ .../control-center-networkpolicy.yaml | 76 +++++++ .../templates/control-center-rbac.yaml | 176 ++++++++++++++++ .../templates/control-center.yaml | 196 ++++++++++++++++++ .../charts/simplyblock-operator/values.yaml | 152 ++++++++++++++ 6 files changed, 757 insertions(+) create mode 100644 helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl create mode 100644 helm-charts/charts/simplyblock-operator/templates/control-center-mock.yaml create mode 100644 helm-charts/charts/simplyblock-operator/templates/control-center-networkpolicy.yaml create mode 100644 helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml create mode 100644 helm-charts/charts/simplyblock-operator/templates/control-center.yaml diff --git a/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl b/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl new file mode 100644 index 000000000..97a31717e --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl @@ -0,0 +1,61 @@ +{{/* + Control Center helpers — self-contained on purpose. + + The console ships as a fragment (see control-center/ in the monorepo) and + deliberately does NOT call this chart's other helpers: everything is + namespaced under `sbcc.` so it cannot collide, and the fragment renders + unchanged if it is ever lifted out of this chart again. +*/}} + +{{- define "sbcc.name" -}} +{{- default "control-center" .Values.controlCenter.nameOverride | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{- define "sbcc.fullname" -}} +{{- if .Values.controlCenter.fullnameOverride -}} +{{- .Values.controlCenter.fullnameOverride | trunc 63 | trimSuffix "-" -}} +{{- else -}} +{{- printf "%s-%s" .Release.Name (include "sbcc.name" .) | trunc 63 | trimSuffix "-" -}} +{{- end -}} +{{- end -}} + +{{- define "sbcc.selectorLabels" -}} +app.kubernetes.io/name: {{ include "sbcc.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +{{- end -}} + +{{- define "sbcc.labels" -}} +{{ include "sbcc.selectorLabels" . }} +app.kubernetes.io/component: ui +app.kubernetes.io/part-of: simplyblock +app.kubernetes.io/managed-by: {{ .Release.Service }} +{{- with .Chart }} +helm.sh/chart: {{ printf "%s-%s" .Name .Version | replace "+" "_" | trunc 63 | trimSuffix "-" }} +{{- end }} +{{- end -}} + +{{/* Image reference. The registry falls back to the public one and the tag to + the chart's appVersion — pin controlCenter.image.tag for production, this + chart's appVersion is a floating tag. */}} +{{- define "sbcc.image" -}} +{{- $i := .Values.controlCenter.image -}} +{{- $reg := $i.registry | default "quay.io" -}} +{{- $tag := $i.tag | default (.Chart.AppVersion | default "latest") -}} +{{- printf "%s/%s:%s" $reg $i.repository $tag -}} +{{- end -}} + +{{/* Default upstream URLs. This chart names its objects statically + (simplyblock-operator, simplyblock-prometheus, …) rather than deriving + them from the release, so the fallbacks here are static too. */}} +{{- define "sbcc.operatorUrl" -}} +{{- .Values.controlCenter.operatorUrl | default "http://simplyblock-operator:8080" -}} +{{- end -}} + +{{- define "sbcc.prometheusUrl" -}} +{{- if .Values.controlCenter.prometheusUrl -}} +{{- .Values.controlCenter.prometheusUrl -}} +{{- else -}} +{{- $p := ((.Values.prometheus).simplyblock) | default dict -}} +{{- printf "http://%s:%v" ($p.prometheusURL | default "simplyblock-prometheus") ($p.prometheusPORT | default 9090) -}} +{{- end -}} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center-mock.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center-mock.yaml new file mode 100644 index 000000000..8cc406a13 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center-mock.yaml @@ -0,0 +1,96 @@ +{{/* + sb-mock — the Control Center's test backend (control-center/mock). + + Deployed only for testing and demos: it impersonates the Kubernetes API, + the operator API and Prometheus with generated data, and the console's + Deployment points its proxy here when controlCenter.mock.enabled is true. + Writes persist into the mock's memory and perform no real change; nothing + in this pod touches the actual cluster, and it needs no RBAC at all. +*/}} +{{- if and .Values.controlCenter.enabled .Values.controlCenter.mock.enabled }} +{{- $cc := .Values.controlCenter }} +{{- $mock := $cc.mock }} +{{- $name := printf "%s-mock" (include "sbcc.fullname" .) }} +{{- $reg := $mock.image.registry | default "quay.io" }} +{{- $tag := $mock.image.tag | default (.Chart.AppVersion | default "latest") }} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} + app.kubernetes.io/component: ui-mock +spec: + # one replica on purpose: the world lives in this pod's memory + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: {{ include "sbcc.name" . }}-mock + app.kubernetes.io/instance: {{ .Release.Name }} + template: + metadata: + labels: + app.kubernetes.io/name: {{ include "sbcc.name" . }}-mock + app.kubernetes.io/instance: {{ .Release.Name }} + spec: + automountServiceAccountToken: false + securityContext: + runAsNonRoot: true + runAsUser: 65532 + runAsGroup: 65532 + seccompProfile: + type: RuntimeDefault + {{- with $cc.nodeSelector }} + nodeSelector: {{- toYaml . | nindent 8 }} + {{- end }} + {{- with $cc.tolerations }} + tolerations: {{- toYaml . | nindent 8 }} + {{- end }} + containers: + - name: mock + image: "{{ $reg }}/{{ $mock.image.repository }}:{{ $tag }}" + imagePullPolicy: {{ $mock.image.pullPolicy }} + args: + - --listen=:8080 + - --dataset={{ $mock.dataset }} + {{- if $mock.seed }} + - --seed={{ $mock.seed }} + {{- end }} + - --sim-interval={{ $mock.simInterval }} + - --fail-rate={{ $mock.failRate }} + - --namespace={{ .Release.Namespace }} + ports: + - name: http + containerPort: 8080 + resources: {{- toYaml $mock.resources | nindent 12 }} + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + readinessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 5 + livenessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 20 +--- +apiVersion: v1 +kind: Service +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} + app.kubernetes.io/component: ui-mock +spec: + type: ClusterIP + ports: + - name: http + port: 8080 + targetPort: http + selector: + app.kubernetes.io/name: {{ include "sbcc.name" . }}-mock + app.kubernetes.io/instance: {{ .Release.Name }} +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center-networkpolicy.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center-networkpolicy.yaml new file mode 100644 index 000000000..969fabc3e --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center-networkpolicy.yaml @@ -0,0 +1,76 @@ +{{/* + NetworkPolicy for the Control Center. + + The console has exactly three upstreams — the Kubernetes API, the operator + API and Prometheus — and no route to the internet: React, Babel, the + typefaces and the brand mark are vendored into the image. This caps egress + to those upstreams plus DNS. + + Off by default because the API-server rule is cluster-specific: it ships as + the RFC1918 ranges, and you should narrow controlCenter.networkPolicy + .apiServerCidrs to your API server endpoint (`kubectl get endpoints + kubernetes -n default`) or your service CIDR. A rule with ports but no `to:` + would allow 443 to the whole internet, which is why one is never rendered. +*/}} +{{- if and .Values.controlCenter.enabled .Values.controlCenter.networkPolicy.enabled }} +{{- $cc := .Values.controlCenter }} +{{- $name := include "sbcc.fullname" . }} +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +spec: + podSelector: + matchLabels: + {{- include "sbcc.selectorLabels" . | nindent 6 }} + policyTypes: ["Ingress", "Egress"] + ingress: + # Who may reach the console. Defaults to the release namespace (which + # covers kubectl port-forward); add your ingress controller's namespace + # when the Ingress is enabled. + - from: + {{- range $ns := $cc.networkPolicy.ingressFromNamespaces }} + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: {{ $ns }} + {{- end }} + - podSelector: {} + ports: + - protocol: TCP + port: 8080 + egress: + # The operator API and Prometheus, inside the release namespace. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: {{ .Release.Namespace }} + ports: + - protocol: TCP + port: 8080 + - protocol: TCP + port: 9090 + # The Kubernetes API server. Narrow these CIDRs — see the header comment. + - to: + {{- range $cidr := $cc.networkPolicy.apiServerCidrs }} + - ipBlock: + cidr: {{ $cidr }} + {{- end }} + ports: + - protocol: TCP + port: 443 + - protocol: TCP + port: 6443 + # DNS + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml new file mode 100644 index 000000000..c1bd77089 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml @@ -0,0 +1,176 @@ +{{/* + RBAC for the Control Center. + + In serviceaccount mode this role IS the console's authority, for anyone who + can reach the Service. In passthrough mode the pod needs nothing at all — + Kubernetes enforces each user's own RBAC — so set controlCenter.rbac.create + to false there. + + Kept in step with control-center/deploy/k8s/rbac.yaml; that file is the + plain-manifest equivalent of this one. +*/}} +{{- if and .Values.controlCenter.enabled .Values.controlCenter.rbac.create }} +{{- $name := include "sbcc.fullname" . }} +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: {{ $name }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +rules: + # ---- simplyblock entities: read ---- + - apiGroups: ["storage.simplyblock.io"] + resources: + - storageclusters + - storagenodes + - storagenodesets + - storagedevices + - storagepools + - storagebackups + - backuppolicies + - backuprestores + - backupimports + - controlplanes + - replicationpairs + - replicationpolicies + - replicationslots + - volumemigrations + - tasks + - simplyblockdrivers + - clusterdeploymentconfigs + verbs: ["get", "list", "watch"] + - apiGroups: ["storage.simplyblock.io"] + resources: + - storageclusters/status + - storagenodes/status + - storagepools/status + - replicationpolicies/status + - replicationslots/status + verbs: ["get"] + + # ---- day-2 operations ---- + # An action is not a verb against an entity: it is an Ops object the operator + # reconciles. The console creates one, reads its phase, patches spec.abort to + # unwind it, and deletes it only to abort. + - apiGroups: ["storage.simplyblock.io"] + resources: + - storageclusterops + - storagenodeops + - storagedeviceops + - storagepoolops + - storagebackupops + - controlplaneops + - persistentvolumeops + - operatorops + - replicationops + verbs: ["get", "list", "watch", "create", "patch", "delete"] + + # ---- resources the console authors ---- + - apiGroups: ["storage.simplyblock.io"] + resources: ["replicationpairs", "replicationpolicies"] + verbs: ["create", "update", "patch", "delete"] + - apiGroups: ["storage.simplyblock.io"] + resources: ["storagebackups", "backuppolicies", "backuprestores", "backupimports", "volumemigrations"] + verbs: ["create", "update", "patch", "delete"] + - apiGroups: ["storage.simplyblock.io"] + resources: ["storageclusters", "storagepools", "storagenodesets"] + verbs: ["update", "patch"] + + # ---- core Kubernetes ---- + - apiGroups: [""] + resources: ["nodes", "persistentvolumes", "persistentvolumeclaims", "pods", "events", "namespaces"] + verbs: ["get", "list", "watch"] + - apiGroups: [""] + resources: ["pods/log"] + verbs: ["get"] + # A PVC annotation is the whole membership model for replication and for + # backup policies, so patching one is how a volume is attached. + - apiGroups: [""] + resources: ["persistentvolumeclaims"] + verbs: ["patch", "update"] + - apiGroups: ["storage.k8s.io"] + resources: ["storageclasses"] + verbs: ["get", "list", "watch", "patch"] + - apiGroups: ["snapshot.storage.k8s.io"] + resources: ["volumesnapshots", "volumesnapshotcontents", "volumesnapshotclasses"] + verbs: ["get", "list", "watch"] + + # ---- workloads, read only, for the Ramen recipe editor ---- + - apiGroups: ["apps"] + resources: ["deployments", "statefulsets", "daemonsets", "replicasets"] + verbs: ["get", "list"] + - apiGroups: [""] + resources: ["services", "configmaps"] + verbs: ["get", "list"] + - apiGroups: ["networking.k8s.io"] + resources: ["ingresses"] + verbs: ["get", "list"] + - apiGroups: ["kubevirt.io"] + resources: ["virtualmachines", "virtualmachineinstances"] + verbs: ["get", "list"] + + # ---- Ramen: instances only, never the CRDs ---- + - apiGroups: ["ramendr.openshift.io"] + resources: ["drpolicies", "drclusters", "drclusterconfigs", "volumereplicationgroups"] + verbs: ["get", "list", "watch"] + - apiGroups: ["ramendr.openshift.io"] + resources: ["drplacementcontrols"] + verbs: ["get", "list", "watch", "create", "update", "patch", "delete"] + - apiGroups: ["ramendr.openshift.io"] + resources: ["recipes"] + verbs: ["get", "list", "watch", "create", "update", "patch", "delete"] + - apiGroups: ["cluster.open-cluster-management.io"] + resources: ["managedclusters"] + verbs: ["get", "list"] + + # ---- DR hub (dr-simplyblock): every DR screen is a CR ---- + # In serviceaccount mode this is the console's authority for DR: reads on + # everything, the dr-operator writes, the dr-admin writes on plans/paths/ + # restores, and the override verb the hub's webhook checks before admitting a + # readiness override. Absent CRDs cost nothing — the DR section then reports + # that the hub is not installed. Trim to your policy, or use passthrough. + - apiGroups: ["dr.simplyblock.io"] + resources: ["protectionplans", "drpaths", "protectedapplications", "recoveryplans", "recoveryactions", "testbubbles", "testschedules", "restoreactions", "drconfigs"] + verbs: ["get", "list", "watch"] + - apiGroups: ["dr.simplyblock.io"] + resources: ["protectionplans", "drpaths", "protectedapplications", "recoveryplans", "recoveryactions", "testbubbles", "testschedules", "restoreactions"] + verbs: ["create", "update", "patch", "delete"] + - apiGroups: ["dr.simplyblock.io"] + resources: ["recoveryactions"] + verbs: ["override"] + - apiGroups: ["sitemap.simplyblock.io"] + resources: ["siteprofiles", "dhcpservers"] + verbs: ["get", "list", "watch"] + - apiGroups: ["sitemap.simplyblock.io"] + resources: ["dhcpservers"] + verbs: ["create", "update", "patch", "delete"] + # the console asks the API server what its identity may do, to disable controls + - apiGroups: ["authorization.k8s.io"] + resources: ["selfsubjectaccessreviews", "selfsubjectrulesreviews"] + verbs: ["create"] + - apiGroups: ["authentication.k8s.io"] + resources: ["selfsubjectreviews"] + verbs: ["create"] + + # ---- Helm release view ---- + # Helm stores releases as Secrets. This is the only reason the console reads a + # Secret, and a backup credentialsSecretRef is only ever shown by name. + - apiGroups: [""] + resources: ["secrets"] + verbs: ["get", "list"] +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: {{ $name }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: {{ $name }} +subjects: + - kind: ServiceAccount + name: {{ $name }} + namespace: {{ .Release.Namespace }} +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center.yaml new file mode 100644 index 000000000..3a81f5e3b --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center.yaml @@ -0,0 +1,196 @@ +{{/* + Control Center — the simplyblock web UI. + + Installs with the operator when controlCenter.enabled is true, so the console + arrives as part of the deployment rather than as a second thing to install. + Self-contained on purpose: it calls only its own `sbcc.*` helpers (in + _control_center_helpers.tpl), never this chart's. Sources and image build + live in control-center/ at the repository root. +*/}} +{{- if .Values.controlCenter.enabled }} +{{- $cc := .Values.controlCenter }} +{{- $name := include "sbcc.fullname" . }} +apiVersion: v1 +kind: ServiceAccount +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +{{- with $cc.imagePullSecrets }} +imagePullSecrets: + {{- toYaml . | nindent 2 }} +{{- end }} +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +spec: + replicas: {{ $cc.replicas }} + revisionHistoryLimit: 3 + selector: + matchLabels: + {{- include "sbcc.selectorLabels" . | nindent 6 }} + strategy: + type: RollingUpdate + rollingUpdate: + maxUnavailable: 0 + maxSurge: 1 + template: + metadata: + labels: + {{- include "sbcc.selectorLabels" . | nindent 8 }} + {{- with $cc.podAnnotations }} + annotations: + {{- toYaml . | nindent 8 }} + {{- end }} + spec: + serviceAccountName: {{ $name }} + automountServiceAccountToken: true + securityContext: + runAsNonRoot: true + runAsUser: 101 + runAsGroup: 101 + fsGroup: 101 + seccompProfile: + type: RuntimeDefault + {{- with $cc.nodeSelector }} + nodeSelector: {{- toYaml . | nindent 8 }} + {{- end }} + {{- with $cc.tolerations }} + tolerations: {{- toYaml . | nindent 8 }} + {{- end }} + topologySpreadConstraints: + - maxSkew: 1 + topologyKey: kubernetes.io/hostname + whenUnsatisfiable: ScheduleAnyway + labelSelector: + matchLabels: + {{- include "sbcc.selectorLabels" . | nindent 14 }} + containers: + - name: ui + image: {{ include "sbcc.image" . | quote }} + imagePullPolicy: {{ $cc.image.pullPolicy }} + ports: + - name: http + containerPort: 8080 + env: + - name: SB_NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + - name: SB_MODE + value: {{ $cc.mode | default "full" | quote }} + - name: SB_DR_NAMESPACE + value: {{ $cc.drNamespace | default "ramen-ops" | quote }} + - name: SB_AUTH_MODE + value: {{ $cc.authMode | quote }} + {{- /* With the mock enabled every upstream is the sb-mock service: + the console runs unmodified against generated data. */}} + {{- $mockURL := printf "http://%s-mock:8080" $name }} + - name: SB_K8S_API + value: {{ ternary $mockURL $cc.kubernetesApi $cc.mock.enabled | quote }} + # SNI and Host for the API server. Must match a name on the + # apiserver's certificate — see controlCenter.kubernetesApiHost. + # (Ignored for a plain-http upstream, i.e. in mock mode.) + - name: SB_K8S_HOST + value: {{ $cc.kubernetesApiHost | quote }} + - name: SB_K8S_CA_FILE + value: {{ $cc.kubernetesApiCaFile | quote }} + - name: SB_OPERATOR_URL + value: {{ ternary $mockURL (include "sbcc.operatorUrl" .) $cc.mock.enabled | quote }} + - name: SB_HELM_URL + value: {{ ternary $mockURL ($cc.helmUrl | default (include "sbcc.operatorUrl" .)) $cc.mock.enabled | quote }} + - name: SB_PROMETHEUS_URL + value: {{ ternary $mockURL (include "sbcc.prometheusUrl" .) $cc.mock.enabled | quote }} + - name: SB_MOCK + value: "false" + - name: SB_LISTEN_PORT + value: "8080" + - name: SB_TOKEN_REFRESH_SECONDS + value: {{ $cc.tokenRefreshSeconds | quote }} + resources: {{- toYaml $cc.resources | nindent 12 }} + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + startupProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 2 + failureThreshold: 15 + readinessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 10 + livenessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 20 + volumeMounts: + # readOnlyRootFilesystem: the scratch mount holds the generated + # config.js, the proxied token and nginx's temp files; conf.d is + # writable because the image renders its server block at startup. + - name: nginx-tmp + mountPath: /tmp/nginx + - name: nginx-confd + mountPath: /etc/nginx/conf.d + volumes: + - name: nginx-tmp + emptyDir: {medium: Memory, sizeLimit: 16Mi} + - name: nginx-confd + emptyDir: {medium: Memory, sizeLimit: 1Mi} +--- +apiVersion: v1 +kind: Service +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +spec: + type: {{ $cc.service.type }} + ports: + - name: http + port: {{ $cc.service.port }} + targetPort: http + {{- with $cc.service.nodePort }} + nodePort: {{ . }} + {{- end }} + selector: + {{- include "sbcc.selectorLabels" . | nindent 4 }} +{{- if $cc.ingress.enabled }} +--- +apiVersion: networking.k8s.io/v1 +kind: Ingress +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} + {{- with $cc.ingress.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +spec: + {{- with $cc.ingress.className }} + ingressClassName: {{ . }} + {{- end }} + {{- with $cc.ingress.tls }} + tls: {{- toYaml . | nindent 4 }} + {{- end }} + rules: + - host: {{ required "controlCenter.ingress.host is required when the ingress is enabled" $cc.ingress.host }} + http: + paths: + - path: / + pathType: Prefix + backend: + service: + name: {{ $name }} + port: + name: http +{{- end }} +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/values.yaml b/helm-charts/charts/simplyblock-operator/values.yaml index 420bb7951..276d44491 100644 --- a/helm-charts/charts/simplyblock-operator/values.yaml +++ b/helm-charts/charts/simplyblock-operator/values.yaml @@ -974,3 +974,155 @@ tls: # and FoundationDB's peers, so there is nothing left for a deployment to # provision before it can be required. mutual_enabled: true + +# The Control Center — the simplyblock web console (control-center/ in the +# monorepo). Off by default: it proxies the Kubernetes API with the pod's +# ServiceAccount token, so enabling it is a deliberate decision about who may +# reach the Service. See control-center/README.md, "Who can do what". +controlCenter: + enabled: false + + # Naming. The console templates are self-contained and do not borrow the + # chart's helpers, so these control its object names. Default object name: + # -control-center. + nameOverride: "" + fullnameOverride: "" + + image: + # defaults to quay.io when unset + registry: "" + repository: simplyblock-io/control-center + # inherits .Chart.AppVersion when unset. This chart's appVersion is a + # floating tag, so pin a released console tag here for production. + tag: "" + pullPolicy: IfNotPresent + imagePullSecrets: [] + + # Stateless — location lives in the browser — so two replicas cost little and + # keep the console up through a node loss, which is when it matters most. + replicas: 2 + + # full | dr + # + # full: the storage console — clusters, Kubernetes, control plane — with a + # Disaster recovery section that reads the DR hub's dr.simplyblock.io CRDs + # when dr-simplyblock is installed on the same cluster. + # dr: the DR-only console. Nothing but the DR section; no storage CRDs, no + # operator API, no Prometheus. This is what the dr-simplyblock-hub chart + # deploys (console.enabled) on a hub without a simplyblock control plane; + # set it here only to run the same stripped-down console from this chart. + mode: full + # Ramen's ops namespace on the DR hub: where discovered ProtectedApplications + # live and where the console asks the API server what it may do. + drNamespace: ramen-ops + + # serviceaccount | passthrough + # + # serviceaccount: this pod attaches its own token to proxied requests. The + # browser holds no credential. The console's authority is the ClusterRole, + # shared by everyone who can reach it, so put authentication in front. + # + # passthrough: the browser supplies the bearer token and Kubernetes enforces + # that user's own RBAC. Per-user authority and a real audit trail, at the + # cost of needing an auth proxy that injects the token. If you use this, + # set rbac.create=false — the pod needs no permissions of its own. + authMode: serviceaccount + tokenRefreshSeconds: 600 + + kubernetesApi: https://kubernetes.default.svc + # SNI and Host presented to the API server. Must be a name on its + # certificate — change this only together with kubernetesApi. + kubernetesApiHost: kubernetes.default.svc + kubernetesApiCaFile: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + # Empty defaults resolve to this chart's own services: the operator API on + # http://simplyblock-operator:8080 and Prometheus per + # prometheus.simplyblock.prometheusURL/prometheusPORT. + operatorUrl: "" + helmUrl: "" + prometheusUrl: "" + + rbac: + # Required in serviceaccount mode: the proxied token needs these rules or + # every request returns 403 and the console renders empty. Set false only + # with authMode: passthrough, where each user's own RBAC applies instead. + create: true + + service: + type: ClusterIP + port: 80 + # With type NodePort: the node port to pin, else Kubernetes picks one. + nodePort: null + + ingress: + enabled: false + className: nginx + host: "" + annotations: + # Authentication is not optional in serviceaccount mode. Replace this with + # your OIDC/OAuth2 proxy annotations, or keep basic auth as a stop-gap. + nginx.ingress.kubernetes.io/auth-type: basic + nginx.ingress.kubernetes.io/auth-secret: simplyblock-control-center-auth + nginx.ingress.kubernetes.io/auth-realm: simplyblock Control Center + nginx.ingress.kubernetes.io/proxy-body-size: 2m + nginx.ingress.kubernetes.io/proxy-read-timeout: "3600" + tls: [] + + networkPolicy: + # Caps the console's egress to its three upstreams plus DNS. Off by default + # because the API-server CIDRs below are cluster-specific: narrow them to + # your API server endpoint (`kubectl get endpoints kubernetes -n default`) + # or your service CIDR before enabling. + enabled: false + apiServerCidrs: + - 10.0.0.0/8 + - 172.16.0.0/12 + - 192.168.0.0/16 + # Namespaces allowed to reach the console, in addition to the release + # namespace itself — add your ingress controller's namespace when the + # Ingress is enabled, e.g. [ingress-nginx]. + ingressFromNamespaces: [] + + # Deploys sb-mock (control-center/mock) next to the console and points the + # console's proxy at it instead of the real cluster: the Kubernetes API, + # operator API and Prometheus are all impersonated, reads come from a + # generated dataset, and writes persist without performing any change. For + # testing the UI and demo installs only — never enable in production. + mock: + enabled: false + image: + # defaults to quay.io when unset + registry: "" + repository: simplyblock-io/control-center-mock + # inherits .Chart.AppVersion when unset + tag: "" + pullPolicy: IfNotPresent + # small-healthy | medium-degraded | large-scale | dr-failover | chaos, + # or auto for a seeded random pick + dataset: auto + # 0 picks a random seed; a fixed value makes the world reproducible + seed: 0 + # simulator tick (Ops phase progression); "0" disables the simulator so + # e2e suites can drive it deterministically via POST /mockctl/advance + simInterval: 4s + # probability a simulated operation ends Failed + failRate: 0.1 + resources: + requests: + cpu: 10m + memory: 32Mi + limits: + cpu: 200m + memory: 256Mi + + resources: + requests: + cpu: 20m + memory: 48Mi + limits: + cpu: 500m + memory: 192Mi + + nodeSelector: {} + tolerations: [] + podAnnotations: {} + From e356b56c25b4493b0fa224bc648d7b3242eac981 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 14:29:00 +0300 Subject: [PATCH 10/16] csi-driver: a replication-chain walk is bounded by a cycle guard, not by 8 hops The PV keeps the original handle while every relocate and fail-over appends a clone, so the chain behind a volume grows by one per move and is never compacted. Both walks (the node plugin's redirect to the active volume and the controller's resolveChain) stopped after 8 hops; the ninth move of a volume -- an unplanned fail-over on 2026-10-03 -- left its clone one hop out of reach, the node fell back to the stashed context and attached the original on the partitioned site, and the VM never started. The walks now track the members visited and stop on a loop, with a bound of 256 as a guard far above any real chain. Tests: a 12-move chain alternating clusters, and a looping pair. Co-Authored-By: Claude Fable 5.1 --- .../internal/csi/controller/replication.go | 18 +++- csi-driver/internal/csi/node/stats.go | 21 ++++- csi-driver/internal/csi/node/stats_test.go | 82 +++++++++++++++++++ 3 files changed, 118 insertions(+), 3 deletions(-) diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index e7b483ecf..85f5b21c4 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -9,6 +9,7 @@ package controller import ( "context" "errors" + "fmt" "sort" "github.com/csi-addons/spec/lib/go/replication" @@ -139,7 +140,15 @@ func resolveReplica( func resolveChain(ctx context.Context, h *lvol.Handle, client *atlascp.Client) ([]chainHop, bool, error) { known := false hops := []chainHop{{h: h, client: client}} - for range 8 { // one hop per past fail-over; capped far above any real chain + // One hop per past fail-over, never compacted (the PV keeps the original + // handle): a cap of 8 ended the walk one hop short of a ninth move's + // clone (2026-10-03). The bound is a cycle guard, not a length estimate. + visited := map[lvol.VolumeHandle]bool{} + for range maxChainHops { + if visited[h.Handle()] { + return nil, false, fmt.Errorf("replication chain of %s loops at %s", hops[0].h.Handle(), h.Handle()) + } + visited[h.Handle()] = true rel, err := client.GetVolumeReplicationRelationship(ctx, h.Handle()) if err != nil { if errors.Is(err, errs.ErrNotFound) { @@ -162,9 +171,16 @@ func resolveChain(ctx context.Context, h *lvol.Handle, client *atlascp.Client) ( break } } + if len(hops) > maxChainHops { + return nil, false, fmt.Errorf("replication chain of %s did not converge within %d hops", hops[0].h.Handle(), maxChainHops) + } return hops, known, nil } +// maxChainHops bounds a replication-chain walk: a guard against a looping +// record, far above any chain a volume accumulates in its lifetime. +const maxChainHops = 256 + // activeEndFallback is where a Resync or a status read goes when the local // member of the chain is reaped: the chain's active end -- the live primary // on the other site -- and, as the cluster to fail back to, the local one. diff --git a/csi-driver/internal/csi/node/stats.go b/csi-driver/internal/csi/node/stats.go index 6185e9257..0e24a3085 100644 --- a/csi-driver/internal/csi/node/stats.go +++ b/csi-driver/internal/csi/node/stats.go @@ -129,7 +129,19 @@ func redirectToActiveVolume( vc map[string]string, ) map[string]string { client, lvolID := srcClient, srcLvolID - for range 8 { // one hop per past fail-over; capped far above any real chain + // One hop per past fail-over, and the chain never shrinks: the PV keeps + // the original handle while every relocate and fail-over appends a clone, + // so a volume moved nine times is nine hops out. A cap of 8 stranded a + // fail-over's clone behind the ninth hop and the node attached the + // partitioned original instead (2026-10-03, WordPress's fifth move of + // the day). The bound is a cycle guard now, not a length estimate. + visited := map[string]bool{} + for range maxChainHops { + if visited[lvolID] { + klog.Warningf("replication chain for deleted volume %s loops at %s", volumeID, lvolID) + return nil + } + visited[lvolID] = true rel, err := client.GetRelationship(ctx, lvolID) if err != nil || rel == nil { klog.Warningf("replication relationship lookup failed for deleted volume %s (at hop %s): %v", @@ -169,6 +181,11 @@ func redirectToActiveVolume( connInfo["poolID"] = rel.TargetPoolID return connInfo } - klog.Warningf("replication chain for deleted volume %s did not converge within 8 hops", volumeID) + klog.Warningf("replication chain for deleted volume %s did not converge within %d hops", volumeID, maxChainHops) return nil } + +// maxChainHops bounds a replication-chain walk. A chain grows by one member +// per move and is never compacted, so this is a guard against a looping +// record, far above any chain a volume accumulates in its lifetime. +const maxChainHops = 256 diff --git a/csi-driver/internal/csi/node/stats_test.go b/csi-driver/internal/csi/node/stats_test.go index 0a78fa2b7..bb01510f7 100644 --- a/csi-driver/internal/csi/node/stats_test.go +++ b/csi-driver/internal/csi/node/stats_test.go @@ -7,6 +7,7 @@ package node import ( "context" "errors" + "fmt" "testing" "github.com/simplyblock/csi-driver/internal/controlplane" @@ -159,3 +160,84 @@ func TestRedirectToActiveVolumeSinglePairingIsUnchanged(t *testing.T) { t.Errorf("cluster_id = %q, want %q", got, clusterB) } } + +// A volume moved many times: the PV keeps the original handle while every +// relocate and fail-over appends a clone, alternating between the two +// clusters, so the live copy sits one hop further out after each move. The +// walk must reach it however long the chain has grown; a cap of 8 stranded +// the ninth move's clone and the node attached the original on the +// partitioned site instead (2026-10-03). +func TestRedirectToActiveVolumeFollowsALongChain(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + clusterB = "bbbbbbbb-0000-0000-0000-000000000001" + poolA = "aaaaaaaa-0000-0000-0000-00000000000a" + poolB = "bbbbbbbb-0000-0000-0000-00000000000b" + moves = 12 + ) + member := func(i int) string { return fmt.Sprintf("%08d-0000-0000-0000-000000000000", i) } + cluster := func(i int) (string, string) { + if i%2 == 0 { + return clusterA, poolA + } + return clusterB, poolB + } + active := member(moves) + clients := map[string]*fakeRelationshipAPI{ + clusterA + "/" + poolA: {rels: map[string]*controlplane.ReplicationRelationship{}, conn: map[string]map[string]string{}}, + clusterB + "/" + poolB: {rels: map[string]*controlplane.ReplicationRelationship{}, conn: map[string]map[string]string{}}, + } + for i := 0; i < moves; i++ { + srcC, srcP := cluster(i) + tgtC, tgtP := cluster(i + 1) + clients[srcC+"/"+srcP].rels[member(i)] = &controlplane.ReplicationRelationship{ + SourceLvolID: member(i), TargetLvolID: member(i + 1), + SourceClusterID: srcC, TargetClusterID: tgtC, TargetPoolID: tgtP, + ActiveLvolID: active, + } + } + activeC, activeP := cluster(moves) + clients[activeC+"/"+activeP].conn[active] = map[string]string{"nqn": "nqn.test:" + active} + + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(_ context.Context, clusterID, poolID string) (controlplane.ClusterAPI, error) { + if c, ok := clients[clusterID+"/"+poolID]; ok { + return c, nil + } + return nil, errors.New("unexpected cluster " + clusterID + "/" + poolID) + } + + connInfo := redirectToActiveVolume(context.Background(), clients[clusterA+"/"+poolA], member(0), + clusterA+":"+poolA+":"+member(0), map[string]string{"hostNQN": "nqn.host"}) + if connInfo == nil { + t.Fatalf("redirect returned nil: the walk gave up before the %d-hop chain's active volume", moves) + } + if got := connInfo["nqn"]; got != "nqn.test:"+active { + t.Errorf("connection nqn = %q, want the active volume's %q", got, "nqn.test:"+active) + } + if got := connInfo[csicommon.ParamClusterID]; got != activeC { + t.Errorf("cluster_id = %q, want the active volume's cluster %q", got, activeC) + } +} + +// A relationship that points back at a member already walked must end the +// walk instead of spinning to the bound. +func TestRedirectToActiveVolumeStopsOnALoop(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + poolA = "aaaaaaaa-0000-0000-0000-00000000000a" + x = "11111111-1111-1111-1111-111111111111" + y = "22222222-2222-2222-2222-222222222222" + ) + client := &fakeRelationshipAPI{rels: map[string]*controlplane.ReplicationRelationship{ + x: {SourceLvolID: x, TargetLvolID: y, SourceClusterID: clusterA, TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz"}, + y: {SourceLvolID: y, TargetLvolID: x, SourceClusterID: clusterA, TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz"}, + }} + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(context.Context, string, string) (controlplane.ClusterAPI, error) { return client, nil } + if got := redirectToActiveVolume(context.Background(), client, x, clusterA+":"+poolA+":"+x, map[string]string{}); got != nil { + t.Fatalf("a looping chain returned %v, want nil", got) + } +} From f8760ac5af3af96cd21b5bcc146621160da412d8 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 14:31:26 +0300 Subject: [PATCH 11/16] operator: TestFailover CRD as controller-gen renders it The hand-written sourceVolumeMode entry differed from the generated one in wrapping; the Manifests check keeps the three copies identical to the generator's output. Co-Authored-By: Claude Fable 5.1 --- .../crds/storage.simplyblock.io_testfailovers.yaml | 14 +++++++------- .../storage.simplyblock.io_testfailovers.yaml | 14 +++++++------- .../storage.simplyblock.io_testfailovers.yaml | 14 +++++++------- 3 files changed, 21 insertions(+), 21 deletions(-) diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml index 1cca0647d..d8b008df6 100644 --- a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml @@ -185,13 +185,6 @@ spec: empty fsType makes the node plugin default to ext4 and refuse to mount an XFS volume. type: string - sourceVolumeMode: - description: |- - SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), - carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble - claim that omitted the mode defaulted to Filesystem and the kubelet asked - the node plugin to mount a raw guest disk (2026-10-03). - type: string sourceHandle: description: SourceHandle is the source volume's backend handle, read from its PV. @@ -213,6 +206,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object diff --git a/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml index 1cca0647d..d8b008df6 100644 --- a/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml +++ b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml @@ -185,13 +185,6 @@ spec: empty fsType makes the node plugin default to ext4 and refuse to mount an XFS volume. type: string - sourceVolumeMode: - description: |- - SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), - carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble - claim that omitted the mode defaulted to Filesystem and the kubelet asked - the node plugin to mount a raw guest disk (2026-10-03). - type: string sourceHandle: description: SourceHandle is the source volume's backend handle, read from its PV. @@ -213,6 +206,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml index 1cca0647d..d8b008df6 100644 --- a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml @@ -185,13 +185,6 @@ spec: empty fsType makes the node plugin default to ext4 and refuse to mount an XFS volume. type: string - sourceVolumeMode: - description: |- - SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), - carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble - claim that omitted the mode defaulted to Filesystem and the kubelet asked - the node plugin to mount a raw guest disk (2026-10-03). - type: string sourceHandle: description: SourceHandle is the source volume's backend handle, read from its PV. @@ -213,6 +206,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object From 8233d4bba45563abf7a45e21c83213df9706a160 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 14:59:47 +0300 Subject: [PATCH 12/16] operator: StorageSiteDeployment, a managed site's storage deployed from the hub A site's storage cluster is built from objects on the site's API server (the OperatorOps discovery, the ClusterDeploymentConfig draft it writes, the StorageCluster the approved draft expands into), which a hub managing the site through OCM does not reach. StorageSiteDeployment is the hub-side request: the site, the discovery filters, the sizing, and the one-way approval. Its controller (hub-only, registered with TestFailover when the ManifestWork API is served) carries the request through one ManifestWork in the site's hub namespace -- the discovery first, then a server-side apply of the sizing and the approval onto the draft once it names nodes -- and projects the draft, the StorageCluster and its nodes back through ManagedClusterViews. Phases Pending, Discovering, Drafted, Deploying, Online, Failed; conditions Delivered, Discovered, Approved, Ready; the cluster's uuid and pool are in the status, which is what a StorageClass names. Deleting the request orphans the work's resources: the storage cluster is never torn down by withdrawing the request. The console's ClusterRole may manage the kind and read ManagedClusters. The TestFailover CRD copies are the generator's rendering (as on #618). Design: simplyblock-dr docs/design/control-center-managed-discovery.md. Co-Authored-By: Claude Opus 5.5 --- ...simplyblock.io_storagesitedeployments.yaml | 1057 +++++++++++++++++ .../templates/control-center-rbac.yaml | 8 + .../templates/roles/manager_role.yaml | 11 + .../v1alpha2/storagesitedeployment_types.go | 298 +++++ .../api/v1alpha2/zz_generated.deepcopy.go | 251 ++++ operator/cmd/main.go | 12 +- ...simplyblock.io_storagesitedeployments.yaml | 1057 +++++++++++++++++ operator/config/crd/kustomization.yaml | 1 + operator/config/rbac/role.yaml | 11 + .../storagesitedeployment_controller.go | 719 +++++++++++ ...ragesitedeployment_controller_unit_test.go | 335 ++++++ ...simplyblock.io_storagesitedeployments.yaml | 1057 +++++++++++++++++ 12 files changed, 4816 insertions(+), 1 deletion(-) create mode 100644 helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml create mode 100644 operator/api/v1alpha2/storagesitedeployment_types.go create mode 100644 operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml create mode 100644 operator/internal/controller/storagesitedeployment_controller.go create mode 100644 operator/internal/controller/storagesitedeployment_controller_unit_test.go create mode 100644 operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..9c2718a8f --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1057 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml index c1bd77089..4a638e58c 100644 --- a/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml +++ b/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml @@ -144,6 +144,14 @@ rules: - apiGroups: ["sitemap.simplyblock.io"] resources: ["dhcpservers"] verbs: ["create", "update", "patch", "delete"] + # a managed site's storage deployment is requested, sized and approved + # from the hub console (StorageSiteDeployment, carried to the site by OCM) + - apiGroups: ["storage.simplyblock.io"] + resources: ["storagesitedeployments"] + verbs: ["get", "list", "watch", "create", "update", "patch", "delete"] + - apiGroups: ["cluster.open-cluster-management.io"] + resources: ["managedclusters"] + verbs: ["get", "list", "watch"] # the console asks the API server what its identity may do, to disable controls - apiGroups: ["authorization.k8s.io"] resources: ["selfsubjectaccessreviews", "selfsubjectrulesreviews"] diff --git a/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml b/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml index c6801eb7f..3f8c28526 100644 --- a/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml +++ b/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml @@ -176,6 +176,14 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch - apiGroups: - coordination.k8s.io resources: @@ -325,6 +333,7 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks - testfailovers - volumemigrations @@ -361,6 +370,7 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers - testfailovers/finalizers - volumemigrations/finalizers @@ -394,6 +404,7 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status - testfailovers/status - volumegroupsnapshotops/status diff --git a/operator/api/v1alpha2/storagesitedeployment_types.go b/operator/api/v1alpha2/storagesitedeployment_types.go new file mode 100644 index 000000000..93a18006c --- /dev/null +++ b/operator/api/v1alpha2/storagesitedeployment_types.go @@ -0,0 +1,298 @@ +// The storage deployment of a managed site, requested from the hub. +// +// A site's storage cluster is built from objects that live on the site's API +// server: an OperatorOps discovery, the ClusterDeploymentConfig draft it +// writes, and the StorageCluster the approved draft expands into. A hub that +// manages the site through Open Cluster Management does not reach that API +// server, so this kind is the hub-side request: it names the site and the +// sizing, and a controller carries the request to the site through a +// ManifestWork and projects the site's answer back through ManagedClusterViews. +// Approval is the same one-way gate the draft has on the site; it is flipped +// here and delivered there. +// +// Deleting the object withdraws nothing on the site: the storage cluster it +// requested stays, as a storage cluster is never torn down by deleting a +// request. Specified by docs/design/control-center-managed-discovery.md of the +// simplyblock-dr repository. + +package v1alpha2 + +import ( + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +// StorageSiteDeploymentPhase is the request's own progress. +// +kubebuilder:validation:Enum=Pending;Discovering;Drafted;Deploying;Online;Failed +type StorageSiteDeploymentPhase string + +const ( + // StorageSiteDeploymentPhasePending is the request before the hub delivered + // anything to the site. + StorageSiteDeploymentPhasePending StorageSiteDeploymentPhase = "Pending" + + // StorageSiteDeploymentPhaseDiscovering is the discovery running on the site: + // the draft is not written yet, or names no node yet. + StorageSiteDeploymentPhaseDiscovering StorageSiteDeploymentPhase = "Discovering" + + // StorageSiteDeploymentPhaseDrafted is a draft with nodes on the site, sized + // as the request says, awaiting approval. + StorageSiteDeploymentPhaseDrafted StorageSiteDeploymentPhase = "Drafted" + + // StorageSiteDeploymentPhaseDeploying is an approved draft expanding into a + // StorageCluster that is not Online yet. + StorageSiteDeploymentPhaseDeploying StorageSiteDeploymentPhase = "Deploying" + + // StorageSiteDeploymentPhaseOnline is the StorageCluster Online on the site. + StorageSiteDeploymentPhaseOnline StorageSiteDeploymentPhase = "Online" + + // StorageSiteDeploymentPhaseFailed is the site's own failure: the draft or the + // StorageCluster failed, or the hub could not deliver the request. + StorageSiteDeploymentPhaseFailed StorageSiteDeploymentPhase = "Failed" +) + +// StorageSiteDiscovery is the discovery the site runs: which nodes are +// inspected. It is the hub-side form of OperatorOps.spec.discover. +type StorageSiteDiscovery struct { + // EnableControlPlaneNodes lets the discovery consider the nodes that run the + // API server. Every server of a small distribution is one, so a three-node + // site has no storage without it. + // +optional + EnableControlPlaneNodes *bool `json:"enableControlPlaneNodes,omitempty"` + + // Workers limits the discovery to these nodes. Empty is every worker. + // +optional + // +listType=set + Workers []string `json:"workers,omitempty"` + + // NodeSelector limits the discovery to the nodes carrying these labels. + // +optional + NodeSelector map[string]string `json:"nodeSelector,omitempty"` +} + +// StorageSiteSizing is the cluster template written onto the draft before it +// is approved: the fields of ClusterDeploymentConfig.spec.cluster a reviewer +// decides. Absent fields keep what the discovery wrote. +type StorageSiteSizing struct { + // Name is the StorageCluster's name on the site. + // +kubebuilder:validation:MaxLength=63 + // +optional + Name string `json:"name,omitempty"` + + // VCPUCount is the number of vCPUs each storage node takes. + // +kubebuilder:validation:Minimum=1 + // +optional + VCPUCount *int32 `json:"vcpuCount,omitempty"` + + // MinHugePagesSize is the hugepage memory each storage node takes, as a + // quantity ("8G"). + // +optional + MinHugePagesSize string `json:"minHugePagesSize,omitempty"` + + // MaxSubsystemCount is the number of NVMe-oF subsystems each node serves. + // +kubebuilder:validation:Minimum=1 + // +optional + MaxSubsystemCount *int32 `json:"maxSubsystemCount,omitempty"` + + // EnableDriveFormat lets the deployment format the devices it takes. + // +optional + EnableDriveFormat *bool `json:"enableDriveFormat,omitempty"` + + // EnableJournalDevice dedicates one device per node to the journal. + // +optional + EnableJournalDevice *bool `json:"enableJournalDevice,omitempty"` + + // Stripe is the erasure-coding layout. + // +optional + Stripe *StripeSpec `json:"stripe,omitempty"` +} + +// StorageSiteDeploymentSpec is the request for one site's storage cluster. +// +kubebuilder:validation:XValidation:rule="!has(oldSelf.approved) || !oldSelf.approved || self.approved",message="approval is one-way: an approved deployment cannot be un-approved" +type StorageSiteDeploymentSpec struct { + // Cluster is the OCM ManagedCluster the storage is deployed on. The request's + // ManifestWork and views live in its namespace on the hub. Immutable. + // +kubebuilder:validation:Required + // +kubebuilder:validation:MinLength=1 + // +kubebuilder:validation:MaxLength=63 + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="cluster is immutable" + Cluster string `json:"cluster"` + + // SiteNamespace is the simplyblock operator's namespace on the site, where + // the discovery and the draft live. + // +kubebuilder:default=simplyblock + // +kubebuilder:validation:MaxLength=63 + // +optional + SiteNamespace string `json:"siteNamespace,omitempty"` + + // DraftName is the ClusterDeploymentConfig the discovery writes on the site + // and the request sizes and approves. Immutable. + // +kubebuilder:default=site-draft + // +kubebuilder:validation:MaxLength=63 + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="draftName is immutable" + // +optional + DraftName string `json:"draftName,omitempty"` + + // Discover is the discovery the site runs first. Changing it runs another + // discovery, which rewrites the draft. + // +optional + Discover StorageSiteDiscovery `json:"discover,omitempty"` + + // Sizing is written onto the draft's cluster template once the draft exists, + // so the reviewer sees the sized draft before approving it. + // +optional + Sizing *StorageSiteSizing `json:"sizing,omitempty"` + + // Approved is the review gate, delivered to the draft on the site. One-way, + // as the draft's own gate is. + // +kubebuilder:default=false + // +optional + Approved bool `json:"approved"` +} + +// StorageSiteDraft is the draft as the site reports it. +type StorageSiteDraft struct { + // Name is the ClusterDeploymentConfig on the site. + Name string `json:"name"` + + // Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + // Failed). + // +optional + Phase string `json:"phase,omitempty"` + + // Message is what the site says about the draft: validation findings while + // it is a draft, the expansion's step afterwards. + // +optional + Message string `json:"message,omitempty"` + + // Approved is whether the draft is approved on the site. + // +optional + Approved bool `json:"approved,omitempty"` + + // Cluster is the draft's cluster template, with the sizing applied. + // +optional + Cluster *ClusterTemplate `json:"cluster,omitempty"` + + // NodeSets are the nodes and devices the discovery found, for review. + // +optional + NodeSets []NodeSet `json:"nodeSets,omitempty"` + + // NodeRefs are the StorageNode objects the expansion created. + // +optional + // +listType=set + NodeRefs []string `json:"nodeRefs,omitempty"` +} + +// StorageSiteNode is one storage node of the deployed cluster, as the site +// reports it. +type StorageSiteNode struct { + // Name is the StorageNode object on the site. + Name string `json:"name"` + + // Phase is the node's phase on the site. + // +optional + Phase string `json:"phase,omitempty"` + + // Hostname is the Kubernetes node it runs on. + // +optional + Hostname string `json:"hostname,omitempty"` +} + +// StorageSiteCluster is the StorageCluster the approved draft produced. +type StorageSiteCluster struct { + // Name is the StorageCluster object on the site. + Name string `json:"name"` + + // UUID is the storage cluster's id in the control plane, which a + // StorageClass names in cluster_id. + // +optional + UUID string `json:"uuid,omitempty"` + + // Phase is the StorageCluster's phase on the site. + // +optional + Phase string `json:"phase,omitempty"` + + // Pool is the pool the cluster was created with, which a StorageClass names + // in pool_name. + // +optional + Pool string `json:"pool,omitempty"` + + // Nodes are the cluster's storage nodes. + // +optional + // +listType=map + // +listMapKey=name + Nodes []StorageSiteNode `json:"nodes,omitempty"` +} + +// StorageSiteDeploymentStatus is what the site reports back, projected. +type StorageSiteDeploymentStatus struct { + // Phase is the request's own progress. + // +optional + Phase StorageSiteDeploymentPhase `json:"phase,omitempty"` + + // Message is the reason the phase is what it is: one sentence, replaced as + // the request moves, and never a log. + // +optional + Message string `json:"message,omitempty"` + + // ObservedGeneration is the generation the rest of this status was computed + // from. + // +optional + ObservedGeneration int64 `json:"observedGeneration,omitempty"` + + // WorkName is the ManifestWork carrying the request to the site. + // +optional + WorkName string `json:"workName,omitempty"` + + // Draft is the draft as the site reports it. + // +optional + Draft *StorageSiteDraft `json:"draft,omitempty"` + + // StorageCluster is the cluster the approved draft produced. + // +optional + StorageCluster *StorageSiteCluster `json:"storageCluster,omitempty"` + + // Conditions: Delivered (the work is applied on the site), Discovered (the + // draft names nodes), Approved (the site's draft is approved), Ready (the + // StorageCluster is Online). + // +optional + // +listType=map + // +listMapKey=type + Conditions []metav1.Condition `json:"conditions,omitempty"` +} + +// +kubebuilder:object:root=true +// +kubebuilder:subresource:status +// +kubebuilder:resource:scope=Namespaced,shortName=sbsd +// +kubebuilder:printcolumn:name="Cluster",type=string,JSONPath=".spec.cluster" +// +kubebuilder:printcolumn:name="Approved",type=boolean,JSONPath=".spec.approved" +// +kubebuilder:printcolumn:name="Phase",type=string,JSONPath=".status.phase" +// +kubebuilder:printcolumn:name="Draft",type=string,JSONPath=".status.draft.phase" +// +kubebuilder:printcolumn:name="Storage",type=string,JSONPath=".status.storageCluster.phase" +// +kubebuilder:printcolumn:name="Message",type=string,JSONPath=".status.message",priority=1 +// +kubebuilder:printcolumn:name="Age",type=date,JSONPath=".metadata.creationTimestamp" + +// StorageSiteDeployment requests a managed site's storage cluster from the hub: +// a discovery on the site, the sizing of the draft it writes, and the approval +// that expands the draft into a StorageCluster. The hub carries the request +// through OCM and projects the site's draft and cluster into the status. +// Deleting the request leaves the storage cluster alone. +type StorageSiteDeployment struct { + metav1.TypeMeta `json:",inline"` + metav1.ObjectMeta `json:"metadata,omitempty"` + + Spec StorageSiteDeploymentSpec `json:"spec,omitempty"` + Status StorageSiteDeploymentStatus `json:"status,omitempty"` +} + +// +kubebuilder:object:root=true + +// StorageSiteDeploymentList contains a list of StorageSiteDeployment. +type StorageSiteDeploymentList struct { + metav1.TypeMeta `json:",inline"` + metav1.ListMeta `json:"metadata,omitempty"` + Items []StorageSiteDeployment `json:"items"` +} + +func init() { + SchemeBuilder.Register(&StorageSiteDeployment{}, &StorageSiteDeploymentList{}) +} diff --git a/operator/api/v1alpha2/zz_generated.deepcopy.go b/operator/api/v1alpha2/zz_generated.deepcopy.go index 4913a6e6f..9dc34a411 100644 --- a/operator/api/v1alpha2/zz_generated.deepcopy.go +++ b/operator/api/v1alpha2/zz_generated.deepcopy.go @@ -3566,6 +3566,257 @@ func (in *StoragePoolStatus) DeepCopy() *StoragePoolStatus { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteCluster) DeepCopyInto(out *StorageSiteCluster) { + *out = *in + if in.Nodes != nil { + in, out := &in.Nodes, &out.Nodes + *out = make([]StorageSiteNode, len(*in)) + copy(*out, *in) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteCluster. +func (in *StorageSiteCluster) DeepCopy() *StorageSiteCluster { + if in == nil { + return nil + } + out := new(StorageSiteCluster) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeployment) DeepCopyInto(out *StorageSiteDeployment) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ObjectMeta.DeepCopyInto(&out.ObjectMeta) + in.Spec.DeepCopyInto(&out.Spec) + in.Status.DeepCopyInto(&out.Status) +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeployment. +func (in *StorageSiteDeployment) DeepCopy() *StorageSiteDeployment { + if in == nil { + return nil + } + out := new(StorageSiteDeployment) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *StorageSiteDeployment) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentList) DeepCopyInto(out *StorageSiteDeploymentList) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ListMeta.DeepCopyInto(&out.ListMeta) + if in.Items != nil { + in, out := &in.Items, &out.Items + *out = make([]StorageSiteDeployment, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentList. +func (in *StorageSiteDeploymentList) DeepCopy() *StorageSiteDeploymentList { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentList) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *StorageSiteDeploymentList) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentSpec) DeepCopyInto(out *StorageSiteDeploymentSpec) { + *out = *in + in.Discover.DeepCopyInto(&out.Discover) + if in.Sizing != nil { + in, out := &in.Sizing, &out.Sizing + *out = new(StorageSiteSizing) + (*in).DeepCopyInto(*out) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentSpec. +func (in *StorageSiteDeploymentSpec) DeepCopy() *StorageSiteDeploymentSpec { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentSpec) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentStatus) DeepCopyInto(out *StorageSiteDeploymentStatus) { + *out = *in + if in.Draft != nil { + in, out := &in.Draft, &out.Draft + *out = new(StorageSiteDraft) + (*in).DeepCopyInto(*out) + } + if in.StorageCluster != nil { + in, out := &in.StorageCluster, &out.StorageCluster + *out = new(StorageSiteCluster) + (*in).DeepCopyInto(*out) + } + if in.Conditions != nil { + in, out := &in.Conditions, &out.Conditions + *out = make([]metav1.Condition, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentStatus. +func (in *StorageSiteDeploymentStatus) DeepCopy() *StorageSiteDeploymentStatus { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentStatus) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDiscovery) DeepCopyInto(out *StorageSiteDiscovery) { + *out = *in + if in.EnableControlPlaneNodes != nil { + in, out := &in.EnableControlPlaneNodes, &out.EnableControlPlaneNodes + *out = new(bool) + **out = **in + } + if in.Workers != nil { + in, out := &in.Workers, &out.Workers + *out = make([]string, len(*in)) + copy(*out, *in) + } + if in.NodeSelector != nil { + in, out := &in.NodeSelector, &out.NodeSelector + *out = make(map[string]string, len(*in)) + for key, val := range *in { + (*out)[key] = val + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDiscovery. +func (in *StorageSiteDiscovery) DeepCopy() *StorageSiteDiscovery { + if in == nil { + return nil + } + out := new(StorageSiteDiscovery) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDraft) DeepCopyInto(out *StorageSiteDraft) { + *out = *in + if in.Cluster != nil { + in, out := &in.Cluster, &out.Cluster + *out = new(ClusterTemplate) + (*in).DeepCopyInto(*out) + } + if in.NodeSets != nil { + in, out := &in.NodeSets, &out.NodeSets + *out = make([]NodeSet, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } + if in.NodeRefs != nil { + in, out := &in.NodeRefs, &out.NodeRefs + *out = make([]string, len(*in)) + copy(*out, *in) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDraft. +func (in *StorageSiteDraft) DeepCopy() *StorageSiteDraft { + if in == nil { + return nil + } + out := new(StorageSiteDraft) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteNode) DeepCopyInto(out *StorageSiteNode) { + *out = *in +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteNode. +func (in *StorageSiteNode) DeepCopy() *StorageSiteNode { + if in == nil { + return nil + } + out := new(StorageSiteNode) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteSizing) DeepCopyInto(out *StorageSiteSizing) { + *out = *in + if in.VCPUCount != nil { + in, out := &in.VCPUCount, &out.VCPUCount + *out = new(int32) + **out = **in + } + if in.MaxSubsystemCount != nil { + in, out := &in.MaxSubsystemCount, &out.MaxSubsystemCount + *out = new(int32) + **out = **in + } + if in.EnableDriveFormat != nil { + in, out := &in.EnableDriveFormat, &out.EnableDriveFormat + *out = new(bool) + **out = **in + } + if in.EnableJournalDevice != nil { + in, out := &in.EnableJournalDevice, &out.EnableJournalDevice + *out = new(bool) + **out = **in + } + if in.Stripe != nil { + in, out := &in.Stripe, &out.Stripe + *out = new(StripeSpec) + (*in).DeepCopyInto(*out) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteSizing. +func (in *StorageSiteSizing) DeepCopy() *StorageSiteSizing { + if in == nil { + return nil + } + out := new(StorageSiteSizing) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *StripeSpec) DeepCopyInto(out *StripeSpec) { *out = *in diff --git a/operator/cmd/main.go b/operator/cmd/main.go index b3258ecd9..42b37093a 100644 --- a/operator/cmd/main.go +++ b/operator/cmd/main.go @@ -962,8 +962,18 @@ func main() { setupLog.Error(err, "unable to create controller", "controller", "TestFailover") os.Exit(1) } + // A managed site's storage deployment is requested from the hub through + // the same work API, so the controller is hub-only too. + if err := (&controller.StorageSiteDeploymentReconciler{ + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorder("storagesitedeployment-controller"), + }).SetupWithManager(mgr); err != nil { + setupLog.Error(err, "unable to create controller", "controller", "StorageSiteDeployment") + os.Exit(1) + } } else { - setupLog.Info("OCM ManifestWork resource not served; skipping TestFailover controller (hub-only)", + setupLog.Info("OCM ManifestWork resource not served; skipping the TestFailover and StorageSiteDeployment controllers (hub-only)", "groupVersion", ocmWorkGroupVersion, "resource", ocmManifestWorkResource) } // +kubebuilder:scaffold:builder diff --git a/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml b/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..9c2718a8f --- /dev/null +++ b/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1057 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/operator/config/crd/kustomization.yaml b/operator/config/crd/kustomization.yaml index 760b4ffb6..520d8f4be 100644 --- a/operator/config/crd/kustomization.yaml +++ b/operator/config/crd/kustomization.yaml @@ -31,6 +31,7 @@ resources: - bases/storage.simplyblock.io_controlplaneops.yaml - bases/storage.simplyblock.io_storagedeviceops.yaml - bases/storage.simplyblock.io_testfailovers.yaml +- bases/storage.simplyblock.io_storagesitedeployments.yaml # +kubebuilder:scaffold:crdkustomizeresource patches: [] diff --git a/operator/config/rbac/role.yaml b/operator/config/rbac/role.yaml index 4d6684b03..1335dea16 100644 --- a/operator/config/rbac/role.yaml +++ b/operator/config/rbac/role.yaml @@ -176,6 +176,14 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch - apiGroups: - coordination.k8s.io resources: @@ -325,6 +333,7 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks - testfailovers - volumemigrations @@ -361,6 +370,7 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers - testfailovers/finalizers - volumemigrations/finalizers @@ -394,6 +404,7 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status - testfailovers/status - volumegroupsnapshotops/status diff --git a/operator/internal/controller/storagesitedeployment_controller.go b/operator/internal/controller/storagesitedeployment_controller.go new file mode 100644 index 000000000..c4a24c091 --- /dev/null +++ b/operator/internal/controller/storagesitedeployment_controller.go @@ -0,0 +1,719 @@ +// The StorageSiteDeployment controller carries a managed site's storage +// deployment request from the hub to the site and projects the site's answer +// back. +// +// It never holds a site kubeconfig: every write to the site is a ManifestWork +// in the site's hub namespace, every read a ManagedClusterView there, the same +// two primitives the TestFailover controller uses. The work carries the +// OperatorOps discovery first; once the site has written a draft with nodes, it +// carries a server-side apply of the draft's sizing, and when the request is +// approved, the draft's approval. The views project the draft, the +// StorageCluster the approved draft expands into, and that cluster's nodes. +// +// The request withdraws nothing on deletion: the work is released with its +// resources orphaned, so a storage cluster is never torn down by deleting the +// request that asked for it. See docs/design/control-center-managed-discovery.md +// in the simplyblock-dr repository. + +package controller + +import ( + "context" + "crypto/sha256" + "encoding/json" + "fmt" + "sort" + "time" + + corev1 "k8s.io/api/core/v1" + apiequality "k8s.io/apimachinery/pkg/api/equality" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/client-go/tools/events" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" + logf "sigs.k8s.io/controller-runtime/pkg/log" + + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/controllers/pool" +) + +// storageSiteDeploymentIDLabel tags the work and the views of one request, so +// its release can enumerate them. +const storageSiteDeploymentIDLabel = "storage.simplyblock.io/site-deployment" + +// finalizerStorageSiteDeployment holds the request until its work and views +// are released. The work is released with its resources orphaned: the +// discovery, the draft and the storage cluster stay on the site. +const finalizerStorageSiteDeployment = "storage.simplyblock.io/storagesitedeployment-release" + +// hubDeployFieldManager is the field manager the work agent applies the +// draft's sizing and approval with, so the discovery's own fields on the draft +// are left to their owner. +const hubDeployFieldManager = "hub-deploy" + +// storageSiteDeploymentRequeue is how long the reconcile waits before reading +// the site's views again while the request is in progress. +const storageSiteDeploymentRequeue = 15 * time.Second + +// storageSiteDeploymentOnlineRequeue keeps an Online request's projection of +// the storage cluster fresh without polling the site hard. +const storageSiteDeploymentOnlineRequeue = 2 * time.Minute + +// maxStorageSiteNodeViews bounds the per-node views a request keeps: list +// views are not supported by ManagedClusterView, so there is one per node +// named in the draft's nodeRefs. +const maxStorageSiteNodeViews = 64 + +// annotationStorageClusterDefaultPool is the annotation the StorageCluster +// controller records the cluster's first pool under (controllers/cluster). +const annotationStorageClusterDefaultPool = "storage.simplyblock.io/default-pool" + +// Conditions of a request. +const ( + ConditionStorageSiteDelivered = "Delivered" + ConditionStorageSiteDiscovered = "Discovered" + ConditionStorageSiteApproved = "Approved" + ConditionStorageSiteReady = "Ready" +) + +// StorageSiteDeploymentReconciler reconciles a StorageSiteDeployment object. +type StorageSiteDeploymentReconciler struct { + client.Client + Scheme *runtime.Scheme + Recorder events.EventRecorder +} + +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments/status,verbs=get;update;patch +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments/finalizers,verbs=update +// +kubebuilder:rbac:groups=work.open-cluster-management.io,resources=manifestworks,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=view.open-cluster-management.io,resources=managedclusterviews,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=cluster.open-cluster-management.io,resources=managedclusters,verbs=get;list;watch +// +kubebuilder:rbac:groups=events.k8s.io,resources=events,verbs=create;patch + +// Reconcile carries the request to the site and projects the site's answer: +// it ensures the finalizer and the work, reads the draft, the storage cluster +// and its nodes through views, and derives the phase from what they report. +func (r *StorageSiteDeploymentReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) { + log := logf.FromContext(ctx) + + var sd simplyblockv1alpha2.StorageSiteDeployment + if err := r.Get(ctx, req.NamespacedName, &sd); err != nil { + return ctrl.Result{}, client.IgnoreNotFound(err) + } + + if !sd.DeletionTimestamp.IsZero() { + return r.reconcileDeletion(ctx, &sd) + } + if !controllerutil.ContainsFinalizer(&sd, finalizerStorageSiteDeployment) { + controllerutil.AddFinalizer(&sd, finalizerStorageSiteDeployment) + if err := r.Update(ctx, &sd); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{Requeue: true}, nil + } + + work, err := r.ensureWork(ctx, &sd) + if err != nil { + log.Error(err, "ensure ManifestWork") + return ctrl.Result{}, err + } + return r.project(ctx, &sd, work) +} + +// ensureWork creates or updates the request's ManifestWork: the discovery +// always, and the draft's sizing and approval once the site has a draft with +// nodes to apply them to. +func (r *StorageSiteDeploymentReconciler) ensureWork(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment) (*workv1.ManifestWork, error) { + want, err := r.manifestWork(sd) + if err != nil { + return nil, err + } + var have workv1.ManifestWork + err = r.Get(ctx, client.ObjectKeyFromObject(want), &have) + if apierrors.IsNotFound(err) { + if err := r.Create(ctx, want); err != nil { + return nil, err + } + return want, nil + } + if err != nil { + return nil, err + } + if !apiequality.Semantic.DeepEqual(have.Spec, want.Spec) || !apiequality.Semantic.DeepEqual(have.Labels, want.Labels) { + have.Spec = want.Spec + have.Labels = want.Labels + if err := r.Update(ctx, &have); err != nil { + return nil, err + } + } + return &have, nil +} + +// manifestWork builds the request's work as it should be now. Its resources +// are orphaned on delete: the discovery, the draft and what it expanded into +// are the site's, and deleting the request must not take them away. +func (r *StorageSiteDeploymentReconciler) manifestWork(sd *simplyblockv1alpha2.StorageSiteDeployment) (*workv1.ManifestWork, error) { + ns := siteNamespace(sd) + draft := draftName(sd) + labels := map[string]string{storageSiteDeploymentIDLabel: string(sd.UID)} + + ops := map[string]any{ + "apiVersion": simplyblockv1alpha2.GroupVersion.String(), + "kind": "OperatorOps", + "metadata": map[string]any{"name": discoveryName(sd), "namespace": ns, "labels": labels}, + "spec": map[string]any{ + "action": string(simplyblockv1alpha2.OperatorOpsActionDiscover), + "discover": discoverSpec(sd), + }, + } + manifests := []workv1.Manifest{} + var configs []workv1.ManifestConfigOption + raw, err := json.Marshal(ops) + if err != nil { + return nil, fmt.Errorf("marshal discovery: %w", err) + } + manifests = append(manifests, workv1.Manifest{RawExtension: runtime.RawExtension{Raw: raw}}) + + // The sizing and the approval are applied onto the draft the discovery + // wrote, never before it exists: an apply that created the draft would + // make a document with no nodes, which the site refuses. + if draftHasNodes(sd.Status.Draft) && (sd.Spec.Sizing != nil || sd.Spec.Approved) { + spec := map[string]any{"approved": sd.Spec.Approved} + if tpl := sizingTemplate(sd.Spec.Sizing); len(tpl) > 0 { + spec["cluster"] = tpl + } + cdc := map[string]any{ + "apiVersion": simplyblockv1alpha2.GroupVersion.String(), + "kind": "ClusterDeploymentConfig", + "metadata": map[string]any{"name": draft, "namespace": ns}, + "spec": spec, + } + raw, err := json.Marshal(cdc) + if err != nil { + return nil, fmt.Errorf("marshal draft apply: %w", err) + } + manifests = append(manifests, workv1.Manifest{RawExtension: runtime.RawExtension{Raw: raw}}) + configs = append(configs, workv1.ManifestConfigOption{ + ResourceIdentifier: workv1.ResourceIdentifier{ + Group: simplyblockv1alpha2.GroupVersion.Group, Resource: "clusterdeploymentconfigs", Namespace: ns, Name: draft, + }, + UpdateStrategy: &workv1.UpdateStrategy{ + Type: workv1.UpdateStrategyTypeServerSideApply, + ServerSideApply: &workv1.ServerSideApplyConfig{ + Force: true, + FieldManager: hubDeployFieldManager, + }, + }, + FeedbackRules: []workv1.FeedbackRule{{ + Type: workv1.JSONPathsType, + JsonPaths: []workv1.JsonPath{ + {Name: "phase", Path: ".status.phase"}, + {Name: "approved", Path: ".spec.approved"}, + }, + }}, + }) + } + + return &workv1.ManifestWork{ + ObjectMeta: metav1.ObjectMeta{ + Name: workName(sd), + Namespace: sd.Spec.Cluster, + Labels: labels, + }, + Spec: workv1.ManifestWorkSpec{ + Workload: workv1.ManifestsTemplate{Manifests: manifests}, + ManifestConfigs: configs, + DeleteOption: &workv1.DeleteOption{PropagationPolicy: workv1.DeletePropagationPolicyTypeOrphan}, + }, + }, nil +} + +// discoverSpec is the OperatorOps discover block of the request. +func discoverSpec(sd *simplyblockv1alpha2.StorageSiteDeployment) map[string]any { + d := map[string]any{"configName": draftName(sd)} + if sd.Spec.Discover.EnableControlPlaneNodes != nil { + d["enableControlPlaneNodes"] = *sd.Spec.Discover.EnableControlPlaneNodes + } + if len(sd.Spec.Discover.Workers) > 0 { + d["workers"] = sd.Spec.Discover.Workers + } + if len(sd.Spec.Discover.NodeSelector) > 0 { + d["nodeSelector"] = sd.Spec.Discover.NodeSelector + } + return d +} + +// sizingTemplate is the draft's cluster template fields the request sets. +// Only the stated fields are applied, so what the discovery wrote stays. +func sizingTemplate(s *simplyblockv1alpha2.StorageSiteSizing) map[string]any { + tpl := map[string]any{} + if s == nil { + return tpl + } + if s.Name != "" { + tpl["name"] = s.Name + } + if s.VCPUCount != nil { + tpl["vcpuCount"] = *s.VCPUCount + } + if s.MinHugePagesSize != "" { + tpl["minHugePagesSize"] = s.MinHugePagesSize + } + if s.MaxSubsystemCount != nil { + tpl["maxSubsystemCount"] = *s.MaxSubsystemCount + } + if s.EnableDriveFormat != nil { + tpl["enableDriveFormat"] = *s.EnableDriveFormat + } + if s.EnableJournalDevice != nil { + tpl["enableJournalDevice"] = *s.EnableJournalDevice + } + if s.Stripe != nil { + stripe := map[string]any{} + if s.Stripe.DataChunks != nil { + stripe["dataChunks"] = *s.Stripe.DataChunks + } + if s.Stripe.ParityChunks != nil { + stripe["parityChunks"] = *s.Stripe.ParityChunks + } + tpl["stripe"] = stripe + } + return tpl +} + +// project reads the site's views and derives the request's phase. +func (r *StorageSiteDeploymentReconciler) project(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, work *workv1.ManifestWork) (ctrl.Result, error) { + delivered, deliveryMessage := workDelivery(work) + + draftObj, haveDraft, err := r.projected(ctx, sd, "draft", "clusterdeploymentconfigs", draftName(sd), siteNamespace(sd)) + if err != nil { + return ctrl.Result{}, err + } + if !haveDraft { + if deliveryMessage != "" { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, deliveryMessage, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + setCondition(s, ConditionStorageSiteDelivered, false, "NotApplied", deliveryMessage) + }) + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering, + fmt.Sprintf("waiting for site %s to write draft %s/%s", sd.Spec.Cluster, siteNamespace(sd), draftName(sd)), + func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + s.WorkName = work.Name + setCondition(s, ConditionStorageSiteDelivered, delivered, deliveryReason(delivered), deliveryNote(delivered, deliveryMessage)) + }) + } + + var cdc simplyblockv1alpha2.ClusterDeploymentConfig + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(draftObj, &cdc); err != nil { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + fmt.Sprintf("the site's draft could not be read: %v", err), nil) + } + draft := projectDraft(&cdc) + base := func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + s.WorkName = work.Name + s.Draft = draft + setCondition(s, ConditionStorageSiteDelivered, delivered, deliveryReason(delivered), deliveryNote(delivered, deliveryMessage)) + setCondition(s, ConditionStorageSiteDiscovered, draftHasNodes(draft), "Nodes", fmt.Sprintf("%d node(s) in the draft", draftNodeCount(draft))) + setCondition(s, ConditionStorageSiteApproved, draft.Approved, "SiteDraft", fmt.Sprintf("the site's draft approved=%t", draft.Approved)) + } + + if !draftHasNodes(draft) { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering, + "the site's draft names no node yet: discovery is running", base) + } + if deliveryMessage != "" { + // The draft exists, so the message is about the sizing or the approval + // the work could not apply: the site refused it. + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, deliveryMessage, base) + } + switch { + case cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseFailed: + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + "the site's draft failed: "+orDefault(cdc.Status.Message, "no message"), base) + case !cdc.Spec.Approved: + msg := "the draft awaits approval" + if sd.Spec.Approved { + msg = "approval requested; waiting for the site's draft to take it" + } else if sd.Spec.Sizing != nil && !sizingApplied(sd.Spec.Sizing, cdc.Spec.Cluster) { + msg = "the draft awaits approval; the sizing is being applied" + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted, msg, base) + } + + // Approved on the site: follow the StorageCluster it expands into. + clusterName := cdc.Status.ClusterRef + if clusterName == "" && cdc.Spec.Cluster != nil { + clusterName = cdc.Spec.Cluster.Name + } + if clusterName == "" { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying, + "the draft is approved; waiting for the site to name its StorageCluster", base) + } + scObj, haveSC, err := r.projected(ctx, sd, "cluster", "storageclusters", clusterName, siteNamespace(sd)) + if err != nil { + return ctrl.Result{}, err + } + sc := &simplyblockv1alpha2.StorageSiteCluster{Name: clusterName} + if haveSC { + var cluster simplyblockv1alpha2.StorageCluster + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(scObj, &cluster); err == nil { + sc.UUID = cluster.Status.UUID + sc.Phase = string(cluster.Status.Phase) + sc.Pool = cluster.Annotations[annotationStorageClusterDefaultPool] + } + } + if sc.Pool == "" { + sc.Pool = pool.DefaultPoolName(clusterName) + } + nodes, err := r.projectNodes(ctx, sd, draft.NodeRefs) + if err != nil { + return ctrl.Result{}, err + } + sc.Nodes = nodes + withCluster := func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + base(s) + s.StorageCluster = sc + } + + switch { + case sc.Phase == string(simplyblockv1alpha2.StorageClusterPhaseOnline) && sc.UUID != "": + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline, + fmt.Sprintf("StorageCluster %s is Online (%d node(s))", clusterName, len(nodes)), func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + withCluster(s) + setCondition(s, ConditionStorageSiteReady, true, "Online", "the StorageCluster is Online") + }) + case cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanded && sc.Phase == string(simplyblockv1alpha2.StorageClusterPhaseUnavailable): + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + fmt.Sprintf("StorageCluster %s is Unavailable after the expansion", clusterName), withCluster) + } + msg := fmt.Sprintf("draft %s, StorageCluster %s %s", orDefault(string(cdc.Status.Phase), "Expanding"), clusterName, orDefault(sc.Phase, "not reported yet")) + if step := cdc.Status.Step.State; step != "" && cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanding { + msg = fmt.Sprintf("draft Expanding (%s), StorageCluster %s %s", step, clusterName, orDefault(sc.Phase, "not reported yet")) + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying, msg, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + withCluster(s) + setCondition(s, ConditionStorageSiteReady, false, "Deploying", msg) + }) +} + +// projectNodes projects the draft's StorageNodes, one view each. +func (r *StorageSiteDeploymentReconciler) projectNodes(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, refs []string) ([]simplyblockv1alpha2.StorageSiteNode, error) { + sorted := append([]string(nil), refs...) + sort.Strings(sorted) + if len(sorted) > maxStorageSiteNodeViews { + sorted = sorted[:maxStorageSiteNodeViews] + } + nodes := make([]simplyblockv1alpha2.StorageSiteNode, 0, len(sorted)) + for i, name := range sorted { + obj, ok, err := r.projected(ctx, sd, fmt.Sprintf("node-%d", i), "storagenodes", name, siteNamespace(sd)) + if err != nil { + return nil, err + } + n := simplyblockv1alpha2.StorageSiteNode{Name: name} + if ok { + var sn simplyblockv1alpha2.StorageNode + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(obj, &sn); err == nil { + n.Phase = string(sn.Status.Phase) + n.Hostname = sn.Status.Hostname + } + } + nodes = append(nodes, n) + } + return nodes, nil +} + +// projected reads one of the request's views, creating it when it is missing, +// and reports whether the site has projected the object yet. +func (r *StorageSiteDeploymentReconciler) projected(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, suffix, resource, name, namespace string) (map[string]interface{}, bool, error) { + viewName := storageSiteViewName(sd, suffix) + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + getErr := r.Get(ctx, client.ObjectKey{Namespace: sd.Spec.Cluster, Name: viewName}, view) + if apierrors.IsNotFound(getErr) { + want := newStorageSiteView(sd, viewName, resource, name, namespace) + if createErr := r.Create(ctx, want); createErr != nil { + return nil, false, createErr + } + return nil, false, nil + } + if getErr != nil { + return nil, false, getErr + } + // A view that names another object (the draft's cluster changed) is + // pointed at the right one. + scope, _, _ := unstructured.NestedMap(view.Object, "spec", "scope") + if scope["name"] != name || scope["resource"] != resource { + want := newStorageSiteView(sd, viewName, resource, name, namespace) + view.Object["spec"] = want.Object["spec"] + if err := r.Update(ctx, view); err != nil { + return nil, false, err + } + return nil, false, nil + } + result, found, nestedErr := unstructured.NestedMap(view.Object, "status", "result") + if nestedErr != nil || !found || len(result) == 0 { + return nil, false, nil + } + return result, true, nil +} + +// newStorageSiteView asks the site to project one object back to the hub. +func newStorageSiteView(sd *simplyblockv1alpha2.StorageSiteDeployment, name, resource, targetName, targetNamespace string) *unstructured.Unstructured { + scope := map[string]interface{}{"resource": resource, "name": targetName} + if targetNamespace != "" { + scope["namespace"] = targetNamespace + } + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + view.SetNamespace(sd.Spec.Cluster) + view.SetName(name) + view.SetLabels(map[string]string{storageSiteDeploymentIDLabel: string(sd.UID)}) + _ = unstructured.SetNestedMap(view.Object, scope, "spec", "scope") + return view +} + +// setPhase writes the phase, the message and the mutation, and records a +// phase change as an event. +func (r *StorageSiteDeploymentReconciler) setPhase(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, phase simplyblockv1alpha2.StorageSiteDeploymentPhase, message string, mutate func(*simplyblockv1alpha2.StorageSiteDeploymentStatus)) (ctrl.Result, error) { + previous := sd.Status.Phase + if err := r.patchStatus(ctx, sd, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + if mutate != nil { + mutate(s) + } + s.Phase = phase + s.Message = message + }); err != nil { + return ctrl.Result{}, err + } + if previous != phase && r.Recorder != nil { + kind := corev1.EventTypeNormal + if phase == simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed { + kind = corev1.EventTypeWarning + } + r.Recorder.Eventf(sd, nil, kind, string(phase), string(phase), "%s", message) + } + switch phase { + case simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline: + return ctrl.Result{RequeueAfter: storageSiteDeploymentOnlineRequeue}, nil + case simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed: + // A failure on the site may clear (a node comes back, a draft is + // corrected on the site): keep reading at the slow cadence. + return ctrl.Result{RequeueAfter: storageSiteDeploymentOnlineRequeue}, nil + } + return ctrl.Result{RequeueAfter: storageSiteDeploymentRequeue}, nil +} + +// patchStatus applies mutate to the status and writes it with the generation +// it was computed from. +func (r *StorageSiteDeploymentReconciler) patchStatus(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, mutate func(*simplyblockv1alpha2.StorageSiteDeploymentStatus)) error { + base := client.MergeFrom(sd.DeepCopy()) + mutate(&sd.Status) + sd.Status.ObservedGeneration = sd.Generation + return r.Status().Patch(ctx, sd, base) +} + +// reconcileDeletion releases the request's work and views and removes the +// finalizer. The work orphans its resources, so nothing on the site goes. +func (r *StorageSiteDeploymentReconciler) reconcileDeletion(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment) (ctrl.Result, error) { + if !controllerutil.ContainsFinalizer(sd, finalizerStorageSiteDeployment) { + return ctrl.Result{}, nil + } + var work workv1.ManifestWork + err := r.Get(ctx, client.ObjectKey{Namespace: sd.Spec.Cluster, Name: workName(sd)}, &work) + switch { + case apierrors.IsNotFound(err): + case err != nil: + return ctrl.Result{}, err + case work.DeletionTimestamp.IsZero(): + if err := client.IgnoreNotFound(r.Delete(ctx, &work)); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + default: + // Deleting: wait for the work agent to release it. + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + } + views := &unstructured.UnstructuredList{} + listGVK := managedClusterViewGVK + listGVK.Kind += "List" + views.SetGroupVersionKind(listGVK) + if err := r.List(ctx, views, client.InNamespace(sd.Spec.Cluster), client.MatchingLabels{storageSiteDeploymentIDLabel: string(sd.UID)}); err != nil && !meta.IsNoMatchError(err) { + return ctrl.Result{}, err + } + for i := range views.Items { + if err := client.IgnoreNotFound(r.Delete(ctx, &views.Items[i])); err != nil { + return ctrl.Result{}, err + } + } + controllerutil.RemoveFinalizer(sd, finalizerStorageSiteDeployment) + return ctrl.Result{}, r.Update(ctx, sd) +} + +// workDelivery reads the work's status: whether every manifest is applied, +// and the first manifest's refusal when one is not. +func workDelivery(work *workv1.ManifestWork) (applied bool, message string) { + if work == nil { + return false, "" + } + for _, m := range work.Status.ResourceStatus.Manifests { + for _, c := range m.Conditions { + if c.Type == workv1.ManifestApplied && c.Status == metav1.ConditionFalse { + return false, fmt.Sprintf("the site did not apply %s %s: %s", m.ResourceMeta.Kind, m.ResourceMeta.Name, c.Message) + } + } + } + for _, c := range work.Status.Conditions { + if c.Type == workv1.WorkApplied { + return c.Status == metav1.ConditionTrue, "" + } + } + return false, "" +} + +func deliveryReason(delivered bool) string { + if delivered { + return "Applied" + } + return "Pending" +} + +func deliveryNote(delivered bool, message string) string { + switch { + case message != "": + return message + case delivered: + return "the work is applied on the site" + } + return "the work is not applied on the site yet" +} + +// projectDraft is the draft as the status carries it. +func projectDraft(cdc *simplyblockv1alpha2.ClusterDeploymentConfig) *simplyblockv1alpha2.StorageSiteDraft { + d := &simplyblockv1alpha2.StorageSiteDraft{ + Name: cdc.Name, + Phase: string(cdc.Status.Phase), + Message: cdc.Status.Message, + Approved: cdc.Spec.Approved, + NodeSets: cdc.Spec.NodeSets, + NodeRefs: cdc.Status.NodeRefs, + } + if cdc.Spec.Cluster != nil { + d.Cluster = cdc.Spec.Cluster.DeepCopy() + } + if d.Phase == "" { + d.Phase = string(simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft) + } + return d +} + +// draftHasNodes is whether the draft names at least one worker. +func draftHasNodes(d *simplyblockv1alpha2.StorageSiteDraft) bool { + return draftNodeCount(d) > 0 +} + +func draftNodeCount(d *simplyblockv1alpha2.StorageSiteDraft) int { + if d == nil { + return 0 + } + n := 0 + for _, set := range d.NodeSets { + for _, g := range set.Groups { + n += len(g.Workers) + } + } + return n +} + +// sizingApplied is whether the draft's template carries the request's sizing. +func sizingApplied(s *simplyblockv1alpha2.StorageSiteSizing, tpl *simplyblockv1alpha2.ClusterTemplate) bool { + if s == nil { + return true + } + if tpl == nil { + return false + } + eq32 := func(want, have *int32) bool { return want == nil || (have != nil && *have == *want) } + eqBool := func(want, have *bool) bool { return want == nil || (have != nil && *have == *want) } + if s.Name != "" && tpl.Name != s.Name { + return false + } + if s.MinHugePagesSize != "" && tpl.MinHugePagesSize != s.MinHugePagesSize { + return false + } + if !eq32(s.VCPUCount, tpl.VCPUCount) || !eq32(s.MaxSubsystemCount, tpl.MaxSubsystemCount) || + !eqBool(s.EnableDriveFormat, tpl.EnableDriveFormat) || !eqBool(s.EnableJournalDevice, tpl.EnableJournalDevice) { + return false + } + if s.Stripe != nil { + if tpl.Stripe == nil || !eq32(s.Stripe.DataChunks, tpl.Stripe.DataChunks) || !eq32(s.Stripe.ParityChunks, tpl.Stripe.ParityChunks) { + return false + } + } + return true +} + +func setCondition(s *simplyblockv1alpha2.StorageSiteDeploymentStatus, kind string, ok bool, reason, message string) { + status := metav1.ConditionFalse + if ok { + status = metav1.ConditionTrue + } + meta.SetStatusCondition(&s.Conditions, metav1.Condition{Type: kind, Status: status, Reason: reason, Message: message, ObservedGeneration: s.ObservedGeneration}) +} + +func orDefault(s, d string) string { + if s == "" { + return d + } + return s +} + +func siteNamespace(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return orDefault(sd.Spec.SiteNamespace, "simplyblock") +} + +func draftName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return orDefault(sd.Spec.DraftName, "site-draft") +} + +// storageSiteHash is a short, deterministic id of the request for the names +// of its work and views, which must be bounded and found again on a restart. +func storageSiteHash(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + h := sha256.Sum256([]byte(sd.Namespace + "/" + sd.Name)) + return fmt.Sprintf("%x", h[:6]) +} + +func workName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return "sbsd-" + storageSiteHash(sd) +} + +func storageSiteViewName(sd *simplyblockv1alpha2.StorageSiteDeployment, suffix string) string { + return "sbsd-" + storageSiteHash(sd) + "-" + suffix +} + +// discoveryName is the OperatorOps on the site. It carries a hash of the +// discovery's parameters, so a changed discovery is a new run rather than an +// edit of a finished one. +func discoveryName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + raw, _ := json.Marshal(sd.Spec.Discover) + h := sha256.Sum256(raw) + return fmt.Sprintf("hub-discover-%s-%x", draftName(sd), h[:3]) +} + +// SetupWithManager registers the controller. The work and the views live in +// the site's namespace on the hub, where an owner reference to the request +// cannot point, so the site's answers are read on the requeue cadence of the +// phase rather than through a watch. +func (r *StorageSiteDeploymentReconciler) SetupWithManager(mgr ctrl.Manager) error { + return ctrl.NewControllerManagedBy(mgr). + For(&simplyblockv1alpha2.StorageSiteDeployment{}). + Named("storagesitedeployment"). + Complete(r) +} diff --git a/operator/internal/controller/storagesitedeployment_controller_unit_test.go b/operator/internal/controller/storagesitedeployment_controller_unit_test.go new file mode 100644 index 000000000..63f9bbf76 --- /dev/null +++ b/operator/internal/controller/storagesitedeployment_controller_unit_test.go @@ -0,0 +1,335 @@ +package controller + +import ( + "context" + "encoding/json" + "testing" + + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/types" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" +) + +const ( + testSDNamespace = "simplyblock" + testSDName = "site-a" + testSDCluster = "site-a" +) + +func newStorageSiteDeploymentReconciler(t *testing.T, objects ...client.Object) (*StorageSiteDeploymentReconciler, client.Client) { + t.Helper() + scheme := newTestScheme(t) + scheme.AddKnownTypeWithName(managedClusterViewGVK, &unstructured.Unstructured{}) + listGVK := managedClusterViewGVK + listGVK.Kind += "List" + scheme.AddKnownTypeWithName(listGVK, &unstructured.UnstructuredList{}) + if err := workv1.Install(scheme); err != nil { + t.Fatalf("register work/v1 scheme: %v", err) + } + cl := newTestClient(t, scheme, + []client.Object{&simplyblockv1alpha2.StorageSiteDeployment{}, &workv1.ManifestWork{}}, + objects...) + return &StorageSiteDeploymentReconciler{Client: cl, Scheme: scheme, Recorder: &fakeRecorder{}}, cl +} + +func newSiteDeployment(mutate func(*simplyblockv1alpha2.StorageSiteDeployment)) *simplyblockv1alpha2.StorageSiteDeployment { + enable := true + sd := &simplyblockv1alpha2.StorageSiteDeployment{ + ObjectMeta: metav1.ObjectMeta{Name: testSDName, Namespace: testSDNamespace, UID: types.UID("uid-site-a")}, + Spec: simplyblockv1alpha2.StorageSiteDeploymentSpec{ + Cluster: testSDCluster, + Discover: simplyblockv1alpha2.StorageSiteDiscovery{EnableControlPlaneNodes: &enable}, + }, + } + if mutate != nil { + mutate(sd) + } + return sd +} + +// reconcileSD runs the reconcile n times (the first adds the finalizer) and +// returns the request as stored. +func reconcileSD(t *testing.T, r *StorageSiteDeploymentReconciler, cl client.Client, n int) *simplyblockv1alpha2.StorageSiteDeployment { + t.Helper() + req := ctrl.Request{NamespacedName: types.NamespacedName{Namespace: testSDNamespace, Name: testSDName}} + for i := 0; i < n; i++ { + if _, err := r.Reconcile(context.Background(), req); err != nil { + t.Fatalf("reconcile %d: %v", i, err) + } + } + var sd simplyblockv1alpha2.StorageSiteDeployment + if err := cl.Get(context.Background(), req.NamespacedName, &sd); err != nil { + t.Fatalf("get request: %v", err) + } + return &sd +} + +func getSiteWork(t *testing.T, cl client.Client, sd *simplyblockv1alpha2.StorageSiteDeployment) *workv1.ManifestWork { + t.Helper() + var w workv1.ManifestWork + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: sd.Spec.Cluster, Name: workName(sd)}, &w); err != nil { + t.Fatalf("get ManifestWork: %v", err) + } + return &w +} + +// workManifests decodes the work's manifests by kind. +func workManifests(t *testing.T, w *workv1.ManifestWork) map[string]map[string]any { + t.Helper() + out := map[string]map[string]any{} + for _, m := range w.Spec.Workload.Manifests { + obj := map[string]any{} + if err := json.Unmarshal(m.Raw, &obj); err != nil { + t.Fatalf("decode manifest: %v", err) + } + out[obj["kind"].(string)] = obj + } + return out +} + +// projectSiteObject writes what the site's view controller would project. +func projectSiteObject(t *testing.T, cl client.Client, sd *simplyblockv1alpha2.StorageSiteDeployment, suffix string, obj any) { + t.Helper() + v := getView(t, cl, sd.Spec.Cluster, storageSiteViewName(sd, suffix)) + result, err := runtime.DefaultUnstructuredConverter.ToUnstructured(obj) + if err != nil { + t.Fatalf("convert projection: %v", err) + } + setViewResult(t, cl, v, result) +} + +func siteDraft(approved bool, phase simplyblockv1alpha2.ClusterDeploymentConfigPhase, mutate func(*simplyblockv1alpha2.ClusterDeploymentConfig)) *simplyblockv1alpha2.ClusterDeploymentConfig { + cdc := &simplyblockv1alpha2.ClusterDeploymentConfig{ + TypeMeta: metav1.TypeMeta{APIVersion: simplyblockv1alpha2.GroupVersion.String(), Kind: "ClusterDeploymentConfig"}, + ObjectMeta: metav1.ObjectMeta{Name: "site-draft", Namespace: "simplyblock"}, + Spec: simplyblockv1alpha2.ClusterDeploymentConfigSpec{ + Approved: approved, + NodeSets: []simplyblockv1alpha2.NodeSet{{ + Name: "default", + Groups: []simplyblockv1alpha2.NodeGroup{{Workers: []string{"n1", "n2", "n3"}}}, + }}, + }, + } + cdc.Status.Phase = phase + if mutate != nil { + mutate(cdc) + } + return cdc +} + +func TestStorageSiteDeploymentDeliversTheDiscoveryAndWaitsForTheDraft(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(nil)) + sd := reconcileSD(t, r, cl, 2) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering { + t.Fatalf("phase = %q, want Discovering", sd.Status.Phase) + } + w := getSiteWork(t, cl, sd) + if w.Spec.DeleteOption == nil || w.Spec.DeleteOption.PropagationPolicy != workv1.DeletePropagationPolicyTypeOrphan { + t.Errorf("work delete option = %+v, want Orphan: a request must never take the site's storage away", w.Spec.DeleteOption) + } + ms := workManifests(t, w) + ops, ok := ms["OperatorOps"] + if !ok || len(ms) != 1 { + t.Fatalf("manifests = %v, want the discovery alone before the draft exists", keys(ms)) + } + discover := ops["spec"].(map[string]any)["discover"].(map[string]any) + if discover["configName"] != "site-draft" || discover["enableControlPlaneNodes"] != true { + t.Errorf("discover = %v, want configName site-draft and control-plane nodes enabled", discover) + } + // The draft view exists so the site can project the draft. + getView(t, cl, testSDCluster, storageSiteViewName(sd, "draft")) +} + +func TestStorageSiteDeploymentSizesTheDraftOnceItNamesNodes(t *testing.T) { + vcpu := int32(8) + data, parity := int32(1), int32(1) + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Sizing = &simplyblockv1alpha2.StorageSiteSizing{ + Name: "sb-site-a", VCPUCount: &vcpu, MinHugePagesSize: "8G", + Stripe: &simplyblockv1alpha2.StripeSpec{DataChunks: &data, ParityChunks: &parity}, + } + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, nil)) + sd = reconcileSD(t, r, cl, 2) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted { + t.Fatalf("phase = %q (%s), want Drafted", sd.Status.Phase, sd.Status.Message) + } + if got := draftNodeCount(sd.Status.Draft); got != 3 { + t.Errorf("projected draft nodes = %d, want 3", got) + } + ms := workManifests(t, getSiteWork(t, cl, sd)) + cdc, ok := ms["ClusterDeploymentConfig"] + if !ok { + t.Fatalf("manifests = %v, want the draft's sizing applied", keys(ms)) + } + spec := cdc["spec"].(map[string]any) + if spec["approved"] != false { + t.Errorf("approved = %v, want false before the request is approved", spec["approved"]) + } + cluster := spec["cluster"].(map[string]any) + if cluster["name"] != "sb-site-a" || cluster["vcpuCount"] != float64(8) || cluster["minHugePagesSize"] != "8G" { + t.Errorf("cluster template = %v, want the sizing", cluster) + } + if _, ok := spec["nodeSets"]; ok { + t.Error("the sizing apply must not carry nodeSets: they are the discovery's") + } + if !meta.IsStatusConditionTrue(sd.Status.Conditions, ConditionStorageSiteDiscovered) { + t.Error("Discovered condition is not True for a draft with nodes") + } +} + +func TestStorageSiteDeploymentApprovalFollowsTheClusterToOnline(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, nil)) + // One pass projects the draft, the next delivers the approval onto it. + sd = reconcileSD(t, r, cl, 2) + + ms := workManifests(t, getSiteWork(t, cl, sd)) + if ms["ClusterDeploymentConfig"]["spec"].(map[string]any)["approved"] != true { + t.Fatalf("the work does not carry the approval: %v", ms["ClusterDeploymentConfig"]) + } + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted { + t.Fatalf("phase = %q, want Drafted until the site's draft takes the approval", sd.Status.Phase) + } + + // The site's draft takes it and starts expanding. + projectSiteObject(t, cl, sd, "draft", siteDraft(true, simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanding, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Status.ClusterRef = "sb-site-a" + c.Status.NodeRefs = []string{"sn-1", "sn-2", "sn-3"} + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying { + t.Fatalf("phase = %q (%s), want Deploying", sd.Status.Phase, sd.Status.Message) + } + + // The cluster comes Online. + sc := &simplyblockv1alpha2.StorageCluster{ + TypeMeta: metav1.TypeMeta{APIVersion: simplyblockv1alpha2.GroupVersion.String(), Kind: "StorageCluster"}, + ObjectMeta: metav1.ObjectMeta{Name: "sb-site-a", Namespace: "simplyblock"}, + } + sc.Status.UUID = "8f8dd277-1544-4177-9a74-e0f66eb2672c" + sc.Status.Phase = simplyblockv1alpha2.StorageClusterPhaseOnline + projectSiteObject(t, cl, sd, "cluster", sc) + sd = reconcileSD(t, r, cl, 1) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline { + t.Fatalf("phase = %q (%s), want Online", sd.Status.Phase, sd.Status.Message) + } + if sd.Status.StorageCluster == nil || sd.Status.StorageCluster.UUID != sc.Status.UUID { + t.Fatalf("storageCluster = %+v, want the site's uuid", sd.Status.StorageCluster) + } + if sd.Status.StorageCluster.Pool == "" { + t.Error("storageCluster.pool is empty: a StorageClass needs it") + } + if n := len(sd.Status.StorageCluster.Nodes); n != 3 { + t.Errorf("projected nodes = %d, want 3", n) + } + if !meta.IsStatusConditionTrue(sd.Status.Conditions, ConditionStorageSiteReady) { + t.Error("Ready condition is not True for an Online cluster") + } +} + +func TestStorageSiteDeploymentReportsAFailedDraft(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(true, simplyblockv1alpha2.ClusterDeploymentConfigPhaseFailed, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Status.Message = "node n2 has no free device" + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed { + t.Fatalf("phase = %q, want Failed", sd.Status.Phase) + } + if sd.Status.Message != "the site's draft failed: node n2 has no free device" { + t.Errorf("message = %q, want the site's own reason", sd.Status.Message) + } +} + +func TestStorageSiteDeploymentADraftWithoutNodesIsStillDiscovering(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Spec.NodeSets = nil + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering { + t.Fatalf("phase = %q, want Discovering", sd.Status.Phase) + } + if _, ok := workManifests(t, getSiteWork(t, cl, sd))["ClusterDeploymentConfig"]; ok { + t.Error("the approval was delivered onto a draft without nodes") + } +} + +func TestStorageSiteDeploymentAChangedDiscoveryIsANewRun(t *testing.T) { + a := newSiteDeployment(nil) + b := newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Discover.Workers = []string{"n1"} + }) + if discoveryName(a) == discoveryName(b) { + t.Fatalf("discovery name %q did not change with the discovery's parameters", discoveryName(a)) + } + if discoveryName(a) != discoveryName(newSiteDeployment(nil)) { + t.Fatal("the discovery name is not stable for the same parameters") + } +} + +func TestStorageSiteDeploymentDeletionOrphansTheSitesStorage(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(nil)) + sd := reconcileSD(t, r, cl, 2) + if err := cl.Delete(context.Background(), sd); err != nil { + t.Fatalf("delete request: %v", err) + } + // First pass deletes the work; the fake client removes it at once. + reconcileOnce := func() { + req := ctrl.Request{NamespacedName: types.NamespacedName{Namespace: testSDNamespace, Name: testSDName}} + if _, err := r.Reconcile(context.Background(), req); err != nil { + t.Fatalf("reconcile: %v", err) + } + } + reconcileOnce() + reconcileOnce() + + var w workv1.ManifestWork + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: testSDCluster, Name: workName(sd)}, &w); err == nil { + t.Error("the work is still there after the request was deleted") + } + views := &unstructured.UnstructuredList{} + listGVK := managedClusterViewGVK + listGVK.Kind += "List" + views.SetGroupVersionKind(listGVK) + if err := cl.List(context.Background(), views, client.InNamespace(testSDCluster)); err != nil { + t.Fatalf("list views: %v", err) + } + if len(views.Items) != 0 { + t.Errorf("%d view(s) left after the request was deleted", len(views.Items)) + } + var gone simplyblockv1alpha2.StorageSiteDeployment + if err := cl.Get(context.Background(), client.ObjectKeyFromObject(sd), &gone); err == nil { + t.Error("the request is still there: the finalizer was not released") + } +} + +func keys(m map[string]map[string]any) []string { + out := make([]string, 0, len(m)) + for k := range m { + out = append(out, k) + } + return out +} diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..9c2718a8f --- /dev/null +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1057 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} From 2ab649a73213cbaeca1c934c9316dd70ab61f5ef Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 17:14:37 +0300 Subject: [PATCH 13/16] csi-driver: wrap lines over 120 characters; operator: regenerate dist/install.yaml Lint (lll) on the chain-walk change and its tests; the installer bundle carries the TestFailover CRD's sourceVolumeMode. Co-Authored-By: Claude Opus 5.5 --- .../internal/csi/controller/replication.go | 3 ++- csi-driver/internal/csi/node/stats_test.go | 23 +++++++++++++++---- operator/dist/install.yaml | 7 ++++++ 3 files changed, 27 insertions(+), 6 deletions(-) diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index 85f5b21c4..bdc9045c5 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -172,7 +172,8 @@ func resolveChain(ctx context.Context, h *lvol.Handle, client *atlascp.Client) ( } } if len(hops) > maxChainHops { - return nil, false, fmt.Errorf("replication chain of %s did not converge within %d hops", hops[0].h.Handle(), maxChainHops) + return nil, false, fmt.Errorf("replication chain of %s did not converge within %d hops", + hops[0].h.Handle(), maxChainHops) } return hops, known, nil } diff --git a/csi-driver/internal/csi/node/stats_test.go b/csi-driver/internal/csi/node/stats_test.go index bb01510f7..2ddc16854 100644 --- a/csi-driver/internal/csi/node/stats_test.go +++ b/csi-driver/internal/csi/node/stats_test.go @@ -184,8 +184,14 @@ func TestRedirectToActiveVolumeFollowsALongChain(t *testing.T) { } active := member(moves) clients := map[string]*fakeRelationshipAPI{ - clusterA + "/" + poolA: {rels: map[string]*controlplane.ReplicationRelationship{}, conn: map[string]map[string]string{}}, - clusterB + "/" + poolB: {rels: map[string]*controlplane.ReplicationRelationship{}, conn: map[string]map[string]string{}}, + clusterA + "/" + poolA: { + rels: map[string]*controlplane.ReplicationRelationship{}, + conn: map[string]map[string]string{}, + }, + clusterB + "/" + poolB: { + rels: map[string]*controlplane.ReplicationRelationship{}, + conn: map[string]map[string]string{}, + }, } for i := 0; i < moves; i++ { srcC, srcP := cluster(i) @@ -231,13 +237,20 @@ func TestRedirectToActiveVolumeStopsOnALoop(t *testing.T) { y = "22222222-2222-2222-2222-222222222222" ) client := &fakeRelationshipAPI{rels: map[string]*controlplane.ReplicationRelationship{ - x: {SourceLvolID: x, TargetLvolID: y, SourceClusterID: clusterA, TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz"}, - y: {SourceLvolID: y, TargetLvolID: x, SourceClusterID: clusterA, TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz"}, + x: { + SourceLvolID: x, TargetLvolID: y, SourceClusterID: clusterA, + TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz", + }, + y: { + SourceLvolID: y, TargetLvolID: x, SourceClusterID: clusterA, + TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz", + }, }} orig := clusterClientFor defer func() { clusterClientFor = orig }() clusterClientFor = func(context.Context, string, string) (controlplane.ClusterAPI, error) { return client, nil } - if got := redirectToActiveVolume(context.Background(), client, x, clusterA+":"+poolA+":"+x, map[string]string{}); got != nil { + got := redirectToActiveVolume(context.Background(), client, x, clusterA+":"+poolA+":"+x, map[string]string{}) + if got != nil { t.Fatalf("a looping chain returned %v, want nil", got) } } diff --git a/operator/dist/install.yaml b/operator/dist/install.yaml index 97cc7fead..c7c996599 100644 --- a/operator/dist/install.yaml +++ b/operator/dist/install.yaml @@ -11463,6 +11463,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object From cce7357fb80c335695409e0463b844e34d253e91 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 17:32:22 +0300 Subject: [PATCH 14/16] =?UTF-8?q?operator:=20lint=20=E2=80=94=20a=20list-k?= =?UTF-8?q?ind=20suffix=20constant;=20wrap=20the=20hub-only=20log=20line?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5.5 --- operator/cmd/main.go | 3 ++- .../internal/controller/storagesitedeployment_controller.go | 6 +++++- .../storagesitedeployment_controller_unit_test.go | 4 ++-- .../controller/testfailover_controller_unit_test.go | 4 ++-- 4 files changed, 11 insertions(+), 6 deletions(-) diff --git a/operator/cmd/main.go b/operator/cmd/main.go index 42b37093a..004db856e 100644 --- a/operator/cmd/main.go +++ b/operator/cmd/main.go @@ -973,7 +973,8 @@ func main() { os.Exit(1) } } else { - setupLog.Info("OCM ManifestWork resource not served; skipping the TestFailover and StorageSiteDeployment controllers (hub-only)", + setupLog.Info("OCM ManifestWork resource not served; skipping the TestFailover and "+ + "StorageSiteDeployment controllers (hub-only)", "groupVersion", ocmWorkGroupVersion, "resource", ocmManifestWorkResource) } // +kubebuilder:scaffold:builder diff --git a/operator/internal/controller/storagesitedeployment_controller.go b/operator/internal/controller/storagesitedeployment_controller.go index c4a24c091..87108d4fc 100644 --- a/operator/internal/controller/storagesitedeployment_controller.go +++ b/operator/internal/controller/storagesitedeployment_controller.go @@ -543,7 +543,7 @@ func (r *StorageSiteDeploymentReconciler) reconcileDeletion(ctx context.Context, } views := &unstructured.UnstructuredList{} listGVK := managedClusterViewGVK - listGVK.Kind += "List" + listGVK.Kind += listKindSuffix views.SetGroupVersionKind(listGVK) if err := r.List(ctx, views, client.InNamespace(sd.Spec.Cluster), client.MatchingLabels{storageSiteDeploymentIDLabel: string(sd.UID)}); err != nil && !meta.IsNoMatchError(err) { return ctrl.Result{}, err @@ -717,3 +717,7 @@ func (r *StorageSiteDeploymentReconciler) SetupWithManager(mgr ctrl.Manager) err Named("storagesitedeployment"). Complete(r) } + +// listKindSuffix turns a kind into its list kind (StorageCluster -> +// StorageClusterList) for an unstructured list read. +const listKindSuffix = "List" diff --git a/operator/internal/controller/storagesitedeployment_controller_unit_test.go b/operator/internal/controller/storagesitedeployment_controller_unit_test.go index 63f9bbf76..4bb445d6f 100644 --- a/operator/internal/controller/storagesitedeployment_controller_unit_test.go +++ b/operator/internal/controller/storagesitedeployment_controller_unit_test.go @@ -29,7 +29,7 @@ func newStorageSiteDeploymentReconciler(t *testing.T, objects ...client.Object) scheme := newTestScheme(t) scheme.AddKnownTypeWithName(managedClusterViewGVK, &unstructured.Unstructured{}) listGVK := managedClusterViewGVK - listGVK.Kind += "List" + listGVK.Kind += listKindSuffix scheme.AddKnownTypeWithName(listGVK, &unstructured.UnstructuredList{}) if err := workv1.Install(scheme); err != nil { t.Fatalf("register work/v1 scheme: %v", err) @@ -312,7 +312,7 @@ func TestStorageSiteDeploymentDeletionOrphansTheSitesStorage(t *testing.T) { } views := &unstructured.UnstructuredList{} listGVK := managedClusterViewGVK - listGVK.Kind += "List" + listGVK.Kind += listKindSuffix views.SetGroupVersionKind(listGVK) if err := cl.List(context.Background(), views, client.InNamespace(testSDCluster)); err != nil { t.Fatalf("list views: %v", err) diff --git a/operator/internal/controller/testfailover_controller_unit_test.go b/operator/internal/controller/testfailover_controller_unit_test.go index 2df684ed3..7c8effd6d 100644 --- a/operator/internal/controller/testfailover_controller_unit_test.go +++ b/operator/internal/controller/testfailover_controller_unit_test.go @@ -53,7 +53,7 @@ func newTestFailoverReconciler(t *testing.T, objects ...client.Object) (*TestFai // GVK (and list GVK) registered to create and read it. scheme.AddKnownTypeWithName(managedClusterViewGVK, &unstructured.Unstructured{}) listGVK := managedClusterViewGVK - listGVK.Kind += "List" + listGVK.Kind += listKindSuffix scheme.AddKnownTypeWithName(listGVK, &unstructured.UnstructuredList{}) if err := workv1.Install(scheme); err != nil { t.Fatalf("register work/v1 scheme: %v", err) @@ -443,7 +443,7 @@ func TestFailoverResolvingSourceReuseViewOnRestart(t *testing.T) { list := &unstructured.UnstructuredList{} gvk := managedClusterViewGVK - gvk.Kind += "List" + gvk.Kind += listKindSuffix list.SetGroupVersionKind(gvk) if err := cl.List(ctx, list, client.InNamespace(tf.Spec.SourceCluster)); err != nil { t.Fatalf("list views: %v", err) From 16f9713754db642cb0e6e2d1738683cb165fc2b4 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 17:46:29 +0300 Subject: [PATCH 15/16] operator: regenerate dist/install.yaml with the StorageSiteDeployment CRD Co-Authored-By: Claude Opus 5.5 --- operator/dist/install.yaml | 1068 ++++++++++++++++++++++++++++++++++++ 1 file changed, 1068 insertions(+) diff --git a/operator/dist/install.yaml b/operator/dist/install.yaml index c7c996599..c91db2421 100644 --- a/operator/dist/install.yaml +++ b/operator/dist/install.yaml @@ -11152,6 +11152,1063 @@ spec: --- apiVersion: apiextensions.k8s.io/v1 kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition metadata: annotations: controller-gen.kubebuilder.io/version: v0.21.0 @@ -12270,6 +13327,14 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch - apiGroups: - coordination.k8s.io resources: @@ -12419,6 +13484,7 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks - testfailovers - volumemigrations @@ -12455,6 +13521,7 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers - testfailovers/finalizers - volumemigrations/finalizers @@ -12488,6 +13555,7 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status - testfailovers/status - volumegroupsnapshotops/status From 81cdbf5f081d2c54792d8b1fee1bf0d66780a456 Mon Sep 17 00:00:00 2001 From: michael Date: Sat, 3 Oct 2026 17:49:20 +0300 Subject: [PATCH 16/16] operator: the CSV owns StorageSiteDeployment Co-Authored-By: Claude Opus 5.5 --- .../bases/simplyblock-operator.clusterserviceversion.yaml | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml b/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml index fc70292e9..ced5c9b1d 100644 --- a/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml +++ b/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml @@ -601,6 +601,14 @@ spec: kind: TestFailover name: testfailovers.storage.simplyblock.io version: v1alpha2 + - description: Deploys a managed site's storage from the hub, through OCM. + Discovers the site's nodes, applies the requested sizing to the draft and, + once approved, deploys the storage cluster; status projects the draft and + the resulting storage cluster. + displayName: Storage Site Deployment + kind: StorageSiteDeployment + name: storagesitedeployments.storage.simplyblock.io + version: v1alpha2 description: The Simplyblock Operator helps with installation, operation, and management of Simplyblock Control Planes, Storage Planes, and the CSI Driver. displayName: Simplyblock Operator