diff --git a/atlas-lib/volstack/layers/filesystem.go b/atlas-lib/volstack/layers/filesystem.go index 7d799c903..10d8a472f 100644 --- a/atlas-lib/volstack/layers/filesystem.go +++ b/atlas-lib/volstack/layers/filesystem.go @@ -63,8 +63,19 @@ type FilesystemConfig struct { // formatted as, and it is also the only filesystem the layer will mount: a // device carrying another is refused, because neither reformatting it nor // serving what is on it is safe. + // + // Empty, the plan expresses no opinion: a device carrying a filesystem is + // mounted as what it carries, and a blank one is formatted as DefaultFsType. + // That is what a PersistentVolume without fsType means -- a static PV that + // adopts an existing volume (a test fail-over's clone, 2026-10-03), or one + // Ramen restored without the field -- and turning it into "ext4" before the + // device was looked at made the layer refuse every such XFS volume. FsType string + // DefaultFsType is what a blank device is formatted as when FsType names + // nothing. Empty means ext4. + DefaultFsType string + // StagingPath is where the filesystem is mounted. StagingPath string @@ -107,6 +118,43 @@ type FilesystemConfig struct { // the volume is, and refuses every other device. type Filesystem struct { cfg FilesystemConfig + // detected is the filesystem found on the device when the plan named none. + detected string +} + +// effective is the filesystem this layer acts with: the one the plan named, +// else the one the device carries, else the default for a blank device. +func (f *Filesystem) effective(reading blockdev.Reading) string { + if f.cfg.FsType != "" { + return f.cfg.FsType + } + if reading.Content == blockdev.ContentFilesystem && reading.Type != "" { + f.detected = reading.Type + return reading.Type + } + if f.detected != "" { + return f.detected + } + return f.defaultFsType() +} + +// known is the filesystem this layer stands for once it has acted: named, +// detected, or the default it formats with. +func (f *Filesystem) known() string { + if f.cfg.FsType != "" { + return f.cfg.FsType + } + if f.detected != "" { + return f.detected + } + return f.defaultFsType() +} + +func (f *Filesystem) defaultFsType() string { + if f.cfg.DefaultFsType != "" { + return f.cfg.DefaultFsType + } + return "ext4" } // NewFilesystem returns the filesystem layer for one volume. @@ -206,11 +254,12 @@ func (f *Filesystem) Ensure(ctx context.Context, below volstack.Artifact) (volst // decide the state and again to act on it. The reading itself is not needed // past that, because the filesystem to act on is the one the plan named and // observe has already refused every device carrying another. - state, _, own, err := f.observe(ctx, below) + state, reading, own, err := f.observe(ctx, below) if err != nil { return volstack.Artifact{}, err } if state == volstack.StateReady { + f.effective(reading) return own, nil } @@ -219,7 +268,7 @@ func (f *Filesystem) Ensure(ctx context.Context, below volstack.Artifact) (volst // disagreement, which is the point: the only two ways to reconcile one are to // reformat, which destroys the volume, and to serve the other filesystem, // which hides the misconfiguration until something else acts on it. - fsType := f.cfg.FsType + fsType := f.effective(reading) if state == volstack.StateAbsent { if err := f.cfg.Ops.Format(ctx, dev.Path, fsType, f.formatOptions(below)); err != nil { return volstack.Artifact{}, fmt.Errorf("filesystem: format %s as %s: %w", dev.Path, fsType, err) @@ -321,7 +370,7 @@ func (f *Filesystem) Heal(ctx context.Context, below, _ volstack.Artifact) error return err } - if err := f.cfg.Ops.Mount(ctx, dev.Path, f.cfg.StagingPath, f.cfg.FsType, f.mountFlags()); err != nil { + if err := f.cfg.Ops.Mount(ctx, dev.Path, f.cfg.StagingPath, f.effective(reading), f.mountFlags()); err != nil { return fmt.Errorf("filesystem: remount %s at %s: %w", dev.Path, f.cfg.StagingPath, err) } return nil @@ -379,7 +428,7 @@ type FilesystemParams struct { // recorded is the one the volume asked for, and a teardown needs no more than // that: what is actually on the device is read from the device. func (f *Filesystem) Params() any { - return FilesystemParams{FsType: f.cfg.FsType} + return FilesystemParams{FsType: f.known()} } // agrees reports whether the filesystem on the device is the one the plan asked @@ -436,13 +485,16 @@ func (f *Filesystem) blank( switch { case prior == "": return volstack.StateAbsent, reading, volstack.Artifact{}, nil - case prior != f.cfg.FsType: + case f.cfg.FsType != "" && prior != f.cfg.FsType: return volstack.StateAbsent, reading, volstack.Artifact{}, fmt.Errorf( "filesystem: refusing to stage %s, which is recorded as carrying %s where the plan "+ "asks for %s: reformatting would destroy the volume, and mounting it as %s would "+ "serve a filesystem the plan does not declare", deviceOf(below), prior, f.cfg.FsType, prior) default: + if f.cfg.FsType == "" { + f.detected = prior + } // Recorded as formatted while nothing was found on it: the reading is a // failed probe rather than an empty device, so the filesystem is treated as // present and unmounted. Mounting it is the honest next step, and a mount @@ -486,7 +538,7 @@ func (f *Filesystem) mountFlags() []string { // asked for. That is also the only one the layer acts on, since a device // carrying another is refused rather than reconciled. func (f *Filesystem) strategy() FilesystemLayerStrategy { - return FilesystemStrategyFor(f.cfg.FsType) + return FilesystemStrategyFor(f.known()) } // deviceOf names the device below for an error message, without asserting there diff --git a/atlas-lib/volstack/layers/filesystem_test.go b/atlas-lib/volstack/layers/filesystem_test.go index aa107980b..742791b16 100644 --- a/atlas-lib/volstack/layers/filesystem_test.go +++ b/atlas-lib/volstack/layers/filesystem_test.go @@ -592,3 +592,41 @@ func TestReleaseForcesWhenAPlainUnmountRefuses(t *testing.T) { t.Fatal("a plain unmount refused and the release did not fall back to its force path") } } + +// A plan that names no filesystem -- a static PV without fsType, such as a +// test fail-over's clone or a PV Ramen restored without the field -- mounts +// what the device carries instead of refusing it for not being ext4 +// (2026-10-03), and formats a blank device as the default. +func TestAPlanNamingNoFilesystemMountsWhatTheDeviceCarries(t *testing.T) { + fs := newFakeFS() + l := newFSAsking(t, fs, "", blockdev.Reading{Content: blockdev.ContentFilesystem, Type: "xfs"}, nil) + if _, err := l.Ensure(context.Background(), belowArtifact()); err != nil { + t.Fatal(err) + } + if len(fs.formatted) != 0 { + t.Fatalf("formatted a device that carries a filesystem: %+v", fs.formatted) + } + if len(fs.mounted) != 1 || fs.mounted[0].fsType != "xfs" { + t.Fatalf("mounted %+v, want once as xfs", fs.mounted) + } + if p, _ := l.Params().(FilesystemParams); p.FsType != "xfs" { + t.Fatalf("params %+v, want the detected xfs recorded", p) + } +} + +func TestAPlanNamingNoFilesystemFormatsABlankDeviceAsTheDefault(t *testing.T) { + fs := newFakeFS() + l := NewFilesystem(FilesystemConfig{ + FsType: "", DefaultFsType: "ext4", StagingPath: stagingPath, Ops: fs, + Content: fakeReader{reading: blockdev.Reading{Content: blockdev.ContentBlank}}, + }) + if _, err := l.Ensure(context.Background(), belowArtifact()); err != nil { + t.Fatal(err) + } + if len(fs.formatted) != 1 || fs.formatted[0].fsType != "ext4" { + t.Fatalf("formatted %+v, want once as ext4", fs.formatted) + } + if len(fs.mounted) != 1 || fs.mounted[0].fsType != "ext4" { + t.Fatalf("mounted %+v, want once as ext4", fs.mounted) + } +} diff --git a/atlas-lib/volstack/plans/node.go b/atlas-lib/volstack/plans/node.go index 5aa5c02d8..f31045d40 100644 --- a/atlas-lib/volstack/plans/node.go +++ b/atlas-lib/volstack/plans/node.go @@ -109,6 +109,7 @@ func (n *Node) fabric(connection lvol.Connection) volstack.Layer { func (n *Node) filesystem(volume Volume) volstack.Layer { return layers.NewFilesystem(layers.FilesystemConfig{ FsType: volume.FsType, + DefaultFsType: volume.DefaultFsType, StagingPath: volume.StagingPath, MountFlags: volume.MountFlags, FormatOptions: volume.FormatOptions, diff --git a/atlas-lib/volstack/plans/plans.go b/atlas-lib/volstack/plans/plans.go index c514dae4a..8395a160f 100644 --- a/atlas-lib/volstack/plans/plans.go +++ b/atlas-lib/volstack/plans/plans.go @@ -52,9 +52,14 @@ type Volume struct { // FsType is the filesystem this volume is. It decides what a blank device is // formatted as, and it is also the only filesystem that will be mounted: a - // device carrying another is refused. + // device carrying another is refused. Empty: whatever the device carries, + // and DefaultFsType for a blank one. FsType string + // DefaultFsType is what a blank device is formatted as when FsType names + // nothing. + DefaultFsType string + // MountFlags are the flags the volume asked for, ahead of the ones the // filesystem layer derives from the filesystem itself. MountFlags []string diff --git a/csi-driver/internal/clusters/clusters.go b/csi-driver/internal/clusters/clusters.go index a71616866..daadbacf6 100644 --- a/csi-driver/internal/clusters/clusters.go +++ b/csi-driver/internal/clusters/clusters.go @@ -39,6 +39,10 @@ type Config struct { ClusterID string `json:"cluster_id"` ClusterEndpoint string `json:"cluster_endpoint"` ClusterSecret string `json:"cluster_secret"` + // Local marks a cluster of the site this driver runs on (written by the + // site's operator); an entry without it belongs to another site, kept so + // that a failed-over volume's handle still resolves. + Local bool `json:"local,omitempty"` } // Info is the secret file as a whole. @@ -86,6 +90,24 @@ func Load() (Info, error) { return clusters, nil } +// Local returns the ids of the clusters the secret marks local, and whether +// the secret marks any: a secret written by an operator that predates the +// flag marks none, and callers then fall back to treating every cluster as +// local. +func Local() (map[string]bool, bool, error) { + clusters, err := Load() + if err != nil { + return nil, false, err + } + local := map[string]bool{} + for _, cluster := range clusters.Clusters { + if cluster.Local { + local[cluster.ClusterID] = true + } + } + return local, len(local) > 0, nil +} + // List returns the ID of every cluster in the secret. func List() ([]string, error) { clusters, err := Load() diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go index 242763056..bdc9045c5 100644 --- a/csi-driver/internal/csi/controller/replication.go +++ b/csi-driver/internal/csi/controller/replication.go @@ -9,6 +9,8 @@ package controller import ( "context" "errors" + "fmt" + "sort" "github.com/csi-addons/spec/lib/go/replication" "google.golang.org/grpc/codes" @@ -93,31 +95,153 @@ func volumeIDFrom(req volumeIDCarrier) string { // combining active_lvol_id with ANOTHER record's cluster is exactly the bug // this walk exists to avoid. An empty ActiveLvolID (backend predating the // field) stops after the first hop, the old single-step behavior. +// +// The chain behind a handle alternates between the sites: every fail-over +// adds a hop to the other side. The volume this driver must act on is the +// chain's last member on a LOCAL cluster (the secret marks the site's own +// clusters, clusters.Local), not the chain's end: after an unplanned +// fail-over A->B, Ramen makes the old primary on A secondary, and the +// chain's end is the NEW primary on B. Resolving to the end demoted -- and +// on VR deletion detached -- the live production volume on the other site +// (2026-10-02, WordPress: the demote fenced the live primary's paths and +// took demote snapshots of it; the fail-back never got PeerReady). A secret +// that marks no cluster local (an operator predating the flag) keeps the +// previous behaviour, the chain's active end. func resolveToLocalReplica( ctx context.Context, h *lvol.Handle, client *atlascp.Client, ) (*lvol.Handle, *atlascp.Client, error) { - for range 8 { // one hop per past fail-over; capped far above any real chain + h, client, _, err := resolveReplica(ctx, h, client) + return h, client, err +} + +// resolveReplica is resolveToLocalReplica reporting also whether h has a +// replication relationship at all (known): the member it resolves to is then +// one of a chain the backend records, and a volume of that chain that no +// longer exists is a superseded, reaped old primary -- nothing left to demote +// or detach -- rather than an unknown handle. +func resolveReplica( + ctx context.Context, h *lvol.Handle, client *atlascp.Client, +) (*lvol.Handle, *atlascp.Client, bool, error) { + hops, known, err := resolveChain(ctx, h, client) + if err != nil { + return nil, nil, false, err + } + local, flagged, err := clusters.Local() + if err != nil { + return nil, nil, false, err + } + pick := chooseReplica(hops, local, flagged) + return pick.h, pick.client, known, nil +} + +// resolveChain walks the replication chain behind h: h itself, then every +// IsSource->target hop to the active end. known is whether h has a +// relationship at all. +func resolveChain(ctx context.Context, h *lvol.Handle, client *atlascp.Client) ([]chainHop, bool, error) { + known := false + hops := []chainHop{{h: h, client: client}} + // One hop per past fail-over, never compacted (the PV keeps the original + // handle): a cap of 8 ended the walk one hop short of a ninth move's + // clone (2026-10-03). The bound is a cycle guard, not a length estimate. + visited := map[lvol.VolumeHandle]bool{} + for range maxChainHops { + if visited[h.Handle()] { + return nil, false, fmt.Errorf("replication chain of %s loops at %s", hops[0].h.Handle(), h.Handle()) + } + visited[h.Handle()] = true rel, err := client.GetVolumeReplicationRelationship(ctx, h.Handle()) if err != nil { if errors.Is(err, errs.ErrNotFound) { - return h, client, nil + break } - return nil, nil, err + return nil, false, err } + known = true if !rel.IsSource { - return h, client, nil + break } target := &lvol.Handle{ClusterID: rel.TargetClusterID, PoolRef: rel.TargetPoolID, VolumeID: rel.TargetLvolID} targetClient, err := clusters.ReplicationClient(ctx, target.ClusterID) if err != nil { - return nil, nil, err + return nil, false, err } h, client = target, targetClient + hops = append(hops, chainHop{h: h, client: client}) if rel.ActiveLvolID == "" || rel.ActiveLvolID == rel.TargetLvolID { - return h, client, nil + break } } - return h, client, nil + if len(hops) > maxChainHops { + return nil, false, fmt.Errorf("replication chain of %s did not converge within %d hops", + hops[0].h.Handle(), maxChainHops) + } + return hops, known, nil +} + +// maxChainHops bounds a replication-chain walk: a guard against a looping +// record, far above any chain a volume accumulates in its lifetime. +const maxChainHops = 256 + +// activeEndFallback is where a Resync or a status read goes when the local +// member of the chain is reaped: the chain's active end -- the live primary +// on the other site -- and, as the cluster to fail back to, the local one. +// sbcli's replication_failback is addressed to the failed-over clone and +// re-aims its replication at the original site's node (the recovered-source +// case: only the delta ships), which is exactly the fail-back of a site that +// lost its primary (live 2026-10-02, site A after the unplanned fail-over of +// WordPress: the old primary 80e3e748 was reaped, the clone e3d439ca on B +// holds the data). +func activeEndFallback(hops []chainHop, sourceClusterID string) (chainHop, string) { + end := hops[len(hops)-1] + if local, flagged, err := clusters.Local(); err == nil && flagged { + ids := make([]string, 0, len(local)) + for id := range local { + ids = append(ids, id) + } + sort.Strings(ids) + for _, hop := range hops { + if local[hop.h.ClusterID] { + return end, hop.h.ClusterID + } + } + if len(ids) > 0 { + return end, ids[0] + } + } + return end, sourceClusterID +} + +// reapedChainMember is whether a Replication verb on a resolved chain member +// found the volume gone (404): a superseded old primary the control plane +// has reaped after its fail-over completed (deferred removal; live +// 2026-10-02 on site A). Demoting or detaching it is a no-op that succeeds; +// a 404 on a handle with no relationship stays NotFound. +func reapedChainMember(known bool, ce classifiedError) bool { + return known && status.Code(ce) == codes.NotFound +} + +// chainHop is one member of a replication chain, with the client of its +// cluster. +type chainHop struct { + h *lvol.Handle + client *atlascp.Client +} + +// chooseReplica picks the chain member a Replication RPC acts on: the last +// member on a local cluster when the secret marks local clusters (and the +// chain's end when none of the members is local, e.g. a volume that only +// ever lived elsewhere), else the chain's end. +func chooseReplica(hops []chainHop, local map[string]bool, flagged bool) chainHop { + end := hops[len(hops)-1] + if !flagged { + return end + } + for i := len(hops) - 1; i >= 0; i-- { + if local[hops[i].h.ClusterID] { + return hops[i] + } + } + return end } // EnableVolumeReplication attaches the volume to the policy named by the @@ -224,12 +348,14 @@ func (cs *Server) DisableVolumeReplication( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + h, client, known, err := resolveReplica(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } if err := client.DisableVolumeReplication(ctx, h.Handle()); err != nil { - return nil, classifyDisableVolumeReplicationError(err) + if ce := classifyDisableVolumeReplicationError(err); !reapedChainMember(known, ce) { + return nil, ce + } } return &replication.DisableVolumeReplicationResponse{}, nil } @@ -271,10 +397,16 @@ func (cs *Server) GetVolumeReplicationInfo( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + // The pairing's status is the ACTIVE END's: the volume that holds the + // data and replicates. On the primary site that is the local volume; on + // the secondary site the local member is the demoted or reaped old + // primary, whose status says nothing about the pipe back to this site. + hops, _, err := resolveChain(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } + end := hops[len(hops)-1] + h, client = end.h, end.client info, err := client.GetVolumeReplicationInfo(ctx, h.Handle()) if err != nil { return nil, classifyGetVolumeReplicationInfoError(err) @@ -321,10 +453,20 @@ func (cs *Server) PromoteVolume( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + // Promote is addressed to the chain's ACTIVE END, never to the local + // member: the control plane's failover endpoint takes the volume that + // currently holds the data (the source of the pairing) and creates the + // clone on the replication target -- this site. The local member here + // is the volume being replaced: a demoted old primary, or one the + // control plane already reaped (2026-10-02: promote on site A hit the + // reaped 80e3e748 and 404ed while the live primary e3d439ca on B held + // the data). + hops, _, err := resolveChain(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } + end := hops[len(hops)-1] + h, client = end.h, end.client if err := client.PromoteVolume(ctx, h.Handle(), req.GetForce()); err != nil { return nil, classifyPromoteVolumeError(err) } @@ -375,13 +517,17 @@ func (cs *Server) DemoteVolume( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + h, client, known, err := resolveReplica(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } done, err := client.DemoteVolume(ctx, h.Handle()) if err != nil { - return nil, classifyDemoteVolumeError(err) + ce := classifyDemoteVolumeError(err) + if !reapedChainMember(known, ce) { + return nil, ce + } + done = true } if !done { return nil, status.Error(codes.Aborted, "demote is still converging") @@ -428,11 +574,19 @@ func (cs *Server) ResyncVolume( if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - h, client, err = resolveToLocalReplica(ctx, h, client) + // Resync is addressed to the chain's ACTIVE END with this site as the + // cluster to fail back to: sbcli's replication_failback takes the volume + // that holds the data (the failed-over clone) and re-aims its replication + // at the recovered site's node, shipping only the delta. The local member + // is the demoted or reaped old primary; re-aiming IT configured nothing + // for the live clone (2026-10-02, Gitea after the unplanned fail-over: + // the clones on A had no replication, lastGroupSyncTime stayed empty). + hops, _, err := resolveChain(ctx, h, client) if err != nil { return nil, status.Error(codes.Unavailable, err.Error()) } - sourceClusterID := req.GetParameters()[sourceClusterIDParam] + end, sourceClusterID := activeEndFallback(hops, req.GetParameters()[sourceClusterIDParam]) + h, client = end.h, end.client if err := client.ResyncVolume(ctx, h.Handle(), sourceClusterID); err != nil { return nil, classifyResyncVolumeError(err) } diff --git a/csi-driver/internal/csi/controller/replication_local_test.go b/csi-driver/internal/csi/controller/replication_local_test.go new file mode 100644 index 000000000..83f93c855 --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_local_test.go @@ -0,0 +1,98 @@ +package controller + +import ( + "context" + "testing" + + "github.com/csi-addons/spec/lib/go/replication" + "github.com/simplyblock/atlas/lvol" +) + +// The chain of 2026-10-02 (realbed, WordPress): created on A (0aea, gone), +// failed over to B (6e83), relocated back to A (80e3), failed over to B +// (e3d4, the live primary). Ramen then made the old primary on A secondary. +const ( + siteA, siteB = "A", "B" + oldPrimaryOnA = "80e3" + livePrimaryOnB = "e3d4" +) + +func liveChain() []chainHop { + mk := func(cluster, id string) chainHop { + return chainHop{h: &lvol.Handle{ClusterID: cluster, PoolRef: "p", VolumeID: id}} + } + return []chainHop{mk(siteA, "0aea"), mk(siteB, "6e83"), mk(siteA, oldPrimaryOnA), mk(siteB, livePrimaryOnB)} +} + +func TestChooseReplicaOnTheOldPrimarysSiteIsTheOldPrimaryNotTheLivePrimary(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{siteA: true}, true) + if got.h.VolumeID != oldPrimaryOnA { + t.Fatalf("site A acts on %s, want 80e3 (its own, superseded primary); e3d4 is the live primary on B", got.h.VolumeID) + } +} + +func TestChooseReplicaOnTheNewPrimarysSiteIsTheLivePrimary(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{siteB: true}, true) + if got.h.VolumeID != livePrimaryOnB { + t.Fatalf("site B acts on %s, want e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaWithoutLocalFlagsKeepsTheChainsEnd(t *testing.T) { + got := chooseReplica(liveChain(), nil, false) + if got.h.VolumeID != livePrimaryOnB { + t.Fatalf("unflagged secret: %s, want the chain's end e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaWithNoLocalMemberKeepsTheChainsEnd(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{"C": true}, true) + if got.h.VolumeID != livePrimaryOnB { + t.Fatalf("no member on C: %s, want the chain's end e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaOfAVolumeWithoutARelationshipIsTheVolume(t *testing.T) { + one := liveChain()[:1] + if got := chooseReplica(one, map[string]bool{siteB: true}, true); got.h.VolumeID != "0aea" { + t.Fatalf("got %s", got.h.VolumeID) + } +} + +// The old primary of an unplanned fail-over is reaped by the control plane +// once its fail-over completed; Ramen still demotes it (and deletes its VR) +// when the site returns. Nothing is left to demote: success, not NotFound +// (live 2026-10-02: the VR on site A stayed Degraded on a 404). +func TestDemoteAndDisableOfAReapedChainMemberSucceed(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + gone := "99999999-aaaa-bbbb-cccc-dddddddddddd" + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": gone, + "active_lvol_id": gone, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + ctx := context.Background() + if _, err := cs.DemoteVolume(ctx, &replication.DemoteVolumeRequest{VolumeId: testReplVolID}); err != nil { + t.Fatalf("demote of a reaped chain member: %v", err) + } + disable := &replication.DisableVolumeReplicationRequest{VolumeId: testReplVolID} + if _, err := cs.DisableVolumeReplication(ctx, disable); err != nil { + t.Fatalf("disable of a reaped chain member: %v", err) + } +} + +func TestActiveEndFallbackAimsAtTheChainsEndFromTheLocalSite(t *testing.T) { + // Not flagged in the test secret: the class parameter stays. + end, src := activeEndFallback(liveChain(), "param") + if end.h.VolumeID != livePrimaryOnB || src != "param" { + t.Fatalf("end %s source %s", end.h.VolumeID, src) + } +} diff --git a/csi-driver/internal/csi/node/plan.go b/csi-driver/internal/csi/node/plan.go index e0dcaa747..aa86c704d 100644 --- a/csi-driver/internal/csi/node/plan.go +++ b/csi-driver/internal/csi/node/plan.go @@ -159,6 +159,10 @@ func stackVolume( vc map[string]string, volCap *csi.VolumeCapability, ) plans.Volume { + // Named by the record or the capability; empty otherwise, so the layer + // mounts what the device carries (a static PV without fsType: a test + // fail-over's clone, a PV Ramen restored without the field) and formats + // a blank device as ext4. fsType := stagedFsType(vc, volCap) return plans.Volume{ UUID: deviceLvolID(vc), @@ -167,8 +171,9 @@ func stackVolume( PVCName: vc[csicommon.CSIStorageNameKey], StagingPath: stagingPath, FsType: fsType, + DefaultFsType: "ext4", MountFlags: volumeMountFlags(volCap), - FormatOptions: mount.FormatOptions(fsType, vc, wantsVDO(vc)), + FormatOptions: mount.FormatOptions(fsTypeOrDefault(volCap), vc, wantsVDO(vc)), ReservedBlocksPercent: vc["tune2fs_reserved_blocks"], Encrypted: boolFromContext(vc[csicommon.ParamEncryption]), } diff --git a/csi-driver/internal/csi/node/stage.go b/csi-driver/internal/csi/node/stage.go index 2e52d0cd1..004fdb243 100644 --- a/csi-driver/internal/csi/node/stage.go +++ b/csi-driver/internal/csi/node/stage.go @@ -102,7 +102,7 @@ func (ns *Server) NodeStageVolume( return nil, status.Error(codes.Internal, err.Error()) } - ns.rememberStagedVolume(ctx, volumeID, vc, artifact, req.GetVolumeCapability()) + ns.rememberStagedVolume(ctx, volumeID, vc, artifact, plan, req.GetVolumeCapability()) // The CSI spec passes VolumeContext to this RPC and to nothing after it, so // what the later RPCs need is written beside the staging path. @@ -619,6 +619,7 @@ func (ns *Server) rememberStagedVolume( volumeID string, vc map[string]string, artifact volstack.Artifact, + plan volstack.Plan, volCap *csi.VolumeCapability, ) { if device, ok := artifact.Device(); ok { @@ -636,10 +637,33 @@ func (ns *Server) rememberStagedVolume( // The device carries this filesystem, because the layer either put it there // or refused to stage a device carrying another. fsType := stagedFsType(vc, volCap) + if fsType == "" { + // Nobody named it: the layer mounted what the device carries, or + // formatted a blank device as its default, and knows which. + fsType = planFsType(plan) + } + if fsType == "" { + return + } vc[stagedFsTypeKey] = fsType ns.recordOnDiskFilesystem(ctx, volumeID, vc, fsType) } +// planFsType is the filesystem the plan's filesystem layer stands for after +// it acted (layers.FilesystemParams), "" when the plan has none. +func planFsType(plan volstack.Plan) string { + for _, l := range plan { + recorded, ok := l.(interface{ Params() any }) + if !ok { + continue + } + if p, ok := recorded.Params().(layers.FilesystemParams); ok { + return p.FsType + } + } + return "" +} + // priorFormat is what the volume is recorded as carrying, for the layer that // has to decide whether a device reading blank is empty or merely unreadable. // @@ -704,13 +728,15 @@ func (ns *Server) volumeIsBeingDeleted(ctx context.Context, volumeID string) boo } // stagedFsType returns the filesystem a volume was staged with: the one -// recorded at stage time when it is there, and otherwise the one the volume -// capability asks for, which is all a volume staged by an older driver has. +// recorded at stage time when it is there, else the one the volume capability +// asks for, else "" -- no opinion, which lets the filesystem layer mount what +// the device carries instead of refusing an XFS volume for not being the ext4 +// nobody asked for (a static PV without fsType, 2026-10-03). func stagedFsType(volumeContext map[string]string, volCap *csi.VolumeCapability) string { if fsType := strings.TrimSpace(volumeContext[stagedFsTypeKey]); fsType != "" { return fsType } - return fsTypeOrDefault(volCap) + return volCap.GetMount().GetFsType() } // fsTypeOrDefault returns the requested filesystem type, defaulting to ext4. diff --git a/csi-driver/internal/csi/node/stats.go b/csi-driver/internal/csi/node/stats.go index 6185e9257..0e24a3085 100644 --- a/csi-driver/internal/csi/node/stats.go +++ b/csi-driver/internal/csi/node/stats.go @@ -129,7 +129,19 @@ func redirectToActiveVolume( vc map[string]string, ) map[string]string { client, lvolID := srcClient, srcLvolID - for range 8 { // one hop per past fail-over; capped far above any real chain + // One hop per past fail-over, and the chain never shrinks: the PV keeps + // the original handle while every relocate and fail-over appends a clone, + // so a volume moved nine times is nine hops out. A cap of 8 stranded a + // fail-over's clone behind the ninth hop and the node attached the + // partitioned original instead (2026-10-03, WordPress's fifth move of + // the day). The bound is a cycle guard now, not a length estimate. + visited := map[string]bool{} + for range maxChainHops { + if visited[lvolID] { + klog.Warningf("replication chain for deleted volume %s loops at %s", volumeID, lvolID) + return nil + } + visited[lvolID] = true rel, err := client.GetRelationship(ctx, lvolID) if err != nil || rel == nil { klog.Warningf("replication relationship lookup failed for deleted volume %s (at hop %s): %v", @@ -169,6 +181,11 @@ func redirectToActiveVolume( connInfo["poolID"] = rel.TargetPoolID return connInfo } - klog.Warningf("replication chain for deleted volume %s did not converge within 8 hops", volumeID) + klog.Warningf("replication chain for deleted volume %s did not converge within %d hops", volumeID, maxChainHops) return nil } + +// maxChainHops bounds a replication-chain walk. A chain grows by one member +// per move and is never compacted, so this is a guard against a looping +// record, far above any chain a volume accumulates in its lifetime. +const maxChainHops = 256 diff --git a/csi-driver/internal/csi/node/stats_test.go b/csi-driver/internal/csi/node/stats_test.go index 0a78fa2b7..2ddc16854 100644 --- a/csi-driver/internal/csi/node/stats_test.go +++ b/csi-driver/internal/csi/node/stats_test.go @@ -7,6 +7,7 @@ package node import ( "context" "errors" + "fmt" "testing" "github.com/simplyblock/csi-driver/internal/controlplane" @@ -159,3 +160,97 @@ func TestRedirectToActiveVolumeSinglePairingIsUnchanged(t *testing.T) { t.Errorf("cluster_id = %q, want %q", got, clusterB) } } + +// A volume moved many times: the PV keeps the original handle while every +// relocate and fail-over appends a clone, alternating between the two +// clusters, so the live copy sits one hop further out after each move. The +// walk must reach it however long the chain has grown; a cap of 8 stranded +// the ninth move's clone and the node attached the original on the +// partitioned site instead (2026-10-03). +func TestRedirectToActiveVolumeFollowsALongChain(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + clusterB = "bbbbbbbb-0000-0000-0000-000000000001" + poolA = "aaaaaaaa-0000-0000-0000-00000000000a" + poolB = "bbbbbbbb-0000-0000-0000-00000000000b" + moves = 12 + ) + member := func(i int) string { return fmt.Sprintf("%08d-0000-0000-0000-000000000000", i) } + cluster := func(i int) (string, string) { + if i%2 == 0 { + return clusterA, poolA + } + return clusterB, poolB + } + active := member(moves) + clients := map[string]*fakeRelationshipAPI{ + clusterA + "/" + poolA: { + rels: map[string]*controlplane.ReplicationRelationship{}, + conn: map[string]map[string]string{}, + }, + clusterB + "/" + poolB: { + rels: map[string]*controlplane.ReplicationRelationship{}, + conn: map[string]map[string]string{}, + }, + } + for i := 0; i < moves; i++ { + srcC, srcP := cluster(i) + tgtC, tgtP := cluster(i + 1) + clients[srcC+"/"+srcP].rels[member(i)] = &controlplane.ReplicationRelationship{ + SourceLvolID: member(i), TargetLvolID: member(i + 1), + SourceClusterID: srcC, TargetClusterID: tgtC, TargetPoolID: tgtP, + ActiveLvolID: active, + } + } + activeC, activeP := cluster(moves) + clients[activeC+"/"+activeP].conn[active] = map[string]string{"nqn": "nqn.test:" + active} + + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(_ context.Context, clusterID, poolID string) (controlplane.ClusterAPI, error) { + if c, ok := clients[clusterID+"/"+poolID]; ok { + return c, nil + } + return nil, errors.New("unexpected cluster " + clusterID + "/" + poolID) + } + + connInfo := redirectToActiveVolume(context.Background(), clients[clusterA+"/"+poolA], member(0), + clusterA+":"+poolA+":"+member(0), map[string]string{"hostNQN": "nqn.host"}) + if connInfo == nil { + t.Fatalf("redirect returned nil: the walk gave up before the %d-hop chain's active volume", moves) + } + if got := connInfo["nqn"]; got != "nqn.test:"+active { + t.Errorf("connection nqn = %q, want the active volume's %q", got, "nqn.test:"+active) + } + if got := connInfo[csicommon.ParamClusterID]; got != activeC { + t.Errorf("cluster_id = %q, want the active volume's cluster %q", got, activeC) + } +} + +// A relationship that points back at a member already walked must end the +// walk instead of spinning to the bound. +func TestRedirectToActiveVolumeStopsOnALoop(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + poolA = "aaaaaaaa-0000-0000-0000-00000000000a" + x = "11111111-1111-1111-1111-111111111111" + y = "22222222-2222-2222-2222-222222222222" + ) + client := &fakeRelationshipAPI{rels: map[string]*controlplane.ReplicationRelationship{ + x: { + SourceLvolID: x, TargetLvolID: y, SourceClusterID: clusterA, + TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz", + }, + y: { + SourceLvolID: y, TargetLvolID: x, SourceClusterID: clusterA, + TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz", + }, + }} + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(context.Context, string, string) (controlplane.ClusterAPI, error) { return client, nil } + got := redirectToActiveVolume(context.Background(), client, x, clusterA+":"+poolA+":"+x, map[string]string{}) + if got != nil { + t.Fatalf("a looping chain returned %v, want nil", got) + } +} diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..9c2718a8f --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1057 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml index d68cf6ad7..d8b008df6 100644 --- a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml @@ -206,6 +206,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object diff --git a/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl b/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl new file mode 100644 index 000000000..97a31717e --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl @@ -0,0 +1,61 @@ +{{/* + Control Center helpers — self-contained on purpose. + + The console ships as a fragment (see control-center/ in the monorepo) and + deliberately does NOT call this chart's other helpers: everything is + namespaced under `sbcc.` so it cannot collide, and the fragment renders + unchanged if it is ever lifted out of this chart again. +*/}} + +{{- define "sbcc.name" -}} +{{- default "control-center" .Values.controlCenter.nameOverride | trunc 63 | trimSuffix "-" -}} +{{- end -}} + +{{- define "sbcc.fullname" -}} +{{- if .Values.controlCenter.fullnameOverride -}} +{{- .Values.controlCenter.fullnameOverride | trunc 63 | trimSuffix "-" -}} +{{- else -}} +{{- printf "%s-%s" .Release.Name (include "sbcc.name" .) | trunc 63 | trimSuffix "-" -}} +{{- end -}} +{{- end -}} + +{{- define "sbcc.selectorLabels" -}} +app.kubernetes.io/name: {{ include "sbcc.name" . }} +app.kubernetes.io/instance: {{ .Release.Name }} +{{- end -}} + +{{- define "sbcc.labels" -}} +{{ include "sbcc.selectorLabels" . }} +app.kubernetes.io/component: ui +app.kubernetes.io/part-of: simplyblock +app.kubernetes.io/managed-by: {{ .Release.Service }} +{{- with .Chart }} +helm.sh/chart: {{ printf "%s-%s" .Name .Version | replace "+" "_" | trunc 63 | trimSuffix "-" }} +{{- end }} +{{- end -}} + +{{/* Image reference. The registry falls back to the public one and the tag to + the chart's appVersion — pin controlCenter.image.tag for production, this + chart's appVersion is a floating tag. */}} +{{- define "sbcc.image" -}} +{{- $i := .Values.controlCenter.image -}} +{{- $reg := $i.registry | default "quay.io" -}} +{{- $tag := $i.tag | default (.Chart.AppVersion | default "latest") -}} +{{- printf "%s/%s:%s" $reg $i.repository $tag -}} +{{- end -}} + +{{/* Default upstream URLs. This chart names its objects statically + (simplyblock-operator, simplyblock-prometheus, …) rather than deriving + them from the release, so the fallbacks here are static too. */}} +{{- define "sbcc.operatorUrl" -}} +{{- .Values.controlCenter.operatorUrl | default "http://simplyblock-operator:8080" -}} +{{- end -}} + +{{- define "sbcc.prometheusUrl" -}} +{{- if .Values.controlCenter.prometheusUrl -}} +{{- .Values.controlCenter.prometheusUrl -}} +{{- else -}} +{{- $p := ((.Values.prometheus).simplyblock) | default dict -}} +{{- printf "http://%s:%v" ($p.prometheusURL | default "simplyblock-prometheus") ($p.prometheusPORT | default 9090) -}} +{{- end -}} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center-mock.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center-mock.yaml new file mode 100644 index 000000000..8cc406a13 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center-mock.yaml @@ -0,0 +1,96 @@ +{{/* + sb-mock — the Control Center's test backend (control-center/mock). + + Deployed only for testing and demos: it impersonates the Kubernetes API, + the operator API and Prometheus with generated data, and the console's + Deployment points its proxy here when controlCenter.mock.enabled is true. + Writes persist into the mock's memory and perform no real change; nothing + in this pod touches the actual cluster, and it needs no RBAC at all. +*/}} +{{- if and .Values.controlCenter.enabled .Values.controlCenter.mock.enabled }} +{{- $cc := .Values.controlCenter }} +{{- $mock := $cc.mock }} +{{- $name := printf "%s-mock" (include "sbcc.fullname" .) }} +{{- $reg := $mock.image.registry | default "quay.io" }} +{{- $tag := $mock.image.tag | default (.Chart.AppVersion | default "latest") }} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} + app.kubernetes.io/component: ui-mock +spec: + # one replica on purpose: the world lives in this pod's memory + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: {{ include "sbcc.name" . }}-mock + app.kubernetes.io/instance: {{ .Release.Name }} + template: + metadata: + labels: + app.kubernetes.io/name: {{ include "sbcc.name" . }}-mock + app.kubernetes.io/instance: {{ .Release.Name }} + spec: + automountServiceAccountToken: false + securityContext: + runAsNonRoot: true + runAsUser: 65532 + runAsGroup: 65532 + seccompProfile: + type: RuntimeDefault + {{- with $cc.nodeSelector }} + nodeSelector: {{- toYaml . | nindent 8 }} + {{- end }} + {{- with $cc.tolerations }} + tolerations: {{- toYaml . | nindent 8 }} + {{- end }} + containers: + - name: mock + image: "{{ $reg }}/{{ $mock.image.repository }}:{{ $tag }}" + imagePullPolicy: {{ $mock.image.pullPolicy }} + args: + - --listen=:8080 + - --dataset={{ $mock.dataset }} + {{- if $mock.seed }} + - --seed={{ $mock.seed }} + {{- end }} + - --sim-interval={{ $mock.simInterval }} + - --fail-rate={{ $mock.failRate }} + - --namespace={{ .Release.Namespace }} + ports: + - name: http + containerPort: 8080 + resources: {{- toYaml $mock.resources | nindent 12 }} + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + readinessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 5 + livenessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 20 +--- +apiVersion: v1 +kind: Service +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} + app.kubernetes.io/component: ui-mock +spec: + type: ClusterIP + ports: + - name: http + port: 8080 + targetPort: http + selector: + app.kubernetes.io/name: {{ include "sbcc.name" . }}-mock + app.kubernetes.io/instance: {{ .Release.Name }} +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center-networkpolicy.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center-networkpolicy.yaml new file mode 100644 index 000000000..969fabc3e --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center-networkpolicy.yaml @@ -0,0 +1,76 @@ +{{/* + NetworkPolicy for the Control Center. + + The console has exactly three upstreams — the Kubernetes API, the operator + API and Prometheus — and no route to the internet: React, Babel, the + typefaces and the brand mark are vendored into the image. This caps egress + to those upstreams plus DNS. + + Off by default because the API-server rule is cluster-specific: it ships as + the RFC1918 ranges, and you should narrow controlCenter.networkPolicy + .apiServerCidrs to your API server endpoint (`kubectl get endpoints + kubernetes -n default`) or your service CIDR. A rule with ports but no `to:` + would allow 443 to the whole internet, which is why one is never rendered. +*/}} +{{- if and .Values.controlCenter.enabled .Values.controlCenter.networkPolicy.enabled }} +{{- $cc := .Values.controlCenter }} +{{- $name := include "sbcc.fullname" . }} +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +spec: + podSelector: + matchLabels: + {{- include "sbcc.selectorLabels" . | nindent 6 }} + policyTypes: ["Ingress", "Egress"] + ingress: + # Who may reach the console. Defaults to the release namespace (which + # covers kubectl port-forward); add your ingress controller's namespace + # when the Ingress is enabled. + - from: + {{- range $ns := $cc.networkPolicy.ingressFromNamespaces }} + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: {{ $ns }} + {{- end }} + - podSelector: {} + ports: + - protocol: TCP + port: 8080 + egress: + # The operator API and Prometheus, inside the release namespace. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: {{ .Release.Namespace }} + ports: + - protocol: TCP + port: 8080 + - protocol: TCP + port: 9090 + # The Kubernetes API server. Narrow these CIDRs — see the header comment. + - to: + {{- range $cidr := $cc.networkPolicy.apiServerCidrs }} + - ipBlock: + cidr: {{ $cidr }} + {{- end }} + ports: + - protocol: TCP + port: 443 + - protocol: TCP + port: 6443 + # DNS + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml new file mode 100644 index 000000000..4a638e58c --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center-rbac.yaml @@ -0,0 +1,184 @@ +{{/* + RBAC for the Control Center. + + In serviceaccount mode this role IS the console's authority, for anyone who + can reach the Service. In passthrough mode the pod needs nothing at all — + Kubernetes enforces each user's own RBAC — so set controlCenter.rbac.create + to false there. + + Kept in step with control-center/deploy/k8s/rbac.yaml; that file is the + plain-manifest equivalent of this one. +*/}} +{{- if and .Values.controlCenter.enabled .Values.controlCenter.rbac.create }} +{{- $name := include "sbcc.fullname" . }} +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: {{ $name }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +rules: + # ---- simplyblock entities: read ---- + - apiGroups: ["storage.simplyblock.io"] + resources: + - storageclusters + - storagenodes + - storagenodesets + - storagedevices + - storagepools + - storagebackups + - backuppolicies + - backuprestores + - backupimports + - controlplanes + - replicationpairs + - replicationpolicies + - replicationslots + - volumemigrations + - tasks + - simplyblockdrivers + - clusterdeploymentconfigs + verbs: ["get", "list", "watch"] + - apiGroups: ["storage.simplyblock.io"] + resources: + - storageclusters/status + - storagenodes/status + - storagepools/status + - replicationpolicies/status + - replicationslots/status + verbs: ["get"] + + # ---- day-2 operations ---- + # An action is not a verb against an entity: it is an Ops object the operator + # reconciles. The console creates one, reads its phase, patches spec.abort to + # unwind it, and deletes it only to abort. + - apiGroups: ["storage.simplyblock.io"] + resources: + - storageclusterops + - storagenodeops + - storagedeviceops + - storagepoolops + - storagebackupops + - controlplaneops + - persistentvolumeops + - operatorops + - replicationops + verbs: ["get", "list", "watch", "create", "patch", "delete"] + + # ---- resources the console authors ---- + - apiGroups: ["storage.simplyblock.io"] + resources: ["replicationpairs", "replicationpolicies"] + verbs: ["create", "update", "patch", "delete"] + - apiGroups: ["storage.simplyblock.io"] + resources: ["storagebackups", "backuppolicies", "backuprestores", "backupimports", "volumemigrations"] + verbs: ["create", "update", "patch", "delete"] + - apiGroups: ["storage.simplyblock.io"] + resources: ["storageclusters", "storagepools", "storagenodesets"] + verbs: ["update", "patch"] + + # ---- core Kubernetes ---- + - apiGroups: [""] + resources: ["nodes", "persistentvolumes", "persistentvolumeclaims", "pods", "events", "namespaces"] + verbs: ["get", "list", "watch"] + - apiGroups: [""] + resources: ["pods/log"] + verbs: ["get"] + # A PVC annotation is the whole membership model for replication and for + # backup policies, so patching one is how a volume is attached. + - apiGroups: [""] + resources: ["persistentvolumeclaims"] + verbs: ["patch", "update"] + - apiGroups: ["storage.k8s.io"] + resources: ["storageclasses"] + verbs: ["get", "list", "watch", "patch"] + - apiGroups: ["snapshot.storage.k8s.io"] + resources: ["volumesnapshots", "volumesnapshotcontents", "volumesnapshotclasses"] + verbs: ["get", "list", "watch"] + + # ---- workloads, read only, for the Ramen recipe editor ---- + - apiGroups: ["apps"] + resources: ["deployments", "statefulsets", "daemonsets", "replicasets"] + verbs: ["get", "list"] + - apiGroups: [""] + resources: ["services", "configmaps"] + verbs: ["get", "list"] + - apiGroups: ["networking.k8s.io"] + resources: ["ingresses"] + verbs: ["get", "list"] + - apiGroups: ["kubevirt.io"] + resources: ["virtualmachines", "virtualmachineinstances"] + verbs: ["get", "list"] + + # ---- Ramen: instances only, never the CRDs ---- + - apiGroups: ["ramendr.openshift.io"] + resources: ["drpolicies", "drclusters", "drclusterconfigs", "volumereplicationgroups"] + verbs: ["get", "list", "watch"] + - apiGroups: ["ramendr.openshift.io"] + resources: ["drplacementcontrols"] + verbs: ["get", "list", "watch", "create", "update", "patch", "delete"] + - apiGroups: ["ramendr.openshift.io"] + resources: ["recipes"] + verbs: ["get", "list", "watch", "create", "update", "patch", "delete"] + - apiGroups: ["cluster.open-cluster-management.io"] + resources: ["managedclusters"] + verbs: ["get", "list"] + + # ---- DR hub (dr-simplyblock): every DR screen is a CR ---- + # In serviceaccount mode this is the console's authority for DR: reads on + # everything, the dr-operator writes, the dr-admin writes on plans/paths/ + # restores, and the override verb the hub's webhook checks before admitting a + # readiness override. Absent CRDs cost nothing — the DR section then reports + # that the hub is not installed. Trim to your policy, or use passthrough. + - apiGroups: ["dr.simplyblock.io"] + resources: ["protectionplans", "drpaths", "protectedapplications", "recoveryplans", "recoveryactions", "testbubbles", "testschedules", "restoreactions", "drconfigs"] + verbs: ["get", "list", "watch"] + - apiGroups: ["dr.simplyblock.io"] + resources: ["protectionplans", "drpaths", "protectedapplications", "recoveryplans", "recoveryactions", "testbubbles", "testschedules", "restoreactions"] + verbs: ["create", "update", "patch", "delete"] + - apiGroups: ["dr.simplyblock.io"] + resources: ["recoveryactions"] + verbs: ["override"] + - apiGroups: ["sitemap.simplyblock.io"] + resources: ["siteprofiles", "dhcpservers"] + verbs: ["get", "list", "watch"] + - apiGroups: ["sitemap.simplyblock.io"] + resources: ["dhcpservers"] + verbs: ["create", "update", "patch", "delete"] + # a managed site's storage deployment is requested, sized and approved + # from the hub console (StorageSiteDeployment, carried to the site by OCM) + - apiGroups: ["storage.simplyblock.io"] + resources: ["storagesitedeployments"] + verbs: ["get", "list", "watch", "create", "update", "patch", "delete"] + - apiGroups: ["cluster.open-cluster-management.io"] + resources: ["managedclusters"] + verbs: ["get", "list", "watch"] + # the console asks the API server what its identity may do, to disable controls + - apiGroups: ["authorization.k8s.io"] + resources: ["selfsubjectaccessreviews", "selfsubjectrulesreviews"] + verbs: ["create"] + - apiGroups: ["authentication.k8s.io"] + resources: ["selfsubjectreviews"] + verbs: ["create"] + + # ---- Helm release view ---- + # Helm stores releases as Secrets. This is the only reason the console reads a + # Secret, and a backup credentialsSecretRef is only ever shown by name. + - apiGroups: [""] + resources: ["secrets"] + verbs: ["get", "list"] +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: {{ $name }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: {{ $name }} +subjects: + - kind: ServiceAccount + name: {{ $name }} + namespace: {{ .Release.Namespace }} +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center.yaml new file mode 100644 index 000000000..3a81f5e3b --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/control-center.yaml @@ -0,0 +1,196 @@ +{{/* + Control Center — the simplyblock web UI. + + Installs with the operator when controlCenter.enabled is true, so the console + arrives as part of the deployment rather than as a second thing to install. + Self-contained on purpose: it calls only its own `sbcc.*` helpers (in + _control_center_helpers.tpl), never this chart's. Sources and image build + live in control-center/ at the repository root. +*/}} +{{- if .Values.controlCenter.enabled }} +{{- $cc := .Values.controlCenter }} +{{- $name := include "sbcc.fullname" . }} +apiVersion: v1 +kind: ServiceAccount +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +{{- with $cc.imagePullSecrets }} +imagePullSecrets: + {{- toYaml . | nindent 2 }} +{{- end }} +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +spec: + replicas: {{ $cc.replicas }} + revisionHistoryLimit: 3 + selector: + matchLabels: + {{- include "sbcc.selectorLabels" . | nindent 6 }} + strategy: + type: RollingUpdate + rollingUpdate: + maxUnavailable: 0 + maxSurge: 1 + template: + metadata: + labels: + {{- include "sbcc.selectorLabels" . | nindent 8 }} + {{- with $cc.podAnnotations }} + annotations: + {{- toYaml . | nindent 8 }} + {{- end }} + spec: + serviceAccountName: {{ $name }} + automountServiceAccountToken: true + securityContext: + runAsNonRoot: true + runAsUser: 101 + runAsGroup: 101 + fsGroup: 101 + seccompProfile: + type: RuntimeDefault + {{- with $cc.nodeSelector }} + nodeSelector: {{- toYaml . | nindent 8 }} + {{- end }} + {{- with $cc.tolerations }} + tolerations: {{- toYaml . | nindent 8 }} + {{- end }} + topologySpreadConstraints: + - maxSkew: 1 + topologyKey: kubernetes.io/hostname + whenUnsatisfiable: ScheduleAnyway + labelSelector: + matchLabels: + {{- include "sbcc.selectorLabels" . | nindent 14 }} + containers: + - name: ui + image: {{ include "sbcc.image" . | quote }} + imagePullPolicy: {{ $cc.image.pullPolicy }} + ports: + - name: http + containerPort: 8080 + env: + - name: SB_NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + - name: SB_MODE + value: {{ $cc.mode | default "full" | quote }} + - name: SB_DR_NAMESPACE + value: {{ $cc.drNamespace | default "ramen-ops" | quote }} + - name: SB_AUTH_MODE + value: {{ $cc.authMode | quote }} + {{- /* With the mock enabled every upstream is the sb-mock service: + the console runs unmodified against generated data. */}} + {{- $mockURL := printf "http://%s-mock:8080" $name }} + - name: SB_K8S_API + value: {{ ternary $mockURL $cc.kubernetesApi $cc.mock.enabled | quote }} + # SNI and Host for the API server. Must match a name on the + # apiserver's certificate — see controlCenter.kubernetesApiHost. + # (Ignored for a plain-http upstream, i.e. in mock mode.) + - name: SB_K8S_HOST + value: {{ $cc.kubernetesApiHost | quote }} + - name: SB_K8S_CA_FILE + value: {{ $cc.kubernetesApiCaFile | quote }} + - name: SB_OPERATOR_URL + value: {{ ternary $mockURL (include "sbcc.operatorUrl" .) $cc.mock.enabled | quote }} + - name: SB_HELM_URL + value: {{ ternary $mockURL ($cc.helmUrl | default (include "sbcc.operatorUrl" .)) $cc.mock.enabled | quote }} + - name: SB_PROMETHEUS_URL + value: {{ ternary $mockURL (include "sbcc.prometheusUrl" .) $cc.mock.enabled | quote }} + - name: SB_MOCK + value: "false" + - name: SB_LISTEN_PORT + value: "8080" + - name: SB_TOKEN_REFRESH_SECONDS + value: {{ $cc.tokenRefreshSeconds | quote }} + resources: {{- toYaml $cc.resources | nindent 12 }} + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + startupProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 2 + failureThreshold: 15 + readinessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 10 + livenessProbe: + httpGet: {path: /healthz, port: http} + periodSeconds: 20 + volumeMounts: + # readOnlyRootFilesystem: the scratch mount holds the generated + # config.js, the proxied token and nginx's temp files; conf.d is + # writable because the image renders its server block at startup. + - name: nginx-tmp + mountPath: /tmp/nginx + - name: nginx-confd + mountPath: /etc/nginx/conf.d + volumes: + - name: nginx-tmp + emptyDir: {medium: Memory, sizeLimit: 16Mi} + - name: nginx-confd + emptyDir: {medium: Memory, sizeLimit: 1Mi} +--- +apiVersion: v1 +kind: Service +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} +spec: + type: {{ $cc.service.type }} + ports: + - name: http + port: {{ $cc.service.port }} + targetPort: http + {{- with $cc.service.nodePort }} + nodePort: {{ . }} + {{- end }} + selector: + {{- include "sbcc.selectorLabels" . | nindent 4 }} +{{- if $cc.ingress.enabled }} +--- +apiVersion: networking.k8s.io/v1 +kind: Ingress +metadata: + name: {{ $name }} + namespace: {{ .Release.Namespace }} + labels: + {{- include "sbcc.labels" . | nindent 4 }} + {{- with $cc.ingress.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +spec: + {{- with $cc.ingress.className }} + ingressClassName: {{ . }} + {{- end }} + {{- with $cc.ingress.tls }} + tls: {{- toYaml . | nindent 4 }} + {{- end }} + rules: + - host: {{ required "controlCenter.ingress.host is required when the ingress is enabled" $cc.ingress.host }} + http: + paths: + - path: / + pathType: Prefix + backend: + service: + name: {{ $name }} + port: + name: http +{{- end }} +{{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml b/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml index c6801eb7f..3f8c28526 100644 --- a/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml +++ b/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml @@ -176,6 +176,14 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch - apiGroups: - coordination.k8s.io resources: @@ -325,6 +333,7 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks - testfailovers - volumemigrations @@ -361,6 +370,7 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers - testfailovers/finalizers - volumemigrations/finalizers @@ -394,6 +404,7 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status - testfailovers/status - volumegroupsnapshotops/status diff --git a/helm-charts/charts/simplyblock-operator/values.yaml b/helm-charts/charts/simplyblock-operator/values.yaml index 420bb7951..276d44491 100644 --- a/helm-charts/charts/simplyblock-operator/values.yaml +++ b/helm-charts/charts/simplyblock-operator/values.yaml @@ -974,3 +974,155 @@ tls: # and FoundationDB's peers, so there is nothing left for a deployment to # provision before it can be required. mutual_enabled: true + +# The Control Center — the simplyblock web console (control-center/ in the +# monorepo). Off by default: it proxies the Kubernetes API with the pod's +# ServiceAccount token, so enabling it is a deliberate decision about who may +# reach the Service. See control-center/README.md, "Who can do what". +controlCenter: + enabled: false + + # Naming. The console templates are self-contained and do not borrow the + # chart's helpers, so these control its object names. Default object name: + # -control-center. + nameOverride: "" + fullnameOverride: "" + + image: + # defaults to quay.io when unset + registry: "" + repository: simplyblock-io/control-center + # inherits .Chart.AppVersion when unset. This chart's appVersion is a + # floating tag, so pin a released console tag here for production. + tag: "" + pullPolicy: IfNotPresent + imagePullSecrets: [] + + # Stateless — location lives in the browser — so two replicas cost little and + # keep the console up through a node loss, which is when it matters most. + replicas: 2 + + # full | dr + # + # full: the storage console — clusters, Kubernetes, control plane — with a + # Disaster recovery section that reads the DR hub's dr.simplyblock.io CRDs + # when dr-simplyblock is installed on the same cluster. + # dr: the DR-only console. Nothing but the DR section; no storage CRDs, no + # operator API, no Prometheus. This is what the dr-simplyblock-hub chart + # deploys (console.enabled) on a hub without a simplyblock control plane; + # set it here only to run the same stripped-down console from this chart. + mode: full + # Ramen's ops namespace on the DR hub: where discovered ProtectedApplications + # live and where the console asks the API server what it may do. + drNamespace: ramen-ops + + # serviceaccount | passthrough + # + # serviceaccount: this pod attaches its own token to proxied requests. The + # browser holds no credential. The console's authority is the ClusterRole, + # shared by everyone who can reach it, so put authentication in front. + # + # passthrough: the browser supplies the bearer token and Kubernetes enforces + # that user's own RBAC. Per-user authority and a real audit trail, at the + # cost of needing an auth proxy that injects the token. If you use this, + # set rbac.create=false — the pod needs no permissions of its own. + authMode: serviceaccount + tokenRefreshSeconds: 600 + + kubernetesApi: https://kubernetes.default.svc + # SNI and Host presented to the API server. Must be a name on its + # certificate — change this only together with kubernetesApi. + kubernetesApiHost: kubernetes.default.svc + kubernetesApiCaFile: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + # Empty defaults resolve to this chart's own services: the operator API on + # http://simplyblock-operator:8080 and Prometheus per + # prometheus.simplyblock.prometheusURL/prometheusPORT. + operatorUrl: "" + helmUrl: "" + prometheusUrl: "" + + rbac: + # Required in serviceaccount mode: the proxied token needs these rules or + # every request returns 403 and the console renders empty. Set false only + # with authMode: passthrough, where each user's own RBAC applies instead. + create: true + + service: + type: ClusterIP + port: 80 + # With type NodePort: the node port to pin, else Kubernetes picks one. + nodePort: null + + ingress: + enabled: false + className: nginx + host: "" + annotations: + # Authentication is not optional in serviceaccount mode. Replace this with + # your OIDC/OAuth2 proxy annotations, or keep basic auth as a stop-gap. + nginx.ingress.kubernetes.io/auth-type: basic + nginx.ingress.kubernetes.io/auth-secret: simplyblock-control-center-auth + nginx.ingress.kubernetes.io/auth-realm: simplyblock Control Center + nginx.ingress.kubernetes.io/proxy-body-size: 2m + nginx.ingress.kubernetes.io/proxy-read-timeout: "3600" + tls: [] + + networkPolicy: + # Caps the console's egress to its three upstreams plus DNS. Off by default + # because the API-server CIDRs below are cluster-specific: narrow them to + # your API server endpoint (`kubectl get endpoints kubernetes -n default`) + # or your service CIDR before enabling. + enabled: false + apiServerCidrs: + - 10.0.0.0/8 + - 172.16.0.0/12 + - 192.168.0.0/16 + # Namespaces allowed to reach the console, in addition to the release + # namespace itself — add your ingress controller's namespace when the + # Ingress is enabled, e.g. [ingress-nginx]. + ingressFromNamespaces: [] + + # Deploys sb-mock (control-center/mock) next to the console and points the + # console's proxy at it instead of the real cluster: the Kubernetes API, + # operator API and Prometheus are all impersonated, reads come from a + # generated dataset, and writes persist without performing any change. For + # testing the UI and demo installs only — never enable in production. + mock: + enabled: false + image: + # defaults to quay.io when unset + registry: "" + repository: simplyblock-io/control-center-mock + # inherits .Chart.AppVersion when unset + tag: "" + pullPolicy: IfNotPresent + # small-healthy | medium-degraded | large-scale | dr-failover | chaos, + # or auto for a seeded random pick + dataset: auto + # 0 picks a random seed; a fixed value makes the world reproducible + seed: 0 + # simulator tick (Ops phase progression); "0" disables the simulator so + # e2e suites can drive it deterministically via POST /mockctl/advance + simInterval: 4s + # probability a simulated operation ends Failed + failRate: 0.1 + resources: + requests: + cpu: 10m + memory: 32Mi + limits: + cpu: 200m + memory: 256Mi + + resources: + requests: + cpu: 20m + memory: 48Mi + limits: + cpu: 500m + memory: 192Mi + + nodeSelector: {} + tolerations: [] + podAnnotations: {} + diff --git a/operator/api/v1alpha2/storagesitedeployment_types.go b/operator/api/v1alpha2/storagesitedeployment_types.go new file mode 100644 index 000000000..93a18006c --- /dev/null +++ b/operator/api/v1alpha2/storagesitedeployment_types.go @@ -0,0 +1,298 @@ +// The storage deployment of a managed site, requested from the hub. +// +// A site's storage cluster is built from objects that live on the site's API +// server: an OperatorOps discovery, the ClusterDeploymentConfig draft it +// writes, and the StorageCluster the approved draft expands into. A hub that +// manages the site through Open Cluster Management does not reach that API +// server, so this kind is the hub-side request: it names the site and the +// sizing, and a controller carries the request to the site through a +// ManifestWork and projects the site's answer back through ManagedClusterViews. +// Approval is the same one-way gate the draft has on the site; it is flipped +// here and delivered there. +// +// Deleting the object withdraws nothing on the site: the storage cluster it +// requested stays, as a storage cluster is never torn down by deleting a +// request. Specified by docs/design/control-center-managed-discovery.md of the +// simplyblock-dr repository. + +package v1alpha2 + +import ( + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +// StorageSiteDeploymentPhase is the request's own progress. +// +kubebuilder:validation:Enum=Pending;Discovering;Drafted;Deploying;Online;Failed +type StorageSiteDeploymentPhase string + +const ( + // StorageSiteDeploymentPhasePending is the request before the hub delivered + // anything to the site. + StorageSiteDeploymentPhasePending StorageSiteDeploymentPhase = "Pending" + + // StorageSiteDeploymentPhaseDiscovering is the discovery running on the site: + // the draft is not written yet, or names no node yet. + StorageSiteDeploymentPhaseDiscovering StorageSiteDeploymentPhase = "Discovering" + + // StorageSiteDeploymentPhaseDrafted is a draft with nodes on the site, sized + // as the request says, awaiting approval. + StorageSiteDeploymentPhaseDrafted StorageSiteDeploymentPhase = "Drafted" + + // StorageSiteDeploymentPhaseDeploying is an approved draft expanding into a + // StorageCluster that is not Online yet. + StorageSiteDeploymentPhaseDeploying StorageSiteDeploymentPhase = "Deploying" + + // StorageSiteDeploymentPhaseOnline is the StorageCluster Online on the site. + StorageSiteDeploymentPhaseOnline StorageSiteDeploymentPhase = "Online" + + // StorageSiteDeploymentPhaseFailed is the site's own failure: the draft or the + // StorageCluster failed, or the hub could not deliver the request. + StorageSiteDeploymentPhaseFailed StorageSiteDeploymentPhase = "Failed" +) + +// StorageSiteDiscovery is the discovery the site runs: which nodes are +// inspected. It is the hub-side form of OperatorOps.spec.discover. +type StorageSiteDiscovery struct { + // EnableControlPlaneNodes lets the discovery consider the nodes that run the + // API server. Every server of a small distribution is one, so a three-node + // site has no storage without it. + // +optional + EnableControlPlaneNodes *bool `json:"enableControlPlaneNodes,omitempty"` + + // Workers limits the discovery to these nodes. Empty is every worker. + // +optional + // +listType=set + Workers []string `json:"workers,omitempty"` + + // NodeSelector limits the discovery to the nodes carrying these labels. + // +optional + NodeSelector map[string]string `json:"nodeSelector,omitempty"` +} + +// StorageSiteSizing is the cluster template written onto the draft before it +// is approved: the fields of ClusterDeploymentConfig.spec.cluster a reviewer +// decides. Absent fields keep what the discovery wrote. +type StorageSiteSizing struct { + // Name is the StorageCluster's name on the site. + // +kubebuilder:validation:MaxLength=63 + // +optional + Name string `json:"name,omitempty"` + + // VCPUCount is the number of vCPUs each storage node takes. + // +kubebuilder:validation:Minimum=1 + // +optional + VCPUCount *int32 `json:"vcpuCount,omitempty"` + + // MinHugePagesSize is the hugepage memory each storage node takes, as a + // quantity ("8G"). + // +optional + MinHugePagesSize string `json:"minHugePagesSize,omitempty"` + + // MaxSubsystemCount is the number of NVMe-oF subsystems each node serves. + // +kubebuilder:validation:Minimum=1 + // +optional + MaxSubsystemCount *int32 `json:"maxSubsystemCount,omitempty"` + + // EnableDriveFormat lets the deployment format the devices it takes. + // +optional + EnableDriveFormat *bool `json:"enableDriveFormat,omitempty"` + + // EnableJournalDevice dedicates one device per node to the journal. + // +optional + EnableJournalDevice *bool `json:"enableJournalDevice,omitempty"` + + // Stripe is the erasure-coding layout. + // +optional + Stripe *StripeSpec `json:"stripe,omitempty"` +} + +// StorageSiteDeploymentSpec is the request for one site's storage cluster. +// +kubebuilder:validation:XValidation:rule="!has(oldSelf.approved) || !oldSelf.approved || self.approved",message="approval is one-way: an approved deployment cannot be un-approved" +type StorageSiteDeploymentSpec struct { + // Cluster is the OCM ManagedCluster the storage is deployed on. The request's + // ManifestWork and views live in its namespace on the hub. Immutable. + // +kubebuilder:validation:Required + // +kubebuilder:validation:MinLength=1 + // +kubebuilder:validation:MaxLength=63 + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="cluster is immutable" + Cluster string `json:"cluster"` + + // SiteNamespace is the simplyblock operator's namespace on the site, where + // the discovery and the draft live. + // +kubebuilder:default=simplyblock + // +kubebuilder:validation:MaxLength=63 + // +optional + SiteNamespace string `json:"siteNamespace,omitempty"` + + // DraftName is the ClusterDeploymentConfig the discovery writes on the site + // and the request sizes and approves. Immutable. + // +kubebuilder:default=site-draft + // +kubebuilder:validation:MaxLength=63 + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="draftName is immutable" + // +optional + DraftName string `json:"draftName,omitempty"` + + // Discover is the discovery the site runs first. Changing it runs another + // discovery, which rewrites the draft. + // +optional + Discover StorageSiteDiscovery `json:"discover,omitempty"` + + // Sizing is written onto the draft's cluster template once the draft exists, + // so the reviewer sees the sized draft before approving it. + // +optional + Sizing *StorageSiteSizing `json:"sizing,omitempty"` + + // Approved is the review gate, delivered to the draft on the site. One-way, + // as the draft's own gate is. + // +kubebuilder:default=false + // +optional + Approved bool `json:"approved"` +} + +// StorageSiteDraft is the draft as the site reports it. +type StorageSiteDraft struct { + // Name is the ClusterDeploymentConfig on the site. + Name string `json:"name"` + + // Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + // Failed). + // +optional + Phase string `json:"phase,omitempty"` + + // Message is what the site says about the draft: validation findings while + // it is a draft, the expansion's step afterwards. + // +optional + Message string `json:"message,omitempty"` + + // Approved is whether the draft is approved on the site. + // +optional + Approved bool `json:"approved,omitempty"` + + // Cluster is the draft's cluster template, with the sizing applied. + // +optional + Cluster *ClusterTemplate `json:"cluster,omitempty"` + + // NodeSets are the nodes and devices the discovery found, for review. + // +optional + NodeSets []NodeSet `json:"nodeSets,omitempty"` + + // NodeRefs are the StorageNode objects the expansion created. + // +optional + // +listType=set + NodeRefs []string `json:"nodeRefs,omitempty"` +} + +// StorageSiteNode is one storage node of the deployed cluster, as the site +// reports it. +type StorageSiteNode struct { + // Name is the StorageNode object on the site. + Name string `json:"name"` + + // Phase is the node's phase on the site. + // +optional + Phase string `json:"phase,omitempty"` + + // Hostname is the Kubernetes node it runs on. + // +optional + Hostname string `json:"hostname,omitempty"` +} + +// StorageSiteCluster is the StorageCluster the approved draft produced. +type StorageSiteCluster struct { + // Name is the StorageCluster object on the site. + Name string `json:"name"` + + // UUID is the storage cluster's id in the control plane, which a + // StorageClass names in cluster_id. + // +optional + UUID string `json:"uuid,omitempty"` + + // Phase is the StorageCluster's phase on the site. + // +optional + Phase string `json:"phase,omitempty"` + + // Pool is the pool the cluster was created with, which a StorageClass names + // in pool_name. + // +optional + Pool string `json:"pool,omitempty"` + + // Nodes are the cluster's storage nodes. + // +optional + // +listType=map + // +listMapKey=name + Nodes []StorageSiteNode `json:"nodes,omitempty"` +} + +// StorageSiteDeploymentStatus is what the site reports back, projected. +type StorageSiteDeploymentStatus struct { + // Phase is the request's own progress. + // +optional + Phase StorageSiteDeploymentPhase `json:"phase,omitempty"` + + // Message is the reason the phase is what it is: one sentence, replaced as + // the request moves, and never a log. + // +optional + Message string `json:"message,omitempty"` + + // ObservedGeneration is the generation the rest of this status was computed + // from. + // +optional + ObservedGeneration int64 `json:"observedGeneration,omitempty"` + + // WorkName is the ManifestWork carrying the request to the site. + // +optional + WorkName string `json:"workName,omitempty"` + + // Draft is the draft as the site reports it. + // +optional + Draft *StorageSiteDraft `json:"draft,omitempty"` + + // StorageCluster is the cluster the approved draft produced. + // +optional + StorageCluster *StorageSiteCluster `json:"storageCluster,omitempty"` + + // Conditions: Delivered (the work is applied on the site), Discovered (the + // draft names nodes), Approved (the site's draft is approved), Ready (the + // StorageCluster is Online). + // +optional + // +listType=map + // +listMapKey=type + Conditions []metav1.Condition `json:"conditions,omitempty"` +} + +// +kubebuilder:object:root=true +// +kubebuilder:subresource:status +// +kubebuilder:resource:scope=Namespaced,shortName=sbsd +// +kubebuilder:printcolumn:name="Cluster",type=string,JSONPath=".spec.cluster" +// +kubebuilder:printcolumn:name="Approved",type=boolean,JSONPath=".spec.approved" +// +kubebuilder:printcolumn:name="Phase",type=string,JSONPath=".status.phase" +// +kubebuilder:printcolumn:name="Draft",type=string,JSONPath=".status.draft.phase" +// +kubebuilder:printcolumn:name="Storage",type=string,JSONPath=".status.storageCluster.phase" +// +kubebuilder:printcolumn:name="Message",type=string,JSONPath=".status.message",priority=1 +// +kubebuilder:printcolumn:name="Age",type=date,JSONPath=".metadata.creationTimestamp" + +// StorageSiteDeployment requests a managed site's storage cluster from the hub: +// a discovery on the site, the sizing of the draft it writes, and the approval +// that expands the draft into a StorageCluster. The hub carries the request +// through OCM and projects the site's draft and cluster into the status. +// Deleting the request leaves the storage cluster alone. +type StorageSiteDeployment struct { + metav1.TypeMeta `json:",inline"` + metav1.ObjectMeta `json:"metadata,omitempty"` + + Spec StorageSiteDeploymentSpec `json:"spec,omitempty"` + Status StorageSiteDeploymentStatus `json:"status,omitempty"` +} + +// +kubebuilder:object:root=true + +// StorageSiteDeploymentList contains a list of StorageSiteDeployment. +type StorageSiteDeploymentList struct { + metav1.TypeMeta `json:",inline"` + metav1.ListMeta `json:"metadata,omitempty"` + Items []StorageSiteDeployment `json:"items"` +} + +func init() { + SchemeBuilder.Register(&StorageSiteDeployment{}, &StorageSiteDeploymentList{}) +} diff --git a/operator/api/v1alpha2/testfailover_types.go b/operator/api/v1alpha2/testfailover_types.go index de033a936..cf2e1e852 100644 --- a/operator/api/v1alpha2/testfailover_types.go +++ b/operator/api/v1alpha2/testfailover_types.go @@ -175,6 +175,12 @@ type TestFailoverClone struct { // volume. // +optional SourceFSType string `json:"sourceFSType,omitempty"` + // SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + // carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + // claim that omitted the mode defaulted to Filesystem and the kubelet asked + // the node plugin to mount a raw guest disk (2026-10-03). + // +optional + SourceVolumeMode string `json:"sourceVolumeMode,omitempty"` } // TestFailoverReport is the evidence a drill produces. diff --git a/operator/api/v1alpha2/zz_generated.deepcopy.go b/operator/api/v1alpha2/zz_generated.deepcopy.go index 4913a6e6f..9dc34a411 100644 --- a/operator/api/v1alpha2/zz_generated.deepcopy.go +++ b/operator/api/v1alpha2/zz_generated.deepcopy.go @@ -3566,6 +3566,257 @@ func (in *StoragePoolStatus) DeepCopy() *StoragePoolStatus { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteCluster) DeepCopyInto(out *StorageSiteCluster) { + *out = *in + if in.Nodes != nil { + in, out := &in.Nodes, &out.Nodes + *out = make([]StorageSiteNode, len(*in)) + copy(*out, *in) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteCluster. +func (in *StorageSiteCluster) DeepCopy() *StorageSiteCluster { + if in == nil { + return nil + } + out := new(StorageSiteCluster) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeployment) DeepCopyInto(out *StorageSiteDeployment) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ObjectMeta.DeepCopyInto(&out.ObjectMeta) + in.Spec.DeepCopyInto(&out.Spec) + in.Status.DeepCopyInto(&out.Status) +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeployment. +func (in *StorageSiteDeployment) DeepCopy() *StorageSiteDeployment { + if in == nil { + return nil + } + out := new(StorageSiteDeployment) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *StorageSiteDeployment) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentList) DeepCopyInto(out *StorageSiteDeploymentList) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ListMeta.DeepCopyInto(&out.ListMeta) + if in.Items != nil { + in, out := &in.Items, &out.Items + *out = make([]StorageSiteDeployment, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentList. +func (in *StorageSiteDeploymentList) DeepCopy() *StorageSiteDeploymentList { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentList) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *StorageSiteDeploymentList) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentSpec) DeepCopyInto(out *StorageSiteDeploymentSpec) { + *out = *in + in.Discover.DeepCopyInto(&out.Discover) + if in.Sizing != nil { + in, out := &in.Sizing, &out.Sizing + *out = new(StorageSiteSizing) + (*in).DeepCopyInto(*out) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentSpec. +func (in *StorageSiteDeploymentSpec) DeepCopy() *StorageSiteDeploymentSpec { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentSpec) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentStatus) DeepCopyInto(out *StorageSiteDeploymentStatus) { + *out = *in + if in.Draft != nil { + in, out := &in.Draft, &out.Draft + *out = new(StorageSiteDraft) + (*in).DeepCopyInto(*out) + } + if in.StorageCluster != nil { + in, out := &in.StorageCluster, &out.StorageCluster + *out = new(StorageSiteCluster) + (*in).DeepCopyInto(*out) + } + if in.Conditions != nil { + in, out := &in.Conditions, &out.Conditions + *out = make([]metav1.Condition, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentStatus. +func (in *StorageSiteDeploymentStatus) DeepCopy() *StorageSiteDeploymentStatus { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentStatus) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDiscovery) DeepCopyInto(out *StorageSiteDiscovery) { + *out = *in + if in.EnableControlPlaneNodes != nil { + in, out := &in.EnableControlPlaneNodes, &out.EnableControlPlaneNodes + *out = new(bool) + **out = **in + } + if in.Workers != nil { + in, out := &in.Workers, &out.Workers + *out = make([]string, len(*in)) + copy(*out, *in) + } + if in.NodeSelector != nil { + in, out := &in.NodeSelector, &out.NodeSelector + *out = make(map[string]string, len(*in)) + for key, val := range *in { + (*out)[key] = val + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDiscovery. +func (in *StorageSiteDiscovery) DeepCopy() *StorageSiteDiscovery { + if in == nil { + return nil + } + out := new(StorageSiteDiscovery) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDraft) DeepCopyInto(out *StorageSiteDraft) { + *out = *in + if in.Cluster != nil { + in, out := &in.Cluster, &out.Cluster + *out = new(ClusterTemplate) + (*in).DeepCopyInto(*out) + } + if in.NodeSets != nil { + in, out := &in.NodeSets, &out.NodeSets + *out = make([]NodeSet, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } + if in.NodeRefs != nil { + in, out := &in.NodeRefs, &out.NodeRefs + *out = make([]string, len(*in)) + copy(*out, *in) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDraft. +func (in *StorageSiteDraft) DeepCopy() *StorageSiteDraft { + if in == nil { + return nil + } + out := new(StorageSiteDraft) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteNode) DeepCopyInto(out *StorageSiteNode) { + *out = *in +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteNode. +func (in *StorageSiteNode) DeepCopy() *StorageSiteNode { + if in == nil { + return nil + } + out := new(StorageSiteNode) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteSizing) DeepCopyInto(out *StorageSiteSizing) { + *out = *in + if in.VCPUCount != nil { + in, out := &in.VCPUCount, &out.VCPUCount + *out = new(int32) + **out = **in + } + if in.MaxSubsystemCount != nil { + in, out := &in.MaxSubsystemCount, &out.MaxSubsystemCount + *out = new(int32) + **out = **in + } + if in.EnableDriveFormat != nil { + in, out := &in.EnableDriveFormat, &out.EnableDriveFormat + *out = new(bool) + **out = **in + } + if in.EnableJournalDevice != nil { + in, out := &in.EnableJournalDevice, &out.EnableJournalDevice + *out = new(bool) + **out = **in + } + if in.Stripe != nil { + in, out := &in.Stripe, &out.Stripe + *out = new(StripeSpec) + (*in).DeepCopyInto(*out) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteSizing. +func (in *StorageSiteSizing) DeepCopy() *StorageSiteSizing { + if in == nil { + return nil + } + out := new(StorageSiteSizing) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *StripeSpec) DeepCopyInto(out *StripeSpec) { *out = *in diff --git a/operator/cmd/main.go b/operator/cmd/main.go index b3258ecd9..004db856e 100644 --- a/operator/cmd/main.go +++ b/operator/cmd/main.go @@ -962,8 +962,19 @@ func main() { setupLog.Error(err, "unable to create controller", "controller", "TestFailover") os.Exit(1) } + // A managed site's storage deployment is requested from the hub through + // the same work API, so the controller is hub-only too. + if err := (&controller.StorageSiteDeploymentReconciler{ + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorder("storagesitedeployment-controller"), + }).SetupWithManager(mgr); err != nil { + setupLog.Error(err, "unable to create controller", "controller", "StorageSiteDeployment") + os.Exit(1) + } } else { - setupLog.Info("OCM ManifestWork resource not served; skipping TestFailover controller (hub-only)", + setupLog.Info("OCM ManifestWork resource not served; skipping the TestFailover and "+ + "StorageSiteDeployment controllers (hub-only)", "groupVersion", ocmWorkGroupVersion, "resource", ocmManifestWorkResource) } // +kubebuilder:scaffold:builder diff --git a/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml b/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..9c2718a8f --- /dev/null +++ b/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1057 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml index d68cf6ad7..d8b008df6 100644 --- a/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml +++ b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml @@ -206,6 +206,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object diff --git a/operator/config/crd/kustomization.yaml b/operator/config/crd/kustomization.yaml index 760b4ffb6..520d8f4be 100644 --- a/operator/config/crd/kustomization.yaml +++ b/operator/config/crd/kustomization.yaml @@ -31,6 +31,7 @@ resources: - bases/storage.simplyblock.io_controlplaneops.yaml - bases/storage.simplyblock.io_storagedeviceops.yaml - bases/storage.simplyblock.io_testfailovers.yaml +- bases/storage.simplyblock.io_storagesitedeployments.yaml # +kubebuilder:scaffold:crdkustomizeresource patches: [] diff --git a/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml b/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml index fc70292e9..ced5c9b1d 100644 --- a/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml +++ b/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml @@ -601,6 +601,14 @@ spec: kind: TestFailover name: testfailovers.storage.simplyblock.io version: v1alpha2 + - description: Deploys a managed site's storage from the hub, through OCM. + Discovers the site's nodes, applies the requested sizing to the draft and, + once approved, deploys the storage cluster; status projects the draft and + the resulting storage cluster. + displayName: Storage Site Deployment + kind: StorageSiteDeployment + name: storagesitedeployments.storage.simplyblock.io + version: v1alpha2 description: The Simplyblock Operator helps with installation, operation, and management of Simplyblock Control Planes, Storage Planes, and the CSI Driver. displayName: Simplyblock Operator diff --git a/operator/config/rbac/role.yaml b/operator/config/rbac/role.yaml index 4d6684b03..1335dea16 100644 --- a/operator/config/rbac/role.yaml +++ b/operator/config/rbac/role.yaml @@ -176,6 +176,14 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch - apiGroups: - coordination.k8s.io resources: @@ -325,6 +333,7 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks - testfailovers - volumemigrations @@ -361,6 +370,7 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers - testfailovers/finalizers - volumemigrations/finalizers @@ -394,6 +404,7 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status - testfailovers/status - volumegroupsnapshotops/status diff --git a/operator/dist/install.yaml b/operator/dist/install.yaml index 97cc7fead..c91db2421 100644 --- a/operator/dist/install.yaml +++ b/operator/dist/install.yaml @@ -11152,6 +11152,1063 @@ spec: --- apiVersion: apiextensions.k8s.io/v1 kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition metadata: annotations: controller-gen.kubebuilder.io/version: v0.21.0 @@ -11463,6 +12520,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object @@ -12263,6 +13327,14 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch - apiGroups: - coordination.k8s.io resources: @@ -12412,6 +13484,7 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks - testfailovers - volumemigrations @@ -12448,6 +13521,7 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers - testfailovers/finalizers - volumemigrations/finalizers @@ -12481,6 +13555,7 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status - testfailovers/status - volumegroupsnapshotops/status diff --git a/operator/internal/controller/storagesitedeployment_controller.go b/operator/internal/controller/storagesitedeployment_controller.go new file mode 100644 index 000000000..87108d4fc --- /dev/null +++ b/operator/internal/controller/storagesitedeployment_controller.go @@ -0,0 +1,723 @@ +// The StorageSiteDeployment controller carries a managed site's storage +// deployment request from the hub to the site and projects the site's answer +// back. +// +// It never holds a site kubeconfig: every write to the site is a ManifestWork +// in the site's hub namespace, every read a ManagedClusterView there, the same +// two primitives the TestFailover controller uses. The work carries the +// OperatorOps discovery first; once the site has written a draft with nodes, it +// carries a server-side apply of the draft's sizing, and when the request is +// approved, the draft's approval. The views project the draft, the +// StorageCluster the approved draft expands into, and that cluster's nodes. +// +// The request withdraws nothing on deletion: the work is released with its +// resources orphaned, so a storage cluster is never torn down by deleting the +// request that asked for it. See docs/design/control-center-managed-discovery.md +// in the simplyblock-dr repository. + +package controller + +import ( + "context" + "crypto/sha256" + "encoding/json" + "fmt" + "sort" + "time" + + corev1 "k8s.io/api/core/v1" + apiequality "k8s.io/apimachinery/pkg/api/equality" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/client-go/tools/events" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" + logf "sigs.k8s.io/controller-runtime/pkg/log" + + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/controllers/pool" +) + +// storageSiteDeploymentIDLabel tags the work and the views of one request, so +// its release can enumerate them. +const storageSiteDeploymentIDLabel = "storage.simplyblock.io/site-deployment" + +// finalizerStorageSiteDeployment holds the request until its work and views +// are released. The work is released with its resources orphaned: the +// discovery, the draft and the storage cluster stay on the site. +const finalizerStorageSiteDeployment = "storage.simplyblock.io/storagesitedeployment-release" + +// hubDeployFieldManager is the field manager the work agent applies the +// draft's sizing and approval with, so the discovery's own fields on the draft +// are left to their owner. +const hubDeployFieldManager = "hub-deploy" + +// storageSiteDeploymentRequeue is how long the reconcile waits before reading +// the site's views again while the request is in progress. +const storageSiteDeploymentRequeue = 15 * time.Second + +// storageSiteDeploymentOnlineRequeue keeps an Online request's projection of +// the storage cluster fresh without polling the site hard. +const storageSiteDeploymentOnlineRequeue = 2 * time.Minute + +// maxStorageSiteNodeViews bounds the per-node views a request keeps: list +// views are not supported by ManagedClusterView, so there is one per node +// named in the draft's nodeRefs. +const maxStorageSiteNodeViews = 64 + +// annotationStorageClusterDefaultPool is the annotation the StorageCluster +// controller records the cluster's first pool under (controllers/cluster). +const annotationStorageClusterDefaultPool = "storage.simplyblock.io/default-pool" + +// Conditions of a request. +const ( + ConditionStorageSiteDelivered = "Delivered" + ConditionStorageSiteDiscovered = "Discovered" + ConditionStorageSiteApproved = "Approved" + ConditionStorageSiteReady = "Ready" +) + +// StorageSiteDeploymentReconciler reconciles a StorageSiteDeployment object. +type StorageSiteDeploymentReconciler struct { + client.Client + Scheme *runtime.Scheme + Recorder events.EventRecorder +} + +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments/status,verbs=get;update;patch +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments/finalizers,verbs=update +// +kubebuilder:rbac:groups=work.open-cluster-management.io,resources=manifestworks,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=view.open-cluster-management.io,resources=managedclusterviews,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=cluster.open-cluster-management.io,resources=managedclusters,verbs=get;list;watch +// +kubebuilder:rbac:groups=events.k8s.io,resources=events,verbs=create;patch + +// Reconcile carries the request to the site and projects the site's answer: +// it ensures the finalizer and the work, reads the draft, the storage cluster +// and its nodes through views, and derives the phase from what they report. +func (r *StorageSiteDeploymentReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) { + log := logf.FromContext(ctx) + + var sd simplyblockv1alpha2.StorageSiteDeployment + if err := r.Get(ctx, req.NamespacedName, &sd); err != nil { + return ctrl.Result{}, client.IgnoreNotFound(err) + } + + if !sd.DeletionTimestamp.IsZero() { + return r.reconcileDeletion(ctx, &sd) + } + if !controllerutil.ContainsFinalizer(&sd, finalizerStorageSiteDeployment) { + controllerutil.AddFinalizer(&sd, finalizerStorageSiteDeployment) + if err := r.Update(ctx, &sd); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{Requeue: true}, nil + } + + work, err := r.ensureWork(ctx, &sd) + if err != nil { + log.Error(err, "ensure ManifestWork") + return ctrl.Result{}, err + } + return r.project(ctx, &sd, work) +} + +// ensureWork creates or updates the request's ManifestWork: the discovery +// always, and the draft's sizing and approval once the site has a draft with +// nodes to apply them to. +func (r *StorageSiteDeploymentReconciler) ensureWork(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment) (*workv1.ManifestWork, error) { + want, err := r.manifestWork(sd) + if err != nil { + return nil, err + } + var have workv1.ManifestWork + err = r.Get(ctx, client.ObjectKeyFromObject(want), &have) + if apierrors.IsNotFound(err) { + if err := r.Create(ctx, want); err != nil { + return nil, err + } + return want, nil + } + if err != nil { + return nil, err + } + if !apiequality.Semantic.DeepEqual(have.Spec, want.Spec) || !apiequality.Semantic.DeepEqual(have.Labels, want.Labels) { + have.Spec = want.Spec + have.Labels = want.Labels + if err := r.Update(ctx, &have); err != nil { + return nil, err + } + } + return &have, nil +} + +// manifestWork builds the request's work as it should be now. Its resources +// are orphaned on delete: the discovery, the draft and what it expanded into +// are the site's, and deleting the request must not take them away. +func (r *StorageSiteDeploymentReconciler) manifestWork(sd *simplyblockv1alpha2.StorageSiteDeployment) (*workv1.ManifestWork, error) { + ns := siteNamespace(sd) + draft := draftName(sd) + labels := map[string]string{storageSiteDeploymentIDLabel: string(sd.UID)} + + ops := map[string]any{ + "apiVersion": simplyblockv1alpha2.GroupVersion.String(), + "kind": "OperatorOps", + "metadata": map[string]any{"name": discoveryName(sd), "namespace": ns, "labels": labels}, + "spec": map[string]any{ + "action": string(simplyblockv1alpha2.OperatorOpsActionDiscover), + "discover": discoverSpec(sd), + }, + } + manifests := []workv1.Manifest{} + var configs []workv1.ManifestConfigOption + raw, err := json.Marshal(ops) + if err != nil { + return nil, fmt.Errorf("marshal discovery: %w", err) + } + manifests = append(manifests, workv1.Manifest{RawExtension: runtime.RawExtension{Raw: raw}}) + + // The sizing and the approval are applied onto the draft the discovery + // wrote, never before it exists: an apply that created the draft would + // make a document with no nodes, which the site refuses. + if draftHasNodes(sd.Status.Draft) && (sd.Spec.Sizing != nil || sd.Spec.Approved) { + spec := map[string]any{"approved": sd.Spec.Approved} + if tpl := sizingTemplate(sd.Spec.Sizing); len(tpl) > 0 { + spec["cluster"] = tpl + } + cdc := map[string]any{ + "apiVersion": simplyblockv1alpha2.GroupVersion.String(), + "kind": "ClusterDeploymentConfig", + "metadata": map[string]any{"name": draft, "namespace": ns}, + "spec": spec, + } + raw, err := json.Marshal(cdc) + if err != nil { + return nil, fmt.Errorf("marshal draft apply: %w", err) + } + manifests = append(manifests, workv1.Manifest{RawExtension: runtime.RawExtension{Raw: raw}}) + configs = append(configs, workv1.ManifestConfigOption{ + ResourceIdentifier: workv1.ResourceIdentifier{ + Group: simplyblockv1alpha2.GroupVersion.Group, Resource: "clusterdeploymentconfigs", Namespace: ns, Name: draft, + }, + UpdateStrategy: &workv1.UpdateStrategy{ + Type: workv1.UpdateStrategyTypeServerSideApply, + ServerSideApply: &workv1.ServerSideApplyConfig{ + Force: true, + FieldManager: hubDeployFieldManager, + }, + }, + FeedbackRules: []workv1.FeedbackRule{{ + Type: workv1.JSONPathsType, + JsonPaths: []workv1.JsonPath{ + {Name: "phase", Path: ".status.phase"}, + {Name: "approved", Path: ".spec.approved"}, + }, + }}, + }) + } + + return &workv1.ManifestWork{ + ObjectMeta: metav1.ObjectMeta{ + Name: workName(sd), + Namespace: sd.Spec.Cluster, + Labels: labels, + }, + Spec: workv1.ManifestWorkSpec{ + Workload: workv1.ManifestsTemplate{Manifests: manifests}, + ManifestConfigs: configs, + DeleteOption: &workv1.DeleteOption{PropagationPolicy: workv1.DeletePropagationPolicyTypeOrphan}, + }, + }, nil +} + +// discoverSpec is the OperatorOps discover block of the request. +func discoverSpec(sd *simplyblockv1alpha2.StorageSiteDeployment) map[string]any { + d := map[string]any{"configName": draftName(sd)} + if sd.Spec.Discover.EnableControlPlaneNodes != nil { + d["enableControlPlaneNodes"] = *sd.Spec.Discover.EnableControlPlaneNodes + } + if len(sd.Spec.Discover.Workers) > 0 { + d["workers"] = sd.Spec.Discover.Workers + } + if len(sd.Spec.Discover.NodeSelector) > 0 { + d["nodeSelector"] = sd.Spec.Discover.NodeSelector + } + return d +} + +// sizingTemplate is the draft's cluster template fields the request sets. +// Only the stated fields are applied, so what the discovery wrote stays. +func sizingTemplate(s *simplyblockv1alpha2.StorageSiteSizing) map[string]any { + tpl := map[string]any{} + if s == nil { + return tpl + } + if s.Name != "" { + tpl["name"] = s.Name + } + if s.VCPUCount != nil { + tpl["vcpuCount"] = *s.VCPUCount + } + if s.MinHugePagesSize != "" { + tpl["minHugePagesSize"] = s.MinHugePagesSize + } + if s.MaxSubsystemCount != nil { + tpl["maxSubsystemCount"] = *s.MaxSubsystemCount + } + if s.EnableDriveFormat != nil { + tpl["enableDriveFormat"] = *s.EnableDriveFormat + } + if s.EnableJournalDevice != nil { + tpl["enableJournalDevice"] = *s.EnableJournalDevice + } + if s.Stripe != nil { + stripe := map[string]any{} + if s.Stripe.DataChunks != nil { + stripe["dataChunks"] = *s.Stripe.DataChunks + } + if s.Stripe.ParityChunks != nil { + stripe["parityChunks"] = *s.Stripe.ParityChunks + } + tpl["stripe"] = stripe + } + return tpl +} + +// project reads the site's views and derives the request's phase. +func (r *StorageSiteDeploymentReconciler) project(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, work *workv1.ManifestWork) (ctrl.Result, error) { + delivered, deliveryMessage := workDelivery(work) + + draftObj, haveDraft, err := r.projected(ctx, sd, "draft", "clusterdeploymentconfigs", draftName(sd), siteNamespace(sd)) + if err != nil { + return ctrl.Result{}, err + } + if !haveDraft { + if deliveryMessage != "" { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, deliveryMessage, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + setCondition(s, ConditionStorageSiteDelivered, false, "NotApplied", deliveryMessage) + }) + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering, + fmt.Sprintf("waiting for site %s to write draft %s/%s", sd.Spec.Cluster, siteNamespace(sd), draftName(sd)), + func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + s.WorkName = work.Name + setCondition(s, ConditionStorageSiteDelivered, delivered, deliveryReason(delivered), deliveryNote(delivered, deliveryMessage)) + }) + } + + var cdc simplyblockv1alpha2.ClusterDeploymentConfig + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(draftObj, &cdc); err != nil { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + fmt.Sprintf("the site's draft could not be read: %v", err), nil) + } + draft := projectDraft(&cdc) + base := func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + s.WorkName = work.Name + s.Draft = draft + setCondition(s, ConditionStorageSiteDelivered, delivered, deliveryReason(delivered), deliveryNote(delivered, deliveryMessage)) + setCondition(s, ConditionStorageSiteDiscovered, draftHasNodes(draft), "Nodes", fmt.Sprintf("%d node(s) in the draft", draftNodeCount(draft))) + setCondition(s, ConditionStorageSiteApproved, draft.Approved, "SiteDraft", fmt.Sprintf("the site's draft approved=%t", draft.Approved)) + } + + if !draftHasNodes(draft) { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering, + "the site's draft names no node yet: discovery is running", base) + } + if deliveryMessage != "" { + // The draft exists, so the message is about the sizing or the approval + // the work could not apply: the site refused it. + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, deliveryMessage, base) + } + switch { + case cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseFailed: + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + "the site's draft failed: "+orDefault(cdc.Status.Message, "no message"), base) + case !cdc.Spec.Approved: + msg := "the draft awaits approval" + if sd.Spec.Approved { + msg = "approval requested; waiting for the site's draft to take it" + } else if sd.Spec.Sizing != nil && !sizingApplied(sd.Spec.Sizing, cdc.Spec.Cluster) { + msg = "the draft awaits approval; the sizing is being applied" + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted, msg, base) + } + + // Approved on the site: follow the StorageCluster it expands into. + clusterName := cdc.Status.ClusterRef + if clusterName == "" && cdc.Spec.Cluster != nil { + clusterName = cdc.Spec.Cluster.Name + } + if clusterName == "" { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying, + "the draft is approved; waiting for the site to name its StorageCluster", base) + } + scObj, haveSC, err := r.projected(ctx, sd, "cluster", "storageclusters", clusterName, siteNamespace(sd)) + if err != nil { + return ctrl.Result{}, err + } + sc := &simplyblockv1alpha2.StorageSiteCluster{Name: clusterName} + if haveSC { + var cluster simplyblockv1alpha2.StorageCluster + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(scObj, &cluster); err == nil { + sc.UUID = cluster.Status.UUID + sc.Phase = string(cluster.Status.Phase) + sc.Pool = cluster.Annotations[annotationStorageClusterDefaultPool] + } + } + if sc.Pool == "" { + sc.Pool = pool.DefaultPoolName(clusterName) + } + nodes, err := r.projectNodes(ctx, sd, draft.NodeRefs) + if err != nil { + return ctrl.Result{}, err + } + sc.Nodes = nodes + withCluster := func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + base(s) + s.StorageCluster = sc + } + + switch { + case sc.Phase == string(simplyblockv1alpha2.StorageClusterPhaseOnline) && sc.UUID != "": + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline, + fmt.Sprintf("StorageCluster %s is Online (%d node(s))", clusterName, len(nodes)), func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + withCluster(s) + setCondition(s, ConditionStorageSiteReady, true, "Online", "the StorageCluster is Online") + }) + case cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanded && sc.Phase == string(simplyblockv1alpha2.StorageClusterPhaseUnavailable): + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + fmt.Sprintf("StorageCluster %s is Unavailable after the expansion", clusterName), withCluster) + } + msg := fmt.Sprintf("draft %s, StorageCluster %s %s", orDefault(string(cdc.Status.Phase), "Expanding"), clusterName, orDefault(sc.Phase, "not reported yet")) + if step := cdc.Status.Step.State; step != "" && cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanding { + msg = fmt.Sprintf("draft Expanding (%s), StorageCluster %s %s", step, clusterName, orDefault(sc.Phase, "not reported yet")) + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying, msg, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + withCluster(s) + setCondition(s, ConditionStorageSiteReady, false, "Deploying", msg) + }) +} + +// projectNodes projects the draft's StorageNodes, one view each. +func (r *StorageSiteDeploymentReconciler) projectNodes(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, refs []string) ([]simplyblockv1alpha2.StorageSiteNode, error) { + sorted := append([]string(nil), refs...) + sort.Strings(sorted) + if len(sorted) > maxStorageSiteNodeViews { + sorted = sorted[:maxStorageSiteNodeViews] + } + nodes := make([]simplyblockv1alpha2.StorageSiteNode, 0, len(sorted)) + for i, name := range sorted { + obj, ok, err := r.projected(ctx, sd, fmt.Sprintf("node-%d", i), "storagenodes", name, siteNamespace(sd)) + if err != nil { + return nil, err + } + n := simplyblockv1alpha2.StorageSiteNode{Name: name} + if ok { + var sn simplyblockv1alpha2.StorageNode + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(obj, &sn); err == nil { + n.Phase = string(sn.Status.Phase) + n.Hostname = sn.Status.Hostname + } + } + nodes = append(nodes, n) + } + return nodes, nil +} + +// projected reads one of the request's views, creating it when it is missing, +// and reports whether the site has projected the object yet. +func (r *StorageSiteDeploymentReconciler) projected(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, suffix, resource, name, namespace string) (map[string]interface{}, bool, error) { + viewName := storageSiteViewName(sd, suffix) + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + getErr := r.Get(ctx, client.ObjectKey{Namespace: sd.Spec.Cluster, Name: viewName}, view) + if apierrors.IsNotFound(getErr) { + want := newStorageSiteView(sd, viewName, resource, name, namespace) + if createErr := r.Create(ctx, want); createErr != nil { + return nil, false, createErr + } + return nil, false, nil + } + if getErr != nil { + return nil, false, getErr + } + // A view that names another object (the draft's cluster changed) is + // pointed at the right one. + scope, _, _ := unstructured.NestedMap(view.Object, "spec", "scope") + if scope["name"] != name || scope["resource"] != resource { + want := newStorageSiteView(sd, viewName, resource, name, namespace) + view.Object["spec"] = want.Object["spec"] + if err := r.Update(ctx, view); err != nil { + return nil, false, err + } + return nil, false, nil + } + result, found, nestedErr := unstructured.NestedMap(view.Object, "status", "result") + if nestedErr != nil || !found || len(result) == 0 { + return nil, false, nil + } + return result, true, nil +} + +// newStorageSiteView asks the site to project one object back to the hub. +func newStorageSiteView(sd *simplyblockv1alpha2.StorageSiteDeployment, name, resource, targetName, targetNamespace string) *unstructured.Unstructured { + scope := map[string]interface{}{"resource": resource, "name": targetName} + if targetNamespace != "" { + scope["namespace"] = targetNamespace + } + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + view.SetNamespace(sd.Spec.Cluster) + view.SetName(name) + view.SetLabels(map[string]string{storageSiteDeploymentIDLabel: string(sd.UID)}) + _ = unstructured.SetNestedMap(view.Object, scope, "spec", "scope") + return view +} + +// setPhase writes the phase, the message and the mutation, and records a +// phase change as an event. +func (r *StorageSiteDeploymentReconciler) setPhase(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, phase simplyblockv1alpha2.StorageSiteDeploymentPhase, message string, mutate func(*simplyblockv1alpha2.StorageSiteDeploymentStatus)) (ctrl.Result, error) { + previous := sd.Status.Phase + if err := r.patchStatus(ctx, sd, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + if mutate != nil { + mutate(s) + } + s.Phase = phase + s.Message = message + }); err != nil { + return ctrl.Result{}, err + } + if previous != phase && r.Recorder != nil { + kind := corev1.EventTypeNormal + if phase == simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed { + kind = corev1.EventTypeWarning + } + r.Recorder.Eventf(sd, nil, kind, string(phase), string(phase), "%s", message) + } + switch phase { + case simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline: + return ctrl.Result{RequeueAfter: storageSiteDeploymentOnlineRequeue}, nil + case simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed: + // A failure on the site may clear (a node comes back, a draft is + // corrected on the site): keep reading at the slow cadence. + return ctrl.Result{RequeueAfter: storageSiteDeploymentOnlineRequeue}, nil + } + return ctrl.Result{RequeueAfter: storageSiteDeploymentRequeue}, nil +} + +// patchStatus applies mutate to the status and writes it with the generation +// it was computed from. +func (r *StorageSiteDeploymentReconciler) patchStatus(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, mutate func(*simplyblockv1alpha2.StorageSiteDeploymentStatus)) error { + base := client.MergeFrom(sd.DeepCopy()) + mutate(&sd.Status) + sd.Status.ObservedGeneration = sd.Generation + return r.Status().Patch(ctx, sd, base) +} + +// reconcileDeletion releases the request's work and views and removes the +// finalizer. The work orphans its resources, so nothing on the site goes. +func (r *StorageSiteDeploymentReconciler) reconcileDeletion(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment) (ctrl.Result, error) { + if !controllerutil.ContainsFinalizer(sd, finalizerStorageSiteDeployment) { + return ctrl.Result{}, nil + } + var work workv1.ManifestWork + err := r.Get(ctx, client.ObjectKey{Namespace: sd.Spec.Cluster, Name: workName(sd)}, &work) + switch { + case apierrors.IsNotFound(err): + case err != nil: + return ctrl.Result{}, err + case work.DeletionTimestamp.IsZero(): + if err := client.IgnoreNotFound(r.Delete(ctx, &work)); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + default: + // Deleting: wait for the work agent to release it. + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + } + views := &unstructured.UnstructuredList{} + listGVK := managedClusterViewGVK + listGVK.Kind += listKindSuffix + views.SetGroupVersionKind(listGVK) + if err := r.List(ctx, views, client.InNamespace(sd.Spec.Cluster), client.MatchingLabels{storageSiteDeploymentIDLabel: string(sd.UID)}); err != nil && !meta.IsNoMatchError(err) { + return ctrl.Result{}, err + } + for i := range views.Items { + if err := client.IgnoreNotFound(r.Delete(ctx, &views.Items[i])); err != nil { + return ctrl.Result{}, err + } + } + controllerutil.RemoveFinalizer(sd, finalizerStorageSiteDeployment) + return ctrl.Result{}, r.Update(ctx, sd) +} + +// workDelivery reads the work's status: whether every manifest is applied, +// and the first manifest's refusal when one is not. +func workDelivery(work *workv1.ManifestWork) (applied bool, message string) { + if work == nil { + return false, "" + } + for _, m := range work.Status.ResourceStatus.Manifests { + for _, c := range m.Conditions { + if c.Type == workv1.ManifestApplied && c.Status == metav1.ConditionFalse { + return false, fmt.Sprintf("the site did not apply %s %s: %s", m.ResourceMeta.Kind, m.ResourceMeta.Name, c.Message) + } + } + } + for _, c := range work.Status.Conditions { + if c.Type == workv1.WorkApplied { + return c.Status == metav1.ConditionTrue, "" + } + } + return false, "" +} + +func deliveryReason(delivered bool) string { + if delivered { + return "Applied" + } + return "Pending" +} + +func deliveryNote(delivered bool, message string) string { + switch { + case message != "": + return message + case delivered: + return "the work is applied on the site" + } + return "the work is not applied on the site yet" +} + +// projectDraft is the draft as the status carries it. +func projectDraft(cdc *simplyblockv1alpha2.ClusterDeploymentConfig) *simplyblockv1alpha2.StorageSiteDraft { + d := &simplyblockv1alpha2.StorageSiteDraft{ + Name: cdc.Name, + Phase: string(cdc.Status.Phase), + Message: cdc.Status.Message, + Approved: cdc.Spec.Approved, + NodeSets: cdc.Spec.NodeSets, + NodeRefs: cdc.Status.NodeRefs, + } + if cdc.Spec.Cluster != nil { + d.Cluster = cdc.Spec.Cluster.DeepCopy() + } + if d.Phase == "" { + d.Phase = string(simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft) + } + return d +} + +// draftHasNodes is whether the draft names at least one worker. +func draftHasNodes(d *simplyblockv1alpha2.StorageSiteDraft) bool { + return draftNodeCount(d) > 0 +} + +func draftNodeCount(d *simplyblockv1alpha2.StorageSiteDraft) int { + if d == nil { + return 0 + } + n := 0 + for _, set := range d.NodeSets { + for _, g := range set.Groups { + n += len(g.Workers) + } + } + return n +} + +// sizingApplied is whether the draft's template carries the request's sizing. +func sizingApplied(s *simplyblockv1alpha2.StorageSiteSizing, tpl *simplyblockv1alpha2.ClusterTemplate) bool { + if s == nil { + return true + } + if tpl == nil { + return false + } + eq32 := func(want, have *int32) bool { return want == nil || (have != nil && *have == *want) } + eqBool := func(want, have *bool) bool { return want == nil || (have != nil && *have == *want) } + if s.Name != "" && tpl.Name != s.Name { + return false + } + if s.MinHugePagesSize != "" && tpl.MinHugePagesSize != s.MinHugePagesSize { + return false + } + if !eq32(s.VCPUCount, tpl.VCPUCount) || !eq32(s.MaxSubsystemCount, tpl.MaxSubsystemCount) || + !eqBool(s.EnableDriveFormat, tpl.EnableDriveFormat) || !eqBool(s.EnableJournalDevice, tpl.EnableJournalDevice) { + return false + } + if s.Stripe != nil { + if tpl.Stripe == nil || !eq32(s.Stripe.DataChunks, tpl.Stripe.DataChunks) || !eq32(s.Stripe.ParityChunks, tpl.Stripe.ParityChunks) { + return false + } + } + return true +} + +func setCondition(s *simplyblockv1alpha2.StorageSiteDeploymentStatus, kind string, ok bool, reason, message string) { + status := metav1.ConditionFalse + if ok { + status = metav1.ConditionTrue + } + meta.SetStatusCondition(&s.Conditions, metav1.Condition{Type: kind, Status: status, Reason: reason, Message: message, ObservedGeneration: s.ObservedGeneration}) +} + +func orDefault(s, d string) string { + if s == "" { + return d + } + return s +} + +func siteNamespace(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return orDefault(sd.Spec.SiteNamespace, "simplyblock") +} + +func draftName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return orDefault(sd.Spec.DraftName, "site-draft") +} + +// storageSiteHash is a short, deterministic id of the request for the names +// of its work and views, which must be bounded and found again on a restart. +func storageSiteHash(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + h := sha256.Sum256([]byte(sd.Namespace + "/" + sd.Name)) + return fmt.Sprintf("%x", h[:6]) +} + +func workName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return "sbsd-" + storageSiteHash(sd) +} + +func storageSiteViewName(sd *simplyblockv1alpha2.StorageSiteDeployment, suffix string) string { + return "sbsd-" + storageSiteHash(sd) + "-" + suffix +} + +// discoveryName is the OperatorOps on the site. It carries a hash of the +// discovery's parameters, so a changed discovery is a new run rather than an +// edit of a finished one. +func discoveryName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + raw, _ := json.Marshal(sd.Spec.Discover) + h := sha256.Sum256(raw) + return fmt.Sprintf("hub-discover-%s-%x", draftName(sd), h[:3]) +} + +// SetupWithManager registers the controller. The work and the views live in +// the site's namespace on the hub, where an owner reference to the request +// cannot point, so the site's answers are read on the requeue cadence of the +// phase rather than through a watch. +func (r *StorageSiteDeploymentReconciler) SetupWithManager(mgr ctrl.Manager) error { + return ctrl.NewControllerManagedBy(mgr). + For(&simplyblockv1alpha2.StorageSiteDeployment{}). + Named("storagesitedeployment"). + Complete(r) +} + +// listKindSuffix turns a kind into its list kind (StorageCluster -> +// StorageClusterList) for an unstructured list read. +const listKindSuffix = "List" diff --git a/operator/internal/controller/storagesitedeployment_controller_unit_test.go b/operator/internal/controller/storagesitedeployment_controller_unit_test.go new file mode 100644 index 000000000..4bb445d6f --- /dev/null +++ b/operator/internal/controller/storagesitedeployment_controller_unit_test.go @@ -0,0 +1,335 @@ +package controller + +import ( + "context" + "encoding/json" + "testing" + + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/types" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" +) + +const ( + testSDNamespace = "simplyblock" + testSDName = "site-a" + testSDCluster = "site-a" +) + +func newStorageSiteDeploymentReconciler(t *testing.T, objects ...client.Object) (*StorageSiteDeploymentReconciler, client.Client) { + t.Helper() + scheme := newTestScheme(t) + scheme.AddKnownTypeWithName(managedClusterViewGVK, &unstructured.Unstructured{}) + listGVK := managedClusterViewGVK + listGVK.Kind += listKindSuffix + scheme.AddKnownTypeWithName(listGVK, &unstructured.UnstructuredList{}) + if err := workv1.Install(scheme); err != nil { + t.Fatalf("register work/v1 scheme: %v", err) + } + cl := newTestClient(t, scheme, + []client.Object{&simplyblockv1alpha2.StorageSiteDeployment{}, &workv1.ManifestWork{}}, + objects...) + return &StorageSiteDeploymentReconciler{Client: cl, Scheme: scheme, Recorder: &fakeRecorder{}}, cl +} + +func newSiteDeployment(mutate func(*simplyblockv1alpha2.StorageSiteDeployment)) *simplyblockv1alpha2.StorageSiteDeployment { + enable := true + sd := &simplyblockv1alpha2.StorageSiteDeployment{ + ObjectMeta: metav1.ObjectMeta{Name: testSDName, Namespace: testSDNamespace, UID: types.UID("uid-site-a")}, + Spec: simplyblockv1alpha2.StorageSiteDeploymentSpec{ + Cluster: testSDCluster, + Discover: simplyblockv1alpha2.StorageSiteDiscovery{EnableControlPlaneNodes: &enable}, + }, + } + if mutate != nil { + mutate(sd) + } + return sd +} + +// reconcileSD runs the reconcile n times (the first adds the finalizer) and +// returns the request as stored. +func reconcileSD(t *testing.T, r *StorageSiteDeploymentReconciler, cl client.Client, n int) *simplyblockv1alpha2.StorageSiteDeployment { + t.Helper() + req := ctrl.Request{NamespacedName: types.NamespacedName{Namespace: testSDNamespace, Name: testSDName}} + for i := 0; i < n; i++ { + if _, err := r.Reconcile(context.Background(), req); err != nil { + t.Fatalf("reconcile %d: %v", i, err) + } + } + var sd simplyblockv1alpha2.StorageSiteDeployment + if err := cl.Get(context.Background(), req.NamespacedName, &sd); err != nil { + t.Fatalf("get request: %v", err) + } + return &sd +} + +func getSiteWork(t *testing.T, cl client.Client, sd *simplyblockv1alpha2.StorageSiteDeployment) *workv1.ManifestWork { + t.Helper() + var w workv1.ManifestWork + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: sd.Spec.Cluster, Name: workName(sd)}, &w); err != nil { + t.Fatalf("get ManifestWork: %v", err) + } + return &w +} + +// workManifests decodes the work's manifests by kind. +func workManifests(t *testing.T, w *workv1.ManifestWork) map[string]map[string]any { + t.Helper() + out := map[string]map[string]any{} + for _, m := range w.Spec.Workload.Manifests { + obj := map[string]any{} + if err := json.Unmarshal(m.Raw, &obj); err != nil { + t.Fatalf("decode manifest: %v", err) + } + out[obj["kind"].(string)] = obj + } + return out +} + +// projectSiteObject writes what the site's view controller would project. +func projectSiteObject(t *testing.T, cl client.Client, sd *simplyblockv1alpha2.StorageSiteDeployment, suffix string, obj any) { + t.Helper() + v := getView(t, cl, sd.Spec.Cluster, storageSiteViewName(sd, suffix)) + result, err := runtime.DefaultUnstructuredConverter.ToUnstructured(obj) + if err != nil { + t.Fatalf("convert projection: %v", err) + } + setViewResult(t, cl, v, result) +} + +func siteDraft(approved bool, phase simplyblockv1alpha2.ClusterDeploymentConfigPhase, mutate func(*simplyblockv1alpha2.ClusterDeploymentConfig)) *simplyblockv1alpha2.ClusterDeploymentConfig { + cdc := &simplyblockv1alpha2.ClusterDeploymentConfig{ + TypeMeta: metav1.TypeMeta{APIVersion: simplyblockv1alpha2.GroupVersion.String(), Kind: "ClusterDeploymentConfig"}, + ObjectMeta: metav1.ObjectMeta{Name: "site-draft", Namespace: "simplyblock"}, + Spec: simplyblockv1alpha2.ClusterDeploymentConfigSpec{ + Approved: approved, + NodeSets: []simplyblockv1alpha2.NodeSet{{ + Name: "default", + Groups: []simplyblockv1alpha2.NodeGroup{{Workers: []string{"n1", "n2", "n3"}}}, + }}, + }, + } + cdc.Status.Phase = phase + if mutate != nil { + mutate(cdc) + } + return cdc +} + +func TestStorageSiteDeploymentDeliversTheDiscoveryAndWaitsForTheDraft(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(nil)) + sd := reconcileSD(t, r, cl, 2) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering { + t.Fatalf("phase = %q, want Discovering", sd.Status.Phase) + } + w := getSiteWork(t, cl, sd) + if w.Spec.DeleteOption == nil || w.Spec.DeleteOption.PropagationPolicy != workv1.DeletePropagationPolicyTypeOrphan { + t.Errorf("work delete option = %+v, want Orphan: a request must never take the site's storage away", w.Spec.DeleteOption) + } + ms := workManifests(t, w) + ops, ok := ms["OperatorOps"] + if !ok || len(ms) != 1 { + t.Fatalf("manifests = %v, want the discovery alone before the draft exists", keys(ms)) + } + discover := ops["spec"].(map[string]any)["discover"].(map[string]any) + if discover["configName"] != "site-draft" || discover["enableControlPlaneNodes"] != true { + t.Errorf("discover = %v, want configName site-draft and control-plane nodes enabled", discover) + } + // The draft view exists so the site can project the draft. + getView(t, cl, testSDCluster, storageSiteViewName(sd, "draft")) +} + +func TestStorageSiteDeploymentSizesTheDraftOnceItNamesNodes(t *testing.T) { + vcpu := int32(8) + data, parity := int32(1), int32(1) + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Sizing = &simplyblockv1alpha2.StorageSiteSizing{ + Name: "sb-site-a", VCPUCount: &vcpu, MinHugePagesSize: "8G", + Stripe: &simplyblockv1alpha2.StripeSpec{DataChunks: &data, ParityChunks: &parity}, + } + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, nil)) + sd = reconcileSD(t, r, cl, 2) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted { + t.Fatalf("phase = %q (%s), want Drafted", sd.Status.Phase, sd.Status.Message) + } + if got := draftNodeCount(sd.Status.Draft); got != 3 { + t.Errorf("projected draft nodes = %d, want 3", got) + } + ms := workManifests(t, getSiteWork(t, cl, sd)) + cdc, ok := ms["ClusterDeploymentConfig"] + if !ok { + t.Fatalf("manifests = %v, want the draft's sizing applied", keys(ms)) + } + spec := cdc["spec"].(map[string]any) + if spec["approved"] != false { + t.Errorf("approved = %v, want false before the request is approved", spec["approved"]) + } + cluster := spec["cluster"].(map[string]any) + if cluster["name"] != "sb-site-a" || cluster["vcpuCount"] != float64(8) || cluster["minHugePagesSize"] != "8G" { + t.Errorf("cluster template = %v, want the sizing", cluster) + } + if _, ok := spec["nodeSets"]; ok { + t.Error("the sizing apply must not carry nodeSets: they are the discovery's") + } + if !meta.IsStatusConditionTrue(sd.Status.Conditions, ConditionStorageSiteDiscovered) { + t.Error("Discovered condition is not True for a draft with nodes") + } +} + +func TestStorageSiteDeploymentApprovalFollowsTheClusterToOnline(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, nil)) + // One pass projects the draft, the next delivers the approval onto it. + sd = reconcileSD(t, r, cl, 2) + + ms := workManifests(t, getSiteWork(t, cl, sd)) + if ms["ClusterDeploymentConfig"]["spec"].(map[string]any)["approved"] != true { + t.Fatalf("the work does not carry the approval: %v", ms["ClusterDeploymentConfig"]) + } + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted { + t.Fatalf("phase = %q, want Drafted until the site's draft takes the approval", sd.Status.Phase) + } + + // The site's draft takes it and starts expanding. + projectSiteObject(t, cl, sd, "draft", siteDraft(true, simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanding, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Status.ClusterRef = "sb-site-a" + c.Status.NodeRefs = []string{"sn-1", "sn-2", "sn-3"} + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying { + t.Fatalf("phase = %q (%s), want Deploying", sd.Status.Phase, sd.Status.Message) + } + + // The cluster comes Online. + sc := &simplyblockv1alpha2.StorageCluster{ + TypeMeta: metav1.TypeMeta{APIVersion: simplyblockv1alpha2.GroupVersion.String(), Kind: "StorageCluster"}, + ObjectMeta: metav1.ObjectMeta{Name: "sb-site-a", Namespace: "simplyblock"}, + } + sc.Status.UUID = "8f8dd277-1544-4177-9a74-e0f66eb2672c" + sc.Status.Phase = simplyblockv1alpha2.StorageClusterPhaseOnline + projectSiteObject(t, cl, sd, "cluster", sc) + sd = reconcileSD(t, r, cl, 1) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline { + t.Fatalf("phase = %q (%s), want Online", sd.Status.Phase, sd.Status.Message) + } + if sd.Status.StorageCluster == nil || sd.Status.StorageCluster.UUID != sc.Status.UUID { + t.Fatalf("storageCluster = %+v, want the site's uuid", sd.Status.StorageCluster) + } + if sd.Status.StorageCluster.Pool == "" { + t.Error("storageCluster.pool is empty: a StorageClass needs it") + } + if n := len(sd.Status.StorageCluster.Nodes); n != 3 { + t.Errorf("projected nodes = %d, want 3", n) + } + if !meta.IsStatusConditionTrue(sd.Status.Conditions, ConditionStorageSiteReady) { + t.Error("Ready condition is not True for an Online cluster") + } +} + +func TestStorageSiteDeploymentReportsAFailedDraft(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(true, simplyblockv1alpha2.ClusterDeploymentConfigPhaseFailed, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Status.Message = "node n2 has no free device" + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed { + t.Fatalf("phase = %q, want Failed", sd.Status.Phase) + } + if sd.Status.Message != "the site's draft failed: node n2 has no free device" { + t.Errorf("message = %q, want the site's own reason", sd.Status.Message) + } +} + +func TestStorageSiteDeploymentADraftWithoutNodesIsStillDiscovering(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Spec.NodeSets = nil + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering { + t.Fatalf("phase = %q, want Discovering", sd.Status.Phase) + } + if _, ok := workManifests(t, getSiteWork(t, cl, sd))["ClusterDeploymentConfig"]; ok { + t.Error("the approval was delivered onto a draft without nodes") + } +} + +func TestStorageSiteDeploymentAChangedDiscoveryIsANewRun(t *testing.T) { + a := newSiteDeployment(nil) + b := newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Discover.Workers = []string{"n1"} + }) + if discoveryName(a) == discoveryName(b) { + t.Fatalf("discovery name %q did not change with the discovery's parameters", discoveryName(a)) + } + if discoveryName(a) != discoveryName(newSiteDeployment(nil)) { + t.Fatal("the discovery name is not stable for the same parameters") + } +} + +func TestStorageSiteDeploymentDeletionOrphansTheSitesStorage(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(nil)) + sd := reconcileSD(t, r, cl, 2) + if err := cl.Delete(context.Background(), sd); err != nil { + t.Fatalf("delete request: %v", err) + } + // First pass deletes the work; the fake client removes it at once. + reconcileOnce := func() { + req := ctrl.Request{NamespacedName: types.NamespacedName{Namespace: testSDNamespace, Name: testSDName}} + if _, err := r.Reconcile(context.Background(), req); err != nil { + t.Fatalf("reconcile: %v", err) + } + } + reconcileOnce() + reconcileOnce() + + var w workv1.ManifestWork + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: testSDCluster, Name: workName(sd)}, &w); err == nil { + t.Error("the work is still there after the request was deleted") + } + views := &unstructured.UnstructuredList{} + listGVK := managedClusterViewGVK + listGVK.Kind += listKindSuffix + views.SetGroupVersionKind(listGVK) + if err := cl.List(context.Background(), views, client.InNamespace(testSDCluster)); err != nil { + t.Fatalf("list views: %v", err) + } + if len(views.Items) != 0 { + t.Errorf("%d view(s) left after the request was deleted", len(views.Items)) + } + var gone simplyblockv1alpha2.StorageSiteDeployment + if err := cl.Get(context.Background(), client.ObjectKeyFromObject(sd), &gone); err == nil { + t.Error("the request is still there: the finalizer was not released") + } +} + +func keys(m map[string]map[string]any) []string { + out := make([]string, 0, len(m)) + for k := range m { + out = append(out, k) + } + return out +} diff --git a/operator/internal/controller/testfailover_controller.go b/operator/internal/controller/testfailover_controller.go index 53cab8521..3f13340be 100644 --- a/operator/internal/controller/testfailover_controller.go +++ b/operator/internal/controller/testfailover_controller.go @@ -241,6 +241,7 @@ func (r *TestFailoverReconciler) resolveSource(ctx context.Context, tf *simplybl srcAttrs, _, _ := unstructured.NestedStringMap(pv, "spec", "csi", "volumeAttributes") bubbleVC := bubbleVolumeContext(srcAttrs) fsType, _, _ := unstructured.NestedString(pv, "spec", "csi", "fsType") + volumeMode, _, _ := unstructured.NestedString(pv, "spec", "volumeMode") if err := r.transitionTo(ctx, tf, simplyblockv1alpha2.TestFailoverStepResolvingPoint, func(s *simplyblockv1alpha2.TestFailoverStatus) { s.Clones = []simplyblockv1alpha2.TestFailoverClone{{ @@ -248,6 +249,7 @@ func (r *TestFailoverReconciler) resolveSource(ctx context.Context, tf *simplybl SourceHandle: handle, SourceVolumeContext: bubbleVC, SourceFSType: fsType, + SourceVolumeMode: volumeMode, }} s.Message = "resolved the source volume; resolving the recovery point" }); err != nil { @@ -322,6 +324,7 @@ func (r *TestFailoverReconciler) resolveSourceGroup(ctx context.Context, tf *sim srcAttrs, _, _ := unstructured.NestedStringMap(pv, "spec", "csi", "volumeAttributes") bubbleVC := bubbleVolumeContext(srcAttrs) fsType, _, _ := unstructured.NestedString(pv, "spec", "csi", "fsType") + volumeMode, _, _ := unstructured.NestedString(pv, "spec", "volumeMode") clones := make([]simplyblockv1alpha2.TestFailoverClone, 0, len(memberIDs)) for _, id := range memberIDs { @@ -330,6 +333,7 @@ func (r *TestFailoverReconciler) resolveSourceGroup(ctx context.Context, tf *sim SourceRef: v.PVCName, SourceHandle: srcUUID + ":" + v.PoolID + ":" + v.LvolID, SourceFSType: fsType, + SourceVolumeMode: volumeMode, SourceVolumeContext: bubbleVC, SizeBytes: v.Size, }) @@ -998,6 +1002,17 @@ func (r *TestFailoverReconciler) placeBubble(ctx context.Context, tf *simplybloc return ctrl.Result{}, nil } +// volumeModeOf is the bubble PV's and PVC's volumeMode: the source's, so a +// VM's Block disk stays a block device instead of being mounted as a +// filesystem; nil (the default, Filesystem) when the source did not say. +func volumeModeOf(clone simplyblockv1alpha2.TestFailoverClone) *corev1.PersistentVolumeMode { + if clone.SourceVolumeMode == "" { + return nil + } + mode := corev1.PersistentVolumeMode(clone.SourceVolumeMode) + return &mode +} + // bubbleVolumeContextStripKeys are the source PV volumeAttributes that must NOT // be carried onto the bubble PV: they identify the SOURCE volume and its NVMe-oF // target. The node plugin re-resolves the clone's own identity from the clone @@ -1078,6 +1093,7 @@ func (r *TestFailoverReconciler) bubbleManifestWork(tf *simplyblockv1alpha2.Test AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, PersistentVolumeReclaimPolicy: corev1.PersistentVolumeReclaimRetain, StorageClassName: scName, + VolumeMode: volumeModeOf(clone), ClaimRef: &corev1.ObjectReference{ Kind: "PersistentVolumeClaim", APIVersion: "v1", Namespace: ns, Name: pvcName, }, @@ -1099,6 +1115,7 @@ func (r *TestFailoverReconciler) bubbleManifestWork(tf *simplyblockv1alpha2.Test Resources: corev1.VolumeResourceRequirements{Requests: corev1.ResourceList{corev1.ResourceStorage: capacity}}, StorageClassName: &scName, VolumeName: pvName, + VolumeMode: volumeModeOf(clone), }, } for _, obj := range []client.Object{pv, pvc} { diff --git a/operator/internal/controller/testfailover_controller_unit_test.go b/operator/internal/controller/testfailover_controller_unit_test.go index 2df684ed3..7c8effd6d 100644 --- a/operator/internal/controller/testfailover_controller_unit_test.go +++ b/operator/internal/controller/testfailover_controller_unit_test.go @@ -53,7 +53,7 @@ func newTestFailoverReconciler(t *testing.T, objects ...client.Object) (*TestFai // GVK (and list GVK) registered to create and read it. scheme.AddKnownTypeWithName(managedClusterViewGVK, &unstructured.Unstructured{}) listGVK := managedClusterViewGVK - listGVK.Kind += "List" + listGVK.Kind += listKindSuffix scheme.AddKnownTypeWithName(listGVK, &unstructured.UnstructuredList{}) if err := workv1.Install(scheme); err != nil { t.Fatalf("register work/v1 scheme: %v", err) @@ -443,7 +443,7 @@ func TestFailoverResolvingSourceReuseViewOnRestart(t *testing.T) { list := &unstructured.UnstructuredList{} gvk := managedClusterViewGVK - gvk.Kind += "List" + gvk.Kind += listKindSuffix list.SetGroupVersionKind(gvk) if err := cl.List(ctx, list, client.InNamespace(tf.Spec.SourceCluster)); err != nil { t.Fatalf("list views: %v", err) diff --git a/operator/internal/controllers/cluster/storagecluster_controller.go b/operator/internal/controllers/cluster/storagecluster_controller.go index 6f30bef45..14a963ce3 100644 --- a/operator/internal/controllers/cluster/storagecluster_controller.go +++ b/operator/internal/controllers/cluster/storagecluster_controller.go @@ -201,6 +201,14 @@ type CSIClusterEntry struct { ClusterID string `json:"cluster_id"` ClusterEndpoint string `json:"cluster_endpoint"` ClusterSecret string `json:"cluster_secret"` + // Local marks a cluster this operator manages, i.e. the storage of the + // Kubernetes cluster the driver runs on, as opposed to an entry another + // site registered so that a failed-over volume's handle still resolves. + // The driver's csi-addons Replication RPCs act on the LOCAL member of a + // replication chain: a volume's PV keeps the handle it was created with + // across fail-overs, and the chain of relationships behind it alternates + // between the sites. + Local bool `json:"local,omitempty"` } // +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storageclusters,verbs=get;list;watch;create;update;patch;delete @@ -1126,6 +1134,7 @@ func (r *StorageClusterReconciler) upsertCSICredentials( ClusterID: clusterID, ClusterEndpoint: r.API.Endpoint(ctx), ClusterSecret: clusterSecret, + Local: true, } for i := range creds.Clusters { if creds.Clusters[i].ClusterID == clusterID { diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..9c2718a8f --- /dev/null +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1057 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + prefix: + description: |- + Prefix narrows the store to one key prefix, so that several clusters can + share a bucket without each walking the others' backups. + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: Count is the number of journal managers + to configure. + format: int32 + minimum: 1 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml index d68cf6ad7..d8b008df6 100644 --- a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml @@ -206,6 +206,13 @@ spec: identity keys are dropped so a failed clone lookup can never point the mount back at the source. type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string required: - sourceRef type: object